feat: classify sensitive columns locally

This commit is contained in:
Codex
2026-09-03 02:11:13 +02:00
parent 7b87e95427
commit f114d0065a
57 changed files with 4038 additions and 1149 deletions
+20 -9
View File
@@ -98,6 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr
description history, and sensitive-field review/history use the production APIs in right-side
drawers rather than prototype fixtures; closing a history drawer does not stop its background run.
Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a
catalog model. The backend reads selected source tables through read-only, database-specific
adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript
`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to
five seconds per table before sampling and the whole request to sixty seconds. Draft assessments
remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is
CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by
default because its extra findings did not offset the coverage lost to inference within the global
deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`.
Physical membership, source
comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are
projections of the external schema. They cannot be created, renamed, or structurally edited by
@@ -155,15 +166,15 @@ Semantic aliases, value descriptions, synonyms, and concepts remain deferred to
slices.
AI Description Generation uses the catalog's human-owned Sensitive Data Flag. The flag defaults to
`false`, including for newly synchronized columns. An administrator may request an AI proposal based
only on structural metadata for one selected database, selected tables, or selected columns. The
backend divides large scopes into deterministic model requests of at most ten columns, also bounded
by helper message size, and combines their results, but the proposal remains an unsaved draft until
the human reviews and saves it.
Each started suggestion attempt records a separate Sensitive Data Suggestion Run with aggregate
counters and safe ordered events. This operational history never stores per-column proposals,
prompts, raw model output, or provider diagnostics; reloading still discards an unsaved review
draft.
`false`, including for newly synchronized columns. An administrator may request a local sensitivity
analysis for one selected database, selected tables, or selected columns. One deterministic
TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER;
no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments
remain an unsaved draft until the human reviews and saves any chosen flag changes, including a
downgrade to non-sensitive.
Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe
ordered events. This operational history never stores per-column assessments, source values,
matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft.
For unprotected columns, up to five source rows and five representative non-null values may be sent
transiently to the configured model provider. Protected columns are omitted from source reads and
replaced in the prompt by deterministic plausible values derived only from their metadata. Existing
+25
View File
@@ -12,14 +12,17 @@
"@types/pg": "^8.20.3",
"fastify": "^5.0.0",
"kysely": "^0.29.5",
"libphonenumber-js": "1.13.12",
"openid-client": "6.8.5",
"pg": "^8.22.0",
"validator": "13.15.35",
"yaml": "^2.9.0",
"zod": "^4.4.3"
},
"devDependencies": {
"@testcontainers/postgresql": "^12.1.0",
"@types/node": "24.13.3",
"@types/validator": "13.15.10",
"tsx": "^4.19.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
@@ -1329,6 +1332,13 @@
"dev": true,
"license": "MIT"
},
"node_modules/@types/validator": {
"version": "13.15.10",
"resolved": "https://registry.npmjs.org/@types/validator/-/validator-13.15.10.tgz",
"integrity": "sha512-T8L6i7wCuyoK8A/ZeLYt1+q0ty3Zb9+qbSSvrIVitzT3YjZqkTZ40IbRsPanlB4h1QB3JVL1SYCdR6ngtFYcuA==",
"dev": true,
"license": "MIT"
},
"node_modules/@vitest/expect": {
"version": "2.1.9",
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-2.1.9.tgz",
@@ -2824,6 +2834,12 @@
"safe-buffer": "~5.1.0"
}
},
"node_modules/libphonenumber-js": {
"version": "1.13.12",
"resolved": "https://registry.npmjs.org/libphonenumber-js/-/libphonenumber-js-1.13.12.tgz",
"integrity": "sha512-uLVeV1c9OTk6qkdqnj+mpMD+ZdnZ0szVyWu58HwMmpwkHA1gCEkyjd3veZQXDnuw9KEwSRjcc9B1pS9XKIN1fA==",
"license": "MIT"
},
"node_modules/light-my-request": {
"version": "6.6.0",
"resolved": "https://registry.npmjs.org/light-my-request/-/light-my-request-6.6.0.tgz",
@@ -4121,6 +4137,15 @@
"dev": true,
"license": "MIT"
},
"node_modules/validator": {
"version": "13.15.35",
"resolved": "https://registry.npmjs.org/validator/-/validator-13.15.35.tgz",
"integrity": "sha512-TQ5pAGhd5whStmqWvYF4OjQROlmv9SMFVt37qoCBdqRffuuklWYQlCNnEs2ZaIBD1kZRNnikiZOS1eqgkar0iw==",
"license": "MIT",
"engines": {
"node": ">= 0.10"
}
},
"node_modules/vite": {
"version": "5.4.21",
"resolved": "https://registry.npmjs.org/vite/-/vite-5.4.21.tgz",
+4
View File
@@ -7,6 +7,7 @@
"prebuild": "node scripts/clean-dist.mjs",
"build": "tsc -p tsconfig.json",
"catalog:migrate": "node dist/catalog/migrate.js",
"sensitivity:shadow": "node dist/catalog/sensitivity-shadow.js",
"test": "vitest run",
"start": "node dist/server.js",
"test:schema-v4-verifier": "python3 -I -B scripts/test_revision_state_policy.py && node --test scripts/verify-workspace-descriptor-files.test.mjs scripts/revision-state-policy.test.mjs",
@@ -19,14 +20,17 @@
"@types/pg": "^8.20.3",
"fastify": "^5.0.0",
"kysely": "^0.29.5",
"libphonenumber-js": "1.13.12",
"openid-client": "6.8.5",
"pg": "^8.22.0",
"validator": "13.15.35",
"yaml": "^2.9.0",
"zod": "^4.4.3"
},
"devDependencies": {
"@testcontainers/postgresql": "^12.1.0",
"@types/node": "24.13.3",
"@types/validator": "13.15.10",
"tsx": "^4.19.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
@@ -0,0 +1,8 @@
34448b82c17d60fec9b65b1f093c115ddbaadc04beb1b0140b6bfed2e012a930 ./.gitattributes
4d9344c58a2a2ea4bb4ff4f7c611a853cf413205fc10d0cace564eba06f73828 ./README.md
180f0a10d1d5ed5ce3318db0bcb0b1b7780d79a52f0a8fc3acbd27f74536d0e4 ./THOTHII_MODEL_REVISION
164f17362bcf9d114067d3465e7374bfdd79ce6b605acb745de5a49dabb9595c ./config.json
f27dd63cc43a248d2566f0b6ad7a115db353676ce0561dcbca45bac766464c1a ./encoder_config/config.json
0280f6f39f6012da50b6640bad438d9b7e763a1b0102094115d1b710c4dd79b6 ./model.safetensors
f6df10ec83bea993035b2dd7c39345a3d4fcf23421c2adb6cb4ffc1e6d1bc4b5 ./tokenizer.json
233beed1f1095cccfc7907cde31a8d90a0c6aa4fdfaf6493f8e55fd162e81ae6 ./tokenizer_config.json
@@ -0,0 +1,34 @@
# Optional offline CPU pack. Fully version-locked in its own venv; not part of the base image.
--extra-index-url https://download.pytorch.org/whl/cpu
accelerate==1.14.0
annotated-types==0.8.0
certifi==2026.7.22
charset-normalizer==3.5.1
filelock==3.32.5
fsspec==2026.7.0
gliner2[local]==2.0.0
hf-xet==1.6.0
huggingface-hub==0.36.2
idna==3.19
Jinja2==3.1.6
MarkupSafe==3.0.3
mpmath==1.3.0
networkx==3.6.1
numpy==2.5.2
packaging==26.3
peft==0.20.0
psutil==7.2.2
pydantic==2.13.5
pydantic-core==2.46.5
PyYAML==6.0.3
regex==2026.9.3
requests==2.34.2
safetensors==0.8.0
sympy==1.14.0
tokenizers==0.22.2
torch==2.14.0+cpu
tqdm==4.70.0
transformers==4.57.6
typing-extensions==4.16.0
typing-inspection==0.4.4
urllib3==2.7.0
+301
View File
@@ -0,0 +1,301 @@
"""Offline, CPU-only JSONL worker for optional sensitivity NER evidence."""
from __future__ import annotations
import argparse
import contextlib
import ctypes
import errno
import hashlib
import json
import os
import socket
import sys
import tempfile
from pathlib import Path
from typing import Any
PII_LABELS = [
"person",
"full_name",
"first_name",
"middle_name",
"last_name",
"date_of_birth",
"email",
"phone_number",
"address",
"street_address",
"city",
"state_or_region",
"postal_code",
"country",
"government_id",
"national_id_number",
"passport_number",
"drivers_license_number",
"license_number",
"tax_id",
"tax_number",
"bank_account",
"account_number",
"routing_number",
"iban",
"payment_card",
"card_number",
"card_expiry",
"card_cvv",
"username",
"ip_address",
"account_id",
"sensitive_account_id",
"password",
"secret",
"api_key",
"access_token",
"recovery_code",
"sensitive_date",
"document_date",
"expiration_date",
"transaction_date",
]
_MODEL_COMPAT_DIRECTORY: tempfile.TemporaryDirectory[str] | None = None
_EXPECTED_MODEL_REVISION = "c153999da5f4c509df4322b0c6a1baf3d2c284d7"
def _arguments() -> argparse.Namespace:
parser = argparse.ArgumentParser(add_help=False)
parser.add_argument("--model", required=True)
parser.add_argument("--threads", type=int, default=2)
return parser.parse_args()
def _disable_network() -> None:
libc = ctypes.CDLL(None, use_errno=True)
libc.prctl.argtypes = [
ctypes.c_int,
ctypes.c_ulong,
ctypes.c_ulong,
ctypes.c_ulong,
ctypes.c_ulong,
]
libc.prctl.restype = ctypes.c_int
if libc.prctl(38, 1, 0, 0, 0) != 0: # PR_SET_NO_NEW_PRIVS
raise RuntimeError("cannot enable no-new-privileges for network isolation")
try:
seccomp = ctypes.CDLL("libseccomp.so.2", use_errno=True)
except OSError as error:
raise RuntimeError("libseccomp is required for network isolation") from error
seccomp.seccomp_init.argtypes = [ctypes.c_uint32]
seccomp.seccomp_init.restype = ctypes.c_void_p
seccomp.seccomp_syscall_resolve_name.argtypes = [ctypes.c_char_p]
seccomp.seccomp_syscall_resolve_name.restype = ctypes.c_int
seccomp.seccomp_rule_add.argtypes = [
ctypes.c_void_p,
ctypes.c_uint32,
ctypes.c_int,
ctypes.c_uint,
]
seccomp.seccomp_rule_add.restype = ctypes.c_int
seccomp.seccomp_load.argtypes = [ctypes.c_void_p]
seccomp.seccomp_load.restype = ctypes.c_int
seccomp.seccomp_release.argtypes = [ctypes.c_void_p]
seccomp.seccomp_release.restype = None
allow = 0x7FFF0000 # SCMP_ACT_ALLOW
deny = 0x00050000 | errno.EPERM # SCMP_ACT_ERRNO(EPERM)
filter_context = seccomp.seccomp_init(allow)
if not filter_context:
raise RuntimeError("cannot initialize network syscall filter")
try:
for syscall in (
"socket",
"connect",
"sendto",
"sendmsg",
"sendmmsg",
"bind",
"listen",
"accept",
"accept4",
):
syscall_number = seccomp.seccomp_syscall_resolve_name(syscall.encode("ascii"))
if syscall_number < 0:
raise RuntimeError(f"cannot resolve network syscall: {syscall}")
if seccomp.seccomp_rule_add(filter_context, deny, syscall_number, 0) != 0:
raise RuntimeError(f"cannot block network syscall: {syscall}")
if seccomp.seccomp_load(filter_context) != 0:
raise RuntimeError("cannot activate network syscall filter")
finally:
seccomp.seccomp_release(filter_context)
def blocked(*_args: Any, **_kwargs: Any) -> Any:
raise PermissionError(errno.EPERM, "network disabled")
socket.socket = blocked # type: ignore[assignment]
socket.create_connection = blocked # type: ignore[assignment]
def _verify_model(path: Path) -> None:
revision_path = path / "THOTHII_MODEL_REVISION"
try:
revision = revision_path.read_text(encoding="utf-8").strip()
except OSError as error:
raise RuntimeError("model revision marker is unavailable") from error
if revision != _EXPECTED_MODEL_REVISION:
raise RuntimeError("model revision is not approved")
manifest_path = Path(__file__).with_name("sensitivity-ner-model-sha256.txt")
try:
manifest = manifest_path.read_text(encoding="utf-8").splitlines()
except OSError as error:
raise RuntimeError("model checksum manifest is unavailable") from error
for line in manifest:
checksum, separator, relative_name = line.partition(" ")
if not separator or len(checksum) != 64 or not relative_name.startswith("./"):
raise RuntimeError("model checksum manifest is invalid")
relative_path = Path(relative_name[2:])
if relative_path.is_absolute() or ".." in relative_path.parts:
raise RuntimeError("model checksum path is invalid")
model_file = path / relative_path
if not model_file.is_file() or model_file.is_symlink():
raise RuntimeError("approved model file is unavailable")
digest = hashlib.sha256()
with model_file.open("rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(chunk)
if digest.hexdigest() != checksum:
raise RuntimeError("approved model checksum does not match")
def _transformers4_model_path(path: Path) -> Path:
"""Adapt tokenizer metadata emitted by Transformers 5 without changing pinned weights.
GLiNER2 2.0.0 officially requires Transformers <5, while current Fastino checkpoints were
saved by Transformers 5.8.0. Transformers 4 calls the same list
``additional_special_tokens``; Transformers 5 renamed it to ``extra_special_tokens`` and
changed its type. Keep the downloaded model immutable and create a temporary symlink view
containing only the compatibility metadata needed by the supported GLiNER2 dependency set.
"""
tokenizer_path = path / "tokenizer_config.json"
try:
tokenizer = json.loads(tokenizer_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as error:
raise RuntimeError("invalid tokenizer configuration") from error
extra_tokens = tokenizer.get("extra_special_tokens")
if extra_tokens is None:
return path
if not isinstance(extra_tokens, list) or not all(isinstance(token, str) for token in extra_tokens):
raise RuntimeError("unsupported extra_special_tokens configuration")
if "additional_special_tokens" in tokenizer:
raise RuntimeError("ambiguous special-token configuration")
global _MODEL_COMPAT_DIRECTORY
_MODEL_COMPAT_DIRECTORY = tempfile.TemporaryDirectory(prefix="thothii-ner-model-")
compatible_path = Path(_MODEL_COMPAT_DIRECTORY.name)
for child in path.iterdir():
if child.name == tokenizer_path.name:
continue
(compatible_path / child.name).symlink_to(child, target_is_directory=child.is_dir())
tokenizer["additional_special_tokens"] = tokenizer.pop("extra_special_tokens")
(compatible_path / tokenizer_path.name).write_text(
json.dumps(tokenizer, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return compatible_path
def _load_model(model_path: str, threads: int) -> Any:
path = Path(model_path).resolve(strict=True)
if not path.is_dir():
raise RuntimeError("model path must be a local directory")
_verify_model(path)
os.environ["CUDA_VISIBLE_DEVICES"] = ""
os.environ["HIP_VISIBLE_DEVICES"] = ""
os.environ["HF_HUB_OFFLINE"] = "1"
os.environ["TRANSFORMERS_OFFLINE"] = "1"
import torch
from gliner2 import AutoExtractor
torch.set_num_threads(max(1, min(threads, 8)))
torch.set_num_interop_threads(1)
compatible_path = _transformers4_model_path(path)
with contextlib.redirect_stdout(sys.stderr):
model = AutoExtractor.from_pretrained(str(compatible_path), map_location="cpu")
_disable_network()
return model
def _request(value: Any) -> tuple[str, list[dict[str, str]]]:
if not isinstance(value, dict) or not isinstance(value.get("id"), str):
raise ValueError("invalid request")
candidates = value.get("candidates")
if not isinstance(candidates, list) or not 1 <= len(candidates) <= 128:
raise ValueError("invalid candidates")
parsed: list[dict[str, str]] = []
for candidate in candidates:
if not isinstance(candidate, dict):
raise ValueError("invalid candidate")
column_id = candidate.get("columnId")
text = candidate.get("text")
if not isinstance(column_id, str) or not isinstance(text, str) or not 1 <= len(text) <= 500:
raise ValueError("invalid candidate")
parsed.append({"columnId": column_id, "text": text})
return value["id"], parsed
def _detect(model: Any, candidates: list[dict[str, str]]) -> list[dict[str, Any]]:
evidence: list[dict[str, Any]] = []
for candidate in candidates:
result = model.extract_entities(
candidate["text"],
PII_LABELS,
threshold=0.5,
include_confidence=True,
)
entities = result.get("entities", {}) if isinstance(result, dict) else {}
best: tuple[str, float] | None = None
if isinstance(entities, dict):
for label, matches in entities.items():
if label not in PII_LABELS or not isinstance(matches, list):
continue
for match in matches:
if not isinstance(match, dict):
continue
confidence = match.get("confidence")
if not isinstance(confidence, (int, float)) or not 0 <= confidence <= 1:
continue
if best is None or confidence > best[1]:
best = (label, float(confidence))
if best is not None:
evidence.append(
{
"columnId": candidate["columnId"],
"label": best[0],
"confidence": best[1],
}
)
return evidence
def main() -> int:
args = _arguments()
model = _load_model(args.model, args.threads)
print(json.dumps({"ready": True}, separators=(",", ":")), flush=True)
for line in sys.stdin:
request_id = "invalid"
try:
request_id, candidates = _request(json.loads(line))
response = {"id": request_id, "ok": True, "evidence": _detect(model, candidates)}
except Exception:
response = {"id": request_id, "ok": False, "error": "detection_failed"}
print(json.dumps(response, separators=(",", ":")), flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+34 -8
View File
@@ -3,6 +3,7 @@ import cors from "@fastify/cors";
import cookie from "@fastify/cookie";
import rateLimit from "@fastify/rate-limit";
import { dirname, isAbsolute, join } from "node:path";
import { fileURLToPath } from "node:url";
import { tmpdir } from "node:os";
import type { AppConfig } from "./config.js";
import { ThtRunner } from "./tht/tht-runner.js";
@@ -57,8 +58,11 @@ import { metadataGenerationModelRoutes } from "./routes/metadata-generation-mode
import { catalogDescriptionConsolidationRoutes } from "./routes/catalog-description-consolidation.js";
import { PythonModelCompleter, type ModelCompleter } from "./catalog/model-completer.js";
import { DescriptionGenerationWorker } from "./catalog/description-generation-worker.js";
import { SensitiveDataSuggester } from "./catalog/sensitive-data-suggester.js";
import { SensitiveDataSuggestionRunner } from "./catalog/sensitive-data-suggestion-runner.js";
import { SensitivityAnalysisService } from "./catalog/sensitivity-analysis-service.js";
import { SensitivityAnalysisRunner } from "./catalog/sensitivity-analysis-runner.js";
import { SensitivityClassifier, type LocalNerDetector, type SensitivityValueSource } from "./catalog/sensitivity-classifier.js";
import { ConcreteSensitivityValueSource } from "./catalog/sensitivity-value-source.js";
import { PythonLocalNerDetector } from "./catalog/local-ner-detector.js";
import {
ConcreteDescriptionSourceSampler,
type DescriptionSourceSampler,
@@ -93,6 +97,8 @@ export interface BuildAppDeps {
runtimeModelCatalog?: RuntimeModelCatalog;
modelCompleter?: ModelCompleter;
descriptionSourceSampler?: DescriptionSourceSampler;
sensitivityValueSource?: SensitivityValueSource;
localNerDetector?: LocalNerDetector;
workspaceRuntimeSupport?: (workspace: WorkspaceDescriptor) => boolean;
maintenanceBarrier?: MaintenanceBarrier;
piManagement?: PiManagementService;
@@ -194,12 +200,24 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
catalogOperationCoordinator,
descriptionSourceSampler,
);
const sensitiveDataSuggester = new SensitiveDataSuggester(
const sensitivityValueSource = deps?.sensitivityValueSource
?? new ConcreteSensitivityValueSource(catalogPostgresAccess, workspaceSecretStore);
const configuredNerWorker = config.sensitivityNer?.workerScript
?? fileURLToPath(new URL("../python/sensitivity_ner_worker.py", import.meta.url));
const localNerDetector = deps?.localNerDetector ?? (config.sensitivityNer
? new PythonLocalNerDetector({
pythonExecutable: config.sensitivityNer.pythonExecutable,
workerScript: configuredNerWorker,
modelPath: config.sensitivityNer.modelPath,
cwd: dirname(configuredNerWorker),
threads: config.sensitivityNer.threads,
})
: undefined);
const sensitiveDataSuggester = new SensitivityAnalysisService(
catalogRepository,
metadataGenerationModels,
modelCompleter,
new SensitivityClassifier(sensitivityValueSource, localNerDetector),
);
const sensitiveDataSuggestionRunner = new SensitiveDataSuggestionRunner(
const sensitivityAnalysisRunner = new SensitivityAnalysisRunner(
catalogRepository,
sensitiveDataSuggester,
);
@@ -235,12 +253,20 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
);
app.addHook("onReady", async () => { await catalogSyncWorker.initialize(); });
app.addHook("onReady", async () => { await descriptionGenerationWorker.initialize(); });
app.addHook("onReady", async () => { await sensitiveDataSuggestionRunner.initialize(); });
app.addHook("onReady", async () => { await sensitivityAnalysisRunner.initialize(); });
if (localNerDetector?.warmup) {
app.addHook("onReady", async () => {
void localNerDetector.warmup?.().catch(() => undefined);
});
}
if (!deps?.catalogRepository && catalogRepository.close) {
app.addHook("onClose", async () => { await catalogRepository.close?.(); });
}
app.addHook("onClose", async () => { await catalogSyncWorker.stop(); });
app.addHook("onClose", async () => { await descriptionGenerationWorker.stop(); });
if (localNerDetector?.close) {
app.addHook("onClose", async () => { await localNerDetector.close?.(); });
}
const workspaceDiagnoser = deps?.workspaceDiagnoser
?? createProductionWorkspaceDiagnoser(config.workspaceDiagnosticTimeoutMs, undefined, {
internalQdrantUrl: config.internalQdrantUrl,
@@ -483,7 +509,7 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
catalogDescriptionGenerationRoutes(app, {
repository: catalogRepository,
worker: descriptionGenerationWorker,
sensitiveDataSuggestionRunner,
sensitivityAnalysisRunner,
});
settingsRoutes(app, { cfg: config, getSettings });
piManagementRoutes(app, { service: piManagement });
+254
View File
@@ -0,0 +1,254 @@
import { randomUUID } from "node:crypto";
import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process";
import { tmpdir } from "node:os";
import { z } from "zod";
import type {
LocalNerCandidate,
LocalNerDetector,
LocalNerEvidence,
} from "./sensitivity-classifier.js";
const MAX_LINE_BYTES = 64 * 1024;
const candidateSchema = z.object({
columnId: z.uuid(),
text: z.string().min(1).max(500),
}).strict();
const workerMessageSchema = z.union([
z.object({ ready: z.literal(true) }).strict(),
z.object({
id: z.uuid(),
ok: z.literal(true),
evidence: z.array(z.object({
columnId: z.uuid(),
label: z.string().min(1).max(80),
confidence: z.number().min(0).max(1),
}).strict()).max(1_000),
}).strict(),
z.object({ id: z.uuid(), ok: z.literal(false), error: z.string().min(1).max(80) }).strict(),
]);
export class LocalNerUnavailableError extends Error {
constructor() {
super("local NER is unavailable");
this.name = "LocalNerUnavailableError";
}
}
interface PendingRequest {
resolve: (value: readonly LocalNerEvidence[]) => void;
reject: (error: Error) => void;
timer: ReturnType<typeof setTimeout>;
signal: AbortSignal;
cancel: () => void;
}
/** Persistent JSONL adapter for the optional, CPU-only Python NER worker. */
export class PythonLocalNerDetector implements LocalNerDetector {
private child?: ChildProcessWithoutNullStreams;
private ready?: Promise<void>;
private readyResolve?: () => void;
private readyReject?: (error: Error) => void;
private workerReady = false;
private stdout = "";
private readonly pending = new Map<string, PendingRequest>();
constructor(private readonly options: {
pythonExecutable: string;
workerScript: string;
modelPath: string;
cwd: string;
threads?: number;
startupTimeoutMs?: number;
}) {}
async warmup(): Promise<void> {
await this.ensureStarted();
}
isReady(): boolean {
return this.workerReady
&& this.child !== undefined
&& this.child.exitCode === null
&& this.child.signalCode === null;
}
async detect(
candidates: readonly LocalNerCandidate[],
signal: AbortSignal,
deadline: number,
): Promise<readonly LocalNerEvidence[]> {
const parsed = z.array(candidateSchema).min(1).max(128).parse(candidates);
if (signal.aborted || deadline <= Date.now()) throw new LocalNerUnavailableError();
await this.ensureStartedWithin(signal, deadline);
if (!this.child || this.child.exitCode !== null || this.child.signalCode !== null) {
throw new LocalNerUnavailableError();
}
const id = randomUUID();
return await new Promise<readonly LocalNerEvidence[]>((resolve, reject) => {
const fail = () => {
this.finishPending(id);
reject(new LocalNerUnavailableError());
this.stopWorker();
};
const timer = setTimeout(fail, Math.max(1, Math.floor(deadline - Date.now())));
const cancel = fail;
const pending: PendingRequest = { resolve, reject, timer, signal, cancel };
this.pending.set(id, pending);
signal.addEventListener("abort", cancel, { once: true });
this.child!.stdin.write(`${JSON.stringify({ id, candidates: parsed })}\n`, (error) => {
if (error) fail();
});
});
}
async close(): Promise<void> {
const child = this.child;
if (!child || child.exitCode !== null || child.signalCode !== null) return;
await new Promise<void>((resolve) => {
child.once("close", () => resolve());
child.kill("SIGTERM");
setTimeout(() => {
if (child.exitCode === null && child.signalCode === null) child.kill("SIGKILL");
}, 250).unref();
});
}
private async ensureStarted(): Promise<void> {
if (this.ready) return await this.ready;
this.ready = new Promise<void>((resolve, reject) => {
this.readyResolve = resolve;
this.readyReject = reject;
});
const threads = String(this.options.threads ?? 2);
const inheritedRuntimeEnvironment = Object.fromEntries([
"PATH", "SystemRoot", "WINDIR", "PATHEXT", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL",
].flatMap((name) => process.env[name] === undefined ? [] : [[name, process.env[name]!]]));
const child = spawn(this.options.pythonExecutable, [
"-I",
"-B",
this.options.workerScript,
"--model",
this.options.modelPath,
"--threads",
threads,
], {
cwd: this.options.cwd,
stdio: ["pipe", "pipe", "pipe"],
env: {
...inheritedRuntimeEnvironment,
HOME: process.env.HOME ?? tmpdir(),
CUDA_VISIBLE_DEVICES: "",
HIP_VISIBLE_DEVICES: "",
HF_HUB_OFFLINE: "1",
HF_HUB_DISABLE_TELEMETRY: "1",
TRANSFORMERS_OFFLINE: "1",
TOKENIZERS_PARALLELISM: "false",
PYTHONNOUSERSITE: "1",
OMP_NUM_THREADS: threads,
MKL_NUM_THREADS: threads,
OPENBLAS_NUM_THREADS: threads,
HTTP_PROXY: "",
HTTPS_PROXY: "",
ALL_PROXY: "",
NO_PROXY: "*",
},
});
this.child = child;
child.stdout.setEncoding("utf8");
child.stdout.on("data", (chunk: string) => this.receive(chunk));
child.stderr.resume();
child.once("error", () => this.failWorker());
child.once("close", () => this.failWorker());
const startupTimer = setTimeout(() => this.failWorker(), this.options.startupTimeoutMs ?? 120_000);
startupTimer.unref();
try {
await this.ready;
} finally {
clearTimeout(startupTimer);
}
}
private async ensureStartedWithin(signal: AbortSignal, deadline: number): Promise<void> {
const started = this.ensureStarted();
await new Promise<void>((resolve, reject) => {
let settled = false;
const finish = (error?: Error, stopWorker = false) => {
if (settled) return;
settled = true;
clearTimeout(timer);
signal.removeEventListener("abort", cancel);
if (stopWorker) this.failWorker();
if (error) reject(error);
else resolve();
};
const cancel = () => finish(new LocalNerUnavailableError(), true);
const timer = setTimeout(cancel, Math.max(1, Math.floor(deadline - Date.now())));
signal.addEventListener("abort", cancel, { once: true });
void started.then(
() => finish(),
() => finish(new LocalNerUnavailableError()),
);
});
}
private receive(chunk: string): void {
this.stdout += chunk;
if (Buffer.byteLength(this.stdout, "utf8") > MAX_LINE_BYTES) {
this.failWorker();
return;
}
let newline: number;
while ((newline = this.stdout.indexOf("\n")) >= 0) {
const line = this.stdout.slice(0, newline);
this.stdout = this.stdout.slice(newline + 1);
if (!line) continue;
try {
const message = workerMessageSchema.parse(JSON.parse(line));
if ("ready" in message) {
this.workerReady = true;
this.readyResolve?.();
this.readyResolve = undefined;
this.readyReject = undefined;
continue;
}
const pending = this.pending.get(message.id);
if (!pending) continue;
this.finishPending(message.id);
if (message.ok) pending.resolve(message.evidence);
else pending.reject(new LocalNerUnavailableError());
} catch {
this.failWorker();
return;
}
}
}
private finishPending(id: string): void {
const pending = this.pending.get(id);
if (!pending) return;
clearTimeout(pending.timer);
pending.signal.removeEventListener("abort", pending.cancel);
this.pending.delete(id);
}
private stopWorker(): void {
const child = this.child;
if (child && child.exitCode === null && child.signalCode === null) child.kill("SIGTERM");
}
private failWorker(): void {
const error = new LocalNerUnavailableError();
this.readyReject?.(error);
this.readyResolve = undefined;
this.readyReject = undefined;
for (const [id, pending] of this.pending) {
this.finishPending(id);
pending.reject(error);
}
this.stopWorker();
this.child = undefined;
this.ready = undefined;
this.workerReady = false;
this.stdout = "";
}
}
+50 -47
View File
@@ -35,10 +35,10 @@ import {
type DescriptionGenerationRun,
type DescriptionGenerationRunUpdate,
type DescriptionGenerationScope,
type SensitiveDataSuggestionEvent,
type SensitiveDataSuggestionRun,
type SensitiveDataSuggestionRunUpdate,
type SensitiveDataSuggestionScope,
type SensitivityAnalysisEvent,
type SensitivityAnalysisRun,
type SensitivityAnalysisRunUpdate,
type SensitivityAnalysisScope,
type TableSyncRepositoryResult,
type WorkspaceDatabase,
} from "./types.js";
@@ -56,8 +56,8 @@ export class MemoryCatalogRepository implements CatalogRepository {
private readonly logicalRelationships = new Map<string, CatalogLogicalRelationship>();
private readonly descriptionGenerationRuns = new Map<string, DescriptionGenerationRun>();
private readonly descriptionGenerationEvents = new Map<string, DescriptionGenerationEvent[]>();
private readonly sensitiveDataSuggestionRuns = new Map<string, SensitiveDataSuggestionRun>();
private readonly sensitiveDataSuggestionEvents = new Map<string, SensitiveDataSuggestionEvent[]>();
private readonly sensitivityAnalysisRuns = new Map<string, SensitivityAnalysisRun>();
private readonly sensitivityAnalysisEvents = new Map<string, SensitivityAnalysisEvent[]>();
private readonly syncRuns = new Map<string, CatalogSyncRun>();
private readonly syncEvents = new Map<string, CatalogSyncEvent[]>();
@@ -183,10 +183,10 @@ export class MemoryCatalogRepository implements CatalogRepository {
this.descriptionGenerationRuns.delete(runId);
this.descriptionGenerationEvents.delete(runId);
}
for (const [runId, run] of this.sensitiveDataSuggestionRuns) {
for (const [runId, run] of this.sensitivityAnalysisRuns) {
if (run.databaseId !== id) continue;
this.sensitiveDataSuggestionRuns.delete(runId);
this.sensitiveDataSuggestionEvents.delete(runId);
this.sensitivityAnalysisRuns.delete(runId);
this.sensitivityAnalysisEvents.delete(runId);
}
return this.records.delete(id);
}
@@ -433,93 +433,96 @@ export class MemoryCatalogRepository implements CatalogRepository {
.map((event) => structuredClone(event));
}
async createSensitiveDataSuggestionRun(
async createSensitivityAnalysisRun(
databaseId: string,
scope: SensitiveDataSuggestionScope,
modelId: string,
): Promise<SensitiveDataSuggestionRun> {
scope: SensitivityAnalysisScope,
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
): Promise<SensitivityAnalysisRun> {
const now = new Date().toISOString();
const run: SensitiveDataSuggestionRun = {
const run: SensitivityAnalysisRun = {
id: randomUUID(),
databaseId,
scope,
modelId,
engine: origin.engine,
modelId: origin.engine === "llm" ? origin.modelId : null,
policyVersion: origin.engine === "local" ? origin.policyVersion : null,
status: "running",
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
inputTokens: 0,
cacheReadTokens: 0,
outputTokens: 0,
suggestedNonSensitive: 0,
unknown: 0,
inputTokens: 0,
cacheReadTokens: 0,
outputTokens: 0,
createdAt: now,
startedAt: now,
updatedAt: now,
finishedAt: null,
errorSummary: null,
};
this.sensitiveDataSuggestionRuns.set(run.id, run);
this.sensitivityAnalysisRuns.set(run.id, run);
return structuredClone(run);
}
async getSensitiveDataSuggestionRun(
async getSensitivityAnalysisRun(
runId: string,
): Promise<SensitiveDataSuggestionRun | undefined> {
const run = this.sensitiveDataSuggestionRuns.get(runId);
): Promise<SensitivityAnalysisRun | undefined> {
const run = this.sensitivityAnalysisRuns.get(runId);
return run ? structuredClone(run) : undefined;
}
async listSensitiveDataSuggestionRuns(limit = 50): Promise<SensitiveDataSuggestionRun[]> {
return [...this.sensitiveDataSuggestionRuns.values()]
async listSensitivityAnalysisRuns(limit = 50): Promise<SensitivityAnalysisRun[]> {
return [...this.sensitivityAnalysisRuns.values()]
.sort((a, b) => b.createdAt.localeCompare(a.createdAt) || b.id.localeCompare(a.id))
.slice(0, limit)
.map((run) => structuredClone(run));
}
async interruptActiveSensitiveDataSuggestionRuns(
async interruptActiveSensitivityAnalysisRuns(
errorSummary: string,
): Promise<SensitiveDataSuggestionRun[]> {
const interrupted: SensitiveDataSuggestionRun[] = [];
for (const run of this.sensitiveDataSuggestionRuns.values()) {
): Promise<SensitivityAnalysisRun[]> {
const interrupted: SensitivityAnalysisRun[] = [];
for (const run of this.sensitivityAnalysisRuns.values()) {
if (run.status !== "running") continue;
const now = new Date().toISOString();
const updated: SensitiveDataSuggestionRun = {
const updated: SensitivityAnalysisRun = {
...run,
status: "interrupted",
updatedAt: now,
finishedAt: now,
errorSummary,
};
this.sensitiveDataSuggestionRuns.set(run.id, updated);
this.sensitivityAnalysisRuns.set(run.id, updated);
interrupted.push(structuredClone(updated));
}
return interrupted;
}
async updateSensitiveDataSuggestionRun(
async updateSensitivityAnalysisRun(
runId: string,
update: SensitiveDataSuggestionRunUpdate,
): Promise<SensitiveDataSuggestionRun | undefined> {
const current = this.sensitiveDataSuggestionRuns.get(runId);
update: SensitivityAnalysisRunUpdate,
): Promise<SensitivityAnalysisRun | undefined> {
const current = this.sensitivityAnalysisRuns.get(runId);
if (!current) return undefined;
const updated = {
...current,
...structuredClone(update),
updatedAt: new Date().toISOString(),
};
this.sensitiveDataSuggestionRuns.set(runId, updated);
this.sensitivityAnalysisRuns.set(runId, updated);
return structuredClone(updated);
}
async appendSensitiveDataSuggestionEvent(
async appendSensitivityAnalysisEvent(
runId: string,
level: SensitiveDataSuggestionEvent["level"],
level: SensitivityAnalysisEvent["level"],
message: string,
): Promise<SensitiveDataSuggestionEvent> {
if (!this.sensitiveDataSuggestionRuns.has(runId)) {
throw new CatalogConflictError("Sensitive Data Suggestion Run does not exist");
): Promise<SensitivityAnalysisEvent> {
if (!this.sensitivityAnalysisRuns.has(runId)) {
throw new CatalogConflictError("Sensitivity Analysis Run does not exist");
}
const events = this.sensitiveDataSuggestionEvents.get(runId) ?? [];
const event: SensitiveDataSuggestionEvent = {
const events = this.sensitivityAnalysisEvents.get(runId) ?? [];
const event: SensitivityAnalysisEvent = {
runId,
sequence: events.length + 1,
level,
@@ -527,15 +530,15 @@ export class MemoryCatalogRepository implements CatalogRepository {
createdAt: new Date().toISOString(),
};
events.push(event);
this.sensitiveDataSuggestionEvents.set(runId, events);
this.sensitivityAnalysisEvents.set(runId, events);
return structuredClone(event);
}
async listSensitiveDataSuggestionEvents(
async listSensitivityAnalysisEvents(
runId: string,
afterSequence = 0,
): Promise<SensitiveDataSuggestionEvent[]> {
return (this.sensitiveDataSuggestionEvents.get(runId) ?? [])
): Promise<SensitivityAnalysisEvent[]> {
return (this.sensitivityAnalysisEvents.get(runId) ?? [])
.filter((event) => event.sequence > afterSequence)
.map((event) => structuredClone(event));
}
+4 -2
View File
@@ -9,10 +9,11 @@ import * as catalogSchemaSyncMigration from "./migrations/003_catalog_schema_syn
import * as catalogRuntimeSequencePrivilegesMigration from "./migrations/004_catalog_runtime_sequence_privileges.js";
import * as descriptionGenerationRunsMigration from "./migrations/005_description_generation_runs.js";
import * as sensitiveDataFlagMigration from "./migrations/006_sensitive_data_flag.js";
import * as sensitiveDataSuggestionRunsMigration from "./migrations/007_sensitive_data_suggestion_runs.js";
import * as sensitivityAnalysisRunsMigration from "./migrations/007_sensitive_data_suggestion_runs.js";
import * as catalogLogicalRelationshipsMigration from "./migrations/008_catalog_logical_relationships.js";
import * as aiTokenUsageMigration from "./migrations/009_ai_token_usage.js";
import * as canonicalModelIdsMigration from "./migrations/010_canonical_model_ids.js";
import * as localSensitivityAnalysisMigration from "./migrations/011_local_sensitivity_analysis.js";
const connectionString = process.env.THT_CATALOG_MIGRATOR_DATABASE_URL;
const host = process.env.THT_CATALOG_DB_HOST;
@@ -44,10 +45,11 @@ const provider: MigrationProvider = {
"004_catalog_runtime_sequence_privileges": catalogRuntimeSequencePrivilegesMigration,
"005_description_generation_runs": descriptionGenerationRunsMigration,
"006_sensitive_data_flag": sensitiveDataFlagMigration,
"007_sensitive_data_suggestion_runs": sensitiveDataSuggestionRunsMigration,
"007_sensitive_data_suggestion_runs": sensitivityAnalysisRunsMigration,
"008_catalog_logical_relationships": catalogLogicalRelationshipsMigration,
"009_ai_token_usage": aiTokenUsageMigration,
"010_canonical_model_ids": canonicalModelIdsMigration,
"011_local_sensitivity_analysis": localSensitivityAnalysisMigration,
};
},
};
@@ -0,0 +1,42 @@
import { sql, type Kysely } from "kysely";
import type { CatalogDatabase } from "../repository.js";
export async function up(db: Kysely<CatalogDatabase>): Promise<void> {
await sql.raw(`alter table sensitive_data_suggestion_runs
alter column model_id drop not null,
add column engine text not null default 'llm',
add column policy_version text,
add column unknown integer not null default 0,
drop constraint sensitive_data_suggestion_runs_counters_check,
add constraint sensitive_data_suggestion_runs_counters_check
check (total >= 0
and suggested_sensitive >= 0
and suggested_non_sensitive >= 0
and unknown >= 0
and suggested_sensitive + suggested_non_sensitive + unknown <= total),
add constraint sensitive_data_suggestion_runs_engine_check
check (engine in ('llm', 'local')),
add constraint sensitive_data_suggestion_runs_origin_check
check ((engine = 'llm' and model_id is not null and policy_version is null)
or (engine = 'local' and model_id is null
and policy_version ~ '^[a-z][a-z0-9._-]{0,63}$'))`).execute(db);
}
export async function down(db: Kysely<CatalogDatabase>): Promise<void> {
await sql.raw(`alter table sensitive_data_suggestion_runs
drop constraint sensitive_data_suggestion_runs_origin_check,
drop constraint sensitive_data_suggestion_runs_engine_check,
drop constraint sensitive_data_suggestion_runs_counters_check`).execute(db);
await sql.raw(`update sensitive_data_suggestion_runs
set model_id = coalesce(model_id, 'local/sensitivity-v1')`).execute(db);
await sql.raw(`alter table sensitive_data_suggestion_runs
drop column unknown,
drop column policy_version,
drop column engine,
alter column model_id set not null,
add constraint sensitive_data_suggestion_runs_counters_check
check (total >= 0
and suggested_sensitive >= 0
and suggested_non_sensitive >= 0
and suggested_sensitive + suggested_non_sensitive <= total)`).execute(db);
}
+58 -51
View File
@@ -45,10 +45,10 @@ import {
type DescriptionGenerationScope,
type ObservedCatalogTable,
type ObservedSchemaSnapshot,
type SensitiveDataSuggestionEvent,
type SensitiveDataSuggestionRun,
type SensitiveDataSuggestionRunUpdate,
type SensitiveDataSuggestionScope,
type SensitivityAnalysisEvent,
type SensitivityAnalysisRun,
type SensitivityAnalysisRunUpdate,
type SensitivityAnalysisScope,
type TableSyncRepositoryResult,
type WorkspaceDatabase,
} from "./types.js";
@@ -189,15 +189,18 @@ interface DescriptionGenerationEventTable {
createdAt: Timestamp;
}
interface SensitiveDataSuggestionRunTable {
interface SensitivityAnalysisRunTable {
id: string;
databaseId: string;
scope: SensitiveDataSuggestionScope;
modelId: string;
status: SensitiveDataSuggestionRun["status"];
scope: SensitivityAnalysisScope;
engine: SensitivityAnalysisRun["engine"];
modelId: string | null;
policyVersion: string | null;
status: SensitivityAnalysisRun["status"];
total: number;
suggestedSensitive: number;
suggestedNonSensitive: number;
unknown: number;
inputTokens: number;
cacheReadTokens: number;
outputTokens: number;
@@ -208,10 +211,10 @@ interface SensitiveDataSuggestionRunTable {
errorSummary: string | null;
}
interface SensitiveDataSuggestionEventTable {
interface SensitivityAnalysisEventTable {
runId: string;
sequence: number;
level: SensitiveDataSuggestionEvent["level"];
level: SensitivityAnalysisEvent["level"];
message: string;
createdAt: Timestamp;
}
@@ -264,8 +267,9 @@ export interface CatalogDatabase {
catalogLogicalRelationships: CatalogLogicalRelationshipTable;
descriptionGenerationRuns: DescriptionGenerationRunTable;
descriptionGenerationEvents: DescriptionGenerationEventTable;
sensitiveDataSuggestionRuns: SensitiveDataSuggestionRunTable;
sensitiveDataSuggestionEvents: SensitiveDataSuggestionEventTable;
// Legacy physical table names retained for migration and storage compatibility.
sensitiveDataSuggestionRuns: SensitivityAnalysisRunTable;
sensitiveDataSuggestionEvents: SensitivityAnalysisEventTable;
catalogSyncRuns: CatalogSyncRunTable;
catalogSyncEvents: CatalogSyncEventTable;
}
@@ -402,9 +406,9 @@ function serializeDescriptionGenerationEvent(
return { ...row, createdAt: new Date(row.createdAt).toISOString() };
}
function serializeSensitiveDataSuggestionRun(
row: Selectable<SensitiveDataSuggestionRunTable>,
): SensitiveDataSuggestionRun {
function serializeSensitivityAnalysisRun(
row: Selectable<SensitivityAnalysisRunTable>,
): SensitivityAnalysisRun {
const stamp = (value: Date | string | null) => value === null ? null : new Date(value).toISOString();
return {
...row,
@@ -415,9 +419,9 @@ function serializeSensitiveDataSuggestionRun(
};
}
function serializeSensitiveDataSuggestionEvent(
row: Selectable<SensitiveDataSuggestionEventTable>,
): SensitiveDataSuggestionEvent {
function serializeSensitivityAnalysisEvent(
row: Selectable<SensitivityAnalysisEventTable>,
): SensitivityAnalysisEvent {
return { ...row, createdAt: new Date(row.createdAt).toISOString() };
}
@@ -938,52 +942,55 @@ export class KyselyCatalogRepository implements CatalogRepository {
return rows.map(serializeDescriptionGenerationEvent);
}
async createSensitiveDataSuggestionRun(
async createSensitivityAnalysisRun(
databaseId: string,
scope: SensitiveDataSuggestionScope,
modelId: string,
): Promise<SensitiveDataSuggestionRun> {
scope: SensitivityAnalysisScope,
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
): Promise<SensitivityAnalysisRun> {
const row = await this.db.insertInto("sensitiveDataSuggestionRuns").values({
id: randomUUID(),
databaseId,
scope,
modelId,
engine: origin.engine,
modelId: origin.engine === "llm" ? origin.modelId : null,
policyVersion: origin.engine === "local" ? origin.policyVersion : null,
status: "running",
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
unknown: 0,
inputTokens: 0,
cacheReadTokens: 0,
outputTokens: 0,
finishedAt: null,
errorSummary: null,
}).returningAll().executeTakeFirstOrThrow();
return serializeSensitiveDataSuggestionRun(row);
return serializeSensitivityAnalysisRun(row);
}
async getSensitiveDataSuggestionRun(
async getSensitivityAnalysisRun(
runId: string,
): Promise<SensitiveDataSuggestionRun | undefined> {
): Promise<SensitivityAnalysisRun | undefined> {
const row = await this.db.selectFrom("sensitiveDataSuggestionRuns")
.selectAll()
.where("id", "=", runId)
.executeTakeFirst();
return row ? serializeSensitiveDataSuggestionRun(row) : undefined;
return row ? serializeSensitivityAnalysisRun(row) : undefined;
}
async listSensitiveDataSuggestionRuns(limit = 50): Promise<SensitiveDataSuggestionRun[]> {
async listSensitivityAnalysisRuns(limit = 50): Promise<SensitivityAnalysisRun[]> {
const rows = await this.db.selectFrom("sensitiveDataSuggestionRuns")
.selectAll()
.orderBy("createdAt", "desc")
.orderBy("id", "desc")
.limit(limit)
.execute();
return rows.map(serializeSensitiveDataSuggestionRun);
return rows.map(serializeSensitivityAnalysisRun);
}
async interruptActiveSensitiveDataSuggestionRuns(
async interruptActiveSensitivityAnalysisRuns(
errorSummary: string,
): Promise<SensitiveDataSuggestionRun[]> {
): Promise<SensitivityAnalysisRun[]> {
const rows = await this.db.updateTable("sensitiveDataSuggestionRuns")
.set({
status: "interrupted",
@@ -994,34 +1001,34 @@ export class KyselyCatalogRepository implements CatalogRepository {
.where("status", "=", "running")
.returningAll()
.execute();
return rows.map(serializeSensitiveDataSuggestionRun);
return rows.map(serializeSensitivityAnalysisRun);
}
async updateSensitiveDataSuggestionRun(
async updateSensitivityAnalysisRun(
runId: string,
update: SensitiveDataSuggestionRunUpdate,
): Promise<SensitiveDataSuggestionRun | undefined> {
update: SensitivityAnalysisRunUpdate,
): Promise<SensitivityAnalysisRun | undefined> {
const values: any = { ...update, updatedAt: sql`now()` };
const row = await this.db.updateTable("sensitiveDataSuggestionRuns")
.set(values)
.where("id", "=", runId)
.returningAll()
.executeTakeFirst();
return row ? serializeSensitiveDataSuggestionRun(row) : undefined;
return row ? serializeSensitivityAnalysisRun(row) : undefined;
}
async appendSensitiveDataSuggestionEvent(
async appendSensitivityAnalysisEvent(
runId: string,
level: SensitiveDataSuggestionEvent["level"],
level: SensitivityAnalysisEvent["level"],
message: string,
): Promise<SensitiveDataSuggestionEvent> {
): Promise<SensitivityAnalysisEvent> {
return await this.db.transaction().execute(async (trx) => {
const run = await trx.selectFrom("sensitiveDataSuggestionRuns")
.select("id")
.where("id", "=", runId)
.forUpdate()
.executeTakeFirst();
if (!run) throw new CatalogConflictError("Sensitive Data Suggestion Run does not exist");
if (!run) throw new CatalogConflictError("Sensitivity Analysis Run does not exist");
const current = await trx.selectFrom("sensitiveDataSuggestionEvents")
.select(sql<number>`coalesce(max(sequence), 0)::int`.as("sequence"))
.where("runId", "=", runId)
@@ -1032,21 +1039,21 @@ export class KyselyCatalogRepository implements CatalogRepository {
level,
message,
}).returningAll().executeTakeFirstOrThrow();
return serializeSensitiveDataSuggestionEvent(row);
return serializeSensitivityAnalysisEvent(row);
});
}
async listSensitiveDataSuggestionEvents(
async listSensitivityAnalysisEvents(
runId: string,
afterSequence = 0,
): Promise<SensitiveDataSuggestionEvent[]> {
): Promise<SensitivityAnalysisEvent[]> {
const rows = await this.db.selectFrom("sensitiveDataSuggestionEvents")
.selectAll()
.where("runId", "=", runId)
.where("sequence", ">", afterSequence)
.orderBy("sequence")
.execute();
return rows.map(serializeSensitiveDataSuggestionEvent);
return rows.map(serializeSensitivityAnalysisEvent);
}
async listRelationships(databaseId: string): Promise<CatalogPhysicalRelationship[]> {
@@ -1792,13 +1799,13 @@ export class UnavailableCatalogRepository implements CatalogRepository {
async updateDescriptionGenerationRun(): Promise<DescriptionGenerationRun | undefined> { return this.fail(); }
async appendDescriptionGenerationEvent(): Promise<DescriptionGenerationEvent> { return this.fail(); }
async listDescriptionGenerationEvents(): Promise<DescriptionGenerationEvent[]> { return this.fail(); }
async createSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun> { return this.fail(); }
async getSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun | undefined> { return this.fail(); }
async listSensitiveDataSuggestionRuns(): Promise<SensitiveDataSuggestionRun[]> { return this.fail(); }
async interruptActiveSensitiveDataSuggestionRuns(): Promise<SensitiveDataSuggestionRun[]> { return this.fail(); }
async updateSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun | undefined> { return this.fail(); }
async appendSensitiveDataSuggestionEvent(): Promise<SensitiveDataSuggestionEvent> { return this.fail(); }
async listSensitiveDataSuggestionEvents(): Promise<SensitiveDataSuggestionEvent[]> { return this.fail(); }
async createSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun> { return this.fail(); }
async getSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun | undefined> { return this.fail(); }
async listSensitivityAnalysisRuns(): Promise<SensitivityAnalysisRun[]> { return this.fail(); }
async interruptActiveSensitivityAnalysisRuns(): Promise<SensitivityAnalysisRun[]> { return this.fail(); }
async updateSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun | undefined> { return this.fail(); }
async appendSensitivityAnalysisEvent(): Promise<SensitivityAnalysisEvent> { return this.fail(); }
async listSensitivityAnalysisEvents(): Promise<SensitivityAnalysisEvent[]> { return this.fail(); }
async listRelationships(): Promise<CatalogPhysicalRelationship[]> { return this.fail(); }
async listLogicalRelationships(): Promise<CatalogLogicalRelationship[]> { return this.fail(); }
async getLogicalRelationshipContext(): Promise<CatalogLogicalRelationshipContext | undefined> { return this.fail(); }
@@ -1,254 +0,0 @@
import { z } from "zod";
import type { MetadataGenerationModels } from "./metadata-generation-models.js";
import type { ModelCompleter, ModelCompletionMessage, ModelCompletionResult, ModelCompletionUsage } from "./model-completer.js";
import type {
CatalogColumn,
CatalogRepository,
CatalogTable,
SensitiveDataSuggestionScope,
} from "./types.js";
export type { SensitiveDataSuggestionScope } from "./types.js";
// The helper accepts at most 64 KiB per message. Keep the same safety margin used by
// Description Generation so UTF-8 structural metadata never reaches that hard limit.
const MAX_USER_MESSAGE_BYTES = 60 * 1024;
// Preserve ThothAI's proven completion granularity: small batches keep generation time and
// structured-output accuracy predictable even when the helper byte limit would allow much more.
const MAX_COLUMNS_PER_BATCH = 10;
const responseSchema = z.object({
suggestions: z.array(z.object({
columnId: z.uuid(),
sensitive: z.boolean(),
}).strict()),
}).strict();
interface StructuralColumn {
columnId: string;
tableId: string;
table: string;
column: string;
dataType: string;
nullable: boolean;
primaryKey: boolean;
foreignKey: boolean;
version: number;
currentSensitive: boolean;
}
export interface SensitiveDataSuggestion {
columnId: string;
tableId: string;
tableName: string;
columnName: string;
version: number;
currentSensitive: boolean;
sensitive: boolean;
}
export class SensitiveDataSuggestionTargetNotFoundError extends Error {
constructor(readonly target: "database" | "table" | "column") {
super(`${target} not found`);
this.name = "SensitiveDataSuggestionTargetNotFoundError";
}
}
export class SensitiveDataSuggestionDuplicateTargetIdsError extends Error {
constructor() {
super("sensitive-data suggestion target IDs must be unique");
this.name = "SensitiveDataSuggestionDuplicateTargetIdsError";
}
}
export class SensitiveDataSuggestionNoEligibleColumnsError extends Error {
constructor(readonly scope: SensitiveDataSuggestionScope) {
super("selected scope has no catalog columns");
this.name = "SensitiveDataSuggestionNoEligibleColumnsError";
}
}
export class SensitiveDataSuggestionPayloadTooLargeError extends Error {
constructor() {
super("sensitive-data suggestion structural metadata is too large");
this.name = "SensitiveDataSuggestionPayloadTooLargeError";
}
}
export class SensitiveDataSuggestionInvalidResponseError extends Error {
constructor() {
super("sensitive-data suggestion response is invalid");
this.name = "SensitiveDataSuggestionInvalidResponseError";
}
}
function userContent(
database: { databaseName: string; schema: string },
columns: readonly StructuralColumn[],
): string {
return JSON.stringify({
database: database.databaseName,
schema: database.schema,
columns: columns.map((column) => ({
columnId: column.columnId,
table: column.table,
column: column.column,
dataType: column.dataType,
nullable: column.nullable,
primaryKey: column.primaryKey,
foreignKey: column.foreignKey,
})),
});
}
function batchesFor(
database: { databaseName: string; schema: string },
columns: readonly StructuralColumn[],
): StructuralColumn[][] {
const batches: StructuralColumn[][] = [];
let current: StructuralColumn[] = [];
for (const column of columns) {
if (current.length === MAX_COLUMNS_PER_BATCH) {
batches.push(current);
current = [];
}
const candidate = [...current, column];
if (Buffer.byteLength(userContent(database, candidate), "utf8") <= MAX_USER_MESSAGE_BYTES) {
current = candidate;
continue;
}
if (current.length === 0) throw new SensitiveDataSuggestionPayloadTooLargeError();
batches.push(current);
current = [column];
if (Buffer.byteLength(userContent(database, current), "utf8") > MAX_USER_MESSAGE_BYTES) {
throw new SensitiveDataSuggestionPayloadTooLargeError();
}
}
if (current.length > 0) batches.push(current);
return batches;
}
function structuralColumn(table: CatalogTable, column: CatalogColumn): StructuralColumn {
return {
columnId: column.id,
tableId: table.id,
table: table.name,
column: column.name,
dataType: column.dataType,
nullable: column.isNullable,
primaryKey: column.isPrimaryKey,
foreignKey: column.isForeignKey,
version: column.version,
currentSensitive: column.sensitive,
};
}
const systemMessage: ModelCompletionMessage = {
role: "system",
content: [
"Classify whether each database column is likely to contain sensitive source values.",
"Use only the supplied structural metadata. Return strict JSON with this exact shape:",
'{"suggestions":[{"columnId":"uuid","sensitive":true}]}',
"Return every supplied column exactly once. Do not add explanations or markdown.",
].join("\n"),
};
export class SensitiveDataSuggester {
constructor(
private readonly repository: CatalogRepository,
private readonly models: MetadataGenerationModels,
private readonly completer: ModelCompleter,
) {}
private async selectColumns(
databaseId: string,
scope: SensitiveDataSuggestionScope,
targetIds: readonly string[],
): Promise<StructuralColumn[]> {
if (new Set(targetIds).size !== targetIds.length) {
throw new SensitiveDataSuggestionDuplicateTargetIdsError();
}
const tables = await this.repository.listTables(databaseId);
const tableIds = new Set(targetIds);
const selectedTables = scope === "selected_tables"
? tables.filter((table) => tableIds.has(table.id))
: tables;
if (scope === "selected_tables" && selectedTables.length !== targetIds.length) {
throw new SensitiveDataSuggestionTargetNotFoundError("table");
}
const columns = (await Promise.all(selectedTables.map(async (table) => (
(await this.repository.listColumns(databaseId, table.id)).map((column) => (
structuralColumn(table, column)
))
)))).flat();
const columnIds = new Set(targetIds);
const selectedColumns = scope === "selected_columns"
? columns.filter((column) => columnIds.has(column.columnId))
: columns;
if (scope === "selected_columns" && selectedColumns.length !== targetIds.length) {
throw new SensitiveDataSuggestionTargetNotFoundError("column");
}
if (selectedColumns.length === 0) {
throw new SensitiveDataSuggestionNoEligibleColumnsError(scope);
}
return selectedColumns;
}
async suggest(
databaseId: string,
modelId: string,
scope: SensitiveDataSuggestionScope,
targetIds: readonly string[],
signal: AbortSignal,
onPrepared?: (total: number) => void | Promise<void>,
onProgress?: (processed: number, suggestions: readonly SensitiveDataSuggestion[]) => void | Promise<void>,
onUsage?: (usage: ModelCompletionUsage) => void | Promise<void>,
): Promise<readonly SensitiveDataSuggestion[]> {
const database = await this.repository.get(databaseId);
if (!database) throw new SensitiveDataSuggestionTargetNotFoundError("database");
const columns = await this.selectColumns(databaseId, scope, targetIds);
await onPrepared?.(columns.length);
const model = this.models.resolve(modelId);
const suggestions: SensitiveDataSuggestion[] = [];
for (const batch of batchesFor(database, columns)) {
let received: Map<string, { columnId: string; sensitive: boolean }> | undefined;
for (let attempt = 0; attempt < 2 && !received; attempt += 1) {
const completion = await this.completer.complete({
model,
signal,
messages: [systemMessage, { role: "user", content: userContent(database, batch) }],
});
const result: ModelCompletionResult = typeof completion === "string"
? { content: completion, usage: { input: 0, cacheRead: 0, output: 0 } }
: completion;
await onUsage?.(result.usage);
const content = result.content;
try {
const parsed = responseSchema.parse(JSON.parse(content));
const expected = new Set(batch.map((column) => column.columnId));
const candidate = new Map(parsed.suggestions.map((suggestion) => [suggestion.columnId, suggestion]));
if (candidate.size !== parsed.suggestions.length
|| candidate.size !== expected.size
|| [...candidate.keys()].some((columnId) => !expected.has(columnId))) {
throw new SensitiveDataSuggestionInvalidResponseError();
}
received = candidate;
} catch {
if (attempt === 1) throw new SensitiveDataSuggestionInvalidResponseError();
}
}
suggestions.push(...batch.map((column) => ({
columnId: column.columnId,
tableId: column.tableId,
tableName: column.table,
columnName: column.column,
version: column.version,
currentSensitive: column.currentSensitive,
sensitive: received!.get(column.columnId)!.sensitive,
})));
await onProgress?.(suggestions.length, suggestions.slice(-batch.length));
}
return suggestions;
}
}
@@ -1,136 +0,0 @@
import type {
SensitiveDataSuggestion,
} from "./sensitive-data-suggester.js";
import {
SensitiveDataSuggester,
SensitiveDataSuggestionTargetNotFoundError,
} from "./sensitive-data-suggester.js";
import type {
CatalogRepository,
SensitiveDataSuggestionRun,
SensitiveDataSuggestionScope,
} from "./types.js";
import type { ModelCompletionUsage } from "./model-completer.js";
const interruptedMessage = "Sensitive-field suggestion generation was interrupted by backend restart.";
const failedMessage = "Sensitive-field suggestion generation failed.";
export interface SensitiveDataSuggestionRunResult {
suggestions: readonly SensitiveDataSuggestion[];
run: SensitiveDataSuggestionRun;
}
export class SensitiveDataSuggestionRunner {
constructor(
private readonly repository: CatalogRepository,
private readonly suggester: SensitiveDataSuggester,
) {}
async initialize(): Promise<void> {
if (!(await this.repository.available())) return;
const interrupted = await this.repository.interruptActiveSensitiveDataSuggestionRuns(
interruptedMessage,
);
for (const run of interrupted) {
await this.repository.appendSensitiveDataSuggestionEvent(
run.id,
"warning",
interruptedMessage,
);
}
}
async run(
databaseId: string,
modelId: string,
scope: SensitiveDataSuggestionScope,
targetIds: readonly string[],
signal: AbortSignal,
): Promise<SensitiveDataSuggestionRunResult> {
if (!(await this.repository.get(databaseId))) {
throw new SensitiveDataSuggestionTargetNotFoundError("database");
}
const started = await this.repository.createSensitiveDataSuggestionRun(
databaseId,
scope,
modelId,
);
try {
await this.repository.appendSensitiveDataSuggestionEvent(
started.id,
"info",
"Sensitive-field suggestion generation started.",
);
const suggestions = await this.suggester.suggest(
databaseId,
modelId,
scope,
targetIds,
signal,
async (total) => {
const prepared = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
total,
});
if (!prepared) throw new Error("Sensitive Data Suggestion Run disappeared");
},
async (processed, batch) => {
const suggestedSensitive = batch.filter((suggestion) => suggestion.sensitive).length;
const suggestedNonSensitive = batch.length - suggestedSensitive;
const current = await this.repository.getSensitiveDataSuggestionRun(started.id);
if (!current) throw new Error("Sensitive Data Suggestion Run disappeared");
const progress = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
});
if (!progress) throw new Error("Sensitive Data Suggestion Run disappeared");
await this.repository.appendSensitiveDataSuggestionEvent(
started.id,
"info",
`Classified ${processed} of ${progress.total} columns.`,
);
},
async (usage: ModelCompletionUsage) => {
const current = await this.repository.getSensitiveDataSuggestionRun(started.id);
if (!current) throw new Error("Sensitive Data Suggestion Run disappeared");
await this.repository.updateSensitiveDataSuggestionRun(started.id, {
inputTokens: current.inputTokens + usage.input,
cacheReadTokens: current.cacheReadTokens + usage.cacheRead,
outputTokens: current.outputTokens + usage.output,
});
},
);
const suggestedSensitive = suggestions.filter((suggestion) => suggestion.sensitive).length;
const suggestedNonSensitive = suggestions.length - suggestedSensitive;
await this.repository.appendSensitiveDataSuggestionEvent(
started.id,
"info",
`Sensitive-field suggestion generation completed for ${suggestions.length} column${
suggestions.length === 1 ? "" : "s"
}.`,
);
const completed = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
status: "completed",
total: suggestions.length,
suggestedSensitive,
suggestedNonSensitive,
finishedAt: new Date().toISOString(),
errorSummary: null,
});
if (!completed) throw new Error("Sensitive Data Suggestion Run disappeared");
return { suggestions, run: completed };
} catch (error) {
await this.repository.updateSensitiveDataSuggestionRun(started.id, {
status: "failed",
finishedAt: new Date().toISOString(),
errorSummary: failedMessage,
}).catch(() => undefined);
await this.repository.appendSensitiveDataSuggestionEvent(
started.id,
"error",
failedMessage,
).catch(() => undefined);
throw error;
}
}
}
@@ -0,0 +1,172 @@
import type {
SensitivityReviewItem,
} from "./sensitivity-analysis-service.js";
import {
SENSITIVITY_POLICY_VERSION,
SensitivityAnalysisInterruptedError,
SensitivityAnalysisService,
SensitivityAnalysisTargetNotFoundError,
} from "./sensitivity-analysis-service.js";
import type {
CatalogRepository,
SensitivityAnalysisRun,
SensitivityAnalysisScope,
} from "./types.js";
const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart.";
const deadlineMessage = "Local sensitivity analysis reached its time limit.";
const failedMessage = "Local sensitivity analysis failed.";
function ensureActive(signal: AbortSignal): void {
if (signal.aborted) throw new SensitivityAnalysisInterruptedError();
}
export interface SensitivityAnalysisRunResult {
suggestions: readonly SensitivityReviewItem[];
run: SensitivityAnalysisRun;
}
export class SensitivityAnalysisRunner {
constructor(
private readonly repository: CatalogRepository,
private readonly analysis: SensitivityAnalysisService,
) {}
async initialize(): Promise<void> {
if (!(await this.repository.available())) return;
const interrupted = await this.repository.interruptActiveSensitivityAnalysisRuns(
interruptedMessage,
);
for (const run of interrupted) {
await this.repository.appendSensitivityAnalysisEvent(
run.id,
"warning",
interruptedMessage,
);
}
}
async run(
databaseId: string,
scope: SensitivityAnalysisScope,
targetIds: readonly string[],
signal: AbortSignal,
): Promise<SensitivityAnalysisRunResult> {
ensureActive(signal);
const database = await this.repository.get(databaseId);
ensureActive(signal);
if (!database) {
throw new SensitivityAnalysisTargetNotFoundError("database");
}
ensureActive(signal);
const started = await this.repository.createSensitivityAnalysisRun(
databaseId,
scope,
{ engine: "local", policyVersion: SENSITIVITY_POLICY_VERSION },
);
let preparedTotal = 0;
let processedSensitive = 0;
let processedNonSensitive = 0;
try {
ensureActive(signal);
await this.repository.appendSensitivityAnalysisEvent(
started.id,
"info",
"Local sensitivity analysis started.",
);
ensureActive(signal);
const suggestions = await this.analysis.analyze(
databaseId,
scope,
targetIds,
signal,
async (total) => {
ensureActive(signal);
preparedTotal = total;
const prepared = await this.repository.updateSensitivityAnalysisRun(started.id, {
total,
});
ensureActive(signal);
if (!prepared) throw new Error("Sensitivity Analysis Run disappeared");
},
async (processed, batch) => {
ensureActive(signal);
const suggestedSensitive = batch.filter(
(suggestion) => suggestion.assessment === "sensitive",
).length;
const suggestedNonSensitive = batch.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length;
const current = await this.repository.getSensitivityAnalysisRun(started.id);
ensureActive(signal);
if (!current) throw new Error("Sensitivity Analysis Run disappeared");
const progress = await this.repository.updateSensitivityAnalysisRun(started.id, {
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
unknown: current.unknown + unknown,
});
if (!progress) throw new Error("Sensitivity Analysis Run disappeared");
processedSensitive += suggestedSensitive;
processedNonSensitive += suggestedNonSensitive;
ensureActive(signal);
await this.repository.appendSensitivityAnalysisEvent(
started.id,
"info",
`Assessed ${processed} of ${progress.total} columns locally.`,
);
ensureActive(signal);
},
);
ensureActive(signal);
const suggestedSensitive = suggestions.filter(
(suggestion) => suggestion.assessment === "sensitive",
).length;
const suggestedNonSensitive = suggestions.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length;
await this.repository.appendSensitivityAnalysisEvent(
started.id,
"info",
`Local sensitivity analysis completed for ${suggestions.length} column${
suggestions.length === 1 ? "" : "s"
}.`,
);
ensureActive(signal);
const completed = await this.repository.updateSensitivityAnalysisRun(started.id, {
status: "completed",
total: suggestions.length,
suggestedSensitive,
suggestedNonSensitive,
unknown,
finishedAt: new Date().toISOString(),
errorSummary: null,
});
ensureActive(signal);
if (!completed) throw new Error("Sensitivity Analysis Run disappeared");
return { suggestions, run: completed };
} catch (error) {
const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError;
const message = interrupted ? deadlineMessage : failedMessage;
await this.repository.updateSensitivityAnalysisRun(started.id, {
status: interrupted ? "interrupted" : "failed",
...(interrupted ? {
total: preparedTotal,
suggestedSensitive: processedSensitive,
suggestedNonSensitive: processedNonSensitive,
unknown: Math.max(0, preparedTotal - processedSensitive - processedNonSensitive),
} : {}),
finishedAt: new Date().toISOString(),
errorSummary: message,
}).catch(() => undefined);
await this.repository.appendSensitivityAnalysisEvent(
started.id,
interrupted ? "warning" : "error",
message,
).catch(() => undefined);
throw error;
}
}
}
@@ -0,0 +1,175 @@
import type {
SensitivityClassifier,
SensitivityColumnAssessment,
SensitivityEvidence,
SensitivityNerBudget,
} from "./sensitivity-classifier.js";
import type {
CatalogColumn,
CatalogRepository,
CatalogTable,
SensitivityAnalysisScope,
} from "./types.js";
export type { SensitivityAnalysisScope } from "./types.js";
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1";
interface SelectedColumn {
table: CatalogTable;
column: CatalogColumn;
}
export interface SensitivityReviewItem {
columnId: string;
tableId: string;
tableName: string;
columnName: string;
version: number;
currentSensitive: boolean;
sensitive: boolean;
assessment: SensitivityColumnAssessment["assessment"];
evidence: readonly SensitivityEvidence[];
observedValues: number;
}
export class SensitivityAnalysisTargetNotFoundError extends Error {
constructor(readonly target: "database" | "table" | "column") {
super(`${target} not found`);
this.name = "SensitivityAnalysisTargetNotFoundError";
}
}
export class SensitivityAnalysisDuplicateTargetIdsError extends Error {
constructor() {
super("sensitivity analysis target IDs must be unique");
this.name = "SensitivityAnalysisDuplicateTargetIdsError";
}
}
export class SensitivityAnalysisInterruptedError extends Error {
constructor() {
super("sensitivity analysis deadline exceeded");
this.name = "SensitivityAnalysisInterruptedError";
}
}
function ensureActive(signal: AbortSignal): void {
if (signal.aborted) throw new SensitivityAnalysisInterruptedError();
}
export class SensitivityAnalysisNoEligibleColumnsError extends Error {
constructor(readonly scope: SensitivityAnalysisScope) {
super("selected scope has no catalog columns");
this.name = "SensitivityAnalysisNoEligibleColumnsError";
}
}
/** Selection and table orchestration around the single SensitivityClassifier decision module. */
export class SensitivityAnalysisService {
constructor(
private readonly repository: CatalogRepository,
private readonly classifier: SensitivityClassifier,
private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {},
) {}
private async selectColumns(
databaseId: string,
scope: SensitivityAnalysisScope,
targetIds: readonly string[],
signal: AbortSignal,
): Promise<readonly SelectedColumn[]> {
ensureActive(signal);
if (new Set(targetIds).size !== targetIds.length) {
throw new SensitivityAnalysisDuplicateTargetIdsError();
}
const tables = await this.repository.listTables(databaseId);
ensureActive(signal);
const tableIds = new Set(targetIds);
const selectedTables = scope === "selected_tables"
? tables.filter((table) => tableIds.has(table.id))
: tables;
if (scope === "selected_tables" && selectedTables.length !== targetIds.length) {
throw new SensitivityAnalysisTargetNotFoundError("table");
}
const columns = (await Promise.all(selectedTables.map(async (table) => (
(await this.repository.listColumns(databaseId, table.id)).map((column) => ({ table, column }))
)))).flat();
ensureActive(signal);
const columnIds = new Set(targetIds);
const selectedColumns = scope === "selected_columns"
? columns.filter(({ column }) => columnIds.has(column.id))
: columns;
if (scope === "selected_columns" && selectedColumns.length !== targetIds.length) {
throw new SensitivityAnalysisTargetNotFoundError("column");
}
if (selectedColumns.length === 0) {
throw new SensitivityAnalysisNoEligibleColumnsError(scope);
}
return selectedColumns;
}
async analyze(
databaseId: string,
scope: SensitivityAnalysisScope,
targetIds: readonly string[],
signal: AbortSignal,
onPrepared?: (total: number) => void | Promise<void>,
onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise<void>,
): Promise<readonly SensitivityReviewItem[]> {
const now = this.options.now ?? Date.now;
const deadline = now() + (this.options.runBudgetMs ?? 60_000);
const configuredNerBudget = this.options.nerBudgetMs ?? 10_000;
const nerBudget: SensitivityNerBudget = {
remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0
? configuredNerBudget
: 10_000,
};
ensureActive(signal);
const database = await this.repository.get(databaseId);
ensureActive(signal);
if (!database) throw new SensitivityAnalysisTargetNotFoundError("database");
const selected = await this.selectColumns(databaseId, scope, targetIds, signal);
await onPrepared?.(selected.length);
ensureActive(signal);
const byTable = new Map<string, SelectedColumn[]>();
for (const item of selected) {
const items = byTable.get(item.table.id) ?? [];
items.push(item);
byTable.set(item.table.id, items);
}
const suggestions: SensitivityReviewItem[] = [];
for (const items of byTable.values()) {
ensureActive(signal);
const first = items[0]!;
const assessments = await this.classifier.assessTable({
database,
table: first.table,
columns: items.map(({ column }) => column),
}, signal, deadline, nerBudget);
ensureActive(signal);
const assessmentById = new Map(assessments.map((assessment) => [
assessment.columnId,
assessment,
]));
const batch = items.map(({ table, column }) => {
const assessment = assessmentById.get(column.id)!;
return {
columnId: column.id,
tableId: table.id,
tableName: table.name,
columnName: column.name,
version: column.version,
currentSensitive: column.sensitive,
sensitive: assessment.proposedSensitive,
assessment: assessment.assessment,
evidence: assessment.evidence,
observedValues: assessment.observedValues,
};
});
suggestions.push(...batch);
await onProgress?.(suggestions.length, batch);
ensureActive(signal);
}
return suggestions;
}
}
@@ -0,0 +1,441 @@
import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js";
import { findPhoneNumbersInText } from "libphonenumber-js/max";
import validator from "validator";
export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown";
export interface SensitivityEvidence {
kind: "metadata" | "content" | "length" | "ner" | "coverage";
ruleId: string;
label?: string;
confidence?: number;
}
export interface SensitivityValueObservation {
columnId: string;
value: string | null;
characterLength: number | null;
}
export interface SensitivityScanCoverage {
kind: "complete" | "sampled" | "unavailable";
observedRows: number;
}
export interface SensitivityTableScan {
batches: readonly (readonly SensitivityValueObservation[])[];
coverage: SensitivityScanCoverage;
}
export interface SensitivityScanRequest {
database: WorkspaceDatabase;
table: CatalogTable;
columns: readonly CatalogColumn[];
fullScanBudgetMs: number;
deadline: number;
}
export interface SensitivityValueSource {
scanTable(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
signal: AbortSignal,
): Promise<SensitivityScanCoverage>;
}
export interface LocalNerCandidate {
columnId: string;
text: string;
}
export interface LocalNerEvidence {
columnId: string;
label: string;
confidence: number;
}
export interface SensitivityNerBudget {
remainingMs: number;
}
/** Optional local detector. It returns evidence only; it never decides a column assessment. */
export interface LocalNerDetector {
warmup?(): Promise<void>;
isReady?(): boolean;
detect(
candidates: readonly LocalNerCandidate[],
signal: AbortSignal,
deadline: number,
): Promise<readonly LocalNerEvidence[]>;
close?(): Promise<void>;
}
export interface SensitivityColumnAssessment {
columnId: string;
assessment: SensitivityAssessment;
proposedSensitive: boolean;
evidence: readonly SensitivityEvidence[];
observedValues: number;
}
export interface SensitivityTableTarget {
database: WorkspaceDatabase;
table: CatalogTable;
columns: readonly CatalogColumn[];
}
const EMAIL = /(?<![\p{L}\p{N}._%+-])[\p{L}\p{N}._%+-]+@[\p{L}\p{N}.-]+\.[\p{L}]{2,63}(?![\p{L}\p{N}._%+-])/giu;
const DIRECT_IDENTIFIER_NAMES = new Set([
"address", "birth_date", "codice_fiscale", "date_of_birth", "dob", "email", "e_mail",
"bic", "first_name", "fiscal_code", "full_name", "iban", "indirizzo", "last_name", "mobile",
"nome", "passport", "phone", "surname", "swift", "swift_code", "tax_id", "telefono",
]);
const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_key|pwd|secret|token)(?:_|$)/u;
const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u;
const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu;
const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu;
const MAX_NER_CANDIDATES_PER_REQUEST = 128;
function normalizedName(value: string): string {
return value.normalize("NFKD")
.replace(/[\u0300-\u036f]/g, "")
.replace(/([a-z0-9])([A-Z])/g, "$1_$2")
.toLocaleLowerCase("en-US")
.replace(/[^a-z0-9]+/g, "_")
.replace(/^_+|_+$/g, "");
}
function boundedCount(value: number | undefined, fallback: number, maximum: number): number {
return value === undefined || !Number.isSafeInteger(value)
? fallback
: Math.max(1, Math.min(value, maximum));
}
function metadataEvidence(column: CatalogColumn): SensitivityEvidence | undefined {
const ruleId = sensitiveNameRule(column.name);
return ruleId ? { kind: "metadata", ruleId } : undefined;
}
function sensitiveNameRule(value: string): string | undefined {
const name = normalizedName(value);
if (DIRECT_IDENTIFIER_NAMES.has(name)) {
return "metadata.direct_identifier";
}
if (CREDENTIAL_NAME.test(name)) {
return "metadata.credential";
}
if (HEALTH_NAME.test(name)) {
return "metadata.health";
}
return undefined;
}
const ITALIAN_FISCAL_CODE = /(?<![A-Z0-9])[A-Z]{6}[0-9LMNPQRSTUV]{2}[ABCDEHLMPRST][0-9LMNPQRSTUV]{2}[A-Z][0-9LMNPQRSTUV]{3}[A-Z](?![A-Z0-9])/giu;
const FISCAL_ODD: Record<string, number> = {
"0": 1, "1": 0, "2": 5, "3": 7, "4": 9, "5": 13, "6": 15, "7": 17, "8": 19, "9": 21,
A: 1, B: 0, C: 5, D: 7, E: 9, F: 13, G: 15, H: 17, I: 19, J: 21,
K: 2, L: 4, M: 18, N: 20, O: 11, P: 3, Q: 6, R: 8, S: 12, T: 14,
U: 16, V: 10, W: 22, X: 25, Y: 24, Z: 23,
};
function validItalianFiscalCode(candidate: string): boolean {
const value = candidate.toUpperCase();
if (value.length !== 16) return false;
let sum = 0;
for (let index = 0; index < 15; index += 1) {
const character = value[index]!;
if (index % 2 === 0) sum += FISCAL_ODD[character] ?? -1000;
else sum += /\d/u.test(character) ? Number(character) : character.charCodeAt(0) - 65;
}
return String.fromCharCode(65 + (sum % 26)) === value[15];
}
function validIban(candidate: string): boolean {
const value = candidate.replace(/\s+/gu, "").toUpperCase();
if (!/^[A-Z]{2}\d{2}[A-Z0-9]{11,30}$/u.test(value)) return false;
const rearranged = value.slice(4) + value.slice(0, 4);
let remainder = 0;
for (const character of rearranged) {
const digits = /\d/u.test(character) ? character : String(character.charCodeAt(0) - 55);
for (const digit of digits) remainder = (remainder * 10 + Number(digit)) % 97;
}
return remainder === 1;
}
function validPaymentCard(candidate: string): boolean {
const digits = candidate.replace(/[ -]/gu, "");
if (!/^\d{13,19}$/u.test(digits) || /^(\d)\1+$/u.test(digits)) return false;
let sum = 0;
let double = false;
for (let index = digits.length - 1; index >= 0; index -= 1) {
let digit = Number(digits[index]);
if (double) {
digit *= 2;
if (digit > 9) digit -= 9;
}
sum += digit;
double = !double;
}
return sum % 10 === 0;
}
function jsonHasSensitiveKey(value: string): boolean {
const trimmed = value.trim();
if (!(trimmed.startsWith("{") || trimmed.startsWith("["))) return false;
try {
const pending: Array<{ value: unknown; depth: number }> = [{ value: JSON.parse(trimmed), depth: 0 }];
let visited = 0;
while (pending.length > 0 && visited < 1_000) {
const item = pending.pop()!;
visited += 1;
if (item.depth > 8 || item.value === null || typeof item.value !== "object") continue;
if (Array.isArray(item.value)) {
for (const child of item.value) pending.push({ value: child, depth: item.depth + 1 });
continue;
}
for (const [key, child] of Object.entries(item.value)) {
if (sensitiveNameRule(key)) return true;
pending.push({ value: child, depth: item.depth + 1 });
}
}
} catch {
return false;
}
return false;
}
function contentEvidence(value: string): SensitivityEvidence | undefined {
if (/-----BEGIN (?:[A-Z0-9]+ )?PRIVATE KEY-----/u.test(value)) {
return { kind: "content", ruleId: "credential.private_key" };
}
if (/(?:^|[^A-Z0-9])AKIA[A-Z0-9]{16}(?![A-Z0-9])/u.test(value)
|| /(?:^|[^A-Za-z0-9_])gh[pousr]_[A-Za-z0-9_]{30,}(?![A-Za-z0-9_])/u.test(value)
|| /(?:^|[^A-Za-z0-9_-])eyJ[A-Za-z0-9_-]{5,}\.[A-Za-z0-9_-]{5,}\.[A-Za-z0-9_-]{5,}(?![A-Za-z0-9_-])/u.test(value)) {
return { kind: "content", ruleId: "credential.access_key" };
}
if (/(?:^|[^\p{L}\p{N}_])(?:api[_ -]?key|access[_ -]?token|password|passwd|pwd|secret)\s*[:=]\s*[^\s,;]{4,}/iu.test(value)) {
return { kind: "content", ruleId: "credential.key_value" };
}
if (CLINICAL_TERM.test(value)) return { kind: "content", ruleId: "health.clinical_term" };
for (const match of value.matchAll(EMAIL)) {
if (validator.isEmail(match[0])) return { kind: "content", ruleId: "pii.email" };
}
for (const match of value.matchAll(ITALIAN_FISCAL_CODE)) {
if (validItalianFiscalCode(match[0])) {
return { kind: "content", ruleId: "pii.italian_fiscal_code" };
}
}
for (const match of value.matchAll(/\b(?:passaporto|passport)(?:\s+(?:numero|number|n\.?))?\s*[:#-]?\s*([A-Z0-9]{9})\b/giu)) {
if (validator.isPassportNumber(match[1]!, "IT")) {
return { kind: "content", ruleId: "pii.passport_number" };
}
}
for (const match of value.matchAll(/\bC[A-Z]\d{5}[A-Z]{2}\b/giu)) {
if (validator.isIdentityCard(match[0], "IT")) {
return { kind: "content", ruleId: "pii.identity_card" };
}
}
if (/\b(?:patente(?:\s+di\s+guida)?|driving\s+licen[cs]e)(?:\s+(?:numero|number|n\.?))?\s*[:#-]?\s*[A-Z0-9]{8,12}\b/iu.test(value)) {
return { kind: "content", ruleId: "pii.drivers_license_number" };
}
for (const match of value.matchAll(/(?<![A-Z0-9])[A-Z]{2}\d{2}(?:\s?[A-Z0-9]){11,30}(?![A-Z0-9])/giu)) {
if (validIban(match[0])) return { kind: "content", ruleId: "financial.iban" };
}
for (const match of value.matchAll(/(?<!\d)(?:\d[ -]?){13,19}(?!\d)/gu)) {
if (validPaymentCard(match[0])) {
return { kind: "content", ruleId: "financial.payment_card" };
}
}
for (const match of value.matchAll(/(?<![A-Z0-9])[A-Z]{6}[A-Z0-9]{2}(?:[A-Z0-9]{3})?(?![A-Z0-9])/giu)) {
const before = value.slice(Math.max(0, (match.index ?? 0) - 24), match.index ?? 0);
if (/\b(?:bic|swift)\s*[:=-]?\s*$/iu.test(before) && validator.isBIC(match[0])) {
return { kind: "content", ruleId: "financial.bic" };
}
}
for (const match of value.matchAll(/(?<!\d)(?:IT[ .-]?)?\d{11}(?!\d)/giu)) {
const candidate = match[0].replace(/[ .-]/gu, "");
if (validator.isVAT(candidate.replace(/^IT/iu, ""), "IT")) {
return { kind: "content", ruleId: "pii.italian_vat" };
}
}
for (const match of value.matchAll(/(?<![A-F0-9])(?:[A-F0-9]{2}[:-]){5}[A-F0-9]{2}(?![A-F0-9])/giu)) {
if (validator.isMACAddress(match[0])) {
return { kind: "content", ruleId: "network.mac_address" };
}
}
for (const match of value.matchAll(/(?<![A-F0-9:.])[A-F0-9:.]{3,45}(?![A-F0-9:.])/giu)) {
if (validator.isIP(match[0])) return { kind: "content", ruleId: "network.ip_address" };
}
for (const match of value.matchAll(/(?<![A-F0-9-])[0-9A-F]{8}-[0-9A-F]{4}-[1-8][0-9A-F]{3}-[89AB][0-9A-F]{3}-[0-9A-F]{12}(?![A-F0-9-])/giu)) {
if (validator.isUUID(match[0])) return { kind: "content", ruleId: "pii.uuid" };
}
for (const match of value.matchAll(/\b(?:https?|ftp):\/\/[^\s<>"']+/giu)) {
const candidate = match[0].replace(/[.,;:!?\])}]+$/u, "");
if (validator.isURL(candidate, { require_protocol: true })) {
return { kind: "content", ruleId: "network.url" };
}
}
if (findPhoneNumbersInText(value, "IT").some((match) => match.number.isValid())) {
return { kind: "content", ruleId: "pii.phone_number" };
}
if (jsonHasSensitiveKey(value)) {
return { kind: "content", ruleId: "pii.json_sensitive_key" };
}
return undefined;
}
/** Sole decision module for local column-level sensitivity assessments. */
export class SensitivityClassifier {
constructor(
private readonly values: SensitivityValueSource,
private readonly detector?: LocalNerDetector,
private readonly options: {
fullScanBudgetMs?: number;
runBudgetMs?: number;
nerConfidenceThreshold?: number;
maxNerValuesPerColumn?: number;
maxNerCandidatesPerTable?: number;
now?: () => number;
} = {},
) {}
async assessTable(
target: SensitivityTableTarget,
signal: AbortSignal,
runDeadline?: number,
nerBudget?: SensitivityNerBudget,
): Promise<readonly SensitivityColumnAssessment[]> {
const now = this.options.now ?? Date.now;
const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000);
const evidence = new Map(target.columns.map((column) => {
const match = metadataEvidence(column);
return [column.id, match ? [match] : [] as SensitivityEvidence[]];
}));
const observed = new Map(target.columns.map((column) => [column.id, 0]));
const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]]));
const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8);
const unsupported = new Set(target.columns
.filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType))
.map((column) => column.id));
const scannableColumns = target.columns.filter((column) => (
!unsupported.has(column.id) && evidence.get(column.id)!.length === 0
));
let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 };
if (scannableColumns.length > 0 && now() < deadline) {
try {
coverage = await this.values.scanTable({
...target,
columns: scannableColumns,
fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000,
deadline,
}, (batch) => {
for (const item of batch) {
if (!evidence.has(item.columnId) || item.value === null) continue;
observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1);
const matches = evidence.get(item.columnId)!;
if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) {
matches.push({ kind: "length", ruleId: "text.over_500_characters" });
} else if (matches.length === 0) {
const match = contentEvidence(item.value);
if (match) matches.push(match);
else {
const candidates = nerCandidates.get(item.columnId)!;
if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) {
candidates.push(item.value);
}
}
}
}
}, signal);
} catch (error) {
if (!(error instanceof CatalogConnectorError)) throw error;
}
}
if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
&& now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) {
const candidates: LocalNerCandidate[] = [];
const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
for (const column of target.columns) {
if (evidence.get(column.id)!.length > 0) continue;
const text = nerCandidates.get(column.id)![valueIndex];
if (text === undefined) continue;
candidates.push({ columnId: column.id, text });
if (candidates.length >= maxCandidates) break candidateSelection;
}
}
if (candidates.length > 0) {
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
const nerStartedAt = now();
const allowedNerMs = nerBudget
? Math.max(0, nerBudget.remainingMs)
: Math.max(0, deadline - nerStartedAt);
const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs);
try {
for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) {
if (signal.aborted || now() >= nerDeadline) break;
try {
const detected = await this.detector.detect(
candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST),
signal,
nerDeadline,
);
for (const item of detected) {
const matches = evidence.get(item.columnId);
if (!matches || matches.length > 0 || !Number.isFinite(item.confidence)
|| item.confidence < threshold || item.confidence > 1) continue;
const label = normalizedName(item.label).slice(0, 80);
if (!label) continue;
matches.push({
kind: "ner",
ruleId: "ner.entity",
label,
confidence: item.confidence,
});
}
} catch {
// NER is optional: deterministic findings and scan coverage remain authoritative.
break;
}
}
} finally {
if (nerBudget) {
const elapsedMs = Math.max(1, now() - nerStartedAt);
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs);
}
}
}
}
return target.columns.map((column) => {
const matches = evidence.get(column.id)!;
const count = observed.get(column.id) ?? 0;
const assessment: SensitivityAssessment = matches.length > 0
? "sensitive"
: unsupported.has(column.id) || count === 0 || coverage.kind !== "complete"
? "unknown"
: "non_sensitive";
return {
columnId: column.id,
assessment,
proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive",
evidence: matches.length > 0
? matches
: assessment === "unknown"
? [{
kind: "coverage",
ruleId: unsupported.has(column.id)
? "coverage.unsupported_type"
: coverage.kind === "unavailable"
? "coverage.unavailable"
: count === 0
? "coverage.no_values"
: "coverage.incomplete",
}]
: [],
observedValues: count,
};
});
}
}
+94
View File
@@ -0,0 +1,94 @@
import { dirname } from "node:path";
import { fileURLToPath } from "node:url";
import { loadConfig } from "../config.js";
import { WorkspaceSecretStore } from "../workspaces/secret-store.js";
import { PythonLocalNerDetector } from "./local-ner-detector.js";
import { ConcreteCatalogPostgresAccess } from "./postgres-access.js";
import { createCatalogRepository } from "./repository.js";
import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js";
import { SensitivityClassifier } from "./sensitivity-classifier.js";
import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js";
const WORKSPACE_ID = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/u;
async function main(): Promise<void> {
const workspaceId = process.argv[2];
if (!workspaceId || !WORKSPACE_ID.test(workspaceId)) {
process.stderr.write("Usage: sensitivity-shadow <workspace-id>\n");
process.exitCode = 2;
return;
}
let detector: PythonLocalNerDetector | undefined;
let stage = "configuration";
try {
const config = loadConfig(process.env);
stage = "catalog";
const repository = createCatalogRepository(config.catalogDatabase);
if (!(await repository.available())) throw new Error("catalog unavailable");
const database = await repository.getByWorkspace(workspaceId);
if (!database) throw new Error("database unavailable");
stage = "source";
const secretStore = new WorkspaceSecretStore({
root: config.workspaceSecretStoreRoot,
runtimeRoot: config.workspaceSecretRuntimeRoot,
installationId: config.workspaceRegistry.installationId,
});
const access = new ConcreteCatalogPostgresAccess(secretStore, {
connectTimeoutMs: config.workspaceDiagnosticTimeoutMs,
});
const source = new ConcreteSensitivityValueSource(access, secretStore);
if (config.sensitivityNer) {
const workerScript = config.sensitivityNer.workerScript
?? fileURLToPath(new URL("../../python/sensitivity_ner_worker.py", import.meta.url));
detector = new PythonLocalNerDetector({
pythonExecutable: config.sensitivityNer.pythonExecutable,
workerScript,
modelPath: config.sensitivityNer.modelPath,
cwd: dirname(workerScript),
threads: config.sensitivityNer.threads,
});
try {
await detector.warmup();
} catch {
await detector.close();
detector = undefined;
}
}
const startedAt = Date.now();
stage = "analysis";
const suggestions = await new SensitivityAnalysisService(
repository,
new SensitivityClassifier(source, detector),
).analyze(database.id, "all", [], AbortSignal.timeout(65_000));
const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 };
const rules = new Map<string, number>();
for (const suggestion of suggestions) {
if (suggestion.assessment === "sensitive") assessments.sensitive += 1;
else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1;
else assessments.unknown += 1;
for (const evidence of suggestion.evidence) {
rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1);
}
}
process.stdout.write(`${JSON.stringify({
ok: true,
policyVersion: "sensitivity-v1",
nerEnabled: detector !== undefined,
total: suggestions.length,
assessments,
rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))),
elapsedMs: Date.now() - startedAt,
})}\n`);
} catch {
process.stdout.write(`${JSON.stringify({
ok: false,
code: `sensitivity_shadow_${stage}_failed`,
})}\n`);
process.exitCode = 1;
} finally {
await detector?.close();
}
}
await main();
@@ -0,0 +1,259 @@
import { readFile } from "node:fs/promises";
import type { WorkspaceSecretStore } from "../workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "./secrets.js";
import type { CatalogPostgresAccess } from "./postgres-access.js";
import type {
SensitivityScanCoverage,
SensitivityScanRequest,
SensitivityValueObservation,
SensitivityValueSource,
} from "./sensitivity-classifier.js";
import { CatalogConnectorError } from "./types.js";
const MAX_VALUE_CHARACTERS = 501;
const DEFAULT_BATCH_ROWS = 200;
const DEFAULT_SAMPLE_ROWS = 200;
function quoteIdentifier(identifier: string): string {
return `"${identifier.replaceAll('"', '""')}"`;
}
function projections(request: SensitivityScanRequest): string {
return request.columns.flatMap((column, index) => {
const identifier = quoteIdentifier(column.name);
return [
`LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`,
`CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`,
];
}).join(", ");
}
function observations(
request: SensitivityScanRequest,
rows: readonly Record<string, unknown>[],
): SensitivityValueObservation[] {
return rows.flatMap((row) => request.columns.map((column, index) => {
const sourceValue = row[`__value_${index}`];
const sourceLength = row[`__length_${index}`];
const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue);
const parsedLength = sourceLength === null || sourceLength === undefined
? null
: Number(sourceLength);
return {
columnId: column.id,
value,
characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0
? parsedLength
: value?.length ?? null,
};
}));
}
function cancelled(error: unknown): boolean {
return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014");
}
interface SensitivityValueSourceOptions {
now?: () => number;
batchRows?: number;
sampleRows?: number;
}
/**
* PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a
* sensitivity decision.
*/
export class ConcreteSensitivityValueSource implements SensitivityValueSource {
private readonly now: () => number;
private readonly batchRows: number;
private readonly sampleRows: number;
constructor(
private readonly access: CatalogPostgresAccess,
private readonly secretStore?: Pick<WorkspaceSecretStore, "materialize">,
options: SensitivityValueSourceOptions = {},
) {
this.now = options.now ?? Date.now;
this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS;
this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS;
}
async scanTable(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
signal: AbortSignal,
): Promise<SensitivityScanCoverage> {
if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 };
if (request.database.binding.transport === "rest_api") {
return await this.scanRest(request, consume, signal);
}
return await this.scanPostgres(request, consume, signal);
}
private async scanPostgres(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
signal: AbortSignal,
): Promise<SensitivityScanCoverage> {
const client = await this.access.connect(request.database, signal);
let transactionOpen = false;
const startedAt = this.now();
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
let observedRows = 0;
let cursorOpen = false;
try {
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows: 0 };
}
await client.query("BEGIN TRANSACTION READ ONLY", []);
transactionOpen = true;
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`,
]);
await client.query("SAVEPOINT sensitivity_full_scan", []);
const cursor = [
"DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR",
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
].join(" ");
await client.query(cursor, []);
cursorOpen = true;
while (!signal.aborted && this.now() < fullDeadline) {
let rows: Array<Record<string, unknown>>;
try {
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`,
]);
rows = (await client.query(
`FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`,
[],
)).rows;
} catch (error) {
if (!cancelled(error)) throw error;
await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []);
cursorOpen = false;
break;
}
if (rows.length > 0) {
observedRows += rows.length;
await consume(observations(request, rows));
}
if (rows.length < this.batchRows) {
return { kind: "complete", observedRows };
}
}
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows };
}
if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []);
await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []);
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(request.deadline - this.now()))}ms`,
]);
const sampleSql = [
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
"LIMIT $1",
].join(" ");
const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows;
observedRows += sampledRows.length;
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
return { kind: "sampled", observedRows };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("Sensitivity source scan failed");
} finally {
if (transactionOpen) await client.query("ROLLBACK", []).catch(() => undefined);
await client.end().catch(() => undefined);
}
}
private async scanRest(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
signal: AbortSignal,
): Promise<SensitivityScanCoverage> {
if (!this.secretStore) throw new CatalogConnectorError("REST sensitivity scanning is not configured");
const auth = request.database.binding.restAuth ?? "bearer";
const materialized = this.secretStore.materialize(
request.database.workspaceId,
auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey],
);
const startedAt = this.now();
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
let observedRows = 0;
try {
const headers: Record<string, string> = { "content-type": "application/json" };
if (auth !== "none") {
const credentialFile = materialized.files.get(CATALOG_SECRET_IDS.apiKey);
if (!credentialFile) throw new CatalogConnectorError("REST API key is not configured");
const credential = (await readFile(credentialFile, "utf8")).trim();
if (auth === "bearer") headers.authorization = `Bearer ${credential}`;
else headers["x-api-key"] = credential;
}
const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, "");
if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete");
const runQuery = async (sql: string, deadline: number): Promise<Array<Record<string, unknown>>> => {
const response = await fetch(`${baseUrl}/rpc/run_query`, {
method: "POST",
headers,
body: JSON.stringify({ query_text: sql }),
signal: AbortSignal.any([
signal,
AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))),
]),
});
if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
const body: unknown = await response.json();
if (!Array.isArray(body)
|| body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) {
throw new CatalogConnectorError("REST sensitivity source response is invalid");
}
return body as Array<Record<string, unknown>>;
};
let offset = 0;
const baseSelect = [
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
].join(" ");
while (!signal.aborted) {
let rows: Array<Record<string, unknown>>;
try {
rows = await runQuery(
`${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`,
fullDeadline,
);
} catch (error) {
if (signal.aborted || this.now() < fullDeadline) throw error;
break;
}
observedRows += rows.length;
if (rows.length > 0) await consume(observations(request, rows));
if (rows.length < this.batchRows) {
return { kind: offset === 0 ? "complete" : "sampled", observedRows };
}
offset += rows.length;
if (this.now() >= fullDeadline) break;
}
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows };
}
const sampleSql = [
baseSelect,
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
`LIMIT ${this.sampleRows}`,
].join(" ");
const sampledRows = await runQuery(sampleSql, request.deadline);
observedRows += sampledRows.length;
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
return { kind: "sampled", observedRows };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("REST sensitivity source scan failed");
} finally {
materialized.release();
}
}
}
+29 -25
View File
@@ -266,18 +266,21 @@ export interface DescriptionGenerationEvent {
createdAt: string;
}
export type SensitiveDataSuggestionScope = "all" | "selected_tables" | "selected_columns";
export type SensitiveDataSuggestionStatus = "running" | "completed" | "failed" | "interrupted";
export type SensitivityAnalysisScope = "all" | "selected_tables" | "selected_columns";
export type SensitivityAnalysisStatus = "running" | "completed" | "failed" | "interrupted";
export interface SensitiveDataSuggestionRun {
export interface SensitivityAnalysisRun {
id: string;
databaseId: string;
scope: SensitiveDataSuggestionScope;
modelId: string;
status: SensitiveDataSuggestionStatus;
scope: SensitivityAnalysisScope;
engine: "llm" | "local";
modelId: string | null;
policyVersion: string | null;
status: SensitivityAnalysisStatus;
total: number;
suggestedSensitive: number;
suggestedNonSensitive: number;
unknown: number;
inputTokens: number;
cacheReadTokens: number;
outputTokens: number;
@@ -288,11 +291,12 @@ export interface SensitiveDataSuggestionRun {
errorSummary: string | null;
}
export interface SensitiveDataSuggestionRunUpdate {
status?: SensitiveDataSuggestionStatus;
export interface SensitivityAnalysisRunUpdate {
status?: SensitivityAnalysisStatus;
total?: number;
suggestedSensitive?: number;
suggestedNonSensitive?: number;
unknown?: number;
finishedAt?: string | null;
errorSummary?: string | null;
inputTokens?: number;
@@ -300,7 +304,7 @@ export interface SensitiveDataSuggestionRunUpdate {
outputTokens?: number;
}
export interface SensitiveDataSuggestionEvent {
export interface SensitivityAnalysisEvent {
runId: string;
sequence: number;
level: "info" | "warning" | "error";
@@ -491,29 +495,29 @@ export interface CatalogRepository {
runId: string,
afterSequence?: number,
): Promise<DescriptionGenerationEvent[]>;
createSensitiveDataSuggestionRun(
createSensitivityAnalysisRun(
databaseId: string,
scope: SensitiveDataSuggestionScope,
modelId: string,
): Promise<SensitiveDataSuggestionRun>;
getSensitiveDataSuggestionRun(runId: string): Promise<SensitiveDataSuggestionRun | undefined>;
listSensitiveDataSuggestionRuns(limit?: number): Promise<SensitiveDataSuggestionRun[]>;
interruptActiveSensitiveDataSuggestionRuns(
scope: SensitivityAnalysisScope,
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
): Promise<SensitivityAnalysisRun>;
getSensitivityAnalysisRun(runId: string): Promise<SensitivityAnalysisRun | undefined>;
listSensitivityAnalysisRuns(limit?: number): Promise<SensitivityAnalysisRun[]>;
interruptActiveSensitivityAnalysisRuns(
errorSummary: string,
): Promise<SensitiveDataSuggestionRun[]>;
updateSensitiveDataSuggestionRun(
): Promise<SensitivityAnalysisRun[]>;
updateSensitivityAnalysisRun(
runId: string,
update: SensitiveDataSuggestionRunUpdate,
): Promise<SensitiveDataSuggestionRun | undefined>;
appendSensitiveDataSuggestionEvent(
update: SensitivityAnalysisRunUpdate,
): Promise<SensitivityAnalysisRun | undefined>;
appendSensitivityAnalysisEvent(
runId: string,
level: SensitiveDataSuggestionEvent["level"],
level: SensitivityAnalysisEvent["level"],
message: string,
): Promise<SensitiveDataSuggestionEvent>;
listSensitiveDataSuggestionEvents(
): Promise<SensitivityAnalysisEvent>;
listSensitivityAnalysisEvents(
runId: string,
afterSequence?: number,
): Promise<SensitiveDataSuggestionEvent[]>;
): Promise<SensitivityAnalysisEvent[]>;
listRelationships(databaseId: string): Promise<CatalogPhysicalRelationship[]>;
listLogicalRelationships(databaseId: string): Promise<CatalogLogicalRelationship[]>;
getLogicalRelationshipContext(databaseId: string): Promise<CatalogLogicalRelationshipContext | undefined>;
+36
View File
@@ -33,6 +33,12 @@ export interface AppConfig {
secretsFile?: string;
installationConfigFile?: string;
modelCatalogFile?: string;
sensitivityNer?: {
pythonExecutable: string;
modelPath: string;
workerScript?: string;
threads: number;
};
piAuthFile?: string;
secretFiles: Readonly<Record<string, string | undefined>>;
modelApiKeyFile?: string;
@@ -354,6 +360,35 @@ export function loadConfig(
|| modelCatalogFile.includes("\0")
|| !path.isAbsolute(modelCatalogFile)
)) throw new Error("runtime model catalog configuration is invalid");
const sensitivityNerModelPath = env.THT_SENSITIVITY_NER_MODEL_PATH;
const sensitivityNerPython = env.THT_SENSITIVITY_NER_PYTHON;
const sensitivityNerWorker = env.THT_SENSITIVITY_NER_WORKER;
for (const [value, label] of [
[sensitivityNerModelPath, "model path"],
[sensitivityNerPython, "Python executable"],
[sensitivityNerWorker, "worker path"],
] as const) {
if (value !== undefined && (
value.length === 0 || value.trim() !== value || value.includes("\0") || !path.isAbsolute(value)
)) throw new Error(`sensitivity NER ${label} configuration is invalid`);
}
if (sensitivityNerModelPath === undefined && (
sensitivityNerPython !== undefined
|| sensitivityNerWorker !== undefined
|| env.THT_SENSITIVITY_NER_THREADS !== undefined
)) throw new Error("sensitivity NER settings require a model path");
const sensitivityNerThreads = Number(env.THT_SENSITIVITY_NER_THREADS ?? 2);
if (!Number.isSafeInteger(sensitivityNerThreads) || sensitivityNerThreads < 1 || sensitivityNerThreads > 8) {
throw new Error("sensitivity NER thread configuration is invalid");
}
const sensitivityNer = sensitivityNerModelPath === undefined
? undefined
: {
modelPath: sensitivityNerModelPath,
pythonExecutable: sensitivityNerPython ?? "/opt/sensitivity-ner/bin/python",
...(sensitivityNerWorker ? { workerScript: sensitivityNerWorker } : {}),
threads: sensitivityNerThreads,
};
const piAuthFile = env.THT_PI_AUTH_FILE;
if (piAuthFile !== undefined && (
piAuthFile.trim() !== piAuthFile || piAuthFile.length === 0 || piAuthFile.includes("\0")
@@ -442,6 +477,7 @@ export function loadConfig(
secretsFile,
installationConfigFile,
modelCatalogFile,
sensitivityNer,
piAuthFile,
secretFiles,
modelApiKeyFile,
@@ -11,38 +11,35 @@ import {
type DescriptionGenerationWorker,
} from "../catalog/description-generation-worker.js";
import { MetadataGenerationModelUnavailableError } from "../catalog/metadata-generation-models.js";
import { ModelCompletionProviderError } from "../catalog/model-completer.js";
import {
SensitiveDataSuggestionDuplicateTargetIdsError,
SensitiveDataSuggestionInvalidResponseError,
SensitiveDataSuggestionNoEligibleColumnsError,
SensitiveDataSuggestionPayloadTooLargeError,
SensitiveDataSuggestionTargetNotFoundError,
} from "../catalog/sensitive-data-suggester.js";
import type { SensitiveDataSuggestionRunner } from "../catalog/sensitive-data-suggestion-runner.js";
SensitivityAnalysisDuplicateTargetIdsError,
SensitivityAnalysisInterruptedError,
SensitivityAnalysisNoEligibleColumnsError,
SensitivityAnalysisTargetNotFoundError,
} from "../catalog/sensitivity-analysis-service.js";
import type { SensitivityAnalysisRunner } from "../catalog/sensitivity-analysis-runner.js";
import {
CatalogOperationInProgressError,
CatalogConnectorError,
CatalogUnavailableError,
DescriptionGenerationRunActiveError,
type CatalogRepository,
type DescriptionGenerationEvent,
type DescriptionGenerationRun,
type SensitiveDataSuggestionEvent,
type SensitiveDataSuggestionRun,
type SensitivityAnalysisEvent,
type SensitivityAnalysisRun,
} from "../catalog/types.js";
const idSchema = z.uuid();
const modelIdSchema = z.string().regex(/^[a-z][a-z0-9._-]{0,63}\/[A-Za-z0-9][A-Za-z0-9._:-]{0,255}$/);
const selectedTargetIdsSchema = z.array(idSchema).min(1);
const suggestionSchema = z.discriminatedUnion("scope", [
z.object({ modelId: modelIdSchema, scope: z.literal("all") }).strict(),
z.object({ scope: z.literal("all") }).strict(),
z.object({
modelId: modelIdSchema,
scope: z.literal("selected_tables"),
targetIds: selectedTargetIdsSchema,
}).strict(),
z.object({
modelId: modelIdSchema,
scope: z.literal("selected_columns"),
targetIds: selectedTargetIdsSchema,
}).strict(),
@@ -112,7 +109,7 @@ function publicRun(run: DescriptionGenerationRun) {
};
}
function publicSensitiveDataSuggestionEvent(event: SensitiveDataSuggestionEvent) {
function publicSensitivityAnalysisEvent(event: SensitivityAnalysisEvent) {
return {
runId: event.runId,
sequence: event.sequence,
@@ -122,16 +119,19 @@ function publicSensitiveDataSuggestionEvent(event: SensitiveDataSuggestionEvent)
};
}
function publicSensitiveDataSuggestionRun(run: SensitiveDataSuggestionRun) {
function publicSensitivityAnalysisRun(run: SensitivityAnalysisRun) {
return {
id: run.id,
databaseId: run.databaseId,
scope: run.scope,
engine: run.engine,
modelId: run.modelId,
policyVersion: run.policyVersion,
status: run.status,
total: run.total,
suggestedSensitive: run.suggestedSensitive,
suggestedNonSensitive: run.suggestedNonSensitive,
unknown: run.unknown,
inputTokens: run.inputTokens,
cacheReadTokens: run.cacheReadTokens,
outputTokens: run.outputTokens,
@@ -232,16 +232,10 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
if (error instanceof CatalogUnavailableError) {
return reply.code(503).send({
code: "catalog_unavailable",
message: "The database catalog is unavailable, so no sensitive-field suggestions were prepared.",
message: "The database catalog is unavailable, so no sensitivity assessments were prepared.",
});
}
if (error instanceof MetadataGenerationModelUnavailableError) {
return reply.code(409).send({
code: "metadata_generation_model_unavailable",
message: "The selected metadata-generation model is unavailable.",
});
}
if (error instanceof SensitiveDataSuggestionTargetNotFoundError) {
if (error instanceof SensitivityAnalysisTargetNotFoundError) {
const code = error.target === "database"
? "database_not_found"
: error.target === "table"
@@ -254,45 +248,65 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
: "One or more selected Catalog Columns were not found in this database.";
return reply.code(404).send({ code, message });
}
if (error instanceof SensitiveDataSuggestionDuplicateTargetIdsError) {
if (error instanceof SensitivityAnalysisDuplicateTargetIdsError) {
return reply.code(400).send({
code: "sensitive_data_suggestion_target_ids_duplicate",
message: "Each selected table or column must appear only once.",
});
}
if (error instanceof SensitiveDataSuggestionNoEligibleColumnsError) {
if (error instanceof SensitivityAnalysisNoEligibleColumnsError) {
return reply.code(409).send({
code: "sensitive_data_suggestion_no_columns",
message: "The selected scope contains no Catalog Columns to classify.",
message: "The selected scope contains no Catalog Columns to assess.",
});
}
if (error instanceof SensitiveDataSuggestionPayloadTooLargeError) {
return reply.code(413).send({
code: "sensitive_data_suggestion_payload_too_large",
message: "The selected structural metadata cannot be divided into safe LLM requests.",
if (error instanceof SensitivityAnalysisInterruptedError) {
return reply.code(504).send({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
});
}
if (error instanceof SensitiveDataSuggestionInvalidResponseError) {
if (error instanceof CatalogConnectorError) {
return reply.code(502).send({
code: "sensitive_data_suggestion_invalid_response",
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
});
}
if (error instanceof ModelCompletionProviderError) {
return reply.code(502).send({
code: "sensitive_data_suggestion_provider_unavailable",
message: "The selected LLM service could not complete the request. No suggestions were applied.",
code: "sensitivity_source_unavailable",
message: "The source values could not be inspected safely. No assessments were applied.",
});
}
if (error instanceof z.ZodError) {
return reply.code(400).send({
code: "sensitive_data_suggestion_request_invalid",
message: "Choose a database, one or more tables, or one or more columns to classify.",
message: "Choose a database, one or more tables, or one or more columns to assess.",
});
}
return reply.code(500).send({
code: "sensitive_data_suggestion_failed",
message: "Sensitive-field suggestions failed before review. No changes were applied.",
message: "Local sensitivity analysis failed before review. No changes were applied.",
});
}
function untilAborted<T>(operation: Promise<T>, signal: AbortSignal): Promise<T> {
if (signal.aborted) {
void operation.catch(() => undefined);
return Promise.reject(new SensitivityAnalysisInterruptedError());
}
return new Promise<T>((resolve, reject) => {
const abort = () => reject(new SensitivityAnalysisInterruptedError());
signal.addEventListener("abort", abort, { once: true });
if (signal.aborted) {
void operation.catch(() => undefined);
abort();
return;
}
operation.then(
(value) => {
signal.removeEventListener("abort", abort);
resolve(value);
},
(error: unknown) => {
signal.removeEventListener("abort", abort);
reject(error);
},
);
});
}
@@ -300,18 +314,18 @@ function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) {
if (error instanceof CatalogUnavailableError) {
return reply.code(503).send({
code: "catalog_unavailable",
message: "Sensitive Data Suggestion history is unavailable because the database catalog is unavailable.",
message: "Sensitivity Analysis history is unavailable because the database catalog is unavailable.",
});
}
if (error instanceof z.ZodError) {
return reply.code(400).send({
code: "sensitive_data_suggestion_history_request_invalid",
message: "Sensitive Data Suggestion history parameters are invalid.",
message: "Sensitivity Analysis history parameters are invalid.",
});
}
return reply.code(500).send({
code: "sensitive_data_suggestion_history_failed",
message: "Sensitive Data Suggestion history could not be loaded.",
message: "Sensitivity Analysis history could not be loaded.",
});
}
@@ -320,7 +334,7 @@ export function catalogDescriptionGenerationRoutes(
deps: {
repository: CatalogRepository;
worker: DescriptionGenerationWorker;
sensitiveDataSuggestionRunner: SensitiveDataSuggestionRunner;
sensitivityAnalysisRunner: SensitivityAnalysisRunner;
},
): void {
app.post("/catalog/databases/:databaseId/sensitive-data-suggestions", async (request, reply) => {
@@ -328,16 +342,16 @@ export function catalogDescriptionGenerationRoutes(
try {
const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId);
const input = suggestionSchema.parse(request.body);
const result = await deps.sensitiveDataSuggestionRunner.run(
const signal = AbortSignal.timeout(60_000);
const result = await untilAborted(deps.sensitivityAnalysisRunner.run(
databaseId,
input.modelId,
input.scope,
"targetIds" in input ? input.targetIds : [],
new AbortController().signal,
);
signal,
), signal);
return {
suggestions: result.suggestions,
run: publicSensitiveDataSuggestionRun(result.run),
run: publicSensitivityAnalysisRun(result.run),
};
} catch (error) {
return safeSuggestionError(reply, error);
@@ -348,8 +362,8 @@ export function catalogDescriptionGenerationRoutes(
if (!manage(request, reply)) return reply;
try {
const { limit } = historyQuerySchema.parse(request.query);
return (await deps.repository.listSensitiveDataSuggestionRuns(limit))
.map(publicSensitiveDataSuggestionRun);
return (await deps.repository.listSensitivityAnalysisRuns(limit))
.map(publicSensitivityAnalysisRun);
} catch (error) {
return safeSuggestionHistoryError(reply, error);
}
@@ -359,12 +373,12 @@ export function catalogDescriptionGenerationRoutes(
if (!manage(request, reply)) return reply;
try {
const runId = idSchema.parse((request.params as { runId?: unknown }).runId);
const run = await deps.repository.getSensitiveDataSuggestionRun(runId);
const run = await deps.repository.getSensitivityAnalysisRun(runId);
if (!run) return reply.code(404).send({
code: "sensitive_data_suggestion_run_not_found",
message: "Sensitive Data Suggestion Run was not found.",
message: "Sensitivity Analysis Run was not found.",
});
return publicSensitiveDataSuggestionRun(run);
return publicSensitivityAnalysisRun(run);
} catch (error) {
return safeSuggestionHistoryError(reply, error);
}
@@ -375,14 +389,14 @@ export function catalogDescriptionGenerationRoutes(
try {
const runId = idSchema.parse((request.params as { runId?: unknown }).runId);
const { after } = eventQuerySchema.parse(request.query);
if (!(await deps.repository.getSensitiveDataSuggestionRun(runId))) {
if (!(await deps.repository.getSensitivityAnalysisRun(runId))) {
return reply.code(404).send({
code: "sensitive_data_suggestion_run_not_found",
message: "Sensitive Data Suggestion Run was not found.",
message: "Sensitivity Analysis Run was not found.",
});
}
return (await deps.repository.listSensitiveDataSuggestionEvents(runId, after))
.map(publicSensitiveDataSuggestionEvent);
return (await deps.repository.listSensitivityAnalysisEvents(runId, after))
.map(publicSensitivityAnalysisEvent);
} catch (error) {
return safeSuggestionHistoryError(reply, error);
}
@@ -14,6 +14,7 @@ import {
type ModelCompletionRequest,
} from "../src/catalog/model-completer.js";
import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js";
import type { SensitivityValueSource } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogDatabaseClient,
CatalogPostgresAccess,
@@ -69,6 +70,16 @@ async function setup(
sample: vi.fn(async () => []),
},
catalogPostgresAccess?: CatalogPostgresAccess,
sensitivityValueSource: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume(request.columns.map((column) => ({
columnId: column.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "complete", observedRows: 1 };
}),
},
) {
const repository = new MemoryCatalogRepository();
const database = await repository.create({
@@ -115,10 +126,11 @@ async function setup(
catalogOperationCoordinator: operations,
metadataGenerationModels: models(),
modelCompleter,
sensitivityValueSource,
...(descriptionSourceSampler ? { descriptionSourceSampler } : {}),
...(catalogPostgresAccess ? { catalogPostgresAccess } : {}),
});
return { app, repository, database, table, column, operations };
return { app, repository, database, table, column, operations, sensitivityValueSource };
}
async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: string) {
@@ -136,22 +148,15 @@ async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: strin
throw new Error(`Description Generation Run ${runId} did not finish`);
}
test("suggests sensitive flags from structural metadata without persisting them", async () => {
const modelCompleter = {
complete: vi.fn(async () => JSON.stringify({
suggestions: [{ columnId: expect.any(String), sensitive: true }],
})),
};
test("assesses sensitive flags locally without persisting them or calling an LLM", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, repository, database, table, column } = await setup(modelCompleter);
modelCompleter.complete.mockResolvedValueOnce(JSON.stringify({
suggestions: [{ columnId: column.id, sensitive: true }],
}));
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
payload: { scope: "all" },
});
expect(response.statusCode).toBe(200);
@@ -160,11 +165,14 @@ test("suggests sensitive flags from structural metadata without persisting them"
run: {
databaseId: database.id,
scope: "all",
modelId: configuredModel.id,
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "completed",
total: 1,
suggestedSensitive: 1,
suggestedNonSensitive: 0,
unknown: 0,
errorSummary: null,
},
suggestions: [{
@@ -175,6 +183,8 @@ test("suggests sensitive flags from structural metadata without persisting them"
version: column.version,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}],
});
expect(await repository.getColumn(database.id, column.tableId, column.id))
@@ -207,49 +217,56 @@ test("suggests sensitive flags from structural metadata without persisting them"
runId: responseBody.run.id,
sequence: 1,
level: "info",
message: "Sensitive-field suggestion generation started.",
message: "Local sensitivity analysis started.",
},
{
runId: responseBody.run.id,
sequence: 2,
level: "info",
message: "Classified 1 of 1 columns.",
message: "Assessed 1 of 1 columns locally.",
},
{
runId: responseBody.run.id,
sequence: 3,
level: "info",
message: "Sensitive-field suggestion generation completed for 1 column.",
message: "Local sensitivity analysis completed for 1 column.",
},
]);
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
const prompt = request.messages.map((message) => message.content).join("\n");
expect(prompt).toContain("patients");
expect(prompt).toContain("birth_date");
expect(prompt).toContain("date");
expect(prompt).not.toContain("Patient date of birth");
expect(prompt).not.toContain("test-provider-secret");
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("limits sensitive-data suggestions to the selected tables or columns", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
const payload = JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
columns: Array<{ columnId: string; column: string }>;
};
return JSON.stringify({
suggestions: payload.columns.map((column) => ({
columnId: column.columnId,
sensitive: column.column.includes("name") || column.column.includes("note"),
})),
});
}),
};
const { app, repository, database } = await setup(modelCompleter);
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
const controller = new AbortController();
controller.abort();
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { scope: "all" },
});
expect(response.statusCode).toBe(504);
expect(response.json()).toEqual({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
});
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
} finally {
timeout.mockRestore();
await app.close();
}
});
test("limits sensitivity analysis to the selected tables or columns", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, repository, database, sensitivityValueSource } = await setup(modelCompleter);
await repository.applySchemaSync(database.id, database.version, "all", [], {
schemaVersion: 1,
capabilities: { tables: "available", columns: "available", relationships: "available" },
@@ -281,7 +298,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: [visits.id, patients.id],
},
@@ -301,7 +317,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_columns",
targetIds: [clinicalNote.id, status.id],
},
@@ -313,44 +328,41 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }),
]));
const prompts = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => (
JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
columns: Array<{ columnId: string }>;
}
));
expect(prompts[0]!.columns.map((column) => column.columnId).sort()).toEqual(
[...patientColumns, ...visitColumns].map((column) => column.id).sort(),
);
expect(prompts[0]!.columns.map((column) => column.columnId)).not.toContain(billingColumns[0]!.id);
expect(prompts[1]!.columns.map((column) => column.columnId).sort()).toEqual(
[status.id, clinicalNote.id].sort(),
const scannedColumnIds = vi.mocked(sensitivityValueSource.scanTable).mock.calls.flatMap(
([request]) => request.columns.map((column) => column.id),
);
expect(scannedColumnIds).toEqual([status.id, status.id]);
expect(scannedColumnIds).not.toContain(patientColumns.find(
(column) => column.name === "patient_name",
)!.id);
expect(scannedColumnIds).not.toContain(clinicalNote.id);
expect(scannedColumnIds).not.toContain(billingColumns[0]!.id);
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("explains invalid sensitive-data suggestion selections without calling the model", async () => {
test("explains invalid sensitivity-analysis selections without reading source values", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, database, table } = await setup(modelCompleter);
const { app, database, table, sensitivityValueSource } = await setup(modelCompleter);
try {
const empty = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [] },
payload: { scope: "selected_tables", targetIds: [] },
});
expect(empty.statusCode).toBe(400);
expect(empty.json()).toEqual({
code: "sensitive_data_suggestion_request_invalid",
message: "Choose a database, one or more tables, or one or more columns to classify.",
message: "Choose a database, one or more tables, or one or more columns to assess.",
});
const duplicate = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: [table.id, table.id],
},
@@ -365,7 +377,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: ["00000000-0000-4000-8000-000000000001"],
},
@@ -380,7 +391,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_columns",
targetIds: ["00000000-0000-4000-8000-000000000002"],
},
@@ -391,205 +401,7 @@ test("explains invalid sensitive-data suggestion selections without calling the
message: "One or more selected Catalog Columns were not found in this database.",
});
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("batches sensitive-data suggestions for schemas larger than one helper message", async () => {
const maxHelperMessageBytes = 64 * 1024;
const seenColumnIds: string[] = [];
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
const userMessage = request.messages.find((message) => message.role === "user")!;
expect(Buffer.byteLength(userMessage.content, "utf8")).toBeLessThanOrEqual(maxHelperMessageBytes);
const payload = JSON.parse(userMessage.content) as {
columns: Array<{ columnId: string; column: string }>;
};
expect(payload.columns.length).toBeLessThanOrEqual(10);
seenColumnIds.push(...payload.columns.map((column) => column.columnId));
return JSON.stringify({
suggestions: payload.columns.map((column) => ({
columnId: column.columnId,
sensitive: column.column.endsWith("_private"),
})),
});
}),
};
const { app, repository, database } = await setup(modelCompleter);
const columnCount = 900;
await repository.applySchemaSync(database.id, database.version, "all", [], {
schemaVersion: 1,
capabilities: { tables: "available", columns: "available", relationships: "available" },
tables: [{ name: "wide_table", sourceComment: null }],
columns: Array.from({ length: columnCount }, (_, index) => ({
tableName: "wide_table",
name: `field_${index.toString().padStart(4, "0")}${index % 10 === 0 ? "_private" : ""}`,
ordinalPosition: index + 1,
dataType: "character varying(255)",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
sourceComment: null,
})),
relationships: [],
});
const wideTable = (await repository.listTables(database.id)).find((table) => table.name === "wide_table")!;
const expectedColumnIds = (await repository.listColumns(database.id, wideTable.id)).map((column) => column.id);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(200);
const suggestions = response.json().suggestions as Array<{
columnName: string;
currentSensitive: boolean;
sensitive: boolean;
}>;
expect(suggestions).toHaveLength(columnCount);
expect(suggestions).toEqual(expect.arrayContaining([
expect.objectContaining({ columnName: "field_0000_private", currentSensitive: false, sensitive: true }),
expect.objectContaining({ columnName: "field_0001", currentSensitive: false, sensitive: false }),
]));
expect(vi.mocked(modelCompleter.complete).mock.calls.length).toBeGreaterThan(1);
expect(seenColumnIds.slice().sort()).toEqual(expectedColumnIds.slice().sort());
expect(new Set(seenColumnIds).size).toBe(columnCount);
} finally {
await app.close();
}
});
test("retries one invalid sensitive-data classification before returning the review draft", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => "unused"),
};
const { app, database, column } = await setup(modelCompleter);
vi.mocked(modelCompleter.complete)
.mockResolvedValueOnce("not-json")
.mockResolvedValueOnce(JSON.stringify({
suggestions: [{ columnId: column.id, sensitive: true }],
}));
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(200);
expect(response.json().suggestions).toEqual([
expect.objectContaining({ columnId: column.id, sensitive: true }),
]);
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
} finally {
await app.close();
}
});
test.each(["malformed", "incomplete", "duplicate"] as const)(
"fails safely when sensitive-data suggestions are %s",
async (kind) => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => "unused"),
};
const { app, repository, database, column } = await setup(modelCompleter);
const rawResponse = kind === "malformed"
? "RAW_PROVIDER_RESPONSE_DO_NOT_EXPOSE_{"
: kind === "incomplete"
? JSON.stringify({ suggestions: [] })
: JSON.stringify({
suggestions: [
{ columnId: column.id, sensitive: true },
{ columnId: column.id, sensitive: true },
],
});
vi.mocked(modelCompleter.complete).mockResolvedValueOnce(rawResponse);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(502);
expect(response.json()).toEqual({
code: "sensitive_data_suggestion_invalid_response",
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
});
expect(response.body).not.toContain(rawResponse);
expect(await repository.getColumn(database.id, column.tableId, column.id))
.toMatchObject({ sensitive: false });
} finally {
await app.close();
}
},
);
test("explains a sensitive-data suggestion provider failure without exposing provider details", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => {
throw new ModelCompletionProviderError();
}),
};
const { app, repository, database, column } = await setup(modelCompleter);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(502);
expect(response.json()).toEqual({
code: "sensitive_data_suggestion_provider_unavailable",
message: "The selected LLM service could not complete the request. No suggestions were applied.",
});
expect(response.body).not.toContain("model completion failed");
expect(await repository.getColumn(database.id, column.tableId, column.id))
.toMatchObject({ sensitive: false });
const history = await app.inject({
method: "GET",
url: "/catalog/sensitive-data-suggestion-runs",
});
expect(history.statusCode).toBe(200);
const [failedRun] = history.json();
expect(failedRun).toMatchObject({
databaseId: database.id,
status: "failed",
total: 1,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
errorSummary: "Sensitive-field suggestion generation failed.",
});
const events = await app.inject({
method: "GET",
url: `/catalog/sensitive-data-suggestion-runs/${failedRun.id}/events-list`,
});
expect(events.statusCode).toBe(200);
expect(events.json()).toMatchObject([
{
runId: failedRun.id,
sequence: 1,
level: "info",
message: "Sensitive-field suggestion generation started.",
},
{
runId: failedRun.id,
sequence: 2,
level: "error",
message: "Sensitive-field suggestion generation failed.",
},
]);
expect(events.body).not.toContain("model completion failed");
expect(sensitivityValueSource.scanTable).not.toHaveBeenCalled();
} finally {
await app.close();
}
@@ -15,6 +15,7 @@ import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensiti
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
import { KyselyCatalogRepository, type CatalogDatabase } from "../src/catalog/repository.js";
import { loadConfig } from "../src/config.js";
import type { WorkspaceRegistry } from "../src/workspaces/registry.js";
@@ -52,6 +53,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
const repository = new KyselyCatalogRepository(db);
const database = await repository.create({
workspaceId: "psd-clinical",
@@ -0,0 +1,131 @@
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { afterEach, expect, test, vi } from "vitest";
import { PythonLocalNerDetector } from "../src/catalog/local-ner-detector.js";
const roots: string[] = [];
afterEach(() => {
vi.unstubAllEnvs();
for (const root of roots.splice(0)) rmSync(root, { recursive: true, force: true });
});
test("keeps a CPU-only local worker warm and returns sanitized evidence", async () => {
vi.stubEnv("THT_MODEL_API_KEY", "must-not-reach-worker");
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-"));
roots.push(root);
const helper = join(root, "fake_ner_worker.py");
writeFileSync(helper, `
import json
import os
import pathlib
import sys
root = pathlib.Path.cwd()
root.joinpath("runtime.json").write_text(json.dumps({
"argv": sys.argv,
"cuda": os.environ.get("CUDA_VISIBLE_DEVICES"),
"hip": os.environ.get("HIP_VISIBLE_DEVICES"),
"offline": os.environ.get("HF_HUB_OFFLINE"),
"inherited_secret": os.environ.get("THT_MODEL_API_KEY"),
"pid": os.getpid(),
}), encoding="utf-8")
print(json.dumps({"ready": True}), flush=True)
for line in sys.stdin:
request = json.loads(line)
root.joinpath("request.json").write_text(json.dumps(request), encoding="utf-8")
print(json.dumps({
"id": request["id"],
"ok": True,
"evidence": [{
"columnId": request["candidates"][0]["columnId"],
"label": "person",
"confidence": 0.93,
}],
}), flush=True)
`, "utf8");
const detector = new PythonLocalNerDetector({
pythonExecutable: "python3",
workerScript: helper,
modelPath: join(root, "pinned-model"),
cwd: root,
threads: 2,
startupTimeoutMs: 5_000,
});
const candidate = {
columnId: "33333333-3333-4333-8333-333333333333",
text: "Dimesso Mario Rossi",
};
try {
expect(detector.isReady()).toBe(false);
await detector.warmup();
expect(detector.isReady()).toBe(true);
expect(existsSync(join(root, "request.json"))).toBe(false);
await expect(detector.detect(
[candidate],
new AbortController().signal,
Date.now() + 5_000,
)).resolves.toEqual([{
columnId: candidate.columnId,
label: "person",
confidence: 0.93,
}]);
const firstRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
expect(firstRuntime).toMatchObject({
cuda: "",
hip: "",
offline: "1",
inherited_secret: null,
});
expect(JSON.stringify(firstRuntime.argv)).not.toContain(candidate.text);
expect(JSON.parse(readFileSync(join(root, "request.json"), "utf8")).candidates).toEqual([candidate]);
await detector.detect([candidate], new AbortController().signal, Date.now() + 5_000);
const secondRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
expect(secondRuntime.pid).toBe(firstRuntime.pid);
} finally {
await detector.close();
}
});
test("bounds worker startup by the caller deadline", async () => {
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-deadline-"));
roots.push(root);
const helper = join(root, "slow_ner_worker.py");
writeFileSync(helper, `
import json
import sys
import time
time.sleep(2)
print(json.dumps({"ready": True}), flush=True)
for line in sys.stdin:
request = json.loads(line)
print(json.dumps({"id": request["id"], "ok": True, "evidence": []}), flush=True)
`, "utf8");
const detector = new PythonLocalNerDetector({
pythonExecutable: "python3",
workerScript: helper,
modelPath: join(root, "pinned-model"),
cwd: root,
startupTimeoutMs: 5_000,
});
const startedAt = Date.now();
try {
await expect(detector.detect(
[{
columnId: "33333333-3333-4333-8333-333333333333",
text: "Dimesso Mario Rossi",
}],
new AbortController().signal,
startedAt + 50,
)).rejects.toThrow("local NER is unavailable");
expect(Date.now() - startedAt).toBeLessThan(1_000);
} finally {
await detector.close();
}
});
@@ -16,6 +16,7 @@ import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_s
import { up as upLogicalRelationships } from "../src/catalog/migrations/008_catalog_logical_relationships.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
const dockerAvailable = spawnSync("docker", ["info"], { stdio: "ignore" }).status === 0;
@@ -47,12 +48,23 @@ test.skipIf(!dockerAvailable)("PostgreSQL migration enforces one database per wo
modelId: "openai-mini", language: "en", status: "completed", total: 1,
processed: 1, generated: 1,
}).execute();
const historicalSuggestionRunId = randomUUID();
await db.insertInto("sensitiveDataSuggestionRuns").values({
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
id: historicalSuggestionRunId, databaseId: historicalDatabaseId, scope: "all",
modelId: "openai-mini", status: "completed", total: 1,
suggestedSensitive: 1,
}).execute();
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
await expect(db.selectFrom("sensitiveDataSuggestionRuns")
.select(["engine", "modelId", "policyVersion", "unknown"])
.where("id", "=", historicalSuggestionRunId)
.executeTakeFirstOrThrow()).resolves.toMatchObject({
engine: "llm",
modelId: "openai-mini",
policyVersion: null,
unknown: 0,
});
await expect(db.insertInto("descriptionGenerationRuns").values({
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
modelId: "openai/gpt-5-mini", language: "en", status: "completed", total: 1,
@@ -418,6 +430,7 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
const repository = new KyselyCatalogRepository(db);
const firstDatabase = await repository.create({
workspaceId: "generation-one",
@@ -586,10 +599,10 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
]);
expect(await repository.getActiveDescriptionGenerationRun()).toBeUndefined();
const suggestionRun = await repository.createSensitiveDataSuggestionRun(
const suggestionRun = await repository.createSensitivityAnalysisRun(
firstDatabase.id,
"selected_columns",
"openai/gpt-4.1-mini",
{ engine: "local", policyVersion: "sensitivity-v1" },
);
expect(suggestionRun).toMatchObject({
databaseId: firstDatabase.id,
@@ -597,43 +610,49 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
unknown: 0,
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
startedAt: expect.any(String),
});
await repository.appendSensitiveDataSuggestionEvent(
await repository.appendSensitivityAnalysisEvent(
suggestionRun.id,
"info",
"Sensitive-field suggestion generation started.",
);
await repository.appendSensitiveDataSuggestionEvent(
await repository.appendSensitivityAnalysisEvent(
suggestionRun.id,
"info",
"Sensitive-field suggestion generation completed for 2 columns.",
);
expect(await repository.updateSensitiveDataSuggestionRun(suggestionRun.id, {
expect(await repository.updateSensitivityAnalysisRun(suggestionRun.id, {
status: "completed",
total: 2,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
suggestedNonSensitive: 0,
unknown: 1,
finishedAt: new Date().toISOString(),
})).toMatchObject({
status: "completed",
total: 2,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
suggestedNonSensitive: 0,
unknown: 1,
});
expect(await repository.listSensitiveDataSuggestionEvents(suggestionRun.id, 1)).toEqual([
expect(await repository.listSensitivityAnalysisEvents(suggestionRun.id, 1)).toEqual([
expect.objectContaining({ sequence: 2, level: "info" }),
]);
expect((await repository.listSensitiveDataSuggestionRuns(1))[0]).toMatchObject({
expect((await repository.listSensitivityAnalysisRuns(1))[0]).toMatchObject({
id: suggestionRun.id,
});
const interruptedSuggestionRun = await repository.createSensitiveDataSuggestionRun(
const interruptedSuggestionRun = await repository.createSensitivityAnalysisRun(
secondDatabase.id,
"all",
"openai/gpt-4.1-mini",
{ engine: "local", policyVersion: "sensitivity-v1" },
);
expect(await repository.interruptActiveSensitiveDataSuggestionRuns(
expect(await repository.interruptActiveSensitivityAnalysisRuns(
"Sensitive-field suggestion generation was interrupted by backend restart.",
)).toEqual([
expect.objectContaining({
@@ -0,0 +1,99 @@
import { expect, test, vi } from "vitest";
import {
SensitivityAnalysisInterruptedError,
SensitivityAnalysisService,
} from "../src/catalog/sensitivity-analysis-service.js";
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogRepository,
SensitivityAnalysisRun,
WorkspaceDatabase,
} from "../src/catalog/types.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "public",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const running: SensitivityAnalysisRun = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "running",
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
unknown: 0,
inputTokens: 0,
cacheReadTokens: 0,
outputTokens: 0,
createdAt: "2026-09-02T08:00:00Z",
startedAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
finishedAt: null,
errorSummary: null,
};
test("stops catalog selection when the request expires during a catalog read", async () => {
const controller = new AbortController();
const listTables = vi.fn();
const repository = {
get: vi.fn(async () => {
controller.abort();
return database;
}),
listTables,
} as unknown as CatalogRepository;
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
const analysis = new SensitivityAnalysisService(repository, classifier);
await expect(analysis.analyze(
database.id,
"all",
[],
controller.signal,
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(listTables).not.toHaveBeenCalled();
expect(classifier.assessTable).not.toHaveBeenCalled();
});
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
const controller = new AbortController();
const update = vi.fn(async (_runId: string, changes: Partial<SensitivityAnalysisRun>) => ({
...running,
...changes,
}));
const repository = {
get: vi.fn(async () => database),
createSensitivityAnalysisRun: vi.fn(async () => {
controller.abort();
return running;
}),
updateSensitivityAnalysisRun: update,
appendSensitivityAnalysisEvent: vi.fn(async () => undefined),
} as unknown as CatalogRepository;
const analysis = { analyze: vi.fn() } as unknown as SensitivityAnalysisService;
const runner = new SensitivityAnalysisRunner(repository, analysis);
await expect(runner.run(database.id, "all", [], controller.signal))
.rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(analysis.analyze).not.toHaveBeenCalled();
expect(update).toHaveBeenCalledWith(running.id, expect.objectContaining({
status: "interrupted",
total: 0,
unknown: 0,
errorSummary: "Local sensitivity analysis reached its time limit.",
}));
});
@@ -0,0 +1,450 @@
import { expect, test, vi } from "vitest";
import {
SensitivityClassifier,
type LocalNerDetector,
type SensitivityNerBudget,
type SensitivityTableScan,
type SensitivityValueSource,
} from "../src/catalog/sensitivity-classifier.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import { CatalogConnectorError } from "../src/catalog/types.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "public",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const table = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
name: "observations",
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
} satisfies CatalogTable;
function column(overrides: Partial<CatalogColumn> = {}): CatalogColumn {
return {
id: "33333333-3333-4333-8333-333333333333",
tableId: table.id,
name: "note",
ordinalPosition: 1,
dataType: "character varying",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
...overrides,
};
}
function source(scan: SensitivityTableScan): SensitivityValueSource {
return { scanTable: vi.fn(async (_request, consume) => {
for (const batch of scan.batches) await consume(batch);
return scan.coverage;
}) };
}
test("one email hidden in a generically named column makes the whole column sensitive", async () => {
const target = column();
const values = source({
batches: [[
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedRows: 2 },
});
const classifier = new SensitivityClassifier(values);
const [assessment] = await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
columnId: target.id,
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("one text value longer than 500 characters makes the whole column sensitive", async () => {
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "length", ruleId: "text.over_500_characters" }],
});
});
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
id: "66666666-6666-4666-8666-666666666666",
name: "category",
sensitive: true,
});
const values = source({
batches: [[
{ columnId: benign.id, value: "active", characterLength: 6 },
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedRows: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [benign, empty, humanProtected] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
expect.objectContaining({
columnId: humanProtected.id,
assessment: "non_sensitive",
proposedSensitive: false,
}),
]);
});
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: true,
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
});
});
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
name: "codice_fiscale",
});
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
throw new CatalogConnectorError("upstream detail must not escape");
}),
};
const assessments = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({
columnId: unresolved.id,
assessment: "unknown",
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
}),
expect.objectContaining({
columnId: metadataMatch.id,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}),
]);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedRows: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});
test.each([
["RSSMRA85T10A562S", "pii.italian_fiscal_code"],
["IT60 X054 2811 1010 0000 0123 456", "financial.iban"],
["4111 1111 1111 1111", "financial.payment_card"],
["SWIFT DEUTDEFF500", "financial.bic"],
["Partita IVA 00743110157", "pii.italian_vat"],
["Passaporto YA1234567", "pii.passport_number"],
["Carta d'identità CA12345AA", "pii.identity_card"],
["Patente di guida U11234567A", "pii.drivers_license_number"],
["Chiamare +39 347 123 4567", "pii.phone_number"],
["Client 192.168.1.5", "network.ip_address"],
["Device 00:1B:44:11:3A:B7", "network.mac_address"],
["https://example.org/profiles/mario", "network.url"],
["550e8400-e29b-41d4-a716-446655440000", "pii.uuid"],
["AWS key AKIAIOSFODNN7EXAMPLE", "credential.access_key"],
["Diagnosi: carcinoma mammario con metastasi ossee", "health.clinical_term"],
["-----BEGIN PRIVATE KEY----- secret -----END PRIVATE KEY-----", "credential.private_key"],
['{"profile":{"email":"not yet supplied"}}', "pii.json_sensitive_key"],
] as const)("recognizes validated sensitive content without relying on the column name: %s", async (
value,
ruleId,
) => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId }],
});
});
test("does not make a malformed email decisive", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedRows: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
});
test("finds a valid email after a malformed candidate in the same value", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedRows: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("optional local NER evidence can make otherwise ambiguous Italian text sensitive", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
};
const [assessment] = await new SensitivityClassifier(values, detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledWith(
[{ columnId: target.id, text: "Dimesso Mario Rossi" }],
expect.any(AbortSignal),
expect.any(Number),
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "ner", ruleId: "ner.entity", label: "person_name", confidence: 0.91 }],
});
});
test("does not wait for an optional NER worker that is still warming", async () => {
const target = column();
const detector: LocalNerDetector = {
isReady: () => false,
detect: vi.fn(async () => [{ columnId: target.id, label: "person", confidence: 0.99 }]),
};
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({ assessment: "unknown" });
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
const columns = Array.from({ length: 17 }, (_, index) => column({
id: `00000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const observations = columns.flatMap((item, columnIndex) => Array.from(
{ length: 8 },
(_, valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
}),
));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedRows: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledTimes(2);
expect(vi.mocked(detector.detect).mock.calls.map(([candidates]) => candidates.length)).toEqual([
128,
8,
]);
});
test("limits default NER work to two candidates spread across a wide table", async () => {
const columns = Array.from({ length: 10 }, (_, index) => column({
id: `10000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [columns.flatMap((item, columnIndex) => [0, 1].map((valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedRows: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledOnce();
const submitted = vi.mocked(detector.detect).mock.calls[0]![0];
expect(submitted).toHaveLength(2);
expect(new Set(submitted.map((candidate) => candidate.columnId)).size).toBe(2);
});
test("shares a bounded NER time allowance across tables in one analysis run", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
await new Promise((resolve) => setTimeout(resolve, 20));
return [];
}),
};
const classifier = new SensitivityClassifier(values, detector);
const nerBudget: SensitivityNerBudget = { remainingMs: 1 };
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
expect(detector.detect).toHaveBeenCalledOnce();
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content remains unknown after complete coverage", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
});
});
@@ -0,0 +1,316 @@
import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: 'clinical"data',
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const table = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
name: 'patient"facts',
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
} satisfies CatalogTable;
function column(id: string, name: string): CatalogColumn {
return {
id,
tableId: table.id,
name,
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
const fullRows = Array.from({ length: 200 }, () => ({
__value_0: "ordinary",
__length_0: "8",
__value_1: null,
__length_1: null,
}));
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
}
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
let clockCalls = 0;
const values = new ConcreteSensitivityValueSource(access, undefined, {
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
});
const consumed: unknown[] = [];
const coverage = await values.scanTable({
database,
table,
columns: [note, contact],
fullScanBudgetMs: 5_000,
deadline: 61_000,
}, (batch) => consumed.push(...batch), new AbortController().signal);
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
expect(query.mock.calls.some(([sql]) => (
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql) === (
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
expect(query.mock.calls.some(([sql]) => (
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
))).toBe(true);
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
expect(end).toHaveBeenCalledOnce();
});
test("reports complete coverage when the final full-scan page is short", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
: { rows: [] });
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
expect(query.mock.calls.filter(([sql]) => (
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toHaveLength(1);
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
]);
});
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
let fullScanAttempts = 0;
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
}
if (sql.startsWith("FETCH FORWARD")) {
fullScanAttempts += 1;
throw Object.assign(new Error("statement timeout"), { code: "57014" });
}
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(fullScanAttempts).toBe(1);
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
"SAVEPOINT sensitivity_full_scan",
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
]));
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "sample", characterLength: 6 },
]);
});
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const release = vi.fn();
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release,
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
};
const values = new ConcreteSensitivityValueSource(access, secretStore);
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root/",
restPath: "/health",
restAuth: "x-api-key",
},
};
const note = column("33333333-3333-4333-8333-333333333333", "note");
const consume = vi.fn();
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal)).resolves.toEqual({
kind: "complete",
observedRows: 1,
});
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith(
"https://dwh.example.test/root/rpc/run_query",
expect.objectContaining({
method: "POST",
headers: { "content-type": "application/json", "x-api-key": "test-api-key" },
}),
);
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
expect(release).toHaveBeenCalledOnce();
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("keeps multi-request REST scans conservative without a source transaction", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release: vi.fn(),
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn()
.mockResolvedValueOnce(new Response(JSON.stringify([
{ __value_0: "ordinary", __length_0: 8 },
]), { status: 200 }))
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
vi.stubGlobal("fetch", fetchMock);
const values = new ConcreteSensitivityValueSource({
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
}, secretStore, { batchRows: 1 });
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root",
restPath: "/health",
restAuth: "x-api-key",
},
};
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 1,
});
expect(fetchMock).toHaveBeenCalledTimes(2);
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
const query = vi.fn(async () => ({ rows: [] }));
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const now = vi.fn()
.mockReturnValueOnce(1_000)
.mockReturnValue(61_000);
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
await expect(values.scanTable({
database,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 0,
});
expect(query).not.toHaveBeenCalled();
expect(end).toHaveBeenCalledOnce();
});
+29
View File
@@ -81,6 +81,35 @@ test("loadConfig keeps local development defaults", () => {
expect(loadConfig({}).dataRoot).toBeUndefined();
});
test("loadConfig keeps local NER disabled unless an absolute model path is configured", () => {
expect(loadConfig({}).sensitivityNer).toBeUndefined();
expect(loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
}).sensitivityNer?.pythonExecutable).toBe("/opt/sensitivity-ner/bin/python");
expect(loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
THT_SENSITIVITY_NER_PYTHON: "/opt/sensitivity-ner/bin/python",
THT_SENSITIVITY_NER_WORKER: "/app/backend/python/sensitivity_ner_worker.py",
THT_SENSITIVITY_NER_THREADS: "3",
}).sensitivityNer).toEqual({
modelPath: "/models/gliner2-pii",
pythonExecutable: "/opt/sensitivity-ner/bin/python",
workerScript: "/app/backend/python/sensitivity_ner_worker.py",
threads: 3,
});
});
test("loadConfig rejects ambiguous or unsafe local NER configuration", () => {
expect(() => loadConfig({ THT_SENSITIVITY_NER_MODEL_PATH: "fastino/model" }))
.toThrow("sensitivity NER model path configuration is invalid");
expect(() => loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
THT_SENSITIVITY_NER_THREADS: "0",
})).toThrow("sensitivity NER thread configuration is invalid");
expect(() => loadConfig({ THT_SENSITIVITY_NER_PYTHON: "/opt/ner/bin/python" }))
.toThrow("sensitivity NER settings require a model path");
});
test("loadConfig allows none and mock only outside production when auth.yaml is absent", () => {
const originalNodeEnvironment = process.env.NODE_ENV;
delete process.env.NODE_ENV;
+21
View File
@@ -0,0 +1,21 @@
services:
core:
image: thothii-core:sensitivity-ner
build:
args:
INSTALL_SENSITIVITY_NER: "true"
environment:
THT_SENSITIVITY_NER_MODEL_PATH: /opt/thothii/models/gliner2-privacy-filter-pii-multi
THT_SENSITIVITY_NER_PYTHON: /opt/sensitivity-ner/bin/python
THT_SENSITIVITY_NER_THREADS: ${THT_SENSITIVITY_NER_THREADS:-2}
volumes:
- type: bind
source: ${THT_SENSITIVITY_NER_MODEL_DIR:?set THT_SENSITIVITY_NER_MODEL_DIR}
target: /opt/thothii/models/gliner2-privacy-filter-pii-multi
read_only: true
catalog-migrate:
image: thothii-core:sensitivity-ner
workspace-maintenance:
image: thothii-core:sensitivity-ner
+14 -1
View File
@@ -41,6 +41,7 @@ RUN npm run build
FROM python:3.12-slim-bookworm@sha256:d50fb7611f86d04a3b0471b46d7557818d88983fc3136726336b2a4c657aa30b AS runtime
ARG PI_VERSION
ARG IMAGE_VERSION
ARG INSTALL_SENSITIVITY_NER=false
LABEL org.opencontainers.image.title="thothii-core" \
org.opencontainers.image.version="${IMAGE_VERSION}" \
org.opencontainers.image.description="ThothII core with its embedded Pi runtime" \
@@ -48,7 +49,7 @@ LABEL org.opencontainers.image.title="thothii-core" \
# Runtime tools
RUN set -eux; \
runtime_packages="curl ca-certificates ripgrep fd-find tini git openssh-client"; \
runtime_packages="curl ca-certificates ripgrep fd-find tini git openssh-client libseccomp2"; \
if ! command -v flock >/dev/null 2>&1; then \
runtime_packages="$runtime_packages util-linux"; \
fi; \
@@ -93,10 +94,22 @@ RUN mkdir -p /app/harness/config \
# PiProcessManager (backend) prepende harnessDir/.venv/bin al PATH del child Pi → symlink al venv reale
RUN ln -s /opt/venv /app/harness/.venv
# The optional NER dependency layer is independent of backend source and build artifacts so it can
# be reused when only TypeScript or worker code changes.
COPY backend/python/sensitivity-ner-requirements.txt /app/backend/python/sensitivity-ner-requirements.txt
RUN if [ "$INSTALL_SENSITIVITY_NER" = "true" ]; then \
python -m venv /opt/sensitivity-ner; \
/opt/sensitivity-ner/bin/pip install --no-cache-dir --upgrade pip; \
/opt/sensitivity-ner/bin/pip install --no-cache-dir -r /app/backend/python/sensitivity-ner-requirements.txt; \
elif [ "$INSTALL_SENSITIVITY_NER" != "false" ]; then \
echo "INSTALL_SENSITIVITY_NER must be true or false" >&2; exit 2; \
fi
# Backend: dist + node_modules (stesso Node major 24 + glibc bookworm → compatibili)
COPY --from=backend-build /src/backend/dist /app/backend/dist
COPY --from=backend-build /src/backend/node_modules /app/backend/node_modules
COPY backend/scripts/ssh-askpass.mjs /app/backend/scripts/ssh-askpass.mjs
COPY backend/python /app/backend/python
COPY backend/package*.json /app/backend/
RUN chmod 0755 /app/backend/scripts/ssh-askpass.mjs
@@ -21,13 +21,21 @@ completed scan with no finding may produce `non_sensitive`; an incomplete scan w
produces `unknown`.
Ambiguous text may additionally be sent to an optional local NER detector only while time remains.
The detector runs on CPU, receives no tools or network access, does not persist source values, and
returns evidence rather than the column decision. The initial supported detector is
By default it receives at most two candidates per table and shares a ten-second allowance across
the entire analysis run. The detector runs on CPU, receives no tools or network access (enforced
inside the worker with a fail-closed seccomp network-syscall filter), does not
persist source values, and returns evidence rather than the column decision. The initial supported detector is
[`fastino/gliner2-privacy-filter-PII-multi`](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi),
used through the Apache-2.0 GLiNER2 Python library with a pinned model revision. Its model weights
and GLiNER2 code are Apache-2.0, and its mDeBERTa base model is MIT. It is trained for seven
languages including Italian and can run on CPU without using the installation's GPUs.
The selected checkpoint currently carries Transformers 5 tokenizer metadata while the released
GLiNER2 2.0.0 runtime requires Transformers 4. ThothII may bridge only that known key rename in a
temporary view of the immutable, checksummed artifact; unexpected or ambiguous metadata fails
closed. Removing the compatibility bridge requires an offline smoke test against a corrected,
pinned upstream release.
The NER detector is an optional installation asset because its weights and runtime are materially
larger than the deterministic TypeScript engine. It is invoked only for otherwise unresolved text,
never for values already classified by a decisive rule. If it is disabled, unavailable, times out,
+13 -5
View File
@@ -104,11 +104,19 @@ or second orchestration subsystem. A target receives at most one provider retry;
exhausted technical batches fail the run. Stale work is marked interrupted at startup and must be
explicitly unlocked; it never resumes automatically.
Sensitive-field suggestion generation remains a synchronous administrative request, but each
attempt has its own durable run and ordered sanitized events. This history is separate from
Description Generation because its lifecycle and counters differ. Only execution metadata and
aggregate counts are stored; proposed flags, prompts, raw model output, and provider diagnostics
remain transient.
Sensitivity analysis is a synchronous administrative request and does not use the installation
model catalog. Database-specific adapters stream bounded normalized values from read-only source
connections; the TypeScript `SensitivityClassifier` is the single decision point for
`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted
for at most five seconds per table, then the adapter samples within the sixty-second request budget.
An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it
cannot make or persist the decision itself.
Each attempt has its own durable run and ordered sanitized events, separate from Description
Generation because its lifecycle and counters differ. The run records the local policy version,
coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and
worker diagnostics remain transient. Only an explicit administrator save changes the human-owned
Sensitive Data Flag.
## Main backend classes
+10 -7
View File
@@ -137,14 +137,17 @@ Generated descriptions can be requested for selected tables, selected columns, e
target, or targets with a missing generated description. The backend accepts one installation-wide
run and processes targets sequentially. Every catalog column has a **Sensitive** flag, which defaults
to `false`, including after a newly discovered column is synchronized. Before generation, an
administrator can ask the configured model to suggest flags from structural metadata only (database,
schema, table and column names, data types, nullability, primary keys, and foreign keys). Suggestions
remain an unsaved draft until a human reviews and saves them.
administrator can run local sensitivity analysis over the selected database, tables, or columns.
The analysis combines structural metadata with bounded read-only inspection of source values. It
uses no generative AI and no installation-catalog model. Assessments remain an unsaved draft until
a human reviews and saves them; the reviewer may reverse any proposal.
The page exposes separate histories for description generation and sensitive-field suggestions.
Sensitive-suggestion history stores the selected model, scope, status, aggregate counts, timestamps,
and sanitized events. It does not store the proposed per-column flags, prompts, raw model output, or
provider diagnostics; closing an unsaved review still discards that draft.
The page exposes separate histories for description generation and sensitivity analysis. Analysis
history stores the local policy version, scope, status, aggregate `sensitive`, `non_sensitive`, and
`unknown` counts, timestamps, and sanitized events. It does not store source values, per-column
proposals, NER spans, or worker diagnostics; closing an unsaved review discards that draft. The
rules, time bounds, and optional CPU-only NER profile are documented in
[Local sensitivity analysis](sensitivity-analysis.md).
For a column with `sensitive=false`, the worker may read at most five source rows and five
representative non-null values through a read-only connector. For `sensitive=true`, the source query
+128
View File
@@ -0,0 +1,128 @@
# Local sensitivity analysis
Database Management can assess selected columns without sending their metadata or contents to a
generative model. The feature is advisory: it creates a transient review draft, while the catalog's
Sensitive Data Flag changes only when an administrator explicitly saves a choice. The administrator
may set either value, including overriding a `sensitive` proposal.
## Default policy
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1`
policy combines:
- normalized column-name rules for direct identifiers, credentials, and health data;
- validated content rules for email, Italian fiscal code and VAT, passport, identity-card and
driving-licence identifiers, phone numbers, IBAN/BIC, payment-card checksums, IP/MAC addresses,
URLs, UUIDs, access keys, private-key markers, sensitive keys inside bounded recursive JSON, and
a reviewed Italian clinical-term dictionary;
- a conservative length rule: any observed textual value longer than 500 characters makes the
entire column sensitive.
One decisive value is enough to classify the column as `sensitive`. A complete scan with no match
may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled
no-match columns are `unknown`; an `unknown` draft preserves the current human flag.
Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and
REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never
persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses
a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a
read-only transaction and always end with rollback. A REST scan can claim complete coverage only
when it finishes in one request; multi-request pagination has no shared source transaction and is
therefore conservatively reported as sampled.
The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and
after catalog selection, source access, progress writes, and every table. If it expires after a run
has been created, that run is finalized as `interrupted` and all not-decisively-processed columns
are counted as `unknown`; no review payload is returned from the timed-out request.
History stores only the policy version, aggregate outcomes, timestamps, and fixed operational
events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted.
Neither path stores values, matched spans, prompts, or free-form model output.
## Optional CPU-only GLiNER2 evidence
The deterministic engine works without Python NER. An installation may opt into
`fastino/gliner2-privacy-filter-PII-multi` for unresolved short text. It runs in a persistent local
Python worker, adds sanitized evidence, and never becomes a second decision point. The worker:
- loads a local model directory only and forces Hugging Face/Transformers offline mode;
- starts warming in the background when the backend starts; an analysis never waits for warm-up
and skips NER until the worker is ready, so loading cannot consume the run's NER allowance;
- hides CUDA and HIP devices and loads weights with `map_location="cpu"`;
- starts with a scrubbed environment, then installs a fail-closed seccomp filter that denies
network syscalls before accepting source text (the Python socket API is disabled as defense in depth);
- receives at most two 500-character candidates per table by default, selected breadth-first
across unresolved columns, and shares a ten-second NER allowance across the whole run;
- returns only column ID, normalized label, and confidence; source text and entity spans are not
returned or stored;
- is skipped on timeout, startup failure, invalid output, or absent configuration. No LLM fallback
is selected.
The pinned model revision is `c153999da5f4c509df4322b0c6a1baf3d2c284d7`. GLiNER2 and the model
are Apache-2.0; the published mDeBERTa base is MIT. The optional runtime pins
`gliner2[local]==2.0.0`, `transformers==4.57.6`, and the CPU-only PyTorch wheel
`torch==2.14.0+cpu`. It lives in `/opt/sensitivity-ner`, is not installed in the default core image,
and does not install CUDA packages.
The current upstream checkpoint was saved by Transformers 5.8 even though GLiNER2 2.0.0 officially
requires Transformers `<5`; the resulting tokenizer error is independently reported in
[GLiNER2 issue 145](https://github.com/fastino-ai/GLiNER2/issues/145). At startup ThothII leaves the
pinned model directory unchanged and creates a temporary symlink view that maps the checkpoint's
`extra_special_tokens` list to the Transformers 4 name `additional_special_tokens`. Any other or
ambiguous shape fails closed and leaves the optional NER unavailable. The offline CPU smoke test
must remain part of every dependency or model revision update.
## Prepare and enable the optional profile
Download happens during explicit installation, never during inference:
```bash
./scripts/fetch-sensitivity-ner-model.sh /absolute/path/to/gliner2-pii
```
The script builds the separate `thothii-core:sensitivity-ner` image, downloads the exact revision, and writes
`MODEL_SHA256SUMS`. Keep the model directory outside the repository. Then set:
```bash
export THOTH_ENABLE_SENSITIVITY_NER=1
export THT_SENSITIVITY_NER_MODEL_DIR=/absolute/path/to/gliner2-pii
export THT_SENSITIVITY_NER_THREADS=2
./scripts/run-stack.sh
```
For an operator-managed Compose invocation, include `deploy/compose.sensitivity-ner.yaml` after the
base and installation overlays. The core build argument `INSTALL_SENSITIVITY_NER=true` installs the
optional Python dependencies into their isolated virtualenv. The model mount is read-only. Values
above eight threads are rejected; start with two so classification cannot contend heavily with
other CPU workloads.
## Acceptance on a real database
Run the first evaluation in shadow mode: read the source with its existing read-only role, do not
save proposed flags, and report only aggregate counts, rule IDs, coverage, and timings. Never copy
matched values into test output. Use a separately approved, labeled Italian corpus to calculate
precision and recall; raw PSD values must remain inside the authorized environment.
Inside the configured core runtime, the non-mutating command is:
```bash
npm run sensitivity:shadow -- psd-clinical
```
It reads catalog metadata and source values but emits one aggregate JSON object with no database,
table, column, or source-value detail. It neither creates an analysis run nor updates a flag.
Enabling NER by default requires all of these gates:
1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory;
2. the Python dependency/license inventory contains only redistribution-compatible licenses;
3. the CPU benchmark stays within the configured deadlines and does not use a GPU;
4. the labeled Italian evaluation meets thresholds approved by the product owner.
If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved
columns remain `unknown` rather than being sent to an internal or external LLM.
The first aggregate PSD shadow comparison is recorded in
[`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the
local CPU runner, NER found additional entities but reduced total coverage inside the 60-second
deadline, so the accepted setting remains disabled by default.
@@ -0,0 +1,37 @@
# PSD sensitivity shadow evaluation
Date: 2026-09-02
This report records an aggregate, non-mutating evaluation of `sensitivity-v1` against the PSD
workspace. The source data warehouse was accessed through the configured read-only connector. The
shadow command did not create an analysis run, update catalog metadata, or save Sensitive Data
Flags. No database, table, column, source value, matched span, or free-form diagnostic was emitted.
The runner was the local Docker `arm64` CPU environment connected to the PSD source; this was not a
benchmark of the PSD production server. Both runs used the same 2,275 catalog columns and a
60-second analysis deadline.
| Profile | Sensitive | Non-sensitive | Unknown | NER findings | Analysis time |
| --- | ---: | ---: | ---: | ---: | ---: |
| Deterministic policy | 57 | 8 | 2,210 | 0 | 60,017 ms |
| CPU NER, pre-warmed, two candidates/table, 10 s shared allowance | 68 | 0 | 2,207 | 11 | 60,022 ms |
The deterministic run produced findings from metadata, phone-number, and Italian clinical-term
rules. The optional NER run identified eleven additional unresolved text candidates, but its
inference time reduced the source coverage reached before the global deadline. The number of
definitive non-sensitive assessments consequently fell from eight to zero, so this broad shadow
run does not justify enabling NER by default.
The separate offline synthetic Italian smoke test succeeded with a `full_name` finding at high
confidence. The image dependency check reported no broken requirements, and PyTorch reported
`cuda=False`, no CUDA runtime, and zero GPU devices. A defense-in-depth test retained a raw socket
constructor before Python-level blocking and confirmed that the worker's seccomp filter still
rejected the socket syscall with `EPERM`.
## Acceptance outcome
- Keep the deterministic TypeScript policy enabled by default.
- Keep GLiNER2 available only through the explicit CPU-only installation profile.
- Do not enable NER by default for PSD on the basis of this shadow run.
- Reconsider the PSD setting only after a benchmark on the actual target CPU and a labeled Italian
corpus demonstrate a useful precision/recall gain without unacceptable coverage loss.
@@ -0,0 +1,36 @@
# Optional sensitivity NER license inventory
This inventory covers the isolated `/opt/sensitivity-ner` Python environment built from
`backend/python/sensitivity-ner-requirements.txt` on 2 September 2026. It is a technical release
gate, not legal advice. Every dependency is version-locked; changing any version requires
regenerating this inventory and rerunning the offline CPU smoke test.
The optional runtime also dynamically links Debian's `libseccomp2` (LGPL-2.1-only) solely to
install its kernel-enforced network syscall filter; no libseccomp source is incorporated into ThothII.
No dependency or selected model uses a non-commercial, research-only, source-available, GPL, or
AGPL license. MPL-2.0, PSF-2.0, and the permissive composite licenses below allow free-of-charge and
commercial use, but distributors must still preserve their applicable notices and license texts.
| License family | Locked packages |
| --- | --- |
| Apache-2.0 | `accelerate==1.14.0`, `gliner2==2.0.0`, `hf-xet==1.6.0`, `huggingface-hub==0.36.2`, `peft==0.20.0`, `requests==2.34.2`, `safetensors==0.8.0`, `tokenizers==0.22.2`, `transformers==4.57.6` |
| MIT | `annotated-types==0.8.0`, `charset-normalizer==3.5.1`, `filelock==3.32.5`, `pydantic==2.13.5`, `pydantic-core==2.46.5`, `PyYAML==6.0.3`, `typing-inspection==0.4.4`, `urllib3==2.7.0` |
| BSD-2/3-Clause | `fsspec==2026.7.0`, `idna==3.19`, `Jinja2==3.1.6`, `MarkupSafe==3.0.3`, `mpmath==1.3.0`, `networkx==3.6.1`, `psutil==7.2.2`, `sympy==1.14.0` |
| MPL-2.0 or mixed MPL/MIT | `certifi==2026.7.22`, `tqdm==4.70.0` |
| PSF-2.0 | `typing-extensions==4.16.0` |
| Composite permissive | `numpy==2.5.2` (BSD-3-Clause, 0BSD, MIT, Zlib, CC0), `packaging==26.3` (Apache-2.0 or BSD-2-Clause), `regex==2026.9.3` (Apache-2.0 and CNRI-Python), `torch==2.14.0+cpu` (Apache-2.0, LLVM exception, BSD, BSL-1.0, MIT) |
The selected `fastino/gliner2-privacy-filter-PII-multi` weights at revision
`c153999da5f4c509df4322b0c6a1baf3d2c284d7` are marked Apache-2.0 in the
[model card](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi). Its published
`microsoft/mdeberta-v3-base` base model is MIT. The Fastino training corpus is described as
synthetic but is not published, so the training process is not independently reproducible.
Before distributing the optional image or model pack:
1. retain the upstream license and notice files for all packaged wheels, system libraries, and weights;
2. archive `THOTHII_MODEL_REVISION` and the verified `MODEL_SHA256SUMS` beside the model;
3. verify that `pip check` succeeds in the isolated environment;
4. compare the installed distribution/version set with this inventory;
5. repeat the licensing review if an upstream artifact or dependency changes.
@@ -179,6 +179,16 @@ Urchade model. This does not prove Italian clinical accuracy: the published SPY
English and the training corpus is synthetic. Those are quality and reproducibility limitations,
not a reason to reintroduce a generative LLM into the classifier.
An implementation smoke test found a packaging incompatibility in the selected upstream versions:
the checkpoint was written by Transformers 5.8, while `gliner2[local]==2.0.0` requires
`transformers>=4.38,<5`. Every published checkpoint revision has the same tokenizer metadata, and
[upstream issue 145](https://github.com/fastino-ai/GLiNER2/issues/145) reports the identical failure.
ThothII therefore pins Transformers 4.57.6 and performs the minimal documented-key conversion in a
temporary local view, without changing the downloaded model or its checksums. This compatibility
shim was accepted only after an offline CPU smoke test detected Italian names, dates, locations,
and usernames; any unexpected metadata shape fails closed. A future upstream fix must replace,
not silently stack on, this shim.
## Fit with current ThothII design
The current backend already owns source sampling and the human-owned `Sensitive Data Flag`; source
@@ -84,7 +84,7 @@ stato ripreso nella descrizione. Deve quindi essere ripetuta sul comportamento p
- scope di sincronizzazione `tables`, `columns`, `relationships` e `all`;
- run durevoli, conferma delle differenze distruttive, cancellazione, recovery, eventi SSE e
fallback polling;
- Sensitive Data Flag, suggerimenti AI strutturali, review draft e storico operativo;
- Sensitive Data Flag, analisi locale di metadati e contenuti, review draft e storico operativo;
- scope di generazione `selected_columns`, `selected_tables`, `all` e `missing`;
- campionamento read-only, valori sintetici per colonne protette, batching, retry, stop, Unlock,
storico e consolidamento;
@@ -211,17 +211,17 @@ riapplicare le migrazioni e ripartire dal database reale.
| ID | P | Livello | Scenario | Risultato atteso |
| --- | --- | --- | --- | --- |
| PRV-01 | P0 | Repository/API | Prima sincronizzazione, re-sync e aggiunta di una colonna. | Il default è `false`; il valore umano delle colonne esistenti è preservato; la nuova colonna è esplicitamente da riesaminare ma non riceve uno stato audit inventato. |
| PRV-02 | P0 | API | Suggerimento per un database, tabelle selezionate e colonne selezionate; database multipli, target duplicati o mancanti. | Il provider riceve esattamente le colonne dello scope. Input ambigui sono rifiutati prima della chiamata e nessun flag cambia. |
| PRV-03 | P0 | Contratto | Ispezione del messaggio al classifier. | Sono presenti solo database, schema, tabella, colonna, tipo, nullabilità, PK e FK. Non compaiono righe, valori, commenti, descrizioni, flag corrente o segreti. |
| PRV-04 | P1 | Contratto | `wide_entity`, identificatori lunghi e limite byte. | Ordine deterministico, batch massimi di dieci colonne e rispetto del limite messaggio; una singola colonna non rappresentabile fallisce prima del provider con errore sicuro. |
| PRV-05 | P0 | API | Risposta valida, fenced/prosa, JSON malformato, target mancante/duplicato/ignoto e provider failure. | Ogni colonna richiesta compare una sola volta. Una classificazione invalida viene ritentata una volta; dopo esaurimento si ottiene errore sanitizzato e nessuna modifica. |
| PRV-02 | P0 | API | Analisi per un database, tabelle selezionate e colonne selezionate; database multipli, target duplicati o mancanti. | Solo i target dello scope raggiungono l'adapter read-only. Input ambigui sono rifiutati prima della lettura e nessun flag cambia. |
| PRV-03 | P0 | Unit/Contratto | Valori con email, codice fiscale italiano valido, IBAN, carta con Luhn, chiave privata, chiave JSON sensibile e testo oltre 500 caratteri. | Un solo riscontro validato rende l'intera colonna `sensitive`; l'evidenza contiene solo rule ID e conteggi sanitizzati, mai il valore. |
| PRV-04 | P0 | Integrazione | Scansione completa oltre cinque secondi, timeout PostgreSQL e budget globale di sessanta secondi. | L'adapter passa al campionamento, ripristina la transazione dopo `statement_timeout`, resta read-only e non supera la deadline. Copertura incompleta senza match produce `unknown`. |
| PRV-05 | P0 | Unit/API | Tabella vuota, colonna all-null, binario non ispezionabile, scan completo senza match e scan incompleto senza match. | Gli esiti sono rispettivamente `unknown`, `unknown`, `unknown`, `non_sensitive` e `unknown`; `unknown` conserva la scelta umana corrente. |
| PRV-06 | P0 | UI/API | Apertura draft, modifica manuale, chiusura/reload e salvataggio. | La proposta non è persistita prima di Save; reload la scarta. Il reviewer può invertire scelte; si salvano solo colonne cambiate con versione ottimistica; un conflitto richiede reload. |
| PRV-07 | P1 | Repository/UI | Tentativi completati, falliti e attivi al restart. | Ogni tentativo ha un run distinto con scope, modello, contatori ed eventi sanitizzati; startup marca `interrupted` i run attivi. Storico newest-first senza target ID, proposte, prompt, output grezzo o diagnostica provider. |
| PRV-07 | P1 | Repository/UI | Tentativi completati, falliti e attivi al restart. | Ogni tentativo ha un run distinto con scope, engine `local`, versione policy, tre contatori ed eventi sanitizzati; startup marca `interrupted` i run attivi. Storico newest-first senza target ID, proposte, valori o diagnostica worker. |
| PRV-08 | P0 | Integrazione | Generazione descrizioni su target con canary protetti. | Le colonne protette sono assenti dalla proiezione SQL, non semplicemente filtrate dopo la lettura. Se non rimangono colonne leggibili non viene eseguita una `SELECT`. Nessun canary protetto esce dal processo. |
| PRV-09 | P0 | Contratto/Integrazione | Tabella mista con colonne sensibili e pubbliche. | Per le sensibili il prompt contiene valori plausibili, deterministici e limitati derivati dai soli metadati, nello stesso formato dei campioni e senza etichettarli al modello come sintetici. Per le pubbliche: massimo cinque righe e cinque valori rappresentativi, valori troncati e transazione read-only chiusa con rollback. |
| PRV-10 | P1 | API | Cambio `false→true→false` dopo una descrizione già generata. | Il testo esistente non viene rigenerato retroattivamente. Solo le generazioni future cambiano fonte del contesto; tornando `false` il campionamento reale torna eleggibile. |
| PRV-11 | P0 | API/UI | Utente senza `database.manage`, modello non configurato, catalogo/provider indisponibile e richiesta interrotta. | Controlli nascosti/disabilitati in UI e rifiuto server-side; errori non espongono dettagli. Un tentativo fallito compare nello storico senza trasformarsi in audit della decisione umana. |
| PRV-12 | P1 | L2 | Corpus strutturale etichettato con identificatori personali, credenziali/token, salute, finanza, localizzazione e controlli non sensibili/ambigui, in inglese e italiano. | Si misurano precisione, recall e falsi negativi per modello. I campi critici mancati sono sottoposti al product owner; la soglia quantitativa va ratificata prima di diventare gate, perché il classifier è advisory e human-in-the-loop. |
| PRV-11 | P0 | API/UI | Utente senza `database.manage`, sorgente/adapter indisponibile, NER assente o in timeout e richiesta interrotta. | Controlli nascosti/disabilitati in UI e rifiuto server-side; errori non espongono dettagli. Il NER opzionale degrada alle regole/coverage senza selezionare un LLM. |
| PRV-12 | P1 | L2 | Corpus etichettato con identificatori personali, credenziali/token, salute, finanza, localizzazione e controlli non sensibili/ambigui, in inglese e italiano. | Si misurano precisione, recall, falsi negativi, copertura e latenza separatamente per policy deterministica e NER CPU. La soglia va ratificata prima di abilitare NER per default; il classifier resta advisory e human-in-the-loop. |
| PRV-13 | P0 | UI/E2E | Modifica di un flag nella review senza Save e tentativo immediato di generare descrizioni. | Gate di rilascio da formalizzare: la generazione deve essere bloccata finché il draft non è salvato o scartato. In alternativa la UI deve dichiarare inequivocabilmente che verrà usato il valore persistito; non è accettabile mostrare “protetto” e campionare come non protetto. |
## 9. Casi di test — generazione e consolidamento dei commenti
@@ -301,9 +301,9 @@ un valore protetto non può mai esserlo.
| Area | Evidenza automatica già presente | Gap principale |
| --- | --- | --- |
| Snapshot e sincronizzazione | `backend/test/catalog-schema-introspector.test.ts`, `catalog-schema-routes.test.ts`, `catalog-table-introspector.test.ts`, `catalog-repository.integration.test.ts` | Introspezione `pg_catalog` realmente end-to-end, parità live dei tre trasporti e un unico E2E con re-scan distruttivo. |
| Privacy | `catalog-description-generation-routes.test.ts`, `catalog-description-generation-worker.test.ts`, `catalog-description-source-sampler.test.ts`, `catalog-synthetic-sample-value.test.ts` | Prova canary integrata query→prompt→API/log e benchmark reale post-ADR 0011. |
| Privacy | `catalog-sensitivity-classifier.test.ts`, `catalog-sensitivity-value-source.test.ts`, `catalog-local-ner-detector.test.ts` e i test di route/review | Prova shadow PSD con report solo aggregato, corpus italiano etichettato e benchmark NER CPU post-ADR 0014. |
| Generazione | `catalog-description-generation-routes.test.ts`, `catalog-description-generation-worker.test.ts`, `catalog-description-generation.integration.test.ts` e test del helper | Accettazione reale aggiornata, cancellazione di una query PostgreSQL bloccata e integrazione ermetica fino all'endpoint LiteLLM locale. |
| UI | `DatabaseManagementPage.test.tsx`, `DescriptionGenerationDrawer.test.tsx`, `SensitiveDataSuggestionHistoryDrawer.test.tsx` | L'E2E Playwright corrente verifica soprattutto il layout, non il workflow funzionale. |
| UI | `DatabaseManagementPage.test.tsx`, `DescriptionGenerationDrawer.test.tsx`, `SensitivityAnalysisHistoryDrawer.test.tsx` | L'E2E Playwright corrente verifica soprattutto il layout, non il workflow funzionale. |
Nuovi asset consigliati:
@@ -320,7 +320,7 @@ Nuovi asset consigliati:
### Wave 1 — contratto rapido
- parser snapshot, introspector, scope e diff;
- classifier strutturale, batching e validazione output;
- classifier locale, validatori/checksum, copertura e fallback al campionamento;
- sampler, valori sintetici, prompt bounds e parser descrizioni;
- autorizzazione, redazione e race del coordinator.
@@ -340,8 +340,8 @@ Nuovi asset consigliati:
### Wave 4 — accettazione L2
- provider reale sul database collegato dopo classificazione e review dei flag;
- benchmark PRV-12 e rubric GEN-15;
- analisi shadow sul database collegato e review dei flag, senza scritture in sorgente;
- benchmark NER CPU PRV-12 e rubric GEN-15 per la generazione descrizioni;
- scansione finale di canary e segreti;
- approvazione del product owner.
@@ -418,7 +418,7 @@ sorgente reale.
mostrare la disclosure anche per tabelle/colonne selezionate; il piano considera entrambi P0.
- L'AbortSignal corrente va provato contro una query PostgreSQL realmente bloccata: la sola
cancellazione del helper non dimostra che la lettura sorgente sia interrompibile.
- Lo storico dei Sensitive Data Suggestion Run è operativo, non un audit delle decisioni umane.
- Lo storico dei Sensitivity Analysis Run è operativo, non un audit delle decisioni umane.
- La policy privacy non è ancora applicata allo schema-linking/LSH; nessun risultato di questo piano
deve essere presentato come copertura di quel percorso.
- Un provider reale resta non deterministico: il rilascio deve dipendere dai gate tecnici e dalla
+32 -22
View File
@@ -126,7 +126,7 @@ export interface CatalogColumn {
updatedAt: string;
}
export interface SensitiveDataSuggestion {
export interface SensitivityReviewItem {
columnId: string;
tableId: string;
tableName: string;
@@ -134,29 +134,40 @@ export interface SensitiveDataSuggestion {
version: number;
currentSensitive: boolean;
sensitive: boolean;
assessment: "sensitive" | "non_sensitive" | "unknown";
evidence: Array<{
kind: "metadata" | "content" | "length" | "ner" | "coverage";
ruleId: string;
label?: string;
confidence?: number;
}>;
observedValues: number;
}
export interface SensitiveDataSuggestions {
suggestions: SensitiveDataSuggestion[];
run: SensitiveDataSuggestionRun;
export interface SensitivityAnalysisResult {
suggestions: SensitivityReviewItem[];
run: SensitivityAnalysisRun;
}
export type SensitiveDataSuggestionRequest =
export type SensitivityAnalysisRequest =
| { scope: "all" }
| { scope: "selected_tables"; targetIds: string[] }
| { scope: "selected_columns"; targetIds: string[] };
export type SensitiveDataSuggestionStatus = "running" | "completed" | "failed" | "interrupted";
export type SensitivityAnalysisStatus = "running" | "completed" | "failed" | "interrupted";
export interface SensitiveDataSuggestionRun {
export interface SensitivityAnalysisRun {
id: string;
databaseId: string;
scope: SensitiveDataSuggestionRequest["scope"];
modelId: string;
status: SensitiveDataSuggestionStatus;
scope: SensitivityAnalysisRequest["scope"];
engine: "llm" | "local";
modelId: string | null;
policyVersion: string | null;
status: SensitivityAnalysisStatus;
total: number;
suggestedSensitive: number;
suggestedNonSensitive: number;
unknown: number;
inputTokens?: number;
cacheReadTokens?: number;
outputTokens?: number;
@@ -167,7 +178,7 @@ export interface SensitiveDataSuggestionRun {
errorSummary: string | null;
}
export interface SensitiveDataSuggestionEvent {
export interface SensitivityAnalysisEvent {
runId: string;
sequence: number;
level: "info" | "warning" | "error";
@@ -458,28 +469,27 @@ export const updateCatalogColumnSensitive = (
{ method: "PATCH", body: JSON.stringify({ version, sensitive }) },
);
export const suggestSensitiveFields = (
export const runSensitivityAnalysis = (
databaseId: string,
modelId: string,
selection: SensitiveDataSuggestionRequest,
selection: SensitivityAnalysisRequest,
) =>
apiFetch<SensitiveDataSuggestions>(
apiFetch<SensitivityAnalysisResult>(
`/catalog/databases/${encodeURIComponent(databaseId)}/sensitive-data-suggestions`,
{ method: "POST", body: JSON.stringify({ modelId, ...selection }) },
{ method: "POST", body: JSON.stringify(selection) },
);
export const getSensitiveDataSuggestionRun = (runId: string) =>
apiFetch<SensitiveDataSuggestionRun>(
export const getSensitivityAnalysisRun = (runId: string) =>
apiFetch<SensitivityAnalysisRun>(
`/catalog/sensitive-data-suggestion-runs/${encodeURIComponent(runId)}`,
);
export const listSensitiveDataSuggestionRuns = (limit = 50) =>
apiFetch<SensitiveDataSuggestionRun[]>(
export const listSensitivityAnalysisRuns = (limit = 50) =>
apiFetch<SensitivityAnalysisRun[]>(
`/catalog/sensitive-data-suggestion-runs?limit=${encodeURIComponent(String(limit))}`,
);
export const listSensitiveDataSuggestionEvents = (runId: string, after = 0) =>
apiFetch<SensitiveDataSuggestionEvent[]>(
export const listSensitivityAnalysisEvents = (runId: string, after = 0) =>
apiFetch<SensitivityAnalysisEvent[]>(
`/catalog/sensitive-data-suggestion-runs/${encodeURIComponent(runId)}/events-list?after=${after}`,
);
@@ -3,13 +3,13 @@ import { server } from "../test/msw";
import {
cancelDescriptionGenerationRun,
descriptionGenerationEventsUrl,
getSensitiveDataSuggestionRun,
getSensitivityAnalysisRun,
listDescriptionGenerationRuns,
listSensitiveDataSuggestionEvents,
listSensitiveDataSuggestionRuns,
listSensitivityAnalysisEvents,
listSensitivityAnalysisRuns,
unlockDescriptionGenerationRun,
type DescriptionGenerationRun,
type SensitiveDataSuggestionRun,
type SensitivityAnalysisRun,
} from "./catalog-databases";
const historicalRun: DescriptionGenerationRun = {
@@ -31,15 +31,18 @@ const historicalRun: DescriptionGenerationRun = {
errorSummary: null,
};
const sensitiveSuggestionRun: SensitiveDataSuggestionRun = {
const sensitiveSuggestionRun: SensitivityAnalysisRun = {
id: "99999999-9999-4999-8999-999999999999",
databaseId: "11111111-1111-4111-8111-111111111111",
scope: "selected_columns",
modelId: "local-qwen",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "completed",
total: 4,
suggestedSensitive: 2,
suggestedNonSensitive: 2,
unknown: 0,
createdAt: "2026-08-28T09:00:00Z",
startedAt: "2026-08-28T09:00:00Z",
updatedAt: "2026-08-28T09:00:01Z",
@@ -119,9 +122,9 @@ test("lists, reads, and replays persisted sensitive-suggestion history", async (
}),
);
await expect(listSensitiveDataSuggestionRuns(50)).resolves.toEqual([sensitiveSuggestionRun]);
await expect(getSensitiveDataSuggestionRun(sensitiveSuggestionRun.id)).resolves.toEqual(sensitiveSuggestionRun);
await expect(listSensitiveDataSuggestionEvents(sensitiveSuggestionRun.id, 2)).resolves.toEqual([event]);
await expect(listSensitivityAnalysisRuns(50)).resolves.toEqual([sensitiveSuggestionRun]);
await expect(getSensitivityAnalysisRun(sensitiveSuggestionRun.id)).resolves.toEqual(sensitiveSuggestionRun);
await expect(listSensitivityAnalysisEvents(sensitiveSuggestionRun.id, 2)).resolves.toEqual([event]);
expect(requestedLimit).toBe("50");
expect(requestedAfter).toBe("2");
});
+5 -5
View File
@@ -147,11 +147,11 @@ test.each([
["description_generation_target_ids_duplicate", "Description generation target IDs must be unique."],
["description_generation_no_eligible_targets", "No eligible catalog tables or columns need description generation."],
["catalog_table_not_found", "One or more selected catalog tables were not found."],
["sensitive_data_suggestion_invalid_response", "The model returned an incomplete or invalid classification. No suggestions were applied."],
["sensitive_data_suggestion_provider_unavailable", "The selected model could not complete the request. No suggestions were applied."],
["sensitive_data_suggestion_history_request_invalid", "Sensitive suggestion history parameters are invalid."],
["sensitive_data_suggestion_history_failed", "Sensitive suggestion history could not be loaded."],
["sensitive_data_suggestion_run_not_found", "The sensitive suggestion run was not found."],
["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."],
["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."],
["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."],
["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."],
["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."],
["relationship_not_found", "The relationship no longer exists. Refresh and try again."],
["relationship_duplicate", "This relationship already exists."],
["relationship_target_not_unique", "The target column must be the only primary-key column of its table."],
+10 -12
View File
@@ -26,9 +26,8 @@ const safeErrorCodes = new Set([
"sensitive_data_suggestion_request_invalid",
"sensitive_data_suggestion_target_ids_duplicate",
"sensitive_data_suggestion_no_columns",
"sensitive_data_suggestion_payload_too_large",
"sensitive_data_suggestion_invalid_response",
"sensitive_data_suggestion_provider_unavailable",
"sensitivity_source_unavailable",
"sensitivity_analysis_timeout",
"sensitive_data_suggestion_failed",
"sensitive_data_suggestion_history_request_invalid",
"sensitive_data_suggestion_history_failed",
@@ -90,16 +89,15 @@ const localCodeMessages: Record<string, string> = {
catalog_column_not_found: "The selected catalog column was not found.",
catalog_table_not_found: "One or more selected catalog tables were not found.",
workspace_configuration_unavailable: "The database workspace configuration is unavailable.",
sensitive_data_suggestion_request_invalid: "Select a database, one or more tables, or one or more columns before requesting sensitive-field suggestions.",
sensitive_data_suggestion_request_invalid: "Select a database, one or more tables, or one or more columns before running sensitivity analysis.",
sensitive_data_suggestion_target_ids_duplicate: "Each selected table or column can be included only once.",
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to classify.",
sensitive_data_suggestion_payload_too_large: "The selected structural metadata cannot be divided into safe model requests.",
sensitive_data_suggestion_invalid_response: "The model returned an incomplete or invalid classification. No suggestions were applied.",
sensitive_data_suggestion_provider_unavailable: "The selected model could not complete the request. No suggestions were applied.",
sensitive_data_suggestion_failed: "Sensitive-field suggestions failed before review. No changes were applied.",
sensitive_data_suggestion_history_request_invalid: "Sensitive suggestion history parameters are invalid.",
sensitive_data_suggestion_history_failed: "Sensitive suggestion history could not be loaded.",
sensitive_data_suggestion_run_not_found: "The sensitive suggestion run was not found.",
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.",
sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.",
sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.",
sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.",
sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.",
sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.",
sensitive_data_suggestion_run_not_found: "The sensitivity analysis run was not found.",
schema_sync_conflict: "A schema synchronization is already active or no longer current.",
schema_introspection_failed: "The database schema could not be read safely.",
schema_request_invalid: "The schema request is invalid.",
@@ -9,7 +9,7 @@ import type {
CatalogSyncRun,
CatalogTable,
DescriptionGenerationRun,
SensitiveDataSuggestionRun,
SensitivityAnalysisRun,
} from "../api/catalog-databases";
import { Toaster } from "../components/ui/sonner";
import { DatabaseManagementPage } from "./DatabaseManagementPage";
@@ -159,18 +159,21 @@ function makeDescriptionGenerationRun(
};
}
function makeSensitiveDataSuggestionRun(
overrides: Partial<SensitiveDataSuggestionRun> = {},
): SensitiveDataSuggestionRun {
function makeSensitivityAnalysisRun(
overrides: Partial<SensitivityAnalysisRun> = {},
): SensitivityAnalysisRun {
return {
id: "99999999-9999-4999-8999-999999999999",
databaseId: "11111111-1111-4111-8111-111111111111",
modelId: "local-qwen",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
scope: "selected_columns",
status: "completed",
total: 2,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
unknown: 0,
createdAt: "2026-08-28T11:00:00Z",
startedAt: "2026-08-28T11:00:00Z",
updatedAt: "2026-08-28T11:00:01Z",
@@ -505,7 +508,7 @@ test("keeps both run-history buttons visible beside the metadata-description sel
name: "View description generation history",
});
const suggestionHistoryButton = within(metadataControls).getByRole("button", {
name: "View sensitive suggestion history",
name: "View sensitivity analysis history",
});
expect(toolbar).toHaveClass("sm:items-end", "sm:justify-between");
@@ -517,7 +520,7 @@ test("keeps both run-history buttons visible beside the metadata-description sel
expect(descriptionHistoryButton).toHaveClass("disabled:opacity-70");
expect(suggestionHistoryButton).toBeVisible();
expect(suggestionHistoryButton).toBeEnabled();
expect(suggestionHistoryButton).toHaveTextContent("View sensitive suggestion history");
expect(suggestionHistoryButton).toHaveTextContent("View sensitivity analysis history");
expect(toolbar.lastElementChild).toBe(actions);
expect(actions).toHaveClass("sm:justify-end");
expect(actions).toContainElement(screen.getByRole("button", { name: "Refresh" }));
@@ -533,8 +536,8 @@ test("keeps both run-history buttons visible beside the metadata-description sel
await user.click(suggestionHistoryButton);
expect(screen.queryByRole("dialog", { name: "Description generation" })).not.toBeInTheDocument();
const suggestionDrawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
expect(within(suggestionDrawer).getByText("No sensitive suggestion runs yet.")).toBeVisible();
const suggestionDrawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
expect(within(suggestionDrawer).getByText("No sensitivity analysis runs yet.")).toBeVisible();
});
test("changes the metadata-description model in page-local state", async () => {
@@ -1642,8 +1645,8 @@ test("observes an active run from another browser and reopens a terminal run fro
expect(await within(drawer).findByRole("heading", { name: "Completed" })).toBeVisible();
});
test("keeps the sensitive suggestion history label stable while showing active status separately", async () => {
const runningRun = makeSensitiveDataSuggestionRun({
test("keeps the sensitivity analysis history label stable while showing active status separately", async () => {
const runningRun = makeSensitivityAnalysisRun({
status: "running",
finishedAt: null,
updatedAt: new Date().toISOString(),
@@ -1655,12 +1658,12 @@ test("keeps the sensitive suggestion history label stable while showing active s
renderPage();
const historyButton = await screen.findByRole("button", {
name: "View sensitive suggestion history",
name: "View sensitivity analysis history",
});
expect(historyButton).toHaveTextContent("View sensitive suggestion history");
expect(historyButton).toHaveTextContent("View sensitivity analysis history");
await waitFor(() => expect(historyButton).toHaveAttribute(
"title",
"Sensitive suggestion generation is active",
"Sensitivity analysis is active",
));
expect(historyButton.querySelector("[aria-hidden='true'].bg-primary")).not.toBeNull();
});
@@ -2015,7 +2018,7 @@ test("starts one selected column with the configured default model", async () =>
expect(await screen.findByText("Description generation started for 1 column")).toBeVisible();
});
test("shows database sensitive suggestions only for a selection and rejects multiple databases clearly", async () => {
test("shows database sensitivity analysis only for a selection and rejects multiple databases clearly", async () => {
const user = userEvent.setup();
let suggestionCalls = 0;
const radiology = makeDatabase({
@@ -2035,24 +2038,24 @@ test("shows database sensitive suggestions only for a selection and rejects mult
);
renderPage({ rows: [makeDatabase(), radiology] });
expect(screen.queryByRole("button", { name: "Suggest sensitive fields" })).not.toBeInTheDocument();
expect(screen.queryByRole("button", { name: "Analyze sensitive fields" })).not.toBeInTheDocument();
const psdRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
const radiologyRow = await screen.findByRole("row", { name: /Radiology/ });
await user.click(within(psdRow).getByRole("checkbox", { name: /toggle row selection/i }));
expect(screen.getByRole("button", { name: "Suggest sensitive fields" })).toBeVisible();
expect(screen.getByRole("button", { name: "Analyze sensitive fields" })).toBeVisible();
await user.click(within(radiologyRow).getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
expect(await screen.findByText("Sensitive-field suggestions can be requested for only one database at a time. Select one database and try again.")).toBeVisible();
expect(await screen.findByText("Sensitivity analysis can run for only one database at a time. Select one database and try again.")).toBeVisible();
expect(suggestionCalls).toBe(0);
});
test("requests database-level sensitive suggestions for the only selected database", async () => {
test("requests database-level sensitivity analysis for the only selected database", async () => {
const user = userEvent.setup();
let suggestionBody: unknown;
let suggestionFinished = false;
let historyCalls = 0;
const run = makeSensitiveDataSuggestionRun({ scope: "all", total: 1 });
const run = makeSensitivityAnalysisRun({ scope: "all", total: 1 });
server.use(
http.get("/api/catalog/metadata-generation/models", () => HttpResponse.json({
models: [{ id: "local-qwen", label: "Local Qwen" }],
@@ -2075,6 +2078,9 @@ test("requests database-level sensitive suggestions for the only selected databa
version: patientIdColumn.version,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
}],
});
}),
@@ -2083,14 +2089,14 @@ test("requests database-level sensitive suggestions for the only selected databa
const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
await waitFor(() => expect(suggestionBody).toEqual({ modelId: "local-qwen", scope: "all" }));
await waitFor(() => expect(suggestionBody).toEqual({ scope: "all" }));
expect(await screen.findByRole("dialog", { name: "Sensitive field review" })).toBeVisible();
await waitFor(() => expect(historyCalls).toBeGreaterThanOrEqual(2));
});
test("refetches sensitive suggestion history after a failed request", async () => {
test("refetches sensitivity analysis history after a failed request", async () => {
const user = userEvent.setup();
let historyCalls = 0;
server.use(
@@ -2104,8 +2110,8 @@ test("refetches sensitive suggestion history after a failed request", async () =
}),
http.post("/api/catalog/databases/:databaseId/sensitive-data-suggestions", () => (
HttpResponse.json({
code: "sensitive_data_suggestion_invalid_response",
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
code: "sensitivity_source_scan_failed",
message: "Sensitivity analysis could not read the source. No assessments were applied.",
}, { status: 502 })
)),
);
@@ -2113,13 +2119,13 @@ test("refetches sensitive suggestion history after a failed request", async () =
const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
await waitFor(() => expect(historyCalls).toBeGreaterThanOrEqual(2));
expect(screen.queryByRole("dialog", { name: "Sensitive field review" })).not.toBeInTheDocument();
});
test("requests sensitive suggestions only for selected tables", async () => {
test("requests sensitivity analysis only for selected tables", async () => {
const user = userEvent.setup();
const visitsTable: CatalogTable = {
...patientsTable,
@@ -2153,6 +2159,9 @@ test("requests sensitive suggestions only for selected tables", async () => {
version: visitColumn.version,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
}],
});
}),
@@ -2163,17 +2172,16 @@ test("requests sensitive suggestions only for selected tables", async () => {
await user.click(screen.getByRole("tab", { name: "Tables" }));
const visitsRow = await screen.findByRole("row", { name: /visits/ });
await user.click(within(visitsRow).getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
await waitFor(() => expect(suggestionBody).toEqual({
modelId: "local-qwen",
scope: "selected_tables",
targetIds: [visitsTable.id],
}));
expect(await screen.findByRole("dialog", { name: "Sensitive field review" })).toBeVisible();
});
test("reviews AI-sensitive-field suggestions as an editable draft and saves only changed columns", async () => {
test("allows a human downgrade and saves only explicit sensitivity changes", async () => {
const user = userEvent.setup();
const idColumn = { ...patientIdColumn, sensitive: false };
const nameColumn = {
@@ -2185,7 +2193,7 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
isPrimaryKey: false,
description: "Patient name",
generatedDescription: "Name of the patient",
sensitive: false,
sensitive: true,
};
const unselectedColumn = {
...patientIdColumn,
@@ -2225,6 +2233,9 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
version: idColumn.version,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
},
{
columnId: nameColumn.id,
@@ -2232,8 +2243,11 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
tableName: patientsTable.name,
columnName: nameColumn.name,
version: nameColumn.version,
currentSensitive: false,
sensitive: true,
currentSensitive: true,
sensitive: false,
assessment: "non_sensitive",
evidence: [],
observedValues: 2,
},
],
});
@@ -2263,8 +2277,8 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
const idSensitive = await screen.findByRole("checkbox", { name: "Sensitive data for id" });
const nameSensitive = await screen.findByRole("checkbox", { name: "Sensitive data for name" });
expect(idSensitive).not.toBeChecked();
expect(nameSensitive).not.toBeChecked();
expect(screen.queryByRole("button", { name: "Suggest sensitive fields" })).not.toBeInTheDocument();
expect(nameSensitive).toBeChecked();
expect(screen.queryByRole("button", { name: "Analyze sensitive fields" })).not.toBeInTheDocument();
const selectableRow = async (name: RegExp) => {
const rows = await screen.findAllByRole("row", { name });
@@ -2274,31 +2288,30 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
.getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(within(await selectableRow(/Patient name/))
.getByRole("checkbox", { name: /toggle row selection/i }));
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
await waitFor(() => expect(suggestionBody).toEqual({
modelId: "local-qwen",
scope: "selected_columns",
targetIds: [idColumn.id, nameColumn.id],
}));
const review = await screen.findByRole("dialog", { name: "Sensitive field review" });
expect(within(review).getByRole("checkbox", { name: "Protect patients.id" })).toBeChecked();
expect(within(review).getByRole("checkbox", { name: "Protect patients.name" })).toBeChecked();
expect(within(review).getByRole("checkbox", { name: "Protect patients.name" })).not.toBeChecked();
expect(patches).toHaveLength(0);
await user.click(within(review).getByRole("checkbox", { name: "Protect patients.name" }));
await user.click(within(review).getByRole("checkbox", { name: "Protect patients.id" }));
await user.click(within(review).getByRole("button", { name: "Save 1" }));
await waitFor(() => expect(patches).toEqual([{
columnId: idColumn.id,
columnId: nameColumn.id,
body: {
version: idColumn.version,
sensitive: true,
version: nameColumn.version,
sensitive: false,
},
}]));
expect(await screen.findByText("Saved 1 sensitive flag")).toBeVisible();
await waitFor(() => expect(screen.queryByRole("dialog", { name: "Sensitive field review" })).not.toBeInTheDocument());
await waitFor(() => expect(screen.getByRole("checkbox", { name: "Sensitive data for id" })).toBeChecked());
await waitFor(() => expect(screen.getByRole("checkbox", { name: "Sensitive data for id" })).not.toBeChecked());
expect(screen.getByRole("checkbox", { name: "Sensitive data for name" })).not.toBeChecked();
});
+58 -68
View File
@@ -18,11 +18,11 @@ import {
listCatalogDatabases,
listDescriptionGenerationRuns,
listMetadataGenerationModels,
listSensitiveDataSuggestionRuns,
listSensitivityAnalysisRuns,
replaceCatalogDatabaseSecrets,
startCatalogSync,
startDescriptionGenerationRun,
suggestSensitiveFields,
runSensitivityAnalysis,
testCatalogDatabase,
updateCatalogDatabase,
type CatalogDatabase,
@@ -34,9 +34,9 @@ import {
type DatabaseBinding,
type DatabaseTransport,
type DescriptionGenerationRun,
type SensitiveDataSuggestion,
type SensitiveDataSuggestionRequest,
type SensitiveDataSuggestionRun,
type SensitivityReviewItem,
type SensitivityAnalysisRequest,
type SensitivityAnalysisRun,
} from "../api/catalog-databases";
import { DatabaseGrid } from "./database-management/DatabaseGrid";
import { DatabaseForm } from "./database-management/DatabaseForm";
@@ -46,7 +46,7 @@ import { DatabaseRelationships } from "./database-management/DatabaseRelationshi
import { CatalogSyncDrawer } from "./database-management/CatalogSyncDrawer";
import { MetadataGenerationModelSelector } from "./database-management/MetadataGenerationModelSelector";
import { DescriptionGenerationDrawer } from "./database-management/DescriptionGenerationDrawer";
import { SensitiveDataSuggestionHistoryDrawer } from "./database-management/SensitiveDataSuggestionHistoryDrawer";
import { SensitivityAnalysisHistoryDrawer } from "./database-management/SensitivityAnalysisHistoryDrawer";
import { SensitiveDataReviewDrawer } from "./database-management/SensitiveDataReviewDrawer";
import {
FleetLedgerBreadcrumb,
@@ -158,15 +158,15 @@ export function DatabaseManagementPage({
const observedActiveDescriptionGenerationRun = descriptionGenerationRuns.find(
(run) => isDescriptionGenerationActive(run),
);
const sensitiveDataSuggestionRunsQuery = useQuery({
const sensitivityAnalysisRunsQuery = useQuery({
queryKey: SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY,
queryFn: () => listSensitiveDataSuggestionRuns(50),
queryFn: () => listSensitivityAnalysisRuns(50),
enabled: canManage,
retry: false,
refetchInterval: 5_000,
});
const sensitiveDataSuggestionRuns = sensitiveDataSuggestionRunsQuery.data ?? [];
const observedActiveSensitiveDataSuggestionRun = sensitiveDataSuggestionRuns.find(
const sensitivityAnalysisRuns = sensitivityAnalysisRunsQuery.data ?? [];
const observedActiveSensitivityAnalysisRun = sensitivityAnalysisRuns.find(
(run) => run.status === "running",
);
@@ -196,12 +196,12 @@ export function DatabaseManagementPage({
const [syncDrawerOpen, setSyncDrawerOpen] = useState(false);
const [activeDescriptionGenerationRun, setActiveDescriptionGenerationRun] = useState<DescriptionGenerationRun | null>(null);
const [descriptionGenerationDrawerOpen, setDescriptionGenerationDrawerOpen] = useState(false);
const [activeSensitiveDataSuggestionRun, setActiveSensitiveDataSuggestionRun] = useState<SensitiveDataSuggestionRun | null>(null);
const [sensitiveDataSuggestionHistoryDrawerOpen, setSensitiveDataSuggestionHistoryDrawerOpen] = useState(false);
const [activeSensitivityAnalysisRun, setActiveSensitivityAnalysisRun] = useState<SensitivityAnalysisRun | null>(null);
const [sensitivityAnalysisHistoryDrawerOpen, setSensitivityAnalysisHistoryDrawerOpen] = useState(false);
const [sensitiveReview, setSensitiveReview] = useState<{
databaseId: string;
scopeLabel: string;
suggestions: SensitiveDataSuggestion[];
suggestions: SensitivityReviewItem[];
} | null>(null);
useEffect(() => {
@@ -694,27 +694,27 @@ export function DatabaseManagementPage({
?? scopedRuns[0]
?? scopedSelectedRun;
if (run) setActiveDescriptionGenerationRun(run);
setSensitiveDataSuggestionHistoryDrawerOpen(false);
setSensitivityAnalysisHistoryDrawerOpen(false);
setDescriptionGenerationDrawerOpen(true);
}, [activeDescriptionGenerationRun, activeRow, descriptionGenerationRuns, observedActiveDescriptionGenerationRun]);
const openSensitiveDataSuggestionHistory = useCallback(() => {
const scopedRuns = activeRow ? sensitiveDataSuggestionRuns.filter((item) => item.databaseId === activeRow.id) : sensitiveDataSuggestionRuns;
const scopedActiveRun = observedActiveSensitiveDataSuggestionRun
&& (!activeRow || observedActiveSensitiveDataSuggestionRun.databaseId === activeRow.id)
? observedActiveSensitiveDataSuggestionRun
const openSensitivityReviewItemHistory = useCallback(() => {
const scopedRuns = activeRow ? sensitivityAnalysisRuns.filter((item) => item.databaseId === activeRow.id) : sensitivityAnalysisRuns;
const scopedActiveRun = observedActiveSensitivityAnalysisRun
&& (!activeRow || observedActiveSensitivityAnalysisRun.databaseId === activeRow.id)
? observedActiveSensitivityAnalysisRun
: undefined;
const scopedSelectedRun = activeSensitiveDataSuggestionRun
&& (!activeRow || activeSensitiveDataSuggestionRun.databaseId === activeRow.id)
? activeSensitiveDataSuggestionRun
const scopedSelectedRun = activeSensitivityAnalysisRun
&& (!activeRow || activeSensitivityAnalysisRun.databaseId === activeRow.id)
? activeSensitivityAnalysisRun
: undefined;
const run = scopedActiveRun
?? scopedRuns[0]
?? scopedSelectedRun;
if (run) setActiveSensitiveDataSuggestionRun(run);
if (run) setActiveSensitivityAnalysisRun(run);
setDescriptionGenerationDrawerOpen(false);
setSensitiveDataSuggestionHistoryDrawerOpen(true);
}, [activeRow, activeSensitiveDataSuggestionRun, observedActiveSensitiveDataSuggestionRun, sensitiveDataSuggestionRuns]);
setSensitivityAnalysisHistoryDrawerOpen(true);
}, [activeRow, activeSensitivityAnalysisRun, observedActiveSensitivityAnalysisRun, sensitivityAnalysisRuns]);
const descriptionGenerationTerminated = useCallback(async (run: DescriptionGenerationRun) => {
await Promise.all([
@@ -788,39 +788,35 @@ export function DatabaseManagementPage({
const requestSensitiveSuggestions = useCallback(async (
database: CatalogDatabase,
selection: SensitiveDataSuggestionRequest,
selection: SensitivityAnalysisRequest,
scopeLabel: string,
) => {
if (!database.id) {
toast.error("The selected database is not configured, so sensitive-field suggestions were not requested.");
toast.error("The selected database is not configured, so sensitivity analysis was not started.");
throw new Error("database is not configured");
}
if (!selectedMetadataModel) {
toast.error("Select a metadata-generation model before requesting sensitive-field suggestions.");
throw new Error("metadata-generation model is not selected");
}
try {
const result = await suggestSensitiveFields(database.id, selectedMetadataModel, selection);
const result = await runSensitivityAnalysis(database.id, selection);
if (result.run) {
setActiveSensitiveDataSuggestionRun(result.run);
queryClient.setQueryData<SensitiveDataSuggestionRun[]>(
setActiveSensitivityAnalysisRun(result.run);
queryClient.setQueryData<SensitivityAnalysisRun[]>(
SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY,
(current = []) => [result.run, ...current.filter((item) => item.id !== result.run.id)],
);
}
setSensitiveReview({ databaseId: database.id, scopeLabel, suggestions: result.suggestions });
toast.success(`Prepared ${result.suggestions.length} sensitive-field suggestion${result.suggestions.length === 1 ? "" : "s"} for review`);
toast.success(`Prepared ${result.suggestions.length} local sensitivity assessment${result.suggestions.length === 1 ? "" : "s"} for review`);
} catch (error) {
toast.error(apiErrorMessage(error));
throw error;
} finally {
await queryClient.invalidateQueries({ queryKey: SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY });
}
}, [queryClient, selectedMetadataModel]);
}, [queryClient]);
const suggestDatabaseSensitiveFields = useCallback(async (selected: CatalogDatabase[]) => {
if (selected.length !== 1) {
toast.error("Sensitive-field suggestions can be requested for only one database at a time. Select one database and try again.");
toast.error("Sensitivity analysis can run for only one database at a time. Select one database and try again.");
throw new Error("more than one database selected");
}
const database = selected[0]!;
@@ -828,11 +824,11 @@ export function DatabaseManagementPage({
}, [requestSensitiveSuggestions]);
const suggestActiveDatabaseSensitiveFields = useCallback(async (
selection: SensitiveDataSuggestionRequest,
selection: SensitivityAnalysisRequest,
scopeLabel: string,
) => {
if (!activeRow) {
toast.error("The database is no longer available, so sensitive-field suggestions were not requested.");
toast.error("The database is no longer available, so sensitivity analysis was not started.");
throw new Error("database is no longer available");
}
await requestSensitiveSuggestions(activeRow, selection, scopeLabel);
@@ -952,10 +948,10 @@ export function DatabaseManagementPage({
? `Synchronization: ${currentActiveRun.phase.replaceAll("_", " ")}`
: descriptionGenerationActive
? "Description generation active"
: observedActiveSensitiveDataSuggestionRun
: observedActiveSensitivityAnalysisRun
? "Sensitive analysis active"
: "Idle";
const operationTone = currentActiveRun || descriptionGenerationActive || observedActiveSensitiveDataSuggestionRun
const operationTone = currentActiveRun || descriptionGenerationActive || observedActiveSensitivityAnalysisRun
? "warning" as const
: "neutral" as const;
const fleetBreadcrumb = screen.kind === "tables" || screen.kind === "relationships"
@@ -987,15 +983,12 @@ export function DatabaseManagementPage({
onRunUpdate={setActiveDescriptionGenerationRun}
onTerminal={(run) => void descriptionGenerationTerminated(run)}
/>
<SensitiveDataSuggestionHistoryDrawer
open={sensitiveDataSuggestionHistoryDrawerOpen}
<SensitivityAnalysisHistoryDrawer
open={sensitivityAnalysisHistoryDrawerOpen}
databaseId={activeRow?.id ?? null}
run={activeSensitiveDataSuggestionRun}
modelLabel={metadataModels.find(
(model) => model.id === activeSensitiveDataSuggestionRun?.modelId,
)?.label ?? activeSensitiveDataSuggestionRun?.modelId ?? ""}
onClose={() => setSensitiveDataSuggestionHistoryDrawerOpen(false)}
onRunUpdate={setActiveSensitiveDataSuggestionRun}
run={activeSensitivityAnalysisRun}
onClose={() => setSensitivityAnalysisHistoryDrawerOpen(false)}
onRunUpdate={setActiveSensitivityAnalysisRun}
/>
<SensitiveDataReviewDrawer
open={Boolean(sensitiveReview)}
@@ -1054,7 +1047,7 @@ export function DatabaseManagementPage({
onSync={(scope) => void syncDatabase(scope)}
onOpenSync={() => openSync(activeRow)}
onOpenDescriptionHistory={openDescriptionGenerationHistory}
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
activeSyncRun={currentActiveRun}
/>
</FleetLedgerDrawer>
@@ -1091,7 +1084,7 @@ export function DatabaseManagementPage({
onRunStarted={rememberSyncRun}
onOpenSync={() => openSync(activeRow)}
onOpenDescriptionHistory={openDescriptionGenerationHistory}
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
onCatalogMetricsChanged={invalidateCatalogMetrics}
onNavigationStateChange={setTablesNavigationState}
/>
@@ -1128,7 +1121,7 @@ export function DatabaseManagementPage({
descriptionGenerationActive={descriptionGenerationActive}
onGenerateDescriptions={generateDatabaseDescriptions}
onSuggestSensitive={suggestDatabaseSensitiveFields}
sensitiveDataSuggestionRuns={sensitiveDataSuggestionRuns}
sensitivityAnalysisRuns={sensitivityAnalysisRuns}
onDeleteMetadataSelected={deleteSelectedMetadata}
onRefresh={refreshList}
/>
@@ -1194,10 +1187,10 @@ export function DatabaseManagementPage({
aria-label="View sensitive analysis progress"
title="View sensitive analysis progress and history"
disabled={!canManage}
onClick={openSensitiveDataSuggestionHistory}
onClick={openSensitivityReviewItemHistory}
>
<History aria-hidden="true" />
{observedActiveSensitiveDataSuggestionRun ? "Sensitive analysis active" : "Sensitive analysis history"}
{observedActiveSensitivityAnalysisRun ? "Sensitive analysis active" : "Sensitive analysis history"}
</Button>
</>
)}
@@ -1276,10 +1269,10 @@ export function DatabaseManagementPage({
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View description generation history" disabled={!canManage} title="View description generation history" onClick={openDescriptionGenerationHistory}>
<History /> View description generation history
</Button>
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View sensitive suggestion history" disabled={!canManage} title={observedActiveSensitiveDataSuggestionRun ? "Sensitive suggestion generation is active" : "View sensitive suggestion history"} onClick={openSensitiveDataSuggestionHistory}>
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View sensitivity analysis history" disabled={!canManage} title={observedActiveSensitivityAnalysisRun ? "Sensitivity analysis is active" : "View sensitivity analysis history"} onClick={openSensitivityReviewItemHistory}>
<History />
{observedActiveSensitiveDataSuggestionRun ? <span aria-hidden="true" className="size-2 rounded-full bg-primary" /> : null}
View sensitive suggestion history
{observedActiveSensitivityAnalysisRun ? <span aria-hidden="true" className="size-2 rounded-full bg-primary" /> : null}
View sensitivity analysis history
</Button>
</> : null}
</div>
@@ -1333,7 +1326,7 @@ export function DatabaseManagementPage({
onSyncSelected={syncSelected}
selectedMetadataModel={selectedMetadataModelAvailable ? selectedMetadataModel : null}
descriptionGenerationActive={descriptionGenerationActive}
sensitiveDataSuggestionRuns={sensitiveDataSuggestionRuns}
sensitivityAnalysisRuns={sensitivityAnalysisRuns}
onGenerateDescriptions={generateDatabaseDescriptions}
onSuggestSensitive={suggestDatabaseSensitiveFields}
onDeleteMetadataSelected={deleteSelectedMetadata}
@@ -1370,7 +1363,7 @@ export function DatabaseManagementPage({
onSync={(scope) => void syncDatabase(scope)}
onOpenSync={() => openSync(activeRow)}
onOpenDescriptionHistory={openDescriptionGenerationHistory}
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
activeSyncRun={currentActiveRun}
/>
) : null}
@@ -1405,7 +1398,7 @@ export function DatabaseManagementPage({
onRunStarted={rememberSyncRun}
onOpenSync={() => openSync(activeRow)}
onOpenDescriptionHistory={openDescriptionGenerationHistory}
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
onCatalogMetricsChanged={invalidateCatalogMetrics}
onNavigationStateChange={setTablesNavigationState}
/>
@@ -1431,15 +1424,12 @@ export function DatabaseManagementPage({
onRunUpdate={setActiveDescriptionGenerationRun}
onTerminal={(run) => void descriptionGenerationTerminated(run)}
/>
<SensitiveDataSuggestionHistoryDrawer
open={sensitiveDataSuggestionHistoryDrawerOpen}
<SensitivityAnalysisHistoryDrawer
open={sensitivityAnalysisHistoryDrawerOpen}
databaseId={activeRow?.id ?? null}
run={activeSensitiveDataSuggestionRun}
modelLabel={metadataModels.find(
(model) => model.id === activeSensitiveDataSuggestionRun?.modelId,
)?.label ?? activeSensitiveDataSuggestionRun?.modelId ?? ""}
onClose={() => setSensitiveDataSuggestionHistoryDrawerOpen(false)}
onRunUpdate={setActiveSensitiveDataSuggestionRun}
run={activeSensitivityAnalysisRun}
onClose={() => setSensitivityAnalysisHistoryDrawerOpen(false)}
onRunUpdate={setActiveSensitivityAnalysisRun}
/>
<SensitiveDataReviewDrawer
open={Boolean(sensitiveReview)}
@@ -17,7 +17,7 @@ import {
type CatalogSyncRun,
type CatalogTable,
type DescriptionGenerationRun,
type SensitiveDataSuggestionRequest,
type SensitivityAnalysisRequest,
} from "../../api/catalog-databases";
import type { DatabaseNavigationState } from "./model";
import { FleetActionSelector, type FleetActionOption } from "./FleetActionSelector";
@@ -34,7 +34,7 @@ interface Props {
onDescriptionGenerationRunStarted: (run: DescriptionGenerationRun) => void;
onCatalogSyncRunStarted?: (run: CatalogSyncRun) => void;
onNavigationStateChange: (state: DatabaseNavigationState) => void;
onSuggestSensitive: (selection: SensitiveDataSuggestionRequest, scopeLabel: string) => Promise<void>;
onSuggestSensitive: (selection: SensitivityAnalysisRequest, scopeLabel: string) => Promise<void>;
catalogOperationActive?: boolean;
onCatalogMetricsChanged?: () => void | Promise<void>;
presentation?: "legacy" | "fleet";
@@ -351,17 +351,15 @@ export function DatabaseColumns({
},
{
id: "suggest-sensitive",
label: "Suggest sensitive fields",
label: "Analyze sensitive fields",
group: "Sensitive data",
runLabel: "Suggest",
disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || catalogOperationActive || busy,
runLabel: "Analyze",
disabled: !canManage || selectedIds.length === 0 || descriptionGenerationActive || catalogOperationActive || busy,
disabledReason: !canManage
? "You do not have permission to review sensitive data."
: selectedIds.length === 0
? "Select at least one column."
: !selectedMetadataModel
? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE
: descriptionGenerationActive || catalogOperationActive
: descriptionGenerationActive || catalogOperationActive
? "Wait for the active catalog operation to finish."
: busy
? "Another action is running."
@@ -504,7 +502,7 @@ export function DatabaseColumns({
</Menu.Positioner>
</Menu.Portal>
</Menu.Root>
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{sensitiveAction === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}</Button>
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{sensitiveAction === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}</Button>
{changedSensitiveColumns.length > 0 ? <Button type="button" disabled={!canManage || busy} onClick={() => void saveSensitive()}><Save />{sensitiveAction === "save" ? "Saving…" : "Save sensitive fields"}</Button> : null}
<Button type="button" variant="ghost" onClick={clearSelection}><X />Clear</Button>
</>
@@ -16,7 +16,7 @@ import { getCatalogMetrics } from "../../api/catalog-databases";
import type {
CatalogDatabase,
CatalogMetrics,
SensitiveDataSuggestionRun,
SensitivityAnalysisRun,
CatalogDatabaseMetadataDeleteTarget,
DescriptionGenerationScope,
} from "../../api/catalog-databases";
@@ -45,7 +45,7 @@ interface DatabaseGridProps {
onSyncSelected: (rows: CatalogDatabase[], scope: DatabaseSyncScope) => Promise<void>;
selectedMetadataModel: string | null;
descriptionGenerationActive: boolean;
sensitiveDataSuggestionRuns?: SensitiveDataSuggestionRun[];
sensitivityAnalysisRuns?: SensitivityAnalysisRun[];
onGenerateDescriptions: (
rows: CatalogDatabase[],
scope: Extract<DescriptionGenerationScope, "all" | "missing">,
@@ -161,7 +161,7 @@ function coverageStatus(total: number, complete: number): { label: string; detai
return { label: "Not started", detail: `0/${total}`, tone: "danger" };
}
function CatalogStatusCells({ row, metrics, sensitiveRun }: { row: CatalogDatabase; metrics?: CatalogMetrics; sensitiveRun?: SensitiveDataSuggestionRun }) {
function CatalogStatusCells({ row, metrics, sensitiveRun }: { row: CatalogDatabase; metrics?: CatalogMetrics; sensitiveRun?: SensitivityAnalysisRun }) {
const access = accessSummary(row);
const synchronized = row.schemaSyncedVersion === row.version;
const syncStatus = !row.schemaSyncedVersion
@@ -337,7 +337,7 @@ export function DatabaseGrid({
onSyncSelected,
selectedMetadataModel,
descriptionGenerationActive,
sensitiveDataSuggestionRuns = [],
sensitivityAnalysisRuns = [],
onGenerateDescriptions,
onSuggestSensitive,
onDeleteMetadataSelected,
@@ -361,8 +361,8 @@ export function DatabaseGrid({
return result;
}, [metricQueries, rows]);
const sensitiveRunsByDatabase = useMemo(() => new Map(
rows.filter((row) => row.id).map((row) => [row.id!, sensitiveDataSuggestionRuns.find((run) => run.databaseId === row.id)]),
), [rows, sensitiveDataSuggestionRuns]);
rows.filter((row) => row.id).map((row) => [row.id!, sensitivityAnalysisRuns.find((run) => run.databaseId === row.id)]),
), [rows, sensitivityAnalysisRuns]);
const compact = useCompactViewport();
const gridRef = useRef<AgGridReact<CatalogDatabase>>(null);
const actionsTriggerRef = useRef<HTMLButtonElement>(null);
@@ -432,7 +432,6 @@ export function DatabaseGrid({
&& selectedRows.every((row) => row.configured && row.id && !row.activeSyncRun);
const canSuggestSensitive = canManage
&& selectedRows.length === 1
&& Boolean(selectedMetadataModel)
&& !descriptionGenerationActive
&& selectedRows.every((row) => row.configured && row.id && !row.activeSyncRun);
const canDeleteMetadataSelection = canManage && selectedRows.length > 0
@@ -462,7 +461,6 @@ export function DatabaseGrid({
: undefined);
const sensitiveReason = permissionReason
?? (selectedRows.length !== 1 ? "Select exactly one database" : undefined)
?? (!selectedMetadataModel ? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE : undefined)
?? (descriptionGenerationActive ? "Wait for the active metadata operation" : undefined)
?? (selectedRows.some((row) => !row.configured || !row.id || row.activeSyncRun)
? "The selected database must be configured and idle"
@@ -479,7 +477,7 @@ export function DatabaseGrid({
{ id: "sync-all", label: "Synchronize all", group: "Synchronization", disabled: !canSyncSelection, disabledReason: syncReason },
{ id: "generate-missing", label: "Generate missing descriptions", group: "Descriptions", disabled: !canGenerateDescriptions, disabledReason: generationReason },
{ id: "generate-all", label: "Generate all descriptions", group: "Descriptions", disabled: !canGenerateDescriptions, disabledReason: generationReason, runLabel: "Review generation" },
{ id: "suggest-sensitive", label: "Suggest sensitive fields", group: "Sensitive data", disabled: !canSuggestSensitive, disabledReason: sensitiveReason, runLabel: "Open review" },
{ id: "suggest-sensitive", label: "Analyze sensitive fields", group: "Sensitive data", disabled: !canSuggestSensitive, disabledReason: sensitiveReason, runLabel: "Open review" },
{ id: "clear-tables", label: "Clear catalog tables", group: "Cleanup", disabled: !canDeleteMetadataSelection, disabledReason: cleanupReason, tone: "destructive", runLabel: "Review cleanup" },
{ id: "clear-relationships", label: "Clear catalog relationships", group: "Cleanup", disabled: !canDeleteMetadataSelection, disabledReason: cleanupReason, tone: "destructive", runLabel: "Review cleanup" },
];
@@ -723,7 +721,7 @@ export function DatabaseGrid({
title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined}
onClick={() => void perform("suggest", () => onSuggestSensitive(selectedRows))}
>
<Sparkles />{action === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}
<Sparkles />{action === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}
</Button>
<Button type="button" variant="ghost" disabled={action !== null} onClick={() => { gridRef.current?.api.deselectAll(); setSelectedRows([]); }}><X />Clear</Button>
</>
@@ -20,7 +20,7 @@ import {
type CatalogTable,
type CatalogTableMetadataDeleteTarget,
type DescriptionGenerationRun,
type SensitiveDataSuggestionRequest,
type SensitivityAnalysisRequest,
} from "../../api/catalog-databases";
import type { DatabaseNavigationState } from "./model";
import { DatabaseColumns } from "./DatabaseColumns";
@@ -43,7 +43,7 @@ interface Props {
onOpenSync: () => void;
onClearCatalogTables?: () => Promise<void>;
onDescriptionGenerationRunStarted: (run: DescriptionGenerationRun) => void;
onSuggestSensitive: (selection: SensitiveDataSuggestionRequest, scopeLabel: string) => Promise<void>;
onSuggestSensitive: (selection: SensitivityAnalysisRequest, scopeLabel: string) => Promise<void>;
onCatalogMetricsChanged?: () => void | Promise<void>;
presentation?: "legacy" | "fleet";
}
@@ -418,17 +418,15 @@ export function DatabaseTables({
},
{
id: "suggest-sensitive",
label: "Suggest sensitive fields",
label: "Analyze sensitive fields",
group: "Sensitive data",
runLabel: "Suggest",
disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || busy !== null || Boolean(currentRun),
runLabel: "Analyze",
disabled: !canManage || selectedIds.length === 0 || descriptionGenerationActive || busy !== null || Boolean(currentRun),
disabledReason: !canManage
? "You do not have permission to review sensitive data."
: selectedIds.length === 0
? "Select at least one table."
: !selectedMetadataModel
? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE
: descriptionGenerationActive || Boolean(currentRun)
: descriptionGenerationActive || Boolean(currentRun)
? "Wait for the active catalog operation to finish."
: busy !== null
? "Another action is running."
@@ -752,7 +750,7 @@ export function DatabaseTables({
</Menu.Positioner>
</Menu.Portal>
</Menu.Root>
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy !== null} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{busy === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}</Button>
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy !== null} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{busy === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}</Button>
<Button type="button" variant="ghost" onClick={clearSelection}><X />Clear</Button>
</>
)
@@ -8,7 +8,7 @@ describe("Recent runs success styling contract", () => {
test("every history renderer exposes its run status to the shared styles", () => {
const renderers = [
"DescriptionGenerationDrawer.tsx",
"SensitiveDataSuggestionHistoryDrawer.tsx",
"SensitivityAnalysisHistoryDrawer.tsx",
"CatalogSyncDrawer.tsx",
];
@@ -0,0 +1,97 @@
import { render, screen, waitFor } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { useState } from "react";
import { vi } from "vitest";
import type { CatalogColumn, SensitivityReviewItem } from "../../api/catalog-databases";
import { SensitiveDataReviewDrawer } from "./SensitiveDataReviewDrawer";
const updateCatalogColumnSensitive = vi.hoisted(() => vi.fn());
vi.mock("../../api/catalog-databases", () => ({ updateCatalogColumnSensitive }));
const suggestions: SensitivityReviewItem[] = [
{
columnId: "11111111-1111-4111-8111-111111111111",
tableId: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
tableName: "patients",
columnName: "email",
version: 3,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
},
{
columnId: "22222222-2222-4222-8222-222222222222",
tableId: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
tableName: "patients",
columnName: "notes",
version: 7,
currentSensitive: true,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
},
];
function savedColumn(suggestion: SensitivityReviewItem): CatalogColumn {
return {
id: suggestion.columnId,
tableId: suggestion.tableId,
name: suggestion.columnName,
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: true,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: suggestion.version + 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:01Z",
};
}
test("preserves failed human choices when another flag in the same save succeeds", async () => {
updateCatalogColumnSensitive
.mockResolvedValueOnce(savedColumn(suggestions[0]!))
.mockRejectedValueOnce(new Error("write failed"));
function Harness() {
const [current, setCurrent] = useState(suggestions);
return (
<SensitiveDataReviewDrawer
open
databaseId="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb"
scopeLabel="All columns"
suggestions={current}
canManage
onClose={() => undefined}
onSaved={(columns) => setCurrent((items) => items.map((item) => {
const saved = columns.find((column) => column.id === item.columnId);
return saved ? { ...item, currentSensitive: saved.sensitive, version: saved.version } : item;
}))}
/>
);
}
const user = userEvent.setup();
render(<Harness />);
await user.click(screen.getByRole("checkbox", { name: "Show all 2 assessed columns" }));
const failedChoice = screen.getByRole("checkbox", { name: "Protect patients.notes" });
await user.click(failedChoice);
await user.click(screen.getByRole("button", { name: "Save 2" }));
await waitFor(() => expect(updateCatalogColumnSensitive).toHaveBeenCalledTimes(2));
await waitFor(() => expect(screen.getByRole("button", { name: "Save 1" })).toBeEnabled());
expect(failedChoice).not.toBeChecked();
});
@@ -1,4 +1,4 @@
import { useEffect, useMemo, useState } from "react";
import { useEffect, useMemo, useRef, useState } from "react";
import { Save } from "lucide-react";
import { toast } from "sonner";
import { Button } from "../../components/ui/button";
@@ -6,7 +6,7 @@ import { apiErrorMessage } from "../../api/client";
import {
updateCatalogColumnSensitive,
type CatalogColumn,
type SensitiveDataSuggestion,
type SensitivityReviewItem,
} from "../../api/catalog-databases";
import { FleetLedgerDrawer } from "./FleetLedgerShell";
@@ -14,7 +14,7 @@ interface Props {
open: boolean;
databaseId: string | null;
scopeLabel: string;
suggestions: SensitiveDataSuggestion[];
suggestions: SensitivityReviewItem[];
canManage: boolean;
onClose: () => void;
onSaved: (columns: CatalogColumn[]) => void;
@@ -33,16 +33,26 @@ export function SensitiveDataReviewDrawer({
const [search, setSearch] = useState("");
const [showAll, setShowAll] = useState(false);
const [saving, setSaving] = useState(false);
const initializedReview = useRef<string | null>(null);
const reviewKey = useMemo(
() => suggestions.map((suggestion) => suggestion.columnId).join(":"),
[suggestions],
);
useEffect(() => {
if (!open) return;
if (!open) {
initializedReview.current = null;
return;
}
if (initializedReview.current === reviewKey) return;
initializedReview.current = reviewKey;
setDrafts(Object.fromEntries(suggestions.map((suggestion) => [
suggestion.columnId,
suggestion.sensitive,
])));
setSearch("");
setShowAll(false);
}, [open, suggestions]);
}, [open, reviewKey, suggestions]);
const changed = useMemo(() => suggestions.filter((suggestion) => (
drafts[suggestion.columnId] !== undefined
@@ -92,8 +102,8 @@ export function SensitiveDataReviewDrawer({
open
ariaLabel="Sensitive field review"
eyebrow="Sensitive data"
title="Review suggested flags"
description={`${scopeLabel}. The model proposed values, but only your save changes the catalog.`}
title="Review local assessments"
description={`${scopeLabel}. Local rules proposed values, but only your save changes the catalog.`}
onClose={close}
closeLabel="Close sensitive field review"
busy={saving}
@@ -101,7 +111,7 @@ export function SensitiveDataReviewDrawer({
footerClassName="thot-catalog-drawer__footer--split"
footer={(
<>
<p className="text-xs text-muted-foreground">Unsaved suggestions never change the catalog.</p>
<p className="text-xs text-muted-foreground">Unsaved assessments never change the catalog.</p>
<div className="flex gap-2">
<Button type="button" variant="outline" disabled={saving} onClick={close}>Cancel</Button>
<Button type="button" disabled={!canManage || saving || changed.length === 0} onClick={() => void save()}>
@@ -115,7 +125,7 @@ export function SensitiveDataReviewDrawer({
<div className="flex items-center gap-3">
<input
className="h-9 min-w-0 flex-1 rounded-md border border-input bg-background px-3 text-sm outline-none focus:border-primary/60 focus:ring-3 focus:ring-ring/15"
aria-label="Search sensitive field suggestions"
aria-label="Search sensitivity assessments"
placeholder="Search table or column"
value={search}
onChange={(event) => setSearch(event.target.value)}
@@ -131,7 +141,7 @@ export function SensitiveDataReviewDrawer({
checked={showAll}
onChange={(event) => setShowAll(event.target.checked)}
/>
Show all {suggestions.length} classified columns
Show all {suggestions.length} assessed columns
</label>
</div>
@@ -144,7 +154,7 @@ export function SensitiveDataReviewDrawer({
</p>
</div>
) : (
<ul className="divide-y divide-border" aria-label="Sensitive field suggestions">
<ul className="divide-y divide-border" aria-label="Sensitivity assessments">
{visible.map((suggestion) => {
const proposed = drafts[suggestion.columnId] ?? suggestion.sensitive;
const changedFromCurrent = proposed !== suggestion.currentSensitive;
@@ -168,6 +178,13 @@ export function SensitiveDataReviewDrawer({
<p className="mt-1 text-xs text-muted-foreground">
Current: {suggestion.currentSensitive ? "protected" : "allowed"}. Proposed: {proposed ? "protected" : "allowed"}.
</p>
<p className="mt-1 text-xs text-muted-foreground">
Assessment: {suggestion.assessment.replace("_", " ")}. Evidence: {suggestion.evidence.length > 0
? suggestion.evidence.map((item) => item.label
? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})`
: item.ruleId).join(", ")
: "no sensitive match"}. Observed values: {suggestion.observedValues}.
</p>
</div>
<span className={`rounded px-2 py-0.5 text-[11px] font-semibold ${changedFromCurrent ? "bg-amber-500/12 text-amber-800 dark:text-amber-300" : "bg-muted text-muted-foreground"}`}>
{changedFromCurrent ? "Change" : "No change"}
@@ -3,19 +3,22 @@ import userEvent from "@testing-library/user-event";
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
import { http, HttpResponse } from "msw";
import { useState } from "react";
import type { SensitiveDataSuggestionRun } from "../../api/catalog-databases";
import type { SensitivityAnalysisRun } from "../../api/catalog-databases";
import { server } from "../../test/msw";
import { SensitiveDataSuggestionHistoryDrawer } from "./SensitiveDataSuggestionHistoryDrawer";
import { SensitivityAnalysisHistoryDrawer } from "./SensitivityAnalysisHistoryDrawer";
const runningRun: SensitiveDataSuggestionRun = {
const runningRun: SensitivityAnalysisRun = {
id: "99999999-9999-4999-8999-999999999999",
databaseId: "11111111-1111-4111-8111-111111111111",
scope: "selected_columns",
modelId: "local-qwen",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "running",
total: 4,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
unknown: 2,
createdAt: "2026-08-28T09:00:00Z",
startedAt: "2026-08-28T09:00:00Z",
updatedAt: "2026-08-28T09:00:01Z",
@@ -23,7 +26,7 @@ const runningRun: SensitiveDataSuggestionRun = {
errorSummary: null,
};
function renderDrawer(initialRun: SensitiveDataSuggestionRun | null) {
function renderDrawer(initialRun: SensitivityAnalysisRun | null) {
const client = new QueryClient({
defaultOptions: { queries: { retry: false, gcTime: Infinity } },
});
@@ -31,10 +34,9 @@ function renderDrawer(initialRun: SensitiveDataSuggestionRun | null) {
function Harness() {
const [run, setRun] = useState(initialRun);
return (
<SensitiveDataSuggestionHistoryDrawer
<SensitivityAnalysisHistoryDrawer
open
run={run}
modelLabel="Local Qwen"
onClose={() => undefined}
onRunUpdate={setRun}
/>
@@ -56,13 +58,13 @@ test("shows an explicit empty state before any sensitive suggestion runs exist",
renderDrawer(null);
const drawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
expect(within(drawer).getByRole("heading", { name: "Suggestion run history" })).toBeVisible();
expect(await within(drawer).findByText("No sensitive suggestion runs yet.")).toBeVisible();
const drawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
expect(within(drawer).getByRole("heading", { name: "Analysis run history" })).toBeVisible();
expect(await within(drawer).findByText("No sensitivity analysis runs yet.")).toBeVisible();
});
test("shows sensitive suggestion results, safe events, and lets operators inspect an older run", async () => {
const completedRun: SensitiveDataSuggestionRun = {
const completedRun: SensitivityAnalysisRun = {
...runningRun,
id: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
scope: "all",
@@ -70,6 +72,7 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
total: 6,
suggestedSensitive: 2,
suggestedNonSensitive: 4,
unknown: 0,
createdAt: "2026-08-28T08:00:00Z",
startedAt: "2026-08-28T08:00:00Z",
updatedAt: "2026-08-28T08:00:01Z",
@@ -95,18 +98,18 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
const user = userEvent.setup();
renderDrawer(runningRun);
const drawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
const drawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
expect(within(drawer).getByRole("heading", { name: "Running" })).toBeVisible();
const progress = within(drawer).getByRole("table", { name: "Sensitive suggestion progress" });
expect(within(progress).getAllByRole("columnheader")).toHaveLength(3);
expect(within(drawer).getByRole("log", { name: "Sensitive suggestion events" })).toHaveClass(
const progress = within(drawer).getByRole("table", { name: "Sensitivity analysis progress" });
expect(within(progress).getAllByRole("columnheader")).toHaveLength(4);
expect(within(drawer).getByRole("log", { name: "Sensitivity analysis events" })).toHaveClass(
"thot-catalog-drawer__event-log",
);
expect(await within(drawer).findByText("Classifying columns")).toBeVisible();
expect(within(drawer).queryByRole("button", { name: "Stop" })).not.toBeInTheDocument();
expect(within(drawer).queryByRole("button", { name: "Unlock stale run" })).not.toBeInTheDocument();
const history = within(drawer).getByRole("region", { name: "Sensitive suggestion run history" });
const history = within(drawer).getByRole("region", { name: "Sensitivity analysis run history" });
await user.click(within(history).getByRole("button", { name: /Completed.*all/i }));
expect(await within(drawer).findByRole("heading", { name: "Completed" })).toBeVisible();
@@ -116,7 +119,7 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
});
test("does not expose unsafe provider details from a final error summary", async () => {
const failedRun: SensitiveDataSuggestionRun = {
const failedRun: SensitivityAnalysisRun = {
...runningRun,
status: "failed",
finishedAt: "2026-08-28T09:00:02Z",
@@ -3,35 +3,34 @@ import { LoaderCircle } from "lucide-react";
import { useQuery, useQueryClient } from "@tanstack/react-query";
import { apiErrorMessage } from "../../api/client";
import {
getSensitiveDataSuggestionRun,
listSensitiveDataSuggestionEvents,
listSensitiveDataSuggestionRuns,
type SensitiveDataSuggestionRun,
getSensitivityAnalysisRun,
listSensitivityAnalysisEvents,
listSensitivityAnalysisRuns,
type SensitivityAnalysisRun,
} from "../../api/catalog-databases";
import { FleetLedgerDrawer } from "./FleetLedgerShell";
interface Props {
open: boolean;
databaseId?: string | null;
run: SensitiveDataSuggestionRun | null;
modelLabel: string;
run: SensitivityAnalysisRun | null;
onClose: () => void;
onRunUpdate: (run: SensitiveDataSuggestionRun) => void;
onRunUpdate: (run: SensitivityAnalysisRun) => void;
}
const HISTORY_QUERY_KEY = ["sensitive-data-suggestion-runs", 50] as const;
const SAFE_FINAL_ERROR = "The run ended before all columns were classified. Review the event log for safe details.";
function statusLabel(run: SensitiveDataSuggestionRun): string {
function statusLabel(run: SensitivityAnalysisRun): string {
const label = run.status.replaceAll("_", " ");
return `${label[0].toUpperCase()}${label.slice(1)}`;
}
function isTerminal(run?: SensitiveDataSuggestionRun): boolean {
function isTerminal(run?: SensitivityAnalysisRun): boolean {
return Boolean(run && ["completed", "failed", "interrupted"].includes(run.status));
}
function outcomeClass(run: SensitiveDataSuggestionRun): string {
function outcomeClass(run: SensitivityAnalysisRun): string {
if (run.status === "completed") return "thot-sensitive-run--success";
if (["failed", "interrupted"].includes(run.status)) return "thot-sensitive-run--failure";
return "";
@@ -52,11 +51,10 @@ function timestamp(value: string | null): string {
return value ? new Date(value).toLocaleString() : "Not available";
}
export function SensitiveDataSuggestionHistoryDrawer({
export function SensitivityAnalysisHistoryDrawer({
open,
databaseId = null,
run: initialRun,
modelLabel,
onClose,
onRunUpdate,
}: Props) {
@@ -64,7 +62,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
const runId = initialRun?.id ?? null;
const runQuery = useQuery({
queryKey: ["sensitive-data-suggestion-run", runId],
queryFn: () => getSensitiveDataSuggestionRun(runId!),
queryFn: () => getSensitivityAnalysisRun(runId!),
enabled: open && Boolean(runId),
initialData: initialRun ?? undefined,
retry: false,
@@ -73,14 +71,14 @@ export function SensitiveDataSuggestionHistoryDrawer({
const run = runQuery.data;
const eventQuery = useQuery({
queryKey: ["sensitive-data-suggestion-events", runId],
queryFn: () => listSensitiveDataSuggestionEvents(runId!),
queryFn: () => listSensitivityAnalysisEvents(runId!),
enabled: open && Boolean(runId),
retry: false,
refetchInterval: run && isTerminal(run) ? false : 1_500,
});
const historyQuery = useQuery({
queryKey: HISTORY_QUERY_KEY,
queryFn: () => listSensitiveDataSuggestionRuns(50),
queryFn: () => listSensitivityAnalysisRuns(50),
enabled: open,
retry: false,
refetchInterval: (query) => query.state.data?.some((item) => !isTerminal(item))
@@ -100,7 +98,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
useEffect(() => {
if (!run) return;
onRunUpdate(run);
queryClient.setQueryData<SensitiveDataSuggestionRun[]>(HISTORY_QUERY_KEY, (current) => {
queryClient.setQueryData<SensitivityAnalysisRun[]>(HISTORY_QUERY_KEY, (current) => {
if (!current) return [run];
return current.some((item) => item.id === run.id)
? current.map((item) => item.id === run.id ? run : item)
@@ -113,11 +111,11 @@ export function SensitiveDataSuggestionHistoryDrawer({
return (
<FleetLedgerDrawer
open
ariaLabel="Sensitive suggestion history"
ariaLabel="Sensitivity analysis history"
eyebrow="Sensitive data"
title="Suggestion run history"
title="Analysis run history"
onClose={onClose}
closeLabel="Close sensitive suggestion history"
closeLabel="Close sensitivity analysis history"
bodyClassName="thot-catalog-drawer__body--empty"
>
{historyQuery.isError ? (
@@ -125,12 +123,12 @@ export function SensitiveDataSuggestionHistoryDrawer({
{apiErrorMessage(historyQuery.error)}
</div>
) : historyQuery.isLoading || (historyQuery.data?.length ?? 0) > 0 ? (
<p className="text-sm text-muted-foreground">Loading suggestion history…</p>
<p className="text-sm text-muted-foreground">Loading analysis history…</p>
) : (
<div className="w-full rounded-md border border-border bg-muted/25 px-4 py-5 text-sm">
<p className="font-semibold">No sensitive suggestion runs yet.</p>
<p className="font-semibold">No sensitivity analysis runs yet.</p>
<p className="mt-1 text-muted-foreground">
Request sensitive-field suggestions to create the first history entry.
Analyze sensitive fields to create the first history entry.
</p>
</div>
)}
@@ -140,20 +138,21 @@ export function SensitiveDataSuggestionHistoryDrawer({
const events = eventQuery.data ?? [];
const finalError = safeFinalError(run.errorSummary);
const modelDisplay = modelLabel && modelLabel !== run.modelId
? `${modelLabel} (${run.modelId})`
: run.modelId;
const engineDisplay = run.engine === "local"
? `Local rules (${run.policyVersion ?? "unknown policy"})`
: `Legacy LLM (${run.modelId ?? "unknown model"})`;
const progressCounters = [
["total", "Total columns"],
["suggestedSensitive", "Sensitive"],
["suggestedNonSensitive", "Not sensitive"],
["unknown", "Unknown"],
] as const;
return (
<FleetLedgerDrawer
open
className={outcomeClass(run)}
ariaLabel="Sensitive suggestion history"
eyebrow="Sensitive data suggestions"
ariaLabel="Sensitivity analysis history"
eyebrow="Sensitivity analysis"
title={(
<span className="inline-flex items-center gap-2">
{!isTerminal(run) ? <LoaderCircle className="size-5 animate-spin" aria-hidden="true" /> : null}
@@ -162,7 +161,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
)}
description={run.scope.replaceAll("_", " ")}
onClose={onClose}
closeLabel="Close sensitive suggestion history"
closeLabel="Close sensitivity analysis history"
bodyClassName="thot-catalog-drawer__history-layout thot-catalog-drawer__run-layout"
>
<div className="thot-catalog-drawer__run-main">
@@ -173,16 +172,18 @@ export function SensitiveDataSuggestionHistoryDrawer({
) : null}
<div className={`grid gap-4 sm:grid-cols-2${runQuery.isError ? " mt-4" : ""}`}>
<section aria-label="AI usage">
<h3 className="thot-label mb-2">AI usage</h3>
<section aria-label="Analysis engine">
<h3 className="thot-label mb-2">Analysis engine</h3>
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1 text-xs">
<dt>Model</dt><dd className="truncate text-right" title={modelDisplay}>{modelDisplay}</dd>
<dt>Input tokens</dt><dd className="text-right tabular-nums">{(run.inputTokens ?? 0).toLocaleString("en-US")}</dd>
<dt>Cache tokens</dt><dd className="text-right tabular-nums">{(run.cacheReadTokens ?? 0).toLocaleString("en-US")}</dd>
<dt>Output tokens</dt><dd className="text-right tabular-nums">{(run.outputTokens ?? 0).toLocaleString("en-US")}</dd>
<dt>Engine</dt><dd className="truncate text-right" title={engineDisplay}>{engineDisplay}</dd>
{run.engine === "llm" ? <>
<dt>Input tokens</dt><dd className="text-right tabular-nums">{(run.inputTokens ?? 0).toLocaleString("en-US")}</dd>
<dt>Cache tokens</dt><dd className="text-right tabular-nums">{(run.cacheReadTokens ?? 0).toLocaleString("en-US")}</dd>
<dt>Output tokens</dt><dd className="text-right tabular-nums">{(run.outputTokens ?? 0).toLocaleString("en-US")}</dd>
</> : null}
</dl>
</section>
<section aria-label="Sensitive suggestion timestamps">
<section aria-label="Sensitivity analysis timestamps">
<h3 className="thot-label mb-2">Timestamps</h3>
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1 text-xs">
<dt className="text-muted-foreground">Created</dt><dd className="text-right tabular-nums">{timestamp(run.createdAt)}</dd>
@@ -193,10 +194,10 @@ export function SensitiveDataSuggestionHistoryDrawer({
</section>
</div>
<section className="mt-4" aria-label="Sensitive suggestion counters">
<section className="mt-4" aria-label="Sensitivity analysis counters">
<h3 className="thot-label mb-2">Results</h3>
<div className="overflow-x-auto rounded-md border border-border bg-muted/20">
<table className="w-full min-w-[20rem] table-fixed" aria-label="Sensitive suggestion progress">
<table className="w-full min-w-[20rem] table-fixed" aria-label="Sensitivity analysis progress">
<thead>
<tr className="border-b border-border bg-muted/45">
{progressCounters.map(([name, label]) => (
@@ -225,14 +226,14 @@ export function SensitiveDataSuggestionHistoryDrawer({
</div>
) : null}
<section className="thot-catalog-drawer__events mt-3" aria-label="Sensitive suggestion log">
<section className="thot-catalog-drawer__events mt-3" aria-label="Sensitivity analysis log">
<div className="mb-2 flex items-center justify-between">
<h3 className="thot-label">Events</h3>
<span className="text-xs tabular-nums text-muted-foreground">{events.length}</span>
</div>
<div
role="log"
aria-label="Sensitive suggestion events"
aria-label="Sensitivity analysis events"
className="thot-catalog-drawer__event-log overflow-y-auto rounded-md bg-zinc-950 p-3 font-mono text-xs leading-5 text-zinc-200"
>
{events.length === 0 ? <p className="text-zinc-500">Waiting for events…</p> : events.map((event) => (
@@ -245,7 +246,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
</section>
</div>
<section className="thot-catalog-drawer__history" aria-label="Sensitive suggestion run history">
<section className="thot-catalog-drawer__history" aria-label="Sensitivity analysis run history">
<div className="mb-2 flex items-center justify-between gap-3">
<h3 className="thot-label">Recent runs</h3>
<span className="text-xs tabular-nums text-muted-foreground">
@@ -255,9 +256,9 @@ export function SensitiveDataSuggestionHistoryDrawer({
{historyQuery.isError ? (
<p role="alert" className="text-sm text-destructive">{apiErrorMessage(historyQuery.error)}</p>
) : historyQuery.isLoading ? (
<p className="text-sm text-muted-foreground">Loading suggestion history…</p>
<p className="text-sm text-muted-foreground">Loading analysis history…</p>
) : visibleHistory.length === 0 ? (
<p className="text-sm text-muted-foreground">No sensitive suggestion runs yet.</p>
<p className="text-sm text-muted-foreground">No sensitivity analysis runs yet.</p>
) : (
<div className="thot-catalog-drawer__history-list divide-y divide-border overflow-y-auto rounded-md border border-border">
{visibleHistory.map((item) => (
@@ -272,7 +273,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
>
<span className="font-medium">{statusLabel(item)}</span>
<span className="shrink-0 text-right text-xs text-muted-foreground">
{item.suggestedSensitive + item.suggestedNonSensitive}/{item.total}<br />
{item.suggestedSensitive + item.suggestedNonSensitive + item.unknown}/{item.total}<br />
{new Date(item.createdAt).toLocaleString()}
</span>
</button>
+5
View File
@@ -50,6 +50,7 @@ nav:
- Generic OIDC: install/authentication-oidc.md
- Authentik: install/authentik.md
- Workspace operations: operations/workspaces.md
- Local sensitivity analysis: operations/sensitivity-analysis.md
- Pi model configuration: general/pi-configuration.md
- Docker installation contexts: installazione-docker-4-contesti.md
- Use ThothII:
@@ -86,7 +87,11 @@ nav:
- 0010 Bounded source samples: adr/0010-allow-bounded-real-source-samples-for-description-generation.md
- 0011 Sensitive data flag: adr/0011-gate-source-samples-with-a-sensitive-data-flag.md
- 0012 Effective relationship authority: adr/0012-use-the-catalog-as-the-logical-relationship-authority.md
- 0013 Installation model catalog: adr/0013-use-one-installation-model-catalog-with-runtime-projections.md
- 0014 Local sensitive-column assessment: adr/0014-assess-sensitive-columns-locally-from-source-content.md
- AI catalog description acceptance: testing/2026-08-29-ai-catalog-description-generation-acceptance.md
- Sensitivity NER license inventory: reports/2026-09-02-sensitivity-ner-license-inventory.md
- PSD sensitivity shadow evaluation: reports/2026-09-02-psd-sensitivity-shadow.md
- Design records:
- Metadata catalog design: plans/2026-08-26-metadata-catalog-from-thothai.md
- Description generation design: plans/2026-08-28-ai-catalog-description-generation.md
+111
View File
@@ -0,0 +1,111 @@
#!/usr/bin/env bash
set -euo pipefail
MODEL_REPOSITORY="fastino/gliner2-privacy-filter-PII-multi"
MODEL_REVISION="c153999da5f4c509df4322b0c6a1baf3d2c284d7"
TARGET="${1:-}"
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
if [[ -z "$TARGET" || "$TARGET" != /* ]]; then
echo "Usage: $0 /absolute/model/directory" >&2
exit 2
fi
if [[ -d "$TARGET" ]] && find "$TARGET" -mindepth 1 -print -quit | grep -q .; then
echo "Target directory must be empty: $TARGET" >&2
exit 2
fi
mkdir -p "$TARGET"
cd "$ROOT"
docker build \
--build-arg INSTALL_SENSITIVITY_NER=true \
--file docker/core.Dockerfile \
--tag thothii-core:sensitivity-ner \
.
docker run --rm \
--user "$(id -u):$(id -g)" \
--env HOME=/tmp \
--env TMPDIR=/tmp \
--env XDG_CACHE_HOME=/tmp/.cache \
--env HF_HOME=/tmp/huggingface \
--env HF_HUB_DISABLE_TELEMETRY=1 \
--env HF_HUB_DISABLE_XET=1 \
--volume "$TARGET:/model" \
--entrypoint /opt/sensitivity-ner/bin/hf \
thothii-core:sensitivity-ner \
download "$MODEL_REPOSITORY" \
--revision "$MODEL_REVISION" \
--local-dir /model
printf '%s\n' "$MODEL_REVISION" > "$TARGET/THOTHII_MODEL_REVISION"
docker run --rm \
--user "$(id -u):$(id -g)" \
--volume "$TARGET:/model" \
--entrypoint /bin/sh \
thothii-core:sensitivity-ner \
-c 'cd /model && sha256sum -c /app/backend/python/sensitivity-ner-model-sha256.txt && cp /app/backend/python/sensitivity-ner-model-sha256.txt MODEL_SHA256SUMS'
docker run --rm \
--network none \
--entrypoint /opt/sensitivity-ner/bin/pip \
thothii-core:sensitivity-ner \
check
GPU_AUDIT="$(docker run --rm \
--network none \
--entrypoint /opt/sensitivity-ner/bin/python \
thothii-core:sensitivity-ner \
-c 'import torch; print(f"{torch.cuda.is_available()}|{torch.version.cuda}|{torch.cuda.device_count()}")')"
if [[ "$GPU_AUDIT" != "False|None|0" ]]; then
echo "The optional runtime is not CPU-only: $GPU_AUDIT" >&2
exit 1
fi
NETWORK_GUARD_AUDIT="$(docker run --rm \
--entrypoint /opt/sensitivity-ner/bin/python \
thothii-core:sensitivity-ner \
-c '
import importlib.util
import socket
spec = importlib.util.spec_from_file_location("worker", "/app/backend/python/sensitivity_ner_worker.py")
worker = importlib.util.module_from_spec(spec)
assert spec.loader is not None
spec.loader.exec_module(worker)
raw_socket = socket.socket
worker._disable_network()
try:
raw_socket(socket.AF_INET, socket.SOCK_STREAM)
except PermissionError as error:
print(error.errno)
else:
raise SystemExit("network syscall filter is inactive")
')"
if [[ "$NETWORK_GUARD_AUDIT" != "1" ]]; then
echo "The optional runtime did not activate its network syscall filter" >&2
exit 1
fi
SMOKE_OUTPUT="$(mktemp)"
SMOKE_ERROR="$(mktemp)"
trap 'rm -f -- "$SMOKE_OUTPUT" "$SMOKE_ERROR"' EXIT
printf '%s\n' '{"id":"smoke","candidates":[{"columnId":"33333333-3333-4333-8333-333333333333","text":"La paziente si chiama Maria Rossi."}]}' \
| docker run --rm \
--network none \
--read-only \
--tmpfs /tmp:rw,noexec,nosuid,size=512m \
--volume "$TARGET:/model:ro" \
--entrypoint /opt/sensitivity-ner/bin/python \
-i thothii-core:sensitivity-ner \
/app/backend/python/sensitivity_ner_worker.py --model /model --threads 2 \
>"$SMOKE_OUTPUT" 2>"$SMOKE_ERROR"
if ! grep -Fxq '{"ready":true}' "$SMOKE_OUTPUT" \
|| ! grep -Eq '"ok":true.*"columnId":"33333333-3333-4333-8333-333333333333".*"label":"full_name"' "$SMOKE_OUTPUT"; then
echo "The offline Italian CPU smoke test failed" >&2
sed -n '1,20p' "$SMOKE_ERROR" >&2
exit 1
fi
echo "Pinned model downloaded to $TARGET"
+3
View File
@@ -20,6 +20,9 @@ compose_files=(-f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml")
if [[ "${THOTH_ENABLE_EMBEDDING_GPU:-0}" == "1" ]]; then
compose_files+=(-f "$ROOT/deploy/compose.embedding-gpu.yaml")
fi
if [[ "${THOTH_ENABLE_SENSITIVITY_NER:-0}" == "1" ]]; then
compose_files+=(-f "$ROOT/deploy/compose.sensitivity-ner.yaml")
fi
compose=(docker compose --env-file "$LOCAL_ENV_FILE" "${compose_files[@]}")