feat: classify sensitive columns locally
This commit is contained in:
+20
-9
@@ -98,6 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr
|
||||
description history, and sensitive-field review/history use the production APIs in right-side
|
||||
drawers rather than prototype fixtures; closing a history drawer does not stop its background run.
|
||||
|
||||
Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a
|
||||
catalog model. The backend reads selected source tables through read-only, database-specific
|
||||
adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript
|
||||
`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to
|
||||
five seconds per table before sampling and the whole request to sixty seconds. Draft assessments
|
||||
remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is
|
||||
CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
|
||||
`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by
|
||||
default because its extra findings did not offset the coverage lost to inference within the global
|
||||
deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`.
|
||||
|
||||
Physical membership, source
|
||||
comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are
|
||||
projections of the external schema. They cannot be created, renamed, or structurally edited by
|
||||
@@ -155,15 +166,15 @@ Semantic aliases, value descriptions, synonyms, and concepts remain deferred to
|
||||
slices.
|
||||
|
||||
AI Description Generation uses the catalog's human-owned Sensitive Data Flag. The flag defaults to
|
||||
`false`, including for newly synchronized columns. An administrator may request an AI proposal based
|
||||
only on structural metadata for one selected database, selected tables, or selected columns. The
|
||||
backend divides large scopes into deterministic model requests of at most ten columns, also bounded
|
||||
by helper message size, and combines their results, but the proposal remains an unsaved draft until
|
||||
the human reviews and saves it.
|
||||
Each started suggestion attempt records a separate Sensitive Data Suggestion Run with aggregate
|
||||
counters and safe ordered events. This operational history never stores per-column proposals,
|
||||
prompts, raw model output, or provider diagnostics; reloading still discards an unsaved review
|
||||
draft.
|
||||
`false`, including for newly synchronized columns. An administrator may request a local sensitivity
|
||||
analysis for one selected database, selected tables, or selected columns. One deterministic
|
||||
TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER;
|
||||
no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments
|
||||
remain an unsaved draft until the human reviews and saves any chosen flag changes, including a
|
||||
downgrade to non-sensitive.
|
||||
Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe
|
||||
ordered events. This operational history never stores per-column assessments, source values,
|
||||
matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft.
|
||||
For unprotected columns, up to five source rows and five representative non-null values may be sent
|
||||
transiently to the configured model provider. Protected columns are omitted from source reads and
|
||||
replaced in the prompt by deterministic plausible values derived only from their metadata. Existing
|
||||
|
||||
Generated
+25
@@ -12,14 +12,17 @@
|
||||
"@types/pg": "^8.20.3",
|
||||
"fastify": "^5.0.0",
|
||||
"kysely": "^0.29.5",
|
||||
"libphonenumber-js": "1.13.12",
|
||||
"openid-client": "6.8.5",
|
||||
"pg": "^8.22.0",
|
||||
"validator": "13.15.35",
|
||||
"yaml": "^2.9.0",
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@testcontainers/postgresql": "^12.1.0",
|
||||
"@types/node": "24.13.3",
|
||||
"@types/validator": "13.15.10",
|
||||
"tsx": "^4.19.0",
|
||||
"typescript": "^5.6.0",
|
||||
"vitest": "^2.1.0"
|
||||
@@ -1329,6 +1332,13 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/validator": {
|
||||
"version": "13.15.10",
|
||||
"resolved": "https://registry.npmjs.org/@types/validator/-/validator-13.15.10.tgz",
|
||||
"integrity": "sha512-T8L6i7wCuyoK8A/ZeLYt1+q0ty3Zb9+qbSSvrIVitzT3YjZqkTZ40IbRsPanlB4h1QB3JVL1SYCdR6ngtFYcuA==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@vitest/expect": {
|
||||
"version": "2.1.9",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-2.1.9.tgz",
|
||||
@@ -2824,6 +2834,12 @@
|
||||
"safe-buffer": "~5.1.0"
|
||||
}
|
||||
},
|
||||
"node_modules/libphonenumber-js": {
|
||||
"version": "1.13.12",
|
||||
"resolved": "https://registry.npmjs.org/libphonenumber-js/-/libphonenumber-js-1.13.12.tgz",
|
||||
"integrity": "sha512-uLVeV1c9OTk6qkdqnj+mpMD+ZdnZ0szVyWu58HwMmpwkHA1gCEkyjd3veZQXDnuw9KEwSRjcc9B1pS9XKIN1fA==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/light-my-request": {
|
||||
"version": "6.6.0",
|
||||
"resolved": "https://registry.npmjs.org/light-my-request/-/light-my-request-6.6.0.tgz",
|
||||
@@ -4121,6 +4137,15 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/validator": {
|
||||
"version": "13.15.35",
|
||||
"resolved": "https://registry.npmjs.org/validator/-/validator-13.15.35.tgz",
|
||||
"integrity": "sha512-TQ5pAGhd5whStmqWvYF4OjQROlmv9SMFVt37qoCBdqRffuuklWYQlCNnEs2ZaIBD1kZRNnikiZOS1eqgkar0iw==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">= 0.10"
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "5.4.21",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-5.4.21.tgz",
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
"prebuild": "node scripts/clean-dist.mjs",
|
||||
"build": "tsc -p tsconfig.json",
|
||||
"catalog:migrate": "node dist/catalog/migrate.js",
|
||||
"sensitivity:shadow": "node dist/catalog/sensitivity-shadow.js",
|
||||
"test": "vitest run",
|
||||
"start": "node dist/server.js",
|
||||
"test:schema-v4-verifier": "python3 -I -B scripts/test_revision_state_policy.py && node --test scripts/verify-workspace-descriptor-files.test.mjs scripts/revision-state-policy.test.mjs",
|
||||
@@ -19,14 +20,17 @@
|
||||
"@types/pg": "^8.20.3",
|
||||
"fastify": "^5.0.0",
|
||||
"kysely": "^0.29.5",
|
||||
"libphonenumber-js": "1.13.12",
|
||||
"openid-client": "6.8.5",
|
||||
"pg": "^8.22.0",
|
||||
"validator": "13.15.35",
|
||||
"yaml": "^2.9.0",
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@testcontainers/postgresql": "^12.1.0",
|
||||
"@types/node": "24.13.3",
|
||||
"@types/validator": "13.15.10",
|
||||
"tsx": "^4.19.0",
|
||||
"typescript": "^5.6.0",
|
||||
"vitest": "^2.1.0"
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
34448b82c17d60fec9b65b1f093c115ddbaadc04beb1b0140b6bfed2e012a930 ./.gitattributes
|
||||
4d9344c58a2a2ea4bb4ff4f7c611a853cf413205fc10d0cace564eba06f73828 ./README.md
|
||||
180f0a10d1d5ed5ce3318db0bcb0b1b7780d79a52f0a8fc3acbd27f74536d0e4 ./THOTHII_MODEL_REVISION
|
||||
164f17362bcf9d114067d3465e7374bfdd79ce6b605acb745de5a49dabb9595c ./config.json
|
||||
f27dd63cc43a248d2566f0b6ad7a115db353676ce0561dcbca45bac766464c1a ./encoder_config/config.json
|
||||
0280f6f39f6012da50b6640bad438d9b7e763a1b0102094115d1b710c4dd79b6 ./model.safetensors
|
||||
f6df10ec83bea993035b2dd7c39345a3d4fcf23421c2adb6cb4ffc1e6d1bc4b5 ./tokenizer.json
|
||||
233beed1f1095cccfc7907cde31a8d90a0c6aa4fdfaf6493f8e55fd162e81ae6 ./tokenizer_config.json
|
||||
@@ -0,0 +1,34 @@
|
||||
# Optional offline CPU pack. Fully version-locked in its own venv; not part of the base image.
|
||||
--extra-index-url https://download.pytorch.org/whl/cpu
|
||||
accelerate==1.14.0
|
||||
annotated-types==0.8.0
|
||||
certifi==2026.7.22
|
||||
charset-normalizer==3.5.1
|
||||
filelock==3.32.5
|
||||
fsspec==2026.7.0
|
||||
gliner2[local]==2.0.0
|
||||
hf-xet==1.6.0
|
||||
huggingface-hub==0.36.2
|
||||
idna==3.19
|
||||
Jinja2==3.1.6
|
||||
MarkupSafe==3.0.3
|
||||
mpmath==1.3.0
|
||||
networkx==3.6.1
|
||||
numpy==2.5.2
|
||||
packaging==26.3
|
||||
peft==0.20.0
|
||||
psutil==7.2.2
|
||||
pydantic==2.13.5
|
||||
pydantic-core==2.46.5
|
||||
PyYAML==6.0.3
|
||||
regex==2026.9.3
|
||||
requests==2.34.2
|
||||
safetensors==0.8.0
|
||||
sympy==1.14.0
|
||||
tokenizers==0.22.2
|
||||
torch==2.14.0+cpu
|
||||
tqdm==4.70.0
|
||||
transformers==4.57.6
|
||||
typing-extensions==4.16.0
|
||||
typing-inspection==0.4.4
|
||||
urllib3==2.7.0
|
||||
@@ -0,0 +1,301 @@
|
||||
"""Offline, CPU-only JSONL worker for optional sensitivity NER evidence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import contextlib
|
||||
import ctypes
|
||||
import errno
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import socket
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
PII_LABELS = [
|
||||
"person",
|
||||
"full_name",
|
||||
"first_name",
|
||||
"middle_name",
|
||||
"last_name",
|
||||
"date_of_birth",
|
||||
"email",
|
||||
"phone_number",
|
||||
"address",
|
||||
"street_address",
|
||||
"city",
|
||||
"state_or_region",
|
||||
"postal_code",
|
||||
"country",
|
||||
"government_id",
|
||||
"national_id_number",
|
||||
"passport_number",
|
||||
"drivers_license_number",
|
||||
"license_number",
|
||||
"tax_id",
|
||||
"tax_number",
|
||||
"bank_account",
|
||||
"account_number",
|
||||
"routing_number",
|
||||
"iban",
|
||||
"payment_card",
|
||||
"card_number",
|
||||
"card_expiry",
|
||||
"card_cvv",
|
||||
"username",
|
||||
"ip_address",
|
||||
"account_id",
|
||||
"sensitive_account_id",
|
||||
"password",
|
||||
"secret",
|
||||
"api_key",
|
||||
"access_token",
|
||||
"recovery_code",
|
||||
"sensitive_date",
|
||||
"document_date",
|
||||
"expiration_date",
|
||||
"transaction_date",
|
||||
]
|
||||
|
||||
_MODEL_COMPAT_DIRECTORY: tempfile.TemporaryDirectory[str] | None = None
|
||||
_EXPECTED_MODEL_REVISION = "c153999da5f4c509df4322b0c6a1baf3d2c284d7"
|
||||
|
||||
|
||||
def _arguments() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(add_help=False)
|
||||
parser.add_argument("--model", required=True)
|
||||
parser.add_argument("--threads", type=int, default=2)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _disable_network() -> None:
|
||||
libc = ctypes.CDLL(None, use_errno=True)
|
||||
libc.prctl.argtypes = [
|
||||
ctypes.c_int,
|
||||
ctypes.c_ulong,
|
||||
ctypes.c_ulong,
|
||||
ctypes.c_ulong,
|
||||
ctypes.c_ulong,
|
||||
]
|
||||
libc.prctl.restype = ctypes.c_int
|
||||
if libc.prctl(38, 1, 0, 0, 0) != 0: # PR_SET_NO_NEW_PRIVS
|
||||
raise RuntimeError("cannot enable no-new-privileges for network isolation")
|
||||
|
||||
try:
|
||||
seccomp = ctypes.CDLL("libseccomp.so.2", use_errno=True)
|
||||
except OSError as error:
|
||||
raise RuntimeError("libseccomp is required for network isolation") from error
|
||||
seccomp.seccomp_init.argtypes = [ctypes.c_uint32]
|
||||
seccomp.seccomp_init.restype = ctypes.c_void_p
|
||||
seccomp.seccomp_syscall_resolve_name.argtypes = [ctypes.c_char_p]
|
||||
seccomp.seccomp_syscall_resolve_name.restype = ctypes.c_int
|
||||
seccomp.seccomp_rule_add.argtypes = [
|
||||
ctypes.c_void_p,
|
||||
ctypes.c_uint32,
|
||||
ctypes.c_int,
|
||||
ctypes.c_uint,
|
||||
]
|
||||
seccomp.seccomp_rule_add.restype = ctypes.c_int
|
||||
seccomp.seccomp_load.argtypes = [ctypes.c_void_p]
|
||||
seccomp.seccomp_load.restype = ctypes.c_int
|
||||
seccomp.seccomp_release.argtypes = [ctypes.c_void_p]
|
||||
seccomp.seccomp_release.restype = None
|
||||
|
||||
allow = 0x7FFF0000 # SCMP_ACT_ALLOW
|
||||
deny = 0x00050000 | errno.EPERM # SCMP_ACT_ERRNO(EPERM)
|
||||
filter_context = seccomp.seccomp_init(allow)
|
||||
if not filter_context:
|
||||
raise RuntimeError("cannot initialize network syscall filter")
|
||||
try:
|
||||
for syscall in (
|
||||
"socket",
|
||||
"connect",
|
||||
"sendto",
|
||||
"sendmsg",
|
||||
"sendmmsg",
|
||||
"bind",
|
||||
"listen",
|
||||
"accept",
|
||||
"accept4",
|
||||
):
|
||||
syscall_number = seccomp.seccomp_syscall_resolve_name(syscall.encode("ascii"))
|
||||
if syscall_number < 0:
|
||||
raise RuntimeError(f"cannot resolve network syscall: {syscall}")
|
||||
if seccomp.seccomp_rule_add(filter_context, deny, syscall_number, 0) != 0:
|
||||
raise RuntimeError(f"cannot block network syscall: {syscall}")
|
||||
if seccomp.seccomp_load(filter_context) != 0:
|
||||
raise RuntimeError("cannot activate network syscall filter")
|
||||
finally:
|
||||
seccomp.seccomp_release(filter_context)
|
||||
|
||||
def blocked(*_args: Any, **_kwargs: Any) -> Any:
|
||||
raise PermissionError(errno.EPERM, "network disabled")
|
||||
|
||||
socket.socket = blocked # type: ignore[assignment]
|
||||
socket.create_connection = blocked # type: ignore[assignment]
|
||||
|
||||
|
||||
def _verify_model(path: Path) -> None:
|
||||
revision_path = path / "THOTHII_MODEL_REVISION"
|
||||
try:
|
||||
revision = revision_path.read_text(encoding="utf-8").strip()
|
||||
except OSError as error:
|
||||
raise RuntimeError("model revision marker is unavailable") from error
|
||||
if revision != _EXPECTED_MODEL_REVISION:
|
||||
raise RuntimeError("model revision is not approved")
|
||||
|
||||
manifest_path = Path(__file__).with_name("sensitivity-ner-model-sha256.txt")
|
||||
try:
|
||||
manifest = manifest_path.read_text(encoding="utf-8").splitlines()
|
||||
except OSError as error:
|
||||
raise RuntimeError("model checksum manifest is unavailable") from error
|
||||
for line in manifest:
|
||||
checksum, separator, relative_name = line.partition(" ")
|
||||
if not separator or len(checksum) != 64 or not relative_name.startswith("./"):
|
||||
raise RuntimeError("model checksum manifest is invalid")
|
||||
relative_path = Path(relative_name[2:])
|
||||
if relative_path.is_absolute() or ".." in relative_path.parts:
|
||||
raise RuntimeError("model checksum path is invalid")
|
||||
model_file = path / relative_path
|
||||
if not model_file.is_file() or model_file.is_symlink():
|
||||
raise RuntimeError("approved model file is unavailable")
|
||||
digest = hashlib.sha256()
|
||||
with model_file.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
if digest.hexdigest() != checksum:
|
||||
raise RuntimeError("approved model checksum does not match")
|
||||
|
||||
|
||||
def _transformers4_model_path(path: Path) -> Path:
|
||||
"""Adapt tokenizer metadata emitted by Transformers 5 without changing pinned weights.
|
||||
|
||||
GLiNER2 2.0.0 officially requires Transformers <5, while current Fastino checkpoints were
|
||||
saved by Transformers 5.8.0. Transformers 4 calls the same list
|
||||
``additional_special_tokens``; Transformers 5 renamed it to ``extra_special_tokens`` and
|
||||
changed its type. Keep the downloaded model immutable and create a temporary symlink view
|
||||
containing only the compatibility metadata needed by the supported GLiNER2 dependency set.
|
||||
"""
|
||||
|
||||
tokenizer_path = path / "tokenizer_config.json"
|
||||
try:
|
||||
tokenizer = json.loads(tokenizer_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as error:
|
||||
raise RuntimeError("invalid tokenizer configuration") from error
|
||||
extra_tokens = tokenizer.get("extra_special_tokens")
|
||||
if extra_tokens is None:
|
||||
return path
|
||||
if not isinstance(extra_tokens, list) or not all(isinstance(token, str) for token in extra_tokens):
|
||||
raise RuntimeError("unsupported extra_special_tokens configuration")
|
||||
if "additional_special_tokens" in tokenizer:
|
||||
raise RuntimeError("ambiguous special-token configuration")
|
||||
|
||||
global _MODEL_COMPAT_DIRECTORY
|
||||
_MODEL_COMPAT_DIRECTORY = tempfile.TemporaryDirectory(prefix="thothii-ner-model-")
|
||||
compatible_path = Path(_MODEL_COMPAT_DIRECTORY.name)
|
||||
for child in path.iterdir():
|
||||
if child.name == tokenizer_path.name:
|
||||
continue
|
||||
(compatible_path / child.name).symlink_to(child, target_is_directory=child.is_dir())
|
||||
tokenizer["additional_special_tokens"] = tokenizer.pop("extra_special_tokens")
|
||||
(compatible_path / tokenizer_path.name).write_text(
|
||||
json.dumps(tokenizer, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return compatible_path
|
||||
|
||||
|
||||
def _load_model(model_path: str, threads: int) -> Any:
|
||||
path = Path(model_path).resolve(strict=True)
|
||||
if not path.is_dir():
|
||||
raise RuntimeError("model path must be a local directory")
|
||||
_verify_model(path)
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = ""
|
||||
os.environ["HIP_VISIBLE_DEVICES"] = ""
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
||||
import torch
|
||||
from gliner2 import AutoExtractor
|
||||
|
||||
torch.set_num_threads(max(1, min(threads, 8)))
|
||||
torch.set_num_interop_threads(1)
|
||||
compatible_path = _transformers4_model_path(path)
|
||||
with contextlib.redirect_stdout(sys.stderr):
|
||||
model = AutoExtractor.from_pretrained(str(compatible_path), map_location="cpu")
|
||||
_disable_network()
|
||||
return model
|
||||
|
||||
|
||||
def _request(value: Any) -> tuple[str, list[dict[str, str]]]:
|
||||
if not isinstance(value, dict) or not isinstance(value.get("id"), str):
|
||||
raise ValueError("invalid request")
|
||||
candidates = value.get("candidates")
|
||||
if not isinstance(candidates, list) or not 1 <= len(candidates) <= 128:
|
||||
raise ValueError("invalid candidates")
|
||||
parsed: list[dict[str, str]] = []
|
||||
for candidate in candidates:
|
||||
if not isinstance(candidate, dict):
|
||||
raise ValueError("invalid candidate")
|
||||
column_id = candidate.get("columnId")
|
||||
text = candidate.get("text")
|
||||
if not isinstance(column_id, str) or not isinstance(text, str) or not 1 <= len(text) <= 500:
|
||||
raise ValueError("invalid candidate")
|
||||
parsed.append({"columnId": column_id, "text": text})
|
||||
return value["id"], parsed
|
||||
|
||||
|
||||
def _detect(model: Any, candidates: list[dict[str, str]]) -> list[dict[str, Any]]:
|
||||
evidence: list[dict[str, Any]] = []
|
||||
for candidate in candidates:
|
||||
result = model.extract_entities(
|
||||
candidate["text"],
|
||||
PII_LABELS,
|
||||
threshold=0.5,
|
||||
include_confidence=True,
|
||||
)
|
||||
entities = result.get("entities", {}) if isinstance(result, dict) else {}
|
||||
best: tuple[str, float] | None = None
|
||||
if isinstance(entities, dict):
|
||||
for label, matches in entities.items():
|
||||
if label not in PII_LABELS or not isinstance(matches, list):
|
||||
continue
|
||||
for match in matches:
|
||||
if not isinstance(match, dict):
|
||||
continue
|
||||
confidence = match.get("confidence")
|
||||
if not isinstance(confidence, (int, float)) or not 0 <= confidence <= 1:
|
||||
continue
|
||||
if best is None or confidence > best[1]:
|
||||
best = (label, float(confidence))
|
||||
if best is not None:
|
||||
evidence.append(
|
||||
{
|
||||
"columnId": candidate["columnId"],
|
||||
"label": best[0],
|
||||
"confidence": best[1],
|
||||
}
|
||||
)
|
||||
return evidence
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = _arguments()
|
||||
model = _load_model(args.model, args.threads)
|
||||
print(json.dumps({"ready": True}, separators=(",", ":")), flush=True)
|
||||
for line in sys.stdin:
|
||||
request_id = "invalid"
|
||||
try:
|
||||
request_id, candidates = _request(json.loads(line))
|
||||
response = {"id": request_id, "ok": True, "evidence": _detect(model, candidates)}
|
||||
except Exception:
|
||||
response = {"id": request_id, "ok": False, "error": "detection_failed"}
|
||||
print(json.dumps(response, separators=(",", ":")), flush=True)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+34
-8
@@ -3,6 +3,7 @@ import cors from "@fastify/cors";
|
||||
import cookie from "@fastify/cookie";
|
||||
import rateLimit from "@fastify/rate-limit";
|
||||
import { dirname, isAbsolute, join } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { tmpdir } from "node:os";
|
||||
import type { AppConfig } from "./config.js";
|
||||
import { ThtRunner } from "./tht/tht-runner.js";
|
||||
@@ -57,8 +58,11 @@ import { metadataGenerationModelRoutes } from "./routes/metadata-generation-mode
|
||||
import { catalogDescriptionConsolidationRoutes } from "./routes/catalog-description-consolidation.js";
|
||||
import { PythonModelCompleter, type ModelCompleter } from "./catalog/model-completer.js";
|
||||
import { DescriptionGenerationWorker } from "./catalog/description-generation-worker.js";
|
||||
import { SensitiveDataSuggester } from "./catalog/sensitive-data-suggester.js";
|
||||
import { SensitiveDataSuggestionRunner } from "./catalog/sensitive-data-suggestion-runner.js";
|
||||
import { SensitivityAnalysisService } from "./catalog/sensitivity-analysis-service.js";
|
||||
import { SensitivityAnalysisRunner } from "./catalog/sensitivity-analysis-runner.js";
|
||||
import { SensitivityClassifier, type LocalNerDetector, type SensitivityValueSource } from "./catalog/sensitivity-classifier.js";
|
||||
import { ConcreteSensitivityValueSource } from "./catalog/sensitivity-value-source.js";
|
||||
import { PythonLocalNerDetector } from "./catalog/local-ner-detector.js";
|
||||
import {
|
||||
ConcreteDescriptionSourceSampler,
|
||||
type DescriptionSourceSampler,
|
||||
@@ -93,6 +97,8 @@ export interface BuildAppDeps {
|
||||
runtimeModelCatalog?: RuntimeModelCatalog;
|
||||
modelCompleter?: ModelCompleter;
|
||||
descriptionSourceSampler?: DescriptionSourceSampler;
|
||||
sensitivityValueSource?: SensitivityValueSource;
|
||||
localNerDetector?: LocalNerDetector;
|
||||
workspaceRuntimeSupport?: (workspace: WorkspaceDescriptor) => boolean;
|
||||
maintenanceBarrier?: MaintenanceBarrier;
|
||||
piManagement?: PiManagementService;
|
||||
@@ -194,12 +200,24 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
|
||||
catalogOperationCoordinator,
|
||||
descriptionSourceSampler,
|
||||
);
|
||||
const sensitiveDataSuggester = new SensitiveDataSuggester(
|
||||
const sensitivityValueSource = deps?.sensitivityValueSource
|
||||
?? new ConcreteSensitivityValueSource(catalogPostgresAccess, workspaceSecretStore);
|
||||
const configuredNerWorker = config.sensitivityNer?.workerScript
|
||||
?? fileURLToPath(new URL("../python/sensitivity_ner_worker.py", import.meta.url));
|
||||
const localNerDetector = deps?.localNerDetector ?? (config.sensitivityNer
|
||||
? new PythonLocalNerDetector({
|
||||
pythonExecutable: config.sensitivityNer.pythonExecutable,
|
||||
workerScript: configuredNerWorker,
|
||||
modelPath: config.sensitivityNer.modelPath,
|
||||
cwd: dirname(configuredNerWorker),
|
||||
threads: config.sensitivityNer.threads,
|
||||
})
|
||||
: undefined);
|
||||
const sensitiveDataSuggester = new SensitivityAnalysisService(
|
||||
catalogRepository,
|
||||
metadataGenerationModels,
|
||||
modelCompleter,
|
||||
new SensitivityClassifier(sensitivityValueSource, localNerDetector),
|
||||
);
|
||||
const sensitiveDataSuggestionRunner = new SensitiveDataSuggestionRunner(
|
||||
const sensitivityAnalysisRunner = new SensitivityAnalysisRunner(
|
||||
catalogRepository,
|
||||
sensitiveDataSuggester,
|
||||
);
|
||||
@@ -235,12 +253,20 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
|
||||
);
|
||||
app.addHook("onReady", async () => { await catalogSyncWorker.initialize(); });
|
||||
app.addHook("onReady", async () => { await descriptionGenerationWorker.initialize(); });
|
||||
app.addHook("onReady", async () => { await sensitiveDataSuggestionRunner.initialize(); });
|
||||
app.addHook("onReady", async () => { await sensitivityAnalysisRunner.initialize(); });
|
||||
if (localNerDetector?.warmup) {
|
||||
app.addHook("onReady", async () => {
|
||||
void localNerDetector.warmup?.().catch(() => undefined);
|
||||
});
|
||||
}
|
||||
if (!deps?.catalogRepository && catalogRepository.close) {
|
||||
app.addHook("onClose", async () => { await catalogRepository.close?.(); });
|
||||
}
|
||||
app.addHook("onClose", async () => { await catalogSyncWorker.stop(); });
|
||||
app.addHook("onClose", async () => { await descriptionGenerationWorker.stop(); });
|
||||
if (localNerDetector?.close) {
|
||||
app.addHook("onClose", async () => { await localNerDetector.close?.(); });
|
||||
}
|
||||
const workspaceDiagnoser = deps?.workspaceDiagnoser
|
||||
?? createProductionWorkspaceDiagnoser(config.workspaceDiagnosticTimeoutMs, undefined, {
|
||||
internalQdrantUrl: config.internalQdrantUrl,
|
||||
@@ -483,7 +509,7 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc
|
||||
catalogDescriptionGenerationRoutes(app, {
|
||||
repository: catalogRepository,
|
||||
worker: descriptionGenerationWorker,
|
||||
sensitiveDataSuggestionRunner,
|
||||
sensitivityAnalysisRunner,
|
||||
});
|
||||
settingsRoutes(app, { cfg: config, getSettings });
|
||||
piManagementRoutes(app, { service: piManagement });
|
||||
|
||||
@@ -0,0 +1,254 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process";
|
||||
import { tmpdir } from "node:os";
|
||||
import { z } from "zod";
|
||||
import type {
|
||||
LocalNerCandidate,
|
||||
LocalNerDetector,
|
||||
LocalNerEvidence,
|
||||
} from "./sensitivity-classifier.js";
|
||||
|
||||
const MAX_LINE_BYTES = 64 * 1024;
|
||||
const candidateSchema = z.object({
|
||||
columnId: z.uuid(),
|
||||
text: z.string().min(1).max(500),
|
||||
}).strict();
|
||||
const workerMessageSchema = z.union([
|
||||
z.object({ ready: z.literal(true) }).strict(),
|
||||
z.object({
|
||||
id: z.uuid(),
|
||||
ok: z.literal(true),
|
||||
evidence: z.array(z.object({
|
||||
columnId: z.uuid(),
|
||||
label: z.string().min(1).max(80),
|
||||
confidence: z.number().min(0).max(1),
|
||||
}).strict()).max(1_000),
|
||||
}).strict(),
|
||||
z.object({ id: z.uuid(), ok: z.literal(false), error: z.string().min(1).max(80) }).strict(),
|
||||
]);
|
||||
|
||||
export class LocalNerUnavailableError extends Error {
|
||||
constructor() {
|
||||
super("local NER is unavailable");
|
||||
this.name = "LocalNerUnavailableError";
|
||||
}
|
||||
}
|
||||
|
||||
interface PendingRequest {
|
||||
resolve: (value: readonly LocalNerEvidence[]) => void;
|
||||
reject: (error: Error) => void;
|
||||
timer: ReturnType<typeof setTimeout>;
|
||||
signal: AbortSignal;
|
||||
cancel: () => void;
|
||||
}
|
||||
|
||||
/** Persistent JSONL adapter for the optional, CPU-only Python NER worker. */
|
||||
export class PythonLocalNerDetector implements LocalNerDetector {
|
||||
private child?: ChildProcessWithoutNullStreams;
|
||||
private ready?: Promise<void>;
|
||||
private readyResolve?: () => void;
|
||||
private readyReject?: (error: Error) => void;
|
||||
private workerReady = false;
|
||||
private stdout = "";
|
||||
private readonly pending = new Map<string, PendingRequest>();
|
||||
|
||||
constructor(private readonly options: {
|
||||
pythonExecutable: string;
|
||||
workerScript: string;
|
||||
modelPath: string;
|
||||
cwd: string;
|
||||
threads?: number;
|
||||
startupTimeoutMs?: number;
|
||||
}) {}
|
||||
|
||||
async warmup(): Promise<void> {
|
||||
await this.ensureStarted();
|
||||
}
|
||||
|
||||
isReady(): boolean {
|
||||
return this.workerReady
|
||||
&& this.child !== undefined
|
||||
&& this.child.exitCode === null
|
||||
&& this.child.signalCode === null;
|
||||
}
|
||||
|
||||
async detect(
|
||||
candidates: readonly LocalNerCandidate[],
|
||||
signal: AbortSignal,
|
||||
deadline: number,
|
||||
): Promise<readonly LocalNerEvidence[]> {
|
||||
const parsed = z.array(candidateSchema).min(1).max(128).parse(candidates);
|
||||
if (signal.aborted || deadline <= Date.now()) throw new LocalNerUnavailableError();
|
||||
await this.ensureStartedWithin(signal, deadline);
|
||||
if (!this.child || this.child.exitCode !== null || this.child.signalCode !== null) {
|
||||
throw new LocalNerUnavailableError();
|
||||
}
|
||||
const id = randomUUID();
|
||||
return await new Promise<readonly LocalNerEvidence[]>((resolve, reject) => {
|
||||
const fail = () => {
|
||||
this.finishPending(id);
|
||||
reject(new LocalNerUnavailableError());
|
||||
this.stopWorker();
|
||||
};
|
||||
const timer = setTimeout(fail, Math.max(1, Math.floor(deadline - Date.now())));
|
||||
const cancel = fail;
|
||||
const pending: PendingRequest = { resolve, reject, timer, signal, cancel };
|
||||
this.pending.set(id, pending);
|
||||
signal.addEventListener("abort", cancel, { once: true });
|
||||
this.child!.stdin.write(`${JSON.stringify({ id, candidates: parsed })}\n`, (error) => {
|
||||
if (error) fail();
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
async close(): Promise<void> {
|
||||
const child = this.child;
|
||||
if (!child || child.exitCode !== null || child.signalCode !== null) return;
|
||||
await new Promise<void>((resolve) => {
|
||||
child.once("close", () => resolve());
|
||||
child.kill("SIGTERM");
|
||||
setTimeout(() => {
|
||||
if (child.exitCode === null && child.signalCode === null) child.kill("SIGKILL");
|
||||
}, 250).unref();
|
||||
});
|
||||
}
|
||||
|
||||
private async ensureStarted(): Promise<void> {
|
||||
if (this.ready) return await this.ready;
|
||||
this.ready = new Promise<void>((resolve, reject) => {
|
||||
this.readyResolve = resolve;
|
||||
this.readyReject = reject;
|
||||
});
|
||||
const threads = String(this.options.threads ?? 2);
|
||||
const inheritedRuntimeEnvironment = Object.fromEntries([
|
||||
"PATH", "SystemRoot", "WINDIR", "PATHEXT", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL",
|
||||
].flatMap((name) => process.env[name] === undefined ? [] : [[name, process.env[name]!]]));
|
||||
const child = spawn(this.options.pythonExecutable, [
|
||||
"-I",
|
||||
"-B",
|
||||
this.options.workerScript,
|
||||
"--model",
|
||||
this.options.modelPath,
|
||||
"--threads",
|
||||
threads,
|
||||
], {
|
||||
cwd: this.options.cwd,
|
||||
stdio: ["pipe", "pipe", "pipe"],
|
||||
env: {
|
||||
...inheritedRuntimeEnvironment,
|
||||
HOME: process.env.HOME ?? tmpdir(),
|
||||
CUDA_VISIBLE_DEVICES: "",
|
||||
HIP_VISIBLE_DEVICES: "",
|
||||
HF_HUB_OFFLINE: "1",
|
||||
HF_HUB_DISABLE_TELEMETRY: "1",
|
||||
TRANSFORMERS_OFFLINE: "1",
|
||||
TOKENIZERS_PARALLELISM: "false",
|
||||
PYTHONNOUSERSITE: "1",
|
||||
OMP_NUM_THREADS: threads,
|
||||
MKL_NUM_THREADS: threads,
|
||||
OPENBLAS_NUM_THREADS: threads,
|
||||
HTTP_PROXY: "",
|
||||
HTTPS_PROXY: "",
|
||||
ALL_PROXY: "",
|
||||
NO_PROXY: "*",
|
||||
},
|
||||
});
|
||||
this.child = child;
|
||||
child.stdout.setEncoding("utf8");
|
||||
child.stdout.on("data", (chunk: string) => this.receive(chunk));
|
||||
child.stderr.resume();
|
||||
child.once("error", () => this.failWorker());
|
||||
child.once("close", () => this.failWorker());
|
||||
const startupTimer = setTimeout(() => this.failWorker(), this.options.startupTimeoutMs ?? 120_000);
|
||||
startupTimer.unref();
|
||||
try {
|
||||
await this.ready;
|
||||
} finally {
|
||||
clearTimeout(startupTimer);
|
||||
}
|
||||
}
|
||||
|
||||
private async ensureStartedWithin(signal: AbortSignal, deadline: number): Promise<void> {
|
||||
const started = this.ensureStarted();
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
let settled = false;
|
||||
const finish = (error?: Error, stopWorker = false) => {
|
||||
if (settled) return;
|
||||
settled = true;
|
||||
clearTimeout(timer);
|
||||
signal.removeEventListener("abort", cancel);
|
||||
if (stopWorker) this.failWorker();
|
||||
if (error) reject(error);
|
||||
else resolve();
|
||||
};
|
||||
const cancel = () => finish(new LocalNerUnavailableError(), true);
|
||||
const timer = setTimeout(cancel, Math.max(1, Math.floor(deadline - Date.now())));
|
||||
signal.addEventListener("abort", cancel, { once: true });
|
||||
void started.then(
|
||||
() => finish(),
|
||||
() => finish(new LocalNerUnavailableError()),
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
private receive(chunk: string): void {
|
||||
this.stdout += chunk;
|
||||
if (Buffer.byteLength(this.stdout, "utf8") > MAX_LINE_BYTES) {
|
||||
this.failWorker();
|
||||
return;
|
||||
}
|
||||
let newline: number;
|
||||
while ((newline = this.stdout.indexOf("\n")) >= 0) {
|
||||
const line = this.stdout.slice(0, newline);
|
||||
this.stdout = this.stdout.slice(newline + 1);
|
||||
if (!line) continue;
|
||||
try {
|
||||
const message = workerMessageSchema.parse(JSON.parse(line));
|
||||
if ("ready" in message) {
|
||||
this.workerReady = true;
|
||||
this.readyResolve?.();
|
||||
this.readyResolve = undefined;
|
||||
this.readyReject = undefined;
|
||||
continue;
|
||||
}
|
||||
const pending = this.pending.get(message.id);
|
||||
if (!pending) continue;
|
||||
this.finishPending(message.id);
|
||||
if (message.ok) pending.resolve(message.evidence);
|
||||
else pending.reject(new LocalNerUnavailableError());
|
||||
} catch {
|
||||
this.failWorker();
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private finishPending(id: string): void {
|
||||
const pending = this.pending.get(id);
|
||||
if (!pending) return;
|
||||
clearTimeout(pending.timer);
|
||||
pending.signal.removeEventListener("abort", pending.cancel);
|
||||
this.pending.delete(id);
|
||||
}
|
||||
|
||||
private stopWorker(): void {
|
||||
const child = this.child;
|
||||
if (child && child.exitCode === null && child.signalCode === null) child.kill("SIGTERM");
|
||||
}
|
||||
|
||||
private failWorker(): void {
|
||||
const error = new LocalNerUnavailableError();
|
||||
this.readyReject?.(error);
|
||||
this.readyResolve = undefined;
|
||||
this.readyReject = undefined;
|
||||
for (const [id, pending] of this.pending) {
|
||||
this.finishPending(id);
|
||||
pending.reject(error);
|
||||
}
|
||||
this.stopWorker();
|
||||
this.child = undefined;
|
||||
this.ready = undefined;
|
||||
this.workerReady = false;
|
||||
this.stdout = "";
|
||||
}
|
||||
}
|
||||
@@ -35,10 +35,10 @@ import {
|
||||
type DescriptionGenerationRun,
|
||||
type DescriptionGenerationRunUpdate,
|
||||
type DescriptionGenerationScope,
|
||||
type SensitiveDataSuggestionEvent,
|
||||
type SensitiveDataSuggestionRun,
|
||||
type SensitiveDataSuggestionRunUpdate,
|
||||
type SensitiveDataSuggestionScope,
|
||||
type SensitivityAnalysisEvent,
|
||||
type SensitivityAnalysisRun,
|
||||
type SensitivityAnalysisRunUpdate,
|
||||
type SensitivityAnalysisScope,
|
||||
type TableSyncRepositoryResult,
|
||||
type WorkspaceDatabase,
|
||||
} from "./types.js";
|
||||
@@ -56,8 +56,8 @@ export class MemoryCatalogRepository implements CatalogRepository {
|
||||
private readonly logicalRelationships = new Map<string, CatalogLogicalRelationship>();
|
||||
private readonly descriptionGenerationRuns = new Map<string, DescriptionGenerationRun>();
|
||||
private readonly descriptionGenerationEvents = new Map<string, DescriptionGenerationEvent[]>();
|
||||
private readonly sensitiveDataSuggestionRuns = new Map<string, SensitiveDataSuggestionRun>();
|
||||
private readonly sensitiveDataSuggestionEvents = new Map<string, SensitiveDataSuggestionEvent[]>();
|
||||
private readonly sensitivityAnalysisRuns = new Map<string, SensitivityAnalysisRun>();
|
||||
private readonly sensitivityAnalysisEvents = new Map<string, SensitivityAnalysisEvent[]>();
|
||||
private readonly syncRuns = new Map<string, CatalogSyncRun>();
|
||||
private readonly syncEvents = new Map<string, CatalogSyncEvent[]>();
|
||||
|
||||
@@ -183,10 +183,10 @@ export class MemoryCatalogRepository implements CatalogRepository {
|
||||
this.descriptionGenerationRuns.delete(runId);
|
||||
this.descriptionGenerationEvents.delete(runId);
|
||||
}
|
||||
for (const [runId, run] of this.sensitiveDataSuggestionRuns) {
|
||||
for (const [runId, run] of this.sensitivityAnalysisRuns) {
|
||||
if (run.databaseId !== id) continue;
|
||||
this.sensitiveDataSuggestionRuns.delete(runId);
|
||||
this.sensitiveDataSuggestionEvents.delete(runId);
|
||||
this.sensitivityAnalysisRuns.delete(runId);
|
||||
this.sensitivityAnalysisEvents.delete(runId);
|
||||
}
|
||||
return this.records.delete(id);
|
||||
}
|
||||
@@ -433,93 +433,96 @@ export class MemoryCatalogRepository implements CatalogRepository {
|
||||
.map((event) => structuredClone(event));
|
||||
}
|
||||
|
||||
async createSensitiveDataSuggestionRun(
|
||||
async createSensitivityAnalysisRun(
|
||||
databaseId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
modelId: string,
|
||||
): Promise<SensitiveDataSuggestionRun> {
|
||||
scope: SensitivityAnalysisScope,
|
||||
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
|
||||
): Promise<SensitivityAnalysisRun> {
|
||||
const now = new Date().toISOString();
|
||||
const run: SensitiveDataSuggestionRun = {
|
||||
const run: SensitivityAnalysisRun = {
|
||||
id: randomUUID(),
|
||||
databaseId,
|
||||
scope,
|
||||
modelId,
|
||||
engine: origin.engine,
|
||||
modelId: origin.engine === "llm" ? origin.modelId : null,
|
||||
policyVersion: origin.engine === "local" ? origin.policyVersion : null,
|
||||
status: "running",
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
inputTokens: 0,
|
||||
cacheReadTokens: 0,
|
||||
outputTokens: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
inputTokens: 0,
|
||||
cacheReadTokens: 0,
|
||||
outputTokens: 0,
|
||||
createdAt: now,
|
||||
startedAt: now,
|
||||
updatedAt: now,
|
||||
finishedAt: null,
|
||||
errorSummary: null,
|
||||
};
|
||||
this.sensitiveDataSuggestionRuns.set(run.id, run);
|
||||
this.sensitivityAnalysisRuns.set(run.id, run);
|
||||
return structuredClone(run);
|
||||
}
|
||||
|
||||
async getSensitiveDataSuggestionRun(
|
||||
async getSensitivityAnalysisRun(
|
||||
runId: string,
|
||||
): Promise<SensitiveDataSuggestionRun | undefined> {
|
||||
const run = this.sensitiveDataSuggestionRuns.get(runId);
|
||||
): Promise<SensitivityAnalysisRun | undefined> {
|
||||
const run = this.sensitivityAnalysisRuns.get(runId);
|
||||
return run ? structuredClone(run) : undefined;
|
||||
}
|
||||
|
||||
async listSensitiveDataSuggestionRuns(limit = 50): Promise<SensitiveDataSuggestionRun[]> {
|
||||
return [...this.sensitiveDataSuggestionRuns.values()]
|
||||
async listSensitivityAnalysisRuns(limit = 50): Promise<SensitivityAnalysisRun[]> {
|
||||
return [...this.sensitivityAnalysisRuns.values()]
|
||||
.sort((a, b) => b.createdAt.localeCompare(a.createdAt) || b.id.localeCompare(a.id))
|
||||
.slice(0, limit)
|
||||
.map((run) => structuredClone(run));
|
||||
}
|
||||
|
||||
async interruptActiveSensitiveDataSuggestionRuns(
|
||||
async interruptActiveSensitivityAnalysisRuns(
|
||||
errorSummary: string,
|
||||
): Promise<SensitiveDataSuggestionRun[]> {
|
||||
const interrupted: SensitiveDataSuggestionRun[] = [];
|
||||
for (const run of this.sensitiveDataSuggestionRuns.values()) {
|
||||
): Promise<SensitivityAnalysisRun[]> {
|
||||
const interrupted: SensitivityAnalysisRun[] = [];
|
||||
for (const run of this.sensitivityAnalysisRuns.values()) {
|
||||
if (run.status !== "running") continue;
|
||||
const now = new Date().toISOString();
|
||||
const updated: SensitiveDataSuggestionRun = {
|
||||
const updated: SensitivityAnalysisRun = {
|
||||
...run,
|
||||
status: "interrupted",
|
||||
updatedAt: now,
|
||||
finishedAt: now,
|
||||
errorSummary,
|
||||
};
|
||||
this.sensitiveDataSuggestionRuns.set(run.id, updated);
|
||||
this.sensitivityAnalysisRuns.set(run.id, updated);
|
||||
interrupted.push(structuredClone(updated));
|
||||
}
|
||||
return interrupted;
|
||||
}
|
||||
|
||||
async updateSensitiveDataSuggestionRun(
|
||||
async updateSensitivityAnalysisRun(
|
||||
runId: string,
|
||||
update: SensitiveDataSuggestionRunUpdate,
|
||||
): Promise<SensitiveDataSuggestionRun | undefined> {
|
||||
const current = this.sensitiveDataSuggestionRuns.get(runId);
|
||||
update: SensitivityAnalysisRunUpdate,
|
||||
): Promise<SensitivityAnalysisRun | undefined> {
|
||||
const current = this.sensitivityAnalysisRuns.get(runId);
|
||||
if (!current) return undefined;
|
||||
const updated = {
|
||||
...current,
|
||||
...structuredClone(update),
|
||||
updatedAt: new Date().toISOString(),
|
||||
};
|
||||
this.sensitiveDataSuggestionRuns.set(runId, updated);
|
||||
this.sensitivityAnalysisRuns.set(runId, updated);
|
||||
return structuredClone(updated);
|
||||
}
|
||||
|
||||
async appendSensitiveDataSuggestionEvent(
|
||||
async appendSensitivityAnalysisEvent(
|
||||
runId: string,
|
||||
level: SensitiveDataSuggestionEvent["level"],
|
||||
level: SensitivityAnalysisEvent["level"],
|
||||
message: string,
|
||||
): Promise<SensitiveDataSuggestionEvent> {
|
||||
if (!this.sensitiveDataSuggestionRuns.has(runId)) {
|
||||
throw new CatalogConflictError("Sensitive Data Suggestion Run does not exist");
|
||||
): Promise<SensitivityAnalysisEvent> {
|
||||
if (!this.sensitivityAnalysisRuns.has(runId)) {
|
||||
throw new CatalogConflictError("Sensitivity Analysis Run does not exist");
|
||||
}
|
||||
const events = this.sensitiveDataSuggestionEvents.get(runId) ?? [];
|
||||
const event: SensitiveDataSuggestionEvent = {
|
||||
const events = this.sensitivityAnalysisEvents.get(runId) ?? [];
|
||||
const event: SensitivityAnalysisEvent = {
|
||||
runId,
|
||||
sequence: events.length + 1,
|
||||
level,
|
||||
@@ -527,15 +530,15 @@ export class MemoryCatalogRepository implements CatalogRepository {
|
||||
createdAt: new Date().toISOString(),
|
||||
};
|
||||
events.push(event);
|
||||
this.sensitiveDataSuggestionEvents.set(runId, events);
|
||||
this.sensitivityAnalysisEvents.set(runId, events);
|
||||
return structuredClone(event);
|
||||
}
|
||||
|
||||
async listSensitiveDataSuggestionEvents(
|
||||
async listSensitivityAnalysisEvents(
|
||||
runId: string,
|
||||
afterSequence = 0,
|
||||
): Promise<SensitiveDataSuggestionEvent[]> {
|
||||
return (this.sensitiveDataSuggestionEvents.get(runId) ?? [])
|
||||
): Promise<SensitivityAnalysisEvent[]> {
|
||||
return (this.sensitivityAnalysisEvents.get(runId) ?? [])
|
||||
.filter((event) => event.sequence > afterSequence)
|
||||
.map((event) => structuredClone(event));
|
||||
}
|
||||
|
||||
@@ -9,10 +9,11 @@ import * as catalogSchemaSyncMigration from "./migrations/003_catalog_schema_syn
|
||||
import * as catalogRuntimeSequencePrivilegesMigration from "./migrations/004_catalog_runtime_sequence_privileges.js";
|
||||
import * as descriptionGenerationRunsMigration from "./migrations/005_description_generation_runs.js";
|
||||
import * as sensitiveDataFlagMigration from "./migrations/006_sensitive_data_flag.js";
|
||||
import * as sensitiveDataSuggestionRunsMigration from "./migrations/007_sensitive_data_suggestion_runs.js";
|
||||
import * as sensitivityAnalysisRunsMigration from "./migrations/007_sensitive_data_suggestion_runs.js";
|
||||
import * as catalogLogicalRelationshipsMigration from "./migrations/008_catalog_logical_relationships.js";
|
||||
import * as aiTokenUsageMigration from "./migrations/009_ai_token_usage.js";
|
||||
import * as canonicalModelIdsMigration from "./migrations/010_canonical_model_ids.js";
|
||||
import * as localSensitivityAnalysisMigration from "./migrations/011_local_sensitivity_analysis.js";
|
||||
|
||||
const connectionString = process.env.THT_CATALOG_MIGRATOR_DATABASE_URL;
|
||||
const host = process.env.THT_CATALOG_DB_HOST;
|
||||
@@ -44,10 +45,11 @@ const provider: MigrationProvider = {
|
||||
"004_catalog_runtime_sequence_privileges": catalogRuntimeSequencePrivilegesMigration,
|
||||
"005_description_generation_runs": descriptionGenerationRunsMigration,
|
||||
"006_sensitive_data_flag": sensitiveDataFlagMigration,
|
||||
"007_sensitive_data_suggestion_runs": sensitiveDataSuggestionRunsMigration,
|
||||
"007_sensitive_data_suggestion_runs": sensitivityAnalysisRunsMigration,
|
||||
"008_catalog_logical_relationships": catalogLogicalRelationshipsMigration,
|
||||
"009_ai_token_usage": aiTokenUsageMigration,
|
||||
"010_canonical_model_ids": canonicalModelIdsMigration,
|
||||
"011_local_sensitivity_analysis": localSensitivityAnalysisMigration,
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
import { sql, type Kysely } from "kysely";
|
||||
import type { CatalogDatabase } from "../repository.js";
|
||||
|
||||
export async function up(db: Kysely<CatalogDatabase>): Promise<void> {
|
||||
await sql.raw(`alter table sensitive_data_suggestion_runs
|
||||
alter column model_id drop not null,
|
||||
add column engine text not null default 'llm',
|
||||
add column policy_version text,
|
||||
add column unknown integer not null default 0,
|
||||
drop constraint sensitive_data_suggestion_runs_counters_check,
|
||||
add constraint sensitive_data_suggestion_runs_counters_check
|
||||
check (total >= 0
|
||||
and suggested_sensitive >= 0
|
||||
and suggested_non_sensitive >= 0
|
||||
and unknown >= 0
|
||||
and suggested_sensitive + suggested_non_sensitive + unknown <= total),
|
||||
add constraint sensitive_data_suggestion_runs_engine_check
|
||||
check (engine in ('llm', 'local')),
|
||||
add constraint sensitive_data_suggestion_runs_origin_check
|
||||
check ((engine = 'llm' and model_id is not null and policy_version is null)
|
||||
or (engine = 'local' and model_id is null
|
||||
and policy_version ~ '^[a-z][a-z0-9._-]{0,63}$'))`).execute(db);
|
||||
}
|
||||
|
||||
export async function down(db: Kysely<CatalogDatabase>): Promise<void> {
|
||||
await sql.raw(`alter table sensitive_data_suggestion_runs
|
||||
drop constraint sensitive_data_suggestion_runs_origin_check,
|
||||
drop constraint sensitive_data_suggestion_runs_engine_check,
|
||||
drop constraint sensitive_data_suggestion_runs_counters_check`).execute(db);
|
||||
await sql.raw(`update sensitive_data_suggestion_runs
|
||||
set model_id = coalesce(model_id, 'local/sensitivity-v1')`).execute(db);
|
||||
await sql.raw(`alter table sensitive_data_suggestion_runs
|
||||
drop column unknown,
|
||||
drop column policy_version,
|
||||
drop column engine,
|
||||
alter column model_id set not null,
|
||||
add constraint sensitive_data_suggestion_runs_counters_check
|
||||
check (total >= 0
|
||||
and suggested_sensitive >= 0
|
||||
and suggested_non_sensitive >= 0
|
||||
and suggested_sensitive + suggested_non_sensitive <= total)`).execute(db);
|
||||
}
|
||||
@@ -45,10 +45,10 @@ import {
|
||||
type DescriptionGenerationScope,
|
||||
type ObservedCatalogTable,
|
||||
type ObservedSchemaSnapshot,
|
||||
type SensitiveDataSuggestionEvent,
|
||||
type SensitiveDataSuggestionRun,
|
||||
type SensitiveDataSuggestionRunUpdate,
|
||||
type SensitiveDataSuggestionScope,
|
||||
type SensitivityAnalysisEvent,
|
||||
type SensitivityAnalysisRun,
|
||||
type SensitivityAnalysisRunUpdate,
|
||||
type SensitivityAnalysisScope,
|
||||
type TableSyncRepositoryResult,
|
||||
type WorkspaceDatabase,
|
||||
} from "./types.js";
|
||||
@@ -189,15 +189,18 @@ interface DescriptionGenerationEventTable {
|
||||
createdAt: Timestamp;
|
||||
}
|
||||
|
||||
interface SensitiveDataSuggestionRunTable {
|
||||
interface SensitivityAnalysisRunTable {
|
||||
id: string;
|
||||
databaseId: string;
|
||||
scope: SensitiveDataSuggestionScope;
|
||||
modelId: string;
|
||||
status: SensitiveDataSuggestionRun["status"];
|
||||
scope: SensitivityAnalysisScope;
|
||||
engine: SensitivityAnalysisRun["engine"];
|
||||
modelId: string | null;
|
||||
policyVersion: string | null;
|
||||
status: SensitivityAnalysisRun["status"];
|
||||
total: number;
|
||||
suggestedSensitive: number;
|
||||
suggestedNonSensitive: number;
|
||||
unknown: number;
|
||||
inputTokens: number;
|
||||
cacheReadTokens: number;
|
||||
outputTokens: number;
|
||||
@@ -208,10 +211,10 @@ interface SensitiveDataSuggestionRunTable {
|
||||
errorSummary: string | null;
|
||||
}
|
||||
|
||||
interface SensitiveDataSuggestionEventTable {
|
||||
interface SensitivityAnalysisEventTable {
|
||||
runId: string;
|
||||
sequence: number;
|
||||
level: SensitiveDataSuggestionEvent["level"];
|
||||
level: SensitivityAnalysisEvent["level"];
|
||||
message: string;
|
||||
createdAt: Timestamp;
|
||||
}
|
||||
@@ -264,8 +267,9 @@ export interface CatalogDatabase {
|
||||
catalogLogicalRelationships: CatalogLogicalRelationshipTable;
|
||||
descriptionGenerationRuns: DescriptionGenerationRunTable;
|
||||
descriptionGenerationEvents: DescriptionGenerationEventTable;
|
||||
sensitiveDataSuggestionRuns: SensitiveDataSuggestionRunTable;
|
||||
sensitiveDataSuggestionEvents: SensitiveDataSuggestionEventTable;
|
||||
// Legacy physical table names retained for migration and storage compatibility.
|
||||
sensitiveDataSuggestionRuns: SensitivityAnalysisRunTable;
|
||||
sensitiveDataSuggestionEvents: SensitivityAnalysisEventTable;
|
||||
catalogSyncRuns: CatalogSyncRunTable;
|
||||
catalogSyncEvents: CatalogSyncEventTable;
|
||||
}
|
||||
@@ -402,9 +406,9 @@ function serializeDescriptionGenerationEvent(
|
||||
return { ...row, createdAt: new Date(row.createdAt).toISOString() };
|
||||
}
|
||||
|
||||
function serializeSensitiveDataSuggestionRun(
|
||||
row: Selectable<SensitiveDataSuggestionRunTable>,
|
||||
): SensitiveDataSuggestionRun {
|
||||
function serializeSensitivityAnalysisRun(
|
||||
row: Selectable<SensitivityAnalysisRunTable>,
|
||||
): SensitivityAnalysisRun {
|
||||
const stamp = (value: Date | string | null) => value === null ? null : new Date(value).toISOString();
|
||||
return {
|
||||
...row,
|
||||
@@ -415,9 +419,9 @@ function serializeSensitiveDataSuggestionRun(
|
||||
};
|
||||
}
|
||||
|
||||
function serializeSensitiveDataSuggestionEvent(
|
||||
row: Selectable<SensitiveDataSuggestionEventTable>,
|
||||
): SensitiveDataSuggestionEvent {
|
||||
function serializeSensitivityAnalysisEvent(
|
||||
row: Selectable<SensitivityAnalysisEventTable>,
|
||||
): SensitivityAnalysisEvent {
|
||||
return { ...row, createdAt: new Date(row.createdAt).toISOString() };
|
||||
}
|
||||
|
||||
@@ -938,52 +942,55 @@ export class KyselyCatalogRepository implements CatalogRepository {
|
||||
return rows.map(serializeDescriptionGenerationEvent);
|
||||
}
|
||||
|
||||
async createSensitiveDataSuggestionRun(
|
||||
async createSensitivityAnalysisRun(
|
||||
databaseId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
modelId: string,
|
||||
): Promise<SensitiveDataSuggestionRun> {
|
||||
scope: SensitivityAnalysisScope,
|
||||
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
|
||||
): Promise<SensitivityAnalysisRun> {
|
||||
const row = await this.db.insertInto("sensitiveDataSuggestionRuns").values({
|
||||
id: randomUUID(),
|
||||
databaseId,
|
||||
scope,
|
||||
modelId,
|
||||
engine: origin.engine,
|
||||
modelId: origin.engine === "llm" ? origin.modelId : null,
|
||||
policyVersion: origin.engine === "local" ? origin.policyVersion : null,
|
||||
status: "running",
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
inputTokens: 0,
|
||||
cacheReadTokens: 0,
|
||||
outputTokens: 0,
|
||||
finishedAt: null,
|
||||
errorSummary: null,
|
||||
}).returningAll().executeTakeFirstOrThrow();
|
||||
return serializeSensitiveDataSuggestionRun(row);
|
||||
return serializeSensitivityAnalysisRun(row);
|
||||
}
|
||||
|
||||
async getSensitiveDataSuggestionRun(
|
||||
async getSensitivityAnalysisRun(
|
||||
runId: string,
|
||||
): Promise<SensitiveDataSuggestionRun | undefined> {
|
||||
): Promise<SensitivityAnalysisRun | undefined> {
|
||||
const row = await this.db.selectFrom("sensitiveDataSuggestionRuns")
|
||||
.selectAll()
|
||||
.where("id", "=", runId)
|
||||
.executeTakeFirst();
|
||||
return row ? serializeSensitiveDataSuggestionRun(row) : undefined;
|
||||
return row ? serializeSensitivityAnalysisRun(row) : undefined;
|
||||
}
|
||||
|
||||
async listSensitiveDataSuggestionRuns(limit = 50): Promise<SensitiveDataSuggestionRun[]> {
|
||||
async listSensitivityAnalysisRuns(limit = 50): Promise<SensitivityAnalysisRun[]> {
|
||||
const rows = await this.db.selectFrom("sensitiveDataSuggestionRuns")
|
||||
.selectAll()
|
||||
.orderBy("createdAt", "desc")
|
||||
.orderBy("id", "desc")
|
||||
.limit(limit)
|
||||
.execute();
|
||||
return rows.map(serializeSensitiveDataSuggestionRun);
|
||||
return rows.map(serializeSensitivityAnalysisRun);
|
||||
}
|
||||
|
||||
async interruptActiveSensitiveDataSuggestionRuns(
|
||||
async interruptActiveSensitivityAnalysisRuns(
|
||||
errorSummary: string,
|
||||
): Promise<SensitiveDataSuggestionRun[]> {
|
||||
): Promise<SensitivityAnalysisRun[]> {
|
||||
const rows = await this.db.updateTable("sensitiveDataSuggestionRuns")
|
||||
.set({
|
||||
status: "interrupted",
|
||||
@@ -994,34 +1001,34 @@ export class KyselyCatalogRepository implements CatalogRepository {
|
||||
.where("status", "=", "running")
|
||||
.returningAll()
|
||||
.execute();
|
||||
return rows.map(serializeSensitiveDataSuggestionRun);
|
||||
return rows.map(serializeSensitivityAnalysisRun);
|
||||
}
|
||||
|
||||
async updateSensitiveDataSuggestionRun(
|
||||
async updateSensitivityAnalysisRun(
|
||||
runId: string,
|
||||
update: SensitiveDataSuggestionRunUpdate,
|
||||
): Promise<SensitiveDataSuggestionRun | undefined> {
|
||||
update: SensitivityAnalysisRunUpdate,
|
||||
): Promise<SensitivityAnalysisRun | undefined> {
|
||||
const values: any = { ...update, updatedAt: sql`now()` };
|
||||
const row = await this.db.updateTable("sensitiveDataSuggestionRuns")
|
||||
.set(values)
|
||||
.where("id", "=", runId)
|
||||
.returningAll()
|
||||
.executeTakeFirst();
|
||||
return row ? serializeSensitiveDataSuggestionRun(row) : undefined;
|
||||
return row ? serializeSensitivityAnalysisRun(row) : undefined;
|
||||
}
|
||||
|
||||
async appendSensitiveDataSuggestionEvent(
|
||||
async appendSensitivityAnalysisEvent(
|
||||
runId: string,
|
||||
level: SensitiveDataSuggestionEvent["level"],
|
||||
level: SensitivityAnalysisEvent["level"],
|
||||
message: string,
|
||||
): Promise<SensitiveDataSuggestionEvent> {
|
||||
): Promise<SensitivityAnalysisEvent> {
|
||||
return await this.db.transaction().execute(async (trx) => {
|
||||
const run = await trx.selectFrom("sensitiveDataSuggestionRuns")
|
||||
.select("id")
|
||||
.where("id", "=", runId)
|
||||
.forUpdate()
|
||||
.executeTakeFirst();
|
||||
if (!run) throw new CatalogConflictError("Sensitive Data Suggestion Run does not exist");
|
||||
if (!run) throw new CatalogConflictError("Sensitivity Analysis Run does not exist");
|
||||
const current = await trx.selectFrom("sensitiveDataSuggestionEvents")
|
||||
.select(sql<number>`coalesce(max(sequence), 0)::int`.as("sequence"))
|
||||
.where("runId", "=", runId)
|
||||
@@ -1032,21 +1039,21 @@ export class KyselyCatalogRepository implements CatalogRepository {
|
||||
level,
|
||||
message,
|
||||
}).returningAll().executeTakeFirstOrThrow();
|
||||
return serializeSensitiveDataSuggestionEvent(row);
|
||||
return serializeSensitivityAnalysisEvent(row);
|
||||
});
|
||||
}
|
||||
|
||||
async listSensitiveDataSuggestionEvents(
|
||||
async listSensitivityAnalysisEvents(
|
||||
runId: string,
|
||||
afterSequence = 0,
|
||||
): Promise<SensitiveDataSuggestionEvent[]> {
|
||||
): Promise<SensitivityAnalysisEvent[]> {
|
||||
const rows = await this.db.selectFrom("sensitiveDataSuggestionEvents")
|
||||
.selectAll()
|
||||
.where("runId", "=", runId)
|
||||
.where("sequence", ">", afterSequence)
|
||||
.orderBy("sequence")
|
||||
.execute();
|
||||
return rows.map(serializeSensitiveDataSuggestionEvent);
|
||||
return rows.map(serializeSensitivityAnalysisEvent);
|
||||
}
|
||||
|
||||
async listRelationships(databaseId: string): Promise<CatalogPhysicalRelationship[]> {
|
||||
@@ -1792,13 +1799,13 @@ export class UnavailableCatalogRepository implements CatalogRepository {
|
||||
async updateDescriptionGenerationRun(): Promise<DescriptionGenerationRun | undefined> { return this.fail(); }
|
||||
async appendDescriptionGenerationEvent(): Promise<DescriptionGenerationEvent> { return this.fail(); }
|
||||
async listDescriptionGenerationEvents(): Promise<DescriptionGenerationEvent[]> { return this.fail(); }
|
||||
async createSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun> { return this.fail(); }
|
||||
async getSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun | undefined> { return this.fail(); }
|
||||
async listSensitiveDataSuggestionRuns(): Promise<SensitiveDataSuggestionRun[]> { return this.fail(); }
|
||||
async interruptActiveSensitiveDataSuggestionRuns(): Promise<SensitiveDataSuggestionRun[]> { return this.fail(); }
|
||||
async updateSensitiveDataSuggestionRun(): Promise<SensitiveDataSuggestionRun | undefined> { return this.fail(); }
|
||||
async appendSensitiveDataSuggestionEvent(): Promise<SensitiveDataSuggestionEvent> { return this.fail(); }
|
||||
async listSensitiveDataSuggestionEvents(): Promise<SensitiveDataSuggestionEvent[]> { return this.fail(); }
|
||||
async createSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun> { return this.fail(); }
|
||||
async getSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun | undefined> { return this.fail(); }
|
||||
async listSensitivityAnalysisRuns(): Promise<SensitivityAnalysisRun[]> { return this.fail(); }
|
||||
async interruptActiveSensitivityAnalysisRuns(): Promise<SensitivityAnalysisRun[]> { return this.fail(); }
|
||||
async updateSensitivityAnalysisRun(): Promise<SensitivityAnalysisRun | undefined> { return this.fail(); }
|
||||
async appendSensitivityAnalysisEvent(): Promise<SensitivityAnalysisEvent> { return this.fail(); }
|
||||
async listSensitivityAnalysisEvents(): Promise<SensitivityAnalysisEvent[]> { return this.fail(); }
|
||||
async listRelationships(): Promise<CatalogPhysicalRelationship[]> { return this.fail(); }
|
||||
async listLogicalRelationships(): Promise<CatalogLogicalRelationship[]> { return this.fail(); }
|
||||
async getLogicalRelationshipContext(): Promise<CatalogLogicalRelationshipContext | undefined> { return this.fail(); }
|
||||
|
||||
@@ -1,254 +0,0 @@
|
||||
import { z } from "zod";
|
||||
import type { MetadataGenerationModels } from "./metadata-generation-models.js";
|
||||
import type { ModelCompleter, ModelCompletionMessage, ModelCompletionResult, ModelCompletionUsage } from "./model-completer.js";
|
||||
import type {
|
||||
CatalogColumn,
|
||||
CatalogRepository,
|
||||
CatalogTable,
|
||||
SensitiveDataSuggestionScope,
|
||||
} from "./types.js";
|
||||
|
||||
export type { SensitiveDataSuggestionScope } from "./types.js";
|
||||
|
||||
// The helper accepts at most 64 KiB per message. Keep the same safety margin used by
|
||||
// Description Generation so UTF-8 structural metadata never reaches that hard limit.
|
||||
const MAX_USER_MESSAGE_BYTES = 60 * 1024;
|
||||
// Preserve ThothAI's proven completion granularity: small batches keep generation time and
|
||||
// structured-output accuracy predictable even when the helper byte limit would allow much more.
|
||||
const MAX_COLUMNS_PER_BATCH = 10;
|
||||
const responseSchema = z.object({
|
||||
suggestions: z.array(z.object({
|
||||
columnId: z.uuid(),
|
||||
sensitive: z.boolean(),
|
||||
}).strict()),
|
||||
}).strict();
|
||||
|
||||
interface StructuralColumn {
|
||||
columnId: string;
|
||||
tableId: string;
|
||||
table: string;
|
||||
column: string;
|
||||
dataType: string;
|
||||
nullable: boolean;
|
||||
primaryKey: boolean;
|
||||
foreignKey: boolean;
|
||||
version: number;
|
||||
currentSensitive: boolean;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestion {
|
||||
columnId: string;
|
||||
tableId: string;
|
||||
tableName: string;
|
||||
columnName: string;
|
||||
version: number;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionTargetNotFoundError extends Error {
|
||||
constructor(readonly target: "database" | "table" | "column") {
|
||||
super(`${target} not found`);
|
||||
this.name = "SensitiveDataSuggestionTargetNotFoundError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionDuplicateTargetIdsError extends Error {
|
||||
constructor() {
|
||||
super("sensitive-data suggestion target IDs must be unique");
|
||||
this.name = "SensitiveDataSuggestionDuplicateTargetIdsError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionNoEligibleColumnsError extends Error {
|
||||
constructor(readonly scope: SensitiveDataSuggestionScope) {
|
||||
super("selected scope has no catalog columns");
|
||||
this.name = "SensitiveDataSuggestionNoEligibleColumnsError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionPayloadTooLargeError extends Error {
|
||||
constructor() {
|
||||
super("sensitive-data suggestion structural metadata is too large");
|
||||
this.name = "SensitiveDataSuggestionPayloadTooLargeError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionInvalidResponseError extends Error {
|
||||
constructor() {
|
||||
super("sensitive-data suggestion response is invalid");
|
||||
this.name = "SensitiveDataSuggestionInvalidResponseError";
|
||||
}
|
||||
}
|
||||
|
||||
function userContent(
|
||||
database: { databaseName: string; schema: string },
|
||||
columns: readonly StructuralColumn[],
|
||||
): string {
|
||||
return JSON.stringify({
|
||||
database: database.databaseName,
|
||||
schema: database.schema,
|
||||
columns: columns.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
table: column.table,
|
||||
column: column.column,
|
||||
dataType: column.dataType,
|
||||
nullable: column.nullable,
|
||||
primaryKey: column.primaryKey,
|
||||
foreignKey: column.foreignKey,
|
||||
})),
|
||||
});
|
||||
}
|
||||
|
||||
function batchesFor(
|
||||
database: { databaseName: string; schema: string },
|
||||
columns: readonly StructuralColumn[],
|
||||
): StructuralColumn[][] {
|
||||
const batches: StructuralColumn[][] = [];
|
||||
let current: StructuralColumn[] = [];
|
||||
for (const column of columns) {
|
||||
if (current.length === MAX_COLUMNS_PER_BATCH) {
|
||||
batches.push(current);
|
||||
current = [];
|
||||
}
|
||||
const candidate = [...current, column];
|
||||
if (Buffer.byteLength(userContent(database, candidate), "utf8") <= MAX_USER_MESSAGE_BYTES) {
|
||||
current = candidate;
|
||||
continue;
|
||||
}
|
||||
if (current.length === 0) throw new SensitiveDataSuggestionPayloadTooLargeError();
|
||||
batches.push(current);
|
||||
current = [column];
|
||||
if (Buffer.byteLength(userContent(database, current), "utf8") > MAX_USER_MESSAGE_BYTES) {
|
||||
throw new SensitiveDataSuggestionPayloadTooLargeError();
|
||||
}
|
||||
}
|
||||
if (current.length > 0) batches.push(current);
|
||||
return batches;
|
||||
}
|
||||
|
||||
function structuralColumn(table: CatalogTable, column: CatalogColumn): StructuralColumn {
|
||||
return {
|
||||
columnId: column.id,
|
||||
tableId: table.id,
|
||||
table: table.name,
|
||||
column: column.name,
|
||||
dataType: column.dataType,
|
||||
nullable: column.isNullable,
|
||||
primaryKey: column.isPrimaryKey,
|
||||
foreignKey: column.isForeignKey,
|
||||
version: column.version,
|
||||
currentSensitive: column.sensitive,
|
||||
};
|
||||
}
|
||||
|
||||
const systemMessage: ModelCompletionMessage = {
|
||||
role: "system",
|
||||
content: [
|
||||
"Classify whether each database column is likely to contain sensitive source values.",
|
||||
"Use only the supplied structural metadata. Return strict JSON with this exact shape:",
|
||||
'{"suggestions":[{"columnId":"uuid","sensitive":true}]}',
|
||||
"Return every supplied column exactly once. Do not add explanations or markdown.",
|
||||
].join("\n"),
|
||||
};
|
||||
|
||||
export class SensitiveDataSuggester {
|
||||
constructor(
|
||||
private readonly repository: CatalogRepository,
|
||||
private readonly models: MetadataGenerationModels,
|
||||
private readonly completer: ModelCompleter,
|
||||
) {}
|
||||
|
||||
private async selectColumns(
|
||||
databaseId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
targetIds: readonly string[],
|
||||
): Promise<StructuralColumn[]> {
|
||||
if (new Set(targetIds).size !== targetIds.length) {
|
||||
throw new SensitiveDataSuggestionDuplicateTargetIdsError();
|
||||
}
|
||||
const tables = await this.repository.listTables(databaseId);
|
||||
const tableIds = new Set(targetIds);
|
||||
const selectedTables = scope === "selected_tables"
|
||||
? tables.filter((table) => tableIds.has(table.id))
|
||||
: tables;
|
||||
if (scope === "selected_tables" && selectedTables.length !== targetIds.length) {
|
||||
throw new SensitiveDataSuggestionTargetNotFoundError("table");
|
||||
}
|
||||
|
||||
const columns = (await Promise.all(selectedTables.map(async (table) => (
|
||||
(await this.repository.listColumns(databaseId, table.id)).map((column) => (
|
||||
structuralColumn(table, column)
|
||||
))
|
||||
)))).flat();
|
||||
const columnIds = new Set(targetIds);
|
||||
const selectedColumns = scope === "selected_columns"
|
||||
? columns.filter((column) => columnIds.has(column.columnId))
|
||||
: columns;
|
||||
if (scope === "selected_columns" && selectedColumns.length !== targetIds.length) {
|
||||
throw new SensitiveDataSuggestionTargetNotFoundError("column");
|
||||
}
|
||||
if (selectedColumns.length === 0) {
|
||||
throw new SensitiveDataSuggestionNoEligibleColumnsError(scope);
|
||||
}
|
||||
return selectedColumns;
|
||||
}
|
||||
|
||||
async suggest(
|
||||
databaseId: string,
|
||||
modelId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
targetIds: readonly string[],
|
||||
signal: AbortSignal,
|
||||
onPrepared?: (total: number) => void | Promise<void>,
|
||||
onProgress?: (processed: number, suggestions: readonly SensitiveDataSuggestion[]) => void | Promise<void>,
|
||||
onUsage?: (usage: ModelCompletionUsage) => void | Promise<void>,
|
||||
): Promise<readonly SensitiveDataSuggestion[]> {
|
||||
const database = await this.repository.get(databaseId);
|
||||
if (!database) throw new SensitiveDataSuggestionTargetNotFoundError("database");
|
||||
const columns = await this.selectColumns(databaseId, scope, targetIds);
|
||||
await onPrepared?.(columns.length);
|
||||
const model = this.models.resolve(modelId);
|
||||
const suggestions: SensitiveDataSuggestion[] = [];
|
||||
|
||||
for (const batch of batchesFor(database, columns)) {
|
||||
let received: Map<string, { columnId: string; sensitive: boolean }> | undefined;
|
||||
for (let attempt = 0; attempt < 2 && !received; attempt += 1) {
|
||||
const completion = await this.completer.complete({
|
||||
model,
|
||||
signal,
|
||||
messages: [systemMessage, { role: "user", content: userContent(database, batch) }],
|
||||
});
|
||||
const result: ModelCompletionResult = typeof completion === "string"
|
||||
? { content: completion, usage: { input: 0, cacheRead: 0, output: 0 } }
|
||||
: completion;
|
||||
await onUsage?.(result.usage);
|
||||
const content = result.content;
|
||||
try {
|
||||
const parsed = responseSchema.parse(JSON.parse(content));
|
||||
const expected = new Set(batch.map((column) => column.columnId));
|
||||
const candidate = new Map(parsed.suggestions.map((suggestion) => [suggestion.columnId, suggestion]));
|
||||
if (candidate.size !== parsed.suggestions.length
|
||||
|| candidate.size !== expected.size
|
||||
|| [...candidate.keys()].some((columnId) => !expected.has(columnId))) {
|
||||
throw new SensitiveDataSuggestionInvalidResponseError();
|
||||
}
|
||||
received = candidate;
|
||||
} catch {
|
||||
if (attempt === 1) throw new SensitiveDataSuggestionInvalidResponseError();
|
||||
}
|
||||
}
|
||||
suggestions.push(...batch.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
tableId: column.tableId,
|
||||
tableName: column.table,
|
||||
columnName: column.column,
|
||||
version: column.version,
|
||||
currentSensitive: column.currentSensitive,
|
||||
sensitive: received!.get(column.columnId)!.sensitive,
|
||||
})));
|
||||
await onProgress?.(suggestions.length, suggestions.slice(-batch.length));
|
||||
}
|
||||
return suggestions;
|
||||
}
|
||||
}
|
||||
@@ -1,136 +0,0 @@
|
||||
import type {
|
||||
SensitiveDataSuggestion,
|
||||
} from "./sensitive-data-suggester.js";
|
||||
import {
|
||||
SensitiveDataSuggester,
|
||||
SensitiveDataSuggestionTargetNotFoundError,
|
||||
} from "./sensitive-data-suggester.js";
|
||||
import type {
|
||||
CatalogRepository,
|
||||
SensitiveDataSuggestionRun,
|
||||
SensitiveDataSuggestionScope,
|
||||
} from "./types.js";
|
||||
import type { ModelCompletionUsage } from "./model-completer.js";
|
||||
|
||||
const interruptedMessage = "Sensitive-field suggestion generation was interrupted by backend restart.";
|
||||
const failedMessage = "Sensitive-field suggestion generation failed.";
|
||||
|
||||
export interface SensitiveDataSuggestionRunResult {
|
||||
suggestions: readonly SensitiveDataSuggestion[];
|
||||
run: SensitiveDataSuggestionRun;
|
||||
}
|
||||
|
||||
export class SensitiveDataSuggestionRunner {
|
||||
constructor(
|
||||
private readonly repository: CatalogRepository,
|
||||
private readonly suggester: SensitiveDataSuggester,
|
||||
) {}
|
||||
|
||||
async initialize(): Promise<void> {
|
||||
if (!(await this.repository.available())) return;
|
||||
const interrupted = await this.repository.interruptActiveSensitiveDataSuggestionRuns(
|
||||
interruptedMessage,
|
||||
);
|
||||
for (const run of interrupted) {
|
||||
await this.repository.appendSensitiveDataSuggestionEvent(
|
||||
run.id,
|
||||
"warning",
|
||||
interruptedMessage,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
async run(
|
||||
databaseId: string,
|
||||
modelId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
targetIds: readonly string[],
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitiveDataSuggestionRunResult> {
|
||||
if (!(await this.repository.get(databaseId))) {
|
||||
throw new SensitiveDataSuggestionTargetNotFoundError("database");
|
||||
}
|
||||
const started = await this.repository.createSensitiveDataSuggestionRun(
|
||||
databaseId,
|
||||
scope,
|
||||
modelId,
|
||||
);
|
||||
|
||||
try {
|
||||
await this.repository.appendSensitiveDataSuggestionEvent(
|
||||
started.id,
|
||||
"info",
|
||||
"Sensitive-field suggestion generation started.",
|
||||
);
|
||||
const suggestions = await this.suggester.suggest(
|
||||
databaseId,
|
||||
modelId,
|
||||
scope,
|
||||
targetIds,
|
||||
signal,
|
||||
async (total) => {
|
||||
const prepared = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
|
||||
total,
|
||||
});
|
||||
if (!prepared) throw new Error("Sensitive Data Suggestion Run disappeared");
|
||||
},
|
||||
async (processed, batch) => {
|
||||
const suggestedSensitive = batch.filter((suggestion) => suggestion.sensitive).length;
|
||||
const suggestedNonSensitive = batch.length - suggestedSensitive;
|
||||
const current = await this.repository.getSensitiveDataSuggestionRun(started.id);
|
||||
if (!current) throw new Error("Sensitive Data Suggestion Run disappeared");
|
||||
const progress = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
|
||||
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
|
||||
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
|
||||
});
|
||||
if (!progress) throw new Error("Sensitive Data Suggestion Run disappeared");
|
||||
await this.repository.appendSensitiveDataSuggestionEvent(
|
||||
started.id,
|
||||
"info",
|
||||
`Classified ${processed} of ${progress.total} columns.`,
|
||||
);
|
||||
},
|
||||
async (usage: ModelCompletionUsage) => {
|
||||
const current = await this.repository.getSensitiveDataSuggestionRun(started.id);
|
||||
if (!current) throw new Error("Sensitive Data Suggestion Run disappeared");
|
||||
await this.repository.updateSensitiveDataSuggestionRun(started.id, {
|
||||
inputTokens: current.inputTokens + usage.input,
|
||||
cacheReadTokens: current.cacheReadTokens + usage.cacheRead,
|
||||
outputTokens: current.outputTokens + usage.output,
|
||||
});
|
||||
},
|
||||
);
|
||||
const suggestedSensitive = suggestions.filter((suggestion) => suggestion.sensitive).length;
|
||||
const suggestedNonSensitive = suggestions.length - suggestedSensitive;
|
||||
await this.repository.appendSensitiveDataSuggestionEvent(
|
||||
started.id,
|
||||
"info",
|
||||
`Sensitive-field suggestion generation completed for ${suggestions.length} column${
|
||||
suggestions.length === 1 ? "" : "s"
|
||||
}.`,
|
||||
);
|
||||
const completed = await this.repository.updateSensitiveDataSuggestionRun(started.id, {
|
||||
status: "completed",
|
||||
total: suggestions.length,
|
||||
suggestedSensitive,
|
||||
suggestedNonSensitive,
|
||||
finishedAt: new Date().toISOString(),
|
||||
errorSummary: null,
|
||||
});
|
||||
if (!completed) throw new Error("Sensitive Data Suggestion Run disappeared");
|
||||
return { suggestions, run: completed };
|
||||
} catch (error) {
|
||||
await this.repository.updateSensitiveDataSuggestionRun(started.id, {
|
||||
status: "failed",
|
||||
finishedAt: new Date().toISOString(),
|
||||
errorSummary: failedMessage,
|
||||
}).catch(() => undefined);
|
||||
await this.repository.appendSensitiveDataSuggestionEvent(
|
||||
started.id,
|
||||
"error",
|
||||
failedMessage,
|
||||
).catch(() => undefined);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
import type {
|
||||
SensitivityReviewItem,
|
||||
} from "./sensitivity-analysis-service.js";
|
||||
import {
|
||||
SENSITIVITY_POLICY_VERSION,
|
||||
SensitivityAnalysisInterruptedError,
|
||||
SensitivityAnalysisService,
|
||||
SensitivityAnalysisTargetNotFoundError,
|
||||
} from "./sensitivity-analysis-service.js";
|
||||
import type {
|
||||
CatalogRepository,
|
||||
SensitivityAnalysisRun,
|
||||
SensitivityAnalysisScope,
|
||||
} from "./types.js";
|
||||
|
||||
const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart.";
|
||||
const deadlineMessage = "Local sensitivity analysis reached its time limit.";
|
||||
const failedMessage = "Local sensitivity analysis failed.";
|
||||
|
||||
function ensureActive(signal: AbortSignal): void {
|
||||
if (signal.aborted) throw new SensitivityAnalysisInterruptedError();
|
||||
}
|
||||
|
||||
export interface SensitivityAnalysisRunResult {
|
||||
suggestions: readonly SensitivityReviewItem[];
|
||||
run: SensitivityAnalysisRun;
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisRunner {
|
||||
constructor(
|
||||
private readonly repository: CatalogRepository,
|
||||
private readonly analysis: SensitivityAnalysisService,
|
||||
) {}
|
||||
|
||||
async initialize(): Promise<void> {
|
||||
if (!(await this.repository.available())) return;
|
||||
const interrupted = await this.repository.interruptActiveSensitivityAnalysisRuns(
|
||||
interruptedMessage,
|
||||
);
|
||||
for (const run of interrupted) {
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
run.id,
|
||||
"warning",
|
||||
interruptedMessage,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
async run(
|
||||
databaseId: string,
|
||||
scope: SensitivityAnalysisScope,
|
||||
targetIds: readonly string[],
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityAnalysisRunResult> {
|
||||
ensureActive(signal);
|
||||
const database = await this.repository.get(databaseId);
|
||||
ensureActive(signal);
|
||||
if (!database) {
|
||||
throw new SensitivityAnalysisTargetNotFoundError("database");
|
||||
}
|
||||
ensureActive(signal);
|
||||
const started = await this.repository.createSensitivityAnalysisRun(
|
||||
databaseId,
|
||||
scope,
|
||||
{ engine: "local", policyVersion: SENSITIVITY_POLICY_VERSION },
|
||||
);
|
||||
let preparedTotal = 0;
|
||||
let processedSensitive = 0;
|
||||
let processedNonSensitive = 0;
|
||||
|
||||
try {
|
||||
ensureActive(signal);
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
started.id,
|
||||
"info",
|
||||
"Local sensitivity analysis started.",
|
||||
);
|
||||
ensureActive(signal);
|
||||
const suggestions = await this.analysis.analyze(
|
||||
databaseId,
|
||||
scope,
|
||||
targetIds,
|
||||
signal,
|
||||
async (total) => {
|
||||
ensureActive(signal);
|
||||
preparedTotal = total;
|
||||
const prepared = await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
total,
|
||||
});
|
||||
ensureActive(signal);
|
||||
if (!prepared) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
},
|
||||
async (processed, batch) => {
|
||||
ensureActive(signal);
|
||||
const suggestedSensitive = batch.filter(
|
||||
(suggestion) => suggestion.assessment === "sensitive",
|
||||
).length;
|
||||
const suggestedNonSensitive = batch.filter(
|
||||
(suggestion) => suggestion.assessment === "non_sensitive",
|
||||
).length;
|
||||
const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length;
|
||||
const current = await this.repository.getSensitivityAnalysisRun(started.id);
|
||||
ensureActive(signal);
|
||||
if (!current) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
const progress = await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
|
||||
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
|
||||
unknown: current.unknown + unknown,
|
||||
});
|
||||
if (!progress) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
processedSensitive += suggestedSensitive;
|
||||
processedNonSensitive += suggestedNonSensitive;
|
||||
ensureActive(signal);
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
started.id,
|
||||
"info",
|
||||
`Assessed ${processed} of ${progress.total} columns locally.`,
|
||||
);
|
||||
ensureActive(signal);
|
||||
},
|
||||
);
|
||||
ensureActive(signal);
|
||||
const suggestedSensitive = suggestions.filter(
|
||||
(suggestion) => suggestion.assessment === "sensitive",
|
||||
).length;
|
||||
const suggestedNonSensitive = suggestions.filter(
|
||||
(suggestion) => suggestion.assessment === "non_sensitive",
|
||||
).length;
|
||||
const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length;
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
started.id,
|
||||
"info",
|
||||
`Local sensitivity analysis completed for ${suggestions.length} column${
|
||||
suggestions.length === 1 ? "" : "s"
|
||||
}.`,
|
||||
);
|
||||
ensureActive(signal);
|
||||
const completed = await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
status: "completed",
|
||||
total: suggestions.length,
|
||||
suggestedSensitive,
|
||||
suggestedNonSensitive,
|
||||
unknown,
|
||||
finishedAt: new Date().toISOString(),
|
||||
errorSummary: null,
|
||||
});
|
||||
ensureActive(signal);
|
||||
if (!completed) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
return { suggestions, run: completed };
|
||||
} catch (error) {
|
||||
const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError;
|
||||
const message = interrupted ? deadlineMessage : failedMessage;
|
||||
await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
status: interrupted ? "interrupted" : "failed",
|
||||
...(interrupted ? {
|
||||
total: preparedTotal,
|
||||
suggestedSensitive: processedSensitive,
|
||||
suggestedNonSensitive: processedNonSensitive,
|
||||
unknown: Math.max(0, preparedTotal - processedSensitive - processedNonSensitive),
|
||||
} : {}),
|
||||
finishedAt: new Date().toISOString(),
|
||||
errorSummary: message,
|
||||
}).catch(() => undefined);
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
started.id,
|
||||
interrupted ? "warning" : "error",
|
||||
message,
|
||||
).catch(() => undefined);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
import type {
|
||||
SensitivityClassifier,
|
||||
SensitivityColumnAssessment,
|
||||
SensitivityEvidence,
|
||||
SensitivityNerBudget,
|
||||
} from "./sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogColumn,
|
||||
CatalogRepository,
|
||||
CatalogTable,
|
||||
SensitivityAnalysisScope,
|
||||
} from "./types.js";
|
||||
|
||||
export type { SensitivityAnalysisScope } from "./types.js";
|
||||
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1";
|
||||
|
||||
interface SelectedColumn {
|
||||
table: CatalogTable;
|
||||
column: CatalogColumn;
|
||||
}
|
||||
|
||||
export interface SensitivityReviewItem {
|
||||
columnId: string;
|
||||
tableId: string;
|
||||
tableName: string;
|
||||
columnName: string;
|
||||
version: number;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
assessment: SensitivityColumnAssessment["assessment"];
|
||||
evidence: readonly SensitivityEvidence[];
|
||||
observedValues: number;
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisTargetNotFoundError extends Error {
|
||||
constructor(readonly target: "database" | "table" | "column") {
|
||||
super(`${target} not found`);
|
||||
this.name = "SensitivityAnalysisTargetNotFoundError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisDuplicateTargetIdsError extends Error {
|
||||
constructor() {
|
||||
super("sensitivity analysis target IDs must be unique");
|
||||
this.name = "SensitivityAnalysisDuplicateTargetIdsError";
|
||||
}
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisInterruptedError extends Error {
|
||||
constructor() {
|
||||
super("sensitivity analysis deadline exceeded");
|
||||
this.name = "SensitivityAnalysisInterruptedError";
|
||||
}
|
||||
}
|
||||
|
||||
function ensureActive(signal: AbortSignal): void {
|
||||
if (signal.aborted) throw new SensitivityAnalysisInterruptedError();
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisNoEligibleColumnsError extends Error {
|
||||
constructor(readonly scope: SensitivityAnalysisScope) {
|
||||
super("selected scope has no catalog columns");
|
||||
this.name = "SensitivityAnalysisNoEligibleColumnsError";
|
||||
}
|
||||
}
|
||||
|
||||
/** Selection and table orchestration around the single SensitivityClassifier decision module. */
|
||||
export class SensitivityAnalysisService {
|
||||
constructor(
|
||||
private readonly repository: CatalogRepository,
|
||||
private readonly classifier: SensitivityClassifier,
|
||||
private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {},
|
||||
) {}
|
||||
|
||||
private async selectColumns(
|
||||
databaseId: string,
|
||||
scope: SensitivityAnalysisScope,
|
||||
targetIds: readonly string[],
|
||||
signal: AbortSignal,
|
||||
): Promise<readonly SelectedColumn[]> {
|
||||
ensureActive(signal);
|
||||
if (new Set(targetIds).size !== targetIds.length) {
|
||||
throw new SensitivityAnalysisDuplicateTargetIdsError();
|
||||
}
|
||||
const tables = await this.repository.listTables(databaseId);
|
||||
ensureActive(signal);
|
||||
const tableIds = new Set(targetIds);
|
||||
const selectedTables = scope === "selected_tables"
|
||||
? tables.filter((table) => tableIds.has(table.id))
|
||||
: tables;
|
||||
if (scope === "selected_tables" && selectedTables.length !== targetIds.length) {
|
||||
throw new SensitivityAnalysisTargetNotFoundError("table");
|
||||
}
|
||||
const columns = (await Promise.all(selectedTables.map(async (table) => (
|
||||
(await this.repository.listColumns(databaseId, table.id)).map((column) => ({ table, column }))
|
||||
)))).flat();
|
||||
ensureActive(signal);
|
||||
const columnIds = new Set(targetIds);
|
||||
const selectedColumns = scope === "selected_columns"
|
||||
? columns.filter(({ column }) => columnIds.has(column.id))
|
||||
: columns;
|
||||
if (scope === "selected_columns" && selectedColumns.length !== targetIds.length) {
|
||||
throw new SensitivityAnalysisTargetNotFoundError("column");
|
||||
}
|
||||
if (selectedColumns.length === 0) {
|
||||
throw new SensitivityAnalysisNoEligibleColumnsError(scope);
|
||||
}
|
||||
return selectedColumns;
|
||||
}
|
||||
|
||||
async analyze(
|
||||
databaseId: string,
|
||||
scope: SensitivityAnalysisScope,
|
||||
targetIds: readonly string[],
|
||||
signal: AbortSignal,
|
||||
onPrepared?: (total: number) => void | Promise<void>,
|
||||
onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise<void>,
|
||||
): Promise<readonly SensitivityReviewItem[]> {
|
||||
const now = this.options.now ?? Date.now;
|
||||
const deadline = now() + (this.options.runBudgetMs ?? 60_000);
|
||||
const configuredNerBudget = this.options.nerBudgetMs ?? 10_000;
|
||||
const nerBudget: SensitivityNerBudget = {
|
||||
remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0
|
||||
? configuredNerBudget
|
||||
: 10_000,
|
||||
};
|
||||
ensureActive(signal);
|
||||
const database = await this.repository.get(databaseId);
|
||||
ensureActive(signal);
|
||||
if (!database) throw new SensitivityAnalysisTargetNotFoundError("database");
|
||||
const selected = await this.selectColumns(databaseId, scope, targetIds, signal);
|
||||
await onPrepared?.(selected.length);
|
||||
ensureActive(signal);
|
||||
const byTable = new Map<string, SelectedColumn[]>();
|
||||
for (const item of selected) {
|
||||
const items = byTable.get(item.table.id) ?? [];
|
||||
items.push(item);
|
||||
byTable.set(item.table.id, items);
|
||||
}
|
||||
const suggestions: SensitivityReviewItem[] = [];
|
||||
for (const items of byTable.values()) {
|
||||
ensureActive(signal);
|
||||
const first = items[0]!;
|
||||
const assessments = await this.classifier.assessTable({
|
||||
database,
|
||||
table: first.table,
|
||||
columns: items.map(({ column }) => column),
|
||||
}, signal, deadline, nerBudget);
|
||||
ensureActive(signal);
|
||||
const assessmentById = new Map(assessments.map((assessment) => [
|
||||
assessment.columnId,
|
||||
assessment,
|
||||
]));
|
||||
const batch = items.map(({ table, column }) => {
|
||||
const assessment = assessmentById.get(column.id)!;
|
||||
return {
|
||||
columnId: column.id,
|
||||
tableId: table.id,
|
||||
tableName: table.name,
|
||||
columnName: column.name,
|
||||
version: column.version,
|
||||
currentSensitive: column.sensitive,
|
||||
sensitive: assessment.proposedSensitive,
|
||||
assessment: assessment.assessment,
|
||||
evidence: assessment.evidence,
|
||||
observedValues: assessment.observedValues,
|
||||
};
|
||||
});
|
||||
suggestions.push(...batch);
|
||||
await onProgress?.(suggestions.length, batch);
|
||||
ensureActive(signal);
|
||||
}
|
||||
return suggestions;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,441 @@
|
||||
import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js";
|
||||
import { findPhoneNumbersInText } from "libphonenumber-js/max";
|
||||
import validator from "validator";
|
||||
|
||||
export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown";
|
||||
|
||||
export interface SensitivityEvidence {
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage";
|
||||
ruleId: string;
|
||||
label?: string;
|
||||
confidence?: number;
|
||||
}
|
||||
|
||||
export interface SensitivityValueObservation {
|
||||
columnId: string;
|
||||
value: string | null;
|
||||
characterLength: number | null;
|
||||
}
|
||||
|
||||
export interface SensitivityScanCoverage {
|
||||
kind: "complete" | "sampled" | "unavailable";
|
||||
observedRows: number;
|
||||
}
|
||||
|
||||
export interface SensitivityTableScan {
|
||||
batches: readonly (readonly SensitivityValueObservation[])[];
|
||||
coverage: SensitivityScanCoverage;
|
||||
}
|
||||
|
||||
export interface SensitivityScanRequest {
|
||||
database: WorkspaceDatabase;
|
||||
table: CatalogTable;
|
||||
columns: readonly CatalogColumn[];
|
||||
fullScanBudgetMs: number;
|
||||
deadline: number;
|
||||
}
|
||||
|
||||
export interface SensitivityValueSource {
|
||||
scanTable(
|
||||
request: SensitivityScanRequest,
|
||||
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityScanCoverage>;
|
||||
}
|
||||
|
||||
export interface LocalNerCandidate {
|
||||
columnId: string;
|
||||
text: string;
|
||||
}
|
||||
|
||||
export interface LocalNerEvidence {
|
||||
columnId: string;
|
||||
label: string;
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export interface SensitivityNerBudget {
|
||||
remainingMs: number;
|
||||
}
|
||||
|
||||
/** Optional local detector. It returns evidence only; it never decides a column assessment. */
|
||||
export interface LocalNerDetector {
|
||||
warmup?(): Promise<void>;
|
||||
isReady?(): boolean;
|
||||
detect(
|
||||
candidates: readonly LocalNerCandidate[],
|
||||
signal: AbortSignal,
|
||||
deadline: number,
|
||||
): Promise<readonly LocalNerEvidence[]>;
|
||||
close?(): Promise<void>;
|
||||
}
|
||||
|
||||
export interface SensitivityColumnAssessment {
|
||||
columnId: string;
|
||||
assessment: SensitivityAssessment;
|
||||
proposedSensitive: boolean;
|
||||
evidence: readonly SensitivityEvidence[];
|
||||
observedValues: number;
|
||||
}
|
||||
|
||||
export interface SensitivityTableTarget {
|
||||
database: WorkspaceDatabase;
|
||||
table: CatalogTable;
|
||||
columns: readonly CatalogColumn[];
|
||||
}
|
||||
|
||||
const EMAIL = /(?<![\p{L}\p{N}._%+-])[\p{L}\p{N}._%+-]+@[\p{L}\p{N}.-]+\.[\p{L}]{2,63}(?![\p{L}\p{N}._%+-])/giu;
|
||||
const DIRECT_IDENTIFIER_NAMES = new Set([
|
||||
"address", "birth_date", "codice_fiscale", "date_of_birth", "dob", "email", "e_mail",
|
||||
"bic", "first_name", "fiscal_code", "full_name", "iban", "indirizzo", "last_name", "mobile",
|
||||
"nome", "passport", "phone", "surname", "swift", "swift_code", "tax_id", "telefono",
|
||||
]);
|
||||
const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_key|pwd|secret|token)(?:_|$)/u;
|
||||
const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u;
|
||||
const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu;
|
||||
const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu;
|
||||
const MAX_NER_CANDIDATES_PER_REQUEST = 128;
|
||||
|
||||
function normalizedName(value: string): string {
|
||||
return value.normalize("NFKD")
|
||||
.replace(/[\u0300-\u036f]/g, "")
|
||||
.replace(/([a-z0-9])([A-Z])/g, "$1_$2")
|
||||
.toLocaleLowerCase("en-US")
|
||||
.replace(/[^a-z0-9]+/g, "_")
|
||||
.replace(/^_+|_+$/g, "");
|
||||
}
|
||||
|
||||
function boundedCount(value: number | undefined, fallback: number, maximum: number): number {
|
||||
return value === undefined || !Number.isSafeInteger(value)
|
||||
? fallback
|
||||
: Math.max(1, Math.min(value, maximum));
|
||||
}
|
||||
|
||||
function metadataEvidence(column: CatalogColumn): SensitivityEvidence | undefined {
|
||||
const ruleId = sensitiveNameRule(column.name);
|
||||
return ruleId ? { kind: "metadata", ruleId } : undefined;
|
||||
}
|
||||
|
||||
function sensitiveNameRule(value: string): string | undefined {
|
||||
const name = normalizedName(value);
|
||||
if (DIRECT_IDENTIFIER_NAMES.has(name)) {
|
||||
return "metadata.direct_identifier";
|
||||
}
|
||||
if (CREDENTIAL_NAME.test(name)) {
|
||||
return "metadata.credential";
|
||||
}
|
||||
if (HEALTH_NAME.test(name)) {
|
||||
return "metadata.health";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const ITALIAN_FISCAL_CODE = /(?<![A-Z0-9])[A-Z]{6}[0-9LMNPQRSTUV]{2}[ABCDEHLMPRST][0-9LMNPQRSTUV]{2}[A-Z][0-9LMNPQRSTUV]{3}[A-Z](?![A-Z0-9])/giu;
|
||||
const FISCAL_ODD: Record<string, number> = {
|
||||
"0": 1, "1": 0, "2": 5, "3": 7, "4": 9, "5": 13, "6": 15, "7": 17, "8": 19, "9": 21,
|
||||
A: 1, B: 0, C: 5, D: 7, E: 9, F: 13, G: 15, H: 17, I: 19, J: 21,
|
||||
K: 2, L: 4, M: 18, N: 20, O: 11, P: 3, Q: 6, R: 8, S: 12, T: 14,
|
||||
U: 16, V: 10, W: 22, X: 25, Y: 24, Z: 23,
|
||||
};
|
||||
|
||||
function validItalianFiscalCode(candidate: string): boolean {
|
||||
const value = candidate.toUpperCase();
|
||||
if (value.length !== 16) return false;
|
||||
let sum = 0;
|
||||
for (let index = 0; index < 15; index += 1) {
|
||||
const character = value[index]!;
|
||||
if (index % 2 === 0) sum += FISCAL_ODD[character] ?? -1000;
|
||||
else sum += /\d/u.test(character) ? Number(character) : character.charCodeAt(0) - 65;
|
||||
}
|
||||
return String.fromCharCode(65 + (sum % 26)) === value[15];
|
||||
}
|
||||
|
||||
function validIban(candidate: string): boolean {
|
||||
const value = candidate.replace(/\s+/gu, "").toUpperCase();
|
||||
if (!/^[A-Z]{2}\d{2}[A-Z0-9]{11,30}$/u.test(value)) return false;
|
||||
const rearranged = value.slice(4) + value.slice(0, 4);
|
||||
let remainder = 0;
|
||||
for (const character of rearranged) {
|
||||
const digits = /\d/u.test(character) ? character : String(character.charCodeAt(0) - 55);
|
||||
for (const digit of digits) remainder = (remainder * 10 + Number(digit)) % 97;
|
||||
}
|
||||
return remainder === 1;
|
||||
}
|
||||
|
||||
function validPaymentCard(candidate: string): boolean {
|
||||
const digits = candidate.replace(/[ -]/gu, "");
|
||||
if (!/^\d{13,19}$/u.test(digits) || /^(\d)\1+$/u.test(digits)) return false;
|
||||
let sum = 0;
|
||||
let double = false;
|
||||
for (let index = digits.length - 1; index >= 0; index -= 1) {
|
||||
let digit = Number(digits[index]);
|
||||
if (double) {
|
||||
digit *= 2;
|
||||
if (digit > 9) digit -= 9;
|
||||
}
|
||||
sum += digit;
|
||||
double = !double;
|
||||
}
|
||||
return sum % 10 === 0;
|
||||
}
|
||||
|
||||
function jsonHasSensitiveKey(value: string): boolean {
|
||||
const trimmed = value.trim();
|
||||
if (!(trimmed.startsWith("{") || trimmed.startsWith("["))) return false;
|
||||
try {
|
||||
const pending: Array<{ value: unknown; depth: number }> = [{ value: JSON.parse(trimmed), depth: 0 }];
|
||||
let visited = 0;
|
||||
while (pending.length > 0 && visited < 1_000) {
|
||||
const item = pending.pop()!;
|
||||
visited += 1;
|
||||
if (item.depth > 8 || item.value === null || typeof item.value !== "object") continue;
|
||||
if (Array.isArray(item.value)) {
|
||||
for (const child of item.value) pending.push({ value: child, depth: item.depth + 1 });
|
||||
continue;
|
||||
}
|
||||
for (const [key, child] of Object.entries(item.value)) {
|
||||
if (sensitiveNameRule(key)) return true;
|
||||
pending.push({ value: child, depth: item.depth + 1 });
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function contentEvidence(value: string): SensitivityEvidence | undefined {
|
||||
if (/-----BEGIN (?:[A-Z0-9]+ )?PRIVATE KEY-----/u.test(value)) {
|
||||
return { kind: "content", ruleId: "credential.private_key" };
|
||||
}
|
||||
if (/(?:^|[^A-Z0-9])AKIA[A-Z0-9]{16}(?![A-Z0-9])/u.test(value)
|
||||
|| /(?:^|[^A-Za-z0-9_])gh[pousr]_[A-Za-z0-9_]{30,}(?![A-Za-z0-9_])/u.test(value)
|
||||
|| /(?:^|[^A-Za-z0-9_-])eyJ[A-Za-z0-9_-]{5,}\.[A-Za-z0-9_-]{5,}\.[A-Za-z0-9_-]{5,}(?![A-Za-z0-9_-])/u.test(value)) {
|
||||
return { kind: "content", ruleId: "credential.access_key" };
|
||||
}
|
||||
if (/(?:^|[^\p{L}\p{N}_])(?:api[_ -]?key|access[_ -]?token|password|passwd|pwd|secret)\s*[:=]\s*[^\s,;]{4,}/iu.test(value)) {
|
||||
return { kind: "content", ruleId: "credential.key_value" };
|
||||
}
|
||||
if (CLINICAL_TERM.test(value)) return { kind: "content", ruleId: "health.clinical_term" };
|
||||
for (const match of value.matchAll(EMAIL)) {
|
||||
if (validator.isEmail(match[0])) return { kind: "content", ruleId: "pii.email" };
|
||||
}
|
||||
for (const match of value.matchAll(ITALIAN_FISCAL_CODE)) {
|
||||
if (validItalianFiscalCode(match[0])) {
|
||||
return { kind: "content", ruleId: "pii.italian_fiscal_code" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/\b(?:passaporto|passport)(?:\s+(?:numero|number|n\.?))?\s*[:#-]?\s*([A-Z0-9]{9})\b/giu)) {
|
||||
if (validator.isPassportNumber(match[1]!, "IT")) {
|
||||
return { kind: "content", ruleId: "pii.passport_number" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/\bC[A-Z]\d{5}[A-Z]{2}\b/giu)) {
|
||||
if (validator.isIdentityCard(match[0], "IT")) {
|
||||
return { kind: "content", ruleId: "pii.identity_card" };
|
||||
}
|
||||
}
|
||||
if (/\b(?:patente(?:\s+di\s+guida)?|driving\s+licen[cs]e)(?:\s+(?:numero|number|n\.?))?\s*[:#-]?\s*[A-Z0-9]{8,12}\b/iu.test(value)) {
|
||||
return { kind: "content", ruleId: "pii.drivers_license_number" };
|
||||
}
|
||||
for (const match of value.matchAll(/(?<![A-Z0-9])[A-Z]{2}\d{2}(?:\s?[A-Z0-9]){11,30}(?![A-Z0-9])/giu)) {
|
||||
if (validIban(match[0])) return { kind: "content", ruleId: "financial.iban" };
|
||||
}
|
||||
for (const match of value.matchAll(/(?<!\d)(?:\d[ -]?){13,19}(?!\d)/gu)) {
|
||||
if (validPaymentCard(match[0])) {
|
||||
return { kind: "content", ruleId: "financial.payment_card" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/(?<![A-Z0-9])[A-Z]{6}[A-Z0-9]{2}(?:[A-Z0-9]{3})?(?![A-Z0-9])/giu)) {
|
||||
const before = value.slice(Math.max(0, (match.index ?? 0) - 24), match.index ?? 0);
|
||||
if (/\b(?:bic|swift)\s*[:=-]?\s*$/iu.test(before) && validator.isBIC(match[0])) {
|
||||
return { kind: "content", ruleId: "financial.bic" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/(?<!\d)(?:IT[ .-]?)?\d{11}(?!\d)/giu)) {
|
||||
const candidate = match[0].replace(/[ .-]/gu, "");
|
||||
if (validator.isVAT(candidate.replace(/^IT/iu, ""), "IT")) {
|
||||
return { kind: "content", ruleId: "pii.italian_vat" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/(?<![A-F0-9])(?:[A-F0-9]{2}[:-]){5}[A-F0-9]{2}(?![A-F0-9])/giu)) {
|
||||
if (validator.isMACAddress(match[0])) {
|
||||
return { kind: "content", ruleId: "network.mac_address" };
|
||||
}
|
||||
}
|
||||
for (const match of value.matchAll(/(?<![A-F0-9:.])[A-F0-9:.]{3,45}(?![A-F0-9:.])/giu)) {
|
||||
if (validator.isIP(match[0])) return { kind: "content", ruleId: "network.ip_address" };
|
||||
}
|
||||
for (const match of value.matchAll(/(?<![A-F0-9-])[0-9A-F]{8}-[0-9A-F]{4}-[1-8][0-9A-F]{3}-[89AB][0-9A-F]{3}-[0-9A-F]{12}(?![A-F0-9-])/giu)) {
|
||||
if (validator.isUUID(match[0])) return { kind: "content", ruleId: "pii.uuid" };
|
||||
}
|
||||
for (const match of value.matchAll(/\b(?:https?|ftp):\/\/[^\s<>"']+/giu)) {
|
||||
const candidate = match[0].replace(/[.,;:!?\])}]+$/u, "");
|
||||
if (validator.isURL(candidate, { require_protocol: true })) {
|
||||
return { kind: "content", ruleId: "network.url" };
|
||||
}
|
||||
}
|
||||
if (findPhoneNumbersInText(value, "IT").some((match) => match.number.isValid())) {
|
||||
return { kind: "content", ruleId: "pii.phone_number" };
|
||||
}
|
||||
if (jsonHasSensitiveKey(value)) {
|
||||
return { kind: "content", ruleId: "pii.json_sensitive_key" };
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Sole decision module for local column-level sensitivity assessments. */
|
||||
export class SensitivityClassifier {
|
||||
constructor(
|
||||
private readonly values: SensitivityValueSource,
|
||||
private readonly detector?: LocalNerDetector,
|
||||
private readonly options: {
|
||||
fullScanBudgetMs?: number;
|
||||
runBudgetMs?: number;
|
||||
nerConfidenceThreshold?: number;
|
||||
maxNerValuesPerColumn?: number;
|
||||
maxNerCandidatesPerTable?: number;
|
||||
now?: () => number;
|
||||
} = {},
|
||||
) {}
|
||||
|
||||
async assessTable(
|
||||
target: SensitivityTableTarget,
|
||||
signal: AbortSignal,
|
||||
runDeadline?: number,
|
||||
nerBudget?: SensitivityNerBudget,
|
||||
): Promise<readonly SensitivityColumnAssessment[]> {
|
||||
const now = this.options.now ?? Date.now;
|
||||
const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000);
|
||||
const evidence = new Map(target.columns.map((column) => {
|
||||
const match = metadataEvidence(column);
|
||||
return [column.id, match ? [match] : [] as SensitivityEvidence[]];
|
||||
}));
|
||||
const observed = new Map(target.columns.map((column) => [column.id, 0]));
|
||||
const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]]));
|
||||
const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8);
|
||||
const unsupported = new Set(target.columns
|
||||
.filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType))
|
||||
.map((column) => column.id));
|
||||
const scannableColumns = target.columns.filter((column) => (
|
||||
!unsupported.has(column.id) && evidence.get(column.id)!.length === 0
|
||||
));
|
||||
let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 };
|
||||
if (scannableColumns.length > 0 && now() < deadline) {
|
||||
try {
|
||||
coverage = await this.values.scanTable({
|
||||
...target,
|
||||
columns: scannableColumns,
|
||||
fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000,
|
||||
deadline,
|
||||
}, (batch) => {
|
||||
for (const item of batch) {
|
||||
if (!evidence.has(item.columnId) || item.value === null) continue;
|
||||
observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1);
|
||||
const matches = evidence.get(item.columnId)!;
|
||||
if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) {
|
||||
matches.push({ kind: "length", ruleId: "text.over_500_characters" });
|
||||
} else if (matches.length === 0) {
|
||||
const match = contentEvidence(item.value);
|
||||
if (match) matches.push(match);
|
||||
else {
|
||||
const candidates = nerCandidates.get(item.columnId)!;
|
||||
if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) {
|
||||
candidates.push(item.value);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}, signal);
|
||||
} catch (error) {
|
||||
if (!(error instanceof CatalogConnectorError)) throw error;
|
||||
}
|
||||
}
|
||||
|
||||
if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
|
||||
&& now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) {
|
||||
const candidates: LocalNerCandidate[] = [];
|
||||
const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
|
||||
candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
|
||||
for (const column of target.columns) {
|
||||
if (evidence.get(column.id)!.length > 0) continue;
|
||||
const text = nerCandidates.get(column.id)![valueIndex];
|
||||
if (text === undefined) continue;
|
||||
candidates.push({ columnId: column.id, text });
|
||||
if (candidates.length >= maxCandidates) break candidateSelection;
|
||||
}
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
|
||||
const nerStartedAt = now();
|
||||
const allowedNerMs = nerBudget
|
||||
? Math.max(0, nerBudget.remainingMs)
|
||||
: Math.max(0, deadline - nerStartedAt);
|
||||
const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs);
|
||||
try {
|
||||
for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) {
|
||||
if (signal.aborted || now() >= nerDeadline) break;
|
||||
try {
|
||||
const detected = await this.detector.detect(
|
||||
candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST),
|
||||
signal,
|
||||
nerDeadline,
|
||||
);
|
||||
for (const item of detected) {
|
||||
const matches = evidence.get(item.columnId);
|
||||
if (!matches || matches.length > 0 || !Number.isFinite(item.confidence)
|
||||
|| item.confidence < threshold || item.confidence > 1) continue;
|
||||
const label = normalizedName(item.label).slice(0, 80);
|
||||
if (!label) continue;
|
||||
matches.push({
|
||||
kind: "ner",
|
||||
ruleId: "ner.entity",
|
||||
label,
|
||||
confidence: item.confidence,
|
||||
});
|
||||
}
|
||||
} catch {
|
||||
// NER is optional: deterministic findings and scan coverage remain authoritative.
|
||||
break;
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
if (nerBudget) {
|
||||
const elapsedMs = Math.max(1, now() - nerStartedAt);
|
||||
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return target.columns.map((column) => {
|
||||
const matches = evidence.get(column.id)!;
|
||||
const count = observed.get(column.id) ?? 0;
|
||||
const assessment: SensitivityAssessment = matches.length > 0
|
||||
? "sensitive"
|
||||
: unsupported.has(column.id) || count === 0 || coverage.kind !== "complete"
|
||||
? "unknown"
|
||||
: "non_sensitive";
|
||||
return {
|
||||
columnId: column.id,
|
||||
assessment,
|
||||
proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive",
|
||||
evidence: matches.length > 0
|
||||
? matches
|
||||
: assessment === "unknown"
|
||||
? [{
|
||||
kind: "coverage",
|
||||
ruleId: unsupported.has(column.id)
|
||||
? "coverage.unsupported_type"
|
||||
: coverage.kind === "unavailable"
|
||||
? "coverage.unavailable"
|
||||
: count === 0
|
||||
? "coverage.no_values"
|
||||
: "coverage.incomplete",
|
||||
}]
|
||||
: [],
|
||||
observedValues: count,
|
||||
};
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
import { dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { loadConfig } from "../config.js";
|
||||
import { WorkspaceSecretStore } from "../workspaces/secret-store.js";
|
||||
import { PythonLocalNerDetector } from "./local-ner-detector.js";
|
||||
import { ConcreteCatalogPostgresAccess } from "./postgres-access.js";
|
||||
import { createCatalogRepository } from "./repository.js";
|
||||
import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js";
|
||||
import { SensitivityClassifier } from "./sensitivity-classifier.js";
|
||||
import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js";
|
||||
|
||||
const WORKSPACE_ID = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/u;
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const workspaceId = process.argv[2];
|
||||
if (!workspaceId || !WORKSPACE_ID.test(workspaceId)) {
|
||||
process.stderr.write("Usage: sensitivity-shadow <workspace-id>\n");
|
||||
process.exitCode = 2;
|
||||
return;
|
||||
}
|
||||
|
||||
let detector: PythonLocalNerDetector | undefined;
|
||||
let stage = "configuration";
|
||||
try {
|
||||
const config = loadConfig(process.env);
|
||||
stage = "catalog";
|
||||
const repository = createCatalogRepository(config.catalogDatabase);
|
||||
if (!(await repository.available())) throw new Error("catalog unavailable");
|
||||
const database = await repository.getByWorkspace(workspaceId);
|
||||
if (!database) throw new Error("database unavailable");
|
||||
stage = "source";
|
||||
const secretStore = new WorkspaceSecretStore({
|
||||
root: config.workspaceSecretStoreRoot,
|
||||
runtimeRoot: config.workspaceSecretRuntimeRoot,
|
||||
installationId: config.workspaceRegistry.installationId,
|
||||
});
|
||||
const access = new ConcreteCatalogPostgresAccess(secretStore, {
|
||||
connectTimeoutMs: config.workspaceDiagnosticTimeoutMs,
|
||||
});
|
||||
const source = new ConcreteSensitivityValueSource(access, secretStore);
|
||||
if (config.sensitivityNer) {
|
||||
const workerScript = config.sensitivityNer.workerScript
|
||||
?? fileURLToPath(new URL("../../python/sensitivity_ner_worker.py", import.meta.url));
|
||||
detector = new PythonLocalNerDetector({
|
||||
pythonExecutable: config.sensitivityNer.pythonExecutable,
|
||||
workerScript,
|
||||
modelPath: config.sensitivityNer.modelPath,
|
||||
cwd: dirname(workerScript),
|
||||
threads: config.sensitivityNer.threads,
|
||||
});
|
||||
try {
|
||||
await detector.warmup();
|
||||
} catch {
|
||||
await detector.close();
|
||||
detector = undefined;
|
||||
}
|
||||
}
|
||||
const startedAt = Date.now();
|
||||
stage = "analysis";
|
||||
const suggestions = await new SensitivityAnalysisService(
|
||||
repository,
|
||||
new SensitivityClassifier(source, detector),
|
||||
).analyze(database.id, "all", [], AbortSignal.timeout(65_000));
|
||||
const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 };
|
||||
const rules = new Map<string, number>();
|
||||
for (const suggestion of suggestions) {
|
||||
if (suggestion.assessment === "sensitive") assessments.sensitive += 1;
|
||||
else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1;
|
||||
else assessments.unknown += 1;
|
||||
for (const evidence of suggestion.evidence) {
|
||||
rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1);
|
||||
}
|
||||
}
|
||||
process.stdout.write(`${JSON.stringify({
|
||||
ok: true,
|
||||
policyVersion: "sensitivity-v1",
|
||||
nerEnabled: detector !== undefined,
|
||||
total: suggestions.length,
|
||||
assessments,
|
||||
rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))),
|
||||
elapsedMs: Date.now() - startedAt,
|
||||
})}\n`);
|
||||
} catch {
|
||||
process.stdout.write(`${JSON.stringify({
|
||||
ok: false,
|
||||
code: `sensitivity_shadow_${stage}_failed`,
|
||||
})}\n`);
|
||||
process.exitCode = 1;
|
||||
} finally {
|
||||
await detector?.close();
|
||||
}
|
||||
}
|
||||
|
||||
await main();
|
||||
@@ -0,0 +1,259 @@
|
||||
import { readFile } from "node:fs/promises";
|
||||
import type { WorkspaceSecretStore } from "../workspaces/secret-store.js";
|
||||
import { CATALOG_SECRET_IDS } from "./secrets.js";
|
||||
import type { CatalogPostgresAccess } from "./postgres-access.js";
|
||||
import type {
|
||||
SensitivityScanCoverage,
|
||||
SensitivityScanRequest,
|
||||
SensitivityValueObservation,
|
||||
SensitivityValueSource,
|
||||
} from "./sensitivity-classifier.js";
|
||||
import { CatalogConnectorError } from "./types.js";
|
||||
|
||||
const MAX_VALUE_CHARACTERS = 501;
|
||||
const DEFAULT_BATCH_ROWS = 200;
|
||||
const DEFAULT_SAMPLE_ROWS = 200;
|
||||
|
||||
function quoteIdentifier(identifier: string): string {
|
||||
return `"${identifier.replaceAll('"', '""')}"`;
|
||||
}
|
||||
|
||||
function projections(request: SensitivityScanRequest): string {
|
||||
return request.columns.flatMap((column, index) => {
|
||||
const identifier = quoteIdentifier(column.name);
|
||||
return [
|
||||
`LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`,
|
||||
`CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`,
|
||||
];
|
||||
}).join(", ");
|
||||
}
|
||||
|
||||
function observations(
|
||||
request: SensitivityScanRequest,
|
||||
rows: readonly Record<string, unknown>[],
|
||||
): SensitivityValueObservation[] {
|
||||
return rows.flatMap((row) => request.columns.map((column, index) => {
|
||||
const sourceValue = row[`__value_${index}`];
|
||||
const sourceLength = row[`__length_${index}`];
|
||||
const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue);
|
||||
const parsedLength = sourceLength === null || sourceLength === undefined
|
||||
? null
|
||||
: Number(sourceLength);
|
||||
return {
|
||||
columnId: column.id,
|
||||
value,
|
||||
characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0
|
||||
? parsedLength
|
||||
: value?.length ?? null,
|
||||
};
|
||||
}));
|
||||
}
|
||||
|
||||
function cancelled(error: unknown): boolean {
|
||||
return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014");
|
||||
}
|
||||
|
||||
interface SensitivityValueSourceOptions {
|
||||
now?: () => number;
|
||||
batchRows?: number;
|
||||
sampleRows?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a
|
||||
* sensitivity decision.
|
||||
*/
|
||||
export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
private readonly now: () => number;
|
||||
private readonly batchRows: number;
|
||||
private readonly sampleRows: number;
|
||||
|
||||
constructor(
|
||||
private readonly access: CatalogPostgresAccess,
|
||||
private readonly secretStore?: Pick<WorkspaceSecretStore, "materialize">,
|
||||
options: SensitivityValueSourceOptions = {},
|
||||
) {
|
||||
this.now = options.now ?? Date.now;
|
||||
this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS;
|
||||
this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS;
|
||||
}
|
||||
|
||||
async scanTable(
|
||||
request: SensitivityScanRequest,
|
||||
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityScanCoverage> {
|
||||
if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 };
|
||||
if (request.database.binding.transport === "rest_api") {
|
||||
return await this.scanRest(request, consume, signal);
|
||||
}
|
||||
return await this.scanPostgres(request, consume, signal);
|
||||
}
|
||||
|
||||
private async scanPostgres(
|
||||
request: SensitivityScanRequest,
|
||||
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityScanCoverage> {
|
||||
const client = await this.access.connect(request.database, signal);
|
||||
let transactionOpen = false;
|
||||
const startedAt = this.now();
|
||||
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
|
||||
let observedRows = 0;
|
||||
let cursorOpen = false;
|
||||
try {
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows: 0 };
|
||||
}
|
||||
await client.query("BEGIN TRANSACTION READ ONLY", []);
|
||||
transactionOpen = true;
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`,
|
||||
]);
|
||||
await client.query("SAVEPOINT sensitivity_full_scan", []);
|
||||
const cursor = [
|
||||
"DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR",
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
].join(" ");
|
||||
await client.query(cursor, []);
|
||||
cursorOpen = true;
|
||||
while (!signal.aborted && this.now() < fullDeadline) {
|
||||
let rows: Array<Record<string, unknown>>;
|
||||
try {
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`,
|
||||
]);
|
||||
rows = (await client.query(
|
||||
`FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`,
|
||||
[],
|
||||
)).rows;
|
||||
} catch (error) {
|
||||
if (!cancelled(error)) throw error;
|
||||
await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []);
|
||||
cursorOpen = false;
|
||||
break;
|
||||
}
|
||||
if (rows.length > 0) {
|
||||
observedRows += rows.length;
|
||||
await consume(observations(request, rows));
|
||||
}
|
||||
if (rows.length < this.batchRows) {
|
||||
return { kind: "complete", observedRows };
|
||||
}
|
||||
}
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows };
|
||||
}
|
||||
if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []);
|
||||
await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []);
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(request.deadline - this.now()))}ms`,
|
||||
]);
|
||||
const sampleSql = [
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
|
||||
"LIMIT $1",
|
||||
].join(" ");
|
||||
const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows;
|
||||
observedRows += sampledRows.length;
|
||||
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
|
||||
return { kind: "sampled", observedRows };
|
||||
} catch (error) {
|
||||
if (error instanceof CatalogConnectorError) throw error;
|
||||
throw new CatalogConnectorError("Sensitivity source scan failed");
|
||||
} finally {
|
||||
if (transactionOpen) await client.query("ROLLBACK", []).catch(() => undefined);
|
||||
await client.end().catch(() => undefined);
|
||||
}
|
||||
}
|
||||
|
||||
private async scanRest(
|
||||
request: SensitivityScanRequest,
|
||||
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityScanCoverage> {
|
||||
if (!this.secretStore) throw new CatalogConnectorError("REST sensitivity scanning is not configured");
|
||||
const auth = request.database.binding.restAuth ?? "bearer";
|
||||
const materialized = this.secretStore.materialize(
|
||||
request.database.workspaceId,
|
||||
auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey],
|
||||
);
|
||||
const startedAt = this.now();
|
||||
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
|
||||
let observedRows = 0;
|
||||
try {
|
||||
const headers: Record<string, string> = { "content-type": "application/json" };
|
||||
if (auth !== "none") {
|
||||
const credentialFile = materialized.files.get(CATALOG_SECRET_IDS.apiKey);
|
||||
if (!credentialFile) throw new CatalogConnectorError("REST API key is not configured");
|
||||
const credential = (await readFile(credentialFile, "utf8")).trim();
|
||||
if (auth === "bearer") headers.authorization = `Bearer ${credential}`;
|
||||
else headers["x-api-key"] = credential;
|
||||
}
|
||||
const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, "");
|
||||
if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete");
|
||||
const runQuery = async (sql: string, deadline: number): Promise<Array<Record<string, unknown>>> => {
|
||||
const response = await fetch(`${baseUrl}/rpc/run_query`, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({ query_text: sql }),
|
||||
signal: AbortSignal.any([
|
||||
signal,
|
||||
AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))),
|
||||
]),
|
||||
});
|
||||
if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
|
||||
const body: unknown = await response.json();
|
||||
if (!Array.isArray(body)
|
||||
|| body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) {
|
||||
throw new CatalogConnectorError("REST sensitivity source response is invalid");
|
||||
}
|
||||
return body as Array<Record<string, unknown>>;
|
||||
};
|
||||
|
||||
let offset = 0;
|
||||
const baseSelect = [
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
].join(" ");
|
||||
while (!signal.aborted) {
|
||||
let rows: Array<Record<string, unknown>>;
|
||||
try {
|
||||
rows = await runQuery(
|
||||
`${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`,
|
||||
fullDeadline,
|
||||
);
|
||||
} catch (error) {
|
||||
if (signal.aborted || this.now() < fullDeadline) throw error;
|
||||
break;
|
||||
}
|
||||
observedRows += rows.length;
|
||||
if (rows.length > 0) await consume(observations(request, rows));
|
||||
if (rows.length < this.batchRows) {
|
||||
return { kind: offset === 0 ? "complete" : "sampled", observedRows };
|
||||
}
|
||||
offset += rows.length;
|
||||
if (this.now() >= fullDeadline) break;
|
||||
}
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows };
|
||||
}
|
||||
const sampleSql = [
|
||||
baseSelect,
|
||||
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
|
||||
`LIMIT ${this.sampleRows}`,
|
||||
].join(" ");
|
||||
const sampledRows = await runQuery(sampleSql, request.deadline);
|
||||
observedRows += sampledRows.length;
|
||||
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
|
||||
return { kind: "sampled", observedRows };
|
||||
} catch (error) {
|
||||
if (error instanceof CatalogConnectorError) throw error;
|
||||
throw new CatalogConnectorError("REST sensitivity source scan failed");
|
||||
} finally {
|
||||
materialized.release();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -266,18 +266,21 @@ export interface DescriptionGenerationEvent {
|
||||
createdAt: string;
|
||||
}
|
||||
|
||||
export type SensitiveDataSuggestionScope = "all" | "selected_tables" | "selected_columns";
|
||||
export type SensitiveDataSuggestionStatus = "running" | "completed" | "failed" | "interrupted";
|
||||
export type SensitivityAnalysisScope = "all" | "selected_tables" | "selected_columns";
|
||||
export type SensitivityAnalysisStatus = "running" | "completed" | "failed" | "interrupted";
|
||||
|
||||
export interface SensitiveDataSuggestionRun {
|
||||
export interface SensitivityAnalysisRun {
|
||||
id: string;
|
||||
databaseId: string;
|
||||
scope: SensitiveDataSuggestionScope;
|
||||
modelId: string;
|
||||
status: SensitiveDataSuggestionStatus;
|
||||
scope: SensitivityAnalysisScope;
|
||||
engine: "llm" | "local";
|
||||
modelId: string | null;
|
||||
policyVersion: string | null;
|
||||
status: SensitivityAnalysisStatus;
|
||||
total: number;
|
||||
suggestedSensitive: number;
|
||||
suggestedNonSensitive: number;
|
||||
unknown: number;
|
||||
inputTokens: number;
|
||||
cacheReadTokens: number;
|
||||
outputTokens: number;
|
||||
@@ -288,11 +291,12 @@ export interface SensitiveDataSuggestionRun {
|
||||
errorSummary: string | null;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestionRunUpdate {
|
||||
status?: SensitiveDataSuggestionStatus;
|
||||
export interface SensitivityAnalysisRunUpdate {
|
||||
status?: SensitivityAnalysisStatus;
|
||||
total?: number;
|
||||
suggestedSensitive?: number;
|
||||
suggestedNonSensitive?: number;
|
||||
unknown?: number;
|
||||
finishedAt?: string | null;
|
||||
errorSummary?: string | null;
|
||||
inputTokens?: number;
|
||||
@@ -300,7 +304,7 @@ export interface SensitiveDataSuggestionRunUpdate {
|
||||
outputTokens?: number;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestionEvent {
|
||||
export interface SensitivityAnalysisEvent {
|
||||
runId: string;
|
||||
sequence: number;
|
||||
level: "info" | "warning" | "error";
|
||||
@@ -491,29 +495,29 @@ export interface CatalogRepository {
|
||||
runId: string,
|
||||
afterSequence?: number,
|
||||
): Promise<DescriptionGenerationEvent[]>;
|
||||
createSensitiveDataSuggestionRun(
|
||||
createSensitivityAnalysisRun(
|
||||
databaseId: string,
|
||||
scope: SensitiveDataSuggestionScope,
|
||||
modelId: string,
|
||||
): Promise<SensitiveDataSuggestionRun>;
|
||||
getSensitiveDataSuggestionRun(runId: string): Promise<SensitiveDataSuggestionRun | undefined>;
|
||||
listSensitiveDataSuggestionRuns(limit?: number): Promise<SensitiveDataSuggestionRun[]>;
|
||||
interruptActiveSensitiveDataSuggestionRuns(
|
||||
scope: SensitivityAnalysisScope,
|
||||
origin: { engine: "llm"; modelId: string } | { engine: "local"; policyVersion: string },
|
||||
): Promise<SensitivityAnalysisRun>;
|
||||
getSensitivityAnalysisRun(runId: string): Promise<SensitivityAnalysisRun | undefined>;
|
||||
listSensitivityAnalysisRuns(limit?: number): Promise<SensitivityAnalysisRun[]>;
|
||||
interruptActiveSensitivityAnalysisRuns(
|
||||
errorSummary: string,
|
||||
): Promise<SensitiveDataSuggestionRun[]>;
|
||||
updateSensitiveDataSuggestionRun(
|
||||
): Promise<SensitivityAnalysisRun[]>;
|
||||
updateSensitivityAnalysisRun(
|
||||
runId: string,
|
||||
update: SensitiveDataSuggestionRunUpdate,
|
||||
): Promise<SensitiveDataSuggestionRun | undefined>;
|
||||
appendSensitiveDataSuggestionEvent(
|
||||
update: SensitivityAnalysisRunUpdate,
|
||||
): Promise<SensitivityAnalysisRun | undefined>;
|
||||
appendSensitivityAnalysisEvent(
|
||||
runId: string,
|
||||
level: SensitiveDataSuggestionEvent["level"],
|
||||
level: SensitivityAnalysisEvent["level"],
|
||||
message: string,
|
||||
): Promise<SensitiveDataSuggestionEvent>;
|
||||
listSensitiveDataSuggestionEvents(
|
||||
): Promise<SensitivityAnalysisEvent>;
|
||||
listSensitivityAnalysisEvents(
|
||||
runId: string,
|
||||
afterSequence?: number,
|
||||
): Promise<SensitiveDataSuggestionEvent[]>;
|
||||
): Promise<SensitivityAnalysisEvent[]>;
|
||||
listRelationships(databaseId: string): Promise<CatalogPhysicalRelationship[]>;
|
||||
listLogicalRelationships(databaseId: string): Promise<CatalogLogicalRelationship[]>;
|
||||
getLogicalRelationshipContext(databaseId: string): Promise<CatalogLogicalRelationshipContext | undefined>;
|
||||
|
||||
@@ -33,6 +33,12 @@ export interface AppConfig {
|
||||
secretsFile?: string;
|
||||
installationConfigFile?: string;
|
||||
modelCatalogFile?: string;
|
||||
sensitivityNer?: {
|
||||
pythonExecutable: string;
|
||||
modelPath: string;
|
||||
workerScript?: string;
|
||||
threads: number;
|
||||
};
|
||||
piAuthFile?: string;
|
||||
secretFiles: Readonly<Record<string, string | undefined>>;
|
||||
modelApiKeyFile?: string;
|
||||
@@ -354,6 +360,35 @@ export function loadConfig(
|
||||
|| modelCatalogFile.includes("\0")
|
||||
|| !path.isAbsolute(modelCatalogFile)
|
||||
)) throw new Error("runtime model catalog configuration is invalid");
|
||||
const sensitivityNerModelPath = env.THT_SENSITIVITY_NER_MODEL_PATH;
|
||||
const sensitivityNerPython = env.THT_SENSITIVITY_NER_PYTHON;
|
||||
const sensitivityNerWorker = env.THT_SENSITIVITY_NER_WORKER;
|
||||
for (const [value, label] of [
|
||||
[sensitivityNerModelPath, "model path"],
|
||||
[sensitivityNerPython, "Python executable"],
|
||||
[sensitivityNerWorker, "worker path"],
|
||||
] as const) {
|
||||
if (value !== undefined && (
|
||||
value.length === 0 || value.trim() !== value || value.includes("\0") || !path.isAbsolute(value)
|
||||
)) throw new Error(`sensitivity NER ${label} configuration is invalid`);
|
||||
}
|
||||
if (sensitivityNerModelPath === undefined && (
|
||||
sensitivityNerPython !== undefined
|
||||
|| sensitivityNerWorker !== undefined
|
||||
|| env.THT_SENSITIVITY_NER_THREADS !== undefined
|
||||
)) throw new Error("sensitivity NER settings require a model path");
|
||||
const sensitivityNerThreads = Number(env.THT_SENSITIVITY_NER_THREADS ?? 2);
|
||||
if (!Number.isSafeInteger(sensitivityNerThreads) || sensitivityNerThreads < 1 || sensitivityNerThreads > 8) {
|
||||
throw new Error("sensitivity NER thread configuration is invalid");
|
||||
}
|
||||
const sensitivityNer = sensitivityNerModelPath === undefined
|
||||
? undefined
|
||||
: {
|
||||
modelPath: sensitivityNerModelPath,
|
||||
pythonExecutable: sensitivityNerPython ?? "/opt/sensitivity-ner/bin/python",
|
||||
...(sensitivityNerWorker ? { workerScript: sensitivityNerWorker } : {}),
|
||||
threads: sensitivityNerThreads,
|
||||
};
|
||||
const piAuthFile = env.THT_PI_AUTH_FILE;
|
||||
if (piAuthFile !== undefined && (
|
||||
piAuthFile.trim() !== piAuthFile || piAuthFile.length === 0 || piAuthFile.includes("\0")
|
||||
@@ -442,6 +477,7 @@ export function loadConfig(
|
||||
secretsFile,
|
||||
installationConfigFile,
|
||||
modelCatalogFile,
|
||||
sensitivityNer,
|
||||
piAuthFile,
|
||||
secretFiles,
|
||||
modelApiKeyFile,
|
||||
|
||||
@@ -11,38 +11,35 @@ import {
|
||||
type DescriptionGenerationWorker,
|
||||
} from "../catalog/description-generation-worker.js";
|
||||
import { MetadataGenerationModelUnavailableError } from "../catalog/metadata-generation-models.js";
|
||||
import { ModelCompletionProviderError } from "../catalog/model-completer.js";
|
||||
import {
|
||||
SensitiveDataSuggestionDuplicateTargetIdsError,
|
||||
SensitiveDataSuggestionInvalidResponseError,
|
||||
SensitiveDataSuggestionNoEligibleColumnsError,
|
||||
SensitiveDataSuggestionPayloadTooLargeError,
|
||||
SensitiveDataSuggestionTargetNotFoundError,
|
||||
} from "../catalog/sensitive-data-suggester.js";
|
||||
import type { SensitiveDataSuggestionRunner } from "../catalog/sensitive-data-suggestion-runner.js";
|
||||
SensitivityAnalysisDuplicateTargetIdsError,
|
||||
SensitivityAnalysisInterruptedError,
|
||||
SensitivityAnalysisNoEligibleColumnsError,
|
||||
SensitivityAnalysisTargetNotFoundError,
|
||||
} from "../catalog/sensitivity-analysis-service.js";
|
||||
import type { SensitivityAnalysisRunner } from "../catalog/sensitivity-analysis-runner.js";
|
||||
import {
|
||||
CatalogOperationInProgressError,
|
||||
CatalogConnectorError,
|
||||
CatalogUnavailableError,
|
||||
DescriptionGenerationRunActiveError,
|
||||
type CatalogRepository,
|
||||
type DescriptionGenerationEvent,
|
||||
type DescriptionGenerationRun,
|
||||
type SensitiveDataSuggestionEvent,
|
||||
type SensitiveDataSuggestionRun,
|
||||
type SensitivityAnalysisEvent,
|
||||
type SensitivityAnalysisRun,
|
||||
} from "../catalog/types.js";
|
||||
|
||||
const idSchema = z.uuid();
|
||||
const modelIdSchema = z.string().regex(/^[a-z][a-z0-9._-]{0,63}\/[A-Za-z0-9][A-Za-z0-9._:-]{0,255}$/);
|
||||
const selectedTargetIdsSchema = z.array(idSchema).min(1);
|
||||
const suggestionSchema = z.discriminatedUnion("scope", [
|
||||
z.object({ modelId: modelIdSchema, scope: z.literal("all") }).strict(),
|
||||
z.object({ scope: z.literal("all") }).strict(),
|
||||
z.object({
|
||||
modelId: modelIdSchema,
|
||||
scope: z.literal("selected_tables"),
|
||||
targetIds: selectedTargetIdsSchema,
|
||||
}).strict(),
|
||||
z.object({
|
||||
modelId: modelIdSchema,
|
||||
scope: z.literal("selected_columns"),
|
||||
targetIds: selectedTargetIdsSchema,
|
||||
}).strict(),
|
||||
@@ -112,7 +109,7 @@ function publicRun(run: DescriptionGenerationRun) {
|
||||
};
|
||||
}
|
||||
|
||||
function publicSensitiveDataSuggestionEvent(event: SensitiveDataSuggestionEvent) {
|
||||
function publicSensitivityAnalysisEvent(event: SensitivityAnalysisEvent) {
|
||||
return {
|
||||
runId: event.runId,
|
||||
sequence: event.sequence,
|
||||
@@ -122,16 +119,19 @@ function publicSensitiveDataSuggestionEvent(event: SensitiveDataSuggestionEvent)
|
||||
};
|
||||
}
|
||||
|
||||
function publicSensitiveDataSuggestionRun(run: SensitiveDataSuggestionRun) {
|
||||
function publicSensitivityAnalysisRun(run: SensitivityAnalysisRun) {
|
||||
return {
|
||||
id: run.id,
|
||||
databaseId: run.databaseId,
|
||||
scope: run.scope,
|
||||
engine: run.engine,
|
||||
modelId: run.modelId,
|
||||
policyVersion: run.policyVersion,
|
||||
status: run.status,
|
||||
total: run.total,
|
||||
suggestedSensitive: run.suggestedSensitive,
|
||||
suggestedNonSensitive: run.suggestedNonSensitive,
|
||||
unknown: run.unknown,
|
||||
inputTokens: run.inputTokens,
|
||||
cacheReadTokens: run.cacheReadTokens,
|
||||
outputTokens: run.outputTokens,
|
||||
@@ -232,16 +232,10 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
|
||||
if (error instanceof CatalogUnavailableError) {
|
||||
return reply.code(503).send({
|
||||
code: "catalog_unavailable",
|
||||
message: "The database catalog is unavailable, so no sensitive-field suggestions were prepared.",
|
||||
message: "The database catalog is unavailable, so no sensitivity assessments were prepared.",
|
||||
});
|
||||
}
|
||||
if (error instanceof MetadataGenerationModelUnavailableError) {
|
||||
return reply.code(409).send({
|
||||
code: "metadata_generation_model_unavailable",
|
||||
message: "The selected metadata-generation model is unavailable.",
|
||||
});
|
||||
}
|
||||
if (error instanceof SensitiveDataSuggestionTargetNotFoundError) {
|
||||
if (error instanceof SensitivityAnalysisTargetNotFoundError) {
|
||||
const code = error.target === "database"
|
||||
? "database_not_found"
|
||||
: error.target === "table"
|
||||
@@ -254,45 +248,65 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
|
||||
: "One or more selected Catalog Columns were not found in this database.";
|
||||
return reply.code(404).send({ code, message });
|
||||
}
|
||||
if (error instanceof SensitiveDataSuggestionDuplicateTargetIdsError) {
|
||||
if (error instanceof SensitivityAnalysisDuplicateTargetIdsError) {
|
||||
return reply.code(400).send({
|
||||
code: "sensitive_data_suggestion_target_ids_duplicate",
|
||||
message: "Each selected table or column must appear only once.",
|
||||
});
|
||||
}
|
||||
if (error instanceof SensitiveDataSuggestionNoEligibleColumnsError) {
|
||||
if (error instanceof SensitivityAnalysisNoEligibleColumnsError) {
|
||||
return reply.code(409).send({
|
||||
code: "sensitive_data_suggestion_no_columns",
|
||||
message: "The selected scope contains no Catalog Columns to classify.",
|
||||
message: "The selected scope contains no Catalog Columns to assess.",
|
||||
});
|
||||
}
|
||||
if (error instanceof SensitiveDataSuggestionPayloadTooLargeError) {
|
||||
return reply.code(413).send({
|
||||
code: "sensitive_data_suggestion_payload_too_large",
|
||||
message: "The selected structural metadata cannot be divided into safe LLM requests.",
|
||||
if (error instanceof SensitivityAnalysisInterruptedError) {
|
||||
return reply.code(504).send({
|
||||
code: "sensitivity_analysis_timeout",
|
||||
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
});
|
||||
}
|
||||
if (error instanceof SensitiveDataSuggestionInvalidResponseError) {
|
||||
if (error instanceof CatalogConnectorError) {
|
||||
return reply.code(502).send({
|
||||
code: "sensitive_data_suggestion_invalid_response",
|
||||
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
|
||||
});
|
||||
}
|
||||
if (error instanceof ModelCompletionProviderError) {
|
||||
return reply.code(502).send({
|
||||
code: "sensitive_data_suggestion_provider_unavailable",
|
||||
message: "The selected LLM service could not complete the request. No suggestions were applied.",
|
||||
code: "sensitivity_source_unavailable",
|
||||
message: "The source values could not be inspected safely. No assessments were applied.",
|
||||
});
|
||||
}
|
||||
if (error instanceof z.ZodError) {
|
||||
return reply.code(400).send({
|
||||
code: "sensitive_data_suggestion_request_invalid",
|
||||
message: "Choose a database, one or more tables, or one or more columns to classify.",
|
||||
message: "Choose a database, one or more tables, or one or more columns to assess.",
|
||||
});
|
||||
}
|
||||
return reply.code(500).send({
|
||||
code: "sensitive_data_suggestion_failed",
|
||||
message: "Sensitive-field suggestions failed before review. No changes were applied.",
|
||||
message: "Local sensitivity analysis failed before review. No changes were applied.",
|
||||
});
|
||||
}
|
||||
|
||||
function untilAborted<T>(operation: Promise<T>, signal: AbortSignal): Promise<T> {
|
||||
if (signal.aborted) {
|
||||
void operation.catch(() => undefined);
|
||||
return Promise.reject(new SensitivityAnalysisInterruptedError());
|
||||
}
|
||||
return new Promise<T>((resolve, reject) => {
|
||||
const abort = () => reject(new SensitivityAnalysisInterruptedError());
|
||||
signal.addEventListener("abort", abort, { once: true });
|
||||
if (signal.aborted) {
|
||||
void operation.catch(() => undefined);
|
||||
abort();
|
||||
return;
|
||||
}
|
||||
operation.then(
|
||||
(value) => {
|
||||
signal.removeEventListener("abort", abort);
|
||||
resolve(value);
|
||||
},
|
||||
(error: unknown) => {
|
||||
signal.removeEventListener("abort", abort);
|
||||
reject(error);
|
||||
},
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -300,18 +314,18 @@ function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) {
|
||||
if (error instanceof CatalogUnavailableError) {
|
||||
return reply.code(503).send({
|
||||
code: "catalog_unavailable",
|
||||
message: "Sensitive Data Suggestion history is unavailable because the database catalog is unavailable.",
|
||||
message: "Sensitivity Analysis history is unavailable because the database catalog is unavailable.",
|
||||
});
|
||||
}
|
||||
if (error instanceof z.ZodError) {
|
||||
return reply.code(400).send({
|
||||
code: "sensitive_data_suggestion_history_request_invalid",
|
||||
message: "Sensitive Data Suggestion history parameters are invalid.",
|
||||
message: "Sensitivity Analysis history parameters are invalid.",
|
||||
});
|
||||
}
|
||||
return reply.code(500).send({
|
||||
code: "sensitive_data_suggestion_history_failed",
|
||||
message: "Sensitive Data Suggestion history could not be loaded.",
|
||||
message: "Sensitivity Analysis history could not be loaded.",
|
||||
});
|
||||
}
|
||||
|
||||
@@ -320,7 +334,7 @@ export function catalogDescriptionGenerationRoutes(
|
||||
deps: {
|
||||
repository: CatalogRepository;
|
||||
worker: DescriptionGenerationWorker;
|
||||
sensitiveDataSuggestionRunner: SensitiveDataSuggestionRunner;
|
||||
sensitivityAnalysisRunner: SensitivityAnalysisRunner;
|
||||
},
|
||||
): void {
|
||||
app.post("/catalog/databases/:databaseId/sensitive-data-suggestions", async (request, reply) => {
|
||||
@@ -328,16 +342,16 @@ export function catalogDescriptionGenerationRoutes(
|
||||
try {
|
||||
const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId);
|
||||
const input = suggestionSchema.parse(request.body);
|
||||
const result = await deps.sensitiveDataSuggestionRunner.run(
|
||||
const signal = AbortSignal.timeout(60_000);
|
||||
const result = await untilAborted(deps.sensitivityAnalysisRunner.run(
|
||||
databaseId,
|
||||
input.modelId,
|
||||
input.scope,
|
||||
"targetIds" in input ? input.targetIds : [],
|
||||
new AbortController().signal,
|
||||
);
|
||||
signal,
|
||||
), signal);
|
||||
return {
|
||||
suggestions: result.suggestions,
|
||||
run: publicSensitiveDataSuggestionRun(result.run),
|
||||
run: publicSensitivityAnalysisRun(result.run),
|
||||
};
|
||||
} catch (error) {
|
||||
return safeSuggestionError(reply, error);
|
||||
@@ -348,8 +362,8 @@ export function catalogDescriptionGenerationRoutes(
|
||||
if (!manage(request, reply)) return reply;
|
||||
try {
|
||||
const { limit } = historyQuerySchema.parse(request.query);
|
||||
return (await deps.repository.listSensitiveDataSuggestionRuns(limit))
|
||||
.map(publicSensitiveDataSuggestionRun);
|
||||
return (await deps.repository.listSensitivityAnalysisRuns(limit))
|
||||
.map(publicSensitivityAnalysisRun);
|
||||
} catch (error) {
|
||||
return safeSuggestionHistoryError(reply, error);
|
||||
}
|
||||
@@ -359,12 +373,12 @@ export function catalogDescriptionGenerationRoutes(
|
||||
if (!manage(request, reply)) return reply;
|
||||
try {
|
||||
const runId = idSchema.parse((request.params as { runId?: unknown }).runId);
|
||||
const run = await deps.repository.getSensitiveDataSuggestionRun(runId);
|
||||
const run = await deps.repository.getSensitivityAnalysisRun(runId);
|
||||
if (!run) return reply.code(404).send({
|
||||
code: "sensitive_data_suggestion_run_not_found",
|
||||
message: "Sensitive Data Suggestion Run was not found.",
|
||||
message: "Sensitivity Analysis Run was not found.",
|
||||
});
|
||||
return publicSensitiveDataSuggestionRun(run);
|
||||
return publicSensitivityAnalysisRun(run);
|
||||
} catch (error) {
|
||||
return safeSuggestionHistoryError(reply, error);
|
||||
}
|
||||
@@ -375,14 +389,14 @@ export function catalogDescriptionGenerationRoutes(
|
||||
try {
|
||||
const runId = idSchema.parse((request.params as { runId?: unknown }).runId);
|
||||
const { after } = eventQuerySchema.parse(request.query);
|
||||
if (!(await deps.repository.getSensitiveDataSuggestionRun(runId))) {
|
||||
if (!(await deps.repository.getSensitivityAnalysisRun(runId))) {
|
||||
return reply.code(404).send({
|
||||
code: "sensitive_data_suggestion_run_not_found",
|
||||
message: "Sensitive Data Suggestion Run was not found.",
|
||||
message: "Sensitivity Analysis Run was not found.",
|
||||
});
|
||||
}
|
||||
return (await deps.repository.listSensitiveDataSuggestionEvents(runId, after))
|
||||
.map(publicSensitiveDataSuggestionEvent);
|
||||
return (await deps.repository.listSensitivityAnalysisEvents(runId, after))
|
||||
.map(publicSensitivityAnalysisEvent);
|
||||
} catch (error) {
|
||||
return safeSuggestionHistoryError(reply, error);
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ import {
|
||||
type ModelCompletionRequest,
|
||||
} from "../src/catalog/model-completer.js";
|
||||
import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js";
|
||||
import type { SensitivityValueSource } from "../src/catalog/sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogDatabaseClient,
|
||||
CatalogPostgresAccess,
|
||||
@@ -69,6 +70,16 @@ async function setup(
|
||||
sample: vi.fn(async () => []),
|
||||
},
|
||||
catalogPostgresAccess?: CatalogPostgresAccess,
|
||||
sensitivityValueSource: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, consume) => {
|
||||
await consume(request.columns.map((column) => ({
|
||||
columnId: column.id,
|
||||
value: "ordinary",
|
||||
characterLength: 8,
|
||||
})));
|
||||
return { kind: "complete", observedRows: 1 };
|
||||
}),
|
||||
},
|
||||
) {
|
||||
const repository = new MemoryCatalogRepository();
|
||||
const database = await repository.create({
|
||||
@@ -115,10 +126,11 @@ async function setup(
|
||||
catalogOperationCoordinator: operations,
|
||||
metadataGenerationModels: models(),
|
||||
modelCompleter,
|
||||
sensitivityValueSource,
|
||||
...(descriptionSourceSampler ? { descriptionSourceSampler } : {}),
|
||||
...(catalogPostgresAccess ? { catalogPostgresAccess } : {}),
|
||||
});
|
||||
return { app, repository, database, table, column, operations };
|
||||
return { app, repository, database, table, column, operations, sensitivityValueSource };
|
||||
}
|
||||
|
||||
async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: string) {
|
||||
@@ -136,22 +148,15 @@ async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: strin
|
||||
throw new Error(`Description Generation Run ${runId} did not finish`);
|
||||
}
|
||||
|
||||
test("suggests sensitive flags from structural metadata without persisting them", async () => {
|
||||
const modelCompleter = {
|
||||
complete: vi.fn(async () => JSON.stringify({
|
||||
suggestions: [{ columnId: expect.any(String), sensitive: true }],
|
||||
})),
|
||||
};
|
||||
test("assesses sensitive flags locally without persisting them or calling an LLM", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, repository, database, table, column } = await setup(modelCompleter);
|
||||
modelCompleter.complete.mockResolvedValueOnce(JSON.stringify({
|
||||
suggestions: [{ columnId: column.id, sensitive: true }],
|
||||
}));
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
payload: { scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
@@ -160,11 +165,14 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
run: {
|
||||
databaseId: database.id,
|
||||
scope: "all",
|
||||
modelId: configuredModel.id,
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "completed",
|
||||
total: 1,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
errorSummary: null,
|
||||
},
|
||||
suggestions: [{
|
||||
@@ -175,6 +183,8 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
version: column.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
}],
|
||||
});
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
@@ -207,49 +217,56 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
runId: responseBody.run.id,
|
||||
sequence: 1,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation started.",
|
||||
message: "Local sensitivity analysis started.",
|
||||
},
|
||||
{
|
||||
runId: responseBody.run.id,
|
||||
sequence: 2,
|
||||
level: "info",
|
||||
message: "Classified 1 of 1 columns.",
|
||||
message: "Assessed 1 of 1 columns locally.",
|
||||
},
|
||||
{
|
||||
runId: responseBody.run.id,
|
||||
sequence: 3,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation completed for 1 column.",
|
||||
message: "Local sensitivity analysis completed for 1 column.",
|
||||
},
|
||||
]);
|
||||
|
||||
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
|
||||
const prompt = request.messages.map((message) => message.content).join("\n");
|
||||
expect(prompt).toContain("patients");
|
||||
expect(prompt).toContain("birth_date");
|
||||
expect(prompt).toContain("date");
|
||||
expect(prompt).not.toContain("Patient date of birth");
|
||||
expect(prompt).not.toContain("test-provider-secret");
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("limits sensitive-data suggestions to the selected tables or columns", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async (request) => {
|
||||
const payload = JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
||||
columns: Array<{ columnId: string; column: string }>;
|
||||
};
|
||||
return JSON.stringify({
|
||||
suggestions: payload.columns.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
sensitive: column.column.includes("name") || column.column.includes("note"),
|
||||
})),
|
||||
});
|
||||
}),
|
||||
};
|
||||
const { app, repository, database } = await setup(modelCompleter);
|
||||
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
|
||||
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(504);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitivity_analysis_timeout",
|
||||
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
});
|
||||
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
|
||||
} finally {
|
||||
timeout.mockRestore();
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("limits sensitivity analysis to the selected tables or columns", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, repository, database, sensitivityValueSource } = await setup(modelCompleter);
|
||||
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
||||
schemaVersion: 1,
|
||||
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
||||
@@ -281,7 +298,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: [visits.id, patients.id],
|
||||
},
|
||||
@@ -301,7 +317,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_columns",
|
||||
targetIds: [clinicalNote.id, status.id],
|
||||
},
|
||||
@@ -313,44 +328,41 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }),
|
||||
]));
|
||||
|
||||
const prompts = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => (
|
||||
JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
||||
columns: Array<{ columnId: string }>;
|
||||
}
|
||||
));
|
||||
expect(prompts[0]!.columns.map((column) => column.columnId).sort()).toEqual(
|
||||
[...patientColumns, ...visitColumns].map((column) => column.id).sort(),
|
||||
);
|
||||
expect(prompts[0]!.columns.map((column) => column.columnId)).not.toContain(billingColumns[0]!.id);
|
||||
expect(prompts[1]!.columns.map((column) => column.columnId).sort()).toEqual(
|
||||
[status.id, clinicalNote.id].sort(),
|
||||
const scannedColumnIds = vi.mocked(sensitivityValueSource.scanTable).mock.calls.flatMap(
|
||||
([request]) => request.columns.map((column) => column.id),
|
||||
);
|
||||
expect(scannedColumnIds).toEqual([status.id, status.id]);
|
||||
expect(scannedColumnIds).not.toContain(patientColumns.find(
|
||||
(column) => column.name === "patient_name",
|
||||
)!.id);
|
||||
expect(scannedColumnIds).not.toContain(clinicalNote.id);
|
||||
expect(scannedColumnIds).not.toContain(billingColumns[0]!.id);
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("explains invalid sensitive-data suggestion selections without calling the model", async () => {
|
||||
test("explains invalid sensitivity-analysis selections without reading source values", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, database, table } = await setup(modelCompleter);
|
||||
const { app, database, table, sensitivityValueSource } = await setup(modelCompleter);
|
||||
|
||||
try {
|
||||
const empty = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [] },
|
||||
payload: { scope: "selected_tables", targetIds: [] },
|
||||
});
|
||||
expect(empty.statusCode).toBe(400);
|
||||
expect(empty.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_request_invalid",
|
||||
message: "Choose a database, one or more tables, or one or more columns to classify.",
|
||||
message: "Choose a database, one or more tables, or one or more columns to assess.",
|
||||
});
|
||||
|
||||
const duplicate = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: [table.id, table.id],
|
||||
},
|
||||
@@ -365,7 +377,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: ["00000000-0000-4000-8000-000000000001"],
|
||||
},
|
||||
@@ -380,7 +391,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_columns",
|
||||
targetIds: ["00000000-0000-4000-8000-000000000002"],
|
||||
},
|
||||
@@ -391,205 +401,7 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
message: "One or more selected Catalog Columns were not found in this database.",
|
||||
});
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("batches sensitive-data suggestions for schemas larger than one helper message", async () => {
|
||||
const maxHelperMessageBytes = 64 * 1024;
|
||||
const seenColumnIds: string[] = [];
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async (request) => {
|
||||
const userMessage = request.messages.find((message) => message.role === "user")!;
|
||||
expect(Buffer.byteLength(userMessage.content, "utf8")).toBeLessThanOrEqual(maxHelperMessageBytes);
|
||||
const payload = JSON.parse(userMessage.content) as {
|
||||
columns: Array<{ columnId: string; column: string }>;
|
||||
};
|
||||
expect(payload.columns.length).toBeLessThanOrEqual(10);
|
||||
seenColumnIds.push(...payload.columns.map((column) => column.columnId));
|
||||
return JSON.stringify({
|
||||
suggestions: payload.columns.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
sensitive: column.column.endsWith("_private"),
|
||||
})),
|
||||
});
|
||||
}),
|
||||
};
|
||||
const { app, repository, database } = await setup(modelCompleter);
|
||||
const columnCount = 900;
|
||||
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
||||
schemaVersion: 1,
|
||||
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
||||
tables: [{ name: "wide_table", sourceComment: null }],
|
||||
columns: Array.from({ length: columnCount }, (_, index) => ({
|
||||
tableName: "wide_table",
|
||||
name: `field_${index.toString().padStart(4, "0")}${index % 10 === 0 ? "_private" : ""}`,
|
||||
ordinalPosition: index + 1,
|
||||
dataType: "character varying(255)",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
sourceComment: null,
|
||||
})),
|
||||
relationships: [],
|
||||
});
|
||||
const wideTable = (await repository.listTables(database.id)).find((table) => table.name === "wide_table")!;
|
||||
const expectedColumnIds = (await repository.listColumns(database.id, wideTable.id)).map((column) => column.id);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
const suggestions = response.json().suggestions as Array<{
|
||||
columnName: string;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
}>;
|
||||
expect(suggestions).toHaveLength(columnCount);
|
||||
expect(suggestions).toEqual(expect.arrayContaining([
|
||||
expect.objectContaining({ columnName: "field_0000_private", currentSensitive: false, sensitive: true }),
|
||||
expect.objectContaining({ columnName: "field_0001", currentSensitive: false, sensitive: false }),
|
||||
]));
|
||||
expect(vi.mocked(modelCompleter.complete).mock.calls.length).toBeGreaterThan(1);
|
||||
expect(seenColumnIds.slice().sort()).toEqual(expectedColumnIds.slice().sort());
|
||||
expect(new Set(seenColumnIds).size).toBe(columnCount);
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("retries one invalid sensitive-data classification before returning the review draft", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => "unused"),
|
||||
};
|
||||
const { app, database, column } = await setup(modelCompleter);
|
||||
vi.mocked(modelCompleter.complete)
|
||||
.mockResolvedValueOnce("not-json")
|
||||
.mockResolvedValueOnce(JSON.stringify({
|
||||
suggestions: [{ columnId: column.id, sensitive: true }],
|
||||
}));
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
expect(response.json().suggestions).toEqual([
|
||||
expect.objectContaining({ columnId: column.id, sensitive: true }),
|
||||
]);
|
||||
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test.each(["malformed", "incomplete", "duplicate"] as const)(
|
||||
"fails safely when sensitive-data suggestions are %s",
|
||||
async (kind) => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => "unused"),
|
||||
};
|
||||
const { app, repository, database, column } = await setup(modelCompleter);
|
||||
const rawResponse = kind === "malformed"
|
||||
? "RAW_PROVIDER_RESPONSE_DO_NOT_EXPOSE_{"
|
||||
: kind === "incomplete"
|
||||
? JSON.stringify({ suggestions: [] })
|
||||
: JSON.stringify({
|
||||
suggestions: [
|
||||
{ columnId: column.id, sensitive: true },
|
||||
{ columnId: column.id, sensitive: true },
|
||||
],
|
||||
});
|
||||
vi.mocked(modelCompleter.complete).mockResolvedValueOnce(rawResponse);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(502);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_invalid_response",
|
||||
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
|
||||
});
|
||||
expect(response.body).not.toContain(rawResponse);
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
.toMatchObject({ sensitive: false });
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
test("explains a sensitive-data suggestion provider failure without exposing provider details", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => {
|
||||
throw new ModelCompletionProviderError();
|
||||
}),
|
||||
};
|
||||
const { app, repository, database, column } = await setup(modelCompleter);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(502);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_provider_unavailable",
|
||||
message: "The selected LLM service could not complete the request. No suggestions were applied.",
|
||||
});
|
||||
expect(response.body).not.toContain("model completion failed");
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
.toMatchObject({ sensitive: false });
|
||||
|
||||
const history = await app.inject({
|
||||
method: "GET",
|
||||
url: "/catalog/sensitive-data-suggestion-runs",
|
||||
});
|
||||
expect(history.statusCode).toBe(200);
|
||||
const [failedRun] = history.json();
|
||||
expect(failedRun).toMatchObject({
|
||||
databaseId: database.id,
|
||||
status: "failed",
|
||||
total: 1,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
errorSummary: "Sensitive-field suggestion generation failed.",
|
||||
});
|
||||
|
||||
const events = await app.inject({
|
||||
method: "GET",
|
||||
url: `/catalog/sensitive-data-suggestion-runs/${failedRun.id}/events-list`,
|
||||
});
|
||||
expect(events.statusCode).toBe(200);
|
||||
expect(events.json()).toMatchObject([
|
||||
{
|
||||
runId: failedRun.id,
|
||||
sequence: 1,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation started.",
|
||||
},
|
||||
{
|
||||
runId: failedRun.id,
|
||||
sequence: 2,
|
||||
level: "error",
|
||||
message: "Sensitive-field suggestion generation failed.",
|
||||
},
|
||||
]);
|
||||
expect(events.body).not.toContain("model completion failed");
|
||||
expect(sensitivityValueSource.scanTable).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensiti
|
||||
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
|
||||
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
|
||||
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
|
||||
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
|
||||
import { KyselyCatalogRepository, type CatalogDatabase } from "../src/catalog/repository.js";
|
||||
import { loadConfig } from "../src/config.js";
|
||||
import type { WorkspaceRegistry } from "../src/workspaces/registry.js";
|
||||
@@ -52,6 +53,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
|
||||
await upSensitiveSuggestionRuns(db);
|
||||
await upAiTokenUsage(db);
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
const repository = new KyselyCatalogRepository(db);
|
||||
const database = await repository.create({
|
||||
workspaceId: "psd-clinical",
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { afterEach, expect, test, vi } from "vitest";
|
||||
import { PythonLocalNerDetector } from "../src/catalog/local-ner-detector.js";
|
||||
|
||||
const roots: string[] = [];
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllEnvs();
|
||||
for (const root of roots.splice(0)) rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test("keeps a CPU-only local worker warm and returns sanitized evidence", async () => {
|
||||
vi.stubEnv("THT_MODEL_API_KEY", "must-not-reach-worker");
|
||||
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-"));
|
||||
roots.push(root);
|
||||
const helper = join(root, "fake_ner_worker.py");
|
||||
writeFileSync(helper, `
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import sys
|
||||
|
||||
root = pathlib.Path.cwd()
|
||||
root.joinpath("runtime.json").write_text(json.dumps({
|
||||
"argv": sys.argv,
|
||||
"cuda": os.environ.get("CUDA_VISIBLE_DEVICES"),
|
||||
"hip": os.environ.get("HIP_VISIBLE_DEVICES"),
|
||||
"offline": os.environ.get("HF_HUB_OFFLINE"),
|
||||
"inherited_secret": os.environ.get("THT_MODEL_API_KEY"),
|
||||
"pid": os.getpid(),
|
||||
}), encoding="utf-8")
|
||||
print(json.dumps({"ready": True}), flush=True)
|
||||
for line in sys.stdin:
|
||||
request = json.loads(line)
|
||||
root.joinpath("request.json").write_text(json.dumps(request), encoding="utf-8")
|
||||
print(json.dumps({
|
||||
"id": request["id"],
|
||||
"ok": True,
|
||||
"evidence": [{
|
||||
"columnId": request["candidates"][0]["columnId"],
|
||||
"label": "person",
|
||||
"confidence": 0.93,
|
||||
}],
|
||||
}), flush=True)
|
||||
`, "utf8");
|
||||
const detector = new PythonLocalNerDetector({
|
||||
pythonExecutable: "python3",
|
||||
workerScript: helper,
|
||||
modelPath: join(root, "pinned-model"),
|
||||
cwd: root,
|
||||
threads: 2,
|
||||
startupTimeoutMs: 5_000,
|
||||
});
|
||||
const candidate = {
|
||||
columnId: "33333333-3333-4333-8333-333333333333",
|
||||
text: "Dimesso Mario Rossi",
|
||||
};
|
||||
|
||||
try {
|
||||
expect(detector.isReady()).toBe(false);
|
||||
await detector.warmup();
|
||||
expect(detector.isReady()).toBe(true);
|
||||
expect(existsSync(join(root, "request.json"))).toBe(false);
|
||||
|
||||
await expect(detector.detect(
|
||||
[candidate],
|
||||
new AbortController().signal,
|
||||
Date.now() + 5_000,
|
||||
)).resolves.toEqual([{
|
||||
columnId: candidate.columnId,
|
||||
label: "person",
|
||||
confidence: 0.93,
|
||||
}]);
|
||||
const firstRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
|
||||
expect(firstRuntime).toMatchObject({
|
||||
cuda: "",
|
||||
hip: "",
|
||||
offline: "1",
|
||||
inherited_secret: null,
|
||||
});
|
||||
expect(JSON.stringify(firstRuntime.argv)).not.toContain(candidate.text);
|
||||
expect(JSON.parse(readFileSync(join(root, "request.json"), "utf8")).candidates).toEqual([candidate]);
|
||||
|
||||
await detector.detect([candidate], new AbortController().signal, Date.now() + 5_000);
|
||||
const secondRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
|
||||
expect(secondRuntime.pid).toBe(firstRuntime.pid);
|
||||
} finally {
|
||||
await detector.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("bounds worker startup by the caller deadline", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-deadline-"));
|
||||
roots.push(root);
|
||||
const helper = join(root, "slow_ner_worker.py");
|
||||
writeFileSync(helper, `
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
|
||||
time.sleep(2)
|
||||
print(json.dumps({"ready": True}), flush=True)
|
||||
for line in sys.stdin:
|
||||
request = json.loads(line)
|
||||
print(json.dumps({"id": request["id"], "ok": True, "evidence": []}), flush=True)
|
||||
`, "utf8");
|
||||
const detector = new PythonLocalNerDetector({
|
||||
pythonExecutable: "python3",
|
||||
workerScript: helper,
|
||||
modelPath: join(root, "pinned-model"),
|
||||
cwd: root,
|
||||
startupTimeoutMs: 5_000,
|
||||
});
|
||||
const startedAt = Date.now();
|
||||
|
||||
try {
|
||||
await expect(detector.detect(
|
||||
[{
|
||||
columnId: "33333333-3333-4333-8333-333333333333",
|
||||
text: "Dimesso Mario Rossi",
|
||||
}],
|
||||
new AbortController().signal,
|
||||
startedAt + 50,
|
||||
)).rejects.toThrow("local NER is unavailable");
|
||||
expect(Date.now() - startedAt).toBeLessThan(1_000);
|
||||
} finally {
|
||||
await detector.close();
|
||||
}
|
||||
});
|
||||
@@ -16,6 +16,7 @@ import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_s
|
||||
import { up as upLogicalRelationships } from "../src/catalog/migrations/008_catalog_logical_relationships.js";
|
||||
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
|
||||
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
|
||||
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
|
||||
|
||||
const dockerAvailable = spawnSync("docker", ["info"], { stdio: "ignore" }).status === 0;
|
||||
|
||||
@@ -47,12 +48,23 @@ test.skipIf(!dockerAvailable)("PostgreSQL migration enforces one database per wo
|
||||
modelId: "openai-mini", language: "en", status: "completed", total: 1,
|
||||
processed: 1, generated: 1,
|
||||
}).execute();
|
||||
const historicalSuggestionRunId = randomUUID();
|
||||
await db.insertInto("sensitiveDataSuggestionRuns").values({
|
||||
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
|
||||
id: historicalSuggestionRunId, databaseId: historicalDatabaseId, scope: "all",
|
||||
modelId: "openai-mini", status: "completed", total: 1,
|
||||
suggestedSensitive: 1,
|
||||
}).execute();
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
await expect(db.selectFrom("sensitiveDataSuggestionRuns")
|
||||
.select(["engine", "modelId", "policyVersion", "unknown"])
|
||||
.where("id", "=", historicalSuggestionRunId)
|
||||
.executeTakeFirstOrThrow()).resolves.toMatchObject({
|
||||
engine: "llm",
|
||||
modelId: "openai-mini",
|
||||
policyVersion: null,
|
||||
unknown: 0,
|
||||
});
|
||||
await expect(db.insertInto("descriptionGenerationRuns").values({
|
||||
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
|
||||
modelId: "openai/gpt-5-mini", language: "en", status: "completed", total: 1,
|
||||
@@ -418,6 +430,7 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
await upSensitiveSuggestionRuns(db);
|
||||
await upAiTokenUsage(db);
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
const repository = new KyselyCatalogRepository(db);
|
||||
const firstDatabase = await repository.create({
|
||||
workspaceId: "generation-one",
|
||||
@@ -586,10 +599,10 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
]);
|
||||
expect(await repository.getActiveDescriptionGenerationRun()).toBeUndefined();
|
||||
|
||||
const suggestionRun = await repository.createSensitiveDataSuggestionRun(
|
||||
const suggestionRun = await repository.createSensitivityAnalysisRun(
|
||||
firstDatabase.id,
|
||||
"selected_columns",
|
||||
"openai/gpt-4.1-mini",
|
||||
{ engine: "local", policyVersion: "sensitivity-v1" },
|
||||
);
|
||||
expect(suggestionRun).toMatchObject({
|
||||
databaseId: firstDatabase.id,
|
||||
@@ -597,43 +610,49 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
startedAt: expect.any(String),
|
||||
});
|
||||
await repository.appendSensitiveDataSuggestionEvent(
|
||||
await repository.appendSensitivityAnalysisEvent(
|
||||
suggestionRun.id,
|
||||
"info",
|
||||
"Sensitive-field suggestion generation started.",
|
||||
);
|
||||
await repository.appendSensitiveDataSuggestionEvent(
|
||||
await repository.appendSensitivityAnalysisEvent(
|
||||
suggestionRun.id,
|
||||
"info",
|
||||
"Sensitive-field suggestion generation completed for 2 columns.",
|
||||
);
|
||||
expect(await repository.updateSensitiveDataSuggestionRun(suggestionRun.id, {
|
||||
expect(await repository.updateSensitivityAnalysisRun(suggestionRun.id, {
|
||||
status: "completed",
|
||||
total: 2,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 1,
|
||||
finishedAt: new Date().toISOString(),
|
||||
})).toMatchObject({
|
||||
status: "completed",
|
||||
total: 2,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 1,
|
||||
});
|
||||
expect(await repository.listSensitiveDataSuggestionEvents(suggestionRun.id, 1)).toEqual([
|
||||
expect(await repository.listSensitivityAnalysisEvents(suggestionRun.id, 1)).toEqual([
|
||||
expect.objectContaining({ sequence: 2, level: "info" }),
|
||||
]);
|
||||
expect((await repository.listSensitiveDataSuggestionRuns(1))[0]).toMatchObject({
|
||||
expect((await repository.listSensitivityAnalysisRuns(1))[0]).toMatchObject({
|
||||
id: suggestionRun.id,
|
||||
});
|
||||
|
||||
const interruptedSuggestionRun = await repository.createSensitiveDataSuggestionRun(
|
||||
const interruptedSuggestionRun = await repository.createSensitivityAnalysisRun(
|
||||
secondDatabase.id,
|
||||
"all",
|
||||
"openai/gpt-4.1-mini",
|
||||
{ engine: "local", policyVersion: "sensitivity-v1" },
|
||||
);
|
||||
expect(await repository.interruptActiveSensitiveDataSuggestionRuns(
|
||||
expect(await repository.interruptActiveSensitivityAnalysisRuns(
|
||||
"Sensitive-field suggestion generation was interrupted by backend restart.",
|
||||
)).toEqual([
|
||||
expect.objectContaining({
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import {
|
||||
SensitivityAnalysisInterruptedError,
|
||||
SensitivityAnalysisService,
|
||||
} from "../src/catalog/sensitivity-analysis-service.js";
|
||||
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
|
||||
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogRepository,
|
||||
SensitivityAnalysisRun,
|
||||
WorkspaceDatabase,
|
||||
} from "../src/catalog/types.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: "public",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const running: SensitivityAnalysisRun = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
scope: "all",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "running",
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
inputTokens: 0,
|
||||
cacheReadTokens: 0,
|
||||
outputTokens: 0,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
startedAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
finishedAt: null,
|
||||
errorSummary: null,
|
||||
};
|
||||
|
||||
test("stops catalog selection when the request expires during a catalog read", async () => {
|
||||
const controller = new AbortController();
|
||||
const listTables = vi.fn();
|
||||
const repository = {
|
||||
get: vi.fn(async () => {
|
||||
controller.abort();
|
||||
return database;
|
||||
}),
|
||||
listTables,
|
||||
} as unknown as CatalogRepository;
|
||||
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
|
||||
const analysis = new SensitivityAnalysisService(repository, classifier);
|
||||
|
||||
await expect(analysis.analyze(
|
||||
database.id,
|
||||
"all",
|
||||
[],
|
||||
controller.signal,
|
||||
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
|
||||
expect(listTables).not.toHaveBeenCalled();
|
||||
expect(classifier.assessTable).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
|
||||
const controller = new AbortController();
|
||||
const update = vi.fn(async (_runId: string, changes: Partial<SensitivityAnalysisRun>) => ({
|
||||
...running,
|
||||
...changes,
|
||||
}));
|
||||
const repository = {
|
||||
get: vi.fn(async () => database),
|
||||
createSensitivityAnalysisRun: vi.fn(async () => {
|
||||
controller.abort();
|
||||
return running;
|
||||
}),
|
||||
updateSensitivityAnalysisRun: update,
|
||||
appendSensitivityAnalysisEvent: vi.fn(async () => undefined),
|
||||
} as unknown as CatalogRepository;
|
||||
const analysis = { analyze: vi.fn() } as unknown as SensitivityAnalysisService;
|
||||
const runner = new SensitivityAnalysisRunner(repository, analysis);
|
||||
|
||||
await expect(runner.run(database.id, "all", [], controller.signal))
|
||||
.rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
|
||||
expect(analysis.analyze).not.toHaveBeenCalled();
|
||||
expect(update).toHaveBeenCalledWith(running.id, expect.objectContaining({
|
||||
status: "interrupted",
|
||||
total: 0,
|
||||
unknown: 0,
|
||||
errorSummary: "Local sensitivity analysis reached its time limit.",
|
||||
}));
|
||||
});
|
||||
@@ -0,0 +1,450 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import {
|
||||
SensitivityClassifier,
|
||||
type LocalNerDetector,
|
||||
type SensitivityNerBudget,
|
||||
type SensitivityTableScan,
|
||||
type SensitivityValueSource,
|
||||
} from "../src/catalog/sensitivity-classifier.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
|
||||
import { CatalogConnectorError } from "../src/catalog/types.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: "public",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const table = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
name: "observations",
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
} satisfies CatalogTable;
|
||||
|
||||
function column(overrides: Partial<CatalogColumn> = {}): CatalogColumn {
|
||||
return {
|
||||
id: "33333333-3333-4333-8333-333333333333",
|
||||
tableId: table.id,
|
||||
name: "note",
|
||||
ordinalPosition: 1,
|
||||
dataType: "character varying",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: false,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function source(scan: SensitivityTableScan): SensitivityValueSource {
|
||||
return { scanTable: vi.fn(async (_request, consume) => {
|
||||
for (const batch of scan.batches) await consume(batch);
|
||||
return scan.coverage;
|
||||
}) };
|
||||
}
|
||||
|
||||
test("one email hidden in a generically named column makes the whole column sensitive", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[
|
||||
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
|
||||
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
});
|
||||
const classifier = new SensitivityClassifier(values);
|
||||
|
||||
const [assessment] = await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
columnId: target.id,
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("one text value longer than 500 characters makes the whole column sensitive", async () => {
|
||||
const target = column({ name: "comment" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "length", ruleId: "text.over_500_characters" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
|
||||
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
|
||||
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
|
||||
const humanProtected = column({
|
||||
id: "66666666-6666-4666-8666-666666666666",
|
||||
name: "category",
|
||||
sensitive: true,
|
||||
});
|
||||
const values = source({
|
||||
batches: [[
|
||||
{ columnId: benign.id, value: "active", characterLength: 6 },
|
||||
{ columnId: empty.id, value: null, characterLength: null },
|
||||
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [benign, empty, humanProtected] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessments).toEqual([
|
||||
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
|
||||
expect.objectContaining({
|
||||
columnId: empty.id,
|
||||
assessment: "unknown",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
|
||||
}),
|
||||
expect.objectContaining({
|
||||
columnId: humanProtected.id,
|
||||
assessment: "non_sensitive",
|
||||
proposedSensitive: false,
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
|
||||
const target = column({ sensitive: true });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
|
||||
const unresolved = column();
|
||||
const metadataMatch = column({
|
||||
id: "44444444-4444-4444-8444-444444444444",
|
||||
name: "codice_fiscale",
|
||||
});
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async () => {
|
||||
throw new CatalogConnectorError("upstream detail must not escape");
|
||||
}),
|
||||
};
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [unresolved, metadataMatch] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessments).toEqual([
|
||||
expect.objectContaining({
|
||||
columnId: unresolved.id,
|
||||
assessment: "unknown",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
|
||||
}),
|
||||
expect.objectContaining({
|
||||
columnId: metadataMatch.id,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
|
||||
const target = column({ name: "codice_fiscale" });
|
||||
const values = source({
|
||||
batches: [],
|
||||
coverage: { kind: "complete", observedRows: 0 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
});
|
||||
expect(values.scanTable).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test.each([
|
||||
["RSSMRA85T10A562S", "pii.italian_fiscal_code"],
|
||||
["IT60 X054 2811 1010 0000 0123 456", "financial.iban"],
|
||||
["4111 1111 1111 1111", "financial.payment_card"],
|
||||
["SWIFT DEUTDEFF500", "financial.bic"],
|
||||
["Partita IVA 00743110157", "pii.italian_vat"],
|
||||
["Passaporto YA1234567", "pii.passport_number"],
|
||||
["Carta d'identità CA12345AA", "pii.identity_card"],
|
||||
["Patente di guida U11234567A", "pii.drivers_license_number"],
|
||||
["Chiamare +39 347 123 4567", "pii.phone_number"],
|
||||
["Client 192.168.1.5", "network.ip_address"],
|
||||
["Device 00:1B:44:11:3A:B7", "network.mac_address"],
|
||||
["https://example.org/profiles/mario", "network.url"],
|
||||
["550e8400-e29b-41d4-a716-446655440000", "pii.uuid"],
|
||||
["AWS key AKIAIOSFODNN7EXAMPLE", "credential.access_key"],
|
||||
["Diagnosi: carcinoma mammario con metastasi ossee", "health.clinical_term"],
|
||||
["-----BEGIN PRIVATE KEY----- secret -----END PRIVATE KEY-----", "credential.private_key"],
|
||||
['{"profile":{"email":"not yet supplied"}}', "pii.json_sensitive_key"],
|
||||
] as const)("recognizes validated sensitive content without relying on the column name: %s", async (
|
||||
value,
|
||||
ruleId,
|
||||
) => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId }],
|
||||
});
|
||||
});
|
||||
|
||||
test("does not make a malformed email decisive", async () => {
|
||||
const target = column();
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{
|
||||
columnId: target.id,
|
||||
value: "contatto a@b..com non valido",
|
||||
characterLength: 28,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
|
||||
});
|
||||
|
||||
test("finds a valid email after a malformed candidate in the same value", async () => {
|
||||
const target = column();
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{
|
||||
columnId: target.id,
|
||||
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
|
||||
characterLength: 58,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("optional local NER evidence can make otherwise ambiguous Italian text sensitive", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values, detector).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledWith(
|
||||
[{ columnId: target.id, text: "Dimesso Mario Rossi" }],
|
||||
expect.any(AbortSignal),
|
||||
expect.any(Number),
|
||||
);
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "ner", ruleId: "ner.entity", label: "person_name", confidence: 0.91 }],
|
||||
});
|
||||
});
|
||||
|
||||
test("does not wait for an optional NER worker that is still warming", async () => {
|
||||
const target = column();
|
||||
const detector: LocalNerDetector = {
|
||||
isReady: () => false,
|
||||
detect: vi.fn(async () => [{ columnId: target.id, label: "person", confidence: 0.99 }]),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).not.toHaveBeenCalled();
|
||||
expect(assessment).toMatchObject({ assessment: "unknown" });
|
||||
});
|
||||
|
||||
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
|
||||
const columns = Array.from({ length: 17 }, (_, index) => column({
|
||||
id: `00000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
|
||||
name: `attribute_${index + 1}`,
|
||||
ordinalPosition: index + 1,
|
||||
}));
|
||||
const observations = columns.flatMap((item, columnIndex) => Array.from(
|
||||
{ length: 8 },
|
||||
(_, valueIndex) => ({
|
||||
columnId: item.id,
|
||||
value: `ordinary-${columnIndex}-${valueIndex}`,
|
||||
characterLength: 13,
|
||||
}),
|
||||
));
|
||||
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
|
||||
|
||||
await new SensitivityClassifier(source({
|
||||
batches: [observations],
|
||||
coverage: { kind: "complete", observedRows: 8 },
|
||||
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledTimes(2);
|
||||
expect(vi.mocked(detector.detect).mock.calls.map(([candidates]) => candidates.length)).toEqual([
|
||||
128,
|
||||
8,
|
||||
]);
|
||||
});
|
||||
|
||||
test("limits default NER work to two candidates spread across a wide table", async () => {
|
||||
const columns = Array.from({ length: 10 }, (_, index) => column({
|
||||
id: `10000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
|
||||
name: `attribute_${index + 1}`,
|
||||
ordinalPosition: index + 1,
|
||||
}));
|
||||
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
|
||||
|
||||
await new SensitivityClassifier(source({
|
||||
batches: [columns.flatMap((item, columnIndex) => [0, 1].map((valueIndex) => ({
|
||||
columnId: item.id,
|
||||
value: `ordinary-${columnIndex}-${valueIndex}`,
|
||||
characterLength: 13,
|
||||
})))],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledOnce();
|
||||
const submitted = vi.mocked(detector.detect).mock.calls[0]![0];
|
||||
expect(submitted).toHaveLength(2);
|
||||
expect(new Set(submitted.map((candidate) => candidate.columnId)).size).toBe(2);
|
||||
});
|
||||
|
||||
test("shares a bounded NER time allowance across tables in one analysis run", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => {
|
||||
await new Promise((resolve) => setTimeout(resolve, 20));
|
||||
return [];
|
||||
}),
|
||||
};
|
||||
const classifier = new SensitivityClassifier(values, detector);
|
||||
const nerBudget: SensitivityNerBudget = { remainingMs: 1 };
|
||||
|
||||
await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
Date.now() + 1_000,
|
||||
nerBudget,
|
||||
);
|
||||
await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
Date.now() + 1_000,
|
||||
nerBudget,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledOnce();
|
||||
expect(nerBudget.remainingMs).toBe(0);
|
||||
});
|
||||
|
||||
test("uninterpretable binary content remains unknown after complete coverage", async () => {
|
||||
const target = column({ dataType: "bytea" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,316 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
|
||||
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
|
||||
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
|
||||
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: 'clinical"data',
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const table = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
name: 'patient"facts',
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
} satisfies CatalogTable;
|
||||
|
||||
function column(id: string, name: string): CatalogColumn {
|
||||
return {
|
||||
id,
|
||||
tableId: table.id,
|
||||
name,
|
||||
ordinalPosition: 1,
|
||||
dataType: "text",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: false,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
};
|
||||
}
|
||||
|
||||
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
|
||||
const fullRows = Array.from({ length: 200 }, () => ({
|
||||
__value_0: "ordinary",
|
||||
__length_0: "8",
|
||||
__value_1: null,
|
||||
__length_1: null,
|
||||
}));
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
let clockCalls = 0;
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, {
|
||||
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
|
||||
});
|
||||
const consumed: unknown[] = [];
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note, contact],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 61_000,
|
||||
}, (batch) => consumed.push(...batch), new AbortController().signal);
|
||||
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
|
||||
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
|
||||
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql) === (
|
||||
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
test("reports complete coverage when the final full-scan page is short", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
|
||||
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
|
||||
: { rows: [] });
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
|
||||
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
|
||||
expect(query.mock.calls.filter(([sql]) => (
|
||||
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toHaveLength(1);
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "ordinary", characterLength: 8 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
let fullScanAttempts = 0;
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) {
|
||||
fullScanAttempts += 1;
|
||||
throw Object.assign(new Error("statement timeout"), { code: "57014" });
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
|
||||
expect(fullScanAttempts).toBe(1);
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
|
||||
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
|
||||
"SAVEPOINT sensitivity_full_scan",
|
||||
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
|
||||
]));
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "sample", characterLength: 6 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
const release = vi.fn();
|
||||
const secretStore = {
|
||||
materialize: vi.fn(() => ({
|
||||
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
|
||||
release,
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
|
||||
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
|
||||
]), { status: 200, headers: { "content-type": "application/json" } }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access, secretStore);
|
||||
const restDatabase: WorkspaceDatabase = {
|
||||
...database,
|
||||
binding: {
|
||||
transport: "rest_api",
|
||||
baseUrl: "https://dwh.example.test/root/",
|
||||
restPath: "/health",
|
||||
restAuth: "x-api-key",
|
||||
},
|
||||
};
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const consume = vi.fn();
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal)).resolves.toEqual({
|
||||
kind: "complete",
|
||||
observedRows: 1,
|
||||
});
|
||||
expect(access.connect).not.toHaveBeenCalled();
|
||||
expect(fetchMock).toHaveBeenCalledWith(
|
||||
"https://dwh.example.test/root/rpc/run_query",
|
||||
expect.objectContaining({
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json", "x-api-key": "test-api-key" },
|
||||
}),
|
||||
);
|
||||
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
|
||||
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]);
|
||||
expect(release).toHaveBeenCalledOnce();
|
||||
} finally {
|
||||
vi.unstubAllGlobals();
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("keeps multi-request REST scans conservative without a source transaction", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
const secretStore = {
|
||||
materialize: vi.fn(() => ({
|
||||
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
|
||||
release: vi.fn(),
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn()
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([
|
||||
{ __value_0: "ordinary", __length_0: 8 },
|
||||
]), { status: 200 }))
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const values = new ConcreteSensitivityValueSource({
|
||||
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
|
||||
}, secretStore, { batchRows: 1 });
|
||||
const restDatabase: WorkspaceDatabase = {
|
||||
...database,
|
||||
binding: {
|
||||
transport: "rest_api",
|
||||
baseUrl: "https://dwh.example.test/root",
|
||||
restPath: "/health",
|
||||
restAuth: "x-api-key",
|
||||
},
|
||||
};
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 1,
|
||||
});
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
} finally {
|
||||
vi.unstubAllGlobals();
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
|
||||
const query = vi.fn(async () => ({ rows: [] }));
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const now = vi.fn()
|
||||
.mockReturnValueOnce(1_000)
|
||||
.mockReturnValue(61_000);
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
|
||||
|
||||
await expect(values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 0,
|
||||
});
|
||||
expect(query).not.toHaveBeenCalled();
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
});
|
||||
@@ -81,6 +81,35 @@ test("loadConfig keeps local development defaults", () => {
|
||||
expect(loadConfig({}).dataRoot).toBeUndefined();
|
||||
});
|
||||
|
||||
test("loadConfig keeps local NER disabled unless an absolute model path is configured", () => {
|
||||
expect(loadConfig({}).sensitivityNer).toBeUndefined();
|
||||
expect(loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
}).sensitivityNer?.pythonExecutable).toBe("/opt/sensitivity-ner/bin/python");
|
||||
expect(loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
THT_SENSITIVITY_NER_PYTHON: "/opt/sensitivity-ner/bin/python",
|
||||
THT_SENSITIVITY_NER_WORKER: "/app/backend/python/sensitivity_ner_worker.py",
|
||||
THT_SENSITIVITY_NER_THREADS: "3",
|
||||
}).sensitivityNer).toEqual({
|
||||
modelPath: "/models/gliner2-pii",
|
||||
pythonExecutable: "/opt/sensitivity-ner/bin/python",
|
||||
workerScript: "/app/backend/python/sensitivity_ner_worker.py",
|
||||
threads: 3,
|
||||
});
|
||||
});
|
||||
|
||||
test("loadConfig rejects ambiguous or unsafe local NER configuration", () => {
|
||||
expect(() => loadConfig({ THT_SENSITIVITY_NER_MODEL_PATH: "fastino/model" }))
|
||||
.toThrow("sensitivity NER model path configuration is invalid");
|
||||
expect(() => loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
THT_SENSITIVITY_NER_THREADS: "0",
|
||||
})).toThrow("sensitivity NER thread configuration is invalid");
|
||||
expect(() => loadConfig({ THT_SENSITIVITY_NER_PYTHON: "/opt/ner/bin/python" }))
|
||||
.toThrow("sensitivity NER settings require a model path");
|
||||
});
|
||||
|
||||
test("loadConfig allows none and mock only outside production when auth.yaml is absent", () => {
|
||||
const originalNodeEnvironment = process.env.NODE_ENV;
|
||||
delete process.env.NODE_ENV;
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
services:
|
||||
core:
|
||||
image: thothii-core:sensitivity-ner
|
||||
build:
|
||||
args:
|
||||
INSTALL_SENSITIVITY_NER: "true"
|
||||
environment:
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: /opt/thothii/models/gliner2-privacy-filter-pii-multi
|
||||
THT_SENSITIVITY_NER_PYTHON: /opt/sensitivity-ner/bin/python
|
||||
THT_SENSITIVITY_NER_THREADS: ${THT_SENSITIVITY_NER_THREADS:-2}
|
||||
volumes:
|
||||
- type: bind
|
||||
source: ${THT_SENSITIVITY_NER_MODEL_DIR:?set THT_SENSITIVITY_NER_MODEL_DIR}
|
||||
target: /opt/thothii/models/gliner2-privacy-filter-pii-multi
|
||||
read_only: true
|
||||
|
||||
catalog-migrate:
|
||||
image: thothii-core:sensitivity-ner
|
||||
|
||||
workspace-maintenance:
|
||||
image: thothii-core:sensitivity-ner
|
||||
+14
-1
@@ -41,6 +41,7 @@ RUN npm run build
|
||||
FROM python:3.12-slim-bookworm@sha256:d50fb7611f86d04a3b0471b46d7557818d88983fc3136726336b2a4c657aa30b AS runtime
|
||||
ARG PI_VERSION
|
||||
ARG IMAGE_VERSION
|
||||
ARG INSTALL_SENSITIVITY_NER=false
|
||||
LABEL org.opencontainers.image.title="thothii-core" \
|
||||
org.opencontainers.image.version="${IMAGE_VERSION}" \
|
||||
org.opencontainers.image.description="ThothII core with its embedded Pi runtime" \
|
||||
@@ -48,7 +49,7 @@ LABEL org.opencontainers.image.title="thothii-core" \
|
||||
|
||||
# Runtime tools
|
||||
RUN set -eux; \
|
||||
runtime_packages="curl ca-certificates ripgrep fd-find tini git openssh-client"; \
|
||||
runtime_packages="curl ca-certificates ripgrep fd-find tini git openssh-client libseccomp2"; \
|
||||
if ! command -v flock >/dev/null 2>&1; then \
|
||||
runtime_packages="$runtime_packages util-linux"; \
|
||||
fi; \
|
||||
@@ -93,10 +94,22 @@ RUN mkdir -p /app/harness/config \
|
||||
# PiProcessManager (backend) prepende harnessDir/.venv/bin al PATH del child Pi → symlink al venv reale
|
||||
RUN ln -s /opt/venv /app/harness/.venv
|
||||
|
||||
# The optional NER dependency layer is independent of backend source and build artifacts so it can
|
||||
# be reused when only TypeScript or worker code changes.
|
||||
COPY backend/python/sensitivity-ner-requirements.txt /app/backend/python/sensitivity-ner-requirements.txt
|
||||
RUN if [ "$INSTALL_SENSITIVITY_NER" = "true" ]; then \
|
||||
python -m venv /opt/sensitivity-ner; \
|
||||
/opt/sensitivity-ner/bin/pip install --no-cache-dir --upgrade pip; \
|
||||
/opt/sensitivity-ner/bin/pip install --no-cache-dir -r /app/backend/python/sensitivity-ner-requirements.txt; \
|
||||
elif [ "$INSTALL_SENSITIVITY_NER" != "false" ]; then \
|
||||
echo "INSTALL_SENSITIVITY_NER must be true or false" >&2; exit 2; \
|
||||
fi
|
||||
|
||||
# Backend: dist + node_modules (stesso Node major 24 + glibc bookworm → compatibili)
|
||||
COPY --from=backend-build /src/backend/dist /app/backend/dist
|
||||
COPY --from=backend-build /src/backend/node_modules /app/backend/node_modules
|
||||
COPY backend/scripts/ssh-askpass.mjs /app/backend/scripts/ssh-askpass.mjs
|
||||
COPY backend/python /app/backend/python
|
||||
COPY backend/package*.json /app/backend/
|
||||
RUN chmod 0755 /app/backend/scripts/ssh-askpass.mjs
|
||||
|
||||
|
||||
@@ -21,13 +21,21 @@ completed scan with no finding may produce `non_sensitive`; an incomplete scan w
|
||||
produces `unknown`.
|
||||
|
||||
Ambiguous text may additionally be sent to an optional local NER detector only while time remains.
|
||||
The detector runs on CPU, receives no tools or network access, does not persist source values, and
|
||||
returns evidence rather than the column decision. The initial supported detector is
|
||||
By default it receives at most two candidates per table and shares a ten-second allowance across
|
||||
the entire analysis run. The detector runs on CPU, receives no tools or network access (enforced
|
||||
inside the worker with a fail-closed seccomp network-syscall filter), does not
|
||||
persist source values, and returns evidence rather than the column decision. The initial supported detector is
|
||||
[`fastino/gliner2-privacy-filter-PII-multi`](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi),
|
||||
used through the Apache-2.0 GLiNER2 Python library with a pinned model revision. Its model weights
|
||||
and GLiNER2 code are Apache-2.0, and its mDeBERTa base model is MIT. It is trained for seven
|
||||
languages including Italian and can run on CPU without using the installation's GPUs.
|
||||
|
||||
The selected checkpoint currently carries Transformers 5 tokenizer metadata while the released
|
||||
GLiNER2 2.0.0 runtime requires Transformers 4. ThothII may bridge only that known key rename in a
|
||||
temporary view of the immutable, checksummed artifact; unexpected or ambiguous metadata fails
|
||||
closed. Removing the compatibility bridge requires an offline smoke test against a corrected,
|
||||
pinned upstream release.
|
||||
|
||||
The NER detector is an optional installation asset because its weights and runtime are materially
|
||||
larger than the deterministic TypeScript engine. It is invoked only for otherwise unresolved text,
|
||||
never for values already classified by a decisive rule. If it is disabled, unavailable, times out,
|
||||
|
||||
@@ -104,11 +104,19 @@ or second orchestration subsystem. A target receives at most one provider retry;
|
||||
exhausted technical batches fail the run. Stale work is marked interrupted at startup and must be
|
||||
explicitly unlocked; it never resumes automatically.
|
||||
|
||||
Sensitive-field suggestion generation remains a synchronous administrative request, but each
|
||||
attempt has its own durable run and ordered sanitized events. This history is separate from
|
||||
Description Generation because its lifecycle and counters differ. Only execution metadata and
|
||||
aggregate counts are stored; proposed flags, prompts, raw model output, and provider diagnostics
|
||||
remain transient.
|
||||
Sensitivity analysis is a synchronous administrative request and does not use the installation
|
||||
model catalog. Database-specific adapters stream bounded normalized values from read-only source
|
||||
connections; the TypeScript `SensitivityClassifier` is the single decision point for
|
||||
`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted
|
||||
for at most five seconds per table, then the adapter samples within the sixty-second request budget.
|
||||
An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it
|
||||
cannot make or persist the decision itself.
|
||||
|
||||
Each attempt has its own durable run and ordered sanitized events, separate from Description
|
||||
Generation because its lifecycle and counters differ. The run records the local policy version,
|
||||
coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and
|
||||
worker diagnostics remain transient. Only an explicit administrator save changes the human-owned
|
||||
Sensitive Data Flag.
|
||||
|
||||
## Main backend classes
|
||||
|
||||
|
||||
@@ -137,14 +137,17 @@ Generated descriptions can be requested for selected tables, selected columns, e
|
||||
target, or targets with a missing generated description. The backend accepts one installation-wide
|
||||
run and processes targets sequentially. Every catalog column has a **Sensitive** flag, which defaults
|
||||
to `false`, including after a newly discovered column is synchronized. Before generation, an
|
||||
administrator can ask the configured model to suggest flags from structural metadata only (database,
|
||||
schema, table and column names, data types, nullability, primary keys, and foreign keys). Suggestions
|
||||
remain an unsaved draft until a human reviews and saves them.
|
||||
administrator can run local sensitivity analysis over the selected database, tables, or columns.
|
||||
The analysis combines structural metadata with bounded read-only inspection of source values. It
|
||||
uses no generative AI and no installation-catalog model. Assessments remain an unsaved draft until
|
||||
a human reviews and saves them; the reviewer may reverse any proposal.
|
||||
|
||||
The page exposes separate histories for description generation and sensitive-field suggestions.
|
||||
Sensitive-suggestion history stores the selected model, scope, status, aggregate counts, timestamps,
|
||||
and sanitized events. It does not store the proposed per-column flags, prompts, raw model output, or
|
||||
provider diagnostics; closing an unsaved review still discards that draft.
|
||||
The page exposes separate histories for description generation and sensitivity analysis. Analysis
|
||||
history stores the local policy version, scope, status, aggregate `sensitive`, `non_sensitive`, and
|
||||
`unknown` counts, timestamps, and sanitized events. It does not store source values, per-column
|
||||
proposals, NER spans, or worker diagnostics; closing an unsaved review discards that draft. The
|
||||
rules, time bounds, and optional CPU-only NER profile are documented in
|
||||
[Local sensitivity analysis](sensitivity-analysis.md).
|
||||
|
||||
For a column with `sensitive=false`, the worker may read at most five source rows and five
|
||||
representative non-null values through a read-only connector. For `sensitive=true`, the source query
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
# Local sensitivity analysis
|
||||
|
||||
Database Management can assess selected columns without sending their metadata or contents to a
|
||||
generative model. The feature is advisory: it creates a transient review draft, while the catalog's
|
||||
Sensitive Data Flag changes only when an administrator explicitly saves a choice. The administrator
|
||||
may set either value, including overriding a `sensitive` proposal.
|
||||
|
||||
## Default policy
|
||||
|
||||
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1`
|
||||
policy combines:
|
||||
|
||||
- normalized column-name rules for direct identifiers, credentials, and health data;
|
||||
- validated content rules for email, Italian fiscal code and VAT, passport, identity-card and
|
||||
driving-licence identifiers, phone numbers, IBAN/BIC, payment-card checksums, IP/MAC addresses,
|
||||
URLs, UUIDs, access keys, private-key markers, sensitive keys inside bounded recursive JSON, and
|
||||
a reviewed Italian clinical-term dictionary;
|
||||
- a conservative length rule: any observed textual value longer than 500 characters makes the
|
||||
entire column sensitive.
|
||||
|
||||
One decisive value is enough to classify the column as `sensitive`. A complete scan with no match
|
||||
may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled
|
||||
no-match columns are `unknown`; an `unknown` draft preserves the current human flag.
|
||||
|
||||
Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and
|
||||
REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never
|
||||
persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses
|
||||
a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a
|
||||
read-only transaction and always end with rollback. A REST scan can claim complete coverage only
|
||||
when it finishes in one request; multi-request pagination has no shared source transaction and is
|
||||
therefore conservatively reported as sampled.
|
||||
|
||||
The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and
|
||||
after catalog selection, source access, progress writes, and every table. If it expires after a run
|
||||
has been created, that run is finalized as `interrupted` and all not-decisively-processed columns
|
||||
are counted as `unknown`; no review payload is returned from the timed-out request.
|
||||
|
||||
History stores only the policy version, aggregate outcomes, timestamps, and fixed operational
|
||||
events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted.
|
||||
Neither path stores values, matched spans, prompts, or free-form model output.
|
||||
|
||||
## Optional CPU-only GLiNER2 evidence
|
||||
|
||||
The deterministic engine works without Python NER. An installation may opt into
|
||||
`fastino/gliner2-privacy-filter-PII-multi` for unresolved short text. It runs in a persistent local
|
||||
Python worker, adds sanitized evidence, and never becomes a second decision point. The worker:
|
||||
|
||||
- loads a local model directory only and forces Hugging Face/Transformers offline mode;
|
||||
- starts warming in the background when the backend starts; an analysis never waits for warm-up
|
||||
and skips NER until the worker is ready, so loading cannot consume the run's NER allowance;
|
||||
- hides CUDA and HIP devices and loads weights with `map_location="cpu"`;
|
||||
- starts with a scrubbed environment, then installs a fail-closed seccomp filter that denies
|
||||
network syscalls before accepting source text (the Python socket API is disabled as defense in depth);
|
||||
- receives at most two 500-character candidates per table by default, selected breadth-first
|
||||
across unresolved columns, and shares a ten-second NER allowance across the whole run;
|
||||
- returns only column ID, normalized label, and confidence; source text and entity spans are not
|
||||
returned or stored;
|
||||
- is skipped on timeout, startup failure, invalid output, or absent configuration. No LLM fallback
|
||||
is selected.
|
||||
|
||||
The pinned model revision is `c153999da5f4c509df4322b0c6a1baf3d2c284d7`. GLiNER2 and the model
|
||||
are Apache-2.0; the published mDeBERTa base is MIT. The optional runtime pins
|
||||
`gliner2[local]==2.0.0`, `transformers==4.57.6`, and the CPU-only PyTorch wheel
|
||||
`torch==2.14.0+cpu`. It lives in `/opt/sensitivity-ner`, is not installed in the default core image,
|
||||
and does not install CUDA packages.
|
||||
|
||||
The current upstream checkpoint was saved by Transformers 5.8 even though GLiNER2 2.0.0 officially
|
||||
requires Transformers `<5`; the resulting tokenizer error is independently reported in
|
||||
[GLiNER2 issue 145](https://github.com/fastino-ai/GLiNER2/issues/145). At startup ThothII leaves the
|
||||
pinned model directory unchanged and creates a temporary symlink view that maps the checkpoint's
|
||||
`extra_special_tokens` list to the Transformers 4 name `additional_special_tokens`. Any other or
|
||||
ambiguous shape fails closed and leaves the optional NER unavailable. The offline CPU smoke test
|
||||
must remain part of every dependency or model revision update.
|
||||
|
||||
## Prepare and enable the optional profile
|
||||
|
||||
Download happens during explicit installation, never during inference:
|
||||
|
||||
```bash
|
||||
./scripts/fetch-sensitivity-ner-model.sh /absolute/path/to/gliner2-pii
|
||||
```
|
||||
|
||||
The script builds the separate `thothii-core:sensitivity-ner` image, downloads the exact revision, and writes
|
||||
`MODEL_SHA256SUMS`. Keep the model directory outside the repository. Then set:
|
||||
|
||||
```bash
|
||||
export THOTH_ENABLE_SENSITIVITY_NER=1
|
||||
export THT_SENSITIVITY_NER_MODEL_DIR=/absolute/path/to/gliner2-pii
|
||||
export THT_SENSITIVITY_NER_THREADS=2
|
||||
./scripts/run-stack.sh
|
||||
```
|
||||
|
||||
For an operator-managed Compose invocation, include `deploy/compose.sensitivity-ner.yaml` after the
|
||||
base and installation overlays. The core build argument `INSTALL_SENSITIVITY_NER=true` installs the
|
||||
optional Python dependencies into their isolated virtualenv. The model mount is read-only. Values
|
||||
above eight threads are rejected; start with two so classification cannot contend heavily with
|
||||
other CPU workloads.
|
||||
|
||||
## Acceptance on a real database
|
||||
|
||||
Run the first evaluation in shadow mode: read the source with its existing read-only role, do not
|
||||
save proposed flags, and report only aggregate counts, rule IDs, coverage, and timings. Never copy
|
||||
matched values into test output. Use a separately approved, labeled Italian corpus to calculate
|
||||
precision and recall; raw PSD values must remain inside the authorized environment.
|
||||
|
||||
Inside the configured core runtime, the non-mutating command is:
|
||||
|
||||
```bash
|
||||
npm run sensitivity:shadow -- psd-clinical
|
||||
```
|
||||
|
||||
It reads catalog metadata and source values but emits one aggregate JSON object with no database,
|
||||
table, column, or source-value detail. It neither creates an analysis run nor updates a flag.
|
||||
|
||||
Enabling NER by default requires all of these gates:
|
||||
|
||||
1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory;
|
||||
2. the Python dependency/license inventory contains only redistribution-compatible licenses;
|
||||
3. the CPU benchmark stays within the configured deadlines and does not use a GPU;
|
||||
4. the labeled Italian evaluation meets thresholds approved by the product owner.
|
||||
|
||||
If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved
|
||||
columns remain `unknown` rather than being sent to an internal or external LLM.
|
||||
|
||||
The first aggregate PSD shadow comparison is recorded in
|
||||
[`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the
|
||||
local CPU runner, NER found additional entities but reduced total coverage inside the 60-second
|
||||
deadline, so the accepted setting remains disabled by default.
|
||||
@@ -0,0 +1,37 @@
|
||||
# PSD sensitivity shadow evaluation
|
||||
|
||||
Date: 2026-09-02
|
||||
|
||||
This report records an aggregate, non-mutating evaluation of `sensitivity-v1` against the PSD
|
||||
workspace. The source data warehouse was accessed through the configured read-only connector. The
|
||||
shadow command did not create an analysis run, update catalog metadata, or save Sensitive Data
|
||||
Flags. No database, table, column, source value, matched span, or free-form diagnostic was emitted.
|
||||
|
||||
The runner was the local Docker `arm64` CPU environment connected to the PSD source; this was not a
|
||||
benchmark of the PSD production server. Both runs used the same 2,275 catalog columns and a
|
||||
60-second analysis deadline.
|
||||
|
||||
| Profile | Sensitive | Non-sensitive | Unknown | NER findings | Analysis time |
|
||||
| --- | ---: | ---: | ---: | ---: | ---: |
|
||||
| Deterministic policy | 57 | 8 | 2,210 | 0 | 60,017 ms |
|
||||
| CPU NER, pre-warmed, two candidates/table, 10 s shared allowance | 68 | 0 | 2,207 | 11 | 60,022 ms |
|
||||
|
||||
The deterministic run produced findings from metadata, phone-number, and Italian clinical-term
|
||||
rules. The optional NER run identified eleven additional unresolved text candidates, but its
|
||||
inference time reduced the source coverage reached before the global deadline. The number of
|
||||
definitive non-sensitive assessments consequently fell from eight to zero, so this broad shadow
|
||||
run does not justify enabling NER by default.
|
||||
|
||||
The separate offline synthetic Italian smoke test succeeded with a `full_name` finding at high
|
||||
confidence. The image dependency check reported no broken requirements, and PyTorch reported
|
||||
`cuda=False`, no CUDA runtime, and zero GPU devices. A defense-in-depth test retained a raw socket
|
||||
constructor before Python-level blocking and confirmed that the worker's seccomp filter still
|
||||
rejected the socket syscall with `EPERM`.
|
||||
|
||||
## Acceptance outcome
|
||||
|
||||
- Keep the deterministic TypeScript policy enabled by default.
|
||||
- Keep GLiNER2 available only through the explicit CPU-only installation profile.
|
||||
- Do not enable NER by default for PSD on the basis of this shadow run.
|
||||
- Reconsider the PSD setting only after a benchmark on the actual target CPU and a labeled Italian
|
||||
corpus demonstrate a useful precision/recall gain without unacceptable coverage loss.
|
||||
@@ -0,0 +1,36 @@
|
||||
# Optional sensitivity NER license inventory
|
||||
|
||||
This inventory covers the isolated `/opt/sensitivity-ner` Python environment built from
|
||||
`backend/python/sensitivity-ner-requirements.txt` on 2 September 2026. It is a technical release
|
||||
gate, not legal advice. Every dependency is version-locked; changing any version requires
|
||||
regenerating this inventory and rerunning the offline CPU smoke test.
|
||||
|
||||
The optional runtime also dynamically links Debian's `libseccomp2` (LGPL-2.1-only) solely to
|
||||
install its kernel-enforced network syscall filter; no libseccomp source is incorporated into ThothII.
|
||||
|
||||
No dependency or selected model uses a non-commercial, research-only, source-available, GPL, or
|
||||
AGPL license. MPL-2.0, PSF-2.0, and the permissive composite licenses below allow free-of-charge and
|
||||
commercial use, but distributors must still preserve their applicable notices and license texts.
|
||||
|
||||
| License family | Locked packages |
|
||||
| --- | --- |
|
||||
| Apache-2.0 | `accelerate==1.14.0`, `gliner2==2.0.0`, `hf-xet==1.6.0`, `huggingface-hub==0.36.2`, `peft==0.20.0`, `requests==2.34.2`, `safetensors==0.8.0`, `tokenizers==0.22.2`, `transformers==4.57.6` |
|
||||
| MIT | `annotated-types==0.8.0`, `charset-normalizer==3.5.1`, `filelock==3.32.5`, `pydantic==2.13.5`, `pydantic-core==2.46.5`, `PyYAML==6.0.3`, `typing-inspection==0.4.4`, `urllib3==2.7.0` |
|
||||
| BSD-2/3-Clause | `fsspec==2026.7.0`, `idna==3.19`, `Jinja2==3.1.6`, `MarkupSafe==3.0.3`, `mpmath==1.3.0`, `networkx==3.6.1`, `psutil==7.2.2`, `sympy==1.14.0` |
|
||||
| MPL-2.0 or mixed MPL/MIT | `certifi==2026.7.22`, `tqdm==4.70.0` |
|
||||
| PSF-2.0 | `typing-extensions==4.16.0` |
|
||||
| Composite permissive | `numpy==2.5.2` (BSD-3-Clause, 0BSD, MIT, Zlib, CC0), `packaging==26.3` (Apache-2.0 or BSD-2-Clause), `regex==2026.9.3` (Apache-2.0 and CNRI-Python), `torch==2.14.0+cpu` (Apache-2.0, LLVM exception, BSD, BSL-1.0, MIT) |
|
||||
|
||||
The selected `fastino/gliner2-privacy-filter-PII-multi` weights at revision
|
||||
`c153999da5f4c509df4322b0c6a1baf3d2c284d7` are marked Apache-2.0 in the
|
||||
[model card](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi). Its published
|
||||
`microsoft/mdeberta-v3-base` base model is MIT. The Fastino training corpus is described as
|
||||
synthetic but is not published, so the training process is not independently reproducible.
|
||||
|
||||
Before distributing the optional image or model pack:
|
||||
|
||||
1. retain the upstream license and notice files for all packaged wheels, system libraries, and weights;
|
||||
2. archive `THOTHII_MODEL_REVISION` and the verified `MODEL_SHA256SUMS` beside the model;
|
||||
3. verify that `pip check` succeeds in the isolated environment;
|
||||
4. compare the installed distribution/version set with this inventory;
|
||||
5. repeat the licensing review if an upstream artifact or dependency changes.
|
||||
@@ -179,6 +179,16 @@ Urchade model. This does not prove Italian clinical accuracy: the published SPY
|
||||
English and the training corpus is synthetic. Those are quality and reproducibility limitations,
|
||||
not a reason to reintroduce a generative LLM into the classifier.
|
||||
|
||||
An implementation smoke test found a packaging incompatibility in the selected upstream versions:
|
||||
the checkpoint was written by Transformers 5.8, while `gliner2[local]==2.0.0` requires
|
||||
`transformers>=4.38,<5`. Every published checkpoint revision has the same tokenizer metadata, and
|
||||
[upstream issue 145](https://github.com/fastino-ai/GLiNER2/issues/145) reports the identical failure.
|
||||
ThothII therefore pins Transformers 4.57.6 and performs the minimal documented-key conversion in a
|
||||
temporary local view, without changing the downloaded model or its checksums. This compatibility
|
||||
shim was accepted only after an offline CPU smoke test detected Italian names, dates, locations,
|
||||
and usernames; any unexpected metadata shape fails closed. A future upstream fix must replace,
|
||||
not silently stack on, this shim.
|
||||
|
||||
## Fit with current ThothII design
|
||||
|
||||
The current backend already owns source sampling and the human-owned `Sensitive Data Flag`; source
|
||||
|
||||
@@ -84,7 +84,7 @@ stato ripreso nella descrizione. Deve quindi essere ripetuta sul comportamento p
|
||||
- scope di sincronizzazione `tables`, `columns`, `relationships` e `all`;
|
||||
- run durevoli, conferma delle differenze distruttive, cancellazione, recovery, eventi SSE e
|
||||
fallback polling;
|
||||
- Sensitive Data Flag, suggerimenti AI strutturali, review draft e storico operativo;
|
||||
- Sensitive Data Flag, analisi locale di metadati e contenuti, review draft e storico operativo;
|
||||
- scope di generazione `selected_columns`, `selected_tables`, `all` e `missing`;
|
||||
- campionamento read-only, valori sintetici per colonne protette, batching, retry, stop, Unlock,
|
||||
storico e consolidamento;
|
||||
@@ -211,17 +211,17 @@ riapplicare le migrazioni e ripartire dal database reale.
|
||||
| ID | P | Livello | Scenario | Risultato atteso |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| PRV-01 | P0 | Repository/API | Prima sincronizzazione, re-sync e aggiunta di una colonna. | Il default è `false`; il valore umano delle colonne esistenti è preservato; la nuova colonna è esplicitamente da riesaminare ma non riceve uno stato audit inventato. |
|
||||
| PRV-02 | P0 | API | Suggerimento per un database, tabelle selezionate e colonne selezionate; database multipli, target duplicati o mancanti. | Il provider riceve esattamente le colonne dello scope. Input ambigui sono rifiutati prima della chiamata e nessun flag cambia. |
|
||||
| PRV-03 | P0 | Contratto | Ispezione del messaggio al classifier. | Sono presenti solo database, schema, tabella, colonna, tipo, nullabilità, PK e FK. Non compaiono righe, valori, commenti, descrizioni, flag corrente o segreti. |
|
||||
| PRV-04 | P1 | Contratto | `wide_entity`, identificatori lunghi e limite byte. | Ordine deterministico, batch massimi di dieci colonne e rispetto del limite messaggio; una singola colonna non rappresentabile fallisce prima del provider con errore sicuro. |
|
||||
| PRV-05 | P0 | API | Risposta valida, fenced/prosa, JSON malformato, target mancante/duplicato/ignoto e provider failure. | Ogni colonna richiesta compare una sola volta. Una classificazione invalida viene ritentata una volta; dopo esaurimento si ottiene errore sanitizzato e nessuna modifica. |
|
||||
| PRV-02 | P0 | API | Analisi per un database, tabelle selezionate e colonne selezionate; database multipli, target duplicati o mancanti. | Solo i target dello scope raggiungono l'adapter read-only. Input ambigui sono rifiutati prima della lettura e nessun flag cambia. |
|
||||
| PRV-03 | P0 | Unit/Contratto | Valori con email, codice fiscale italiano valido, IBAN, carta con Luhn, chiave privata, chiave JSON sensibile e testo oltre 500 caratteri. | Un solo riscontro validato rende l'intera colonna `sensitive`; l'evidenza contiene solo rule ID e conteggi sanitizzati, mai il valore. |
|
||||
| PRV-04 | P0 | Integrazione | Scansione completa oltre cinque secondi, timeout PostgreSQL e budget globale di sessanta secondi. | L'adapter passa al campionamento, ripristina la transazione dopo `statement_timeout`, resta read-only e non supera la deadline. Copertura incompleta senza match produce `unknown`. |
|
||||
| PRV-05 | P0 | Unit/API | Tabella vuota, colonna all-null, binario non ispezionabile, scan completo senza match e scan incompleto senza match. | Gli esiti sono rispettivamente `unknown`, `unknown`, `unknown`, `non_sensitive` e `unknown`; `unknown` conserva la scelta umana corrente. |
|
||||
| PRV-06 | P0 | UI/API | Apertura draft, modifica manuale, chiusura/reload e salvataggio. | La proposta non è persistita prima di Save; reload la scarta. Il reviewer può invertire scelte; si salvano solo colonne cambiate con versione ottimistica; un conflitto richiede reload. |
|
||||
| PRV-07 | P1 | Repository/UI | Tentativi completati, falliti e attivi al restart. | Ogni tentativo ha un run distinto con scope, modello, contatori ed eventi sanitizzati; startup marca `interrupted` i run attivi. Storico newest-first senza target ID, proposte, prompt, output grezzo o diagnostica provider. |
|
||||
| PRV-07 | P1 | Repository/UI | Tentativi completati, falliti e attivi al restart. | Ogni tentativo ha un run distinto con scope, engine `local`, versione policy, tre contatori ed eventi sanitizzati; startup marca `interrupted` i run attivi. Storico newest-first senza target ID, proposte, valori o diagnostica worker. |
|
||||
| PRV-08 | P0 | Integrazione | Generazione descrizioni su target con canary protetti. | Le colonne protette sono assenti dalla proiezione SQL, non semplicemente filtrate dopo la lettura. Se non rimangono colonne leggibili non viene eseguita una `SELECT`. Nessun canary protetto esce dal processo. |
|
||||
| PRV-09 | P0 | Contratto/Integrazione | Tabella mista con colonne sensibili e pubbliche. | Per le sensibili il prompt contiene valori plausibili, deterministici e limitati derivati dai soli metadati, nello stesso formato dei campioni e senza etichettarli al modello come sintetici. Per le pubbliche: massimo cinque righe e cinque valori rappresentativi, valori troncati e transazione read-only chiusa con rollback. |
|
||||
| PRV-10 | P1 | API | Cambio `false→true→false` dopo una descrizione già generata. | Il testo esistente non viene rigenerato retroattivamente. Solo le generazioni future cambiano fonte del contesto; tornando `false` il campionamento reale torna eleggibile. |
|
||||
| PRV-11 | P0 | API/UI | Utente senza `database.manage`, modello non configurato, catalogo/provider indisponibile e richiesta interrotta. | Controlli nascosti/disabilitati in UI e rifiuto server-side; errori non espongono dettagli. Un tentativo fallito compare nello storico senza trasformarsi in audit della decisione umana. |
|
||||
| PRV-12 | P1 | L2 | Corpus strutturale etichettato con identificatori personali, credenziali/token, salute, finanza, localizzazione e controlli non sensibili/ambigui, in inglese e italiano. | Si misurano precisione, recall e falsi negativi per modello. I campi critici mancati sono sottoposti al product owner; la soglia quantitativa va ratificata prima di diventare gate, perché il classifier è advisory e human-in-the-loop. |
|
||||
| PRV-11 | P0 | API/UI | Utente senza `database.manage`, sorgente/adapter indisponibile, NER assente o in timeout e richiesta interrotta. | Controlli nascosti/disabilitati in UI e rifiuto server-side; errori non espongono dettagli. Il NER opzionale degrada alle regole/coverage senza selezionare un LLM. |
|
||||
| PRV-12 | P1 | L2 | Corpus etichettato con identificatori personali, credenziali/token, salute, finanza, localizzazione e controlli non sensibili/ambigui, in inglese e italiano. | Si misurano precisione, recall, falsi negativi, copertura e latenza separatamente per policy deterministica e NER CPU. La soglia va ratificata prima di abilitare NER per default; il classifier resta advisory e human-in-the-loop. |
|
||||
| PRV-13 | P0 | UI/E2E | Modifica di un flag nella review senza Save e tentativo immediato di generare descrizioni. | Gate di rilascio da formalizzare: la generazione deve essere bloccata finché il draft non è salvato o scartato. In alternativa la UI deve dichiarare inequivocabilmente che verrà usato il valore persistito; non è accettabile mostrare “protetto” e campionare come non protetto. |
|
||||
|
||||
## 9. Casi di test — generazione e consolidamento dei commenti
|
||||
@@ -301,9 +301,9 @@ un valore protetto non può mai esserlo.
|
||||
| Area | Evidenza automatica già presente | Gap principale |
|
||||
| --- | --- | --- |
|
||||
| Snapshot e sincronizzazione | `backend/test/catalog-schema-introspector.test.ts`, `catalog-schema-routes.test.ts`, `catalog-table-introspector.test.ts`, `catalog-repository.integration.test.ts` | Introspezione `pg_catalog` realmente end-to-end, parità live dei tre trasporti e un unico E2E con re-scan distruttivo. |
|
||||
| Privacy | `catalog-description-generation-routes.test.ts`, `catalog-description-generation-worker.test.ts`, `catalog-description-source-sampler.test.ts`, `catalog-synthetic-sample-value.test.ts` | Prova canary integrata query→prompt→API/log e benchmark reale post-ADR 0011. |
|
||||
| Privacy | `catalog-sensitivity-classifier.test.ts`, `catalog-sensitivity-value-source.test.ts`, `catalog-local-ner-detector.test.ts` e i test di route/review | Prova shadow PSD con report solo aggregato, corpus italiano etichettato e benchmark NER CPU post-ADR 0014. |
|
||||
| Generazione | `catalog-description-generation-routes.test.ts`, `catalog-description-generation-worker.test.ts`, `catalog-description-generation.integration.test.ts` e test del helper | Accettazione reale aggiornata, cancellazione di una query PostgreSQL bloccata e integrazione ermetica fino all'endpoint LiteLLM locale. |
|
||||
| UI | `DatabaseManagementPage.test.tsx`, `DescriptionGenerationDrawer.test.tsx`, `SensitiveDataSuggestionHistoryDrawer.test.tsx` | L'E2E Playwright corrente verifica soprattutto il layout, non il workflow funzionale. |
|
||||
| UI | `DatabaseManagementPage.test.tsx`, `DescriptionGenerationDrawer.test.tsx`, `SensitivityAnalysisHistoryDrawer.test.tsx` | L'E2E Playwright corrente verifica soprattutto il layout, non il workflow funzionale. |
|
||||
|
||||
Nuovi asset consigliati:
|
||||
|
||||
@@ -320,7 +320,7 @@ Nuovi asset consigliati:
|
||||
### Wave 1 — contratto rapido
|
||||
|
||||
- parser snapshot, introspector, scope e diff;
|
||||
- classifier strutturale, batching e validazione output;
|
||||
- classifier locale, validatori/checksum, copertura e fallback al campionamento;
|
||||
- sampler, valori sintetici, prompt bounds e parser descrizioni;
|
||||
- autorizzazione, redazione e race del coordinator.
|
||||
|
||||
@@ -340,8 +340,8 @@ Nuovi asset consigliati:
|
||||
|
||||
### Wave 4 — accettazione L2
|
||||
|
||||
- provider reale sul database collegato dopo classificazione e review dei flag;
|
||||
- benchmark PRV-12 e rubric GEN-15;
|
||||
- analisi shadow sul database collegato e review dei flag, senza scritture in sorgente;
|
||||
- benchmark NER CPU PRV-12 e rubric GEN-15 per la generazione descrizioni;
|
||||
- scansione finale di canary e segreti;
|
||||
- approvazione del product owner.
|
||||
|
||||
@@ -418,7 +418,7 @@ sorgente reale.
|
||||
mostrare la disclosure anche per tabelle/colonne selezionate; il piano considera entrambi P0.
|
||||
- L'AbortSignal corrente va provato contro una query PostgreSQL realmente bloccata: la sola
|
||||
cancellazione del helper non dimostra che la lettura sorgente sia interrompibile.
|
||||
- Lo storico dei Sensitive Data Suggestion Run è operativo, non un audit delle decisioni umane.
|
||||
- Lo storico dei Sensitivity Analysis Run è operativo, non un audit delle decisioni umane.
|
||||
- La policy privacy non è ancora applicata allo schema-linking/LSH; nessun risultato di questo piano
|
||||
deve essere presentato come copertura di quel percorso.
|
||||
- Un provider reale resta non deterministico: il rilascio deve dipendere dai gate tecnici e dalla
|
||||
|
||||
@@ -126,7 +126,7 @@ export interface CatalogColumn {
|
||||
updatedAt: string;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestion {
|
||||
export interface SensitivityReviewItem {
|
||||
columnId: string;
|
||||
tableId: string;
|
||||
tableName: string;
|
||||
@@ -134,29 +134,40 @@ export interface SensitiveDataSuggestion {
|
||||
version: number;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
assessment: "sensitive" | "non_sensitive" | "unknown";
|
||||
evidence: Array<{
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage";
|
||||
ruleId: string;
|
||||
label?: string;
|
||||
confidence?: number;
|
||||
}>;
|
||||
observedValues: number;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestions {
|
||||
suggestions: SensitiveDataSuggestion[];
|
||||
run: SensitiveDataSuggestionRun;
|
||||
export interface SensitivityAnalysisResult {
|
||||
suggestions: SensitivityReviewItem[];
|
||||
run: SensitivityAnalysisRun;
|
||||
}
|
||||
|
||||
export type SensitiveDataSuggestionRequest =
|
||||
export type SensitivityAnalysisRequest =
|
||||
| { scope: "all" }
|
||||
| { scope: "selected_tables"; targetIds: string[] }
|
||||
| { scope: "selected_columns"; targetIds: string[] };
|
||||
|
||||
export type SensitiveDataSuggestionStatus = "running" | "completed" | "failed" | "interrupted";
|
||||
export type SensitivityAnalysisStatus = "running" | "completed" | "failed" | "interrupted";
|
||||
|
||||
export interface SensitiveDataSuggestionRun {
|
||||
export interface SensitivityAnalysisRun {
|
||||
id: string;
|
||||
databaseId: string;
|
||||
scope: SensitiveDataSuggestionRequest["scope"];
|
||||
modelId: string;
|
||||
status: SensitiveDataSuggestionStatus;
|
||||
scope: SensitivityAnalysisRequest["scope"];
|
||||
engine: "llm" | "local";
|
||||
modelId: string | null;
|
||||
policyVersion: string | null;
|
||||
status: SensitivityAnalysisStatus;
|
||||
total: number;
|
||||
suggestedSensitive: number;
|
||||
suggestedNonSensitive: number;
|
||||
unknown: number;
|
||||
inputTokens?: number;
|
||||
cacheReadTokens?: number;
|
||||
outputTokens?: number;
|
||||
@@ -167,7 +178,7 @@ export interface SensitiveDataSuggestionRun {
|
||||
errorSummary: string | null;
|
||||
}
|
||||
|
||||
export interface SensitiveDataSuggestionEvent {
|
||||
export interface SensitivityAnalysisEvent {
|
||||
runId: string;
|
||||
sequence: number;
|
||||
level: "info" | "warning" | "error";
|
||||
@@ -458,28 +469,27 @@ export const updateCatalogColumnSensitive = (
|
||||
{ method: "PATCH", body: JSON.stringify({ version, sensitive }) },
|
||||
);
|
||||
|
||||
export const suggestSensitiveFields = (
|
||||
export const runSensitivityAnalysis = (
|
||||
databaseId: string,
|
||||
modelId: string,
|
||||
selection: SensitiveDataSuggestionRequest,
|
||||
selection: SensitivityAnalysisRequest,
|
||||
) =>
|
||||
apiFetch<SensitiveDataSuggestions>(
|
||||
apiFetch<SensitivityAnalysisResult>(
|
||||
`/catalog/databases/${encodeURIComponent(databaseId)}/sensitive-data-suggestions`,
|
||||
{ method: "POST", body: JSON.stringify({ modelId, ...selection }) },
|
||||
{ method: "POST", body: JSON.stringify(selection) },
|
||||
);
|
||||
|
||||
export const getSensitiveDataSuggestionRun = (runId: string) =>
|
||||
apiFetch<SensitiveDataSuggestionRun>(
|
||||
export const getSensitivityAnalysisRun = (runId: string) =>
|
||||
apiFetch<SensitivityAnalysisRun>(
|
||||
`/catalog/sensitive-data-suggestion-runs/${encodeURIComponent(runId)}`,
|
||||
);
|
||||
|
||||
export const listSensitiveDataSuggestionRuns = (limit = 50) =>
|
||||
apiFetch<SensitiveDataSuggestionRun[]>(
|
||||
export const listSensitivityAnalysisRuns = (limit = 50) =>
|
||||
apiFetch<SensitivityAnalysisRun[]>(
|
||||
`/catalog/sensitive-data-suggestion-runs?limit=${encodeURIComponent(String(limit))}`,
|
||||
);
|
||||
|
||||
export const listSensitiveDataSuggestionEvents = (runId: string, after = 0) =>
|
||||
apiFetch<SensitiveDataSuggestionEvent[]>(
|
||||
export const listSensitivityAnalysisEvents = (runId: string, after = 0) =>
|
||||
apiFetch<SensitivityAnalysisEvent[]>(
|
||||
`/catalog/sensitive-data-suggestion-runs/${encodeURIComponent(runId)}/events-list?after=${after}`,
|
||||
);
|
||||
|
||||
|
||||
@@ -3,13 +3,13 @@ import { server } from "../test/msw";
|
||||
import {
|
||||
cancelDescriptionGenerationRun,
|
||||
descriptionGenerationEventsUrl,
|
||||
getSensitiveDataSuggestionRun,
|
||||
getSensitivityAnalysisRun,
|
||||
listDescriptionGenerationRuns,
|
||||
listSensitiveDataSuggestionEvents,
|
||||
listSensitiveDataSuggestionRuns,
|
||||
listSensitivityAnalysisEvents,
|
||||
listSensitivityAnalysisRuns,
|
||||
unlockDescriptionGenerationRun,
|
||||
type DescriptionGenerationRun,
|
||||
type SensitiveDataSuggestionRun,
|
||||
type SensitivityAnalysisRun,
|
||||
} from "./catalog-databases";
|
||||
|
||||
const historicalRun: DescriptionGenerationRun = {
|
||||
@@ -31,15 +31,18 @@ const historicalRun: DescriptionGenerationRun = {
|
||||
errorSummary: null,
|
||||
};
|
||||
|
||||
const sensitiveSuggestionRun: SensitiveDataSuggestionRun = {
|
||||
const sensitiveSuggestionRun: SensitivityAnalysisRun = {
|
||||
id: "99999999-9999-4999-8999-999999999999",
|
||||
databaseId: "11111111-1111-4111-8111-111111111111",
|
||||
scope: "selected_columns",
|
||||
modelId: "local-qwen",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "completed",
|
||||
total: 4,
|
||||
suggestedSensitive: 2,
|
||||
suggestedNonSensitive: 2,
|
||||
unknown: 0,
|
||||
createdAt: "2026-08-28T09:00:00Z",
|
||||
startedAt: "2026-08-28T09:00:00Z",
|
||||
updatedAt: "2026-08-28T09:00:01Z",
|
||||
@@ -119,9 +122,9 @@ test("lists, reads, and replays persisted sensitive-suggestion history", async (
|
||||
}),
|
||||
);
|
||||
|
||||
await expect(listSensitiveDataSuggestionRuns(50)).resolves.toEqual([sensitiveSuggestionRun]);
|
||||
await expect(getSensitiveDataSuggestionRun(sensitiveSuggestionRun.id)).resolves.toEqual(sensitiveSuggestionRun);
|
||||
await expect(listSensitiveDataSuggestionEvents(sensitiveSuggestionRun.id, 2)).resolves.toEqual([event]);
|
||||
await expect(listSensitivityAnalysisRuns(50)).resolves.toEqual([sensitiveSuggestionRun]);
|
||||
await expect(getSensitivityAnalysisRun(sensitiveSuggestionRun.id)).resolves.toEqual(sensitiveSuggestionRun);
|
||||
await expect(listSensitivityAnalysisEvents(sensitiveSuggestionRun.id, 2)).resolves.toEqual([event]);
|
||||
expect(requestedLimit).toBe("50");
|
||||
expect(requestedAfter).toBe("2");
|
||||
});
|
||||
|
||||
@@ -147,11 +147,11 @@ test.each([
|
||||
["description_generation_target_ids_duplicate", "Description generation target IDs must be unique."],
|
||||
["description_generation_no_eligible_targets", "No eligible catalog tables or columns need description generation."],
|
||||
["catalog_table_not_found", "One or more selected catalog tables were not found."],
|
||||
["sensitive_data_suggestion_invalid_response", "The model returned an incomplete or invalid classification. No suggestions were applied."],
|
||||
["sensitive_data_suggestion_provider_unavailable", "The selected model could not complete the request. No suggestions were applied."],
|
||||
["sensitive_data_suggestion_history_request_invalid", "Sensitive suggestion history parameters are invalid."],
|
||||
["sensitive_data_suggestion_history_failed", "Sensitive suggestion history could not be loaded."],
|
||||
["sensitive_data_suggestion_run_not_found", "The sensitive suggestion run was not found."],
|
||||
["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."],
|
||||
["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."],
|
||||
["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."],
|
||||
["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."],
|
||||
["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."],
|
||||
["relationship_not_found", "The relationship no longer exists. Refresh and try again."],
|
||||
["relationship_duplicate", "This relationship already exists."],
|
||||
["relationship_target_not_unique", "The target column must be the only primary-key column of its table."],
|
||||
|
||||
+10
-12
@@ -26,9 +26,8 @@ const safeErrorCodes = new Set([
|
||||
"sensitive_data_suggestion_request_invalid",
|
||||
"sensitive_data_suggestion_target_ids_duplicate",
|
||||
"sensitive_data_suggestion_no_columns",
|
||||
"sensitive_data_suggestion_payload_too_large",
|
||||
"sensitive_data_suggestion_invalid_response",
|
||||
"sensitive_data_suggestion_provider_unavailable",
|
||||
"sensitivity_source_unavailable",
|
||||
"sensitivity_analysis_timeout",
|
||||
"sensitive_data_suggestion_failed",
|
||||
"sensitive_data_suggestion_history_request_invalid",
|
||||
"sensitive_data_suggestion_history_failed",
|
||||
@@ -90,16 +89,15 @@ const localCodeMessages: Record<string, string> = {
|
||||
catalog_column_not_found: "The selected catalog column was not found.",
|
||||
catalog_table_not_found: "One or more selected catalog tables were not found.",
|
||||
workspace_configuration_unavailable: "The database workspace configuration is unavailable.",
|
||||
sensitive_data_suggestion_request_invalid: "Select a database, one or more tables, or one or more columns before requesting sensitive-field suggestions.",
|
||||
sensitive_data_suggestion_request_invalid: "Select a database, one or more tables, or one or more columns before running sensitivity analysis.",
|
||||
sensitive_data_suggestion_target_ids_duplicate: "Each selected table or column can be included only once.",
|
||||
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to classify.",
|
||||
sensitive_data_suggestion_payload_too_large: "The selected structural metadata cannot be divided into safe model requests.",
|
||||
sensitive_data_suggestion_invalid_response: "The model returned an incomplete or invalid classification. No suggestions were applied.",
|
||||
sensitive_data_suggestion_provider_unavailable: "The selected model could not complete the request. No suggestions were applied.",
|
||||
sensitive_data_suggestion_failed: "Sensitive-field suggestions failed before review. No changes were applied.",
|
||||
sensitive_data_suggestion_history_request_invalid: "Sensitive suggestion history parameters are invalid.",
|
||||
sensitive_data_suggestion_history_failed: "Sensitive suggestion history could not be loaded.",
|
||||
sensitive_data_suggestion_run_not_found: "The sensitive suggestion run was not found.",
|
||||
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.",
|
||||
sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.",
|
||||
sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.",
|
||||
sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.",
|
||||
sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.",
|
||||
sensitive_data_suggestion_run_not_found: "The sensitivity analysis run was not found.",
|
||||
schema_sync_conflict: "A schema synchronization is already active or no longer current.",
|
||||
schema_introspection_failed: "The database schema could not be read safely.",
|
||||
schema_request_invalid: "The schema request is invalid.",
|
||||
|
||||
@@ -9,7 +9,7 @@ import type {
|
||||
CatalogSyncRun,
|
||||
CatalogTable,
|
||||
DescriptionGenerationRun,
|
||||
SensitiveDataSuggestionRun,
|
||||
SensitivityAnalysisRun,
|
||||
} from "../api/catalog-databases";
|
||||
import { Toaster } from "../components/ui/sonner";
|
||||
import { DatabaseManagementPage } from "./DatabaseManagementPage";
|
||||
@@ -159,18 +159,21 @@ function makeDescriptionGenerationRun(
|
||||
};
|
||||
}
|
||||
|
||||
function makeSensitiveDataSuggestionRun(
|
||||
overrides: Partial<SensitiveDataSuggestionRun> = {},
|
||||
): SensitiveDataSuggestionRun {
|
||||
function makeSensitivityAnalysisRun(
|
||||
overrides: Partial<SensitivityAnalysisRun> = {},
|
||||
): SensitivityAnalysisRun {
|
||||
return {
|
||||
id: "99999999-9999-4999-8999-999999999999",
|
||||
databaseId: "11111111-1111-4111-8111-111111111111",
|
||||
modelId: "local-qwen",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
scope: "selected_columns",
|
||||
status: "completed",
|
||||
total: 2,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
unknown: 0,
|
||||
createdAt: "2026-08-28T11:00:00Z",
|
||||
startedAt: "2026-08-28T11:00:00Z",
|
||||
updatedAt: "2026-08-28T11:00:01Z",
|
||||
@@ -505,7 +508,7 @@ test("keeps both run-history buttons visible beside the metadata-description sel
|
||||
name: "View description generation history",
|
||||
});
|
||||
const suggestionHistoryButton = within(metadataControls).getByRole("button", {
|
||||
name: "View sensitive suggestion history",
|
||||
name: "View sensitivity analysis history",
|
||||
});
|
||||
|
||||
expect(toolbar).toHaveClass("sm:items-end", "sm:justify-between");
|
||||
@@ -517,7 +520,7 @@ test("keeps both run-history buttons visible beside the metadata-description sel
|
||||
expect(descriptionHistoryButton).toHaveClass("disabled:opacity-70");
|
||||
expect(suggestionHistoryButton).toBeVisible();
|
||||
expect(suggestionHistoryButton).toBeEnabled();
|
||||
expect(suggestionHistoryButton).toHaveTextContent("View sensitive suggestion history");
|
||||
expect(suggestionHistoryButton).toHaveTextContent("View sensitivity analysis history");
|
||||
expect(toolbar.lastElementChild).toBe(actions);
|
||||
expect(actions).toHaveClass("sm:justify-end");
|
||||
expect(actions).toContainElement(screen.getByRole("button", { name: "Refresh" }));
|
||||
@@ -533,8 +536,8 @@ test("keeps both run-history buttons visible beside the metadata-description sel
|
||||
|
||||
await user.click(suggestionHistoryButton);
|
||||
expect(screen.queryByRole("dialog", { name: "Description generation" })).not.toBeInTheDocument();
|
||||
const suggestionDrawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
|
||||
expect(within(suggestionDrawer).getByText("No sensitive suggestion runs yet.")).toBeVisible();
|
||||
const suggestionDrawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
|
||||
expect(within(suggestionDrawer).getByText("No sensitivity analysis runs yet.")).toBeVisible();
|
||||
});
|
||||
|
||||
test("changes the metadata-description model in page-local state", async () => {
|
||||
@@ -1642,8 +1645,8 @@ test("observes an active run from another browser and reopens a terminal run fro
|
||||
expect(await within(drawer).findByRole("heading", { name: "Completed" })).toBeVisible();
|
||||
});
|
||||
|
||||
test("keeps the sensitive suggestion history label stable while showing active status separately", async () => {
|
||||
const runningRun = makeSensitiveDataSuggestionRun({
|
||||
test("keeps the sensitivity analysis history label stable while showing active status separately", async () => {
|
||||
const runningRun = makeSensitivityAnalysisRun({
|
||||
status: "running",
|
||||
finishedAt: null,
|
||||
updatedAt: new Date().toISOString(),
|
||||
@@ -1655,12 +1658,12 @@ test("keeps the sensitive suggestion history label stable while showing active s
|
||||
renderPage();
|
||||
|
||||
const historyButton = await screen.findByRole("button", {
|
||||
name: "View sensitive suggestion history",
|
||||
name: "View sensitivity analysis history",
|
||||
});
|
||||
expect(historyButton).toHaveTextContent("View sensitive suggestion history");
|
||||
expect(historyButton).toHaveTextContent("View sensitivity analysis history");
|
||||
await waitFor(() => expect(historyButton).toHaveAttribute(
|
||||
"title",
|
||||
"Sensitive suggestion generation is active",
|
||||
"Sensitivity analysis is active",
|
||||
));
|
||||
expect(historyButton.querySelector("[aria-hidden='true'].bg-primary")).not.toBeNull();
|
||||
});
|
||||
@@ -2015,7 +2018,7 @@ test("starts one selected column with the configured default model", async () =>
|
||||
expect(await screen.findByText("Description generation started for 1 column")).toBeVisible();
|
||||
});
|
||||
|
||||
test("shows database sensitive suggestions only for a selection and rejects multiple databases clearly", async () => {
|
||||
test("shows database sensitivity analysis only for a selection and rejects multiple databases clearly", async () => {
|
||||
const user = userEvent.setup();
|
||||
let suggestionCalls = 0;
|
||||
const radiology = makeDatabase({
|
||||
@@ -2035,24 +2038,24 @@ test("shows database sensitive suggestions only for a selection and rejects mult
|
||||
);
|
||||
renderPage({ rows: [makeDatabase(), radiology] });
|
||||
|
||||
expect(screen.queryByRole("button", { name: "Suggest sensitive fields" })).not.toBeInTheDocument();
|
||||
expect(screen.queryByRole("button", { name: "Analyze sensitive fields" })).not.toBeInTheDocument();
|
||||
const psdRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
|
||||
const radiologyRow = await screen.findByRole("row", { name: /Radiology/ });
|
||||
await user.click(within(psdRow).getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
expect(screen.getByRole("button", { name: "Suggest sensitive fields" })).toBeVisible();
|
||||
expect(screen.getByRole("button", { name: "Analyze sensitive fields" })).toBeVisible();
|
||||
await user.click(within(radiologyRow).getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
|
||||
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
|
||||
|
||||
expect(await screen.findByText("Sensitive-field suggestions can be requested for only one database at a time. Select one database and try again.")).toBeVisible();
|
||||
expect(await screen.findByText("Sensitivity analysis can run for only one database at a time. Select one database and try again.")).toBeVisible();
|
||||
expect(suggestionCalls).toBe(0);
|
||||
});
|
||||
|
||||
test("requests database-level sensitive suggestions for the only selected database", async () => {
|
||||
test("requests database-level sensitivity analysis for the only selected database", async () => {
|
||||
const user = userEvent.setup();
|
||||
let suggestionBody: unknown;
|
||||
let suggestionFinished = false;
|
||||
let historyCalls = 0;
|
||||
const run = makeSensitiveDataSuggestionRun({ scope: "all", total: 1 });
|
||||
const run = makeSensitivityAnalysisRun({ scope: "all", total: 1 });
|
||||
server.use(
|
||||
http.get("/api/catalog/metadata-generation/models", () => HttpResponse.json({
|
||||
models: [{ id: "local-qwen", label: "Local Qwen" }],
|
||||
@@ -2075,6 +2078,9 @@ test("requests database-level sensitive suggestions for the only selected databa
|
||||
version: patientIdColumn.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
}],
|
||||
});
|
||||
}),
|
||||
@@ -2083,14 +2089,14 @@ test("requests database-level sensitive suggestions for the only selected databa
|
||||
|
||||
const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
|
||||
await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
|
||||
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
|
||||
|
||||
await waitFor(() => expect(suggestionBody).toEqual({ modelId: "local-qwen", scope: "all" }));
|
||||
await waitFor(() => expect(suggestionBody).toEqual({ scope: "all" }));
|
||||
expect(await screen.findByRole("dialog", { name: "Sensitive field review" })).toBeVisible();
|
||||
await waitFor(() => expect(historyCalls).toBeGreaterThanOrEqual(2));
|
||||
});
|
||||
|
||||
test("refetches sensitive suggestion history after a failed request", async () => {
|
||||
test("refetches sensitivity analysis history after a failed request", async () => {
|
||||
const user = userEvent.setup();
|
||||
let historyCalls = 0;
|
||||
server.use(
|
||||
@@ -2104,8 +2110,8 @@ test("refetches sensitive suggestion history after a failed request", async () =
|
||||
}),
|
||||
http.post("/api/catalog/databases/:databaseId/sensitive-data-suggestions", () => (
|
||||
HttpResponse.json({
|
||||
code: "sensitive_data_suggestion_invalid_response",
|
||||
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
|
||||
code: "sensitivity_source_scan_failed",
|
||||
message: "Sensitivity analysis could not read the source. No assessments were applied.",
|
||||
}, { status: 502 })
|
||||
)),
|
||||
);
|
||||
@@ -2113,13 +2119,13 @@ test("refetches sensitive suggestion history after a failed request", async () =
|
||||
|
||||
const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ });
|
||||
await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
|
||||
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
|
||||
|
||||
await waitFor(() => expect(historyCalls).toBeGreaterThanOrEqual(2));
|
||||
expect(screen.queryByRole("dialog", { name: "Sensitive field review" })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
test("requests sensitive suggestions only for selected tables", async () => {
|
||||
test("requests sensitivity analysis only for selected tables", async () => {
|
||||
const user = userEvent.setup();
|
||||
const visitsTable: CatalogTable = {
|
||||
...patientsTable,
|
||||
@@ -2153,6 +2159,9 @@ test("requests sensitive suggestions only for selected tables", async () => {
|
||||
version: visitColumn.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
|
||||
observedValues: 0,
|
||||
}],
|
||||
});
|
||||
}),
|
||||
@@ -2163,17 +2172,16 @@ test("requests sensitive suggestions only for selected tables", async () => {
|
||||
await user.click(screen.getByRole("tab", { name: "Tables" }));
|
||||
const visitsRow = await screen.findByRole("row", { name: /visits/ });
|
||||
await user.click(within(visitsRow).getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
|
||||
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
|
||||
|
||||
await waitFor(() => expect(suggestionBody).toEqual({
|
||||
modelId: "local-qwen",
|
||||
scope: "selected_tables",
|
||||
targetIds: [visitsTable.id],
|
||||
}));
|
||||
expect(await screen.findByRole("dialog", { name: "Sensitive field review" })).toBeVisible();
|
||||
});
|
||||
|
||||
test("reviews AI-sensitive-field suggestions as an editable draft and saves only changed columns", async () => {
|
||||
test("allows a human downgrade and saves only explicit sensitivity changes", async () => {
|
||||
const user = userEvent.setup();
|
||||
const idColumn = { ...patientIdColumn, sensitive: false };
|
||||
const nameColumn = {
|
||||
@@ -2185,7 +2193,7 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
|
||||
isPrimaryKey: false,
|
||||
description: "Patient name",
|
||||
generatedDescription: "Name of the patient",
|
||||
sensitive: false,
|
||||
sensitive: true,
|
||||
};
|
||||
const unselectedColumn = {
|
||||
...patientIdColumn,
|
||||
@@ -2225,6 +2233,9 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
|
||||
version: idColumn.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
},
|
||||
{
|
||||
columnId: nameColumn.id,
|
||||
@@ -2232,8 +2243,11 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
|
||||
tableName: patientsTable.name,
|
||||
columnName: nameColumn.name,
|
||||
version: nameColumn.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
currentSensitive: true,
|
||||
sensitive: false,
|
||||
assessment: "non_sensitive",
|
||||
evidence: [],
|
||||
observedValues: 2,
|
||||
},
|
||||
],
|
||||
});
|
||||
@@ -2263,8 +2277,8 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
|
||||
const idSensitive = await screen.findByRole("checkbox", { name: "Sensitive data for id" });
|
||||
const nameSensitive = await screen.findByRole("checkbox", { name: "Sensitive data for name" });
|
||||
expect(idSensitive).not.toBeChecked();
|
||||
expect(nameSensitive).not.toBeChecked();
|
||||
expect(screen.queryByRole("button", { name: "Suggest sensitive fields" })).not.toBeInTheDocument();
|
||||
expect(nameSensitive).toBeChecked();
|
||||
expect(screen.queryByRole("button", { name: "Analyze sensitive fields" })).not.toBeInTheDocument();
|
||||
|
||||
const selectableRow = async (name: RegExp) => {
|
||||
const rows = await screen.findAllByRole("row", { name });
|
||||
@@ -2274,31 +2288,30 @@ test("reviews AI-sensitive-field suggestions as an editable draft and saves only
|
||||
.getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(within(await selectableRow(/Patient name/))
|
||||
.getByRole("checkbox", { name: /toggle row selection/i }));
|
||||
await user.click(screen.getByRole("button", { name: "Suggest sensitive fields" }));
|
||||
await user.click(screen.getByRole("button", { name: "Analyze sensitive fields" }));
|
||||
|
||||
await waitFor(() => expect(suggestionBody).toEqual({
|
||||
modelId: "local-qwen",
|
||||
scope: "selected_columns",
|
||||
targetIds: [idColumn.id, nameColumn.id],
|
||||
}));
|
||||
const review = await screen.findByRole("dialog", { name: "Sensitive field review" });
|
||||
expect(within(review).getByRole("checkbox", { name: "Protect patients.id" })).toBeChecked();
|
||||
expect(within(review).getByRole("checkbox", { name: "Protect patients.name" })).toBeChecked();
|
||||
expect(within(review).getByRole("checkbox", { name: "Protect patients.name" })).not.toBeChecked();
|
||||
expect(patches).toHaveLength(0);
|
||||
|
||||
await user.click(within(review).getByRole("checkbox", { name: "Protect patients.name" }));
|
||||
await user.click(within(review).getByRole("checkbox", { name: "Protect patients.id" }));
|
||||
await user.click(within(review).getByRole("button", { name: "Save 1" }));
|
||||
|
||||
await waitFor(() => expect(patches).toEqual([{
|
||||
columnId: idColumn.id,
|
||||
columnId: nameColumn.id,
|
||||
body: {
|
||||
version: idColumn.version,
|
||||
sensitive: true,
|
||||
version: nameColumn.version,
|
||||
sensitive: false,
|
||||
},
|
||||
}]));
|
||||
expect(await screen.findByText("Saved 1 sensitive flag")).toBeVisible();
|
||||
await waitFor(() => expect(screen.queryByRole("dialog", { name: "Sensitive field review" })).not.toBeInTheDocument());
|
||||
await waitFor(() => expect(screen.getByRole("checkbox", { name: "Sensitive data for id" })).toBeChecked());
|
||||
await waitFor(() => expect(screen.getByRole("checkbox", { name: "Sensitive data for id" })).not.toBeChecked());
|
||||
expect(screen.getByRole("checkbox", { name: "Sensitive data for name" })).not.toBeChecked();
|
||||
});
|
||||
|
||||
|
||||
@@ -18,11 +18,11 @@ import {
|
||||
listCatalogDatabases,
|
||||
listDescriptionGenerationRuns,
|
||||
listMetadataGenerationModels,
|
||||
listSensitiveDataSuggestionRuns,
|
||||
listSensitivityAnalysisRuns,
|
||||
replaceCatalogDatabaseSecrets,
|
||||
startCatalogSync,
|
||||
startDescriptionGenerationRun,
|
||||
suggestSensitiveFields,
|
||||
runSensitivityAnalysis,
|
||||
testCatalogDatabase,
|
||||
updateCatalogDatabase,
|
||||
type CatalogDatabase,
|
||||
@@ -34,9 +34,9 @@ import {
|
||||
type DatabaseBinding,
|
||||
type DatabaseTransport,
|
||||
type DescriptionGenerationRun,
|
||||
type SensitiveDataSuggestion,
|
||||
type SensitiveDataSuggestionRequest,
|
||||
type SensitiveDataSuggestionRun,
|
||||
type SensitivityReviewItem,
|
||||
type SensitivityAnalysisRequest,
|
||||
type SensitivityAnalysisRun,
|
||||
} from "../api/catalog-databases";
|
||||
import { DatabaseGrid } from "./database-management/DatabaseGrid";
|
||||
import { DatabaseForm } from "./database-management/DatabaseForm";
|
||||
@@ -46,7 +46,7 @@ import { DatabaseRelationships } from "./database-management/DatabaseRelationshi
|
||||
import { CatalogSyncDrawer } from "./database-management/CatalogSyncDrawer";
|
||||
import { MetadataGenerationModelSelector } from "./database-management/MetadataGenerationModelSelector";
|
||||
import { DescriptionGenerationDrawer } from "./database-management/DescriptionGenerationDrawer";
|
||||
import { SensitiveDataSuggestionHistoryDrawer } from "./database-management/SensitiveDataSuggestionHistoryDrawer";
|
||||
import { SensitivityAnalysisHistoryDrawer } from "./database-management/SensitivityAnalysisHistoryDrawer";
|
||||
import { SensitiveDataReviewDrawer } from "./database-management/SensitiveDataReviewDrawer";
|
||||
import {
|
||||
FleetLedgerBreadcrumb,
|
||||
@@ -158,15 +158,15 @@ export function DatabaseManagementPage({
|
||||
const observedActiveDescriptionGenerationRun = descriptionGenerationRuns.find(
|
||||
(run) => isDescriptionGenerationActive(run),
|
||||
);
|
||||
const sensitiveDataSuggestionRunsQuery = useQuery({
|
||||
const sensitivityAnalysisRunsQuery = useQuery({
|
||||
queryKey: SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY,
|
||||
queryFn: () => listSensitiveDataSuggestionRuns(50),
|
||||
queryFn: () => listSensitivityAnalysisRuns(50),
|
||||
enabled: canManage,
|
||||
retry: false,
|
||||
refetchInterval: 5_000,
|
||||
});
|
||||
const sensitiveDataSuggestionRuns = sensitiveDataSuggestionRunsQuery.data ?? [];
|
||||
const observedActiveSensitiveDataSuggestionRun = sensitiveDataSuggestionRuns.find(
|
||||
const sensitivityAnalysisRuns = sensitivityAnalysisRunsQuery.data ?? [];
|
||||
const observedActiveSensitivityAnalysisRun = sensitivityAnalysisRuns.find(
|
||||
(run) => run.status === "running",
|
||||
);
|
||||
|
||||
@@ -196,12 +196,12 @@ export function DatabaseManagementPage({
|
||||
const [syncDrawerOpen, setSyncDrawerOpen] = useState(false);
|
||||
const [activeDescriptionGenerationRun, setActiveDescriptionGenerationRun] = useState<DescriptionGenerationRun | null>(null);
|
||||
const [descriptionGenerationDrawerOpen, setDescriptionGenerationDrawerOpen] = useState(false);
|
||||
const [activeSensitiveDataSuggestionRun, setActiveSensitiveDataSuggestionRun] = useState<SensitiveDataSuggestionRun | null>(null);
|
||||
const [sensitiveDataSuggestionHistoryDrawerOpen, setSensitiveDataSuggestionHistoryDrawerOpen] = useState(false);
|
||||
const [activeSensitivityAnalysisRun, setActiveSensitivityAnalysisRun] = useState<SensitivityAnalysisRun | null>(null);
|
||||
const [sensitivityAnalysisHistoryDrawerOpen, setSensitivityAnalysisHistoryDrawerOpen] = useState(false);
|
||||
const [sensitiveReview, setSensitiveReview] = useState<{
|
||||
databaseId: string;
|
||||
scopeLabel: string;
|
||||
suggestions: SensitiveDataSuggestion[];
|
||||
suggestions: SensitivityReviewItem[];
|
||||
} | null>(null);
|
||||
|
||||
useEffect(() => {
|
||||
@@ -694,27 +694,27 @@ export function DatabaseManagementPage({
|
||||
?? scopedRuns[0]
|
||||
?? scopedSelectedRun;
|
||||
if (run) setActiveDescriptionGenerationRun(run);
|
||||
setSensitiveDataSuggestionHistoryDrawerOpen(false);
|
||||
setSensitivityAnalysisHistoryDrawerOpen(false);
|
||||
setDescriptionGenerationDrawerOpen(true);
|
||||
}, [activeDescriptionGenerationRun, activeRow, descriptionGenerationRuns, observedActiveDescriptionGenerationRun]);
|
||||
|
||||
const openSensitiveDataSuggestionHistory = useCallback(() => {
|
||||
const scopedRuns = activeRow ? sensitiveDataSuggestionRuns.filter((item) => item.databaseId === activeRow.id) : sensitiveDataSuggestionRuns;
|
||||
const scopedActiveRun = observedActiveSensitiveDataSuggestionRun
|
||||
&& (!activeRow || observedActiveSensitiveDataSuggestionRun.databaseId === activeRow.id)
|
||||
? observedActiveSensitiveDataSuggestionRun
|
||||
const openSensitivityReviewItemHistory = useCallback(() => {
|
||||
const scopedRuns = activeRow ? sensitivityAnalysisRuns.filter((item) => item.databaseId === activeRow.id) : sensitivityAnalysisRuns;
|
||||
const scopedActiveRun = observedActiveSensitivityAnalysisRun
|
||||
&& (!activeRow || observedActiveSensitivityAnalysisRun.databaseId === activeRow.id)
|
||||
? observedActiveSensitivityAnalysisRun
|
||||
: undefined;
|
||||
const scopedSelectedRun = activeSensitiveDataSuggestionRun
|
||||
&& (!activeRow || activeSensitiveDataSuggestionRun.databaseId === activeRow.id)
|
||||
? activeSensitiveDataSuggestionRun
|
||||
const scopedSelectedRun = activeSensitivityAnalysisRun
|
||||
&& (!activeRow || activeSensitivityAnalysisRun.databaseId === activeRow.id)
|
||||
? activeSensitivityAnalysisRun
|
||||
: undefined;
|
||||
const run = scopedActiveRun
|
||||
?? scopedRuns[0]
|
||||
?? scopedSelectedRun;
|
||||
if (run) setActiveSensitiveDataSuggestionRun(run);
|
||||
if (run) setActiveSensitivityAnalysisRun(run);
|
||||
setDescriptionGenerationDrawerOpen(false);
|
||||
setSensitiveDataSuggestionHistoryDrawerOpen(true);
|
||||
}, [activeRow, activeSensitiveDataSuggestionRun, observedActiveSensitiveDataSuggestionRun, sensitiveDataSuggestionRuns]);
|
||||
setSensitivityAnalysisHistoryDrawerOpen(true);
|
||||
}, [activeRow, activeSensitivityAnalysisRun, observedActiveSensitivityAnalysisRun, sensitivityAnalysisRuns]);
|
||||
|
||||
const descriptionGenerationTerminated = useCallback(async (run: DescriptionGenerationRun) => {
|
||||
await Promise.all([
|
||||
@@ -788,39 +788,35 @@ export function DatabaseManagementPage({
|
||||
|
||||
const requestSensitiveSuggestions = useCallback(async (
|
||||
database: CatalogDatabase,
|
||||
selection: SensitiveDataSuggestionRequest,
|
||||
selection: SensitivityAnalysisRequest,
|
||||
scopeLabel: string,
|
||||
) => {
|
||||
if (!database.id) {
|
||||
toast.error("The selected database is not configured, so sensitive-field suggestions were not requested.");
|
||||
toast.error("The selected database is not configured, so sensitivity analysis was not started.");
|
||||
throw new Error("database is not configured");
|
||||
}
|
||||
if (!selectedMetadataModel) {
|
||||
toast.error("Select a metadata-generation model before requesting sensitive-field suggestions.");
|
||||
throw new Error("metadata-generation model is not selected");
|
||||
}
|
||||
try {
|
||||
const result = await suggestSensitiveFields(database.id, selectedMetadataModel, selection);
|
||||
const result = await runSensitivityAnalysis(database.id, selection);
|
||||
if (result.run) {
|
||||
setActiveSensitiveDataSuggestionRun(result.run);
|
||||
queryClient.setQueryData<SensitiveDataSuggestionRun[]>(
|
||||
setActiveSensitivityAnalysisRun(result.run);
|
||||
queryClient.setQueryData<SensitivityAnalysisRun[]>(
|
||||
SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY,
|
||||
(current = []) => [result.run, ...current.filter((item) => item.id !== result.run.id)],
|
||||
);
|
||||
}
|
||||
setSensitiveReview({ databaseId: database.id, scopeLabel, suggestions: result.suggestions });
|
||||
toast.success(`Prepared ${result.suggestions.length} sensitive-field suggestion${result.suggestions.length === 1 ? "" : "s"} for review`);
|
||||
toast.success(`Prepared ${result.suggestions.length} local sensitivity assessment${result.suggestions.length === 1 ? "" : "s"} for review`);
|
||||
} catch (error) {
|
||||
toast.error(apiErrorMessage(error));
|
||||
throw error;
|
||||
} finally {
|
||||
await queryClient.invalidateQueries({ queryKey: SENSITIVE_DATA_SUGGESTION_HISTORY_QUERY_KEY });
|
||||
}
|
||||
}, [queryClient, selectedMetadataModel]);
|
||||
}, [queryClient]);
|
||||
|
||||
const suggestDatabaseSensitiveFields = useCallback(async (selected: CatalogDatabase[]) => {
|
||||
if (selected.length !== 1) {
|
||||
toast.error("Sensitive-field suggestions can be requested for only one database at a time. Select one database and try again.");
|
||||
toast.error("Sensitivity analysis can run for only one database at a time. Select one database and try again.");
|
||||
throw new Error("more than one database selected");
|
||||
}
|
||||
const database = selected[0]!;
|
||||
@@ -828,11 +824,11 @@ export function DatabaseManagementPage({
|
||||
}, [requestSensitiveSuggestions]);
|
||||
|
||||
const suggestActiveDatabaseSensitiveFields = useCallback(async (
|
||||
selection: SensitiveDataSuggestionRequest,
|
||||
selection: SensitivityAnalysisRequest,
|
||||
scopeLabel: string,
|
||||
) => {
|
||||
if (!activeRow) {
|
||||
toast.error("The database is no longer available, so sensitive-field suggestions were not requested.");
|
||||
toast.error("The database is no longer available, so sensitivity analysis was not started.");
|
||||
throw new Error("database is no longer available");
|
||||
}
|
||||
await requestSensitiveSuggestions(activeRow, selection, scopeLabel);
|
||||
@@ -952,10 +948,10 @@ export function DatabaseManagementPage({
|
||||
? `Synchronization: ${currentActiveRun.phase.replaceAll("_", " ")}`
|
||||
: descriptionGenerationActive
|
||||
? "Description generation active"
|
||||
: observedActiveSensitiveDataSuggestionRun
|
||||
: observedActiveSensitivityAnalysisRun
|
||||
? "Sensitive analysis active"
|
||||
: "Idle";
|
||||
const operationTone = currentActiveRun || descriptionGenerationActive || observedActiveSensitiveDataSuggestionRun
|
||||
const operationTone = currentActiveRun || descriptionGenerationActive || observedActiveSensitivityAnalysisRun
|
||||
? "warning" as const
|
||||
: "neutral" as const;
|
||||
const fleetBreadcrumb = screen.kind === "tables" || screen.kind === "relationships"
|
||||
@@ -987,15 +983,12 @@ export function DatabaseManagementPage({
|
||||
onRunUpdate={setActiveDescriptionGenerationRun}
|
||||
onTerminal={(run) => void descriptionGenerationTerminated(run)}
|
||||
/>
|
||||
<SensitiveDataSuggestionHistoryDrawer
|
||||
open={sensitiveDataSuggestionHistoryDrawerOpen}
|
||||
<SensitivityAnalysisHistoryDrawer
|
||||
open={sensitivityAnalysisHistoryDrawerOpen}
|
||||
databaseId={activeRow?.id ?? null}
|
||||
run={activeSensitiveDataSuggestionRun}
|
||||
modelLabel={metadataModels.find(
|
||||
(model) => model.id === activeSensitiveDataSuggestionRun?.modelId,
|
||||
)?.label ?? activeSensitiveDataSuggestionRun?.modelId ?? ""}
|
||||
onClose={() => setSensitiveDataSuggestionHistoryDrawerOpen(false)}
|
||||
onRunUpdate={setActiveSensitiveDataSuggestionRun}
|
||||
run={activeSensitivityAnalysisRun}
|
||||
onClose={() => setSensitivityAnalysisHistoryDrawerOpen(false)}
|
||||
onRunUpdate={setActiveSensitivityAnalysisRun}
|
||||
/>
|
||||
<SensitiveDataReviewDrawer
|
||||
open={Boolean(sensitiveReview)}
|
||||
@@ -1054,7 +1047,7 @@ export function DatabaseManagementPage({
|
||||
onSync={(scope) => void syncDatabase(scope)}
|
||||
onOpenSync={() => openSync(activeRow)}
|
||||
onOpenDescriptionHistory={openDescriptionGenerationHistory}
|
||||
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
|
||||
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
|
||||
activeSyncRun={currentActiveRun}
|
||||
/>
|
||||
</FleetLedgerDrawer>
|
||||
@@ -1091,7 +1084,7 @@ export function DatabaseManagementPage({
|
||||
onRunStarted={rememberSyncRun}
|
||||
onOpenSync={() => openSync(activeRow)}
|
||||
onOpenDescriptionHistory={openDescriptionGenerationHistory}
|
||||
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
|
||||
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
|
||||
onCatalogMetricsChanged={invalidateCatalogMetrics}
|
||||
onNavigationStateChange={setTablesNavigationState}
|
||||
/>
|
||||
@@ -1128,7 +1121,7 @@ export function DatabaseManagementPage({
|
||||
descriptionGenerationActive={descriptionGenerationActive}
|
||||
onGenerateDescriptions={generateDatabaseDescriptions}
|
||||
onSuggestSensitive={suggestDatabaseSensitiveFields}
|
||||
sensitiveDataSuggestionRuns={sensitiveDataSuggestionRuns}
|
||||
sensitivityAnalysisRuns={sensitivityAnalysisRuns}
|
||||
onDeleteMetadataSelected={deleteSelectedMetadata}
|
||||
onRefresh={refreshList}
|
||||
/>
|
||||
@@ -1194,10 +1187,10 @@ export function DatabaseManagementPage({
|
||||
aria-label="View sensitive analysis progress"
|
||||
title="View sensitive analysis progress and history"
|
||||
disabled={!canManage}
|
||||
onClick={openSensitiveDataSuggestionHistory}
|
||||
onClick={openSensitivityReviewItemHistory}
|
||||
>
|
||||
<History aria-hidden="true" />
|
||||
{observedActiveSensitiveDataSuggestionRun ? "Sensitive analysis active" : "Sensitive analysis history"}
|
||||
{observedActiveSensitivityAnalysisRun ? "Sensitive analysis active" : "Sensitive analysis history"}
|
||||
</Button>
|
||||
</>
|
||||
)}
|
||||
@@ -1276,10 +1269,10 @@ export function DatabaseManagementPage({
|
||||
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View description generation history" disabled={!canManage} title="View description generation history" onClick={openDescriptionGenerationHistory}>
|
||||
<History /> View description generation history
|
||||
</Button>
|
||||
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View sensitive suggestion history" disabled={!canManage} title={observedActiveSensitiveDataSuggestionRun ? "Sensitive suggestion generation is active" : "View sensitive suggestion history"} onClick={openSensitiveDataSuggestionHistory}>
|
||||
<Button type="button" variant="outline" className="whitespace-nowrap disabled:opacity-70" aria-label="View sensitivity analysis history" disabled={!canManage} title={observedActiveSensitivityAnalysisRun ? "Sensitivity analysis is active" : "View sensitivity analysis history"} onClick={openSensitivityReviewItemHistory}>
|
||||
<History />
|
||||
{observedActiveSensitiveDataSuggestionRun ? <span aria-hidden="true" className="size-2 rounded-full bg-primary" /> : null}
|
||||
View sensitive suggestion history
|
||||
{observedActiveSensitivityAnalysisRun ? <span aria-hidden="true" className="size-2 rounded-full bg-primary" /> : null}
|
||||
View sensitivity analysis history
|
||||
</Button>
|
||||
</> : null}
|
||||
</div>
|
||||
@@ -1333,7 +1326,7 @@ export function DatabaseManagementPage({
|
||||
onSyncSelected={syncSelected}
|
||||
selectedMetadataModel={selectedMetadataModelAvailable ? selectedMetadataModel : null}
|
||||
descriptionGenerationActive={descriptionGenerationActive}
|
||||
sensitiveDataSuggestionRuns={sensitiveDataSuggestionRuns}
|
||||
sensitivityAnalysisRuns={sensitivityAnalysisRuns}
|
||||
onGenerateDescriptions={generateDatabaseDescriptions}
|
||||
onSuggestSensitive={suggestDatabaseSensitiveFields}
|
||||
onDeleteMetadataSelected={deleteSelectedMetadata}
|
||||
@@ -1370,7 +1363,7 @@ export function DatabaseManagementPage({
|
||||
onSync={(scope) => void syncDatabase(scope)}
|
||||
onOpenSync={() => openSync(activeRow)}
|
||||
onOpenDescriptionHistory={openDescriptionGenerationHistory}
|
||||
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
|
||||
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
|
||||
activeSyncRun={currentActiveRun}
|
||||
/>
|
||||
) : null}
|
||||
@@ -1405,7 +1398,7 @@ export function DatabaseManagementPage({
|
||||
onRunStarted={rememberSyncRun}
|
||||
onOpenSync={() => openSync(activeRow)}
|
||||
onOpenDescriptionHistory={openDescriptionGenerationHistory}
|
||||
onOpenSensitiveHistory={openSensitiveDataSuggestionHistory}
|
||||
onOpenSensitiveHistory={openSensitivityReviewItemHistory}
|
||||
onCatalogMetricsChanged={invalidateCatalogMetrics}
|
||||
onNavigationStateChange={setTablesNavigationState}
|
||||
/>
|
||||
@@ -1431,15 +1424,12 @@ export function DatabaseManagementPage({
|
||||
onRunUpdate={setActiveDescriptionGenerationRun}
|
||||
onTerminal={(run) => void descriptionGenerationTerminated(run)}
|
||||
/>
|
||||
<SensitiveDataSuggestionHistoryDrawer
|
||||
open={sensitiveDataSuggestionHistoryDrawerOpen}
|
||||
<SensitivityAnalysisHistoryDrawer
|
||||
open={sensitivityAnalysisHistoryDrawerOpen}
|
||||
databaseId={activeRow?.id ?? null}
|
||||
run={activeSensitiveDataSuggestionRun}
|
||||
modelLabel={metadataModels.find(
|
||||
(model) => model.id === activeSensitiveDataSuggestionRun?.modelId,
|
||||
)?.label ?? activeSensitiveDataSuggestionRun?.modelId ?? ""}
|
||||
onClose={() => setSensitiveDataSuggestionHistoryDrawerOpen(false)}
|
||||
onRunUpdate={setActiveSensitiveDataSuggestionRun}
|
||||
run={activeSensitivityAnalysisRun}
|
||||
onClose={() => setSensitivityAnalysisHistoryDrawerOpen(false)}
|
||||
onRunUpdate={setActiveSensitivityAnalysisRun}
|
||||
/>
|
||||
<SensitiveDataReviewDrawer
|
||||
open={Boolean(sensitiveReview)}
|
||||
|
||||
@@ -17,7 +17,7 @@ import {
|
||||
type CatalogSyncRun,
|
||||
type CatalogTable,
|
||||
type DescriptionGenerationRun,
|
||||
type SensitiveDataSuggestionRequest,
|
||||
type SensitivityAnalysisRequest,
|
||||
} from "../../api/catalog-databases";
|
||||
import type { DatabaseNavigationState } from "./model";
|
||||
import { FleetActionSelector, type FleetActionOption } from "./FleetActionSelector";
|
||||
@@ -34,7 +34,7 @@ interface Props {
|
||||
onDescriptionGenerationRunStarted: (run: DescriptionGenerationRun) => void;
|
||||
onCatalogSyncRunStarted?: (run: CatalogSyncRun) => void;
|
||||
onNavigationStateChange: (state: DatabaseNavigationState) => void;
|
||||
onSuggestSensitive: (selection: SensitiveDataSuggestionRequest, scopeLabel: string) => Promise<void>;
|
||||
onSuggestSensitive: (selection: SensitivityAnalysisRequest, scopeLabel: string) => Promise<void>;
|
||||
catalogOperationActive?: boolean;
|
||||
onCatalogMetricsChanged?: () => void | Promise<void>;
|
||||
presentation?: "legacy" | "fleet";
|
||||
@@ -351,17 +351,15 @@ export function DatabaseColumns({
|
||||
},
|
||||
{
|
||||
id: "suggest-sensitive",
|
||||
label: "Suggest sensitive fields",
|
||||
label: "Analyze sensitive fields",
|
||||
group: "Sensitive data",
|
||||
runLabel: "Suggest",
|
||||
disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || catalogOperationActive || busy,
|
||||
runLabel: "Analyze",
|
||||
disabled: !canManage || selectedIds.length === 0 || descriptionGenerationActive || catalogOperationActive || busy,
|
||||
disabledReason: !canManage
|
||||
? "You do not have permission to review sensitive data."
|
||||
: selectedIds.length === 0
|
||||
? "Select at least one column."
|
||||
: !selectedMetadataModel
|
||||
? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE
|
||||
: descriptionGenerationActive || catalogOperationActive
|
||||
: descriptionGenerationActive || catalogOperationActive
|
||||
? "Wait for the active catalog operation to finish."
|
||||
: busy
|
||||
? "Another action is running."
|
||||
@@ -504,7 +502,7 @@ export function DatabaseColumns({
|
||||
</Menu.Positioner>
|
||||
</Menu.Portal>
|
||||
</Menu.Root>
|
||||
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{sensitiveAction === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}</Button>
|
||||
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{sensitiveAction === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}</Button>
|
||||
{changedSensitiveColumns.length > 0 ? <Button type="button" disabled={!canManage || busy} onClick={() => void saveSensitive()}><Save />{sensitiveAction === "save" ? "Saving…" : "Save sensitive fields"}</Button> : null}
|
||||
<Button type="button" variant="ghost" onClick={clearSelection}><X />Clear</Button>
|
||||
</>
|
||||
|
||||
@@ -16,7 +16,7 @@ import { getCatalogMetrics } from "../../api/catalog-databases";
|
||||
import type {
|
||||
CatalogDatabase,
|
||||
CatalogMetrics,
|
||||
SensitiveDataSuggestionRun,
|
||||
SensitivityAnalysisRun,
|
||||
CatalogDatabaseMetadataDeleteTarget,
|
||||
DescriptionGenerationScope,
|
||||
} from "../../api/catalog-databases";
|
||||
@@ -45,7 +45,7 @@ interface DatabaseGridProps {
|
||||
onSyncSelected: (rows: CatalogDatabase[], scope: DatabaseSyncScope) => Promise<void>;
|
||||
selectedMetadataModel: string | null;
|
||||
descriptionGenerationActive: boolean;
|
||||
sensitiveDataSuggestionRuns?: SensitiveDataSuggestionRun[];
|
||||
sensitivityAnalysisRuns?: SensitivityAnalysisRun[];
|
||||
onGenerateDescriptions: (
|
||||
rows: CatalogDatabase[],
|
||||
scope: Extract<DescriptionGenerationScope, "all" | "missing">,
|
||||
@@ -161,7 +161,7 @@ function coverageStatus(total: number, complete: number): { label: string; detai
|
||||
return { label: "Not started", detail: `0/${total}`, tone: "danger" };
|
||||
}
|
||||
|
||||
function CatalogStatusCells({ row, metrics, sensitiveRun }: { row: CatalogDatabase; metrics?: CatalogMetrics; sensitiveRun?: SensitiveDataSuggestionRun }) {
|
||||
function CatalogStatusCells({ row, metrics, sensitiveRun }: { row: CatalogDatabase; metrics?: CatalogMetrics; sensitiveRun?: SensitivityAnalysisRun }) {
|
||||
const access = accessSummary(row);
|
||||
const synchronized = row.schemaSyncedVersion === row.version;
|
||||
const syncStatus = !row.schemaSyncedVersion
|
||||
@@ -337,7 +337,7 @@ export function DatabaseGrid({
|
||||
onSyncSelected,
|
||||
selectedMetadataModel,
|
||||
descriptionGenerationActive,
|
||||
sensitiveDataSuggestionRuns = [],
|
||||
sensitivityAnalysisRuns = [],
|
||||
onGenerateDescriptions,
|
||||
onSuggestSensitive,
|
||||
onDeleteMetadataSelected,
|
||||
@@ -361,8 +361,8 @@ export function DatabaseGrid({
|
||||
return result;
|
||||
}, [metricQueries, rows]);
|
||||
const sensitiveRunsByDatabase = useMemo(() => new Map(
|
||||
rows.filter((row) => row.id).map((row) => [row.id!, sensitiveDataSuggestionRuns.find((run) => run.databaseId === row.id)]),
|
||||
), [rows, sensitiveDataSuggestionRuns]);
|
||||
rows.filter((row) => row.id).map((row) => [row.id!, sensitivityAnalysisRuns.find((run) => run.databaseId === row.id)]),
|
||||
), [rows, sensitivityAnalysisRuns]);
|
||||
const compact = useCompactViewport();
|
||||
const gridRef = useRef<AgGridReact<CatalogDatabase>>(null);
|
||||
const actionsTriggerRef = useRef<HTMLButtonElement>(null);
|
||||
@@ -432,7 +432,6 @@ export function DatabaseGrid({
|
||||
&& selectedRows.every((row) => row.configured && row.id && !row.activeSyncRun);
|
||||
const canSuggestSensitive = canManage
|
||||
&& selectedRows.length === 1
|
||||
&& Boolean(selectedMetadataModel)
|
||||
&& !descriptionGenerationActive
|
||||
&& selectedRows.every((row) => row.configured && row.id && !row.activeSyncRun);
|
||||
const canDeleteMetadataSelection = canManage && selectedRows.length > 0
|
||||
@@ -462,7 +461,6 @@ export function DatabaseGrid({
|
||||
: undefined);
|
||||
const sensitiveReason = permissionReason
|
||||
?? (selectedRows.length !== 1 ? "Select exactly one database" : undefined)
|
||||
?? (!selectedMetadataModel ? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE : undefined)
|
||||
?? (descriptionGenerationActive ? "Wait for the active metadata operation" : undefined)
|
||||
?? (selectedRows.some((row) => !row.configured || !row.id || row.activeSyncRun)
|
||||
? "The selected database must be configured and idle"
|
||||
@@ -479,7 +477,7 @@ export function DatabaseGrid({
|
||||
{ id: "sync-all", label: "Synchronize all", group: "Synchronization", disabled: !canSyncSelection, disabledReason: syncReason },
|
||||
{ id: "generate-missing", label: "Generate missing descriptions", group: "Descriptions", disabled: !canGenerateDescriptions, disabledReason: generationReason },
|
||||
{ id: "generate-all", label: "Generate all descriptions", group: "Descriptions", disabled: !canGenerateDescriptions, disabledReason: generationReason, runLabel: "Review generation" },
|
||||
{ id: "suggest-sensitive", label: "Suggest sensitive fields", group: "Sensitive data", disabled: !canSuggestSensitive, disabledReason: sensitiveReason, runLabel: "Open review" },
|
||||
{ id: "suggest-sensitive", label: "Analyze sensitive fields", group: "Sensitive data", disabled: !canSuggestSensitive, disabledReason: sensitiveReason, runLabel: "Open review" },
|
||||
{ id: "clear-tables", label: "Clear catalog tables", group: "Cleanup", disabled: !canDeleteMetadataSelection, disabledReason: cleanupReason, tone: "destructive", runLabel: "Review cleanup" },
|
||||
{ id: "clear-relationships", label: "Clear catalog relationships", group: "Cleanup", disabled: !canDeleteMetadataSelection, disabledReason: cleanupReason, tone: "destructive", runLabel: "Review cleanup" },
|
||||
];
|
||||
@@ -723,7 +721,7 @@ export function DatabaseGrid({
|
||||
title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined}
|
||||
onClick={() => void perform("suggest", () => onSuggestSensitive(selectedRows))}
|
||||
>
|
||||
<Sparkles />{action === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}
|
||||
<Sparkles />{action === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}
|
||||
</Button>
|
||||
<Button type="button" variant="ghost" disabled={action !== null} onClick={() => { gridRef.current?.api.deselectAll(); setSelectedRows([]); }}><X />Clear</Button>
|
||||
</>
|
||||
|
||||
@@ -20,7 +20,7 @@ import {
|
||||
type CatalogTable,
|
||||
type CatalogTableMetadataDeleteTarget,
|
||||
type DescriptionGenerationRun,
|
||||
type SensitiveDataSuggestionRequest,
|
||||
type SensitivityAnalysisRequest,
|
||||
} from "../../api/catalog-databases";
|
||||
import type { DatabaseNavigationState } from "./model";
|
||||
import { DatabaseColumns } from "./DatabaseColumns";
|
||||
@@ -43,7 +43,7 @@ interface Props {
|
||||
onOpenSync: () => void;
|
||||
onClearCatalogTables?: () => Promise<void>;
|
||||
onDescriptionGenerationRunStarted: (run: DescriptionGenerationRun) => void;
|
||||
onSuggestSensitive: (selection: SensitiveDataSuggestionRequest, scopeLabel: string) => Promise<void>;
|
||||
onSuggestSensitive: (selection: SensitivityAnalysisRequest, scopeLabel: string) => Promise<void>;
|
||||
onCatalogMetricsChanged?: () => void | Promise<void>;
|
||||
presentation?: "legacy" | "fleet";
|
||||
}
|
||||
@@ -418,17 +418,15 @@ export function DatabaseTables({
|
||||
},
|
||||
{
|
||||
id: "suggest-sensitive",
|
||||
label: "Suggest sensitive fields",
|
||||
label: "Analyze sensitive fields",
|
||||
group: "Sensitive data",
|
||||
runLabel: "Suggest",
|
||||
disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || busy !== null || Boolean(currentRun),
|
||||
runLabel: "Analyze",
|
||||
disabled: !canManage || selectedIds.length === 0 || descriptionGenerationActive || busy !== null || Boolean(currentRun),
|
||||
disabledReason: !canManage
|
||||
? "You do not have permission to review sensitive data."
|
||||
: selectedIds.length === 0
|
||||
? "Select at least one table."
|
||||
: !selectedMetadataModel
|
||||
? NO_METADATA_GENERATION_LLM_MODEL_MESSAGE
|
||||
: descriptionGenerationActive || Boolean(currentRun)
|
||||
: descriptionGenerationActive || Boolean(currentRun)
|
||||
? "Wait for the active catalog operation to finish."
|
||||
: busy !== null
|
||||
? "Another action is running."
|
||||
@@ -752,7 +750,7 @@ export function DatabaseTables({
|
||||
</Menu.Positioner>
|
||||
</Menu.Portal>
|
||||
</Menu.Root>
|
||||
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy !== null} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{busy === "suggest" ? "Suggesting…" : "Suggest sensitive fields"}</Button>
|
||||
<Button type="button" variant="outline" disabled={!canManage || descriptionGenerationActive || busy !== null} title={descriptionGenerationActive ? "Wait for the active description generation to finish" : undefined} onClick={() => void suggestSensitive()}><Sparkles />{busy === "suggest" ? "Analyzing…" : "Analyze sensitive fields"}</Button>
|
||||
<Button type="button" variant="ghost" onClick={clearSelection}><X />Clear</Button>
|
||||
</>
|
||||
)
|
||||
|
||||
@@ -8,7 +8,7 @@ describe("Recent runs success styling contract", () => {
|
||||
test("every history renderer exposes its run status to the shared styles", () => {
|
||||
const renderers = [
|
||||
"DescriptionGenerationDrawer.tsx",
|
||||
"SensitiveDataSuggestionHistoryDrawer.tsx",
|
||||
"SensitivityAnalysisHistoryDrawer.tsx",
|
||||
"CatalogSyncDrawer.tsx",
|
||||
];
|
||||
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
import { render, screen, waitFor } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { useState } from "react";
|
||||
import { vi } from "vitest";
|
||||
import type { CatalogColumn, SensitivityReviewItem } from "../../api/catalog-databases";
|
||||
import { SensitiveDataReviewDrawer } from "./SensitiveDataReviewDrawer";
|
||||
|
||||
const updateCatalogColumnSensitive = vi.hoisted(() => vi.fn());
|
||||
|
||||
vi.mock("../../api/catalog-databases", () => ({ updateCatalogColumnSensitive }));
|
||||
|
||||
const suggestions: SensitivityReviewItem[] = [
|
||||
{
|
||||
columnId: "11111111-1111-4111-8111-111111111111",
|
||||
tableId: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
|
||||
tableName: "patients",
|
||||
columnName: "email",
|
||||
version: 3,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
},
|
||||
{
|
||||
columnId: "22222222-2222-4222-8222-222222222222",
|
||||
tableId: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
|
||||
tableName: "patients",
|
||||
columnName: "notes",
|
||||
version: 7,
|
||||
currentSensitive: true,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
|
||||
observedValues: 0,
|
||||
},
|
||||
];
|
||||
|
||||
function savedColumn(suggestion: SensitivityReviewItem): CatalogColumn {
|
||||
return {
|
||||
id: suggestion.columnId,
|
||||
tableId: suggestion.tableId,
|
||||
name: suggestion.columnName,
|
||||
ordinalPosition: 1,
|
||||
dataType: "text",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: true,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: suggestion.version + 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:01Z",
|
||||
};
|
||||
}
|
||||
|
||||
test("preserves failed human choices when another flag in the same save succeeds", async () => {
|
||||
updateCatalogColumnSensitive
|
||||
.mockResolvedValueOnce(savedColumn(suggestions[0]!))
|
||||
.mockRejectedValueOnce(new Error("write failed"));
|
||||
|
||||
function Harness() {
|
||||
const [current, setCurrent] = useState(suggestions);
|
||||
return (
|
||||
<SensitiveDataReviewDrawer
|
||||
open
|
||||
databaseId="bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb"
|
||||
scopeLabel="All columns"
|
||||
suggestions={current}
|
||||
canManage
|
||||
onClose={() => undefined}
|
||||
onSaved={(columns) => setCurrent((items) => items.map((item) => {
|
||||
const saved = columns.find((column) => column.id === item.columnId);
|
||||
return saved ? { ...item, currentSensitive: saved.sensitive, version: saved.version } : item;
|
||||
}))}
|
||||
/>
|
||||
);
|
||||
}
|
||||
|
||||
const user = userEvent.setup();
|
||||
render(<Harness />);
|
||||
await user.click(screen.getByRole("checkbox", { name: "Show all 2 assessed columns" }));
|
||||
const failedChoice = screen.getByRole("checkbox", { name: "Protect patients.notes" });
|
||||
await user.click(failedChoice);
|
||||
await user.click(screen.getByRole("button", { name: "Save 2" }));
|
||||
|
||||
await waitFor(() => expect(updateCatalogColumnSensitive).toHaveBeenCalledTimes(2));
|
||||
await waitFor(() => expect(screen.getByRole("button", { name: "Save 1" })).toBeEnabled());
|
||||
expect(failedChoice).not.toBeChecked();
|
||||
});
|
||||
@@ -1,4 +1,4 @@
|
||||
import { useEffect, useMemo, useState } from "react";
|
||||
import { useEffect, useMemo, useRef, useState } from "react";
|
||||
import { Save } from "lucide-react";
|
||||
import { toast } from "sonner";
|
||||
import { Button } from "../../components/ui/button";
|
||||
@@ -6,7 +6,7 @@ import { apiErrorMessage } from "../../api/client";
|
||||
import {
|
||||
updateCatalogColumnSensitive,
|
||||
type CatalogColumn,
|
||||
type SensitiveDataSuggestion,
|
||||
type SensitivityReviewItem,
|
||||
} from "../../api/catalog-databases";
|
||||
import { FleetLedgerDrawer } from "./FleetLedgerShell";
|
||||
|
||||
@@ -14,7 +14,7 @@ interface Props {
|
||||
open: boolean;
|
||||
databaseId: string | null;
|
||||
scopeLabel: string;
|
||||
suggestions: SensitiveDataSuggestion[];
|
||||
suggestions: SensitivityReviewItem[];
|
||||
canManage: boolean;
|
||||
onClose: () => void;
|
||||
onSaved: (columns: CatalogColumn[]) => void;
|
||||
@@ -33,16 +33,26 @@ export function SensitiveDataReviewDrawer({
|
||||
const [search, setSearch] = useState("");
|
||||
const [showAll, setShowAll] = useState(false);
|
||||
const [saving, setSaving] = useState(false);
|
||||
const initializedReview = useRef<string | null>(null);
|
||||
const reviewKey = useMemo(
|
||||
() => suggestions.map((suggestion) => suggestion.columnId).join(":"),
|
||||
[suggestions],
|
||||
);
|
||||
|
||||
useEffect(() => {
|
||||
if (!open) return;
|
||||
if (!open) {
|
||||
initializedReview.current = null;
|
||||
return;
|
||||
}
|
||||
if (initializedReview.current === reviewKey) return;
|
||||
initializedReview.current = reviewKey;
|
||||
setDrafts(Object.fromEntries(suggestions.map((suggestion) => [
|
||||
suggestion.columnId,
|
||||
suggestion.sensitive,
|
||||
])));
|
||||
setSearch("");
|
||||
setShowAll(false);
|
||||
}, [open, suggestions]);
|
||||
}, [open, reviewKey, suggestions]);
|
||||
|
||||
const changed = useMemo(() => suggestions.filter((suggestion) => (
|
||||
drafts[suggestion.columnId] !== undefined
|
||||
@@ -92,8 +102,8 @@ export function SensitiveDataReviewDrawer({
|
||||
open
|
||||
ariaLabel="Sensitive field review"
|
||||
eyebrow="Sensitive data"
|
||||
title="Review suggested flags"
|
||||
description={`${scopeLabel}. The model proposed values, but only your save changes the catalog.`}
|
||||
title="Review local assessments"
|
||||
description={`${scopeLabel}. Local rules proposed values, but only your save changes the catalog.`}
|
||||
onClose={close}
|
||||
closeLabel="Close sensitive field review"
|
||||
busy={saving}
|
||||
@@ -101,7 +111,7 @@ export function SensitiveDataReviewDrawer({
|
||||
footerClassName="thot-catalog-drawer__footer--split"
|
||||
footer={(
|
||||
<>
|
||||
<p className="text-xs text-muted-foreground">Unsaved suggestions never change the catalog.</p>
|
||||
<p className="text-xs text-muted-foreground">Unsaved assessments never change the catalog.</p>
|
||||
<div className="flex gap-2">
|
||||
<Button type="button" variant="outline" disabled={saving} onClick={close}>Cancel</Button>
|
||||
<Button type="button" disabled={!canManage || saving || changed.length === 0} onClick={() => void save()}>
|
||||
@@ -115,7 +125,7 @@ export function SensitiveDataReviewDrawer({
|
||||
<div className="flex items-center gap-3">
|
||||
<input
|
||||
className="h-9 min-w-0 flex-1 rounded-md border border-input bg-background px-3 text-sm outline-none focus:border-primary/60 focus:ring-3 focus:ring-ring/15"
|
||||
aria-label="Search sensitive field suggestions"
|
||||
aria-label="Search sensitivity assessments"
|
||||
placeholder="Search table or column"
|
||||
value={search}
|
||||
onChange={(event) => setSearch(event.target.value)}
|
||||
@@ -131,7 +141,7 @@ export function SensitiveDataReviewDrawer({
|
||||
checked={showAll}
|
||||
onChange={(event) => setShowAll(event.target.checked)}
|
||||
/>
|
||||
Show all {suggestions.length} classified columns
|
||||
Show all {suggestions.length} assessed columns
|
||||
</label>
|
||||
</div>
|
||||
|
||||
@@ -144,7 +154,7 @@ export function SensitiveDataReviewDrawer({
|
||||
</p>
|
||||
</div>
|
||||
) : (
|
||||
<ul className="divide-y divide-border" aria-label="Sensitive field suggestions">
|
||||
<ul className="divide-y divide-border" aria-label="Sensitivity assessments">
|
||||
{visible.map((suggestion) => {
|
||||
const proposed = drafts[suggestion.columnId] ?? suggestion.sensitive;
|
||||
const changedFromCurrent = proposed !== suggestion.currentSensitive;
|
||||
@@ -168,6 +178,13 @@ export function SensitiveDataReviewDrawer({
|
||||
<p className="mt-1 text-xs text-muted-foreground">
|
||||
Current: {suggestion.currentSensitive ? "protected" : "allowed"}. Proposed: {proposed ? "protected" : "allowed"}.
|
||||
</p>
|
||||
<p className="mt-1 text-xs text-muted-foreground">
|
||||
Assessment: {suggestion.assessment.replace("_", " ")}. Evidence: {suggestion.evidence.length > 0
|
||||
? suggestion.evidence.map((item) => item.label
|
||||
? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})`
|
||||
: item.ruleId).join(", ")
|
||||
: "no sensitive match"}. Observed values: {suggestion.observedValues}.
|
||||
</p>
|
||||
</div>
|
||||
<span className={`rounded px-2 py-0.5 text-[11px] font-semibold ${changedFromCurrent ? "bg-amber-500/12 text-amber-800 dark:text-amber-300" : "bg-muted text-muted-foreground"}`}>
|
||||
{changedFromCurrent ? "Change" : "No change"}
|
||||
|
||||
+20
-17
@@ -3,19 +3,22 @@ import userEvent from "@testing-library/user-event";
|
||||
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
|
||||
import { http, HttpResponse } from "msw";
|
||||
import { useState } from "react";
|
||||
import type { SensitiveDataSuggestionRun } from "../../api/catalog-databases";
|
||||
import type { SensitivityAnalysisRun } from "../../api/catalog-databases";
|
||||
import { server } from "../../test/msw";
|
||||
import { SensitiveDataSuggestionHistoryDrawer } from "./SensitiveDataSuggestionHistoryDrawer";
|
||||
import { SensitivityAnalysisHistoryDrawer } from "./SensitivityAnalysisHistoryDrawer";
|
||||
|
||||
const runningRun: SensitiveDataSuggestionRun = {
|
||||
const runningRun: SensitivityAnalysisRun = {
|
||||
id: "99999999-9999-4999-8999-999999999999",
|
||||
databaseId: "11111111-1111-4111-8111-111111111111",
|
||||
scope: "selected_columns",
|
||||
modelId: "local-qwen",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "running",
|
||||
total: 4,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
unknown: 2,
|
||||
createdAt: "2026-08-28T09:00:00Z",
|
||||
startedAt: "2026-08-28T09:00:00Z",
|
||||
updatedAt: "2026-08-28T09:00:01Z",
|
||||
@@ -23,7 +26,7 @@ const runningRun: SensitiveDataSuggestionRun = {
|
||||
errorSummary: null,
|
||||
};
|
||||
|
||||
function renderDrawer(initialRun: SensitiveDataSuggestionRun | null) {
|
||||
function renderDrawer(initialRun: SensitivityAnalysisRun | null) {
|
||||
const client = new QueryClient({
|
||||
defaultOptions: { queries: { retry: false, gcTime: Infinity } },
|
||||
});
|
||||
@@ -31,10 +34,9 @@ function renderDrawer(initialRun: SensitiveDataSuggestionRun | null) {
|
||||
function Harness() {
|
||||
const [run, setRun] = useState(initialRun);
|
||||
return (
|
||||
<SensitiveDataSuggestionHistoryDrawer
|
||||
<SensitivityAnalysisHistoryDrawer
|
||||
open
|
||||
run={run}
|
||||
modelLabel="Local Qwen"
|
||||
onClose={() => undefined}
|
||||
onRunUpdate={setRun}
|
||||
/>
|
||||
@@ -56,13 +58,13 @@ test("shows an explicit empty state before any sensitive suggestion runs exist",
|
||||
|
||||
renderDrawer(null);
|
||||
|
||||
const drawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
|
||||
expect(within(drawer).getByRole("heading", { name: "Suggestion run history" })).toBeVisible();
|
||||
expect(await within(drawer).findByText("No sensitive suggestion runs yet.")).toBeVisible();
|
||||
const drawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
|
||||
expect(within(drawer).getByRole("heading", { name: "Analysis run history" })).toBeVisible();
|
||||
expect(await within(drawer).findByText("No sensitivity analysis runs yet.")).toBeVisible();
|
||||
});
|
||||
|
||||
test("shows sensitive suggestion results, safe events, and lets operators inspect an older run", async () => {
|
||||
const completedRun: SensitiveDataSuggestionRun = {
|
||||
const completedRun: SensitivityAnalysisRun = {
|
||||
...runningRun,
|
||||
id: "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa",
|
||||
scope: "all",
|
||||
@@ -70,6 +72,7 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
|
||||
total: 6,
|
||||
suggestedSensitive: 2,
|
||||
suggestedNonSensitive: 4,
|
||||
unknown: 0,
|
||||
createdAt: "2026-08-28T08:00:00Z",
|
||||
startedAt: "2026-08-28T08:00:00Z",
|
||||
updatedAt: "2026-08-28T08:00:01Z",
|
||||
@@ -95,18 +98,18 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
|
||||
const user = userEvent.setup();
|
||||
renderDrawer(runningRun);
|
||||
|
||||
const drawer = await screen.findByRole("dialog", { name: "Sensitive suggestion history" });
|
||||
const drawer = await screen.findByRole("dialog", { name: "Sensitivity analysis history" });
|
||||
expect(within(drawer).getByRole("heading", { name: "Running" })).toBeVisible();
|
||||
const progress = within(drawer).getByRole("table", { name: "Sensitive suggestion progress" });
|
||||
expect(within(progress).getAllByRole("columnheader")).toHaveLength(3);
|
||||
expect(within(drawer).getByRole("log", { name: "Sensitive suggestion events" })).toHaveClass(
|
||||
const progress = within(drawer).getByRole("table", { name: "Sensitivity analysis progress" });
|
||||
expect(within(progress).getAllByRole("columnheader")).toHaveLength(4);
|
||||
expect(within(drawer).getByRole("log", { name: "Sensitivity analysis events" })).toHaveClass(
|
||||
"thot-catalog-drawer__event-log",
|
||||
);
|
||||
expect(await within(drawer).findByText("Classifying columns")).toBeVisible();
|
||||
expect(within(drawer).queryByRole("button", { name: "Stop" })).not.toBeInTheDocument();
|
||||
expect(within(drawer).queryByRole("button", { name: "Unlock stale run" })).not.toBeInTheDocument();
|
||||
|
||||
const history = within(drawer).getByRole("region", { name: "Sensitive suggestion run history" });
|
||||
const history = within(drawer).getByRole("region", { name: "Sensitivity analysis run history" });
|
||||
await user.click(within(history).getByRole("button", { name: /Completed.*all/i }));
|
||||
|
||||
expect(await within(drawer).findByRole("heading", { name: "Completed" })).toBeVisible();
|
||||
@@ -116,7 +119,7 @@ test("shows sensitive suggestion results, safe events, and lets operators inspec
|
||||
});
|
||||
|
||||
test("does not expose unsafe provider details from a final error summary", async () => {
|
||||
const failedRun: SensitiveDataSuggestionRun = {
|
||||
const failedRun: SensitivityAnalysisRun = {
|
||||
...runningRun,
|
||||
status: "failed",
|
||||
finishedAt: "2026-08-28T09:00:02Z",
|
||||
+44
-43
@@ -3,35 +3,34 @@ import { LoaderCircle } from "lucide-react";
|
||||
import { useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { apiErrorMessage } from "../../api/client";
|
||||
import {
|
||||
getSensitiveDataSuggestionRun,
|
||||
listSensitiveDataSuggestionEvents,
|
||||
listSensitiveDataSuggestionRuns,
|
||||
type SensitiveDataSuggestionRun,
|
||||
getSensitivityAnalysisRun,
|
||||
listSensitivityAnalysisEvents,
|
||||
listSensitivityAnalysisRuns,
|
||||
type SensitivityAnalysisRun,
|
||||
} from "../../api/catalog-databases";
|
||||
import { FleetLedgerDrawer } from "./FleetLedgerShell";
|
||||
|
||||
interface Props {
|
||||
open: boolean;
|
||||
databaseId?: string | null;
|
||||
run: SensitiveDataSuggestionRun | null;
|
||||
modelLabel: string;
|
||||
run: SensitivityAnalysisRun | null;
|
||||
onClose: () => void;
|
||||
onRunUpdate: (run: SensitiveDataSuggestionRun) => void;
|
||||
onRunUpdate: (run: SensitivityAnalysisRun) => void;
|
||||
}
|
||||
|
||||
const HISTORY_QUERY_KEY = ["sensitive-data-suggestion-runs", 50] as const;
|
||||
const SAFE_FINAL_ERROR = "The run ended before all columns were classified. Review the event log for safe details.";
|
||||
|
||||
function statusLabel(run: SensitiveDataSuggestionRun): string {
|
||||
function statusLabel(run: SensitivityAnalysisRun): string {
|
||||
const label = run.status.replaceAll("_", " ");
|
||||
return `${label[0].toUpperCase()}${label.slice(1)}`;
|
||||
}
|
||||
|
||||
function isTerminal(run?: SensitiveDataSuggestionRun): boolean {
|
||||
function isTerminal(run?: SensitivityAnalysisRun): boolean {
|
||||
return Boolean(run && ["completed", "failed", "interrupted"].includes(run.status));
|
||||
}
|
||||
|
||||
function outcomeClass(run: SensitiveDataSuggestionRun): string {
|
||||
function outcomeClass(run: SensitivityAnalysisRun): string {
|
||||
if (run.status === "completed") return "thot-sensitive-run--success";
|
||||
if (["failed", "interrupted"].includes(run.status)) return "thot-sensitive-run--failure";
|
||||
return "";
|
||||
@@ -52,11 +51,10 @@ function timestamp(value: string | null): string {
|
||||
return value ? new Date(value).toLocaleString() : "Not available";
|
||||
}
|
||||
|
||||
export function SensitiveDataSuggestionHistoryDrawer({
|
||||
export function SensitivityAnalysisHistoryDrawer({
|
||||
open,
|
||||
databaseId = null,
|
||||
run: initialRun,
|
||||
modelLabel,
|
||||
onClose,
|
||||
onRunUpdate,
|
||||
}: Props) {
|
||||
@@ -64,7 +62,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
const runId = initialRun?.id ?? null;
|
||||
const runQuery = useQuery({
|
||||
queryKey: ["sensitive-data-suggestion-run", runId],
|
||||
queryFn: () => getSensitiveDataSuggestionRun(runId!),
|
||||
queryFn: () => getSensitivityAnalysisRun(runId!),
|
||||
enabled: open && Boolean(runId),
|
||||
initialData: initialRun ?? undefined,
|
||||
retry: false,
|
||||
@@ -73,14 +71,14 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
const run = runQuery.data;
|
||||
const eventQuery = useQuery({
|
||||
queryKey: ["sensitive-data-suggestion-events", runId],
|
||||
queryFn: () => listSensitiveDataSuggestionEvents(runId!),
|
||||
queryFn: () => listSensitivityAnalysisEvents(runId!),
|
||||
enabled: open && Boolean(runId),
|
||||
retry: false,
|
||||
refetchInterval: run && isTerminal(run) ? false : 1_500,
|
||||
});
|
||||
const historyQuery = useQuery({
|
||||
queryKey: HISTORY_QUERY_KEY,
|
||||
queryFn: () => listSensitiveDataSuggestionRuns(50),
|
||||
queryFn: () => listSensitivityAnalysisRuns(50),
|
||||
enabled: open,
|
||||
retry: false,
|
||||
refetchInterval: (query) => query.state.data?.some((item) => !isTerminal(item))
|
||||
@@ -100,7 +98,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
useEffect(() => {
|
||||
if (!run) return;
|
||||
onRunUpdate(run);
|
||||
queryClient.setQueryData<SensitiveDataSuggestionRun[]>(HISTORY_QUERY_KEY, (current) => {
|
||||
queryClient.setQueryData<SensitivityAnalysisRun[]>(HISTORY_QUERY_KEY, (current) => {
|
||||
if (!current) return [run];
|
||||
return current.some((item) => item.id === run.id)
|
||||
? current.map((item) => item.id === run.id ? run : item)
|
||||
@@ -113,11 +111,11 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
return (
|
||||
<FleetLedgerDrawer
|
||||
open
|
||||
ariaLabel="Sensitive suggestion history"
|
||||
ariaLabel="Sensitivity analysis history"
|
||||
eyebrow="Sensitive data"
|
||||
title="Suggestion run history"
|
||||
title="Analysis run history"
|
||||
onClose={onClose}
|
||||
closeLabel="Close sensitive suggestion history"
|
||||
closeLabel="Close sensitivity analysis history"
|
||||
bodyClassName="thot-catalog-drawer__body--empty"
|
||||
>
|
||||
{historyQuery.isError ? (
|
||||
@@ -125,12 +123,12 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
{apiErrorMessage(historyQuery.error)}
|
||||
</div>
|
||||
) : historyQuery.isLoading || (historyQuery.data?.length ?? 0) > 0 ? (
|
||||
<p className="text-sm text-muted-foreground">Loading suggestion history…</p>
|
||||
<p className="text-sm text-muted-foreground">Loading analysis history…</p>
|
||||
) : (
|
||||
<div className="w-full rounded-md border border-border bg-muted/25 px-4 py-5 text-sm">
|
||||
<p className="font-semibold">No sensitive suggestion runs yet.</p>
|
||||
<p className="font-semibold">No sensitivity analysis runs yet.</p>
|
||||
<p className="mt-1 text-muted-foreground">
|
||||
Request sensitive-field suggestions to create the first history entry.
|
||||
Analyze sensitive fields to create the first history entry.
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
@@ -140,20 +138,21 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
|
||||
const events = eventQuery.data ?? [];
|
||||
const finalError = safeFinalError(run.errorSummary);
|
||||
const modelDisplay = modelLabel && modelLabel !== run.modelId
|
||||
? `${modelLabel} (${run.modelId})`
|
||||
: run.modelId;
|
||||
const engineDisplay = run.engine === "local"
|
||||
? `Local rules (${run.policyVersion ?? "unknown policy"})`
|
||||
: `Legacy LLM (${run.modelId ?? "unknown model"})`;
|
||||
const progressCounters = [
|
||||
["total", "Total columns"],
|
||||
["suggestedSensitive", "Sensitive"],
|
||||
["suggestedNonSensitive", "Not sensitive"],
|
||||
["unknown", "Unknown"],
|
||||
] as const;
|
||||
return (
|
||||
<FleetLedgerDrawer
|
||||
open
|
||||
className={outcomeClass(run)}
|
||||
ariaLabel="Sensitive suggestion history"
|
||||
eyebrow="Sensitive data suggestions"
|
||||
ariaLabel="Sensitivity analysis history"
|
||||
eyebrow="Sensitivity analysis"
|
||||
title={(
|
||||
<span className="inline-flex items-center gap-2">
|
||||
{!isTerminal(run) ? <LoaderCircle className="size-5 animate-spin" aria-hidden="true" /> : null}
|
||||
@@ -162,7 +161,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
)}
|
||||
description={run.scope.replaceAll("_", " ")}
|
||||
onClose={onClose}
|
||||
closeLabel="Close sensitive suggestion history"
|
||||
closeLabel="Close sensitivity analysis history"
|
||||
bodyClassName="thot-catalog-drawer__history-layout thot-catalog-drawer__run-layout"
|
||||
>
|
||||
<div className="thot-catalog-drawer__run-main">
|
||||
@@ -173,16 +172,18 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
) : null}
|
||||
|
||||
<div className={`grid gap-4 sm:grid-cols-2${runQuery.isError ? " mt-4" : ""}`}>
|
||||
<section aria-label="AI usage">
|
||||
<h3 className="thot-label mb-2">AI usage</h3>
|
||||
<section aria-label="Analysis engine">
|
||||
<h3 className="thot-label mb-2">Analysis engine</h3>
|
||||
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1 text-xs">
|
||||
<dt>Model</dt><dd className="truncate text-right" title={modelDisplay}>{modelDisplay}</dd>
|
||||
<dt>Input tokens</dt><dd className="text-right tabular-nums">{(run.inputTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
<dt>Cache tokens</dt><dd className="text-right tabular-nums">{(run.cacheReadTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
<dt>Output tokens</dt><dd className="text-right tabular-nums">{(run.outputTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
<dt>Engine</dt><dd className="truncate text-right" title={engineDisplay}>{engineDisplay}</dd>
|
||||
{run.engine === "llm" ? <>
|
||||
<dt>Input tokens</dt><dd className="text-right tabular-nums">{(run.inputTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
<dt>Cache tokens</dt><dd className="text-right tabular-nums">{(run.cacheReadTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
<dt>Output tokens</dt><dd className="text-right tabular-nums">{(run.outputTokens ?? 0).toLocaleString("en-US")}</dd>
|
||||
</> : null}
|
||||
</dl>
|
||||
</section>
|
||||
<section aria-label="Sensitive suggestion timestamps">
|
||||
<section aria-label="Sensitivity analysis timestamps">
|
||||
<h3 className="thot-label mb-2">Timestamps</h3>
|
||||
<dl className="grid grid-cols-[auto_1fr] gap-x-4 gap-y-1 text-xs">
|
||||
<dt className="text-muted-foreground">Created</dt><dd className="text-right tabular-nums">{timestamp(run.createdAt)}</dd>
|
||||
@@ -193,10 +194,10 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<section className="mt-4" aria-label="Sensitive suggestion counters">
|
||||
<section className="mt-4" aria-label="Sensitivity analysis counters">
|
||||
<h3 className="thot-label mb-2">Results</h3>
|
||||
<div className="overflow-x-auto rounded-md border border-border bg-muted/20">
|
||||
<table className="w-full min-w-[20rem] table-fixed" aria-label="Sensitive suggestion progress">
|
||||
<table className="w-full min-w-[20rem] table-fixed" aria-label="Sensitivity analysis progress">
|
||||
<thead>
|
||||
<tr className="border-b border-border bg-muted/45">
|
||||
{progressCounters.map(([name, label]) => (
|
||||
@@ -225,14 +226,14 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
<section className="thot-catalog-drawer__events mt-3" aria-label="Sensitive suggestion log">
|
||||
<section className="thot-catalog-drawer__events mt-3" aria-label="Sensitivity analysis log">
|
||||
<div className="mb-2 flex items-center justify-between">
|
||||
<h3 className="thot-label">Events</h3>
|
||||
<span className="text-xs tabular-nums text-muted-foreground">{events.length}</span>
|
||||
</div>
|
||||
<div
|
||||
role="log"
|
||||
aria-label="Sensitive suggestion events"
|
||||
aria-label="Sensitivity analysis events"
|
||||
className="thot-catalog-drawer__event-log overflow-y-auto rounded-md bg-zinc-950 p-3 font-mono text-xs leading-5 text-zinc-200"
|
||||
>
|
||||
{events.length === 0 ? <p className="text-zinc-500">Waiting for events…</p> : events.map((event) => (
|
||||
@@ -245,7 +246,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<section className="thot-catalog-drawer__history" aria-label="Sensitive suggestion run history">
|
||||
<section className="thot-catalog-drawer__history" aria-label="Sensitivity analysis run history">
|
||||
<div className="mb-2 flex items-center justify-between gap-3">
|
||||
<h3 className="thot-label">Recent runs</h3>
|
||||
<span className="text-xs tabular-nums text-muted-foreground">
|
||||
@@ -255,9 +256,9 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
{historyQuery.isError ? (
|
||||
<p role="alert" className="text-sm text-destructive">{apiErrorMessage(historyQuery.error)}</p>
|
||||
) : historyQuery.isLoading ? (
|
||||
<p className="text-sm text-muted-foreground">Loading suggestion history…</p>
|
||||
<p className="text-sm text-muted-foreground">Loading analysis history…</p>
|
||||
) : visibleHistory.length === 0 ? (
|
||||
<p className="text-sm text-muted-foreground">No sensitive suggestion runs yet.</p>
|
||||
<p className="text-sm text-muted-foreground">No sensitivity analysis runs yet.</p>
|
||||
) : (
|
||||
<div className="thot-catalog-drawer__history-list divide-y divide-border overflow-y-auto rounded-md border border-border">
|
||||
{visibleHistory.map((item) => (
|
||||
@@ -272,7 +273,7 @@ export function SensitiveDataSuggestionHistoryDrawer({
|
||||
>
|
||||
<span className="font-medium">{statusLabel(item)}</span>
|
||||
<span className="shrink-0 text-right text-xs text-muted-foreground">
|
||||
{item.suggestedSensitive + item.suggestedNonSensitive}/{item.total}<br />
|
||||
{item.suggestedSensitive + item.suggestedNonSensitive + item.unknown}/{item.total}<br />
|
||||
{new Date(item.createdAt).toLocaleString()}
|
||||
</span>
|
||||
</button>
|
||||
@@ -50,6 +50,7 @@ nav:
|
||||
- Generic OIDC: install/authentication-oidc.md
|
||||
- Authentik: install/authentik.md
|
||||
- Workspace operations: operations/workspaces.md
|
||||
- Local sensitivity analysis: operations/sensitivity-analysis.md
|
||||
- Pi model configuration: general/pi-configuration.md
|
||||
- Docker installation contexts: installazione-docker-4-contesti.md
|
||||
- Use ThothII:
|
||||
@@ -86,7 +87,11 @@ nav:
|
||||
- 0010 Bounded source samples: adr/0010-allow-bounded-real-source-samples-for-description-generation.md
|
||||
- 0011 Sensitive data flag: adr/0011-gate-source-samples-with-a-sensitive-data-flag.md
|
||||
- 0012 Effective relationship authority: adr/0012-use-the-catalog-as-the-logical-relationship-authority.md
|
||||
- 0013 Installation model catalog: adr/0013-use-one-installation-model-catalog-with-runtime-projections.md
|
||||
- 0014 Local sensitive-column assessment: adr/0014-assess-sensitive-columns-locally-from-source-content.md
|
||||
- AI catalog description acceptance: testing/2026-08-29-ai-catalog-description-generation-acceptance.md
|
||||
- Sensitivity NER license inventory: reports/2026-09-02-sensitivity-ner-license-inventory.md
|
||||
- PSD sensitivity shadow evaluation: reports/2026-09-02-psd-sensitivity-shadow.md
|
||||
- Design records:
|
||||
- Metadata catalog design: plans/2026-08-26-metadata-catalog-from-thothai.md
|
||||
- Description generation design: plans/2026-08-28-ai-catalog-description-generation.md
|
||||
|
||||
Executable
+111
@@ -0,0 +1,111 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
MODEL_REPOSITORY="fastino/gliner2-privacy-filter-PII-multi"
|
||||
MODEL_REVISION="c153999da5f4c509df4322b0c6a1baf3d2c284d7"
|
||||
TARGET="${1:-}"
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd -P)"
|
||||
|
||||
if [[ -z "$TARGET" || "$TARGET" != /* ]]; then
|
||||
echo "Usage: $0 /absolute/model/directory" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
if [[ -d "$TARGET" ]] && find "$TARGET" -mindepth 1 -print -quit | grep -q .; then
|
||||
echo "Target directory must be empty: $TARGET" >&2
|
||||
exit 2
|
||||
fi
|
||||
mkdir -p "$TARGET"
|
||||
cd "$ROOT"
|
||||
|
||||
docker build \
|
||||
--build-arg INSTALL_SENSITIVITY_NER=true \
|
||||
--file docker/core.Dockerfile \
|
||||
--tag thothii-core:sensitivity-ner \
|
||||
.
|
||||
|
||||
docker run --rm \
|
||||
--user "$(id -u):$(id -g)" \
|
||||
--env HOME=/tmp \
|
||||
--env TMPDIR=/tmp \
|
||||
--env XDG_CACHE_HOME=/tmp/.cache \
|
||||
--env HF_HOME=/tmp/huggingface \
|
||||
--env HF_HUB_DISABLE_TELEMETRY=1 \
|
||||
--env HF_HUB_DISABLE_XET=1 \
|
||||
--volume "$TARGET:/model" \
|
||||
--entrypoint /opt/sensitivity-ner/bin/hf \
|
||||
thothii-core:sensitivity-ner \
|
||||
download "$MODEL_REPOSITORY" \
|
||||
--revision "$MODEL_REVISION" \
|
||||
--local-dir /model
|
||||
|
||||
printf '%s\n' "$MODEL_REVISION" > "$TARGET/THOTHII_MODEL_REVISION"
|
||||
|
||||
docker run --rm \
|
||||
--user "$(id -u):$(id -g)" \
|
||||
--volume "$TARGET:/model" \
|
||||
--entrypoint /bin/sh \
|
||||
thothii-core:sensitivity-ner \
|
||||
-c 'cd /model && sha256sum -c /app/backend/python/sensitivity-ner-model-sha256.txt && cp /app/backend/python/sensitivity-ner-model-sha256.txt MODEL_SHA256SUMS'
|
||||
|
||||
docker run --rm \
|
||||
--network none \
|
||||
--entrypoint /opt/sensitivity-ner/bin/pip \
|
||||
thothii-core:sensitivity-ner \
|
||||
check
|
||||
|
||||
GPU_AUDIT="$(docker run --rm \
|
||||
--network none \
|
||||
--entrypoint /opt/sensitivity-ner/bin/python \
|
||||
thothii-core:sensitivity-ner \
|
||||
-c 'import torch; print(f"{torch.cuda.is_available()}|{torch.version.cuda}|{torch.cuda.device_count()}")')"
|
||||
if [[ "$GPU_AUDIT" != "False|None|0" ]]; then
|
||||
echo "The optional runtime is not CPU-only: $GPU_AUDIT" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
NETWORK_GUARD_AUDIT="$(docker run --rm \
|
||||
--entrypoint /opt/sensitivity-ner/bin/python \
|
||||
thothii-core:sensitivity-ner \
|
||||
-c '
|
||||
import importlib.util
|
||||
import socket
|
||||
spec = importlib.util.spec_from_file_location("worker", "/app/backend/python/sensitivity_ner_worker.py")
|
||||
worker = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
spec.loader.exec_module(worker)
|
||||
raw_socket = socket.socket
|
||||
worker._disable_network()
|
||||
try:
|
||||
raw_socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
except PermissionError as error:
|
||||
print(error.errno)
|
||||
else:
|
||||
raise SystemExit("network syscall filter is inactive")
|
||||
')"
|
||||
if [[ "$NETWORK_GUARD_AUDIT" != "1" ]]; then
|
||||
echo "The optional runtime did not activate its network syscall filter" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
SMOKE_OUTPUT="$(mktemp)"
|
||||
SMOKE_ERROR="$(mktemp)"
|
||||
trap 'rm -f -- "$SMOKE_OUTPUT" "$SMOKE_ERROR"' EXIT
|
||||
printf '%s\n' '{"id":"smoke","candidates":[{"columnId":"33333333-3333-4333-8333-333333333333","text":"La paziente si chiama Maria Rossi."}]}' \
|
||||
| docker run --rm \
|
||||
--network none \
|
||||
--read-only \
|
||||
--tmpfs /tmp:rw,noexec,nosuid,size=512m \
|
||||
--volume "$TARGET:/model:ro" \
|
||||
--entrypoint /opt/sensitivity-ner/bin/python \
|
||||
-i thothii-core:sensitivity-ner \
|
||||
/app/backend/python/sensitivity_ner_worker.py --model /model --threads 2 \
|
||||
>"$SMOKE_OUTPUT" 2>"$SMOKE_ERROR"
|
||||
if ! grep -Fxq '{"ready":true}' "$SMOKE_OUTPUT" \
|
||||
|| ! grep -Eq '"ok":true.*"columnId":"33333333-3333-4333-8333-333333333333".*"label":"full_name"' "$SMOKE_OUTPUT"; then
|
||||
echo "The offline Italian CPU smoke test failed" >&2
|
||||
sed -n '1,20p' "$SMOKE_ERROR" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Pinned model downloaded to $TARGET"
|
||||
@@ -20,6 +20,9 @@ compose_files=(-f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml")
|
||||
if [[ "${THOTH_ENABLE_EMBEDDING_GPU:-0}" == "1" ]]; then
|
||||
compose_files+=(-f "$ROOT/deploy/compose.embedding-gpu.yaml")
|
||||
fi
|
||||
if [[ "${THOTH_ENABLE_SENSITIVITY_NER:-0}" == "1" ]]; then
|
||||
compose_files+=(-f "$ROOT/deploy/compose.sensitivity-ner.yaml")
|
||||
fi
|
||||
|
||||
compose=(docker compose --env-file "$LOCAL_ENV_FILE" "${compose_files[@]}")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user