From ae053961a3aef17ee7c67632b7ebb89155b81ace Mon Sep 17 00:00:00 2001 From: Codex Date: Wed, 2 Sep 2026 15:58:23 +0200 Subject: [PATCH] feat: refine metadata catalog workflows --- CONTEXT.md | 83 ++++-- ...-model-catalog-with-runtime-projections.md | 67 +++++ ...ive-columns-locally-from-source-content.md | 52 ++++ .../2026-09-02-installation-model-catalog.md | 245 ++++++++++++++++++ ...l-sensitive-column-classifier-libraries.md | 218 ++++++++++++++++ .../prototypes/database-management/fleet.tsx | 8 +- frontend/src/auth/LoginPage.tsx | 2 +- frontend/src/shell/AppShell.tsx | 4 +- .../src/shell/DatabaseManagementPage.test.tsx | 155 +++++------ frontend/src/shell/DatabaseManagementPage.tsx | 10 +- frontend/src/shell/PiManagement.test.tsx | 2 +- frontend/src/shell/PiManagement.tsx | 4 +- frontend/src/shell/SteerInput.tsx | 2 +- .../database-management/DatabaseColumns.tsx | 58 ++++- .../DatabaseFleetQueryErrors.test.tsx | 3 +- .../database-management/DatabaseForm.tsx | 2 +- .../database-management/DatabaseGrid.tsx | 12 +- .../DatabaseRelationships.tsx | 6 +- .../database-management/DatabaseTables.tsx | 82 +++--- .../FleetActionSelector.test.tsx | 2 +- .../FleetActionSelector.typography.test.ts | 6 + .../database-management/FleetLedgerShell.css | 17 +- .../HistoryActionPlacement.test.ts | 9 +- frontend/src/viewers/CtePlanViewer.test.tsx | 2 +- frontend/src/viewers/CtePlanViewer.tsx | 2 +- .../src/viewers/SchemaLinkingViewer.test.tsx | 2 +- frontend/src/viewers/SchemaLinkingViewer.tsx | 2 +- frontend/src/widgets/ArtifactGateWidget.tsx | 2 +- frontend/src/widgets/MultiselectWidget.tsx | 4 +- frontend/src/widgets/SelectWidget.test.tsx | 2 +- frontend/src/widgets/SelectWidget.tsx | 2 +- 31 files changed, 848 insertions(+), 219 deletions(-) create mode 100644 docs/adr/0013-use-one-installation-model-catalog-with-runtime-projections.md create mode 100644 docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md create mode 100644 docs/plans/2026-09-02-installation-model-catalog.md create mode 100644 docs/research/2026-09-02-local-sensitive-column-classifier-libraries.md diff --git a/CONTEXT.md b/CONTEXT.md index 898aaa04..dc9940f9 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -244,6 +244,29 @@ incompatibile o non aggiornato produce nessuna Evidence e un avviso esplicito. I workflow può continuare, ma non usa mai silenziosamente contenuti di una revisione precedente o di un altro workspace. +## Configurazione dei modelli + +**Workspace Descriptor** — La dichiarazione versionata dell'identità e dello scope del +Workspace Database e delle Evidence di un workspace. Non definisce modelli, selezioni di +modello o Database Binding specifiche di un'installazione. + +**Installation Model Catalog** — L'insieme dichiarativo, proprio di un'installazione, dei +modelli disponibili, dei loro Model Usage e dei relativi default. È l'unica autorità per i +modelli di sessione, generazione dei metadati ed embedding e non appartiene a un workspace. +_Avoid_: Model Catalog, Metadata Generation Model Configuration + +**Model Usage** — Lo scopo per cui un modello dell'Installation Model Catalog può essere +usato: `session`, `metadata_generation` oppure `embedding`. L'ammissibilità e il default +dipendono dall'uso, non dal workspace. + +**Model Selection** — La scelta runtime, a livello di installazione, di un modello del +catalogo per uno specifico Model Usage. Riferisce l'identità canonica del modello senza +ridefinirne provider, endpoint o capacità. + +**Model Runtime Projection** — La rappresentazione derivata e non autoritativa +dell'Installation Model Catalog richiesta da uno specifico runtime. Può essere rigenerata +integralmente dalla configurazione dell'installazione. + ## Catalogo dei metadati **Workspace Database** — Il database che appartiene a un solo workspace e non può essere @@ -385,38 +408,56 @@ con la binding corrente. distinti dai fatti strutturali governati dalla sincronizzazione. Possono essere popolati dall'AI, da un'importazione o da una modifica amministrativa senza cambiare il database esterno. -**Metadata Generation Model Configuration** — La configurazione a livello di setup applicativo -che elenca i modelli selezionabili, il default e i riferimenti agli eventuali segreti per la sola -generazione dei metadati. Un modello keyless è ammesso solo con un endpoint esplicito che non -richiede autenticazione. Non appartiene al workspace ed è indipendente dalla configurazione Pi. - **Model Completion Helper** — Il processo Python interno ed effimero che esegue una singola richiesta LiteLLM per conto del backend. Non è un servizio HTTP, non possiede il lifecycle della Description Generation Run e non è una CLI esposta agli utenti. **Catalog Sample** — Un input transitorio composto da un massimo di cinque righe e da valori di esempio bounded di una Catalog Table per la generazione delle descrizioni. Può contenere valori -reali oppure sintetici in base al Sensitive Data Flag della Catalog Column; non viene persistito -e non diventa Catalog Metadata. +reali oppure sintetici in base alla Source Value Disclosure Decision; non viene persistito e non +diventa Catalog Metadata. -**Sensitive Data Flag** — La scelta binaria umana applicata a una Catalog Column: `true` protegge -i valori sorgente e `false` ne consente l'invio al modello. Il valore predefinito è `false`, anche -per le nuove colonne. +**Sensitive Data Flag** — La classificazione binaria umana applicata a una Catalog Column. Può +essere impostata liberamente dall'amministratore anche in contrasto con una valutazione automatica. -**Sensitive Data Policy** — La regola che applica il Sensitive Data Flag ai Catalog Sample: -valori sintetici per una colonna protetta, valori reali per una colonna non protetta. L'AI può -suggerire il flag dai soli metadati tecnici di un database, delle tabelle o delle colonne -esplicitamente selezionate; le richieste ampie vengono divise in batch bounded, ma soltanto -l'utente imposta i flag dopo aver rivisto la proposta completa. +**Local Sensitivity Assessment** — La valutazione locale, non autoritativa e priva di LLM di una +Catalog Column, basata su metadati e contenuto sorgente, con esito `sensitive`, `non_sensitive` +oppure `unknown`. +_Avoid_: AI suggestion, automatic flag + +**Local NER Detector** — Il componente NLP opzionale e CPU-only che esamina soltanto testo ancora +ambiguo e restituisce evidenze al Local Sensitivity Assessment. Non decide lo stato della colonna, +non usa un LLM generativo e non persiste valori sorgente. +_Avoid_: AI classifier, local LLM fallback + +**Model Data Boundary** — La qualificazione amministrativa di un modello come `internal` oppure +`external` rispetto al confine entro cui i valori sorgente possono essere comunicati. +_Avoid_: local model, remote model + +**Source Value Disclosure Decision** — L'unica decisione effettiva che stabilisce se un modello +riceve valori sorgente reali oppure sostituti sintetici, combinando Model Data Boundary e Sensitive +Data Flag. +_Avoid_: sample filter, export flag + +**Sensitive Data Policy** — L'insieme versionato di regole locali generali e specifiche che produce +una Local Sensitivity Assessment. Un singolo riscontro blocca l'intera colonna e qualsiasi valore +testuale più lungo di 500 caratteri rende sensibile la colonna. _Avoid_: PII filter, sample filter -**Sensitive Data Suggestion Run** — Il tentativo amministrativo tracciato con cui il modello -propone Sensitive Data Flag dai soli metadati strutturali. Conserva stato e conteggi aggregati, -ma non i suggerimenti per colonna, che restano una proposta transitoria fino al salvataggio umano. +**Sensitivity Analysis Run** — Il tentativo amministrativo esplicito e tracciato che valuta una +selezione di colonne mediante la Sensitive Data Policy. Conserva stato, copertura e conteggi +aggregati, ma non valori sorgente né esiti per colonna. +_Avoid_: Sensitive Data Suggestion Run, AI analysis -**Sensitive Data Suggestion Event** — Una riga testuale ordinata e sanitizzata che registra -l'avvio, l'esito o l'errore di una Sensitive Data Suggestion Run senza conservare prompt, -risposte grezze del provider o proposte per colonna. +**Sensitivity Review Draft** — La proposta transitoria che associa alle colonne selezionate una +Local Sensitivity Assessment e le relative evidenze sanificate. Non modifica il Sensitive Data Flag +finché l'amministratore non salva le proprie decisioni e viene scartata al reload. +_Avoid_: automatic flag + +**Sensitivity Analysis Event** — Una riga testuale ordinata e sanificata che registra l'avvio, +l'esito o l'errore di una Sensitivity Analysis Run senza conservare contenuti sorgente, output grezzi +del detector o proposte per colonna. +_Avoid_: Sensitive Data Suggestion Event **Introspection Capability** — Una categoria di struttura fisica che una Database Binding può osservare, come tabelle, colonne, relazioni, indici o enum. Una capability non disponibile diff --git a/docs/adr/0013-use-one-installation-model-catalog-with-runtime-projections.md b/docs/adr/0013-use-one-installation-model-catalog-with-runtime-projections.md new file mode 100644 index 00000000..af26f9ef --- /dev/null +++ b/docs/adr/0013-use-one-installation-model-catalog-with-runtime-projections.md @@ -0,0 +1,67 @@ +# Use one Installation Model Catalog with runtime projections + +ThothII currently declares model availability independently in installation +`metadataGeneration`, Pi configuration, and workspace `llm_policy` and embedding settings. The +Installation Model Catalog in `thothii-installation.yaml` becomes the sole authored authority for +session, metadata-generation, and embedding models, including their allowed usages and per-usage +defaults. Workspace descriptors retain database identity and scope plus Evidence concerns, but no +model policy or selection; installation-local Database Bindings remain separate from them. + +Pi, the backend metadata-generation helper, and the embedding runtime consume generated Model +Runtime Projections of that catalog. Runtime Model Selection stores only a canonical catalog model +identity and use-specific controls such as thinking level; endpoint, provider, capabilities, and +credential references remain catalog facts, while secret values remain in protected secret stores. +The metadata-generation helper continues to use LiteLLM independently of Pi, as established by +ADR-0009: unifying model declaration does not unify execution lifecycles. + +The migration is intentionally fail-closed. Workspace schema v4 removes `llm_policy` and the +entire redundant `semantic_index`; collection identity is derived from the workspace identity, +while vector-store and embedding facts come from the installation. A deterministic migration +rewrites existing descriptors. The +installation loader replaces `metadataGeneration` with `modelCatalog` and rejects the legacy form +with an actionable migration error rather than keeping two live sources. Existing Pi +`models.json` and enabled-model settings become generated artifacts and are never edited as +authoritative configuration. + +Catalog identities use the canonical `provider/model` form; runtime-specific upstream names are +adapter facts, not additional ThothII identities. Installation descriptors use schema version 2 +and model-free workspace descriptors use schema version 4. Removing a model never substitutes it +inside an existing session: an unresolvable resume fails explicitly. Published semantic indexes +record the embedding identity and dimensions that produced them and require explicit +reprocessing when those facts change. + +Model eligibility is expressed by the presence of a `session` or `metadataGeneration` block, +without a duplicate usages list. The installation declares one active embedding identity and its +dimensions rather than a selectable embedding catalog. Runtime projections are regenerated +deterministically and atomically at start, so they require no persisted digest and are excluded +from installation backups; restore regenerates them from the validated installation descriptor. + +A provider owns one endpoint, one explicit authentication mode, and only the runtime adapters it +needs. Authentication is either a protected secret-environment reference, Pi-owned authentication +for session-only built-in models, or explicit keyless operation for an explicit endpoint; models +cannot override it. Pi built-in model facts are not copied into the installation. A session default +and the single embedding definition are required, while metadata generation and its default may be +omitted together. The schema deliberately excludes unused abstractions and future properties until +runtime behavior requires them. + +The catalog session default replaces `PI_PROVIDER`, `PI_MODEL`, and persisted installation model +defaults as configuration sources. A user or session selection is only a canonical catalog +reference, and an existing session keeps that reference without silently switching models. A +generated Compose projection supplies the catalog-derived embedding values and projection mounts +to every affected service, so neither the base Compose files nor `operator.env` repeat model facts. + +Workspace v3-to-v4 migration is a deterministic removal of `llm_policy` and `semantic_index`. +Installation migration instead inspects the legacy installation metadata block and both Pi model +files: it emits a v2 candidate only when their identities and settings can be reconciled without +guessing. Conflicts produce an actionable report and leave every source untouched. + +## Considered Options + +- A separate catalog file referenced by the installation was rejected because it adds path, + permission, backup, and atomic-update coordination without a current need for cross-installation + sharing. +- Pi `models.json` was rejected as the authority because it is a Pi-specific projection that does + not express all ThothII usages, built-in providers, metadata-generation controls, or embedding + facts. +- Transitional dual reading was rejected because it would preserve the configuration discrepancy + this decision is intended to eliminate. diff --git a/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md new file mode 100644 index 00000000..43b4143f --- /dev/null +++ b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md @@ -0,0 +1,52 @@ +--- +status: accepted +--- + +# Assess sensitive columns locally from source content + +The Sensitive Data Flag remains a human-owned boolean. An explicit, selection-scoped sensitivity +analysis may propose changes by inspecting both catalog metadata and source values, but it never +writes the flag. The administrator may accept, reject, or reverse every proposal. + +One TypeScript `SensitivityClassifier` is the only component allowed to produce the column-level +assessment `sensitive`, `non_sensitive`, or `unknown`. It applies a versioned Sensitive Data Policy +and consumes values through database-independent streaming adapters. Database-specific code may +read and normalize bounded values, but it may not decide sensitivity. + +The classifier first applies deterministic metadata rules, value validators, checksums, +dictionaries, and length rules. A single validated sensitive match makes the whole column +`sensitive`; any textual value longer than 500 characters is such a match. It attempts a complete +scan, but after five seconds per table it continues by sampling within the remaining run budget. A +completed scan with no finding may produce `non_sensitive`; an incomplete scan with no finding +produces `unknown`. + +Ambiguous text may additionally be sent to an optional local NER detector only while time remains. +The detector runs on CPU, receives no tools or network access, does not persist source values, and +returns evidence rather than the column decision. The initial supported detector is +[`fastino/gliner2-privacy-filter-PII-multi`](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi), +used through the Apache-2.0 GLiNER2 Python library with a pinned model revision. Its model weights +and GLiNER2 code are Apache-2.0, and its mDeBERTa base model is MIT. It is trained for seven +languages including Italian and can run on CPU without using the installation's GPUs. + +The NER detector is an optional installation asset because its weights and runtime are materially +larger than the deterministic TypeScript engine. It is invoked only for otherwise unresolved text, +never for values already classified by a decisive rule. If it is disabled, unavailable, times out, +or returns no qualifying evidence before the deadline, the classifier follows the same coverage +rule and may return `unknown`. + +No generative LLM, internal or external, participates in sensitivity assessment. A fallback to an +installation-local LLM is unnecessary while a permissively licensed local NER implementation is +available, and would reintroduce queue latency, non-deterministic judgments, prompt-injection +surface, and contention with normal inference. Introducing such a fallback would require a new +decision based on evidence that the NER path is unusable. + +This decision supersedes ADR-0011 only where that ADR assigns draft sensitivity suggestions to an +AI using structural metadata. ADR-0011's human authority, transient draft, explicit scope, and +Sensitive Data Flag remain in force. How description generation uses the flag is outside this +decision. + +The selected model's published evaluation is not an Italian production acceptance test. Before +enabling the NER profile by default, ThothII must pin the artifacts, generate a dependency/license +inventory, and pass a CPU benchmark plus a labeled Italian corpus representative of the target +databases. Failure of those gates disables NER; it does not silently select another model or an +LLM. diff --git a/docs/plans/2026-09-02-installation-model-catalog.md b/docs/plans/2026-09-02-installation-model-catalog.md new file mode 100644 index 00000000..48508562 --- /dev/null +++ b/docs/plans/2026-09-02-installation-model-catalog.md @@ -0,0 +1,245 @@ +# Installation Model Catalog + +Status: accepted design; implementation not started. + +## Outcome + +`thothii-installation.yaml` is the only operator-authored source for models used by interactive +sessions, metadata generation, and embedding. Runtime-specific files are deterministic projections, +not additional configuration sources. Workspace descriptors contain database and Evidence concerns +and no model, provider, allowlist, default, embedding, or vector-store configuration. + +This design does not merge execution lifecycles. Pi continues to run interactive sessions, the +short-lived LiteLLM helper continues to perform metadata generation, and the internal Ollama service +continues to provide embeddings. They share model declaration, not execution machinery. + +## Canonical installation shape + +The following example covers all currently required cases: a Pi built-in model, an authenticated +custom endpoint, a keyless internal endpoint, metadata generation, and the single embedding model. + +```yaml +schemaVersion: 2 +profile: server +projectDirectory: /srv/thothii +envFile: /srv/thothii/operator.env + +workspaceRepository: + remote: git@git.example.com:organization/workspaces.git + branch: main + access: ssh + +modelCatalog: + defaults: + session: zai/glm-5.3 + metadataGeneration: local-qwen/qwen3.6-35b-a3b + + embedding: + id: ollama/qwen3-embedding:0.6b + dimensions: 1024 + + providers: + deepseek: + authentication: + mode: pi_auth + session: + mode: pi_builtin + models: + deepseek-v4-pro: + session: {} + deepseek-v4-flash: + session: {} + + zai: + endpoint: + baseUrl: https://api.z.ai/api/coding/paas/v4 + authentication: + mode: secret_env + apiKeyEnv: ZAI_API_KEY + session: + mode: openai_compatible + metadataGeneration: + litellmProvider: openai + models: + glm-5.3: + label: GLM-5.3 + session: + reasoning: true + contextWindow: 200000 + maxTokens: 131072 + metadataGeneration: {} + + local-qwen: + endpoint: + baseUrl: https://ml-aritmolab.policlinicosandonato.it/v1 + authentication: + mode: none + session: + mode: openai_compatible + metadataGeneration: + litellmProvider: openai + models: + qwen3.6-35b-a3b: + label: Qwen3.6 35B A3B + session: + reasoning: false + contextWindow: 131072 + maxTokens: 16384 + compatibility: + supportsDeveloperRole: false + supportsReasoningEffort: false + supportsStore: false + maxTokensField: max_tokens + metadataGeneration: + disableThinking: true + +authentication: + configDirectory: /srv/thothii/auth-canonical + runtimeProjection: + directory: /srv/thothii/auth-runtime + uid: 10001 + gid: 10001 +``` + +The catalog uses maps instead of repeated IDs. The canonical identity of a model is always derived +as `/`. `upstreamModel` may be added to a model only when the endpoint uses +a different identifier. `label` is optional and falls back to the canonical identity. + +Model eligibility is not repeated in an `usages` array. A `session` block makes the model eligible +for sessions; a `metadataGeneration` block makes it eligible for metadata generation. The embedding +is a single required installation value rather than a list plus default. + +## Provider and authentication rules + +A provider owns one endpoint, one authentication mode, and zero or one adapter for each runtime. +Model entries cannot override provider endpoint or credentials. If the same upstream service needs +different endpoints or credentials, the installation declares two provider identities. + +Supported session modes are intentionally closed: + +- `pi_builtin`: Pi already owns the model's technical descriptor; the model's `session` block is + empty and ThothII does not copy context-window or compatibility facts. +- `openai_compatible`: ThothII generates a Pi custom-provider descriptor; each session model supplies + the technical values required by Pi. + +Metadata generation uses the provider-level `litellmProvider`. A model-level +`metadataGeneration.disableThinking: true` is permitted only for an explicit compatible endpoint. +There is no generic adapter or plugin abstraction in schema version 2. + +Exactly one provider authentication mode is allowed: + +- `secret_env` requires an approved API-key environment reference present in the protected secret + bundle. Secret values never enter YAML, generated files, logs, arguments, or API responses. +- `pi_auth` is valid only for session-only `pi_builtin` providers and resolves through Pi's protected + authentication projection. +- `none` is valid only for an explicit endpoint. Runtime projections may supply a fixed non-secret + compatibility placeholder when a client library requires a non-empty key. + +## Defaults and selections + +`defaults.session` and `embedding` are required. `defaults.metadataGeneration` is required exactly +when at least one model has a `metadataGeneration` block; metadata generation may otherwise be +absent and its UI controls are disabled. + +`modelCatalog.defaults.session` is the only configured session-model default. `PI_PROVIDER`, +`PI_MODEL`, and provider/model fields in installation-default settings are removed. A user choice is +a Model Selection containing only the canonical model identity and runtime controls such as thinking +level. A session manifest pins the selected canonical identity. + +Removing the currently selected model causes new-session selection to fall back to the catalog +default with an explicit administrative warning. An existing session is never silently moved to a +different model; resume fails with `model_unavailable` when its pinned identity can no longer be +resolved. + +## Generated runtime projections + +Before Compose starts, `tht` strictly validates schema version 2 and generates installation-local +artifacts below `deploy//generated/`: + +- a normalized catalog JSON consumed defensively by the backend; +- Pi `models.json` for custom providers; +- Pi `settings.json`, combining fixed product settings with the session-eligible canonical IDs; +- a Compose override that mounts the projections and supplies embedding identity and dimensions to + core, preprocessing, and `embedding-model-init`. + +Generation is deterministic and published only after every candidate artifact validates. A failed +generation aborts start before Compose is invoked. `tht doctor` recomputes expected bytes and reports +differences; no digest manifest or separate apply command exists. When projection bytes change, +`tht start` recreates the affected services so they cannot continue with an older bind mount. + +Generated projections are not backed up. Restore validates the canonical installation descriptor, +regenerates every projection, and only then starts services. Base Compose files and `operator.env` +must contain no model identities, defaults, endpoints, or dimensions. + +## Workspace schema v4 + +Workspace schema v4 removes both top-level `llm_policy` and `semantic_index`. The entire latter +block is redundant today: its engine and distance are product constants, its collection duplicates +the workspace ID, and its model and dimensions are installation facts. + +The runtime derives: + +- Qdrant collection identity from the workspace ID; +- engine and distance from the supported product contract; +- embedding identity and dimensions from the Installation Model Catalog. + +The published index generation records the canonical embedding identity and dimensions that created +it. A mismatch makes the index explicitly incompatible and requires operator-triggered +preprocessing. No existing index is deleted or rebuilt automatically. + +The v3-to-v4 workspace migration is deterministic: set `workspace.schema_version` to `4`, remove +`llm_policy`, and remove `semantic_index`. It does not alter database, Evidence, diagnostics, or +binding data. + +## Installation migration + +Legacy installation migration must inspect all three former sources: + +1. `metadataGeneration` in `thothii-installation.yaml`; +2. `deploy/pi/models.json`; +3. `deploy/pi/settings.json`. + +The migrator emits a version-2 candidate only when it can reconcile identities, endpoints, +credentials, and runtime-specific facts without guessing. Ambiguous aliases such as `glm-53`, +`zai/glm-5.3`, and `openai/glm-5.3` are not silently equated. A conflict produces a field-level +report and leaves every input unchanged for operator resolution. + +After migration, the strict loader rejects `metadataGeneration`, workspace `llm_policy`, workspace +`semantic_index`, legacy Pi source files, unknown fields, duplicate YAML keys, invalid defaults, and +incompatible authentication/adapter combinations with an actionable `migration_required` or +validation error. + +## Final simplicity audit + +The accepted design removes every configuration duplication that can be removed without inference: + +- one authored installation file instead of an installation block plus two Pi files; +- one canonical `provider/model` identity instead of display IDs and runtime IDs; +- per-use blocks instead of a duplicated usages list; +- one embedding entry instead of a selectable embedding catalog; +- one catalog session default instead of environment and settings defaults; +- no model or vector-store fields in workspace descriptors; +- provider-level credentials instead of per-model credentials; +- no generic runtime-plugin abstraction; +- no persisted digest, apply command, or backup of generated projections. + +The remaining generated files are necessary boundary adapters, not configuration concepts. Making +the backend parse the authoring YAML independently would remove one file but restore two semantic +validators. Hard-coding embedding values in Compose would remove one projection but restore a model +source outside the catalog. Inferring authentication from missing fields would save one YAML key but +turn a safe explicit choice into ambiguity. These apparent simplifications are therefore rejected. + +No further reduction was found that preserves one authority, strict validation, explicit security, +session determinism, and model-free workspaces. + +## Implementation surface + +Implementation must update the host `tht` installation loader, setup and lifecycle projection, +doctor, backup/restore, Compose mounts and embedding inputs, backend catalog/settings/session model +resolution, workspace schema and migration, runtime rendering and diagnostics, frontend workspace +drafts and model filtering, examples, fixtures, and documentation. Existing session manifests remain +readable and keep their pinned provider/model identity; only resume resolution changes to the new +catalog. + +This document authorizes design only. Software implementation begins only after a separate explicit +request. diff --git a/docs/research/2026-09-02-local-sensitive-column-classifier-libraries.md b/docs/research/2026-09-02-local-sensitive-column-classifier-libraries.md new file mode 100644 index 00000000..e742ace1 --- /dev/null +++ b/docs/research/2026-09-02-local-sensitive-column-classifier-libraries.md @@ -0,0 +1,218 @@ +# Local sensitive-column classifier: TypeScript or Python + +**Date:** 2026-09-02 +**Scope:** library/runtime choice for a local classifier without a generative LLM. Synthetic data generation and +model-boundary policy for description generation are deliberately out of scope. + +## Recommendation + +Implement the mandatory classifier **inside the TypeScript backend**, with no learned model in its +mandatory path. The agreed policy is dominated by deterministic work: +type/declared-size rules, bounded string normalization, patterns, checksums, exact dictionaries, +context terms, and "one match makes the column sensitive". TypeScript is fully adequate for that +work and keeps the classifier in the process which already owns the catalog, source connectors, +run lifecycle, cancellation, and the human override. + +Use small, focused dependencies rather than a general NLP framework: + +- `validator` for syntax/checksum-oriented recognizers such as email, IP, MAC, IBAN, BIC, payment + cards, and locale-aware Italian tax IDs, passports, identity cards and VAT numbers; its official + API exposes these validators individually, so unused validators need not be imported + ([validator.js source and API](https://github.com/validatorjs/validator.js/)). +- `libphonenumber-js/max` for international telephone parsing and digit-pattern validation. The + project's `max` metadata is intentionally more precise than its default/minimal metadata + ([libphonenumber-js validation documentation](https://github.com/catamphetamine/libphonenumber-js/blob/master/README.md#isvalid-boolean)). +- ThothII-owned recognizers for identifiers not covered adequately by a selected validator (for + example the Italian driving licence) and organization-specific identifiers, each with a version, + tests, context words, and checksum/semantic validation where the identifier defines one. +- A safe regular-expression boundary for administrator-authored rules. `node-re2` offers a mostly + `RegExp`-compatible, ReDoS-resistant engine and a set API for matching many patterns, but is a + native addon that downloads or builds a binary during installation. That packaging cost must be + tested against the project images before adoption + ([node-re2 README](https://github.com/uhop/node-re2/blob/master/README.md)). If arbitrary regular + expressions are not exposed, ThothII can instead keep a reviewed built-in pattern set and bound + every inspected value. + +Do **not** add NLP.js, spaCy, Presidio, or a transformer merely to implement regexes and dictionaries. +They do not make deterministic identifiers more semantically understandable. In particular, NLP.js +documents useful Italian tokenization plus enum/regex/built-in entities, but its advertised built-ins +are mostly the same formatted values already covered above; it does not advertise a ready-made +Italian PII model for arbitrary people or clinical concepts +([NLP.js official README](https://github.com/axa-group/nlp.js/)). + +Add an **optional, non-generative statistical NER pack** based on +[`fastino/gliner2-privacy-filter-PII-multi`](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi). +The selected model and its GLiNER2 runtime are Apache-2.0, its mDeBERTa base is MIT, and the model +is trained across seven languages including Italian. Run the official implementation in one warmed, +CPU-only Python worker; do not spawn it once per column or table. It returns entity evidence only. +The TypeScript classifier remains the sole owner of the final column assessment and invokes NER +only for bounded text that deterministic rules did not resolve and only while the scan deadline has +time remaining. + +This is a positive Q29 decision, not a placeholder for later library selection. The production gate +is limited to pinning artifacts, producing an SBOM/license inventory, and measuring the selected +model on PSD fixtures and target CPUs. If the optional pack is absent or misses its deadline, the +normal coverage rule produces `unknown`; ThothII does not substitute a generative LLM. + +## Why the TypeScript baseline is sufficient + +The classifier is not being asked to infer an open-ended privacy judgment from prose. It executes a +versioned policy and produces evidence. Its mandatory recognizers can be expressed as: + +| Rule family | Implementation | Model needed? | +| --- | --- | --- | +| declared `text`/CLOB/unbounded string or declared maximum `>500` | catalog metadata rule | no | +| an observed value longer than 500 characters | length rule, ideally detected in SQL before transferring the value | no | +| email, IP/MAC, URL, UUID, payment card, IBAN/BIC, phone | pattern plus format/checksum validator | no | +| Italian fiscal code/VAT/passport/identity card/driving licence | locale-aware validator where available, otherwise a reviewed country recognizer; pattern, context and checksum where defined | no | +| credentials and technical secrets | anchored prefixes, token shapes, entropy/character-class heuristics, and organization allow/deny rules | no | +| hospital- or customer-specific terms/codes | normalized exact dictionary or phrase matching | no | +| person/location/organization in otherwise unstructured short text | statistical NER is useful but probabilistic | yes, optional | +| clinical meaning in unstructured short Italian text | domain NER or a governed terminology; generic NER is not enough | optional and domain-specific | + +This is not speculative parity with a Python implementation. Presidio's own supported-entity table +shows that its global email, IBAN, credit-card and similar recognizers use pattern matching, +validation, context and checksums, while its Italian fiscal-code/VAT/passport/identity-card/licence +coverage is likewise rule-based +([Presidio supported entities](https://presidio.dataprivacystack.org/supported_entities/)). Those +mechanisms are portable; Presidio provides tested implementations and orchestration, not a unique +Python-language capability. + +For large dictionaries, exact normalized token/phrase matching is still deterministic NLP and does +not require a learned model. The policy format should describe the rule rather than the library, +for example `kind: email`, `kind: checksum_id`, `kind: regex`, `kind: phrase_set`, and +`kind: max_length`. This makes a future engine change possible without changing persisted policy or +assessment evidence. + +## What Python/Presidio would materially add + +Presidio is the strongest Python option if ThothII later needs a broader recognizer ecosystem. Its +Analyzer composes rule-based recognizers, regexes, validation, NER and contextual enhancement; it +also exposes custom recognizers, batch processing and decision tracing +([Presidio Analyzer architecture](https://presidio.dataprivacystack.org/analyzer/)). Its current +catalog includes Italian identifiers and its YAML registry supports custom patterns and language- +specific context. Recent source also contains a `NoOpNlpEngine`, so a deterministic-only Presidio +deployment can avoid loading spaCy; this removes model cost but not the Python runtime/process +boundary +([Presidio `NoOpNlpEngine` integration](https://github.com/data-privacy-stack/presidio/blob/main/presidio-analyzer/presidio_analyzer/recognizer_registry/recognizer_registry.py)). + +`presidio-structured` is conceptually close because it maps tabular columns/JSON keys to detected +entities, but its documented strategies select the most common, highest-confidence, or a mixed +entity. ThothII would still need its own deadline/sampling/`unknown` semantics and human-override +lifecycle; Presidio also lists improved mixed free-text/structured support as future work +([Presidio Structured](https://presidio.dataprivacystack.org/structured/)). + +That benefit does not currently justify using it as ThothII's mandatory engine: + +- The backend currently has no PII/NLP dependencies and is an ESM TypeScript service + (`backend/package.json`); the existing Python environment belongs to the separate harness/core + layer and also has no Presidio or spaCy dependency (`harness/pyproject.toml`). A Python analyzer therefore means a new + ownership and deployment boundary, not simply another import. +- Presidio is Python-based and can be run as a package or REST container. Its REST API intentionally + has no built-in authentication, so a sidecar would have to remain on a protected internal network + and be guarded by infrastructure + ([Presidio FAQ, deployment](https://presidio.dataprivacystack.org/faq/#how-can-i-deploy-presidio-into-my-environment)). +- Presidio defaults to English. Adding Italian requires both an Italian NLP pipeline and adapted + language-specific recognizers/context; language-agnostic regexes alone do not solve that setup + ([Presidio multilingual guidance](https://github.com/data-privacy-stack/presidio/blob/main/docs/analyzer/languages.md)). +- Presidio itself warns that automated detection cannot guarantee finding all sensitive data and + documents an unavoidable false-positive/false-negative trade-off + ([Presidio FAQ](https://presidio.dataprivacystack.org/faq/#what-is-presidio)). It cannot turn a + negative sample into proof that a column is safe. + +## Learned NLP is not an LLM, but it is still a model + +A spaCy NER pipeline does not generate text and is not an LLM. It is a statistical token classifier: +it predicts spans such as `PERSON`, `LOCATION` and `ORGANIZATION`. spaCy explicitly notes that the +predictions depend on the training examples and will not always be correct +([spaCy NER documentation](https://spacy.io/usage/linguistic-features#named-entities)). That makes it +a useful additional **positive detector**, not the authority which declares a sampled column safe. + +spaCy publishes Italian CPU pipelines with an NER component, but they are trained on news/media +data rather than clinical notes. More importantly, all three official packages (`it_core_news_sm`, +`md`, and `lg`) are licensed **CC BY-NC-SA 3.0**, so they are a no-go for a product which can be +used commercially; using the MIT-licensed spaCy runtime does not change the weights' license +([spaCy Italian pipelines](https://spacy.io/models/it), +[`sm` metadata](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_sm-3.8.0.json), +[`md` metadata](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_md-3.8.0.json), +[`lg` metadata](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_lg-3.8.0.json)). A general Italian model also recognizes +names and places, not all sensitive clinical meaning. Training a suitable model would require a +labeled, representative and legally usable corpus; Presidio's own model guidance likewise states +that a labeled PII dataset is required to train a new model +([Presidio spaCy/Stanza guidance](https://github.com/data-privacy-stack/presidio/blob/main/docs/analyzer/nlp_engines/spacy_stanza.md#training-your-own-model)). + +Consequently, the absence of a local LLM is irrelevant to sensitivity assessment. Installations +with a local LLM may let description generation receive broader source content under a separate +model-boundary policy, but **the sensitivity classifier does not call that LLM**. The optional NER +pack is a distinct local classifier capability and is never inferred from the description-model +catalog. + +## Licensing and redistribution gate + +Framework code, model weights, and training data are three separate licensing layers. A permissive +runtime does not grant ThothII permission to bundle weights trained from restricted data. For an +open-source product that may be deployed commercially, the default rule should therefore be: +**reject `NC`, research-only, custom restrictive, missing, or ambiguous model licenses**. Pinning an +approved artifact must also pin its model card, upstream model, training datasets, required notices, +and checksums. This is a technical compatibility screen, not legal advice. + +| Candidate | Framework code | Weights and training-data evidence | Commercial bundling in ThothII | Decision | +| --- | --- | --- | --- | --- | +| Presidio, deterministic recognizers only | MIT; its license permits use and redistribution with the notice ([Presidio license](https://github.com/data-privacy-stack/presidio/blob/main/LICENSE)) | No general Italian model is required with `NoOpNlpEngine`; third-party dependencies still retain their notices | Compatible, but adds a Python process without adding unique detection capability for the agreed baseline | **GO legally; NO-GO architecturally for v1** | +| spaCy runtime | MIT ([spaCy license](https://github.com/explosion/spaCy/blob/master/LICENSE)) | Runtime only; no permission is conferred on downloaded pipelines | Compatible if used with separately approved weights | **GO for code only** | +| Official spaCy Italian `sm`/`md`/`lg` | spaCy code is MIT | Every current package declares CC BY-NC-SA 3.0; the metadata traces part of training to UD Italian ISDT under the same non-commercial license ([`sm`](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_sm-3.8.0.json), [`md`](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_md-3.8.0.json), [`lg`](https://raw.githubusercontent.com/explosion/spacy-models/master/meta/it_core_news_lg-3.8.0.json)) | `NC` excludes the intended commercial deployments; `SA` adds redistribution obligations | **NO-GO** | +| Stanza runtime | Apache-2.0 ([Stanza license](https://github.com/stanfordnlp/stanza/blob/main/LICENSE)) | Stanford says its language packs are ODC-By only to the extent it owns the rights and explicitly tells users to inspect training-data licenses. Italian NER uses FBK's KIND data; KIND annotations are CC BY-NC 4.0 or CC BY-NC-SA 4.0 ([Stanza model licensing note](https://stanfordnlp.github.io/stanza/performance.html), [Italian NER source](https://stanfordnlp.github.io/stanza/ner_models.html), [KIND license](https://github.com/dhfbk/KIND#license)) | The Italian model's non-commercial source data fails the product requirement | **GO for code; NO-GO for Italian weights** | +| Flair runtime and official multilingual NER | MIT ([Flair repository and license](https://github.com/flairNLP/flair)) | The official `ner-multi` model card neither includes Italian among its four trained languages nor declares a model license; a maintainer issue asking whether bundled weights are MIT was closed without an answer ([model card](https://huggingface.co/flair/ner-multi), [license issue](https://github.com/flairNLP/flair/issues/1487)) | Runtime is usable, but these weights are unsuitable and their redistribution terms are not established | **GO for code; NO-GO for these weights** | +| `osiria/flare-it-ner` | Runs in Transformers/PyTorch; the card labels the weights MIT | Trained on CC BY 4.0 WikiNER plus an additional manually annotated custom dataset whose redistribution terms are not identified in the model card ([model card](https://huggingface.co/osiria/flare-it-ner), [WikiNER dataset](https://figshare.com/articles/dataset/Learning_multilingual_named_entity_recognition_from_Wikipedia/5462500)) | The weights look promising, but an MIT label alone does not resolve the undocumented custom training-data rights | **NO-GO until provenance is explicit** | +| Transformers.js / ONNX runtime | Apache-2.0 ([Transformers.js license](https://github.com/huggingface/transformers.js/blob/main/LICENSE)) | ONNX conversion does not replace the source model's license or cure training-data restrictions; each exact model needs its own audit | Compatible runtime, but there is no blanket approval for arbitrary Hub/ONNX weights | **GO for code; weights case by case** | +| `Laibniz/italian-ner-pii-browser-uncased` | Packaged for Transformers.js as quantized ONNX; model card says Apache-2.0 | It is a conversion of an Osiria model and therefore inherits the same undocumented custom-data provenance ([model card](https://huggingface.co/Laibniz/italian-ner-pii-browser-uncased)) | CPU/browser packaging does not cure the upstream licensing gap | **NO-GO** | +| `Ar86Bat/multilang-pii-ner` | XLM-R/Transformers; model card labels weights MIT | Trained on `ai4privacy/open-pii-masking-500k-ai4privacy`, whose card declares CC BY 4.0 ([model card](https://huggingface.co/Ar86Bat/multilang-pii-ner), [dataset card](https://huggingface.co/datasets/ai4privacy/open-pii-masking-500k-ai4privacy)) | Commercial redistribution appears compatible with MIT notice plus CC BY attribution; an ONNX export, dependency audit, and PSD evaluation are still required | **CONDITIONAL GO as optional PII NER** | +| `urchade/gliner_multi_pii-v1` | GLiNER code and model are Apache-2.0 | Its published multilingual synthetic dataset, including Italian, is Apache-2.0 ([model card](https://huggingface.co/urchade/gliner_multi_pii-v1), [dataset card](https://huggingface.co/datasets/urchade/synthetic-pii-ner-mistral-v1)) | The cleanest publicly inspectable license chain, but the published SPY comparison reports substantially lower recall than the selected Fastino model | **GO legally; reserve candidate** | +| `fastino/gliner2-privacy-filter-PII-multi` | GLiNER2 code is Apache-2.0; the direct local dependencies are permissively licensed | Weights are Apache-2.0 and the mDeBERTa base is MIT. The authors report a purpose-built synthetic corpus spanning seven languages, including Italian; the corpus itself is not published, so training is not fully reproducible ([model card](https://huggingface.co/fastino/gliner2-privacy-filter-PII-multi), [paper](https://arxiv.org/abs/2605.09973), [base model](https://huggingface.co/microsoft/mdeberta-v3-base)) | Free local/commercial use and redistribution are allowed subject to Apache notices. It has the best reported exact-span F1 and recall among the compared open detectors on SPY, although that evaluation is not Italian-specific | **SELECTED optional NER pack** | +| `openai/privacy-filter` | Code and weights are Apache-2.0 and it runs locally through Python or Transformers.js | The publisher explicitly permits commercial deployment, but documents the model as primarily English and warns that non-English performance may drop ([model card](https://huggingface.co/openai/privacy-filter), [source](https://github.com/openai/privacy-filter)) | Clean license and direct TypeScript path, but insufficient Italian evidence and a larger one-billion-parameter artifact make it a weaker fit | **GO legally; NO-GO as primary Italian detector** | +| OpenMed Italian PII models | Cards label the weights Apache-2.0 | They are trained on `ai4privacy/pii-masking-400k`, whose license limits use to academic/non-commercial purposes unless a commercial agreement is obtained ([representative model](https://huggingface.co/OpenMed/OpenMed-PII-Italian-BioClinicalBERT-Base-110M-v1), [dataset license](https://huggingface.co/datasets/ai4privacy/pii-masking-400k/blob/main/LICENSE)) | An Apache label on derivative weights does not remove the explicit upstream non-commercial restriction | **NO-GO** | + +The search therefore found a usable, free-of-charge, permissively licensed solution. Fastino +GLiNER2-PII is selected because it combines an Apache code/weight grant, explicit Italian training, +CPU operation, configurable entity labels, and materially higher published recall than the older +Urchade model. This does not prove Italian clinical accuracy: the published SPY evaluation is +English and the training corpus is synthetic. Those are quality and reproducibility limitations, +not a reason to reintroduce a generative LLM into the classifier. + +## Fit with current ThothII design + +The current backend already owns source sampling and the human-owned `Sensitive Data Flag`; source +values are transient and existing description sampling is bounded (`PROJECT_STATE.md` and +`backend/src/catalog/description-source-sampler.ts`). The new +classifier should therefore be a backend module behind the single effective sensitivity-decision +point already being designed, while preserving these agreed semantics: + +1. `sensitive`, `non_sensitive`, and `unknown` are assessment outcomes; `unknown` leaves the + human-owned Sensitive Data Flag unchanged. Its effect on later description generation is out of + scope here. +2. Any positive rule match stops the column scan and yields `sensitive`. +3. A negative time-bounded sample yields `unknown`, never `non_sensitive`. +4. The administrator may set the final binary flag either way; the automated result remains a + proposal with rule/version/coverage evidence and no stored source value. +5. Learned NER, if present, can add positive evidence but cannot convert `unknown` to + `non_sensitive`. + +The five-second scan budget should be enforced at the database-read orchestration layer and not +inside a recognizer library. An in-process deterministic engine maximizes the portion of that budget +available to database reads and avoids model cold-start/IPC costs. It should stream bounded batches, +check cancellation/deadline between batches, and stop at the first match. + +## Acceptance gate for enabling the selected NER pack + +Build a versioned, value-free fixture corpus representing at least: Italian/general identifiers, +false-positive lookalikes, bounded short notes with people/places, credential formats, null/high- +cardinality columns, and PSD-specific clinical codes. Measure per category rather than one aggregate +score. Pin the GLiNER2 package and model revision/checksum, build an SBOM including transitive +dependencies, retain required notices, prohibit model downloads at inference time, and measure warm +latency and memory on the target CPU. The detector is enabled only where those checks pass. + +The acceptance test may tune label descriptions and per-label confidence thresholds, but it does +not reopen the architecture or silently select another Hub model. If quality or performance is +unacceptable, the optional NER pack remains disabled and unresolved sampled columns remain +`unknown`. Replacing the chosen model or adding a local-LLM fallback requires a new documented +decision. diff --git a/frontend/prototypes/database-management/fleet.tsx b/frontend/prototypes/database-management/fleet.tsx index 95059b8d..960e38b3 100644 --- a/frontend/prototypes/database-management/fleet.tsx +++ b/frontend/prototypes/database-management/fleet.tsx @@ -278,16 +278,16 @@ export function FleetSelectionActions({ props, direction = "row" }: { props: Var { id: "databases-delete-relationships", group: "Cleanup", label: "Delete all relationships", icon: , executeLabel: "Review deletion", destructive: true, disabled, onClick: run("Delete relationships confirmation opened", "changes") }, ]; if (state.inventoryLevel === "tables") actions = [ - { id: "tables-generate", group: "Descriptions", label: "Generate descriptions", icon: , disabled, onClick: run(`Description generation started for ${count} table${count === 1 ? "" : "s"}`) }, + { id: "tables-generate", group: "Descriptions", label: "Generate table descriptions", icon: , disabled, onClick: run(`Description generation started for ${count} table${count === 1 ? "" : "s"}`) }, { id: "tables-sync-columns", group: "Synchronization", label: "Synchronize columns", icon: , disabled, onClick: run("Selected table column synchronization started", "running") }, - { id: "tables-consolidate", group: "Descriptions", label: "Move generated to Description", icon: , disabled, onClick: run("Generated descriptions moved to curated Description") }, + { id: "tables-consolidate", group: "Descriptions", label: "Copy generated descriptions to Description", icon: , disabled, onClick: run("Generated descriptions copied to curated Description") }, { id: "tables-suggest-sensitive", group: "Sensitive data", label: "Suggest sensitive fields", icon: , executeLabel: "Open review", disabled, onClick: () => onOpenPanel("sensitive-review") }, { id: "tables-delete-columns", group: "Cleanup", label: "Delete all columns", icon: , executeLabel: "Review deletion", destructive: true, disabled, onClick: run("Delete table columns confirmation opened", "changes") }, { id: "tables-delete-relationships", group: "Cleanup", label: "Delete all relationships", icon: , executeLabel: "Review deletion", destructive: true, disabled, onClick: run("Delete table relationships confirmation opened", "changes") }, ]; if (state.inventoryLevel === "columns") actions = [ - { id: "columns-generate", group: "Descriptions", label: "Generate descriptions", icon: , disabled, onClick: run(`Description generation started for ${count} column${count === 1 ? "" : "s"}`) }, - { id: "columns-consolidate", group: "Descriptions", label: "Move generated to Description", icon: , disabled, onClick: run("Generated column descriptions consolidated") }, + { id: "columns-generate", group: "Descriptions", label: "Generate column descriptions", icon: , disabled, onClick: run(`Description generation started for ${count} column${count === 1 ? "" : "s"}`) }, + { id: "columns-consolidate", group: "Descriptions", label: "Copy generated descriptions to Description", icon: , disabled, onClick: run("Generated column descriptions copied") }, { id: "columns-suggest-sensitive", group: "Sensitive data", label: "Suggest sensitive fields", icon: , executeLabel: "Open review", disabled, onClick: () => onOpenPanel("sensitive-review") }, { id: "columns-save-sensitive", group: "Sensitive data", label: `Save sensitive fields (${Object.keys(state.sensitiveDrafts).length})`, icon: , executeLabel: "Save changes", disabled: Object.keys(state.sensitiveDrafts).length === 0, onClick: onSaveSensitive }, ]; diff --git a/frontend/src/auth/LoginPage.tsx b/frontend/src/auth/LoginPage.tsx index f5645739..bac18dbf 100644 --- a/frontend/src/auth/LoginPage.tsx +++ b/frontend/src/auth/LoginPage.tsx @@ -78,7 +78,7 @@ export function LoginPage({ config, onAuthenticated, onRetry }: LoginPageProps)

- From intent to SQL, with Human In The Loop + From intent to SQL, with a human in the loop

AI generates, you guide and approve.

diff --git a/frontend/src/shell/AppShell.tsx b/frontend/src/shell/AppShell.tsx index 505ffa9d..fa4f4ced 100644 --- a/frontend/src/shell/AppShell.tsx +++ b/frontend/src/shell/AppShell.tsx @@ -777,7 +777,7 @@ export function AppShell({ canLogout }: AppShellProps) { <> {activeSession?.question && (
-

Domanda

+

Question

{activeSession.question}

@@ -849,7 +849,7 @@ export function AppShell({ canLogout }: AppShellProps) {

Datamart Builder with
- Human In The Loop + Human-in-the-loop review

{authenticatedUser && (
diff --git a/frontend/src/shell/DatabaseManagementPage.test.tsx b/frontend/src/shell/DatabaseManagementPage.test.tsx index 172c9c76..698385af 100644 --- a/frontend/src/shell/DatabaseManagementPage.test.tsx +++ b/frontend/src/shell/DatabaseManagementPage.test.tsx @@ -321,7 +321,7 @@ test("shows the configured access type and endpoint in catalog status", async () renderPage({ rows: [direct, rest, ssh], presentation: "fleet" }); const directRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); - expect(within(directRow).getByText("Direct PostgreSQL")).toBeVisible(); + expect(within(directRow).getByText("Direct PostgreSQL connection")).toBeVisible(); expect(within(directRow).getByText("db.internal:5432")).toBeVisible(); const restRow = screen.getByRole("row", { name: /REST laboratory/ }); @@ -502,10 +502,10 @@ test("keeps both run-history buttons visible beside the metadata-description sel const metadataControls = screen.getByRole("group", { name: "Metadata description controls" }); const actions = screen.getByRole("group", { name: "Database management actions" }); const descriptionHistoryButton = within(metadataControls).getByRole("button", { - name: "Run descriptions generation history", + name: "View description generation history", }); const suggestionHistoryButton = within(metadataControls).getByRole("button", { - name: "Run sensitive suggestions history", + name: "View sensitive suggestion history", }); expect(toolbar).toHaveClass("sm:items-end", "sm:justify-between"); @@ -513,11 +513,11 @@ test("keeps both run-history buttons visible beside the metadata-description sel expect(metadataControls).toContainElement(selector); expect(descriptionHistoryButton).toBeVisible(); expect(descriptionHistoryButton).toBeEnabled(); - expect(descriptionHistoryButton).toHaveTextContent("Run descriptions generation history"); + expect(descriptionHistoryButton).toHaveTextContent("View description generation history"); expect(descriptionHistoryButton).toHaveClass("disabled:opacity-70"); expect(suggestionHistoryButton).toBeVisible(); expect(suggestionHistoryButton).toBeEnabled(); - expect(suggestionHistoryButton).toHaveTextContent("Run sensitive suggestions history"); + expect(suggestionHistoryButton).toHaveTextContent("View sensitive suggestion history"); expect(toolbar.lastElementChild).toBe(actions); expect(actions).toHaveClass("sm:justify-end"); expect(actions).toContainElement(screen.getByRole("button", { name: "Refresh" })); @@ -675,7 +675,7 @@ test.each(synchronizationScopes)( }, ); -test("discloses source sampling before starting Generate Missing", async () => { +test("discloses source sampling before starting Generate missing descriptions", async () => { const user = userEvent.setup(); const queuedRun = makeDescriptionGenerationRun({ scope: "missing", total: 3 }); let startBody: unknown; @@ -696,7 +696,7 @@ test("discloses source sampling before starting Generate Missing", async () => { const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate Missing" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate missing descriptions" })); expect(screen.getByText("Generate missing descriptions for Policlinico San Donato?")).toBeVisible(); expect(screen.getByRole("note", { @@ -706,7 +706,7 @@ test("discloses source sampling before starting Generate Missing", async () => { ); expect(startBody).toBeUndefined(); - await user.click(screen.getByRole("button", { name: "Generate Missing" })); + await user.click(screen.getByRole("button", { name: "Generate missing descriptions" })); await waitFor(() => expect(startBody).toEqual({ modelId: "local-qwen", scope: "missing", @@ -715,7 +715,7 @@ test("discloses source sampling before starting Generate Missing", async () => { expect(await screen.findByRole("dialog", { name: "Description generation" })).toBeVisible(); }); -test("keeps Fleet Generate Missing behind the source-data disclosure", async () => { +test("keeps Fleet Generate missing descriptions behind the source-data disclosure", async () => { const user = userEvent.setup(); let generationStarts = 0; server.use( @@ -753,7 +753,7 @@ test("keeps Fleet Generate Missing behind the source-data disclosure", async () expect(generationStarts).toBe(0); }); -test("confirms Generate All replacement, supports cancel, and sends the database-wide scope", async () => { +test("confirms generating all descriptions, supports cancel, and sends the database-wide scope", async () => { const user = userEvent.setup(); const queuedRun = makeDescriptionGenerationRun({ scope: "all", total: 6 }); let startBody: unknown; @@ -774,7 +774,7 @@ test("confirms Generate All replacement, supports cancel, and sends the database const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate All" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate all descriptions" })); expect(screen.getByText("Replace generated descriptions for Policlinico San Donato?")).toBeVisible(); expect(screen.getByText(/existing generated descriptions for eligible tables and columns will be replaced/i)).toBeVisible(); @@ -788,8 +788,8 @@ test("confirms Generate All replacement, supports cancel, and sends the database expect(startBody).toBeUndefined(); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate All" })); - await user.click(screen.getByRole("button", { name: "Generate All" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate all descriptions" })); + await user.click(screen.getByRole("button", { name: "Generate all descriptions" })); await waitFor(() => expect(startBody).toEqual({ modelId: "local-qwen", @@ -815,8 +815,8 @@ test("keeps the database selected and safely explains when no descriptions are e const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate Missing" })); - await user.click(screen.getByRole("button", { name: "Generate Missing" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate missing descriptions" })); + await user.click(screen.getByRole("button", { name: "Generate missing descriptions" })); expect(await screen.findByText( "No eligible catalog tables or columns need description generation.", @@ -876,10 +876,10 @@ test.each([ await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); await user.click(await screen.findByRole("menuitem", { - name: scope === "all" ? "Generate All" : "Generate Missing", + name: scope === "all" ? "Generate all descriptions" : "Generate missing descriptions", })); await user.click(screen.getByRole("button", { - name: scope === "all" ? "Generate All" : "Generate Missing", + name: scope === "all" ? "Generate all descriptions" : "Generate missing descriptions", })); expect(await screen.findByRole("heading", { @@ -954,9 +954,9 @@ test.each([ } await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate All" })) + expect(await screen.findByRole("menuitem", { name: "Generate all descriptions" })) .toHaveAttribute("aria-disabled", "true"); - expect(screen.getByRole("menuitem", { name: "Generate Missing" })) + expect(screen.getByRole("menuitem", { name: "Generate missing descriptions" })) .toHaveAttribute("aria-disabled", "true"); }); @@ -983,16 +983,16 @@ test("disables database-wide generation while a description generation is active let databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate Missing" })); - await user.click(screen.getByRole("button", { name: "Generate Missing" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate missing descriptions" })); + await user.click(screen.getByRole("button", { name: "Generate missing descriptions" })); expect(await screen.findByRole("dialog", { name: "Description generation" })).toBeVisible(); databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate All" })) + expect(await screen.findByRole("menuitem", { name: "Generate all descriptions" })) .toHaveAttribute("aria-disabled", "true"); - expect(screen.getByRole("menuitem", { name: "Generate Missing" })) + expect(screen.getByRole("menuitem", { name: "Generate missing descriptions" })) .toHaveAttribute("aria-disabled", "true"); }); @@ -1472,7 +1472,7 @@ test("filters catalog tables as the operator types", async () => { }); }); -test("selected table Actions exposes cleanup commands and sends synchronization ids", async () => { +test("selected table Actions expose table commands and send synchronization ids", async () => { const user = userEvent.setup(); let startBody: unknown; const run = makeSyncRun("columns", [patientsTable.id]); @@ -1494,10 +1494,11 @@ test("selected table Actions exposes cleanup commands and sends synchronization await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - const synchronizeColumns = await screen.findByRole("menuitem", { name: "Synchronize columns" }); + const synchronizeColumns = await screen.findByRole("menuitem", { name: "Synchronize columns for selected tables" }); expect(synchronizeColumns).toBeEnabled(); - expect(screen.getByRole("menuitem", { name: "Delete all columns" })).toBeEnabled(); - expect(screen.getByRole("menuitem", { name: "Delete all relationships" })).toBeEnabled(); + expect(screen.getByRole("menuitem", { name: "Synchronize database tables" })).toBeEnabled(); + expect(screen.getByRole("menuitem", { name: "Remove synchronized columns" })).toBeEnabled(); + expect(screen.queryByRole("menuitem", { name: "Delete all relationships" })).not.toBeInTheDocument(); await user.click(synchronizeColumns); await waitFor(() => expect(startBody).toEqual({ @@ -1546,10 +1547,10 @@ test("starts description generation for multiple selected tables", async () => { await user.click(within(await screen.findByRole("row", { name: /visits/ })).getByRole("checkbox")); await user.click(screen.getByRole("button", { name: "Actions" })); - const generate = await screen.findByRole("menuitem", { name: "Generate descriptions" }); + const generate = await screen.findByRole("menuitem", { name: "Generate table description" }); expect(generate).toBeEnabled(); expect(screen.getByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })).toBeEnabled(); await user.click(generate); @@ -1628,9 +1629,9 @@ test("observes an active run from another browser and reopens a terminal run fro renderPage(); const observe = await screen.findByRole("button", { - name: "Run descriptions generation history", + name: "View description generation history", }); - expect(observe).toHaveTextContent("Run descriptions generation history"); + expect(observe).toHaveTextContent("View description generation history"); expect(requestedLimit).toBe("50"); await user.click(observe); @@ -1654,9 +1655,9 @@ test("keeps the sensitive suggestion history label stable while showing active s renderPage(); const historyButton = await screen.findByRole("button", { - name: "Run sensitive suggestions history", + name: "View sensitive suggestion history", }); - expect(historyButton).toHaveTextContent("Run sensitive suggestions history"); + expect(historyButton).toHaveTextContent("View sensitive suggestion history"); await waitFor(() => expect(historyButton).toHaveAttribute( "title", "Sensitive suggestion generation is active", @@ -1692,7 +1693,7 @@ test.each([ client.setQueryData(unrelatedColumnKey, []); await user.click(await screen.findByRole("button", { - name: "Run descriptions generation history", + name: "View description generation history", })); await waitFor(() => expect(client.getQueryState(tableKey)?.isInvalidated).toBe(true)); @@ -1731,7 +1732,7 @@ test("keeps selected tables when description generation cannot start", async () await user.click(within(await screen.findByRole("row", { name: /patients/ })).getByRole("checkbox")); await user.click(within(await screen.findByRole("row", { name: /visits/ })).getByRole("checkbox")); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate descriptions" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate table description" })); expect(await screen.findByText("A description generation run is already active.")).toBeVisible(); expect(screen.getByText("2 selected")).toBeVisible(); @@ -1801,7 +1802,7 @@ test("refreshes Catalog Tables and Catalog Columns after table generation comple await user.click(within(await screen.findByRole("row", { name: /patients/ })).getByRole("checkbox")); await user.click(within(await screen.findByRole("row", { name: /visits/ })).getByRole("checkbox")); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate descriptions" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate table description" })); expect(await screen.findByRole("heading", { name: "Completed" })).toBeVisible(); await waitFor(() => expect(tableReads).toBe(2)); @@ -1857,7 +1858,7 @@ test("moves selected generated table descriptions and reports copied and skipped await user.click(within(visitsRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); await user.click(await screen.findByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })); await waitFor(() => expect(consolidationBody).toEqual({ @@ -1945,8 +1946,9 @@ test("selects columns, moves generated descriptions, and refreshes only the affe await user.click(within(idRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(nameRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); + expect(await screen.findByRole("menuitem", { name: "Synchronize columns for this table" })).toBeEnabled(); await user.click(await screen.findByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })); await waitFor(() => expect(consolidationBody).toEqual({ @@ -2000,7 +2002,7 @@ test("starts one selected column with the configured default model", async () => expect(columnRow).toBeDefined(); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); await waitFor(() => expect(startBody).toEqual({ modelId: "local-qwen", @@ -2354,7 +2356,7 @@ test("polls a description run and renders its events in sequence order", async ( .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); const drawer = await screen.findByRole("dialog", { name: "Description generation" }); expect(within(drawer).getByText(/local-qwen/)).toBeVisible(); @@ -2436,7 +2438,7 @@ test("refreshes Catalog Tables and Catalog Columns after column generation compl .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); expect(await screen.findByRole("heading", { name: "Completed" })).toBeVisible(); await waitFor(() => expect(columnReads).toBe(2)); @@ -2469,7 +2471,7 @@ test("keeps the selected column and shows a safe message when generation cannot .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); expect(await screen.findByText("A description generation run is already active.")).toBeVisible(); expect(screen.getByText("1 selected")).toBeVisible(); @@ -2536,7 +2538,7 @@ test("shows a basic failed run without exposing private model or provider fields .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); const drawer = await screen.findByRole("dialog", { name: "Description generation" }); expect(await within(drawer).findByRole("heading", { name: "Failed" })).toBeVisible(); @@ -2590,18 +2592,18 @@ test("offers generation and consolidation for multiple selected columns", async .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(idRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate description" })).toBeEnabled(); + expect(await screen.findByRole("menuitem", { name: "Generate column description" })).toBeEnabled(); expect(screen.getByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })).toBeEnabled(); await user.keyboard("{Escape}"); await user.click(within(nameRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - const generate = await screen.findByRole("menuitem", { name: "Generate descriptions" }); + const generate = await screen.findByRole("menuitem", { name: "Generate column descriptions" }); expect(generate).toBeEnabled(); expect(await screen.findByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })).toBeEnabled(); await user.click(generate); @@ -2637,7 +2639,7 @@ test.each([ const tableRow = await screen.findByRole("row", { name: /patients/ }); await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate description" })) + expect(await screen.findByRole("menuitem", { name: "Generate descriptions for all columns" })) .toHaveAttribute("aria-disabled", "true"); await user.keyboard("{Escape}"); await user.click(screen.getByRole("button", { name: "Clear" })); @@ -2647,7 +2649,7 @@ test.each([ await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate description" })) + expect(await screen.findByRole("menuitem", { name: "Generate column description" })) .toHaveAttribute("aria-disabled", "true"); }); @@ -2680,11 +2682,11 @@ test("disables another generation start while the selected-column run is active" .find((row) => within(row).queryByRole("checkbox", { name: /toggle row selection/i })); await user.click(within(columnRow!).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Generate description" })); + await user.click(await screen.findByRole("menuitem", { name: "Generate column description" })); expect(await screen.findByRole("dialog", { name: "Description generation" })).toBeVisible(); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate description" })) + expect(await screen.findByRole("menuitem", { name: "Generate column description" })) .toHaveAttribute("aria-disabled", "true"); expect(screen.getByText("1 selected")).toBeVisible(); @@ -2692,7 +2694,7 @@ test("disables another generation start while the selected-column run is active" const tableRow = await screen.findByRole("row", { name: /patients/ }); await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - expect(await screen.findByRole("menuitem", { name: "Generate description" })) + expect(await screen.findByRole("menuitem", { name: "Generate descriptions for all columns" })) .toHaveAttribute("aria-disabled", "true"); }); @@ -2720,7 +2722,7 @@ test("disables selected description consolidation without database.manage", asyn await user.click(screen.getByRole("button", { name: "Actions" })); expect(await screen.findByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })).toHaveAttribute("aria-disabled", "true"); expect(consolidationCalls).toBe(0); }); @@ -2748,7 +2750,7 @@ test("shows an operation conflict without clearing selected descriptions or refr await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); await user.click(await screen.findByRole("menuitem", { - name: "Move generated description to Description", + name: "Copy generated descriptions to Description", })); expect(await screen.findByText("A database operation is already in progress.")).toBeVisible(); @@ -2756,52 +2758,22 @@ test("shows an operation conflict without clearing selected descriptions or refr expect(tableReads).toBe(1); }); -test("selected table Actions confirms and deletes incoming and outgoing relationships", async () => { +test("selected table Actions do not expose relationship cleanup", async () => { const user = userEvent.setup(); - let cleanupBody: unknown; server.use( http.get("/api/catalog/databases/:databaseId/tables", () => HttpResponse.json([patientsTable])), - http.post("/api/catalog/databases/:databaseId/tables/metadata-cleanup", async ({ request }) => { - cleanupBody = await request.json(); - return HttpResponse.json({ tables: 0, columns: 0, relationships: 2 }); - }), ); - const { client } = renderPage({ + renderPage({ rows: [makeDatabase({ connectionStatus: "reachable", testedVersion: 3 })], }); - const unselectedColumnQuery = [ - "catalog-columns", - patientsTable.databaseId, - "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb", - ] as const; - client.setQueryData(unselectedColumnQuery, []); await user.click(await screen.findByRole("button", { name: "View Policlinico San Donato" })); await user.click(screen.getByRole("tab", { name: "Tables" })); const tableRow = await screen.findByRole("row", { name: /patients/ }); await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Delete all relationships" })); - - expect(screen.getByText("Delete all relationships for 1 table?")).toBeVisible(); - expect(screen.getByText(/incoming and outgoing relationships/i)).toBeVisible(); - expect(cleanupBody).toBeUndefined(); - - await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); - expect(screen.queryByText("Delete all relationships for 1 table?")).not.toBeInTheDocument(); - expect(cleanupBody).toBeUndefined(); - await user.click(within(tableRow).getByRole("checkbox", { name: /toggle row selection/i })); - await user.click(screen.getByRole("button", { name: "Actions" })); - await user.click(await screen.findByRole("menuitem", { name: "Delete all relationships" })); - await user.click(screen.getByRole("button", { name: "Delete relationships" })); - - await waitFor(() => expect(cleanupBody).toEqual({ - tableIds: [patientsTable.id], - target: "relationships", - })); - await waitFor(() => expect(screen.queryByText("1 selected")).not.toBeInTheDocument()); - expect(client.getQueryState(unselectedColumnQuery)?.isInvalidated).toBe(true); - await waitFor(() => expect(screen.getByRole("textbox", { name: "Search tables" })).toHaveFocus()); + expect(screen.queryByRole("menuitem", { name: "Delete all relationships" })).not.toBeInTheDocument(); + expect(screen.queryByRole("menuitem", { name: "Clear catalog relationships" })).not.toBeInTheDocument(); }); test("navigates purely from a database to its tables and edits review metadata", async () => { @@ -2974,9 +2946,10 @@ test("opens the durable job drawer and confirms its exact destructive plan", asy rows: [makeDatabase({ connectionStatus: "reachable", testedVersion: 3 })], }); - await user.click(await screen.findByRole("button", { name: "View Policlinico San Donato" })); - await user.click(screen.getByRole("tab", { name: "Tables" })); - await user.click(await screen.findByRole("button", { name: "Sync tables" })); + const databaseRow = await screen.findByRole("row", { name: /Policlinico San Donato/ }); + await user.click(within(databaseRow).getByRole("checkbox", { name: /toggle row selection/i })); + await user.click(screen.getByRole("button", { name: "Actions" })); + await user.click(await screen.findByRole("menuitem", { name: "Synchronize tables" })); const synchronizationDrawer = await screen.findByRole("dialog", { name: "Schema synchronization", diff --git a/frontend/src/shell/DatabaseManagementPage.tsx b/frontend/src/shell/DatabaseManagementPage.tsx index a8b82a0a..5d4a1c3d 100644 --- a/frontend/src/shell/DatabaseManagementPage.tsx +++ b/frontend/src/shell/DatabaseManagementPage.tsx @@ -779,7 +779,7 @@ export function DatabaseManagementPage({ scope, ); rememberDescriptionGenerationRun(run); - toast.success(`${scope === "all" ? "Generate All" : "Generate Missing"} started for ${database.workspaceName}`); + toast.success(`${scope === "all" ? "Generate all descriptions" : "Generate missing descriptions"} started for ${database.workspaceName}`); } catch (error) { toast.error(apiErrorMessage(error)); throw error; @@ -1274,13 +1274,13 @@ export function DatabaseManagementPage({ onSelectedModelChange={setSelectedMetadataModel} /> {screen.kind === "list" ? <> - - : null}
diff --git a/frontend/src/shell/PiManagement.test.tsx b/frontend/src/shell/PiManagement.test.tsx index 36d371e9..ca840170 100644 --- a/frontend/src/shell/PiManagement.test.tsx +++ b/frontend/src/shell/PiManagement.test.tsx @@ -356,7 +356,7 @@ test("shows an explicit recoverable incomplete state when no provider model is a renderManagement(); const incomplete = await screen.findByRole("alert", { name: "Pi configuration incomplete" }); - expect(incomplete).toHaveTextContent("No enabled provider and model choices are available"); + expect(incomplete).toHaveTextContent("No provider or model options are available"); expect(incomplete).toHaveTextContent("Check the host-managed Pi model configuration"); expect(screen.queryByText("Loading Pi management…")).not.toBeInTheDocument(); expect(screen.getByRole("button", { name: "Save defaults" })).toBeDisabled(); diff --git a/frontend/src/shell/PiManagement.tsx b/frontend/src/shell/PiManagement.tsx index 5be8f7ad..a7189bc0 100644 --- a/frontend/src/shell/PiManagement.tsx +++ b/frontend/src/shell/PiManagement.tsx @@ -78,7 +78,7 @@ function ReadinessRail({ ready, configured, credentials, smokeState }: { state={credentials === "present" ? "ready" : "attention"} /> @@ -505,7 +505,7 @@ export function PiManagement({ open, onClose }: { open: boolean; onClose: () => ) : (
-

No enabled provider and model choices are available.

+

No provider or model options are available.

Check the host-managed Pi model configuration, then retry this panel.

diff --git a/frontend/src/shell/SteerInput.tsx b/frontend/src/shell/SteerInput.tsx index 2f51ecb6..59c189dd 100644 --- a/frontend/src/shell/SteerInput.tsx +++ b/frontend/src/shell/SteerInput.tsx @@ -296,7 +296,7 @@ export function ComposerFooter() { {formatTokenUsage(tokenUsage)} diff --git a/frontend/src/shell/database-management/DatabaseColumns.tsx b/frontend/src/shell/database-management/DatabaseColumns.tsx index f8deffaa..615a045a 100644 --- a/frontend/src/shell/database-management/DatabaseColumns.tsx +++ b/frontend/src/shell/database-management/DatabaseColumns.tsx @@ -10,9 +10,11 @@ import { ApiError, apiErrorMessage } from "../../api/client"; import { consolidateCatalogDescriptions, listCatalogColumns, + startCatalogSync, startDescriptionGenerationRun, updateCatalogColumnMetadata, type CatalogColumn, + type CatalogSyncRun, type CatalogTable, type DescriptionGenerationRun, type SensitiveDataSuggestionRequest, @@ -25,10 +27,12 @@ import { NO_METADATA_GENERATION_LLM_MODEL_MESSAGE } from "./MetadataGenerationMo interface Props { databaseId: string; table: CatalogTable; + databaseVersion: number; canManage: boolean; selectedMetadataModel: string | null; descriptionGenerationActive: boolean; onDescriptionGenerationRunStarted: (run: DescriptionGenerationRun) => void; + onCatalogSyncRunStarted?: (run: CatalogSyncRun) => void; onNavigationStateChange: (state: DatabaseNavigationState) => void; onSuggestSensitive: (selection: SensitiveDataSuggestionRequest, scopeLabel: string) => Promise; catalogOperationActive?: boolean; @@ -88,10 +92,12 @@ function SensitiveCell({ data, context }: ICellRendererParams[] = [ + { + id: "sync-columns", + label: "Synchronize columns for this table", + group: "Synchronization", + scopeLabel: "Selected table", + runLabel: "Synchronize", + disabled: !canManage || selectedIds.length === 0 || catalogOperationActive || busy, + disabledReason: !canManage + ? "You do not have permission to synchronize the catalog." + : selectedIds.length === 0 + ? "Select at least one column." + : catalogOperationActive + ? "Wait for the active catalog operation to finish." + : busy + ? "Another action is running." + : undefined, + }, { id: "generate-descriptions", - label: "Generate descriptions", + label: "Generate column descriptions", group: "Descriptions", runLabel: "Generate", disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || catalogOperationActive || busy, @@ -311,7 +335,7 @@ export function DatabaseColumns({ }, { id: "consolidate-descriptions", - label: "Move generated to Description", + label: "Copy generated descriptions to Description", group: "Descriptions", runLabel: "Move", disabled: !canManage || selectedIds.length === 0 || catalogOperationActive || busy, @@ -345,14 +369,14 @@ export function DatabaseColumns({ }, { id: "save-sensitive", - label: "Save sensitive-field changes", + label: "Save sensitive field changes", group: "Sensitive data", runLabel: "Save", disabled: !canManage || changedSensitiveColumns.length === 0 || catalogOperationActive || busy, disabledReason: !canManage ? "You do not have permission to update sensitive fields." : changedSensitiveColumns.length === 0 - ? "No sensitive-field changes are waiting to be saved." + ? "No sensitive field changes to save." : catalogOperationActive ? "Wait for the active catalog operation to finish." : busy @@ -362,7 +386,8 @@ export function DatabaseColumns({ ]; const runFleetAction = async (action: FleetColumnAction) => { - if (action === "generate-descriptions") await generateDescriptions(); + if (action === "sync-columns") await synchronizeColumns(); + else if (action === "generate-descriptions") await generateDescriptions(); else if (action === "consolidate-descriptions") await consolidateDescriptions(); else if (action === "suggest-sensitive") await suggestSensitive(); else await saveSensitive(); @@ -373,6 +398,22 @@ export function DatabaseColumns({ setSelectedIds([]); }; + const synchronizeColumns = async () => { + if (selectedIds.length === 0) return; + setBusy(true); + try { + const run = await startCatalogSync(databaseId, databaseVersion, "columns", [table.id]); + onCatalogSyncRunStarted?.(run); + gridRef.current?.api.deselectAll(); + setSelectedIds([]); + toast.success("Column synchronization started"); + } catch (error) { + toast.error(apiErrorMessage(error)); + } finally { + setBusy(false); + } + }; + const columns = useMemo[]>(() => [ { field: "ordinalPosition", headerName: "#", width: 64, maxWidth: 64, filter: "agNumberColumnFilter" }, { field: "sensitive", headerName: "Sensitive", headerTooltip: "Mark this column as sensitive; this does not select it for actions.", minWidth: 110, width: 110, sortable: false, filter: false, resizable: false, cellRenderer: SensitiveCell }, @@ -398,7 +439,7 @@ export function DatabaseColumns({

Catalog columns could not be loaded.

{apiErrorMessage(error)}

@@ -456,8 +497,9 @@ export function DatabaseColumns({ - void generateDescriptions()}>Generate {selectedIds.length === 1 ? "description" : "descriptions"} - void consolidateDescriptions()}>Move generated description to Description + void synchronizeColumns()}>Synchronize columns for this table + void generateDescriptions()}>Generate {selectedIds.length === 1 ? "column description" : "column descriptions"} + void consolidateDescriptions()}>Copy generated descriptions to Description diff --git a/frontend/src/shell/database-management/DatabaseFleetQueryErrors.test.tsx b/frontend/src/shell/database-management/DatabaseFleetQueryErrors.test.tsx index dabada5e..1a1211aa 100644 --- a/frontend/src/shell/database-management/DatabaseFleetQueryErrors.test.tsx +++ b/frontend/src/shell/database-management/DatabaseFleetQueryErrors.test.tsx @@ -110,6 +110,7 @@ test("Fleet columns distinguish a failed query from an empty catalog and retry i presentation="fleet" databaseId={database.id!} table={table} + databaseVersion={database.version} canManage selectedMetadataModel={null} descriptionGenerationActive={false} @@ -124,7 +125,7 @@ test("Fleet columns distinguish a failed query from an empty catalog and retry i expect(alert).toHaveTextContent("The database catalog is unavailable."); expect(screen.queryByText(/No columns synchronized/i)).not.toBeInTheDocument(); - await user.click(screen.getByRole("button", { name: "Retry columns" })); + await user.click(screen.getByRole("button", { name: "Retry loading columns" })); await waitFor(() => expect(requests).toBe(2)); await waitFor(() => expect(screen.queryByText("Catalog columns could not be loaded.")).not.toBeInTheDocument()); }); diff --git a/frontend/src/shell/database-management/DatabaseForm.tsx b/frontend/src/shell/database-management/DatabaseForm.tsx index a22e2306..e9869d11 100644 --- a/frontend/src/shell/database-management/DatabaseForm.tsx +++ b/frontend/src/shell/database-management/DatabaseForm.tsx @@ -281,7 +281,7 @@ export function DatabaseForm({ disabled={readOnly || busy} onChange={(event) => onTransportChange(event.target.value as DatabaseTransport)} > - + diff --git a/frontend/src/shell/database-management/DatabaseGrid.tsx b/frontend/src/shell/database-management/DatabaseGrid.tsx index 8e7e3cda..383628ab 100644 --- a/frontend/src/shell/database-management/DatabaseGrid.tsx +++ b/frontend/src/shell/database-management/DatabaseGrid.tsx @@ -149,13 +149,13 @@ function accessSummary(row: CatalogDatabase): { label: string; detail: string } return { label: "SSH", detail: endpoint(row.binding.sshHost, row.binding.sshPort) }; } return { - label: "Direct PostgreSQL", + label: "Direct PostgreSQL connection", detail: endpoint(row.binding.host, row.binding.port), }; } function coverageStatus(total: number, complete: number): { label: string; detail: string; tone: StateTone } { - if (total === 0) return { label: "Not started", detail: "No catalog items", tone: "danger" }; + if (total === 0) return { label: "Not started", detail: "No catalog objects", tone: "danger" }; if (complete >= total) return { label: "Complete", detail: `${complete}/${total}`, tone: "success" }; if (complete > 0) return { label: "Partial", detail: `${complete}/${total}`, tone: "warning" }; return { label: "Not started", detail: `0/${total}`, tone: "danger" }; @@ -172,7 +172,7 @@ function CatalogStatusCells({ row, metrics, sensitiveRun }: { row: CatalogDataba const descriptions = coverageStatus(metrics?.descriptionTargets ?? 0, metrics?.describedTargets ?? 0); const sensitiveProcessed = sensitiveRun ? sensitiveRun.suggestedSensitive + sensitiveRun.suggestedNonSensitive : 0; const sensitive = !sensitiveRun - ? { label: "Not started", detail: "No analysis run", tone: "danger" as const } + ? { label: "Not started", detail: "No analysis has been run", tone: "danger" as const } : sensitiveRun.status === "completed" && sensitiveProcessed >= sensitiveRun.total ? { label: "Complete", detail: `${sensitiveProcessed}/${sensitiveRun.total}`, tone: "success" as const } : ["failed", "interrupted"].includes(sensitiveRun.status) @@ -608,7 +608,7 @@ export function DatabaseGrid({ () => onGenerateDescriptions(selectedRows, pendingGenerationScope), )} > - {pendingGenerationScope === "all" ? "Generate All" : "Generate Missing"} + {pendingGenerationScope === "all" ? "Generate all descriptions" : "Generate missing descriptions"}
) : pendingDelete ? ( @@ -688,14 +688,14 @@ export function DatabaseGrid({ disabled={!canGenerateDescriptions} onClick={() => setPendingGenerationScope("all")} > - Generate All + Generate all descriptions setPendingGenerationScope("missing")} > - Generate Missing + Generate missing descriptions - + @@ -414,7 +414,7 @@ export function DatabaseRelationships({ {addUnavailable ? {addUnavailable} : null} {selectedActionUnavailable ? {selectedActionUnavailable} : null} - ) : } + ) : } {!bindingReady ?
{presentation === "fleet" ? "Test the current database binding in database details before synchronizing relationships." : "Test the current database binding from Overview before synchronizing relationships."}
: null} {fleetQueryError ??
@@ -430,7 +430,7 @@ export function DatabaseRelationships({ title="Add relationship" description="Connect one source column to one primary-key target column." onClose={closeAdd} - closeLabel="Close add relationship" + closeLabel="Close relationship form" busy={busy} footer={( <> diff --git a/frontend/src/shell/database-management/DatabaseTables.tsx b/frontend/src/shell/database-management/DatabaseTables.tsx index b6d2b0bd..17511d10 100644 --- a/frontend/src/shell/database-management/DatabaseTables.tsx +++ b/frontend/src/shell/database-management/DatabaseTables.tsx @@ -330,13 +330,12 @@ export function DatabaseTables({ | "sync-columns" | "consolidate-descriptions" | "suggest-sensitive" - | "clear-columns" - | "clear-relationships"; + | "clear-columns"; const fleetActions: readonly FleetActionOption[] = [ { id: "generate-descriptions", - label: "Generate descriptions", + label: "Generate table description", group: "Descriptions", runLabel: "Generate", disabled: !canManage || selectedIds.length === 0 || !selectedMetadataModel || descriptionGenerationActive || busy !== null || Boolean(currentRun), @@ -372,7 +371,7 @@ export function DatabaseTables({ }, { id: "consolidate-descriptions", - label: "Move generated to Description", + label: "Copy generated descriptions to Description", group: "Descriptions", runLabel: "Move", disabled: !canManage || selectedIds.length === 0 || busy !== null || Boolean(currentRun), @@ -386,22 +385,24 @@ export function DatabaseTables({ }, { id: "sync-tables", - label: "Synchronize tables", + label: "Synchronize database tables", group: "Synchronization", - scopeLabel: "All tables in this database", + scopeLabel: "Selected database", runLabel: "Synchronize", - disabled: !canManage || !bindingReady || busy !== null || Boolean(currentRun), + disabled: !canManage || selectedIds.length === 0 || !bindingReady || busy !== null || Boolean(currentRun), disabledReason: !canManage ? "You do not have permission to synchronize the catalog." - : !bindingReady - ? "Test the current database binding first." - : busy !== null || Boolean(currentRun) - ? "Wait for the active catalog operation to finish." - : undefined, + : selectedIds.length === 0 + ? "Select at least one table." + : !bindingReady + ? "Test the current database binding first." + : busy !== null || Boolean(currentRun) + ? "Wait for the active catalog operation to finish." + : undefined, }, { id: "sync-columns", - label: "Synchronize columns", + label: "Synchronize columns for selected tables", group: "Synchronization", runLabel: "Synchronize", disabled: !canManage || selectedIds.length === 0 || !bindingReady || busy !== null || Boolean(currentRun), @@ -435,22 +436,7 @@ export function DatabaseTables({ }, { id: "clear-columns", - label: "Clear catalog columns", - group: "Catalog cleanup", - tone: "destructive", - runLabel: "Review cleanup", - disabled: !canManage || selectedIds.length === 0 || busy !== null || Boolean(currentRun), - disabledReason: !canManage - ? "You do not have permission to clear catalog metadata." - : selectedIds.length === 0 - ? "Select at least one table." - : busy !== null || Boolean(currentRun) - ? "Wait for the active catalog operation to finish." - : undefined, - }, - { - id: "clear-relationships", - label: "Clear catalog relationships", + label: "Remove synchronized columns", group: "Catalog cleanup", tone: "destructive", runLabel: "Review cleanup", @@ -473,7 +459,6 @@ export function DatabaseTables({ else if (action === "sync-columns") await synchronize("columns", selectedIds); else if (action === "suggest-sensitive") await suggestSensitive(); else if (action === "clear-columns") setPendingDelete("columns"); - else setPendingDelete("relationships"); }; const clearSelection = () => { @@ -544,10 +529,12 @@ export function DatabaseTables({
} @@ -655,10 +642,12 @@ export function DatabaseTables({

{presentation === "fleet" - ? `Clear catalog ${pendingDelete} for ${selectedIds.length} table${selectedIds.length === 1 ? "" : "s"}?` - : pendingDelete === "columns" - ? `Delete all catalog columns for ${selectedIds.length} table${selectedIds.length === 1 ? "" : "s"}?` - : `Delete all relationships for ${selectedIds.length} table${selectedIds.length === 1 ? "" : "s"}?`} + ? `Remove synchronized columns for ${selectedIds.length} table${selectedIds.length === 1 ? "" : "s"}?` + : `Remove synchronized columns for ${selectedIds.length} table${selectedIds.length === 1 ? "" : "s"}?`}

- {presentation === "fleet" - ? pendingDelete === "columns" - ? "Only synchronized catalog columns for these tables will be cleared. The source database is not modified." - : "Only synchronized catalog relationships for these tables will be cleared. The source database is not modified." - : pendingDelete === "columns" - ? "Only columns belonging to the selected catalog tables will be removed." - : "All incoming and outgoing relationships for the selected catalog tables will be removed."} + Only synchronized catalog columns for these tables will be removed. The source database is not modified.

@@ -733,7 +714,7 @@ export function DatabaseTables({ disabled={busy !== null} onClick={() => void deleteMetadata(pendingDelete)} > - {presentation === "fleet" ? "Clear catalog" : pendingDelete === "columns" ? "Delete columns" : "Delete relationships"} + Remove synchronized columns ) : ( @@ -760,13 +741,13 @@ export function DatabaseTables({ - void generateDescriptions()}>Generate {selectedIds.length === 1 ? "description" : "descriptions"} + void generateDescriptions()}>Generate table description void generateColumnDescriptions()}>Generate descriptions for all columns - void synchronize("columns", selectedIds)}>Synchronize columns - void consolidateDescriptions()}>Move generated description to Description + void synchronize("tables")}>Synchronize database tables + void synchronize("columns", selectedIds)}>Synchronize columns for selected tables + void consolidateDescriptions()}>Copy generated descriptions to Description - setPendingDelete("columns")}>Delete all columns - setPendingDelete("relationships")}>Delete all relationships + setPendingDelete("columns")}>Remove synchronized columns @@ -799,12 +780,11 @@ export function DatabaseTables({ setSearch(event.target.value)} /> {search.trim() ? `${displayedCount} of ${data.length}` : displayedCount} - ) )} - {!bindingReady ?
Test the current database binding from Overview before synchronizing tables.
: null} + {!bindingReady ?
Test the current database binding from Overview before synchronizing columns.
: null} {fleetQueryError ??
@@ -825,7 +805,7 @@ export function DatabaseTables({ rowHeight={44} headerHeight={38} animateRows={false} - overlayNoRowsTemplate="No catalog tables. Test the binding, then sync tables." + overlayNoRowsTemplate="No catalog tables. Synchronize the schema from the database view." />
} diff --git a/frontend/src/shell/database-management/FleetActionSelector.test.tsx b/frontend/src/shell/database-management/FleetActionSelector.test.tsx index 1d4d3970..7360116d 100644 --- a/frontend/src/shell/database-management/FleetActionSelector.test.tsx +++ b/frontend/src/shell/database-management/FleetActionSelector.test.tsx @@ -8,7 +8,7 @@ const actions = [ { id: "sync", label: "Synchronize tables", group: "Synchronization", scopeLabel: "All tables in this database" }, { id: "generate", - label: "Generate descriptions", + label: "Generate column descriptions", group: "Descriptions", disabled: true, disabledReason: "Choose a metadata model first", diff --git a/frontend/src/shell/database-management/FleetActionSelector.typography.test.ts b/frontend/src/shell/database-management/FleetActionSelector.typography.test.ts index 14091530..b736cc90 100644 --- a/frontend/src/shell/database-management/FleetActionSelector.typography.test.ts +++ b/frontend/src/shell/database-management/FleetActionSelector.typography.test.ts @@ -27,4 +27,10 @@ test("uses operational body text for every available-command list", () => { expect(styles).toMatch( /\.thot-fleet-action-selector__select\s*\{[^}]*font-size:\s*0\.9375rem;/s, ); + expect(styles).toMatch( + /\.thot-fleet-action-selector\s*\{[^}]*flex:\s*0 0 auto;/s, + ); + expect(styles).toMatch( + /\.thot-fleet-action-selector__select\s*\{[^}]*width:\s*max-content;/s, + ); }); diff --git a/frontend/src/shell/database-management/FleetLedgerShell.css b/frontend/src/shell/database-management/FleetLedgerShell.css index b71d482b..df36aa19 100644 --- a/frontend/src/shell/database-management/FleetLedgerShell.css +++ b/frontend/src/shell/database-management/FleetLedgerShell.css @@ -381,7 +381,8 @@ .thot-fleet-action-selector { display: flex; min-width: 0; - flex: 1; + max-width: 100%; + flex: 0 0 auto; flex-wrap: wrap; gap: 0.5rem; align-items: center; @@ -402,9 +403,10 @@ } .thot-fleet-action-selector__select { - flex: 1 1 13rem; - width: min(100%, 22rem); - min-width: 13rem; + flex: 0 0 auto; + width: max-content; + min-width: 0; + max-width: 100%; height: 2rem; padding: 0 2rem 0 0.625rem; border: 1px solid oklch(var(--input)); @@ -613,8 +615,8 @@ @media (min-width: 48rem) { .thot-fleet-grid-toolbar > .thot-fleet-action-selector { - flex: 1 1 33rem; - min-width: 33rem; + flex: 0 0 auto; + min-width: 0; } } @@ -963,9 +965,6 @@ width: 100%; } - .thot-fleet-action-selector__select { - width: 100%; - } } @media (max-height: 44rem) and (min-width: 48rem) { diff --git a/frontend/src/shell/database-management/HistoryActionPlacement.test.ts b/frontend/src/shell/database-management/HistoryActionPlacement.test.ts index 1eee77ec..fe1a8fc2 100644 --- a/frontend/src/shell/database-management/HistoryActionPlacement.test.ts +++ b/frontend/src/shell/database-management/HistoryActionPlacement.test.ts @@ -12,8 +12,13 @@ describe("history action placement", () => { expect(fleetHeader).not.toContain("Sensitive history"); }); - test("database views expose metadata histories beside database commands", () => { - for (const file of ["DatabaseForm.tsx", "DatabaseTables.tsx", "DatabaseRelationships.tsx"]) { + test("database tables do not own metadata history actions", () => { + for (const file of ["DatabaseTables.tsx"]) { + const source = componentSource(file); + expect(source, file).not.toContain("Description history"); + expect(source, file).not.toContain("Sensitive history"); + } + for (const file of ["DatabaseForm.tsx", "DatabaseRelationships.tsx"]) { const source = componentSource(file); expect(source, file).toContain("Description history"); expect(source, file).toContain("Sensitive history"); diff --git a/frontend/src/viewers/CtePlanViewer.test.tsx b/frontend/src/viewers/CtePlanViewer.test.tsx index aeab867d..882b0ff0 100644 --- a/frontend/src/viewers/CtePlanViewer.test.tsx +++ b/frontend/src/viewers/CtePlanViewer.test.tsx @@ -176,7 +176,7 @@ test("compacts table, filter, detail, and rationale rows", () => { expect(within(rationale!).getByText("Rationale")).toHaveClass("mb-1"); }); -test("shows 'no dependencies' for a CTE with an empty depends_on", () => { +test("shows 'No dependencies' for a CTE with an empty depends_on", () => { render(); expect(screen.getByText(/no dependencies/i)).toBeInTheDocument(); }); diff --git a/frontend/src/viewers/CtePlanViewer.tsx b/frontend/src/viewers/CtePlanViewer.tsx index 53e5138e..e84e9a8e 100644 --- a/frontend/src/viewers/CtePlanViewer.tsx +++ b/frontend/src/viewers/CtePlanViewer.tsx @@ -93,7 +93,7 @@ function CteCard({ cte, total }: { cte: CtePlanCte; total: number }) { ))} ) : ( -

no dependencies

+

No dependencies

)} diff --git a/frontend/src/viewers/SchemaLinkingViewer.test.tsx b/frontend/src/viewers/SchemaLinkingViewer.test.tsx index a5e1a699..fc82ad47 100644 --- a/frontend/src/viewers/SchemaLinkingViewer.test.tsx +++ b/frontend/src/viewers/SchemaLinkingViewer.test.tsx @@ -52,7 +52,7 @@ test("(b) clicking toggle switches to table listing a promoted table name", asyn test("(c) oversized linking (>45) shows cap notice", () => { render(); expect( - screen.getByText(/too many elements for the graph/i) + screen.getByText(/too many nodes to display in the graph/i) ).toBeInTheDocument(); }); diff --git a/frontend/src/viewers/SchemaLinkingViewer.tsx b/frontend/src/viewers/SchemaLinkingViewer.tsx index 6f7f1a34..bf35a39b 100644 --- a/frontend/src/viewers/SchemaLinkingViewer.tsx +++ b/frontend/src/viewers/SchemaLinkingViewer.tsx @@ -147,7 +147,7 @@ export function SchemaLinkingViewer({ linking }: { linking: SchemaLinking }) { {oversized && (

- Too many elements for the graph, use the table + Too many nodes to display in the graph. Use the table instead.

)} diff --git a/frontend/src/widgets/ArtifactGateWidget.tsx b/frontend/src/widgets/ArtifactGateWidget.tsx index 110bccf1..4d9e5453 100644 --- a/frontend/src/widgets/ArtifactGateWidget.tsx +++ b/frontend/src/widgets/ArtifactGateWidget.tsx @@ -40,7 +40,7 @@ export function ArtifactGateWidget({ descriptor, onRespond, sessionId }: WidgetP {descriptor.artifact ? ( ) : ( -

No artifact.

+

No artifact available.

)} diff --git a/frontend/src/widgets/MultiselectWidget.tsx b/frontend/src/widgets/MultiselectWidget.tsx index a049e973..87e5749b 100644 --- a/frontend/src/widgets/MultiselectWidget.tsx +++ b/frontend/src/widgets/MultiselectWidget.tsx @@ -68,7 +68,7 @@ export function MultiselectWidget({ descriptor, onRespond }: WidgetProps) { {o.rationale && (
- Motivo + Reason
{o.rationale} @@ -78,7 +78,7 @@ export function MultiselectWidget({ descriptor, onRespond }: WidgetProps) { {typeof o.meta?.question_context === "string" && o.meta.question_context && (
- Domanda di contesto + Context question
diff --git a/frontend/src/widgets/SelectWidget.test.tsx b/frontend/src/widgets/SelectWidget.test.tsx index 82b8f971..82f97820 100644 --- a/frontend/src/widgets/SelectWidget.test.tsx +++ b/frontend/src/widgets/SelectWidget.test.tsx @@ -5,7 +5,7 @@ import { SelectWidget } from "./SelectWidget"; test("picking an option responds with its id", async () => { const onRespond = vi.fn(); render(); - expect(screen.getByText(/consigliato/i)).toBeInTheDocument(); + expect(screen.getByText(/recommended/i)).toBeInTheDocument(); await userEvent.click(screen.getByRole("button", { name: /A/ })); expect(onRespond).toHaveBeenCalledWith({ id: "u1", kind: "select", choices: ["a"] }); }); diff --git a/frontend/src/widgets/SelectWidget.tsx b/frontend/src/widgets/SelectWidget.tsx index 6d0b3816..6a108ea5 100644 --- a/frontend/src/widgets/SelectWidget.tsx +++ b/frontend/src/widgets/SelectWidget.tsx @@ -25,7 +25,7 @@ export function SelectWidget({ descriptor, onRespond }: WidgetProps) { {o.label} {o.recommended && ( - consigliato + Recommended )}