From 6ba33367753c3e1965dff7ef50ef40e650516061 Mon Sep 17 00:00:00 2001 From: mptyl Date: Fri, 26 Jun 2026 22:25:00 +0200 Subject: [PATCH] feat(harness): config.py + workspace.py (D3 confine) + chirone.example.yaml - config.py ported from ChironeWp3, PSD_PROFILE -> THOTH_PROFILE; dual vector key (vector_rest/vector_write_rest top-level) preserved verbatim - workspace.py: thin boundary wrapper (D3) around load_config - workspaces/chirone.example.yaml: aligned to the real Config shape - tests/test_workspace.py: 2 tests (env expand + dual key; missing env raises) - .env.example: added DWH + DOCS_ROOT vars All tests pass (2/2). --- harness/.env.example | 8 ++ harness/nsp/config.py | 183 ++++++++++++++++++++++++ harness/nsp/workspace.py | 31 ++++ harness/tests/conftest.py | 12 ++ harness/tests/test_workspace.py | 79 ++++++++++ harness/workspaces/chirone.example.yaml | 90 ++++++++++++ 6 files changed, 403 insertions(+) create mode 100644 harness/nsp/config.py create mode 100644 harness/nsp/workspace.py create mode 100644 harness/tests/conftest.py create mode 100644 harness/tests/test_workspace.py create mode 100644 harness/workspaces/chirone.example.yaml diff --git a/harness/.env.example b/harness/.env.example index 1444248f..38f25a96 100644 --- a/harness/.env.example +++ b/harness/.env.example @@ -4,9 +4,17 @@ THOTH_PROFILE=server # Workspace DB (relational, direct transport) — example: chirone THOTH_DB_HOST= THOTH_DB_PORT=5432 +THOTH_DB_NAME= THOTH_DB_USER= THOTH_DB_PASSWORD= +# DWH via REST (richiesto se database.transport = rest) +THOTH_DWH_REST_URL=https://supabase-aritmolab.policlinicosandonato.it/dwh/ +THOTH_DWH_API_KEY= + +# Evidence source root (cartella curata a mano con le evidence) +THOTH_DOCS_ROOT= + # Vector REST — LETTURA (rpc search_similar) sul Supabase remoto THOTH_VEC_REST_URL=https://host/vector/v1/ THOTH_VEC_API_KEY= diff --git a/harness/nsp/config.py b/harness/nsp/config.py new file mode 100644 index 00000000..d3bc8b23 --- /dev/null +++ b/harness/nsp/config.py @@ -0,0 +1,183 @@ +import os +import re +from pathlib import Path +from typing import Any, Literal + +import yaml +from pydantic import BaseModel, Field, ValidationError + +_ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}") + + +class ConfigError(Exception): + """Errore di configurazione, con messaggio leggibile per l'utente.""" + + +def _expand_env(value: Any) -> Any: + if isinstance(value, str): + def repl(m: re.Match) -> str: + var = m.group(1) + if var not in os.environ: + raise ConfigError( + f"Variabile d'ambiente non definita: {var} " + f"(definiscila nel file .env o nell'ambiente)" + ) + return os.environ[var] + + return _ENV_RE.sub(repl, value) + if isinstance(value, dict): + return {k: _expand_env(v) for k, v in value.items()} + if isinstance(value, list): + return [_expand_env(v) for v in value] + return value + + +class DatabaseConfig(BaseModel): + host: str = "localhost" + port: int = 5432 + database: str + db_schema: str = Field(alias="schema") + user: str + password: str + # transport: `direct` (Postgres via SQLAlchemy) o `rest` (Supabase/PostgREST). + # In `rest` deve esistere la sezione `rest` (validato a livello di Config). + transport: Literal["direct", "rest"] = "direct" + + model_config = {"populate_by_name": True} + + +class RestConfig(BaseModel): + """Accesso al DWH via Supabase/PostgREST. base_url es. https://host/dwh/ .""" + + base_url: str + api_key: str + timeout: int = 30 + ssl_ca: str | None = None # path al certificato CA (per server con CA interna) + + +class PathsConfig(BaseModel): + artifacts: Path = Path("artifacts") + indexes: Path = Path("indexes") + sessions: Path = Path("sessions") + + +class ExamplesConfig(BaseModel): + max_per_column: int = 10 + + +class LshSkipConfig(BaseModel): + # DEPRECATO: l'euristica di lunghezza (skip_column vendored) non è più usata. La + # selezione delle colonne da indicizzare segue il principio di column eligibility + # (sezione `eligibility`). Mantenuto solo per compatibilità con nsp.yaml esistenti. + max_total_chars: int = 50000 + max_avg_length: int = 20 + + +class LshConfig(BaseModel): + signature_size: int = 64 + n_gram: int = 3 + threshold: float = 0.5 + max_values_per_column: int = 1000 + skip: LshSkipConfig = LshSkipConfig() + + +class EligibilityConfig(BaseModel): + # Soglie del principio di column eligibility (testo ampio ignorato ovunque). + max_declared_len: int = 128 # char/varchar dichiarati <= soglia: eligible senza campionare + max_avg_length: int = 40 # fallback data-driven: lunghezza media valori campionati + max_sampled_len: int = 200 # fallback data-driven: lunghezza massima valore campionato + # Colonne di servizio sempre ignorate per nome (match case-insensitive), a prescindere + # dal tipo: metadati ETL/audit non analitici (es. timestamp di ultimo aggiornamento). + ignore_columns: list[str] = ["etl_last_update"] + + +class EvidenceSourcesConfig(BaseModel): + source_root: Path + # cartella curata a mano nell'ETL (relativa a source_root): unica fonte delle + # evidence. Niente piu' estrazione automatica dalle schede tabella: i documenti + # qui dentro sono gia' evidence pronte (frontmatter + corpo), scelte e arricchite + # dall'autore ETL e organizzate in sottocartelle per dominio. + evidence_dir: str = "evidence" + + +class EmbeddingsConfig(BaseModel): + base_url: str + model: str = "nomic-embed-text-v2-moe" + dim: int = 768 + batch_size: int = 32 + timeout: int = 120 + + +class VectorConfig(BaseModel): + max_chunk_chars: int = 4000 + + +class SearchConfig(BaseModel): + rrf_k: int = 60 + top_schema_tables: int = 12 # default `--top` per `nsp search --kind schema` (n. tabelle) + schema_chunk_pool: int = 150 # chunk tabella/colonna fusi prima dell'aggregazione a tabella + + +class ExecutionConfig(BaseModel): + allow: list[str] = ["cte_test", "explain", "preview", "aggregate", "export"] + max_preview_rows: int = 10 + statement_timeout_ms: int = 30000 + warn_execution_ms: int = 5000 + warn_plan_rows: int = 1_000_000 + max_aggregate_cells: int = 20 + max_export_rows: int = 100000 + forbidden_functions: list[str] = [ + "setval", "nextval", "pg_advisory_lock", "pg_advisory_xact_lock", + "dblink", "dblink_exec", "pg_terminate_backend", "pg_cancel_backend", + "lo_import", "lo_export", "pg_reload_conf", + ] + + +class Config(BaseModel): + database: DatabaseConfig + # Profilo dell'installazione, letto da THOTH_PROFILE (.env), non dallo yaml versionato. + # server: ricostruisce i derivati (artefatti, LSH, vettori schema nel vectordb). + # workstation: postazione locale che legge il vectordb via REST; gli upsert remoti + # richiedono vector_write_rest, mentre init/clear/rebuild restano solo-server. + profile: Literal["server", "workstation"] = "server" + paths: PathsConfig = PathsConfig() + examples: ExamplesConfig = ExamplesConfig() + lsh: LshConfig = LshConfig() + eligibility: EligibilityConfig = EligibilityConfig() + evidence: EvidenceSourcesConfig | None = None + embeddings: EmbeddingsConfig | None = None + vector_db: DatabaseConfig | None = None + vector: VectorConfig = VectorConfig() + search: SearchConfig = SearchConfig() + execution: ExecutionConfig = ExecutionConfig() + rest: RestConfig | None = None + # Endpoint REST dedicato per la similarity search del pgvector (rpc search_similar). + # Se presente, `nsp search` legge via REST; altrimenti legge in diretto (dev/test). + vector_rest: RestConfig | None = None + # Endpoint REST dedicato alle scritture controllate del pgvector. E' opzionale e usa + # una API key separata dalla lettura; espone solo upsert/hash via RPC allowlist. + vector_write_rest: RestConfig | None = None + + +def load_config(path: Path) -> Config: + if not path.exists(): + raise ConfigError(f"File di configurazione non trovato: {path}") + raw = yaml.safe_load(path.read_text()) + if not isinstance(raw, dict): + raise ConfigError(f"Configurazione non valida (atteso un mapping YAML): {path}") + try: + cfg = Config.model_validate(_expand_env(raw)) + except ValidationError as e: + raise ConfigError(f"Configurazione non valida in {path}:\n{e}") from e + env_profile = os.environ.get("THOTH_PROFILE") + if env_profile is not None: + if env_profile not in ("server", "workstation"): + raise ConfigError( + f"THOTH_PROFILE non valido: {env_profile!r} (atteso 'server' o 'workstation')." + ) + cfg = cfg.model_copy(update={"profile": env_profile}) + if cfg.database.transport == "rest" and cfg.rest is None: + raise ConfigError( + f"transport: rest richiede la sezione `rest` (base_url, api_key) in {path}." + ) + return cfg diff --git a/harness/nsp/workspace.py b/harness/nsp/workspace.py new file mode 100644 index 00000000..9206d24b --- /dev/null +++ b/harness/nsp/workspace.py @@ -0,0 +1,31 @@ +"""Workspace YAML loading — the single boundary for workspace configuration (spec D3). + +Reads workspaces/.yaml, expands ${VAR} from env, validates via the Config model +(ported from ChironeWp3). Future migration to a DB store would replace only this module. + +La struttura YAML rispecchia esattamente nsp/config.py: + database + rest (DWH), vector_rest + vector_write_rest (pgvector, doppia key top-level), + vector_db (loading diretto, server-only), embeddings, evidence, execution. +""" +from __future__ import annotations + +from pathlib import Path + +from nsp.config import Config, ConfigError, load_config + + +class WorkspaceError(Exception): + """Errore di caricamento del workspace (file mancante, env var non definita, yaml invalido).""" + + +def load_workspace(path: str | Path) -> Config: + """Carica e valida un workspace YAML. + + Espande ${VAR} dall'ambiente; se una variabile referenziata manca, raises WorkspaceError. + Delega a load_config (portato da ChironeWp3) per la validazione del modello Config. + """ + path = Path(path) + try: + return load_config(path) + except ConfigError as e: + raise WorkspaceError(str(e)) from e diff --git a/harness/tests/conftest.py b/harness/tests/conftest.py new file mode 100644 index 00000000..ef41e039 --- /dev/null +++ b/harness/tests/conftest.py @@ -0,0 +1,12 @@ +import os +import sys +from pathlib import Path + +# Permetti `pytest` lanciato da qualsiasi directory di trovare il package `nsp` +# (installato in modalità editable nella venv, ma utile anche senza attivazione). +_ROOT = Path(__file__).resolve().parent.parent +if str(_ROOT) not in sys.path: + sys.path.insert(0, str(_ROOT)) + +# Directory sessions/ risolta relativamente alla root harness (per i test che creano sessioni) +os.environ.setdefault("NSP_HARNESS_ROOT", str(_ROOT)) diff --git a/harness/tests/test_workspace.py b/harness/tests/test_workspace.py new file mode 100644 index 00000000..3285e5fb --- /dev/null +++ b/harness/tests/test_workspace.py @@ -0,0 +1,79 @@ +from pathlib import Path + +from nsp.workspace import load_workspace, WorkspaceError + + +def test_load_workspace_expands_env_vars(monkeypatch, tmp_path): + monkeypatch.setenv("THOTH_VEC_API_KEY", "secret-reader") + monkeypatch.setenv("THOTH_VEC_WRITE_API_KEY", "secret-writer") + monkeypatch.setenv("THOTH_VEC_REST_URL", "https://example/vector/v1/") + monkeypatch.setenv("THOTH_DWH_REST_URL", "https://example/dwh/") + monkeypatch.setenv("THOTH_DWH_API_KEY", "dwh-key") + monkeypatch.setenv("THOTH_DB_HOST", "h") + monkeypatch.setenv("THOTH_DB_NAME", "db") + monkeypatch.setenv("THOTH_DB_USER", "u") + monkeypatch.setenv("THOTH_DB_PASSWORD", "p") + monkeypatch.setenv("THOTH_VEC_HOST", "vh") + monkeypatch.setenv("THOTH_VEC_USER", "vu") + monkeypatch.setenv("THOTH_VEC_PASSWORD", "vp") + monkeypatch.setenv("THOTH_OLLAMA_URL", "http://ollama") + monkeypatch.setenv("THOTH_DOCS_ROOT", str(tmp_path / "docs")) + yaml = tmp_path / "w.yaml" + yaml.write_text( + "database:\n" + " host: ${THOTH_DB_HOST}\n" + " port: 5432\n" + " database: ${THOTH_DB_NAME}\n" + " schema: datawarehouse\n" + " user: ${THOTH_DB_USER}\n" + " password: ${THOTH_DB_PASSWORD}\n" + " transport: rest\n" + "rest:\n" + " base_url: ${THOTH_DWH_REST_URL}\n" + " api_key: ${THOTH_DWH_API_KEY}\n" + "vector_db:\n" + " host: ${THOTH_VEC_HOST}\n" + " port: 5438\n" + " database: postgres\n" + " schema: vectors\n" + " user: ${THOTH_VEC_USER}\n" + " password: ${THOTH_VEC_PASSWORD}\n" + "vector_rest:\n" + " base_url: ${THOTH_VEC_REST_URL}\n" + " api_key: ${THOTH_VEC_API_KEY}\n" + "vector_write_rest:\n" + " base_url: ${THOTH_VEC_REST_URL}\n" + " api_key: ${THOTH_VEC_WRITE_API_KEY}\n" + "embeddings:\n" + " base_url: ${THOTH_OLLAMA_URL}\n" + " model: nomic-embed-text-v2-moe\n" + " dim: 768\n" + "evidence:\n" + " source_root: ${THOTH_DOCS_ROOT}\n" + ) + ws = load_workspace(yaml) + # Dual vector key (top-level, per Config reale): reader and writer separate + assert ws.vector_rest.api_key == "secret-reader" + assert ws.vector_write_rest.api_key == "secret-writer" + # DWH key distinct from vector keys + assert ws.rest.api_key == "dwh-key" + # profile default = server + assert ws.profile == "server" + + +def test_load_workspace_missing_env_raises(monkeypatch, tmp_path): + monkeypatch.delenv("THOTH_VEC_API_KEY", raising=False) + yaml = tmp_path / "w.yaml" + yaml.write_text( + "database:\n" + " host: h\n port: 5432\n database: db\n schema: s\n" + " user: u\n password: p\n transport: direct\n" + "vector_rest:\n" + " base_url: 'https://v/'\n" + " api_key: '${THOTH_VEC_API_KEY}'\n" + ) + try: + load_workspace(yaml) + assert False, "should have raised" + except WorkspaceError as e: + assert "THOTH_VEC_API_KEY" in str(e) diff --git a/harness/workspaces/chirone.example.yaml b/harness/workspaces/chirone.example.yaml new file mode 100644 index 00000000..d363f079 --- /dev/null +++ b/harness/workspaces/chirone.example.yaml @@ -0,0 +1,90 @@ +# Workspace ThothII — Chirone (esempio). I segreti vivono SOLO in .env (${THOTH_*}). +# La struttura rispecchia esattamente nsp/config.py (portato da ChironeWp3): +# database + rest per il DWH; vector_rest/vector_write_rest per il pgvector (doppia key); +# vector_db per il loading diretto (server-only); embeddings + evidence + execution. + +database: + host: ${THOTH_DB_HOST} + port: ${THOTH_DB_PORT} # es. 5437 (Postgres diretto Supabase; 5432 = pooler) + database: ${THOTH_DB_NAME} # es. postgres (lo schema a stella vive in `datawarehouse`) + schema: datawarehouse + user: ${THOTH_DB_USER} + password: ${THOTH_DB_PASSWORD} + transport: rest # direct (Postgres) | rest (Supabase/PostgREST) + +# Accesso al DWH via REST (richiesto se database.transport = rest). +rest: + base_url: ${THOTH_DWH_REST_URL} # es. https://supabase-aritmolab.policlinicosandonato.it/dwh/ + api_key: ${THOTH_DWH_API_KEY} # header X-API-Key, ruolo dwh_reader (read-only) + ssl_ca: ${THOTH_SSL_CA} # path al certificato CA (per server con CA interna) + +paths: + artifacts: artifacts + indexes: indexes + sessions: sessions + +examples: + max_per_column: 10 + +lsh: + signature_size: 64 + n_gram: 3 + threshold: 0.5 + max_values_per_column: 1000 + +eligibility: + max_declared_len: 128 + max_avg_length: 40 + max_sampled_len: 200 + ignore_columns: [etl_last_update] + +evidence: + source_root: ${THOTH_DOCS_ROOT} # es. /Users/mp/Chirone/chirone/etl/docs + evidence_dir: evidence # cartella curata a mano: unica fonte delle evidence + +embeddings: + base_url: ${THOTH_OLLAMA_URL} # es. http://localhost:11434 + model: nomic-embed-text-v2-moe + dim: 768 + batch_size: 32 + +# LOADING del pgvector: connessione diretta, eseguita sul server (profilo server). +# Opzionale su postazione remota (lì la lettura passa da vector_rest). +vector_db: + host: ${THOTH_VEC_HOST} # Postgres locale del server + port: ${THOTH_VEC_PORT} # es. 5437 + database: postgres + schema: vectors + user: ${THOTH_VEC_USER} + password: ${THOTH_VEC_PASSWORD} + +# LETTURA (similarity search) del pgvector via REST remota: rpc search_similar. +vector_rest: + base_url: ${THOTH_VEC_REST_URL} # es. https://host/vector/v1/ + api_key: ${THOTH_VEC_API_KEY} # header X-API-Key, ruolo vector_reader (read-only) + ssl_ca: ${THOTH_SSL_CA} + +# SCRITTURA controllata del pgvector via REST remota: upsert/hash via RPC allowlist, +# niente delete/clear. Usa una API key SEPARATA dalla lettura (ruolo vector_writer). +# OPZIONALE: assente o key vuota = scrittura non abilitata (solo lettura). +# Abilita nsp memory save-one / vector index-schema da postazione remota. +vector_write_rest: + base_url: ${THOTH_VEC_REST_URL} + api_key: ${THOTH_VEC_WRITE_API_KEY} + ssl_ca: ${THOTH_SSL_CA} + +vector: + max_chunk_chars: 4000 + +search: + rrf_k: 60 + top_schema_tables: 12 + schema_chunk_pool: 150 + +execution: + allow: [cte_test, explain, preview, aggregate, export] + max_preview_rows: 10 + max_export_rows: 100000 + statement_timeout_ms: 30000 + warn_execution_ms: 5000 + max_aggregate_cells: 20