feat(harness): config.py + workspace.py (D3 confine) + chirone.example.yaml

- config.py ported from ChironeWp3, PSD_PROFILE -> THOTH_PROFILE; dual vector key
  (vector_rest/vector_write_rest top-level) preserved verbatim
- workspace.py: thin boundary wrapper (D3) around load_config
- workspaces/chirone.example.yaml: aligned to the real Config shape
- tests/test_workspace.py: 2 tests (env expand + dual key; missing env raises)
- .env.example: added DWH + DOCS_ROOT vars
All tests pass (2/2).
This commit is contained in:
2026-06-26 22:25:00 +02:00
parent 72a2c060d4
commit 6ba3336775
6 changed files with 403 additions and 0 deletions
+8
View File
@@ -4,9 +4,17 @@ THOTH_PROFILE=server
# Workspace DB (relational, direct transport) — example: chirone
THOTH_DB_HOST=
THOTH_DB_PORT=5432
THOTH_DB_NAME=
THOTH_DB_USER=
THOTH_DB_PASSWORD=
# DWH via REST (richiesto se database.transport = rest)
THOTH_DWH_REST_URL=https://supabase-aritmolab.policlinicosandonato.it/dwh/
THOTH_DWH_API_KEY=
# Evidence source root (cartella curata a mano con le evidence)
THOTH_DOCS_ROOT=
# Vector REST — LETTURA (rpc search_similar) sul Supabase remoto
THOTH_VEC_REST_URL=https://host/vector/v1/
THOTH_VEC_API_KEY=
+183
View File
@@ -0,0 +1,183 @@
import os
import re
from pathlib import Path
from typing import Any, Literal
import yaml
from pydantic import BaseModel, Field, ValidationError
_ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
class ConfigError(Exception):
"""Errore di configurazione, con messaggio leggibile per l'utente."""
def _expand_env(value: Any) -> Any:
if isinstance(value, str):
def repl(m: re.Match) -> str:
var = m.group(1)
if var not in os.environ:
raise ConfigError(
f"Variabile d'ambiente non definita: {var} "
f"(definiscila nel file .env o nell'ambiente)"
)
return os.environ[var]
return _ENV_RE.sub(repl, value)
if isinstance(value, dict):
return {k: _expand_env(v) for k, v in value.items()}
if isinstance(value, list):
return [_expand_env(v) for v in value]
return value
class DatabaseConfig(BaseModel):
host: str = "localhost"
port: int = 5432
database: str
db_schema: str = Field(alias="schema")
user: str
password: str
# transport: `direct` (Postgres via SQLAlchemy) o `rest` (Supabase/PostgREST).
# In `rest` deve esistere la sezione `rest` (validato a livello di Config).
transport: Literal["direct", "rest"] = "direct"
model_config = {"populate_by_name": True}
class RestConfig(BaseModel):
"""Accesso al DWH via Supabase/PostgREST. base_url es. https://host/dwh/ ."""
base_url: str
api_key: str
timeout: int = 30
ssl_ca: str | None = None # path al certificato CA (per server con CA interna)
class PathsConfig(BaseModel):
artifacts: Path = Path("artifacts")
indexes: Path = Path("indexes")
sessions: Path = Path("sessions")
class ExamplesConfig(BaseModel):
max_per_column: int = 10
class LshSkipConfig(BaseModel):
# DEPRECATO: l'euristica di lunghezza (skip_column vendored) non è più usata. La
# selezione delle colonne da indicizzare segue il principio di column eligibility
# (sezione `eligibility`). Mantenuto solo per compatibilità con nsp.yaml esistenti.
max_total_chars: int = 50000
max_avg_length: int = 20
class LshConfig(BaseModel):
signature_size: int = 64
n_gram: int = 3
threshold: float = 0.5
max_values_per_column: int = 1000
skip: LshSkipConfig = LshSkipConfig()
class EligibilityConfig(BaseModel):
# Soglie del principio di column eligibility (testo ampio ignorato ovunque).
max_declared_len: int = 128 # char/varchar dichiarati <= soglia: eligible senza campionare
max_avg_length: int = 40 # fallback data-driven: lunghezza media valori campionati
max_sampled_len: int = 200 # fallback data-driven: lunghezza massima valore campionato
# Colonne di servizio sempre ignorate per nome (match case-insensitive), a prescindere
# dal tipo: metadati ETL/audit non analitici (es. timestamp di ultimo aggiornamento).
ignore_columns: list[str] = ["etl_last_update"]
class EvidenceSourcesConfig(BaseModel):
source_root: Path
# cartella curata a mano nell'ETL (relativa a source_root): unica fonte delle
# evidence. Niente piu' estrazione automatica dalle schede tabella: i documenti
# qui dentro sono gia' evidence pronte (frontmatter + corpo), scelte e arricchite
# dall'autore ETL e organizzate in sottocartelle per dominio.
evidence_dir: str = "evidence"
class EmbeddingsConfig(BaseModel):
base_url: str
model: str = "nomic-embed-text-v2-moe"
dim: int = 768
batch_size: int = 32
timeout: int = 120
class VectorConfig(BaseModel):
max_chunk_chars: int = 4000
class SearchConfig(BaseModel):
rrf_k: int = 60
top_schema_tables: int = 12 # default `--top` per `nsp search --kind schema` (n. tabelle)
schema_chunk_pool: int = 150 # chunk tabella/colonna fusi prima dell'aggregazione a tabella
class ExecutionConfig(BaseModel):
allow: list[str] = ["cte_test", "explain", "preview", "aggregate", "export"]
max_preview_rows: int = 10
statement_timeout_ms: int = 30000
warn_execution_ms: int = 5000
warn_plan_rows: int = 1_000_000
max_aggregate_cells: int = 20
max_export_rows: int = 100000
forbidden_functions: list[str] = [
"setval", "nextval", "pg_advisory_lock", "pg_advisory_xact_lock",
"dblink", "dblink_exec", "pg_terminate_backend", "pg_cancel_backend",
"lo_import", "lo_export", "pg_reload_conf",
]
class Config(BaseModel):
database: DatabaseConfig
# Profilo dell'installazione, letto da THOTH_PROFILE (.env), non dallo yaml versionato.
# server: ricostruisce i derivati (artefatti, LSH, vettori schema nel vectordb).
# workstation: postazione locale che legge il vectordb via REST; gli upsert remoti
# richiedono vector_write_rest, mentre init/clear/rebuild restano solo-server.
profile: Literal["server", "workstation"] = "server"
paths: PathsConfig = PathsConfig()
examples: ExamplesConfig = ExamplesConfig()
lsh: LshConfig = LshConfig()
eligibility: EligibilityConfig = EligibilityConfig()
evidence: EvidenceSourcesConfig | None = None
embeddings: EmbeddingsConfig | None = None
vector_db: DatabaseConfig | None = None
vector: VectorConfig = VectorConfig()
search: SearchConfig = SearchConfig()
execution: ExecutionConfig = ExecutionConfig()
rest: RestConfig | None = None
# Endpoint REST dedicato per la similarity search del pgvector (rpc search_similar).
# Se presente, `nsp search` legge via REST; altrimenti legge in diretto (dev/test).
vector_rest: RestConfig | None = None
# Endpoint REST dedicato alle scritture controllate del pgvector. E' opzionale e usa
# una API key separata dalla lettura; espone solo upsert/hash via RPC allowlist.
vector_write_rest: RestConfig | None = None
def load_config(path: Path) -> Config:
if not path.exists():
raise ConfigError(f"File di configurazione non trovato: {path}")
raw = yaml.safe_load(path.read_text())
if not isinstance(raw, dict):
raise ConfigError(f"Configurazione non valida (atteso un mapping YAML): {path}")
try:
cfg = Config.model_validate(_expand_env(raw))
except ValidationError as e:
raise ConfigError(f"Configurazione non valida in {path}:\n{e}") from e
env_profile = os.environ.get("THOTH_PROFILE")
if env_profile is not None:
if env_profile not in ("server", "workstation"):
raise ConfigError(
f"THOTH_PROFILE non valido: {env_profile!r} (atteso 'server' o 'workstation')."
)
cfg = cfg.model_copy(update={"profile": env_profile})
if cfg.database.transport == "rest" and cfg.rest is None:
raise ConfigError(
f"transport: rest richiede la sezione `rest` (base_url, api_key) in {path}."
)
return cfg
+31
View File
@@ -0,0 +1,31 @@
"""Workspace YAML loading — the single boundary for workspace configuration (spec D3).
Reads workspaces/<name>.yaml, expands ${VAR} from env, validates via the Config model
(ported from ChironeWp3). Future migration to a DB store would replace only this module.
La struttura YAML rispecchia esattamente nsp/config.py:
database + rest (DWH), vector_rest + vector_write_rest (pgvector, doppia key top-level),
vector_db (loading diretto, server-only), embeddings, evidence, execution.
"""
from __future__ import annotations
from pathlib import Path
from nsp.config import Config, ConfigError, load_config
class WorkspaceError(Exception):
"""Errore di caricamento del workspace (file mancante, env var non definita, yaml invalido)."""
def load_workspace(path: str | Path) -> Config:
"""Carica e valida un workspace YAML.
Espande ${VAR} dall'ambiente; se una variabile referenziata manca, raises WorkspaceError.
Delega a load_config (portato da ChironeWp3) per la validazione del modello Config.
"""
path = Path(path)
try:
return load_config(path)
except ConfigError as e:
raise WorkspaceError(str(e)) from e
+12
View File
@@ -0,0 +1,12 @@
import os
import sys
from pathlib import Path
# Permetti `pytest` lanciato da qualsiasi directory di trovare il package `nsp`
# (installato in modalità editable nella venv, ma utile anche senza attivazione).
_ROOT = Path(__file__).resolve().parent.parent
if str(_ROOT) not in sys.path:
sys.path.insert(0, str(_ROOT))
# Directory sessions/ risolta relativamente alla root harness (per i test che creano sessioni)
os.environ.setdefault("NSP_HARNESS_ROOT", str(_ROOT))
+79
View File
@@ -0,0 +1,79 @@
from pathlib import Path
from nsp.workspace import load_workspace, WorkspaceError
def test_load_workspace_expands_env_vars(monkeypatch, tmp_path):
monkeypatch.setenv("THOTH_VEC_API_KEY", "secret-reader")
monkeypatch.setenv("THOTH_VEC_WRITE_API_KEY", "secret-writer")
monkeypatch.setenv("THOTH_VEC_REST_URL", "https://example/vector/v1/")
monkeypatch.setenv("THOTH_DWH_REST_URL", "https://example/dwh/")
monkeypatch.setenv("THOTH_DWH_API_KEY", "dwh-key")
monkeypatch.setenv("THOTH_DB_HOST", "h")
monkeypatch.setenv("THOTH_DB_NAME", "db")
monkeypatch.setenv("THOTH_DB_USER", "u")
monkeypatch.setenv("THOTH_DB_PASSWORD", "p")
monkeypatch.setenv("THOTH_VEC_HOST", "vh")
monkeypatch.setenv("THOTH_VEC_USER", "vu")
monkeypatch.setenv("THOTH_VEC_PASSWORD", "vp")
monkeypatch.setenv("THOTH_OLLAMA_URL", "http://ollama")
monkeypatch.setenv("THOTH_DOCS_ROOT", str(tmp_path / "docs"))
yaml = tmp_path / "w.yaml"
yaml.write_text(
"database:\n"
" host: ${THOTH_DB_HOST}\n"
" port: 5432\n"
" database: ${THOTH_DB_NAME}\n"
" schema: datawarehouse\n"
" user: ${THOTH_DB_USER}\n"
" password: ${THOTH_DB_PASSWORD}\n"
" transport: rest\n"
"rest:\n"
" base_url: ${THOTH_DWH_REST_URL}\n"
" api_key: ${THOTH_DWH_API_KEY}\n"
"vector_db:\n"
" host: ${THOTH_VEC_HOST}\n"
" port: 5438\n"
" database: postgres\n"
" schema: vectors\n"
" user: ${THOTH_VEC_USER}\n"
" password: ${THOTH_VEC_PASSWORD}\n"
"vector_rest:\n"
" base_url: ${THOTH_VEC_REST_URL}\n"
" api_key: ${THOTH_VEC_API_KEY}\n"
"vector_write_rest:\n"
" base_url: ${THOTH_VEC_REST_URL}\n"
" api_key: ${THOTH_VEC_WRITE_API_KEY}\n"
"embeddings:\n"
" base_url: ${THOTH_OLLAMA_URL}\n"
" model: nomic-embed-text-v2-moe\n"
" dim: 768\n"
"evidence:\n"
" source_root: ${THOTH_DOCS_ROOT}\n"
)
ws = load_workspace(yaml)
# Dual vector key (top-level, per Config reale): reader and writer separate
assert ws.vector_rest.api_key == "secret-reader"
assert ws.vector_write_rest.api_key == "secret-writer"
# DWH key distinct from vector keys
assert ws.rest.api_key == "dwh-key"
# profile default = server
assert ws.profile == "server"
def test_load_workspace_missing_env_raises(monkeypatch, tmp_path):
monkeypatch.delenv("THOTH_VEC_API_KEY", raising=False)
yaml = tmp_path / "w.yaml"
yaml.write_text(
"database:\n"
" host: h\n port: 5432\n database: db\n schema: s\n"
" user: u\n password: p\n transport: direct\n"
"vector_rest:\n"
" base_url: 'https://v/'\n"
" api_key: '${THOTH_VEC_API_KEY}'\n"
)
try:
load_workspace(yaml)
assert False, "should have raised"
except WorkspaceError as e:
assert "THOTH_VEC_API_KEY" in str(e)
+90
View File
@@ -0,0 +1,90 @@
# Workspace ThothII — Chirone (esempio). I segreti vivono SOLO in .env (${THOTH_*}).
# La struttura rispecchia esattamente nsp/config.py (portato da ChironeWp3):
# database + rest per il DWH; vector_rest/vector_write_rest per il pgvector (doppia key);
# vector_db per il loading diretto (server-only); embeddings + evidence + execution.
database:
host: ${THOTH_DB_HOST}
port: ${THOTH_DB_PORT} # es. 5437 (Postgres diretto Supabase; 5432 = pooler)
database: ${THOTH_DB_NAME} # es. postgres (lo schema a stella vive in `datawarehouse`)
schema: datawarehouse
user: ${THOTH_DB_USER}
password: ${THOTH_DB_PASSWORD}
transport: rest # direct (Postgres) | rest (Supabase/PostgREST)
# Accesso al DWH via REST (richiesto se database.transport = rest).
rest:
base_url: ${THOTH_DWH_REST_URL} # es. https://supabase-aritmolab.policlinicosandonato.it/dwh/
api_key: ${THOTH_DWH_API_KEY} # header X-API-Key, ruolo dwh_reader (read-only)
ssl_ca: ${THOTH_SSL_CA} # path al certificato CA (per server con CA interna)
paths:
artifacts: artifacts
indexes: indexes
sessions: sessions
examples:
max_per_column: 10
lsh:
signature_size: 64
n_gram: 3
threshold: 0.5
max_values_per_column: 1000
eligibility:
max_declared_len: 128
max_avg_length: 40
max_sampled_len: 200
ignore_columns: [etl_last_update]
evidence:
source_root: ${THOTH_DOCS_ROOT} # es. /Users/mp/Chirone/chirone/etl/docs
evidence_dir: evidence # cartella curata a mano: unica fonte delle evidence
embeddings:
base_url: ${THOTH_OLLAMA_URL} # es. http://localhost:11434
model: nomic-embed-text-v2-moe
dim: 768
batch_size: 32
# LOADING del pgvector: connessione diretta, eseguita sul server (profilo server).
# Opzionale su postazione remota (lì la lettura passa da vector_rest).
vector_db:
host: ${THOTH_VEC_HOST} # Postgres locale del server
port: ${THOTH_VEC_PORT} # es. 5437
database: postgres
schema: vectors
user: ${THOTH_VEC_USER}
password: ${THOTH_VEC_PASSWORD}
# LETTURA (similarity search) del pgvector via REST remota: rpc search_similar.
vector_rest:
base_url: ${THOTH_VEC_REST_URL} # es. https://host/vector/v1/
api_key: ${THOTH_VEC_API_KEY} # header X-API-Key, ruolo vector_reader (read-only)
ssl_ca: ${THOTH_SSL_CA}
# SCRITTURA controllata del pgvector via REST remota: upsert/hash via RPC allowlist,
# niente delete/clear. Usa una API key SEPARATA dalla lettura (ruolo vector_writer).
# OPZIONALE: assente o key vuota = scrittura non abilitata (solo lettura).
# Abilita nsp memory save-one / vector index-schema da postazione remota.
vector_write_rest:
base_url: ${THOTH_VEC_REST_URL}
api_key: ${THOTH_VEC_WRITE_API_KEY}
ssl_ca: ${THOTH_SSL_CA}
vector:
max_chunk_chars: 4000
search:
rrf_k: 60
top_schema_tables: 12
schema_chunk_pool: 150
execution:
allow: [cte_test, explain, preview, aggregate, export]
max_preview_rows: 10
max_export_rows: 100000
statement_timeout_ms: 30000
warn_execution_ms: 5000
max_aggregate_cells: 20