feat(harness): config.py + workspace.py (D3 confine) + chirone.example.yaml
- config.py ported from ChironeWp3, PSD_PROFILE -> THOTH_PROFILE; dual vector key (vector_rest/vector_write_rest top-level) preserved verbatim - workspace.py: thin boundary wrapper (D3) around load_config - workspaces/chirone.example.yaml: aligned to the real Config shape - tests/test_workspace.py: 2 tests (env expand + dual key; missing env raises) - .env.example: added DWH + DOCS_ROOT vars All tests pass (2/2).
This commit is contained in:
@@ -4,9 +4,17 @@ THOTH_PROFILE=server
|
||||
# Workspace DB (relational, direct transport) — example: chirone
|
||||
THOTH_DB_HOST=
|
||||
THOTH_DB_PORT=5432
|
||||
THOTH_DB_NAME=
|
||||
THOTH_DB_USER=
|
||||
THOTH_DB_PASSWORD=
|
||||
|
||||
# DWH via REST (richiesto se database.transport = rest)
|
||||
THOTH_DWH_REST_URL=https://supabase-aritmolab.policlinicosandonato.it/dwh/
|
||||
THOTH_DWH_API_KEY=
|
||||
|
||||
# Evidence source root (cartella curata a mano con le evidence)
|
||||
THOTH_DOCS_ROOT=
|
||||
|
||||
# Vector REST — LETTURA (rpc search_similar) sul Supabase remoto
|
||||
THOTH_VEC_REST_URL=https://host/vector/v1/
|
||||
THOTH_VEC_API_KEY=
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal
|
||||
|
||||
import yaml
|
||||
from pydantic import BaseModel, Field, ValidationError
|
||||
|
||||
_ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
|
||||
|
||||
|
||||
class ConfigError(Exception):
|
||||
"""Errore di configurazione, con messaggio leggibile per l'utente."""
|
||||
|
||||
|
||||
def _expand_env(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
def repl(m: re.Match) -> str:
|
||||
var = m.group(1)
|
||||
if var not in os.environ:
|
||||
raise ConfigError(
|
||||
f"Variabile d'ambiente non definita: {var} "
|
||||
f"(definiscila nel file .env o nell'ambiente)"
|
||||
)
|
||||
return os.environ[var]
|
||||
|
||||
return _ENV_RE.sub(repl, value)
|
||||
if isinstance(value, dict):
|
||||
return {k: _expand_env(v) for k, v in value.items()}
|
||||
if isinstance(value, list):
|
||||
return [_expand_env(v) for v in value]
|
||||
return value
|
||||
|
||||
|
||||
class DatabaseConfig(BaseModel):
|
||||
host: str = "localhost"
|
||||
port: int = 5432
|
||||
database: str
|
||||
db_schema: str = Field(alias="schema")
|
||||
user: str
|
||||
password: str
|
||||
# transport: `direct` (Postgres via SQLAlchemy) o `rest` (Supabase/PostgREST).
|
||||
# In `rest` deve esistere la sezione `rest` (validato a livello di Config).
|
||||
transport: Literal["direct", "rest"] = "direct"
|
||||
|
||||
model_config = {"populate_by_name": True}
|
||||
|
||||
|
||||
class RestConfig(BaseModel):
|
||||
"""Accesso al DWH via Supabase/PostgREST. base_url es. https://host/dwh/ ."""
|
||||
|
||||
base_url: str
|
||||
api_key: str
|
||||
timeout: int = 30
|
||||
ssl_ca: str | None = None # path al certificato CA (per server con CA interna)
|
||||
|
||||
|
||||
class PathsConfig(BaseModel):
|
||||
artifacts: Path = Path("artifacts")
|
||||
indexes: Path = Path("indexes")
|
||||
sessions: Path = Path("sessions")
|
||||
|
||||
|
||||
class ExamplesConfig(BaseModel):
|
||||
max_per_column: int = 10
|
||||
|
||||
|
||||
class LshSkipConfig(BaseModel):
|
||||
# DEPRECATO: l'euristica di lunghezza (skip_column vendored) non è più usata. La
|
||||
# selezione delle colonne da indicizzare segue il principio di column eligibility
|
||||
# (sezione `eligibility`). Mantenuto solo per compatibilità con nsp.yaml esistenti.
|
||||
max_total_chars: int = 50000
|
||||
max_avg_length: int = 20
|
||||
|
||||
|
||||
class LshConfig(BaseModel):
|
||||
signature_size: int = 64
|
||||
n_gram: int = 3
|
||||
threshold: float = 0.5
|
||||
max_values_per_column: int = 1000
|
||||
skip: LshSkipConfig = LshSkipConfig()
|
||||
|
||||
|
||||
class EligibilityConfig(BaseModel):
|
||||
# Soglie del principio di column eligibility (testo ampio ignorato ovunque).
|
||||
max_declared_len: int = 128 # char/varchar dichiarati <= soglia: eligible senza campionare
|
||||
max_avg_length: int = 40 # fallback data-driven: lunghezza media valori campionati
|
||||
max_sampled_len: int = 200 # fallback data-driven: lunghezza massima valore campionato
|
||||
# Colonne di servizio sempre ignorate per nome (match case-insensitive), a prescindere
|
||||
# dal tipo: metadati ETL/audit non analitici (es. timestamp di ultimo aggiornamento).
|
||||
ignore_columns: list[str] = ["etl_last_update"]
|
||||
|
||||
|
||||
class EvidenceSourcesConfig(BaseModel):
|
||||
source_root: Path
|
||||
# cartella curata a mano nell'ETL (relativa a source_root): unica fonte delle
|
||||
# evidence. Niente piu' estrazione automatica dalle schede tabella: i documenti
|
||||
# qui dentro sono gia' evidence pronte (frontmatter + corpo), scelte e arricchite
|
||||
# dall'autore ETL e organizzate in sottocartelle per dominio.
|
||||
evidence_dir: str = "evidence"
|
||||
|
||||
|
||||
class EmbeddingsConfig(BaseModel):
|
||||
base_url: str
|
||||
model: str = "nomic-embed-text-v2-moe"
|
||||
dim: int = 768
|
||||
batch_size: int = 32
|
||||
timeout: int = 120
|
||||
|
||||
|
||||
class VectorConfig(BaseModel):
|
||||
max_chunk_chars: int = 4000
|
||||
|
||||
|
||||
class SearchConfig(BaseModel):
|
||||
rrf_k: int = 60
|
||||
top_schema_tables: int = 12 # default `--top` per `nsp search --kind schema` (n. tabelle)
|
||||
schema_chunk_pool: int = 150 # chunk tabella/colonna fusi prima dell'aggregazione a tabella
|
||||
|
||||
|
||||
class ExecutionConfig(BaseModel):
|
||||
allow: list[str] = ["cte_test", "explain", "preview", "aggregate", "export"]
|
||||
max_preview_rows: int = 10
|
||||
statement_timeout_ms: int = 30000
|
||||
warn_execution_ms: int = 5000
|
||||
warn_plan_rows: int = 1_000_000
|
||||
max_aggregate_cells: int = 20
|
||||
max_export_rows: int = 100000
|
||||
forbidden_functions: list[str] = [
|
||||
"setval", "nextval", "pg_advisory_lock", "pg_advisory_xact_lock",
|
||||
"dblink", "dblink_exec", "pg_terminate_backend", "pg_cancel_backend",
|
||||
"lo_import", "lo_export", "pg_reload_conf",
|
||||
]
|
||||
|
||||
|
||||
class Config(BaseModel):
|
||||
database: DatabaseConfig
|
||||
# Profilo dell'installazione, letto da THOTH_PROFILE (.env), non dallo yaml versionato.
|
||||
# server: ricostruisce i derivati (artefatti, LSH, vettori schema nel vectordb).
|
||||
# workstation: postazione locale che legge il vectordb via REST; gli upsert remoti
|
||||
# richiedono vector_write_rest, mentre init/clear/rebuild restano solo-server.
|
||||
profile: Literal["server", "workstation"] = "server"
|
||||
paths: PathsConfig = PathsConfig()
|
||||
examples: ExamplesConfig = ExamplesConfig()
|
||||
lsh: LshConfig = LshConfig()
|
||||
eligibility: EligibilityConfig = EligibilityConfig()
|
||||
evidence: EvidenceSourcesConfig | None = None
|
||||
embeddings: EmbeddingsConfig | None = None
|
||||
vector_db: DatabaseConfig | None = None
|
||||
vector: VectorConfig = VectorConfig()
|
||||
search: SearchConfig = SearchConfig()
|
||||
execution: ExecutionConfig = ExecutionConfig()
|
||||
rest: RestConfig | None = None
|
||||
# Endpoint REST dedicato per la similarity search del pgvector (rpc search_similar).
|
||||
# Se presente, `nsp search` legge via REST; altrimenti legge in diretto (dev/test).
|
||||
vector_rest: RestConfig | None = None
|
||||
# Endpoint REST dedicato alle scritture controllate del pgvector. E' opzionale e usa
|
||||
# una API key separata dalla lettura; espone solo upsert/hash via RPC allowlist.
|
||||
vector_write_rest: RestConfig | None = None
|
||||
|
||||
|
||||
def load_config(path: Path) -> Config:
|
||||
if not path.exists():
|
||||
raise ConfigError(f"File di configurazione non trovato: {path}")
|
||||
raw = yaml.safe_load(path.read_text())
|
||||
if not isinstance(raw, dict):
|
||||
raise ConfigError(f"Configurazione non valida (atteso un mapping YAML): {path}")
|
||||
try:
|
||||
cfg = Config.model_validate(_expand_env(raw))
|
||||
except ValidationError as e:
|
||||
raise ConfigError(f"Configurazione non valida in {path}:\n{e}") from e
|
||||
env_profile = os.environ.get("THOTH_PROFILE")
|
||||
if env_profile is not None:
|
||||
if env_profile not in ("server", "workstation"):
|
||||
raise ConfigError(
|
||||
f"THOTH_PROFILE non valido: {env_profile!r} (atteso 'server' o 'workstation')."
|
||||
)
|
||||
cfg = cfg.model_copy(update={"profile": env_profile})
|
||||
if cfg.database.transport == "rest" and cfg.rest is None:
|
||||
raise ConfigError(
|
||||
f"transport: rest richiede la sezione `rest` (base_url, api_key) in {path}."
|
||||
)
|
||||
return cfg
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Workspace YAML loading — the single boundary for workspace configuration (spec D3).
|
||||
|
||||
Reads workspaces/<name>.yaml, expands ${VAR} from env, validates via the Config model
|
||||
(ported from ChironeWp3). Future migration to a DB store would replace only this module.
|
||||
|
||||
La struttura YAML rispecchia esattamente nsp/config.py:
|
||||
database + rest (DWH), vector_rest + vector_write_rest (pgvector, doppia key top-level),
|
||||
vector_db (loading diretto, server-only), embeddings, evidence, execution.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from nsp.config import Config, ConfigError, load_config
|
||||
|
||||
|
||||
class WorkspaceError(Exception):
|
||||
"""Errore di caricamento del workspace (file mancante, env var non definita, yaml invalido)."""
|
||||
|
||||
|
||||
def load_workspace(path: str | Path) -> Config:
|
||||
"""Carica e valida un workspace YAML.
|
||||
|
||||
Espande ${VAR} dall'ambiente; se una variabile referenziata manca, raises WorkspaceError.
|
||||
Delega a load_config (portato da ChironeWp3) per la validazione del modello Config.
|
||||
"""
|
||||
path = Path(path)
|
||||
try:
|
||||
return load_config(path)
|
||||
except ConfigError as e:
|
||||
raise WorkspaceError(str(e)) from e
|
||||
@@ -0,0 +1,12 @@
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Permetti `pytest` lanciato da qualsiasi directory di trovare il package `nsp`
|
||||
# (installato in modalità editable nella venv, ma utile anche senza attivazione).
|
||||
_ROOT = Path(__file__).resolve().parent.parent
|
||||
if str(_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(_ROOT))
|
||||
|
||||
# Directory sessions/ risolta relativamente alla root harness (per i test che creano sessioni)
|
||||
os.environ.setdefault("NSP_HARNESS_ROOT", str(_ROOT))
|
||||
@@ -0,0 +1,79 @@
|
||||
from pathlib import Path
|
||||
|
||||
from nsp.workspace import load_workspace, WorkspaceError
|
||||
|
||||
|
||||
def test_load_workspace_expands_env_vars(monkeypatch, tmp_path):
|
||||
monkeypatch.setenv("THOTH_VEC_API_KEY", "secret-reader")
|
||||
monkeypatch.setenv("THOTH_VEC_WRITE_API_KEY", "secret-writer")
|
||||
monkeypatch.setenv("THOTH_VEC_REST_URL", "https://example/vector/v1/")
|
||||
monkeypatch.setenv("THOTH_DWH_REST_URL", "https://example/dwh/")
|
||||
monkeypatch.setenv("THOTH_DWH_API_KEY", "dwh-key")
|
||||
monkeypatch.setenv("THOTH_DB_HOST", "h")
|
||||
monkeypatch.setenv("THOTH_DB_NAME", "db")
|
||||
monkeypatch.setenv("THOTH_DB_USER", "u")
|
||||
monkeypatch.setenv("THOTH_DB_PASSWORD", "p")
|
||||
monkeypatch.setenv("THOTH_VEC_HOST", "vh")
|
||||
monkeypatch.setenv("THOTH_VEC_USER", "vu")
|
||||
monkeypatch.setenv("THOTH_VEC_PASSWORD", "vp")
|
||||
monkeypatch.setenv("THOTH_OLLAMA_URL", "http://ollama")
|
||||
monkeypatch.setenv("THOTH_DOCS_ROOT", str(tmp_path / "docs"))
|
||||
yaml = tmp_path / "w.yaml"
|
||||
yaml.write_text(
|
||||
"database:\n"
|
||||
" host: ${THOTH_DB_HOST}\n"
|
||||
" port: 5432\n"
|
||||
" database: ${THOTH_DB_NAME}\n"
|
||||
" schema: datawarehouse\n"
|
||||
" user: ${THOTH_DB_USER}\n"
|
||||
" password: ${THOTH_DB_PASSWORD}\n"
|
||||
" transport: rest\n"
|
||||
"rest:\n"
|
||||
" base_url: ${THOTH_DWH_REST_URL}\n"
|
||||
" api_key: ${THOTH_DWH_API_KEY}\n"
|
||||
"vector_db:\n"
|
||||
" host: ${THOTH_VEC_HOST}\n"
|
||||
" port: 5438\n"
|
||||
" database: postgres\n"
|
||||
" schema: vectors\n"
|
||||
" user: ${THOTH_VEC_USER}\n"
|
||||
" password: ${THOTH_VEC_PASSWORD}\n"
|
||||
"vector_rest:\n"
|
||||
" base_url: ${THOTH_VEC_REST_URL}\n"
|
||||
" api_key: ${THOTH_VEC_API_KEY}\n"
|
||||
"vector_write_rest:\n"
|
||||
" base_url: ${THOTH_VEC_REST_URL}\n"
|
||||
" api_key: ${THOTH_VEC_WRITE_API_KEY}\n"
|
||||
"embeddings:\n"
|
||||
" base_url: ${THOTH_OLLAMA_URL}\n"
|
||||
" model: nomic-embed-text-v2-moe\n"
|
||||
" dim: 768\n"
|
||||
"evidence:\n"
|
||||
" source_root: ${THOTH_DOCS_ROOT}\n"
|
||||
)
|
||||
ws = load_workspace(yaml)
|
||||
# Dual vector key (top-level, per Config reale): reader and writer separate
|
||||
assert ws.vector_rest.api_key == "secret-reader"
|
||||
assert ws.vector_write_rest.api_key == "secret-writer"
|
||||
# DWH key distinct from vector keys
|
||||
assert ws.rest.api_key == "dwh-key"
|
||||
# profile default = server
|
||||
assert ws.profile == "server"
|
||||
|
||||
|
||||
def test_load_workspace_missing_env_raises(monkeypatch, tmp_path):
|
||||
monkeypatch.delenv("THOTH_VEC_API_KEY", raising=False)
|
||||
yaml = tmp_path / "w.yaml"
|
||||
yaml.write_text(
|
||||
"database:\n"
|
||||
" host: h\n port: 5432\n database: db\n schema: s\n"
|
||||
" user: u\n password: p\n transport: direct\n"
|
||||
"vector_rest:\n"
|
||||
" base_url: 'https://v/'\n"
|
||||
" api_key: '${THOTH_VEC_API_KEY}'\n"
|
||||
)
|
||||
try:
|
||||
load_workspace(yaml)
|
||||
assert False, "should have raised"
|
||||
except WorkspaceError as e:
|
||||
assert "THOTH_VEC_API_KEY" in str(e)
|
||||
@@ -0,0 +1,90 @@
|
||||
# Workspace ThothII — Chirone (esempio). I segreti vivono SOLO in .env (${THOTH_*}).
|
||||
# La struttura rispecchia esattamente nsp/config.py (portato da ChironeWp3):
|
||||
# database + rest per il DWH; vector_rest/vector_write_rest per il pgvector (doppia key);
|
||||
# vector_db per il loading diretto (server-only); embeddings + evidence + execution.
|
||||
|
||||
database:
|
||||
host: ${THOTH_DB_HOST}
|
||||
port: ${THOTH_DB_PORT} # es. 5437 (Postgres diretto Supabase; 5432 = pooler)
|
||||
database: ${THOTH_DB_NAME} # es. postgres (lo schema a stella vive in `datawarehouse`)
|
||||
schema: datawarehouse
|
||||
user: ${THOTH_DB_USER}
|
||||
password: ${THOTH_DB_PASSWORD}
|
||||
transport: rest # direct (Postgres) | rest (Supabase/PostgREST)
|
||||
|
||||
# Accesso al DWH via REST (richiesto se database.transport = rest).
|
||||
rest:
|
||||
base_url: ${THOTH_DWH_REST_URL} # es. https://supabase-aritmolab.policlinicosandonato.it/dwh/
|
||||
api_key: ${THOTH_DWH_API_KEY} # header X-API-Key, ruolo dwh_reader (read-only)
|
||||
ssl_ca: ${THOTH_SSL_CA} # path al certificato CA (per server con CA interna)
|
||||
|
||||
paths:
|
||||
artifacts: artifacts
|
||||
indexes: indexes
|
||||
sessions: sessions
|
||||
|
||||
examples:
|
||||
max_per_column: 10
|
||||
|
||||
lsh:
|
||||
signature_size: 64
|
||||
n_gram: 3
|
||||
threshold: 0.5
|
||||
max_values_per_column: 1000
|
||||
|
||||
eligibility:
|
||||
max_declared_len: 128
|
||||
max_avg_length: 40
|
||||
max_sampled_len: 200
|
||||
ignore_columns: [etl_last_update]
|
||||
|
||||
evidence:
|
||||
source_root: ${THOTH_DOCS_ROOT} # es. /Users/mp/Chirone/chirone/etl/docs
|
||||
evidence_dir: evidence # cartella curata a mano: unica fonte delle evidence
|
||||
|
||||
embeddings:
|
||||
base_url: ${THOTH_OLLAMA_URL} # es. http://localhost:11434
|
||||
model: nomic-embed-text-v2-moe
|
||||
dim: 768
|
||||
batch_size: 32
|
||||
|
||||
# LOADING del pgvector: connessione diretta, eseguita sul server (profilo server).
|
||||
# Opzionale su postazione remota (lì la lettura passa da vector_rest).
|
||||
vector_db:
|
||||
host: ${THOTH_VEC_HOST} # Postgres locale del server
|
||||
port: ${THOTH_VEC_PORT} # es. 5437
|
||||
database: postgres
|
||||
schema: vectors
|
||||
user: ${THOTH_VEC_USER}
|
||||
password: ${THOTH_VEC_PASSWORD}
|
||||
|
||||
# LETTURA (similarity search) del pgvector via REST remota: rpc search_similar.
|
||||
vector_rest:
|
||||
base_url: ${THOTH_VEC_REST_URL} # es. https://host/vector/v1/
|
||||
api_key: ${THOTH_VEC_API_KEY} # header X-API-Key, ruolo vector_reader (read-only)
|
||||
ssl_ca: ${THOTH_SSL_CA}
|
||||
|
||||
# SCRITTURA controllata del pgvector via REST remota: upsert/hash via RPC allowlist,
|
||||
# niente delete/clear. Usa una API key SEPARATA dalla lettura (ruolo vector_writer).
|
||||
# OPZIONALE: assente o key vuota = scrittura non abilitata (solo lettura).
|
||||
# Abilita nsp memory save-one / vector index-schema da postazione remota.
|
||||
vector_write_rest:
|
||||
base_url: ${THOTH_VEC_REST_URL}
|
||||
api_key: ${THOTH_VEC_WRITE_API_KEY}
|
||||
ssl_ca: ${THOTH_SSL_CA}
|
||||
|
||||
vector:
|
||||
max_chunk_chars: 4000
|
||||
|
||||
search:
|
||||
rrf_k: 60
|
||||
top_schema_tables: 12
|
||||
schema_chunk_pool: 150
|
||||
|
||||
execution:
|
||||
allow: [cte_test, explain, preview, aggregate, export]
|
||||
max_preview_rows: 10
|
||||
max_export_rows: 100000
|
||||
statement_timeout_ms: 30000
|
||||
warn_execution_ms: 5000
|
||||
max_aggregate_cells: 20
|
||||
Reference in New Issue
Block a user