feat(harness): EmbeddingsConfig bin/start_cmd + ensure_ollama preflight orchestration

This commit is contained in:
2026-06-29 19:15:28 +02:00
parent f66e8e41d9
commit 5f5cac05d5
3 changed files with 226 additions and 0 deletions
+116
View File
@@ -0,0 +1,116 @@
"""Tests for the ensure_ollama orchestration (Ollama mocked via injected ops)."""
from types import SimpleNamespace
from tht.config import EmbeddingsConfig
from tht.cli.ollama_cmd import ensure_ollama
def _cfg(**kw):
emb = EmbeddingsConfig(base_url="http://localhost:11434", **kw)
return SimpleNamespace(embeddings=emb)
def test_no_embeddings_config_is_hard_error():
r = ensure_ollama(SimpleNamespace(embeddings=None), timeout=5, no_start=False)
assert r["ok"] is False and r["stage"] == "config"
def test_server_up_model_present_warms_ok():
warmed = []
r = ensure_ollama(
_cfg(model="nomic-embed-text-v2-moe"), timeout=5, no_start=False,
probe=lambda url: True,
installed_models=lambda url: {"nomic-embed-text-v2-moe:latest"},
start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")),
warm=lambda cfg: warmed.append(True),
)
assert r == {"ok": True, "server": "up", "model": "warmed", "model_name": "nomic-embed-text-v2-moe"}
assert warmed == [True]
def test_server_down_then_started_after_poll():
started = []
probes = iter([False, True]) # down, then up after start
r = ensure_ollama(
_cfg(), timeout=5, no_start=False,
probe=lambda url: next(probes),
installed_models=lambda url: {"nomic-embed-text-v2-moe"},
start=lambda cmd: started.append(cmd),
warm=lambda cfg: None,
sleep=lambda s: None,
)
assert r["ok"] is True and r["server"] == "started"
assert started and started[0] == ["ollama", "serve"]
def test_server_unreachable_after_timeout_is_error():
clk = iter([0.0, 1.0, 2.0, 99.0]) # monotonic crosses the deadline
r = ensure_ollama(
_cfg(), timeout=5, no_start=False,
probe=lambda url: False, # never comes up
installed_models=lambda url: set(),
start=lambda cmd: None,
warm=lambda cfg: None,
sleep=lambda s: None,
clock=lambda: next(clk),
)
assert r["ok"] is False and r["stage"] == "server"
def test_no_start_and_down_is_error_without_starting():
r = ensure_ollama(
_cfg(), timeout=5, no_start=True,
probe=lambda url: False,
installed_models=lambda url: set(),
start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")),
warm=lambda cfg: None,
)
assert r["ok"] is False and r["stage"] == "server"
def test_empty_start_cmd_disables_autostart():
r = ensure_ollama(
_cfg(start_cmd=[]), timeout=5, no_start=False,
probe=lambda url: False,
installed_models=lambda url: set(),
start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")),
warm=lambda cfg: None,
)
assert r["ok"] is False and r["stage"] == "server"
def test_model_absent_is_error_with_pull_guidance():
r = ensure_ollama(
_cfg(model="missing-model"), timeout=5, no_start=False,
probe=lambda url: True,
installed_models=lambda url: {"nomic-embed-text-v2-moe"},
start=lambda cmd: None,
warm=lambda cfg: None,
)
assert r["ok"] is False and r["stage"] == "model"
assert "ollama pull missing-model" in r["error"]
def test_warm_failure_is_error():
r = ensure_ollama(
_cfg(), timeout=5, no_start=False,
probe=lambda url: True,
installed_models=lambda url: {"nomic-embed-text-v2-moe"},
start=lambda cmd: None,
warm=lambda cfg: (_ for _ in ()).throw(RuntimeError("boom")),
)
assert r["ok"] is False and r["stage"] == "warm"
def test_custom_start_cmd_used():
started = []
probes = iter([False, True])
ensure_ollama(
_cfg(bin="ollama", start_cmd=["docker", "start", "ollama"]), timeout=5, no_start=False,
probe=lambda url: next(probes),
installed_models=lambda url: {"nomic-embed-text-v2-moe"},
start=lambda cmd: started.append(cmd),
warm=lambda cfg: None,
sleep=lambda s: None,
)
assert started[0] == ["docker", "start", "ollama"]
+108
View File
@@ -0,0 +1,108 @@
"""`tht ollama` -- embeddings preflight (ensure Ollama up + model warm).
The system REQUIRES embeddings: any condition that makes them unavailable is a hard
error (the caller refuses the session). "Load" = warm the already-installed model.
"""
from __future__ import annotations
import subprocess
import time
# --- low-level ops (real implementations; injected as fakes in tests) ----------
def _probe(base_url: str, timeout: float = 2.0) -> bool:
import requests
try:
return requests.get(f"{base_url.rstrip('/')}/api/tags", timeout=timeout).status_code == 200
except requests.RequestException:
return False
def _installed_models(base_url: str, timeout: float = 5.0) -> set[str]:
import requests
resp = requests.get(f"{base_url.rstrip('/')}/api/tags", timeout=timeout)
resp.raise_for_status()
return {m.get("name", "") for m in resp.json().get("models", [])}
def _start(start_cmd: list[str]) -> None:
# Detached so the server outlives this short-lived CLI process.
subprocess.Popen( # noqa: S603
start_cmd, start_new_session=True,
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
)
def _warm(cfg) -> None:
from tht.vectorstore.embeddings import OllamaEmbeddings
OllamaEmbeddings(cfg.embeddings).embed_query("ping")
def _model_present(installed: set[str], model: str) -> bool:
"""Match the configured model against installed names, allowing the implicit ':latest'."""
if model in installed:
return True
base = model.split(":")[0]
return any(name == base or name.split(":")[0] == base for name in installed)
# --- orchestration (pure: returns a result dict, never raises for control flow) ----
def ensure_ollama(
cfg,
*,
timeout: int,
no_start: bool,
probe=_probe,
installed_models=_installed_models,
start=_start,
warm=_warm,
sleep=time.sleep,
clock=time.monotonic,
) -> dict:
if cfg.embeddings is None:
return {"ok": False, "stage": "config",
"error": "il sistema richiede embeddings ma il workspace non li configura"}
emb = cfg.embeddings
base_url = emb.base_url
start_cmd = emb.start_cmd if emb.start_cmd is not None else [emb.bin, "serve"]
server_state = "up"
if not probe(base_url):
if no_start or start_cmd == []:
return {"ok": False, "stage": "server",
"error": f"Ollama non raggiungibile su {base_url} e avvio disabilitato"}
start(start_cmd)
server_state = "started"
deadline = clock() + timeout
up = False
while clock() < deadline:
sleep(1.0)
if probe(base_url):
up = True
break
if not up:
return {"ok": False, "stage": "server",
"error": f"Ollama non raggiungibile su {base_url} entro {timeout}s"}
try:
installed = installed_models(base_url)
except Exception as e: # noqa: BLE001 - any read failure is a hard error
return {"ok": False, "stage": "server",
"error": f"impossibile leggere i modelli da {base_url}: {e}"}
if not _model_present(installed, emb.model):
return {"ok": False, "stage": "model",
"error": f"modello '{emb.model}' non installato in Ollama: "
f"esegui `ollama pull {emb.model}` o importalo"}
try:
warm(cfg)
except Exception as e: # noqa: BLE001 - warm failure is a hard error
return {"ok": False, "stage": "warm",
"error": f"warm del modello '{emb.model}' fallito: {e}"}
return {"ok": True, "server": server_state, "model": "warmed", "model_name": emb.model}
+2
View File
@@ -106,6 +106,8 @@ class EmbeddingsConfig(BaseModel):
dim: int = 768
batch_size: int = 32
timeout: int = 120
bin: str = "ollama"
start_cmd: list[str] | None = None
class VectorConfig(BaseModel):