diff --git a/harness/tests/test_ollama_ensure.py b/harness/tests/test_ollama_ensure.py new file mode 100644 index 00000000..76f9fe10 --- /dev/null +++ b/harness/tests/test_ollama_ensure.py @@ -0,0 +1,116 @@ +"""Tests for the ensure_ollama orchestration (Ollama mocked via injected ops).""" +from types import SimpleNamespace + +from tht.config import EmbeddingsConfig +from tht.cli.ollama_cmd import ensure_ollama + + +def _cfg(**kw): + emb = EmbeddingsConfig(base_url="http://localhost:11434", **kw) + return SimpleNamespace(embeddings=emb) + + +def test_no_embeddings_config_is_hard_error(): + r = ensure_ollama(SimpleNamespace(embeddings=None), timeout=5, no_start=False) + assert r["ok"] is False and r["stage"] == "config" + + +def test_server_up_model_present_warms_ok(): + warmed = [] + r = ensure_ollama( + _cfg(model="nomic-embed-text-v2-moe"), timeout=5, no_start=False, + probe=lambda url: True, + installed_models=lambda url: {"nomic-embed-text-v2-moe:latest"}, + start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")), + warm=lambda cfg: warmed.append(True), + ) + assert r == {"ok": True, "server": "up", "model": "warmed", "model_name": "nomic-embed-text-v2-moe"} + assert warmed == [True] + + +def test_server_down_then_started_after_poll(): + started = [] + probes = iter([False, True]) # down, then up after start + r = ensure_ollama( + _cfg(), timeout=5, no_start=False, + probe=lambda url: next(probes), + installed_models=lambda url: {"nomic-embed-text-v2-moe"}, + start=lambda cmd: started.append(cmd), + warm=lambda cfg: None, + sleep=lambda s: None, + ) + assert r["ok"] is True and r["server"] == "started" + assert started and started[0] == ["ollama", "serve"] + + +def test_server_unreachable_after_timeout_is_error(): + clk = iter([0.0, 1.0, 2.0, 99.0]) # monotonic crosses the deadline + r = ensure_ollama( + _cfg(), timeout=5, no_start=False, + probe=lambda url: False, # never comes up + installed_models=lambda url: set(), + start=lambda cmd: None, + warm=lambda cfg: None, + sleep=lambda s: None, + clock=lambda: next(clk), + ) + assert r["ok"] is False and r["stage"] == "server" + + +def test_no_start_and_down_is_error_without_starting(): + r = ensure_ollama( + _cfg(), timeout=5, no_start=True, + probe=lambda url: False, + installed_models=lambda url: set(), + start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")), + warm=lambda cfg: None, + ) + assert r["ok"] is False and r["stage"] == "server" + + +def test_empty_start_cmd_disables_autostart(): + r = ensure_ollama( + _cfg(start_cmd=[]), timeout=5, no_start=False, + probe=lambda url: False, + installed_models=lambda url: set(), + start=lambda cmd: (_ for _ in ()).throw(AssertionError("must not start")), + warm=lambda cfg: None, + ) + assert r["ok"] is False and r["stage"] == "server" + + +def test_model_absent_is_error_with_pull_guidance(): + r = ensure_ollama( + _cfg(model="missing-model"), timeout=5, no_start=False, + probe=lambda url: True, + installed_models=lambda url: {"nomic-embed-text-v2-moe"}, + start=lambda cmd: None, + warm=lambda cfg: None, + ) + assert r["ok"] is False and r["stage"] == "model" + assert "ollama pull missing-model" in r["error"] + + +def test_warm_failure_is_error(): + r = ensure_ollama( + _cfg(), timeout=5, no_start=False, + probe=lambda url: True, + installed_models=lambda url: {"nomic-embed-text-v2-moe"}, + start=lambda cmd: None, + warm=lambda cfg: (_ for _ in ()).throw(RuntimeError("boom")), + ) + assert r["ok"] is False and r["stage"] == "warm" + + +def test_custom_start_cmd_used(): + started = [] + probes = iter([False, True]) + ensure_ollama( + _cfg(bin="ollama", start_cmd=["docker", "start", "ollama"]), timeout=5, no_start=False, + probe=lambda url: next(probes), + installed_models=lambda url: {"nomic-embed-text-v2-moe"}, + start=lambda cmd: started.append(cmd), + warm=lambda cfg: None, + sleep=lambda s: None, + ) + assert started[0] == ["docker", "start", "ollama"] diff --git a/harness/tht/cli/ollama_cmd.py b/harness/tht/cli/ollama_cmd.py new file mode 100644 index 00000000..13e04b93 --- /dev/null +++ b/harness/tht/cli/ollama_cmd.py @@ -0,0 +1,108 @@ +"""`tht ollama` -- embeddings preflight (ensure Ollama up + model warm). + +The system REQUIRES embeddings: any condition that makes them unavailable is a hard +error (the caller refuses the session). "Load" = warm the already-installed model. +""" +from __future__ import annotations + +import subprocess +import time + + +# --- low-level ops (real implementations; injected as fakes in tests) ---------- + +def _probe(base_url: str, timeout: float = 2.0) -> bool: + import requests + + try: + return requests.get(f"{base_url.rstrip('/')}/api/tags", timeout=timeout).status_code == 200 + except requests.RequestException: + return False + + +def _installed_models(base_url: str, timeout: float = 5.0) -> set[str]: + import requests + + resp = requests.get(f"{base_url.rstrip('/')}/api/tags", timeout=timeout) + resp.raise_for_status() + return {m.get("name", "") for m in resp.json().get("models", [])} + + +def _start(start_cmd: list[str]) -> None: + # Detached so the server outlives this short-lived CLI process. + subprocess.Popen( # noqa: S603 + start_cmd, start_new_session=True, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + + +def _warm(cfg) -> None: + from tht.vectorstore.embeddings import OllamaEmbeddings + + OllamaEmbeddings(cfg.embeddings).embed_query("ping") + + +def _model_present(installed: set[str], model: str) -> bool: + """Match the configured model against installed names, allowing the implicit ':latest'.""" + if model in installed: + return True + base = model.split(":")[0] + return any(name == base or name.split(":")[0] == base for name in installed) + + +# --- orchestration (pure: returns a result dict, never raises for control flow) ---- + +def ensure_ollama( + cfg, + *, + timeout: int, + no_start: bool, + probe=_probe, + installed_models=_installed_models, + start=_start, + warm=_warm, + sleep=time.sleep, + clock=time.monotonic, +) -> dict: + if cfg.embeddings is None: + return {"ok": False, "stage": "config", + "error": "il sistema richiede embeddings ma il workspace non li configura"} + emb = cfg.embeddings + base_url = emb.base_url + start_cmd = emb.start_cmd if emb.start_cmd is not None else [emb.bin, "serve"] + + server_state = "up" + if not probe(base_url): + if no_start or start_cmd == []: + return {"ok": False, "stage": "server", + "error": f"Ollama non raggiungibile su {base_url} e avvio disabilitato"} + start(start_cmd) + server_state = "started" + deadline = clock() + timeout + up = False + while clock() < deadline: + sleep(1.0) + if probe(base_url): + up = True + break + if not up: + return {"ok": False, "stage": "server", + "error": f"Ollama non raggiungibile su {base_url} entro {timeout}s"} + + try: + installed = installed_models(base_url) + except Exception as e: # noqa: BLE001 - any read failure is a hard error + return {"ok": False, "stage": "server", + "error": f"impossibile leggere i modelli da {base_url}: {e}"} + if not _model_present(installed, emb.model): + return {"ok": False, "stage": "model", + "error": f"modello '{emb.model}' non installato in Ollama: " + f"esegui `ollama pull {emb.model}` o importalo"} + + try: + warm(cfg) + except Exception as e: # noqa: BLE001 - warm failure is a hard error + return {"ok": False, "stage": "warm", + "error": f"warm del modello '{emb.model}' fallito: {e}"} + + return {"ok": True, "server": server_state, "model": "warmed", "model_name": emb.model} diff --git a/harness/tht/config.py b/harness/tht/config.py index 2ff845ab..5f4c1bc1 100644 --- a/harness/tht/config.py +++ b/harness/tht/config.py @@ -106,6 +106,8 @@ class EmbeddingsConfig(BaseModel): dim: int = 768 batch_size: int = 32 timeout: int = 120 + bin: str = "ollama" + start_cmd: list[str] | None = None class VectorConfig(BaseModel):