From 1e4bc11418f6ab029440e5ff4bf42d8b0d503c74 Mon Sep 17 00:00:00 2001 From: mptyl Date: Tue, 7 Jul 2026 19:49:55 +0200 Subject: [PATCH] fix(embed): fast-fail + auto-restart Ollama on solved-search hang Embeddings timeout was 120s, causing multi-minute hangs when Ollama was down during F4/F6/F7 solved-search. Now: connect_timeout=5s across all HTTP clients (REST + Ollama), read_timeout reduced to 30s for embeddings, and OllamaEmbeddings auto-restarts the server on ConnectionError before degrading gracefully. Co-Authored-By: Claude Opus 4.6 --- harness/tests/test_rest_client.py | 2 +- harness/tht/config.py | 4 +- harness/tht/rest/client.py | 2 +- harness/tht/vectorstore/embeddings.py | 64 ++++++++++++++++++++++++-- harness/tht/vectorstore/rest_client.py | 2 +- 5 files changed, 65 insertions(+), 9 deletions(-) diff --git a/harness/tests/test_rest_client.py b/harness/tests/test_rest_client.py index 7a0974ad..4487a1d6 100644 --- a/harness/tests/test_rest_client.py +++ b/harness/tests/test_rest_client.py @@ -51,7 +51,7 @@ def test_run_query_payload_and_rows(monkeypatch): assert c["url"] == "https://h/dwh/rpc/run_query" assert c["json"] == {"query_text": "SELECT 1 AS x"} assert c["headers"]["X-API-Key"] == "dwh_k" - assert c["timeout"] == 30 + assert c["timeout"] == (5, 30) def test_explain_query_maps_lines(monkeypatch): diff --git a/harness/tht/config.py b/harness/tht/config.py index 5f4c1bc1..58f2f277 100644 --- a/harness/tht/config.py +++ b/harness/tht/config.py @@ -52,6 +52,7 @@ class RestConfig(BaseModel): base_url: str api_key: str timeout: int = 30 + connect_timeout: int = 5 ssl_ca: str | None = None # path al certificato CA (per server con CA interna) @@ -105,7 +106,8 @@ class EmbeddingsConfig(BaseModel): model: str = "nomic-embed-text-v2-moe" dim: int = 768 batch_size: int = 32 - timeout: int = 120 + timeout: int = 30 + connect_timeout: int = 5 bin: str = "ollama" start_cmd: list[str] | None = None diff --git a/harness/tht/rest/client.py b/harness/tht/rest/client.py index 085a4b4d..8d80333c 100644 --- a/harness/tht/rest/client.py +++ b/harness/tht/rest/client.py @@ -28,7 +28,7 @@ class RestClient: url, json=args, headers={"X-API-Key": self.cfg.api_key}, - timeout=self.cfg.timeout, + timeout=(self.cfg.connect_timeout, self.cfg.timeout), verify=verify, ) except requests.RequestException as e: diff --git a/harness/tht/vectorstore/embeddings.py b/harness/tht/vectorstore/embeddings.py index 147016ee..76dce5c8 100644 --- a/harness/tht/vectorstore/embeddings.py +++ b/harness/tht/vectorstore/embeddings.py @@ -1,3 +1,7 @@ +import subprocess +import sys +import time + import requests from tht.config import EmbeddingsConfig @@ -5,6 +9,9 @@ from tht.config import EmbeddingsConfig DOC_PREFIX = "search_document: " QUERY_PREFIX = "search_query: " +_RESTART_WAIT = 8 # secondi di attesa dopo aver avviato Ollama +_RESTART_POLL = 1.0 + class EmbeddingsError(Exception): pass @@ -17,17 +24,64 @@ class OllamaEmbeddings: def __init__(self, cfg: EmbeddingsConfig): self.cfg = cfg + def _is_up(self) -> bool: + try: + r = requests.get( + f"{self.cfg.base_url.rstrip('/')}/api/tags", + timeout=(self.cfg.connect_timeout, 5), + ) + return r.status_code == 200 + except requests.RequestException: + return False + + def _try_restart(self) -> bool: + """Tenta di avviare Ollama e attende che sia raggiungibile.""" + start_cmd = self.cfg.start_cmd if self.cfg.start_cmd is not None else [self.cfg.bin, "serve"] + if not start_cmd: + return False + try: + subprocess.Popen( # noqa: S603 + start_cmd, start_new_session=True, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + except Exception: # noqa: BLE001 + return False + print("[embeddings] Ollama non raggiungibile, avvio in corso…", file=sys.stderr) + deadline = time.monotonic() + _RESTART_WAIT + while time.monotonic() < deadline: + time.sleep(_RESTART_POLL) + if self._is_up(): + return True + return False + + def _post(self, url: str, batch: list[str]) -> requests.Response: + resp = requests.post( + url, json={"model": self.cfg.model, "input": batch}, + timeout=(self.cfg.connect_timeout, self.cfg.timeout), + ) + resp.raise_for_status() + return resp + def _embed(self, texts: list[str]) -> list[list[float]]: url = f"{self.cfg.base_url.rstrip('/')}/api/embed" out: list[list[float]] = [] for i in range(0, len(texts), self.cfg.batch_size): batch = texts[i : i + self.cfg.batch_size] try: - resp = requests.post( - url, json={"model": self.cfg.model, "input": batch}, - timeout=self.cfg.timeout, - ) - resp.raise_for_status() + resp = self._post(url, batch) + except requests.ConnectionError: + if not self._try_restart(): + raise EmbeddingsError( + f"Ollama non raggiungibile su {self.cfg.base_url} " + f"(modello {self.cfg.model}), avvio automatico fallito" + ) + try: + resp = self._post(url, batch) + except requests.RequestException as e: + raise EmbeddingsError( + f"Ollama non raggiungibile su {self.cfg.base_url} " + f"(modello {self.cfg.model}): {e}" + ) from e except requests.RequestException as e: raise EmbeddingsError( f"Ollama non raggiungibile su {self.cfg.base_url} " diff --git a/harness/tht/vectorstore/rest_client.py b/harness/tht/vectorstore/rest_client.py index e61c7f03..45c006b0 100644 --- a/harness/tht/vectorstore/rest_client.py +++ b/harness/tht/vectorstore/rest_client.py @@ -33,7 +33,7 @@ class VectorRestClient: url, json=args, headers={"X-API-Key": self.cfg.api_key}, - timeout=self.cfg.timeout, + timeout=(self.cfg.connect_timeout, self.cfg.timeout), verify=verify, ) except requests.RequestException as e: