fix(embed): fast-fail + auto-restart Ollama on solved-search hang
Embeddings timeout was 120s, causing multi-minute hangs when Ollama was down during F4/F6/F7 solved-search. Now: connect_timeout=5s across all HTTP clients (REST + Ollama), read_timeout reduced to 30s for embeddings, and OllamaEmbeddings auto-restarts the server on ConnectionError before degrading gracefully. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -51,7 +51,7 @@ def test_run_query_payload_and_rows(monkeypatch):
|
||||
assert c["url"] == "https://h/dwh/rpc/run_query"
|
||||
assert c["json"] == {"query_text": "SELECT 1 AS x"}
|
||||
assert c["headers"]["X-API-Key"] == "dwh_k"
|
||||
assert c["timeout"] == 30
|
||||
assert c["timeout"] == (5, 30)
|
||||
|
||||
|
||||
def test_explain_query_maps_lines(monkeypatch):
|
||||
|
||||
@@ -52,6 +52,7 @@ class RestConfig(BaseModel):
|
||||
base_url: str
|
||||
api_key: str
|
||||
timeout: int = 30
|
||||
connect_timeout: int = 5
|
||||
ssl_ca: str | None = None # path al certificato CA (per server con CA interna)
|
||||
|
||||
|
||||
@@ -105,7 +106,8 @@ class EmbeddingsConfig(BaseModel):
|
||||
model: str = "nomic-embed-text-v2-moe"
|
||||
dim: int = 768
|
||||
batch_size: int = 32
|
||||
timeout: int = 120
|
||||
timeout: int = 30
|
||||
connect_timeout: int = 5
|
||||
bin: str = "ollama"
|
||||
start_cmd: list[str] | None = None
|
||||
|
||||
|
||||
@@ -28,7 +28,7 @@ class RestClient:
|
||||
url,
|
||||
json=args,
|
||||
headers={"X-API-Key": self.cfg.api_key},
|
||||
timeout=self.cfg.timeout,
|
||||
timeout=(self.cfg.connect_timeout, self.cfg.timeout),
|
||||
verify=verify,
|
||||
)
|
||||
except requests.RequestException as e:
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from tht.config import EmbeddingsConfig
|
||||
@@ -5,6 +9,9 @@ from tht.config import EmbeddingsConfig
|
||||
DOC_PREFIX = "search_document: "
|
||||
QUERY_PREFIX = "search_query: "
|
||||
|
||||
_RESTART_WAIT = 8 # secondi di attesa dopo aver avviato Ollama
|
||||
_RESTART_POLL = 1.0
|
||||
|
||||
|
||||
class EmbeddingsError(Exception):
|
||||
pass
|
||||
@@ -17,17 +24,64 @@ class OllamaEmbeddings:
|
||||
def __init__(self, cfg: EmbeddingsConfig):
|
||||
self.cfg = cfg
|
||||
|
||||
def _is_up(self) -> bool:
|
||||
try:
|
||||
r = requests.get(
|
||||
f"{self.cfg.base_url.rstrip('/')}/api/tags",
|
||||
timeout=(self.cfg.connect_timeout, 5),
|
||||
)
|
||||
return r.status_code == 200
|
||||
except requests.RequestException:
|
||||
return False
|
||||
|
||||
def _try_restart(self) -> bool:
|
||||
"""Tenta di avviare Ollama e attende che sia raggiungibile."""
|
||||
start_cmd = self.cfg.start_cmd if self.cfg.start_cmd is not None else [self.cfg.bin, "serve"]
|
||||
if not start_cmd:
|
||||
return False
|
||||
try:
|
||||
subprocess.Popen( # noqa: S603
|
||||
start_cmd, start_new_session=True,
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
)
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
print("[embeddings] Ollama non raggiungibile, avvio in corso…", file=sys.stderr)
|
||||
deadline = time.monotonic() + _RESTART_WAIT
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(_RESTART_POLL)
|
||||
if self._is_up():
|
||||
return True
|
||||
return False
|
||||
|
||||
def _post(self, url: str, batch: list[str]) -> requests.Response:
|
||||
resp = requests.post(
|
||||
url, json={"model": self.cfg.model, "input": batch},
|
||||
timeout=(self.cfg.connect_timeout, self.cfg.timeout),
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp
|
||||
|
||||
def _embed(self, texts: list[str]) -> list[list[float]]:
|
||||
url = f"{self.cfg.base_url.rstrip('/')}/api/embed"
|
||||
out: list[list[float]] = []
|
||||
for i in range(0, len(texts), self.cfg.batch_size):
|
||||
batch = texts[i : i + self.cfg.batch_size]
|
||||
try:
|
||||
resp = requests.post(
|
||||
url, json={"model": self.cfg.model, "input": batch},
|
||||
timeout=self.cfg.timeout,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
resp = self._post(url, batch)
|
||||
except requests.ConnectionError:
|
||||
if not self._try_restart():
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
f"(modello {self.cfg.model}), avvio automatico fallito"
|
||||
)
|
||||
try:
|
||||
resp = self._post(url, batch)
|
||||
except requests.RequestException as e:
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
f"(modello {self.cfg.model}): {e}"
|
||||
) from e
|
||||
except requests.RequestException as e:
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
|
||||
@@ -33,7 +33,7 @@ class VectorRestClient:
|
||||
url,
|
||||
json=args,
|
||||
headers={"X-API-Key": self.cfg.api_key},
|
||||
timeout=self.cfg.timeout,
|
||||
timeout=(self.cfg.connect_timeout, self.cfg.timeout),
|
||||
verify=verify,
|
||||
)
|
||||
except requests.RequestException as e:
|
||||
|
||||
Reference in New Issue
Block a user