Files
ThothII/harness/tht/vectorstore/embeddings.py
T
marcopanandClaude Opus 4.6 1e4bc11418 fix(embed): fast-fail + auto-restart Ollama on solved-search hang
Embeddings timeout was 120s, causing multi-minute hangs when Ollama was
down during F4/F6/F7 solved-search. Now: connect_timeout=5s across all
HTTP clients (REST + Ollama), read_timeout reduced to 30s for embeddings,
and OllamaEmbeddings auto-restarts the server on ConnectionError before
degrading gracefully.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-07-07 19:49:55 +02:00

105 lines
3.7 KiB
Python

import subprocess
import sys
import time
import requests
from tht.config import EmbeddingsConfig
DOC_PREFIX = "search_document: "
QUERY_PREFIX = "search_query: "
_RESTART_WAIT = 8 # secondi di attesa dopo aver avviato Ollama
_RESTART_POLL = 1.0
class EmbeddingsError(Exception):
pass
class OllamaEmbeddings:
"""Client embeddings via Ollama. Applica i prefissi di task richiesti da nomic v2:
ometterli degrada il retrieval in modo silenzioso."""
def __init__(self, cfg: EmbeddingsConfig):
self.cfg = cfg
def _is_up(self) -> bool:
try:
r = requests.get(
f"{self.cfg.base_url.rstrip('/')}/api/tags",
timeout=(self.cfg.connect_timeout, 5),
)
return r.status_code == 200
except requests.RequestException:
return False
def _try_restart(self) -> bool:
"""Tenta di avviare Ollama e attende che sia raggiungibile."""
start_cmd = self.cfg.start_cmd if self.cfg.start_cmd is not None else [self.cfg.bin, "serve"]
if not start_cmd:
return False
try:
subprocess.Popen( # noqa: S603
start_cmd, start_new_session=True,
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
)
except Exception: # noqa: BLE001
return False
print("[embeddings] Ollama non raggiungibile, avvio in corso…", file=sys.stderr)
deadline = time.monotonic() + _RESTART_WAIT
while time.monotonic() < deadline:
time.sleep(_RESTART_POLL)
if self._is_up():
return True
return False
def _post(self, url: str, batch: list[str]) -> requests.Response:
resp = requests.post(
url, json={"model": self.cfg.model, "input": batch},
timeout=(self.cfg.connect_timeout, self.cfg.timeout),
)
resp.raise_for_status()
return resp
def _embed(self, texts: list[str]) -> list[list[float]]:
url = f"{self.cfg.base_url.rstrip('/')}/api/embed"
out: list[list[float]] = []
for i in range(0, len(texts), self.cfg.batch_size):
batch = texts[i : i + self.cfg.batch_size]
try:
resp = self._post(url, batch)
except requests.ConnectionError:
if not self._try_restart():
raise EmbeddingsError(
f"Ollama non raggiungibile su {self.cfg.base_url} "
f"(modello {self.cfg.model}), avvio automatico fallito"
)
try:
resp = self._post(url, batch)
except requests.RequestException as e:
raise EmbeddingsError(
f"Ollama non raggiungibile su {self.cfg.base_url} "
f"(modello {self.cfg.model}): {e}"
) from e
except requests.RequestException as e:
raise EmbeddingsError(
f"Ollama non raggiungibile su {self.cfg.base_url} "
f"(modello {self.cfg.model}): {e}"
) from e
embeddings = resp.json().get("embeddings", [])
for v in embeddings:
if len(v) != self.cfg.dim:
raise EmbeddingsError(
f"dimensione embedding inattesa: {len(v)} != {self.cfg.dim} "
f"(modello {self.cfg.model})"
)
out.extend(embeddings)
return out
def embed_documents(self, texts: list[str]) -> list[list[float]]:
return self._embed([DOC_PREFIX + t for t in texts])
def embed_query(self, text: str) -> list[float]:
return self._embed([QUERY_PREFIX + text])[0]