feat: use internal ollama embeddings
This commit is contained in:
@@ -1,104 +1,73 @@
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import math
|
||||
|
||||
import requests
|
||||
|
||||
from tht.config import EmbeddingsConfig
|
||||
|
||||
DOC_PREFIX = "search_document: "
|
||||
QUERY_PREFIX = "search_query: "
|
||||
|
||||
_RESTART_WAIT = 8 # secondi di attesa dopo aver avviato Ollama
|
||||
_RESTART_POLL = 1.0
|
||||
|
||||
|
||||
class EmbeddingsError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class OllamaEmbeddings:
|
||||
"""Client embeddings via Ollama. Applica i prefissi di task richiesti da nomic v2:
|
||||
ometterli degrada il retrieval in modo silenzioso."""
|
||||
class OllamaInternalEmbeddings:
|
||||
"""Client embeddings for the installation-owned internal Ollama endpoint."""
|
||||
|
||||
def __init__(self, cfg: EmbeddingsConfig):
|
||||
def __init__(self, cfg: EmbeddingsConfig, *, session: requests.Session | None = None):
|
||||
self.cfg = cfg
|
||||
self._session = session or requests.Session()
|
||||
self.base_url = self.cfg.base_url.rstrip("/")
|
||||
self.model = self.cfg.model
|
||||
self.dim = self.cfg.dim
|
||||
self.timeout = (self.cfg.connect_timeout, self.cfg.timeout)
|
||||
self.batch_size = self.cfg.batch_size
|
||||
|
||||
def _is_up(self) -> bool:
|
||||
def _post(self, texts: list[str]) -> list[list[float]]:
|
||||
try:
|
||||
r = requests.get(
|
||||
f"{self.cfg.base_url.rstrip('/')}/api/tags",
|
||||
timeout=(self.cfg.connect_timeout, 5),
|
||||
response = self._session.post(
|
||||
f"{self.base_url}/api/embed",
|
||||
json={"model": self.model, "input": texts},
|
||||
timeout=self.timeout,
|
||||
)
|
||||
return r.status_code == 200
|
||||
except requests.RequestException:
|
||||
return False
|
||||
|
||||
def _try_restart(self) -> bool:
|
||||
"""Tenta di avviare Ollama e attende che sia raggiungibile."""
|
||||
start_cmd = self.cfg.start_cmd if self.cfg.start_cmd is not None else [self.cfg.bin, "serve"]
|
||||
if not start_cmd:
|
||||
return False
|
||||
try:
|
||||
subprocess.Popen( # noqa: S603
|
||||
start_cmd, start_new_session=True,
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
response.raise_for_status()
|
||||
except requests.RequestException as exc:
|
||||
raise EmbeddingsError(
|
||||
f"internal Ollama embeddings request failed for model {self.model}"
|
||||
) from exc
|
||||
payload = response.json()
|
||||
embeddings = payload.get("embeddings")
|
||||
if not isinstance(embeddings, list):
|
||||
raise EmbeddingsError("internal Ollama response is missing embeddings")
|
||||
if len(embeddings) != len(texts):
|
||||
raise EmbeddingsError(
|
||||
f"unexpected embedding count: {len(embeddings)} != {len(texts)}"
|
||||
)
|
||||
except Exception: # noqa: BLE001
|
||||
return False
|
||||
print("[embeddings] Ollama non raggiungibile, avvio in corso…", file=sys.stderr)
|
||||
deadline = time.monotonic() + _RESTART_WAIT
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(_RESTART_POLL)
|
||||
if self._is_up():
|
||||
return True
|
||||
return False
|
||||
|
||||
def _post(self, url: str, batch: list[str]) -> requests.Response:
|
||||
resp = requests.post(
|
||||
url, json={"model": self.cfg.model, "input": batch},
|
||||
timeout=(self.cfg.connect_timeout, self.cfg.timeout),
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp
|
||||
|
||||
def _embed(self, texts: list[str]) -> list[list[float]]:
|
||||
url = f"{self.cfg.base_url.rstrip('/')}/api/embed"
|
||||
out: list[list[float]] = []
|
||||
for i in range(0, len(texts), self.cfg.batch_size):
|
||||
batch = texts[i : i + self.cfg.batch_size]
|
||||
try:
|
||||
resp = self._post(url, batch)
|
||||
except requests.ConnectionError:
|
||||
if not self._try_restart():
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
f"(modello {self.cfg.model}), avvio automatico fallito"
|
||||
)
|
||||
try:
|
||||
resp = self._post(url, batch)
|
||||
except requests.RequestException as e:
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
f"(modello {self.cfg.model}): {e}"
|
||||
) from e
|
||||
except requests.RequestException as e:
|
||||
validated: list[list[float]] = []
|
||||
for vector in embeddings:
|
||||
if not isinstance(vector, list):
|
||||
raise EmbeddingsError("internal Ollama returned a non-vector embedding")
|
||||
if len(vector) != self.dim:
|
||||
raise EmbeddingsError(
|
||||
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
||||
f"(modello {self.cfg.model}): {e}"
|
||||
) from e
|
||||
embeddings = resp.json().get("embeddings", [])
|
||||
for v in embeddings:
|
||||
if len(v) != self.cfg.dim:
|
||||
raise EmbeddingsError(
|
||||
f"dimensione embedding inattesa: {len(v)} != {self.cfg.dim} "
|
||||
f"(modello {self.cfg.model})"
|
||||
)
|
||||
out.extend(embeddings)
|
||||
f"unexpected embedding dimension: {len(vector)} != {self.dim}"
|
||||
)
|
||||
cleaned: list[float] = []
|
||||
for value in vector:
|
||||
if not isinstance(value, (int, float)) or not math.isfinite(value):
|
||||
raise EmbeddingsError("internal Ollama returned a non-finite embedding value")
|
||||
cleaned.append(float(value))
|
||||
validated.append(cleaned)
|
||||
return validated
|
||||
|
||||
def embed(self, texts: list[str]) -> list[list[float]]:
|
||||
out: list[list[float]] = []
|
||||
for i in range(0, len(texts), self.batch_size):
|
||||
out.extend(self._post(texts[i : i + self.batch_size]))
|
||||
return out
|
||||
|
||||
def embed_documents(self, texts: list[str]) -> list[list[float]]:
|
||||
return self._embed([DOC_PREFIX + t for t in texts])
|
||||
return self.embed(texts)
|
||||
|
||||
def embed_query(self, text: str) -> list[float]:
|
||||
return self._embed([QUERY_PREFIX + text])[0]
|
||||
return self.embed([text])[0]
|
||||
|
||||
|
||||
OllamaEmbeddings = OllamaInternalEmbeddings
|
||||
|
||||
Reference in New Issue
Block a user