Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.
Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
implementation', 'the DWH')
Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).
.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).
Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
51 lines
1.7 KiB
Python
51 lines
1.7 KiB
Python
import requests
|
|
|
|
from tht.config import EmbeddingsConfig
|
|
|
|
DOC_PREFIX = "search_document: "
|
|
QUERY_PREFIX = "search_query: "
|
|
|
|
|
|
class EmbeddingsError(Exception):
|
|
pass
|
|
|
|
|
|
class OllamaEmbeddings:
|
|
"""Client embeddings via Ollama. Applica i prefissi di task richiesti da nomic v2:
|
|
ometterli degrada il retrieval in modo silenzioso."""
|
|
|
|
def __init__(self, cfg: EmbeddingsConfig):
|
|
self.cfg = cfg
|
|
|
|
def _embed(self, texts: list[str]) -> list[list[float]]:
|
|
url = f"{self.cfg.base_url.rstrip('/')}/api/embed"
|
|
out: list[list[float]] = []
|
|
for i in range(0, len(texts), self.cfg.batch_size):
|
|
batch = texts[i : i + self.cfg.batch_size]
|
|
try:
|
|
resp = requests.post(
|
|
url, json={"model": self.cfg.model, "input": batch},
|
|
timeout=self.cfg.timeout,
|
|
)
|
|
resp.raise_for_status()
|
|
except requests.RequestException as e:
|
|
raise EmbeddingsError(
|
|
f"Ollama non raggiungibile su {self.cfg.base_url} "
|
|
f"(modello {self.cfg.model}): {e}"
|
|
) from e
|
|
embeddings = resp.json().get("embeddings", [])
|
|
for v in embeddings:
|
|
if len(v) != self.cfg.dim:
|
|
raise EmbeddingsError(
|
|
f"dimensione embedding inattesa: {len(v)} != {self.cfg.dim} "
|
|
f"(modello {self.cfg.model})"
|
|
)
|
|
out.extend(embeddings)
|
|
return out
|
|
|
|
def embed_documents(self, texts: list[str]) -> list[list[float]]:
|
|
return self._embed([DOC_PREFIX + t for t in texts])
|
|
|
|
def embed_query(self, text: str) -> list[float]:
|
|
return self._embed([QUERY_PREFIX + text])[0]
|