Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.
Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
implementation', 'the DWH')
Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).
.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).
Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
91 lines
3.1 KiB
Python
91 lines
3.1 KiB
Python
"""Scrittura controllata del pgvector via REST.
|
|
|
|
Usata dalle postazioni remote solo quando e' configurata una seconda API key di scrittura.
|
|
Mantiene l'upsert incrementale del VectorStore diretto, ma non esegue delete/clear: le
|
|
operazioni distruttive restano solo-server via connessione Postgres diretta.
|
|
"""
|
|
|
|
from tht.vectorstore.records import VectorRecord
|
|
from tht.vectorstore.rest_client import VectorRestClient
|
|
from tht.vectorstore.store import SyncStats, content_hash
|
|
|
|
|
|
KIND_TO_TABLE = {
|
|
"schema_table": "schema_records",
|
|
"schema_column": "schema_records",
|
|
"evidence": "evidence",
|
|
"memory": "memory",
|
|
}
|
|
TABLE_TO_KINDS = {
|
|
"schema_records": {"schema_table", "schema_column"},
|
|
"evidence": {"evidence"},
|
|
"memory": {"memory"},
|
|
}
|
|
|
|
|
|
def pack_metadata(record: VectorRecord) -> dict:
|
|
"""Impacchetta nel metadata tutta la semantica letta poi da `search_similar`."""
|
|
return {
|
|
"kind": record.kind,
|
|
"ref": record.ref,
|
|
"record_key": record.id,
|
|
"title": record.title,
|
|
"content": record.content,
|
|
**record.metadata,
|
|
}
|
|
|
|
|
|
class RestVectorWriter:
|
|
"""Writer table-scoped via RPC REST allowlist.
|
|
|
|
Il metodo `sync` e' volutamente upsert-only: aggiorna/aggiunge record, conta gli stale,
|
|
ma non li elimina. Per cleanup completo usare i comandi server-side con `vector_db`.
|
|
"""
|
|
|
|
def __init__(self, client: VectorRestClient, table: str):
|
|
if table not in TABLE_TO_KINDS:
|
|
raise ValueError(f"Tabella vector non supportata per scrittura REST: {table}")
|
|
self.client = client
|
|
self.table = table
|
|
|
|
def existing_hashes(self, kinds: set[str]) -> dict[str, str]:
|
|
allowed = TABLE_TO_KINDS[self.table]
|
|
bad = kinds - allowed
|
|
if bad:
|
|
raise ValueError(
|
|
f"Kind non ammessi per vectors.{self.table}: {', '.join(sorted(bad))}"
|
|
)
|
|
return self.client.existing_hashes(self.table, sorted(kinds))
|
|
|
|
def sync(self, records: list[VectorRecord], embedder, kinds: set[str]) -> SyncStats:
|
|
stats = SyncStats()
|
|
existing = self.existing_hashes(kinds)
|
|
to_embed: list[VectorRecord] = []
|
|
for record in records:
|
|
h = content_hash(record.content)
|
|
if record.id not in existing:
|
|
to_embed.append(record)
|
|
stats.added += 1
|
|
elif existing[record.id] != h:
|
|
to_embed.append(record)
|
|
stats.updated += 1
|
|
else:
|
|
stats.unchanged += 1
|
|
|
|
stats.deleted = 0
|
|
vectors = embedder.embed_documents([r.content for r in to_embed]) if to_embed else []
|
|
rows = [
|
|
{
|
|
"record_key": record.id,
|
|
"kind": record.kind,
|
|
"content_hash": content_hash(record.content),
|
|
"metadata": pack_metadata(record),
|
|
"embedding": vector,
|
|
}
|
|
for record, vector in zip(to_embed, vectors)
|
|
]
|
|
if rows:
|
|
self.client.upsert_records(self.table, rows)
|
|
# Gli stale non vengono cancellati in REST writer: restano responsabilita' server-side.
|
|
return stats
|