feat: index semantic records in qdrant

This commit is contained in:
2026-08-08 18:03:57 +02:00
parent f61648fb69
commit 5e39cfa347
13 changed files with 516 additions and 29 deletions
+45 -1
View File
@@ -1,8 +1,8 @@
import pytest
from tht.adapters.dwh import PostgresDwhAdapter, ThothRestDwhAdapter
from tht.adapters.vector import PgVectorStore, ThothHttpVectorStore
from tht.adapters.factory import build_dwh, build_vector_store
from tht.adapters.vector import PgVectorStore, QdrantVectorStore, ThothHttpVectorStore
from tht.config import Config, ConfigError
@@ -125,6 +125,50 @@ def test_factory_builds_writer_only_direct_vector_when_write_is_required():
assert store.capabilities.upsert is True
def test_factory_selects_qdrant_for_schema_v3_runtime():
config = Config.model_validate(
{
"dwh": {
"type": "postgres_direct",
"connection": {
"host": "db",
"database": "analytics",
"schema": "mart",
"user": "reader",
"password": "secret",
},
},
"database": {
"host": "db",
"database": "analytics",
"schema": "mart",
"user": "reader",
"password": "secret",
"transport": "direct",
},
"vectors": {
"type": "qdrant",
"base_url": "http://qdrant:6333",
"collection": "psd-clinical",
},
"embeddings": {
"provider": "ollama_internal",
"base_url": "http://embedding:11434",
"model": "qwen3-embedding:0.6b",
"dim": 1024,
},
}
)
config._workspace_id = "psd-clinical"
config._workspace_revision = "a" * 40
store = build_vector_store(config, require_write=True)
assert isinstance(store, QdrantVectorStore)
assert store.capabilities.search is True
assert store.capabilities.upsert is True
def test_factory_reuses_legacy_direct_connection_for_server_writes_only():
server = _config(vector_type="pgvector_direct", writer=False)
server.vectors.connection = server.vectors.reader
+28
View File
@@ -6,6 +6,7 @@ from tht.config import (
ConfigError,
PgvectorDirectConfig,
PostgresDwhConfig,
QdrantConfig,
ThothRestDwhConfig,
ThothVectorHttpConfig,
load_config,
@@ -237,6 +238,33 @@ resources:
assert cfg.embeddings.dim == 1024
def test_accepts_internal_qdrant_resource_contract(tmp_path):
workspace = tmp_path / "workspace.yaml"
workspace.write_text(
"""
dwh:
type: postgres_direct
connection: {database: analytics, schema: mart, user: reader, password: secret}
resources:
vector:
engine: qdrant
base_url: http://qdrant:6333
collection: psd-clinical
embeddings:
provider: ollama_internal
base_url: http://embedding:11434
model: qwen3-embedding:0.6b
dimensions: 1024
"""
)
cfg = load_config(workspace)
assert isinstance(cfg.vectors, QdrantConfig)
assert cfg.vectors.base_url == "http://qdrant:6333"
assert cfg.vectors.collection == "psd-clinical"
@pytest.mark.parametrize(
("snippet", "pattern"),
[
+42 -7
View File
@@ -7,19 +7,28 @@ pgvector as a one-row upsert. This test pins the pure core of that behavior:
- the writer.upsert_records is called once with a single row
- writer.sync is NEVER called (that is the full-resync path)
"""
from datetime import datetime
from datetime import UTC, datetime
from unittest.mock import MagicMock
from tht.adapters.vector.qdrant import point_id
from tht.memory import MemoryRecord, memory_vector_record_for_decision, save_one_memory
from tht.vectorstore.records import qdrant_payload
def _record(seq: int = 7, **kw) -> MemoryRecord:
base = dict(
id="mem-0007", ts=datetime(2025, 1, 1), session_id="s1", decision_seq=seq,
type="concept_clarified", subject="paziente attivo",
detail="flag_attivo = TRUE", rationale="r",
question_context="dammi i pazienti", tables=[], concepts=["paziente attivo"],
)
base = {
"id": "mem-0007",
"ts": datetime(2025, 1, 1, tzinfo=UTC),
"session_id": "s1",
"decision_seq": seq,
"type": "concept_clarified",
"subject": "paziente attivo",
"detail": "flag_attivo = TRUE",
"rationale": "r",
"question_context": "dammi i pazienti",
"tables": [],
"concepts": ["paziente attivo"],
}
base.update(kw)
return MemoryRecord(**base)
@@ -85,3 +94,29 @@ def test_save_one_uses_writer_key_for_upsert():
save_one_memory(records, decision_seq=7, store=writer, embedder=embedder)
# one upsert call, single row, table=memory
assert writer.upsert.call_count == 1
def test_save_one_preserves_semantic_point_identity_fields():
records = [_record(seq=7)]
writer = MagicMock()
writer.existing_hashes.return_value = {}
writer.upsert.return_value = 1
embedder = MagicMock()
embedder.embed_documents.return_value = [[0.0] * 4]
save_one_memory(records, decision_seq=7, store=writer, embedder=embedder)
row = writer.upsert.call_args.args[1][0]
payload = qdrant_payload(
row.record,
content_hash=row.content_hash,
workspace_id="psd-clinical",
workspace_revision="a" * 40,
)
assert point_id("psd-clinical", "memory", row.record.id) == point_id(
"psd-clinical", "memory", "memory:mem-0007"
)
assert payload["kind"] == "memory"
assert payload["workspace_id"] == "psd-clinical"
assert payload["workspace_revision"] == "a" * 40
@@ -166,6 +166,7 @@ def _store(fake: FakeQdrantHttp) -> QdrantVectorStore:
base_url="http://qdrant:6333",
collection="workspace-semantic",
workspace_id="demo",
workspace_revision="a" * 40,
expected_dimension=1024,
request=fake.request,
)
@@ -195,6 +196,7 @@ def test_upsert_creates_collection_and_keyword_indexes_idempotently():
"record_kind",
"vector_generation",
"workspace_id",
"workspace_revision",
}
@@ -239,6 +241,7 @@ def test_upsert_serializes_qdrant_point_payloads(record, semantic_kind):
assert point["id"] == point_id("demo", semantic_kind, record.record.id)
assert point["vector"] == record.embedding
assert point["payload"]["workspace_id"] == "demo"
assert point["payload"]["workspace_revision"] == "a" * 40
assert point["payload"]["kind"] == semantic_kind
assert point["payload"]["record_kind"] == record.record.kind
assert point["payload"]["record_key"] == record.record.id
+14 -5
View File
@@ -1,5 +1,5 @@
import json
from datetime import datetime
from datetime import UTC, datetime
from types import SimpleNamespace
from typer.testing import CliRunner
@@ -7,8 +7,8 @@ from typer.testing import CliRunner
from tht.cli import app
from tht.config import load_config
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
from tht.ports.vector import VectorReadUnavailable
from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
from tht.ports.vector import VectorReadUnavailable
from tht.vectorstore.embeddings import EmbeddingsError
@@ -22,7 +22,11 @@ class _FakeEmbedder:
class _FakeSearcher:
def __init__(self):
self.calls = []
def search(self, vec, top_n, kinds=None):
self.calls.append({"top_n": top_n, "kinds": kinds})
if kinds == ["solved_question"]:
return [SimpleNamespace(
kind="memory", ref="s-1", id="m1", title="q solved",
@@ -47,7 +51,7 @@ class _FakeSearcher:
def _workspace(tmp_path, with_session=None):
physical = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1),
database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={"fact_ablazione": TablePhysical(
comment="Ablazioni", columns={"cod_paz": ColumnPhysical(type="bigint")})},
)
@@ -55,7 +59,7 @@ def _workspace(tmp_path, with_session=None):
cfg.write_text(
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
"vector_db: {database: v, schema: public, user: u, password: p}\n"
"embeddings: {base_url: 'http://localhost:11434', model: nomic-embed-text, dim: 8}\n"
"embeddings: {base_url: 'http://localhost:11434', model: qwen3-embedding:0.6b, dim: 1024}\n"
f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, "
f"sessions: {tmp_path/'sessions'}}}\n"
)
@@ -91,10 +95,15 @@ def _patch(monkeypatch, embedder, searcher):
def test_pack_single_embed_and_sections(tmp_path, monkeypatch):
cfg = _workspace(tmp_path)
emb = _FakeEmbedder()
_patch(monkeypatch, emb, _FakeSearcher())
searcher = _FakeSearcher()
_patch(monkeypatch, emb, searcher)
res = CliRunner().invoke(app, ["search", "pack", "quanti pazienti", "-c", str(cfg)])
assert res.exit_code == 0, res.output
assert emb.calls == 1 # UN solo embedding per le tre ricerche
assert [call["kinds"] for call in searcher.calls] == [
["schema_table", "schema_column"],
["solved_question"],
]
assert "fact_ablazione" in res.output and "Ablazioni" in res.output
# Evidence is fail-closed until an ACTIVE corpus exists; legacy vector rows
# must not leak into a new search pack.
@@ -0,0 +1,222 @@
from __future__ import annotations
import hashlib
from dataclasses import dataclass
from datetime import UTC, datetime
from tht.adapters.vector.qdrant import point_id
from tht.cli.vector_cmd import sync_canonical_records
from tht.corpus.chunk import ChunkPolicy
from tht.corpus.models import CanonicalChunk
from tht.corpus.pipeline import CorpusPipeline
from tht.corpus.store import CorpusStore
from tht.memory import MemoryRecord, save_one_memory
from tht.mschema.models import (
Annotations,
ColumnPhysical,
PhysicalSchema,
TablePhysical,
)
from tht.ports.vector import VectorCapabilities, VectorHealth
from tht.vectorstore.records import qdrant_payload, schema_records
def _sha(content: str) -> str:
return f"sha256:{hashlib.sha256(content.encode('utf-8')).hexdigest()}"
class _Embedder:
def embed_documents(self, documents):
return [[float(index + 1)] * 4 for index, _ in enumerate(documents)]
@dataclass
class _Point:
point_id: str
payload: dict
embedding: list[float]
class FakeVectorStore:
def __init__(self, workspace_id="psd-clinical", workspace_revision=None):
self.workspace_id = workspace_id
self.workspace_revision = workspace_revision or "a" * 40
self.points: dict[str, _Point] = {}
self.search_calls: list[dict] = []
@property
def capabilities(self):
return VectorCapabilities(
search=True,
existing_hashes=True,
upsert=True,
metadata_filter=True,
delete_generation=True,
list_evidence_generations=True,
)
def health(self):
return VectorHealth(ok=True)
def search(self, collections, embedding, *, limit, kinds=None, metadata_filter=None):
self.search_calls.append(
{
"collections": collections,
"embedding": embedding,
"limit": limit,
"kinds": kinds,
"metadata_filter": metadata_filter,
}
)
return []
def existing_hashes(self, collection, kinds):
allowed = set(kinds)
return {
point.payload["record_key"]: point.payload["content_hash"]
for point in self.points.values()
if point.payload["record_kind"] in allowed
}
def upsert(self, collection, records):
for row in records:
semantic_kind = qdrant_payload(
row.record,
content_hash=row.content_hash,
workspace_id=self.workspace_id,
workspace_revision=self.workspace_revision,
)["kind"]
payload = qdrant_payload(
row.record,
content_hash=row.content_hash,
workspace_id=self.workspace_id,
workspace_revision=self.workspace_revision,
)
self.points[point_id(self.workspace_id, semantic_kind, row.record.id)] = _Point(
point_id=point_id(self.workspace_id, semantic_kind, row.record.id),
payload=payload,
embedding=row.embedding,
)
return len(records)
def delete_generation(self, collection, generation, workspace_id):
doomed = [
key
for key, point in self.points.items()
if point.payload.get("record_kind") == "evidence"
and point.payload.get("vector_generation") == generation
and point.payload.get("workspace_id") == workspace_id
]
for key in doomed:
self.points.pop(key)
return len(doomed)
def list_evidence_generations(self, collection, workspace_id):
return sorted(
{
point.payload["vector_generation"]
for point in self.points.values()
if point.payload.get("record_kind") == "evidence"
and point.payload.get("workspace_id") == workspace_id
}
)
def _schema_records():
return schema_records(
PhysicalSchema(
database="analytics",
schema="mart",
introspected_at=datetime.now(UTC),
tables={
"fact_patient": TablePhysical(
comment="Patients",
columns={"id": ColumnPhysical(type="bigint", comment="pk")},
)
},
),
Annotations(),
)
def _memory_records():
return [
MemoryRecord(
id="mem-0001",
ts=datetime(2026, 1, 1, tzinfo=UTC),
session_id="s1",
decision_seq=7,
type="concept_clarified",
subject="paziente attivo",
detail="flag_attivo = true",
rationale="r",
question_context="dammi i pazienti attivi",
tables=[],
concepts=["paziente attivo"],
)
]
def test_schema_and_memory_use_expected_semantic_kinds_and_shared_identity():
store = FakeVectorStore()
embedder = _Embedder()
schema_stats = sync_canonical_records(
"schema_records",
_schema_records(),
store=store,
embedder=embedder,
)
memory_count = save_one_memory(_memory_records(), 7, store=store, embedder=embedder)
assert schema_stats.added == 2
assert memory_count == 1
payloads = {point.payload["record_kind"]: point.payload for point in store.points.values()}
assert payloads["schema_table"]["kind"] == "schema"
assert payloads["schema_column"]["kind"] == "schema"
assert payloads["memory"]["kind"] == "memory"
assert {payload["workspace_id"] for payload in payloads.values()} == {"psd-clinical"}
assert {payload["workspace_revision"] for payload in payloads.values()} == {"a" * 40}
def test_corpus_vector_records_keep_exact_generation_and_retry_is_idempotent(tmp_path):
store = FakeVectorStore()
pipeline = CorpusPipeline(
store=CorpusStore(tmp_path / "corpus"),
sources=[],
embedder=None,
vector_store=store,
embedding_model="qwen3-embedding:0.6b",
embedding_dimensions=1024,
chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=4000),
pipeline_version="evidence-v1",
workspace_id="psd-clinical",
)
chunk = CanonicalChunk(
chunk_id="chunk:1",
document_id="doc:patient-guide",
ordinal=0,
content="Patient evidence",
content_hash=_sha("Patient evidence"),
source_uri="file:///tmp/patient-guide.md",
pipeline_version="evidence-v1",
)
row = pipeline._vector_record(
chunk,
[0.1, 0.2, 0.3, 0.4],
"gen:" + "1" * 32,
"psd-clinical",
)
assert row.record.kind == "evidence"
assert row.record.metadata["vector_generation"] == "gen:" + "1" * 32
store.upsert("evidence", [row])
store.upsert("evidence", [row])
assert len(store.points) == 1
point = next(iter(store.points.values()))
assert point.payload["kind"] == "evidence"
assert point.payload["vector_generation"] == "gen:" + "1" * 32
assert point.payload["workspace_id"] == "psd-clinical"
assert point.payload["workspace_revision"] == "a" * 40