feat: index semantic records in qdrant

This commit is contained in:
2026-08-08 18:03:57 +02:00
parent f61648fb69
commit 5e39cfa347
13 changed files with 516 additions and 29 deletions
+14 -5
View File
@@ -1,5 +1,5 @@
import json
from datetime import datetime
from datetime import UTC, datetime
from types import SimpleNamespace
from typer.testing import CliRunner
@@ -7,8 +7,8 @@ from typer.testing import CliRunner
from tht.cli import app
from tht.config import load_config
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
from tht.ports.vector import VectorReadUnavailable
from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
from tht.ports.vector import VectorReadUnavailable
from tht.vectorstore.embeddings import EmbeddingsError
@@ -22,7 +22,11 @@ class _FakeEmbedder:
class _FakeSearcher:
def __init__(self):
self.calls = []
def search(self, vec, top_n, kinds=None):
self.calls.append({"top_n": top_n, "kinds": kinds})
if kinds == ["solved_question"]:
return [SimpleNamespace(
kind="memory", ref="s-1", id="m1", title="q solved",
@@ -47,7 +51,7 @@ class _FakeSearcher:
def _workspace(tmp_path, with_session=None):
physical = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1),
database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={"fact_ablazione": TablePhysical(
comment="Ablazioni", columns={"cod_paz": ColumnPhysical(type="bigint")})},
)
@@ -55,7 +59,7 @@ def _workspace(tmp_path, with_session=None):
cfg.write_text(
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
"vector_db: {database: v, schema: public, user: u, password: p}\n"
"embeddings: {base_url: 'http://localhost:11434', model: nomic-embed-text, dim: 8}\n"
"embeddings: {base_url: 'http://localhost:11434', model: qwen3-embedding:0.6b, dim: 1024}\n"
f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, "
f"sessions: {tmp_path/'sessions'}}}\n"
)
@@ -91,10 +95,15 @@ def _patch(monkeypatch, embedder, searcher):
def test_pack_single_embed_and_sections(tmp_path, monkeypatch):
cfg = _workspace(tmp_path)
emb = _FakeEmbedder()
_patch(monkeypatch, emb, _FakeSearcher())
searcher = _FakeSearcher()
_patch(monkeypatch, emb, searcher)
res = CliRunner().invoke(app, ["search", "pack", "quanti pazienti", "-c", str(cfg)])
assert res.exit_code == 0, res.output
assert emb.calls == 1 # UN solo embedding per le tre ricerche
assert [call["kinds"] for call in searcher.calls] == [
["schema_table", "schema_column"],
["solved_question"],
]
assert "fact_ablazione" in res.output and "Ablazioni" in res.output
# Evidence is fail-closed until an ACTIVE corpus exists; legacy vector rows
# must not leak into a new search pack.