feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s

This commit is contained in:
Codex
2026-09-06 17:49:35 +02:00
parent 8707ae1d46
commit cffa60772e
141 changed files with 5898 additions and 3015 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ def build_vector_store(cfg: Config, *, require_write: bool = False) -> VectorSto
case "qdrant":
return QdrantVectorStore(
base_url=resource.base_url,
collection=resource.collection,
collections=resource.collections,
workspace_id=cfg._workspace_id,
workspace_revision=cfg._workspace_revision,
expected_dimension=cfg.embeddings.dim if cfg.embeddings is not None else None,
+1 -1
View File
@@ -5,7 +5,7 @@ from __future__ import annotations
from tht.ports.vector import VectorStoreError
COLLECTION_KINDS = {
"schema_records": {"schema_table", "schema_column"},
"schema_records": {"schema_table", "schema_column", "schema_relationship"},
"evidence": {"evidence"},
"memory": {"memory", "solved_question"},
}
+152 -29
View File
@@ -1,4 +1,4 @@
"""Qdrant-backed vector store for one workspace-owned semantic collection."""
"""Qdrant-backed vector store with separate reference and memory lifecycles."""
import re
from collections.abc import Callable
@@ -25,6 +25,9 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
_BM25_LANGUAGES = frozenset({"english", "italian"})
_REFERENCE_KINDS = frozenset({
"schema_table", "schema_column", "schema_relationship", "evidence",
})
_KEYWORD_INDEXES = (
"content_hash",
@@ -58,7 +61,8 @@ class QdrantVectorStore:
self,
*,
base_url: str,
collection: str,
collections: dict[str, str] | None = None,
collection: str | None = None,
workspace_id: str,
workspace_revision: str | None = None,
expected_dimension: int | None = None,
@@ -68,7 +72,17 @@ class QdrantVectorStore:
read_timeout: float = 10.0,
):
self._base_url = base_url.rstrip("/")
self._collection = collection
legacy_constructor = collections is None and collection is not None
if legacy_constructor:
collections = {"reference": collection, "memory": collection}
if collections is None or set(collections) != {"reference", "memory"}:
raise VectorStoreError("Qdrant collections must define reference and memory")
if not legacy_constructor and collections["reference"] == collections["memory"]:
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
self._collections = dict(collections)
self._split_collections = collections["reference"] != collections["memory"]
self._legacy_collection = workspace_id
self._legacy_checked = False
self._workspace_id = workspace_id
self._workspace_revision = None
self._workspace_revision = workspace_revision
@@ -90,7 +104,10 @@ class QdrantVectorStore:
def health(self) -> VectorHealth:
try:
info = self._ensure_collection(strict=False)
infos = {
name: self._ensure_collection(collection, strict=False)
for name, collection in self._collections.items()
}
except VectorStoreError as exc:
return VectorHealth(
ok=False,
@@ -105,8 +122,10 @@ class QdrantVectorStore:
bm25_compatible=None,
)
dimension = info["config"]["params"]["vectors"]["size"]
dimensions = (dimension,)
dimensions = tuple(sorted({
info["config"]["params"]["vectors"]["size"]
for info in infos.values()
}))
compatible = (
None if self._expected_dimension is None else dimensions == (self._expected_dimension,)
)
@@ -119,7 +138,7 @@ class QdrantVectorStore:
expected_dimension=self._expected_dimension,
observed_dimensions=dimensions,
dimension_compatible=compatible,
bm25_compatible=self._bm25_compatible(info),
bm25_compatible=self._bm25_compatible(infos["reference"]),
)
def search(
@@ -139,6 +158,21 @@ class QdrantVectorStore:
allowed_record_kinds = self._allowed_record_kinds(collections, kinds)
if not allowed_record_kinds:
return []
physical_collections = {
self._physical_collection_for_kind(kind) for kind in allowed_record_kinds
}
if len(physical_collections) != 1:
hits: list[VectorHit] = []
for collection in collections:
nested_kinds = sorted(set(allowed_record_kinds) & COLLECTION_KINDS[collection])
if nested_kinds:
hits.extend(self.search(
[collection], embedding, limit=limit, kinds=nested_kinds,
metadata_filter=metadata_filter, query_text=query_text,
query_language=query_language, retrieval_mode=retrieval_mode,
))
return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit]
physical_collection = physical_collections.pop()
filter_must = self._workspace_filter()
filter_must.extend(self._revision_filter(allowed_record_kinds))
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
@@ -193,7 +227,7 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"vector": embedding,
"limit": limit,
@@ -206,11 +240,11 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(strict=False, require_bm25=True)
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
shared_filter = {"must": filter_must}
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"query": self._bm25_document(query_text, query_language),
"using": "bm25",
@@ -224,7 +258,7 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence hybrid query text is required")
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"vector": embedding,
"limit": limit,
@@ -237,11 +271,11 @@ class QdrantVectorStore:
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(strict=False, require_bm25=True)
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
shared_filter = {"must": filter_must}
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"prefetch": [
{"query": embedding, "limit": limit * 2, "filter": shared_filter},
@@ -266,7 +300,9 @@ class QdrantVectorStore:
def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]:
validate_collection(collection)
validate_collection_kinds(collection, kinds)
physical_collection = self._physical_collection_for_logical(collection)
points = self._scroll(
physical_collection,
[
*self._workspace_filter(),
self._semantic_kind_filter(kinds),
@@ -287,7 +323,14 @@ class QdrantVectorStore:
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
validate_collection(collection)
# The first write with the split configuration is also the upgrade cutover. This keeps
# existing runtime memory reachable even when the operator reruns preprocessing without
# invoking the explicit clear operation first.
if self._split_collections:
self._migrate_legacy_memory()
physical_collection = self._physical_collection_for_logical(collection)
self._ensure_collection(
physical_collection,
strict=True,
require_bm25=any(record.sparse_text is not None for record in records),
)
@@ -310,7 +353,7 @@ class QdrantVectorStore:
self._workspace_id,
semantic_kind,
write_record.record.id,
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "evidence") else None,
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "schema_relationship", "evidence") else None,
),
"vector": vector,
"payload": qdrant_payload(
@@ -326,7 +369,7 @@ class QdrantVectorStore:
for start in range(0, len(points), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{self._collection}/points?wait=true",
f"/collections/{physical_collection}/points?wait=true",
{"points": points[start:start + UPSERT_BATCH_SIZE]},
)
return len(records)
@@ -334,15 +377,16 @@ class QdrantVectorStore:
def delete_kinds(self, collection: str, kinds: list[str]) -> int:
validate_collection(collection)
validate_collection_kinds(collection, kinds)
physical_collection = self._physical_collection_for_logical(collection)
must = [
*self._workspace_filter(),
self._semantic_kind_filter(kinds),
{"key": "record_kind", "match": {"any": sorted(kinds)}},
]
before = len(self._scroll(must))
before = len(self._scroll(physical_collection, must))
self._call(
"POST",
f"/collections/{self._collection}/points/delete?wait=true",
f"/collections/{physical_collection}/points/delete?wait=true",
{"filter": {"must": must}},
)
return before
@@ -353,6 +397,7 @@ class QdrantVectorStore:
if _WORKSPACE.fullmatch(workspace_id) is None:
raise VectorStoreError("Invalid Evidence workspace namespace")
self._require_bound_workspace(workspace_id)
physical_collection = self._collections["reference"]
must = [
*self._workspace_filter(),
{"key": "kind", "match": {"value": "evidence"}},
@@ -360,11 +405,11 @@ class QdrantVectorStore:
{"key": "vector_generation", "match": {"value": generation}},
]
before = len(
self._scroll(must)
self._scroll(physical_collection, must)
)
self._call(
"POST",
f"/collections/{self._collection}/points/delete?wait=true",
f"/collections/{physical_collection}/points/delete?wait=true",
{"filter": {"must": must}},
)
return before
@@ -375,7 +420,9 @@ class QdrantVectorStore:
if _WORKSPACE.fullmatch(workspace_id) is None:
raise VectorStoreError("Invalid Evidence workspace namespace")
self._require_bound_workspace(workspace_id)
physical_collection = self._collections["reference"]
points = self._scroll(
physical_collection,
[
*self._workspace_filter(),
{"key": "kind", "match": {"value": "evidence"}},
@@ -394,6 +441,68 @@ class QdrantVectorStore:
def _workspace_filter(self) -> list[dict]:
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
def clear_reference(self) -> bool:
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
legacy_deleted = self._migrate_legacy_memory()
collection = self._collections["reference"]
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
return legacy_deleted
self._call("DELETE", f"/collections/{collection}", None)
return True
def _migrate_legacy_memory(self) -> bool:
"""Preserve memory from the pre-split collection before retiring it."""
legacy = self._legacy_collection
if (
self._legacy_checked
or not self._split_collections
or legacy in self._collections.values()
):
return False
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
if response is None:
self._legacy_checked = True
return False
memory = self._collections["memory"]
self._ensure_collection(memory, strict=True, allow_create=True)
points = self._scroll(
legacy,
[
*self._workspace_filter(),
self._semantic_kind_filter(["memory", "solved_question"]),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
],
with_vector=True,
)
migrated = []
for point in points:
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
raise VectorStoreError("Qdrant returned malformed legacy memory response")
if not isinstance(point.get("payload"), dict):
raise VectorStoreError("Qdrant returned malformed legacy memory response")
migrated.append({
"id": point["id"],
"vector": point["vector"],
"payload": point["payload"],
})
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{memory}/points?wait=true",
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
)
self._call("DELETE", f"/collections/{legacy}", None)
self._legacy_checked = True
return True
def _physical_collection_for_logical(self, collection: str) -> str:
validate_collection(collection)
return self._collections["memory" if collection == "memory" else "reference"]
def _physical_collection_for_kind(self, kind: str) -> str:
return self._collections["reference" if kind in _REFERENCE_KINDS else "memory"]
@staticmethod
def _bm25_document(text: str, language: str) -> dict:
return {
@@ -405,7 +514,7 @@ class QdrantVectorStore:
def _revision_filter(self, kinds: list[str]) -> list[dict]:
if self._workspace_revision is None:
return []
if not any(kind in ("schema_table", "schema_column", "evidence") for kind in kinds):
if not any(kind in ("schema_table", "schema_column", "schema_relationship", "evidence") for kind in kinds):
return []
return [{"key": "workspace_revision", "match": {"value": self._workspace_revision}}]
@@ -450,25 +559,30 @@ class QdrantVectorStore:
bm25 = sparse_vectors.get("bm25")
return isinstance(bm25, dict) and bm25.get("modifier") == "idf"
def _ensure_collection(self, *, strict: bool, require_bm25: bool = False) -> dict | None:
response = self._call("GET", f"/collections/{self._collection}", None, allow_missing=True)
def _ensure_collection(
self, collection: str, *, strict: bool, require_bm25: bool = False,
allow_create: bool = False,
) -> dict | None:
created = False
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
if not strict:
raise VectorStoreError("Qdrant collection is missing")
if self._collection_lifecycle == "require_existing":
if self._collection_lifecycle == "require_existing" and not allow_create:
raise VectorStoreError("semantic_index_incompatible")
self._call(
"PUT",
f"/collections/{self._collection}",
f"/collections/{collection}",
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
)
for field_name in _KEYWORD_INDEXES:
self._call(
"PUT",
f"/collections/{self._collection}/index",
f"/collections/{collection}/index",
{"field_name": field_name, "field_schema": "keyword"},
)
response = self._call("GET", f"/collections/{self._collection}", None)
created = True
response = self._call("GET", f"/collections/{collection}", None)
result = response.get("result") if isinstance(response, dict) else None
config = result.get("config", {}).get("params", {}).get("vectors") if isinstance(result, dict) else None
if not isinstance(config, dict):
@@ -484,29 +598,38 @@ class QdrantVectorStore:
raise VectorStoreError("Qdrant collection configuration mismatch")
for field_name in _KEYWORD_INDEXES:
if field_name not in result.get("payload_schema", {}):
# A successful index-creation response can precede visibility in the
# collection-info payload. The newly-created collection is already safe to
# use; later readiness checks will validate the asynchronously published
# indexes. Existing collections still follow the strict lifecycle policy.
if created:
continue
if strict and self._collection_lifecycle == "require_existing":
raise VectorStoreError("semantic_index_incompatible")
if not strict:
raise VectorStoreError("Qdrant collection payload indexes mismatch")
self._call(
"PUT",
f"/collections/{self._collection}/index",
f"/collections/{collection}/index",
{"field_name": field_name, "field_schema": "keyword"},
)
if require_bm25 and not self._bm25_compatible(result):
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
return result
def _scroll(self, must: list[dict]) -> list[dict]:
def _scroll(
self, collection: str, must: list[dict], *, with_vector: bool = False
) -> list[dict]:
points: list[dict] = []
offset = None
seen_offsets = set()
while True:
response = self._call(
"POST",
f"/collections/{self._collection}/points/scroll",
f"/collections/{collection}/points/scroll",
{
"with_payload": True,
"with_vector": with_vector,
"limit": 10000,
"filter": {"must": must},
"offset": offset,