feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s
Publish documentation / publish (push) Successful in 2m12s
This commit is contained in:
@@ -5,7 +5,7 @@ from __future__ import annotations
|
||||
from tht.ports.vector import VectorStoreError
|
||||
|
||||
COLLECTION_KINDS = {
|
||||
"schema_records": {"schema_table", "schema_column"},
|
||||
"schema_records": {"schema_table", "schema_column", "schema_relationship"},
|
||||
"evidence": {"evidence"},
|
||||
"memory": {"memory", "solved_question"},
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
"""Qdrant-backed vector store for one workspace-owned semantic collection."""
|
||||
"""Qdrant-backed vector store with separate reference and memory lifecycles."""
|
||||
|
||||
import re
|
||||
from collections.abc import Callable
|
||||
@@ -25,6 +25,9 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
|
||||
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
|
||||
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
|
||||
_BM25_LANGUAGES = frozenset({"english", "italian"})
|
||||
_REFERENCE_KINDS = frozenset({
|
||||
"schema_table", "schema_column", "schema_relationship", "evidence",
|
||||
})
|
||||
_KEYWORD_INDEXES = (
|
||||
|
||||
"content_hash",
|
||||
@@ -58,7 +61,8 @@ class QdrantVectorStore:
|
||||
self,
|
||||
*,
|
||||
base_url: str,
|
||||
collection: str,
|
||||
collections: dict[str, str] | None = None,
|
||||
collection: str | None = None,
|
||||
workspace_id: str,
|
||||
workspace_revision: str | None = None,
|
||||
expected_dimension: int | None = None,
|
||||
@@ -68,7 +72,17 @@ class QdrantVectorStore:
|
||||
read_timeout: float = 10.0,
|
||||
):
|
||||
self._base_url = base_url.rstrip("/")
|
||||
self._collection = collection
|
||||
legacy_constructor = collections is None and collection is not None
|
||||
if legacy_constructor:
|
||||
collections = {"reference": collection, "memory": collection}
|
||||
if collections is None or set(collections) != {"reference", "memory"}:
|
||||
raise VectorStoreError("Qdrant collections must define reference and memory")
|
||||
if not legacy_constructor and collections["reference"] == collections["memory"]:
|
||||
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
|
||||
self._collections = dict(collections)
|
||||
self._split_collections = collections["reference"] != collections["memory"]
|
||||
self._legacy_collection = workspace_id
|
||||
self._legacy_checked = False
|
||||
self._workspace_id = workspace_id
|
||||
self._workspace_revision = None
|
||||
self._workspace_revision = workspace_revision
|
||||
@@ -90,7 +104,10 @@ class QdrantVectorStore:
|
||||
|
||||
def health(self) -> VectorHealth:
|
||||
try:
|
||||
info = self._ensure_collection(strict=False)
|
||||
infos = {
|
||||
name: self._ensure_collection(collection, strict=False)
|
||||
for name, collection in self._collections.items()
|
||||
}
|
||||
except VectorStoreError as exc:
|
||||
return VectorHealth(
|
||||
ok=False,
|
||||
@@ -105,8 +122,10 @@ class QdrantVectorStore:
|
||||
bm25_compatible=None,
|
||||
)
|
||||
|
||||
dimension = info["config"]["params"]["vectors"]["size"]
|
||||
dimensions = (dimension,)
|
||||
dimensions = tuple(sorted({
|
||||
info["config"]["params"]["vectors"]["size"]
|
||||
for info in infos.values()
|
||||
}))
|
||||
compatible = (
|
||||
None if self._expected_dimension is None else dimensions == (self._expected_dimension,)
|
||||
)
|
||||
@@ -119,7 +138,7 @@ class QdrantVectorStore:
|
||||
expected_dimension=self._expected_dimension,
|
||||
observed_dimensions=dimensions,
|
||||
dimension_compatible=compatible,
|
||||
bm25_compatible=self._bm25_compatible(info),
|
||||
bm25_compatible=self._bm25_compatible(infos["reference"]),
|
||||
)
|
||||
|
||||
def search(
|
||||
@@ -139,6 +158,21 @@ class QdrantVectorStore:
|
||||
allowed_record_kinds = self._allowed_record_kinds(collections, kinds)
|
||||
if not allowed_record_kinds:
|
||||
return []
|
||||
physical_collections = {
|
||||
self._physical_collection_for_kind(kind) for kind in allowed_record_kinds
|
||||
}
|
||||
if len(physical_collections) != 1:
|
||||
hits: list[VectorHit] = []
|
||||
for collection in collections:
|
||||
nested_kinds = sorted(set(allowed_record_kinds) & COLLECTION_KINDS[collection])
|
||||
if nested_kinds:
|
||||
hits.extend(self.search(
|
||||
[collection], embedding, limit=limit, kinds=nested_kinds,
|
||||
metadata_filter=metadata_filter, query_text=query_text,
|
||||
query_language=query_language, retrieval_mode=retrieval_mode,
|
||||
))
|
||||
return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit]
|
||||
physical_collection = physical_collections.pop()
|
||||
filter_must = self._workspace_filter()
|
||||
filter_must.extend(self._revision_filter(allowed_record_kinds))
|
||||
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
|
||||
@@ -193,7 +227,7 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"vector": embedding,
|
||||
"limit": limit,
|
||||
@@ -206,11 +240,11 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(strict=False, require_bm25=True)
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
shared_filter = {"must": filter_must}
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"query": self._bm25_document(query_text, query_language),
|
||||
"using": "bm25",
|
||||
@@ -224,7 +258,7 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence hybrid query text is required")
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"vector": embedding,
|
||||
"limit": limit,
|
||||
@@ -237,11 +271,11 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
|
||||
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(strict=False, require_bm25=True)
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
shared_filter = {"must": filter_must}
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"prefetch": [
|
||||
{"query": embedding, "limit": limit * 2, "filter": shared_filter},
|
||||
@@ -266,7 +300,9 @@ class QdrantVectorStore:
|
||||
def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]:
|
||||
validate_collection(collection)
|
||||
validate_collection_kinds(collection, kinds)
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
points = self._scroll(
|
||||
physical_collection,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(kinds),
|
||||
@@ -287,7 +323,14 @@ class QdrantVectorStore:
|
||||
|
||||
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
|
||||
validate_collection(collection)
|
||||
# The first write with the split configuration is also the upgrade cutover. This keeps
|
||||
# existing runtime memory reachable even when the operator reruns preprocessing without
|
||||
# invoking the explicit clear operation first.
|
||||
if self._split_collections:
|
||||
self._migrate_legacy_memory()
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
self._ensure_collection(
|
||||
physical_collection,
|
||||
strict=True,
|
||||
require_bm25=any(record.sparse_text is not None for record in records),
|
||||
)
|
||||
@@ -310,7 +353,7 @@ class QdrantVectorStore:
|
||||
self._workspace_id,
|
||||
semantic_kind,
|
||||
write_record.record.id,
|
||||
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "evidence") else None,
|
||||
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "schema_relationship", "evidence") else None,
|
||||
),
|
||||
"vector": vector,
|
||||
"payload": qdrant_payload(
|
||||
@@ -326,7 +369,7 @@ class QdrantVectorStore:
|
||||
for start in range(0, len(points), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/points?wait=true",
|
||||
f"/collections/{physical_collection}/points?wait=true",
|
||||
{"points": points[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
return len(records)
|
||||
@@ -334,15 +377,16 @@ class QdrantVectorStore:
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int:
|
||||
validate_collection(collection)
|
||||
validate_collection_kinds(collection, kinds)
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
must = [
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(kinds),
|
||||
{"key": "record_kind", "match": {"any": sorted(kinds)}},
|
||||
]
|
||||
before = len(self._scroll(must))
|
||||
before = len(self._scroll(physical_collection, must))
|
||||
self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/delete?wait=true",
|
||||
f"/collections/{physical_collection}/points/delete?wait=true",
|
||||
{"filter": {"must": must}},
|
||||
)
|
||||
return before
|
||||
@@ -353,6 +397,7 @@ class QdrantVectorStore:
|
||||
if _WORKSPACE.fullmatch(workspace_id) is None:
|
||||
raise VectorStoreError("Invalid Evidence workspace namespace")
|
||||
self._require_bound_workspace(workspace_id)
|
||||
physical_collection = self._collections["reference"]
|
||||
must = [
|
||||
*self._workspace_filter(),
|
||||
{"key": "kind", "match": {"value": "evidence"}},
|
||||
@@ -360,11 +405,11 @@ class QdrantVectorStore:
|
||||
{"key": "vector_generation", "match": {"value": generation}},
|
||||
]
|
||||
before = len(
|
||||
self._scroll(must)
|
||||
self._scroll(physical_collection, must)
|
||||
)
|
||||
self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/delete?wait=true",
|
||||
f"/collections/{physical_collection}/points/delete?wait=true",
|
||||
{"filter": {"must": must}},
|
||||
)
|
||||
return before
|
||||
@@ -375,7 +420,9 @@ class QdrantVectorStore:
|
||||
if _WORKSPACE.fullmatch(workspace_id) is None:
|
||||
raise VectorStoreError("Invalid Evidence workspace namespace")
|
||||
self._require_bound_workspace(workspace_id)
|
||||
physical_collection = self._collections["reference"]
|
||||
points = self._scroll(
|
||||
physical_collection,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
{"key": "kind", "match": {"value": "evidence"}},
|
||||
@@ -394,6 +441,68 @@ class QdrantVectorStore:
|
||||
def _workspace_filter(self) -> list[dict]:
|
||||
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
|
||||
|
||||
def clear_reference(self) -> bool:
|
||||
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
|
||||
legacy_deleted = self._migrate_legacy_memory()
|
||||
collection = self._collections["reference"]
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
return legacy_deleted
|
||||
self._call("DELETE", f"/collections/{collection}", None)
|
||||
return True
|
||||
|
||||
def _migrate_legacy_memory(self) -> bool:
|
||||
"""Preserve memory from the pre-split collection before retiring it."""
|
||||
legacy = self._legacy_collection
|
||||
if (
|
||||
self._legacy_checked
|
||||
or not self._split_collections
|
||||
or legacy in self._collections.values()
|
||||
):
|
||||
return False
|
||||
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
|
||||
if response is None:
|
||||
self._legacy_checked = True
|
||||
return False
|
||||
memory = self._collections["memory"]
|
||||
self._ensure_collection(memory, strict=True, allow_create=True)
|
||||
points = self._scroll(
|
||||
legacy,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(["memory", "solved_question"]),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
],
|
||||
with_vector=True,
|
||||
)
|
||||
migrated = []
|
||||
for point in points:
|
||||
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
if not isinstance(point.get("payload"), dict):
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
migrated.append({
|
||||
"id": point["id"],
|
||||
"vector": point["vector"],
|
||||
"payload": point["payload"],
|
||||
})
|
||||
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{memory}/points?wait=true",
|
||||
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
self._call("DELETE", f"/collections/{legacy}", None)
|
||||
self._legacy_checked = True
|
||||
return True
|
||||
|
||||
def _physical_collection_for_logical(self, collection: str) -> str:
|
||||
validate_collection(collection)
|
||||
return self._collections["memory" if collection == "memory" else "reference"]
|
||||
|
||||
def _physical_collection_for_kind(self, kind: str) -> str:
|
||||
return self._collections["reference" if kind in _REFERENCE_KINDS else "memory"]
|
||||
|
||||
@staticmethod
|
||||
def _bm25_document(text: str, language: str) -> dict:
|
||||
return {
|
||||
@@ -405,7 +514,7 @@ class QdrantVectorStore:
|
||||
def _revision_filter(self, kinds: list[str]) -> list[dict]:
|
||||
if self._workspace_revision is None:
|
||||
return []
|
||||
if not any(kind in ("schema_table", "schema_column", "evidence") for kind in kinds):
|
||||
if not any(kind in ("schema_table", "schema_column", "schema_relationship", "evidence") for kind in kinds):
|
||||
return []
|
||||
return [{"key": "workspace_revision", "match": {"value": self._workspace_revision}}]
|
||||
|
||||
@@ -450,25 +559,30 @@ class QdrantVectorStore:
|
||||
bm25 = sparse_vectors.get("bm25")
|
||||
return isinstance(bm25, dict) and bm25.get("modifier") == "idf"
|
||||
|
||||
def _ensure_collection(self, *, strict: bool, require_bm25: bool = False) -> dict | None:
|
||||
response = self._call("GET", f"/collections/{self._collection}", None, allow_missing=True)
|
||||
def _ensure_collection(
|
||||
self, collection: str, *, strict: bool, require_bm25: bool = False,
|
||||
allow_create: bool = False,
|
||||
) -> dict | None:
|
||||
created = False
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
if not strict:
|
||||
raise VectorStoreError("Qdrant collection is missing")
|
||||
if self._collection_lifecycle == "require_existing":
|
||||
if self._collection_lifecycle == "require_existing" and not allow_create:
|
||||
raise VectorStoreError("semantic_index_incompatible")
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}",
|
||||
f"/collections/{collection}",
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
|
||||
)
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/index",
|
||||
f"/collections/{collection}/index",
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
response = self._call("GET", f"/collections/{self._collection}", None)
|
||||
created = True
|
||||
response = self._call("GET", f"/collections/{collection}", None)
|
||||
result = response.get("result") if isinstance(response, dict) else None
|
||||
config = result.get("config", {}).get("params", {}).get("vectors") if isinstance(result, dict) else None
|
||||
if not isinstance(config, dict):
|
||||
@@ -484,29 +598,38 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Qdrant collection configuration mismatch")
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
if field_name not in result.get("payload_schema", {}):
|
||||
# A successful index-creation response can precede visibility in the
|
||||
# collection-info payload. The newly-created collection is already safe to
|
||||
# use; later readiness checks will validate the asynchronously published
|
||||
# indexes. Existing collections still follow the strict lifecycle policy.
|
||||
if created:
|
||||
continue
|
||||
if strict and self._collection_lifecycle == "require_existing":
|
||||
raise VectorStoreError("semantic_index_incompatible")
|
||||
if not strict:
|
||||
raise VectorStoreError("Qdrant collection payload indexes mismatch")
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/index",
|
||||
f"/collections/{collection}/index",
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
if require_bm25 and not self._bm25_compatible(result):
|
||||
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
|
||||
return result
|
||||
|
||||
def _scroll(self, must: list[dict]) -> list[dict]:
|
||||
def _scroll(
|
||||
self, collection: str, must: list[dict], *, with_vector: bool = False
|
||||
) -> list[dict]:
|
||||
points: list[dict] = []
|
||||
offset = None
|
||||
seen_offsets = set()
|
||||
while True:
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/scroll",
|
||||
f"/collections/{collection}/points/scroll",
|
||||
{
|
||||
"with_payload": True,
|
||||
"with_vector": with_vector,
|
||||
"limit": 10000,
|
||||
"filter": {"must": must},
|
||||
"offset": offset,
|
||||
|
||||
Reference in New Issue
Block a user