fix: upsert in bounded chunks, recreate Qdrant indexes on rebuild, larger maintenance tmpfs

This commit is contained in:
2026-08-13 19:02:52 +02:00
parent 84b233d937
commit f31b1e61ac
4 changed files with 56 additions and 12 deletions
+11 -5
View File
@@ -25,6 +25,7 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
_KEYWORD_INDEXES = (
"content_hash",
"document_id",
"kind",
@@ -35,6 +36,8 @@ _KEYWORD_INDEXES = (
"workspace_revision",
)
UPSERT_BATCH_SIZE = 256
def point_id(workspace_id: str, kind: str, record_key: str, workspace_revision: str | None = None) -> str:
# P3: schema/Evidence points are revision-scoped; memory/solved remain workspace-wide.
@@ -215,11 +218,14 @@ class QdrantVectorStore:
),
}
)
self._call(
"PUT",
f"/collections/{self._collection}/points?wait=true",
{"points": points},
)
# Qdrant rejects request bodies larger than its JSON limit (32 MiB by default).
# A large schema/Evidence corpus therefore must be upserted in bounded chunks.
for start in range(0, len(points), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{self._collection}/points?wait=true",
{"points": points[start:start + UPSERT_BATCH_SIZE]},
)
return len(records)
def delete_kinds(self, collection: str, kinds: list[str]) -> int: