fix: upsert in bounded chunks, recreate Qdrant indexes on rebuild, larger maintenance tmpfs
This commit is contained in:
@@ -25,6 +25,7 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
|
||||
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
|
||||
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
|
||||
_KEYWORD_INDEXES = (
|
||||
|
||||
"content_hash",
|
||||
"document_id",
|
||||
"kind",
|
||||
@@ -35,6 +36,8 @@ _KEYWORD_INDEXES = (
|
||||
"workspace_revision",
|
||||
)
|
||||
|
||||
UPSERT_BATCH_SIZE = 256
|
||||
|
||||
|
||||
def point_id(workspace_id: str, kind: str, record_key: str, workspace_revision: str | None = None) -> str:
|
||||
# P3: schema/Evidence points are revision-scoped; memory/solved remain workspace-wide.
|
||||
@@ -215,11 +218,14 @@ class QdrantVectorStore:
|
||||
),
|
||||
}
|
||||
)
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/points?wait=true",
|
||||
{"points": points},
|
||||
)
|
||||
# Qdrant rejects request bodies larger than its JSON limit (32 MiB by default).
|
||||
# A large schema/Evidence corpus therefore must be upserted in bounded chunks.
|
||||
for start in range(0, len(points), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/points?wait=true",
|
||||
{"points": points[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
return len(records)
|
||||
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int:
|
||||
|
||||
Reference in New Issue
Block a user