feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s

Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation.

Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
Codex
2026-09-10 10:31:34 +02:00
parent 8fe526dd6e
commit 82e2c91f42
168 changed files with 11914 additions and 1772 deletions
+78 -67
View File
@@ -80,9 +80,6 @@ class QdrantVectorStore:
if not legacy_constructor and collections["reference"] == collections["memory"]:
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
self._collections = dict(collections)
self._split_collections = collections["reference"] != collections["memory"]
self._legacy_collection = workspace_id
self._legacy_checked = False
self._workspace_id = workspace_id
self._workspace_revision = None
self._workspace_revision = workspace_revision
@@ -177,7 +174,13 @@ class QdrantVectorStore:
filter_must.extend(self._revision_filter(allowed_record_kinds))
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
filter_must.append({"key": "record_kind", "match": {"any": allowed_record_kinds}})
if metadata_filter is not None:
if metadata_filter is not None and "memory" in metadata_filter:
if set(metadata_filter) != {"memory"} or not set(allowed_record_kinds) <= {
"memory", "solved_question",
}:
raise VectorStoreError("Unsupported Memory metadata filter")
filter_must.extend(self._memory_filter(metadata_filter["memory"]))
elif metadata_filter is not None:
allowed_filters = {
"vector_generation", "document_ids", "workspace_id", "purpose",
"required_kinds", "required_concepts", "required_tables", "required_columns",
@@ -221,9 +224,9 @@ class QdrantVectorStore:
raise VectorStoreError("Invalid vector metadata filter")
filter_must.extend({"key": payload_key, "match": {"value": item}} for item in values)
if retrieval_mode not in {"fused", "dense", "bm25"}:
raise VectorStoreError("Evidence retrieval mode is invalid")
raise VectorStoreError("Vector retrieval mode is invalid")
if retrieval_mode == "dense":
if allowed_record_kinds != ["evidence"]:
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
response = self._call(
"POST",
@@ -236,7 +239,7 @@ class QdrantVectorStore:
},
)
elif retrieval_mode == "bm25":
if allowed_record_kinds != ["evidence"]:
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
@@ -267,8 +270,8 @@ class QdrantVectorStore:
},
)
else:
if allowed_record_kinds != ["evidence"]:
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Hybrid BM25 is only available for Evidence and Memory")
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
@@ -323,16 +326,13 @@ class QdrantVectorStore:
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
validate_collection(collection)
# The first write with the split configuration is also the upgrade cutover. This keeps
# existing runtime memory reachable even when the operator reruns preprocessing without
# invoking the explicit clear operation first.
if self._split_collections:
self._migrate_legacy_memory()
# Memory is projected only from its authoritative archive. Never import legacy payloads.
physical_collection = self._physical_collection_for_logical(collection)
self._ensure_collection(
physical_collection,
strict=True,
require_bm25=any(record.sparse_text is not None for record in records),
maintain_bm25=collection == "memory",
)
points = []
for write_record in records:
@@ -341,8 +341,8 @@ class QdrantVectorStore:
semantic_kind = qdrant_semantic_kind(write_record.record.kind)
vector: list[float] | dict = write_record.embedding
if write_record.sparse_text is not None:
if semantic_kind != "evidence" or write_record.sparse_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 document is invalid")
if semantic_kind not in {"evidence", "memory"} or write_record.sparse_language not in _BM25_LANGUAGES:
raise VectorStoreError("BM25 document is invalid")
vector = {
"": write_record.embedding,
"bm25": self._bm25_document(write_record.sparse_text, write_record.sparse_language),
@@ -391,6 +391,26 @@ class QdrantVectorStore:
)
return before
def prepare_memory_index(self) -> None:
"""Explicit rebuild may recreate a lost collection; reads never do so."""
self._ensure_collection(self._collections["memory"], strict=True,
require_bm25=True, maintain_bm25=True, allow_create=True)
def delete_memory_records(self, record_keys: list[str]) -> None:
"""Delete exact authoritative Memory projections, never reference vectors."""
if not record_keys or any(not key.startswith("card:mem-") for key in record_keys):
raise VectorStoreError("Exact Memory card keys are required")
collection = self._collections["memory"]
if self._call("GET", f"/collections/{collection}", None, allow_missing=True) is None:
return
self._call("POST", f"/collections/{collection}/points/delete?wait=true", {
"filter": {"must": [
*self._workspace_filter(),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
{"key": "record_key", "match": {"any": record_keys}},
]},
})
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int:
if collection != "evidence" or _GENERATION.fullmatch(generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted")
@@ -442,60 +462,14 @@ class QdrantVectorStore:
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
def clear_reference(self) -> bool:
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
legacy_deleted = self._migrate_legacy_memory()
"""Drop only replaceable schema/Evidence vectors; do not import legacy Memory."""
collection = self._collections["reference"]
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
return legacy_deleted
return False
self._call("DELETE", f"/collections/{collection}", None)
return True
def _migrate_legacy_memory(self) -> bool:
"""Preserve memory from the pre-split collection before retiring it."""
legacy = self._legacy_collection
if (
self._legacy_checked
or not self._split_collections
or legacy in self._collections.values()
):
return False
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
if response is None:
self._legacy_checked = True
return False
memory = self._collections["memory"]
self._ensure_collection(memory, strict=True, allow_create=True)
points = self._scroll(
legacy,
[
*self._workspace_filter(),
self._semantic_kind_filter(["memory", "solved_question"]),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
],
with_vector=True,
)
migrated = []
for point in points:
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
raise VectorStoreError("Qdrant returned malformed legacy memory response")
if not isinstance(point.get("payload"), dict):
raise VectorStoreError("Qdrant returned malformed legacy memory response")
migrated.append({
"id": point["id"],
"vector": point["vector"],
"payload": point["payload"],
})
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{memory}/points?wait=true",
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
)
self._call("DELETE", f"/collections/{legacy}", None)
self._legacy_checked = True
return True
def _physical_collection_for_logical(self, collection: str) -> str:
validate_collection(collection)
return self._collections["memory" if collection == "memory" else "reference"]
@@ -551,6 +525,34 @@ class QdrantVectorStore:
else "Embedding dimension does not match configured dimension"
)
@staticmethod
def _memory_filter(value: object) -> list[dict]:
fields = {"scope", "database", "schema_name", "table", "column"}
if not isinstance(value, dict) or set(value) != fields | {"family", "concepts", "format"}:
raise VectorStoreError("Invalid Memory metadata filter")
if (any(not isinstance(value[key], str) for key in fields)
or value["family"] is not None and not isinstance(value["family"], str)
or value["family"] not in {None, "domain_clarification", "sql_rule",
"solved_question", "explained_error"}
or type(value["format"]) is not int or value["format"] != 2
or not isinstance(value["concepts"], list)
or not all(isinstance(c, str) and c for c in value["concepts"])):
raise VectorStoreError("Invalid Memory metadata filter")
must = [{"key": "memory_format", "match": {"value": value["format"]}}]
for key in ("family", "scope"):
if value[key]:
must.append({"key": f"memory_{key}", "match": {"value": value[key]}})
must.extend({"key": "memory_concepts", "match": {"value": c}} for c in value["concepts"])
if value["database"]:
dependency = [{"key": "database", "match": {"value": value["database"]}}]
dependency.extend({"key": key, "match": {"any": ["", value[key]]}}
for key in ("schema_name", "table", "column") if value[key])
must.append({"should": [
{"is_empty": {"key": "memory_dependencies"}},
{"nested": {"key": "memory_dependencies", "filter": {"must": dependency}}},
]})
return must
@staticmethod
def _bm25_compatible(info: dict) -> bool:
sparse_vectors = info.get("config", {}).get("params", {}).get("sparse_vectors")
@@ -561,7 +563,7 @@ class QdrantVectorStore:
def _ensure_collection(
self, collection: str, *, strict: bool, require_bm25: bool = False,
allow_create: bool = False,
allow_create: bool = False, maintain_bm25: bool = False,
) -> dict | None:
created = False
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
@@ -573,7 +575,9 @@ class QdrantVectorStore:
self._call(
"PUT",
f"/collections/{collection}",
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"},
**({"sparse_vectors": {"bm25": {"modifier": "idf"}}}
if require_bm25 and maintain_bm25 else {})},
)
for field_name in _KEYWORD_INDEXES:
self._call(
@@ -614,7 +618,14 @@ class QdrantVectorStore:
{"field_name": field_name, "field_schema": "keyword"},
)
if require_bm25 and not self._bm25_compatible(result):
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
sparse = result.get("config", {}).get("params", {}).get("sparse_vectors")
if maintain_bm25 and (sparse is None or isinstance(sparse, dict) and "bm25" not in sparse):
# Explicit Memory writes may add the missing sparse vector without
# touching dense points or the separately managed Reference collection.
self._call("PUT", f"/collections/{collection}/vectors/bm25",
{"sparse": {"modifier": "idf"}})
else:
raise VectorStoreError("BM25 collection configuration mismatch")
return result
def _scroll(