fix(evidence): reconcile generations safely

This commit is contained in:
2026-07-12 05:10:19 +02:00
parent f3b49f41c8
commit 459ffa0bcd
14 changed files with 227 additions and 10 deletions
+22
View File
@@ -87,6 +87,7 @@ class PgVectorStore:
upsert=writable,
metadata_filter=self._reader is not None,
delete_generation=writable,
list_evidence_generations=writable,
)
def _probe(
@@ -428,5 +429,26 @@ class PgVectorStore:
if raw is not None:
raw.close()
def list_evidence_generations(self, collection: str) -> list[str]:
if collection != "evidence":
raise VectorStoreError("Only exact Evidence generations may be listed")
raw = None
try:
raw = self._require_writer().raw_connection()
with raw.cursor() as cursor:
cursor.execute(
sql.SQL(
"SELECT DISTINCT metadata->>'vector_generation' FROM {} "
"WHERE kind = 'evidence' AND metadata->>'vector_generation' "
"~ '^gen:[0-9a-f]{{32}}$' ORDER BY 1"
).format(_collection(self._schema, collection))
)
return [row[0] for row in cursor.fetchall()]
except Exception as exc:
raise VectorWriteUnavailable("Vector generation inventory unavailable") from exc
finally:
if raw is not None:
raw.close()
__all__ = ["ALLOWED_COLLECTIONS", "PgVectorStore"]
@@ -42,6 +42,7 @@ class ThothHttpVectorStore:
return VectorCapabilities(
search=self._reader is not None, existing_hashes=writable, upsert=writable,
metadata_filter=self._reader is not None, delete_generation=writable,
list_evidence_generations=writable,
)
def health(self) -> VectorHealth:
@@ -160,6 +161,14 @@ class ThothHttpVectorStore:
except VectorRestError as exc:
raise VectorStoreError(str(exc)) from exc
def list_evidence_generations(self, collection: str) -> list[str]:
if collection != "evidence":
raise VectorStoreError("Only exact Evidence generations may be listed")
try:
return self._require_writer().list_evidence_generations(collection)
except VectorRestError as exc:
raise VectorWriteUnavailable("Vector generation inventory unavailable") from exc
@staticmethod
def _row(write_record: VectorWriteRecord) -> dict:
record = write_record.record
+16 -4
View File
@@ -92,9 +92,16 @@ class CorpusPipeline:
return protected
def gc(self, *, workspace_root: Path, dry_run: bool = False) -> dict:
generations = self.store.list_generations()
published = self.store.published_generations()
list_vectors = getattr(self.vector_store, "list_evidence_generations", None)
vector_generations = set(list_vectors("evidence")) if list_vectors else set()
generations = sorted(set(published) | vector_generations)
protected = self._protected_generations(workspace_root)
keep = set(generations[-self.retain_published_generations:]) | protected
active = self.store.active_generation()
rollback_count = self.retain_published_generations - 1
rollback = [generation for generation in published if generation != active]
keep = ({active} if active else set()) | set(rollback[-rollback_count:] if rollback_count else ())
keep |= protected
evicted, failures = [], []
for generation in generations:
if generation in keep:
@@ -108,7 +115,8 @@ class CorpusPipeline:
failures.append({"generation": generation, "error": "vector cleanup failed"})
continue
try:
self.store.discard(generation)
if generation in self.store.list_generations():
self.store.discard(generation)
evicted.append(generation)
except Exception:
failures.append({"generation": generation, "error": "filesystem cleanup failed"})
@@ -131,7 +139,11 @@ class CorpusPipeline:
with self.store.writer_lock():
return self._run(dry_run=dry_run, resume=resume)
def run_as_job(
def run_as_job(self, **kwargs) -> PipelineResult:
with self.store.writer_lock():
return self._run_as_job(**kwargs)
def _run_as_job(
self,
*,
workspace_id: str,
+37
View File
@@ -10,6 +10,7 @@ import stat
import shutil
import uuid
import hashlib
from datetime import UTC, datetime
from pathlib import Path
from contextlib import contextmanager
@@ -115,6 +116,7 @@ class CorpusStore:
if self.active_generation() == generation:
return generation
previous = self.active_generation()
published_marker = self.generation_path(generation) / "PUBLISHED"
temporary = self.active_path.with_name(f".ACTIVE.{uuid.uuid4().hex}.tmp")
replaced = False
try:
@@ -122,6 +124,10 @@ class CorpusStore:
self._replace(temporary, self.active_path)
replaced = True
self._fsync_directory()
_atomic_write(
published_marker,
(datetime.now(UTC).isoformat().replace("+00:00", "Z") + "\n").encode("ascii"),
)
except BaseException:
temporary.unlink(missing_ok=True)
if replaced:
@@ -179,6 +185,25 @@ class CorpusStore:
values.append(f"gen:{match.group(1)}")
return sorted(values, key=lambda value: self.generation_path(value).stat().st_mtime_ns)
def published_generations(self) -> list[str]:
active = self.active_generation()
published = []
for generation in self.list_generations():
path = self.generation_path(generation)
marker = path / "PUBLISHED"
if generation != active and not marker.is_file():
continue
try:
manifest = self.manifest(generation)
if manifest.manifest_id != generation:
continue
timestamp = marker.read_text(encoding="ascii").strip() if marker.is_file() else ""
key = (timestamp or manifest.created_at.isoformat(), generation)
published.append((key, generation))
except (OSError, ValueError):
continue
return [generation for _, generation in sorted(published)]
def resolve_document(self, document_id: str, generation: str | None = None) -> Path | None:
generation = generation or self.active_generation()
if generation is None:
@@ -221,3 +246,15 @@ class CorpusStore:
if documents_fd is not None:
os.close(documents_fd)
os.close(generation_fd)
def materialize_document(
self, document_id: str, destination: Path, generation: str | None = None,
) -> Path | None:
content = self.read_document(document_id, generation)
if content is None:
return None
destination = Path(destination)
destination.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
_atomic_write(destination, content.encode("utf-8"))
destination.chmod(0o400)
return destination
+3
View File
@@ -14,6 +14,7 @@ class VectorCapabilities:
upsert: bool = False
metadata_filter: bool = False
delete_generation: bool = False
list_evidence_generations: bool = False
@dataclass(frozen=True)
@@ -81,6 +82,8 @@ class VectorStore(Protocol):
def delete_generation(self, collection: str, generation: str) -> int: ...
def list_evidence_generations(self, collection: str) -> list[str]: ...
__all__ = [
"VectorCapabilities",
+6 -5
View File
@@ -56,7 +56,9 @@ def active_evidence_hits(store: CorpusStore, vector_store, embedding, *, limit:
]
def resolve_evidence_file(store: CorpusStore, evidence_id: str) -> str:
def resolve_evidence_file(
store: CorpusStore, evidence_id: str, *, materialized_root=None,
) -> str:
manifest = store.active_manifest()
if manifest is None:
return ""
@@ -64,9 +66,8 @@ def resolve_evidence_file(store: CorpusStore, evidence_id: str) -> str:
frontmatter = document.metadata.get("frontmatter", {})
identifiers = {document.document_id, document.source_id, str(frontmatter.get("id", ""))}
if evidence_id in identifiers:
# Validate the immutable bytes through the dirfd/O_NOFOLLOW reader before
# handing the path to legacy session-artifact consumers.
store.read_document(document.document_id)
path = store.resolve_document(document.document_id)
root = materialized_root or (store.root / "runtime")
filename = document.document_id.removeprefix("doc:") + ".md"
path = store.materialize_document(document.document_id, root / filename)
return str(path) if path else ""
return ""
+4 -1
View File
@@ -11,7 +11,10 @@ def _find_evidence_file(evidence_root: Path, evidence_id: str) -> str:
if corpus_root.exists():
from tht.corpus.store import CorpusStore
from tht.search.evidence import resolve_evidence_file
return resolve_evidence_file(CorpusStore(corpus_root), evidence_id)
return resolve_evidence_file(
CorpusStore(corpus_root), evidence_id,
materialized_root=evidence_root.parent / ".materialized-evidence",
)
for match in evidence_root.rglob(f"{evidence_id}.md"):
return str(match)
return ""
+14
View File
@@ -142,3 +142,17 @@ class VectorRestClient:
if isinstance(payload, dict):
return int(payload.get("deleted", 0))
return 0
def list_evidence_generations(self, table_name: str) -> list[str]:
try:
rows = self._call(
"list_evidence_generations",
{"table_name": table_name, "kind": "evidence"},
) or []
except VectorRestError as error:
if "HTTP 404" in str(error):
raise VectorRestError(
"list_evidence_generations RPC is unavailable; deploy the cleanup migration"
) from None
raise
return sorted({row["generation"] for row in rows if isinstance(row, dict)})