test(evidence): harden generation lifecycle boundaries

This commit is contained in:
2026-07-12 05:15:30 +02:00
parent 459ffa0bcd
commit 9e21cce036
7 changed files with 135 additions and 3 deletions
@@ -53,3 +53,18 @@ legacy-config deprecation warnings.
Fresh verification after the fix wave: full harness `672 passed, 5 deselected`; Docker pgvector, Fresh verification after the fix wave: full harness `672 passed, 5 deselected`; Docker pgvector,
HTTP parity, and migration suites `43 passed`; exact direct inventory/delete integration `1 passed`; HTTP parity, and migration suites `43 passed`; exact direct inventory/delete integration `1 passed`;
changed-file Ruff and `git diff --check` clean. changed-file Ruff and `git diff --check` clean.
## Final hardening verification
- Canonical generation validation is exact (`^gen:[0-9a-f]{32}$`) before HTTP/direct deletion;
malformed HTTP inventory rows fail closed rather than entering the GC candidate set.
- Added explicit protection coverage for running and failed-resumable JobRunner checkpoints, plus
a second-GC idempotence assertion for vector-only orphan reconciliation.
- Added deterministic concurrent locking coverage: a job paused after discovery retains the corpus
writer lock, explicit GC blocks, then completes after publication without deleting the active run.
- Added a descriptor-race regression: replacing the corpus pathname immediately after `read(2)`
leaves the atomically materialized session-owned copy byte-for-byte equal to the validated ACTIVE
document and its manifest hash.
Final fresh evidence: Docker pgvector/HTTP/migration suites `48 passed`; full harness `680 passed,
5 external L2 deselected`; changed-file Ruff and `git diff --check` clean.
@@ -264,3 +264,24 @@ def test_http_list_evidence_generations_exact_rpc_and_legacy_fail_closed(monkeyp
assert client.list_evidence_generations("evidence") == ["gen:" + "a" * 32] assert client.list_evidence_generations("evidence") == ["gen:" + "a" * 32]
assert calls[0][0].endswith("/rpc/list_evidence_generations") assert calls[0][0].endswith("/rpc/list_evidence_generations")
assert calls[0][1] == {"table_name": "evidence", "kind": "evidence"} assert calls[0][1] == {"table_name": "evidence", "kind": "evidence"}
@pytest.mark.parametrize("generation", ["gen:a", "gen:" + "A" * 32, "gen:" + "a" * 33])
def test_http_generation_operations_reject_noncanonical_values(monkeypatch, generation):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: pytest.fail("invalid generation reached transport"),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(ValueError, match="canonical"):
client.delete_generation("evidence", generation)
def test_http_inventory_rejects_malformed_rpc_output(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response([{"generation": "gen:../escape"}]),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(VectorRestError, match="malformed"):
client.list_evidence_generations("evidence")
+56
View File
@@ -3,6 +3,7 @@ import pytest
from tht.corpus.chunk import ChunkPolicy from tht.corpus.chunk import ChunkPolicy
from tht.corpus.pipeline import CorpusPipeline, PipelineError from tht.corpus.pipeline import CorpusPipeline, PipelineError
from tht.corpus.store import CorpusStore from tht.corpus.store import CorpusStore
from tht.corpus.models import CorpusManifest
from tht.ports.evidence import AcquiredDocument, SourceObject from tht.ports.evidence import AcquiredDocument, SourceObject
from tht.ports.vector import VectorCapabilities from tht.ports.vector import VectorCapabilities
@@ -162,6 +163,61 @@ def test_gc_reconciles_vector_only_generation(tmp_path):
report = candidate.gc(workspace_root=tmp_path) report = candidate.gc(workspace_root=tmp_path)
assert report["evicted"] == [orphan] assert report["evicted"] == [orphan]
assert vectors.list_evidence_generations("evidence") == [] assert vectors.list_evidence_generations("evidence") == []
assert candidate.gc(workspace_root=tmp_path)["evicted"] == []
@pytest.mark.parametrize("status", ["running", "failed"])
def test_gc_protects_generations_referenced_by_resumable_checkpoints(tmp_path, status):
generation = "gen:" + "e" * 32
store = CorpusStore(tmp_path / "corpus")
store.stage(CorpusManifest(), {}, generation=generation)
run = tmp_path / ".tht-jobs" / "evidence" / "runs" / ("a" * 32)
(run / "artifacts").mkdir(parents=True)
(run / "checkpoint.json").write_text(__import__("json").dumps({"status": status}))
(run / "artifacts" / "plan.json").write_text(
__import__("json").dumps({"generation": generation})
)
candidate = pipeline(tmp_path, Source([]), vectors=Vectors(), retain=1)
report = candidate.gc(workspace_root=tmp_path)
assert generation in report["protected"]
assert store.generation_path(generation).exists()
def test_explicit_gc_blocks_while_job_holds_corpus_writer_lock(tmp_path):
import threading
candidate = pipeline(tmp_path, Source([(item("one", "a"), "one")]), vectors=Vectors())
entered = threading.Event()
release = threading.Event()
gc_finished = threading.Event()
def pause(_context, stage):
if stage == "discover":
entered.set()
assert release.wait(5)
job = threading.Thread(target=lambda: candidate.run_as_job(
workspace_id="demo", workspace_root=tmp_path,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64,
after_stage_return=pause,
))
job.start()
assert entered.wait(5)
def collect():
with candidate.store.writer_lock():
candidate.gc(workspace_root=tmp_path)
gc_finished.set()
gc_thread = threading.Thread(target=collect)
gc_thread.start()
assert not gc_finished.wait(0.1)
release.set()
job.join(5)
gc_thread.join(5)
assert gc_finished.is_set()
assert candidate.store.active_generation() is not None
def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path): def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path):
+29
View File
@@ -108,3 +108,32 @@ def test_published_inventory_excludes_staged_and_invalid_newer_directories(tmp_p
invalid.mkdir() invalid.mkdir()
(invalid / "PUBLISHED").write_text("2026-01-01T00:00:00Z\n") (invalid / "PUBLISHED").write_text("2026-01-01T00:00:00Z\n")
assert store.published_generations() == [first] assert store.published_generations() == [first]
def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch):
from tht.corpus.models import CanonicalDocument
import hashlib
content = "active bytes"
document = CanonicalDocument(
document_id="doc:" + "c" * 64, source_id="fs:copy", source_uri="file:///copy",
source_fingerprint="sha256:" + "d" * 64,
content_hash="sha256:" + hashlib.sha256(content.encode()).hexdigest(),
content=content, pipeline_version="evidence-v1",
)
store = CorpusStore(tmp_path / "corpus")
generation = store.stage(CorpusManifest(documents=(document,)), {document.document_id: content})
store.publish(generation)
source = store.resolve_document(document.document_id)
real_read = os.read
def replace_after_read(fd, size):
payload = real_read(fd, size)
source.unlink()
source.write_text("replacement")
return payload
monkeypatch.setattr(os, "read", replace_after_read)
owned = store.materialize_document(document.document_id, tmp_path / "session" / "evidence.md")
assert owned.read_text() == content
assert hashlib.sha256(owned.read_bytes()).hexdigest() == document.content_hash.removeprefix("sha256:")
+1 -1
View File
@@ -405,7 +405,7 @@ class PgVectorStore:
return len(records) return len(records)
def delete_generation(self, collection: str, generation: str) -> int: def delete_generation(self, collection: str, generation: str) -> int:
if collection != "evidence" or not generation.startswith("gen:"): if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted") raise VectorStoreError("Only exact Evidence generations may be deleted")
raw = None raw = None
try: try:
+3 -1
View File
@@ -1,5 +1,7 @@
"""Thoth vector HTTP adapter using distinct read and write clients.""" """Thoth vector HTTP adapter using distinct read and write clients."""
import re
from tht.ports.vector import ( from tht.ports.vector import (
VectorCapabilities, VectorCapabilities,
VectorHealth, VectorHealth,
@@ -154,7 +156,7 @@ class ThothHttpVectorStore:
raise VectorStoreError(str(exc)) from exc raise VectorStoreError(str(exc)) from exc
def delete_generation(self, collection: str, generation: str) -> int: def delete_generation(self, collection: str, generation: str) -> int:
if collection != "evidence" or not generation.startswith("gen:"): if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted") raise VectorStoreError("Only exact Evidence generations may be deleted")
try: try:
return self._require_writer().delete_generation(collection, generation) return self._require_writer().delete_generation(collection, generation)
+10 -1
View File
@@ -6,6 +6,7 @@ Errori in italiano e azionabili, stile `rest/client.py`.
""" """
import requests import requests
import re
from tht.config import RestConfig from tht.config import RestConfig
@@ -128,6 +129,8 @@ class VectorRestClient:
return len(rows) return len(rows)
def delete_generation(self, table_name: str, generation: str) -> int: def delete_generation(self, table_name: str, generation: str) -> int:
if table_name != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise ValueError("generation must be canonical")
try: try:
payload = self._call( payload = self._call(
"delete_vector_generation", "delete_vector_generation",
@@ -155,4 +158,10 @@ class VectorRestClient:
"list_evidence_generations RPC is unavailable; deploy the cleanup migration" "list_evidence_generations RPC is unavailable; deploy the cleanup migration"
) from None ) from None
raise raise
return sorted({row["generation"] for row in rows if isinstance(row, dict)}) if not isinstance(rows, list) or any(
not isinstance(row, dict)
or re.fullmatch(r"gen:[0-9a-f]{32}", str(row.get("generation", ""))) is None
for row in rows
):
raise VectorRestError("list_evidence_generations returned malformed data")
return sorted({row["generation"] for row in rows})