test(evidence): harden generation lifecycle boundaries

This commit is contained in:
2026-07-12 05:15:30 +02:00
parent 459ffa0bcd
commit 9e21cce036
7 changed files with 135 additions and 3 deletions
@@ -53,3 +53,18 @@ legacy-config deprecation warnings.
Fresh verification after the fix wave: full harness `672 passed, 5 deselected`; Docker pgvector,
HTTP parity, and migration suites `43 passed`; exact direct inventory/delete integration `1 passed`;
changed-file Ruff and `git diff --check` clean.
## Final hardening verification
- Canonical generation validation is exact (`^gen:[0-9a-f]{32}$`) before HTTP/direct deletion;
malformed HTTP inventory rows fail closed rather than entering the GC candidate set.
- Added explicit protection coverage for running and failed-resumable JobRunner checkpoints, plus
a second-GC idempotence assertion for vector-only orphan reconciliation.
- Added deterministic concurrent locking coverage: a job paused after discovery retains the corpus
writer lock, explicit GC blocks, then completes after publication without deleting the active run.
- Added a descriptor-race regression: replacing the corpus pathname immediately after `read(2)`
leaves the atomically materialized session-owned copy byte-for-byte equal to the validated ACTIVE
document and its manifest hash.
Final fresh evidence: Docker pgvector/HTTP/migration suites `48 passed`; full harness `680 passed,
5 external L2 deselected`; changed-file Ruff and `git diff --check` clean.
@@ -264,3 +264,24 @@ def test_http_list_evidence_generations_exact_rpc_and_legacy_fail_closed(monkeyp
assert client.list_evidence_generations("evidence") == ["gen:" + "a" * 32]
assert calls[0][0].endswith("/rpc/list_evidence_generations")
assert calls[0][1] == {"table_name": "evidence", "kind": "evidence"}
@pytest.mark.parametrize("generation", ["gen:a", "gen:" + "A" * 32, "gen:" + "a" * 33])
def test_http_generation_operations_reject_noncanonical_values(monkeypatch, generation):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: pytest.fail("invalid generation reached transport"),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(ValueError, match="canonical"):
client.delete_generation("evidence", generation)
def test_http_inventory_rejects_malformed_rpc_output(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response([{"generation": "gen:../escape"}]),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(VectorRestError, match="malformed"):
client.list_evidence_generations("evidence")
+56
View File
@@ -3,6 +3,7 @@ import pytest
from tht.corpus.chunk import ChunkPolicy
from tht.corpus.pipeline import CorpusPipeline, PipelineError
from tht.corpus.store import CorpusStore
from tht.corpus.models import CorpusManifest
from tht.ports.evidence import AcquiredDocument, SourceObject
from tht.ports.vector import VectorCapabilities
@@ -162,6 +163,61 @@ def test_gc_reconciles_vector_only_generation(tmp_path):
report = candidate.gc(workspace_root=tmp_path)
assert report["evicted"] == [orphan]
assert vectors.list_evidence_generations("evidence") == []
assert candidate.gc(workspace_root=tmp_path)["evicted"] == []
@pytest.mark.parametrize("status", ["running", "failed"])
def test_gc_protects_generations_referenced_by_resumable_checkpoints(tmp_path, status):
generation = "gen:" + "e" * 32
store = CorpusStore(tmp_path / "corpus")
store.stage(CorpusManifest(), {}, generation=generation)
run = tmp_path / ".tht-jobs" / "evidence" / "runs" / ("a" * 32)
(run / "artifacts").mkdir(parents=True)
(run / "checkpoint.json").write_text(__import__("json").dumps({"status": status}))
(run / "artifacts" / "plan.json").write_text(
__import__("json").dumps({"generation": generation})
)
candidate = pipeline(tmp_path, Source([]), vectors=Vectors(), retain=1)
report = candidate.gc(workspace_root=tmp_path)
assert generation in report["protected"]
assert store.generation_path(generation).exists()
def test_explicit_gc_blocks_while_job_holds_corpus_writer_lock(tmp_path):
import threading
candidate = pipeline(tmp_path, Source([(item("one", "a"), "one")]), vectors=Vectors())
entered = threading.Event()
release = threading.Event()
gc_finished = threading.Event()
def pause(_context, stage):
if stage == "discover":
entered.set()
assert release.wait(5)
job = threading.Thread(target=lambda: candidate.run_as_job(
workspace_id="demo", workspace_root=tmp_path,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64,
after_stage_return=pause,
))
job.start()
assert entered.wait(5)
def collect():
with candidate.store.writer_lock():
candidate.gc(workspace_root=tmp_path)
gc_finished.set()
gc_thread = threading.Thread(target=collect)
gc_thread.start()
assert not gc_finished.wait(0.1)
release.set()
job.join(5)
gc_thread.join(5)
assert gc_finished.is_set()
assert candidate.store.active_generation() is not None
def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path):
+29
View File
@@ -108,3 +108,32 @@ def test_published_inventory_excludes_staged_and_invalid_newer_directories(tmp_p
invalid.mkdir()
(invalid / "PUBLISHED").write_text("2026-01-01T00:00:00Z\n")
assert store.published_generations() == [first]
def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch):
from tht.corpus.models import CanonicalDocument
import hashlib
content = "active bytes"
document = CanonicalDocument(
document_id="doc:" + "c" * 64, source_id="fs:copy", source_uri="file:///copy",
source_fingerprint="sha256:" + "d" * 64,
content_hash="sha256:" + hashlib.sha256(content.encode()).hexdigest(),
content=content, pipeline_version="evidence-v1",
)
store = CorpusStore(tmp_path / "corpus")
generation = store.stage(CorpusManifest(documents=(document,)), {document.document_id: content})
store.publish(generation)
source = store.resolve_document(document.document_id)
real_read = os.read
def replace_after_read(fd, size):
payload = real_read(fd, size)
source.unlink()
source.write_text("replacement")
return payload
monkeypatch.setattr(os, "read", replace_after_read)
owned = store.materialize_document(document.document_id, tmp_path / "session" / "evidence.md")
assert owned.read_text() == content
assert hashlib.sha256(owned.read_bytes()).hexdigest() == document.content_hash.removeprefix("sha256:")
+1 -1
View File
@@ -405,7 +405,7 @@ class PgVectorStore:
return len(records)
def delete_generation(self, collection: str, generation: str) -> int:
if collection != "evidence" or not generation.startswith("gen:"):
if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted")
raw = None
try:
+3 -1
View File
@@ -1,5 +1,7 @@
"""Thoth vector HTTP adapter using distinct read and write clients."""
import re
from tht.ports.vector import (
VectorCapabilities,
VectorHealth,
@@ -154,7 +156,7 @@ class ThothHttpVectorStore:
raise VectorStoreError(str(exc)) from exc
def delete_generation(self, collection: str, generation: str) -> int:
if collection != "evidence" or not generation.startswith("gen:"):
if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted")
try:
return self._require_writer().delete_generation(collection, generation)
+10 -1
View File
@@ -6,6 +6,7 @@ Errori in italiano e azionabili, stile `rest/client.py`.
"""
import requests
import re
from tht.config import RestConfig
@@ -128,6 +129,8 @@ class VectorRestClient:
return len(rows)
def delete_generation(self, table_name: str, generation: str) -> int:
if table_name != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None:
raise ValueError("generation must be canonical")
try:
payload = self._call(
"delete_vector_generation",
@@ -155,4 +158,10 @@ class VectorRestClient:
"list_evidence_generations RPC is unavailable; deploy the cleanup migration"
) from None
raise
return sorted({row["generation"] for row in rows if isinstance(row, dict)})
if not isinstance(rows, list) or any(
not isinstance(row, dict)
or re.fullmatch(r"gen:[0-9a-f]{32}", str(row.get("generation", ""))) is None
for row in rows
):
raise VectorRestError("list_evidence_generations returned malformed data")
return sorted({row["generation"] for row in rows})