fix(preprocess): verify canonical generations and cleanup

This commit is contained in:
2026-07-12 06:18:18 +02:00
parent 4a29086fe4
commit 7d41c4cefc
6 changed files with 86 additions and 18 deletions
+14
View File
@@ -431,6 +431,20 @@ def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path):
assert second_embedder.calls == []
def test_unchanged_job_reuses_active_generation_without_new_directory(tmp_path):
source = Source([(item("one", "a"), "hello")])
candidate = pipeline(tmp_path, source)
args = dict(workspace_id="demo", workspace_root=tmp_path,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64)
first = candidate.run_as_job(**args)
count = len(candidate.store.list_generations())
second = candidate.run_as_job(**args)
assert second.generation == first.generation
assert second.published is False
assert len(candidate.store.list_generations()) == count
def test_removed_documents_are_marked_and_absent_from_new_manifest(tmp_path):
one, two = item("one", "a"), item("two", "b")
pipeline(tmp_path, Source([(one, "one"), (two, "two")])).run()
+9
View File
@@ -104,6 +104,15 @@ def test_s3_rejects_leading_slash_prefix_empty_and_control_keys():
list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover())
@pytest.mark.parametrize("prefix", ["/bad", "x" * 1025, "bad\x00prefix", "bad\x7fprefix"])
def test_s3_rejects_invalid_prefix_before_client_request(prefix):
from tht.adapters.evidence.s3 import S3EvidenceSource
client = Client()
with pytest.raises(ValueError, match="prefix"):
S3EvidenceSource(bucket="evidence", prefix=prefix, client=client)
assert client.list_calls == 0
def test_s3_hard_page_limit_never_requests_page_max_plus_one():
from tht.adapters.evidence.s3 import S3EvidenceSource
client = Client()
+3 -2
View File
@@ -29,8 +29,9 @@ class S3EvidenceSource:
if (not bucket_valid or bucket_is_ip
or any(value < 1 for value in (max_bytes, max_objects, max_pages, page_size))):
raise ValueError("S3 evidence limits and bucket must be non-empty and positive")
if prefix.startswith("/"):
raise ValueError("S3 prefix must not start with a slash")
if (prefix.startswith("/") or len(prefix.encode()) > 1024
or any(ord(char) < 32 or ord(char) == 127 for char in prefix)):
raise ValueError("S3 prefix is invalid")
if endpoint_url:
parsed = urlsplit(endpoint_url)
if parsed.username or parsed.password:
+10
View File
@@ -220,6 +220,16 @@ class CorpusPipeline:
"dimensions": self.embedding_dimensions,
"chunk_policy": asdict(self.chunk_policy),
})
previous = self.store.active_manifest()
if (not dry_run and resume_run_id is None and previous is not None
and previous.metadata.get("compatibility_fingerprint") == compatibility
and previous.metadata.get("fingerprints") == {
item.source_id: item.fingerprint for _, item in discovered
}):
return PipelineResult(
"succeeded", previous.manifest_id, False, (),
tuple(sorted(item.source_id for _, item in discovered)), (), previous,
)
spec = JobSpec(
workspace_id=workspace_id,
job_type="evidence",