fix(preprocess): verify canonical generations and cleanup
This commit is contained in:
@@ -431,6 +431,20 @@ def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path):
|
||||
assert second_embedder.calls == []
|
||||
|
||||
|
||||
def test_unchanged_job_reuses_active_generation_without_new_directory(tmp_path):
|
||||
source = Source([(item("one", "a"), "hello")])
|
||||
candidate = pipeline(tmp_path, source)
|
||||
args = dict(workspace_id="demo", workspace_root=tmp_path,
|
||||
config_fingerprint="sha256:" + "1" * 64,
|
||||
input_fingerprint="sha256:" + "2" * 64)
|
||||
first = candidate.run_as_job(**args)
|
||||
count = len(candidate.store.list_generations())
|
||||
second = candidate.run_as_job(**args)
|
||||
assert second.generation == first.generation
|
||||
assert second.published is False
|
||||
assert len(candidate.store.list_generations()) == count
|
||||
|
||||
|
||||
def test_removed_documents_are_marked_and_absent_from_new_manifest(tmp_path):
|
||||
one, two = item("one", "a"), item("two", "b")
|
||||
pipeline(tmp_path, Source([(one, "one"), (two, "two")])).run()
|
||||
|
||||
@@ -104,6 +104,15 @@ def test_s3_rejects_leading_slash_prefix_empty_and_control_keys():
|
||||
list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover())
|
||||
|
||||
|
||||
@pytest.mark.parametrize("prefix", ["/bad", "x" * 1025, "bad\x00prefix", "bad\x7fprefix"])
|
||||
def test_s3_rejects_invalid_prefix_before_client_request(prefix):
|
||||
from tht.adapters.evidence.s3 import S3EvidenceSource
|
||||
client = Client()
|
||||
with pytest.raises(ValueError, match="prefix"):
|
||||
S3EvidenceSource(bucket="evidence", prefix=prefix, client=client)
|
||||
assert client.list_calls == 0
|
||||
|
||||
|
||||
def test_s3_hard_page_limit_never_requests_page_max_plus_one():
|
||||
from tht.adapters.evidence.s3 import S3EvidenceSource
|
||||
client = Client()
|
||||
|
||||
@@ -29,8 +29,9 @@ class S3EvidenceSource:
|
||||
if (not bucket_valid or bucket_is_ip
|
||||
or any(value < 1 for value in (max_bytes, max_objects, max_pages, page_size))):
|
||||
raise ValueError("S3 evidence limits and bucket must be non-empty and positive")
|
||||
if prefix.startswith("/"):
|
||||
raise ValueError("S3 prefix must not start with a slash")
|
||||
if (prefix.startswith("/") or len(prefix.encode()) > 1024
|
||||
or any(ord(char) < 32 or ord(char) == 127 for char in prefix)):
|
||||
raise ValueError("S3 prefix is invalid")
|
||||
if endpoint_url:
|
||||
parsed = urlsplit(endpoint_url)
|
||||
if parsed.username or parsed.password:
|
||||
|
||||
@@ -220,6 +220,16 @@ class CorpusPipeline:
|
||||
"dimensions": self.embedding_dimensions,
|
||||
"chunk_policy": asdict(self.chunk_policy),
|
||||
})
|
||||
previous = self.store.active_manifest()
|
||||
if (not dry_run and resume_run_id is None and previous is not None
|
||||
and previous.metadata.get("compatibility_fingerprint") == compatibility
|
||||
and previous.metadata.get("fingerprints") == {
|
||||
item.source_id: item.fingerprint for _, item in discovered
|
||||
}):
|
||||
return PipelineResult(
|
||||
"succeeded", previous.manifest_id, False, (),
|
||||
tuple(sorted(item.source_id for _, item in discovered)), (), previous,
|
||||
)
|
||||
spec = JobSpec(
|
||||
workspace_id=workspace_id,
|
||||
job_type="evidence",
|
||||
|
||||
Reference in New Issue
Block a user