363 lines
12 KiB
Python
363 lines
12 KiB
Python
import json
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
from typer.testing import CliRunner
|
|
|
|
from tht.cli import app
|
|
|
|
|
|
def _runtime_config(
|
|
tmp_path: Path, name: str = "workspace.yaml", evidence_schema_version: int | None = None,
|
|
) -> Path:
|
|
path = tmp_path / name
|
|
(tmp_path / "evidence").mkdir(exist_ok=True)
|
|
evidence_version = "" if evidence_schema_version is None else f"\n schema_version: {evidence_schema_version}"
|
|
path.write_text(
|
|
f"""
|
|
runtime_identity:
|
|
workspace_id: psd-clinical
|
|
workspace_revision: {'a' * 40}
|
|
dwh:
|
|
type: postgres_direct
|
|
connection: {{database: analytics, schema: mart, user: reader, password: secret}}
|
|
vectors:
|
|
type: qdrant
|
|
base_url: http://qdrant:6333
|
|
collection: psd-clinical
|
|
embeddings:
|
|
provider: ollama_internal
|
|
base_url: http://embedding:11434
|
|
model: qwen3-embedding:0.6b
|
|
dim: 1024
|
|
evidence:
|
|
{evidence_version}
|
|
sources:
|
|
- type: filesystem
|
|
root: {tmp_path / 'evidence'}
|
|
roots:
|
|
sessions: {tmp_path / 'sessions'}
|
|
artifacts: {tmp_path / 'artifacts'}
|
|
indexes: {tmp_path / 'indexes'}
|
|
"""
|
|
)
|
|
return path
|
|
|
|
|
|
def test_v2_invalid_materialized_corpus_stops_before_vector_store_construction(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
from tht.evidence import ValidationFinding, ValidationReport
|
|
|
|
config = _runtime_config(tmp_path, evidence_schema_version=2)
|
|
vector_store_constructed = False
|
|
monkeypatch.setattr(
|
|
"tht.evidence.validate_workspace_evidence",
|
|
lambda root: ValidationReport((ValidationFinding(
|
|
"error", "manifest_missing", "manifest.yaml", "missing",
|
|
),)),
|
|
)
|
|
|
|
def forbidden_vector_store(*args, **kwargs):
|
|
nonlocal vector_store_constructed
|
|
vector_store_constructed = True
|
|
raise AssertionError("invalid corpus must not reach vector upsert setup")
|
|
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", forbidden_vector_store)
|
|
|
|
with pytest.raises(RuntimeError, match="curated Evidence corpus is invalid"):
|
|
command.run_from_config(config)
|
|
|
|
assert vector_store_constructed is False
|
|
|
|
|
|
def test_v2_valid_materialized_corpus_is_validated_before_preprocessing(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
from tht.evidence import ValidationReport
|
|
|
|
config = _runtime_config(tmp_path, evidence_schema_version=2)
|
|
calls = []
|
|
|
|
class FakePipeline:
|
|
def run_as_job(self, **kwargs):
|
|
calls.append(("run", kwargs))
|
|
return "completed"
|
|
|
|
monkeypatch.setattr("tht.evidence.validate_workspace_evidence", lambda root: calls.append(("validate", root)) or ValidationReport(()))
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: calls.append(("vector", require_write)) or object())
|
|
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
|
monkeypatch.setattr("tht.evidence.build_sources", lambda evidence: [])
|
|
monkeypatch.setattr("tht.evidence.build_preprocessing_pipeline", lambda **kwargs: FakePipeline())
|
|
|
|
assert command.run_from_config(config) == "completed"
|
|
assert calls[:2] == [("validate", tmp_path), ("vector", True)]
|
|
|
|
|
|
def test_preprocess_evidence_json_is_pristine(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "succeeded",
|
|
"generation": "gen:abc",
|
|
"published": True,
|
|
"counts": {"changed": 0, "unchanged": 0, "removed": 0, "documents": 0, "chunks": 0},
|
|
"changed": [],
|
|
"unchanged": [],
|
|
"removed": [],
|
|
"manifest_id": "manifest-1",
|
|
"run_id": "a" * 32,
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 0, response.output
|
|
assert response.stderr == ""
|
|
assert json.loads(response.stdout) == {
|
|
"changed": [],
|
|
"code": "ok",
|
|
"counts": {"changed": 0, "chunks": 0, "documents": 0, "removed": 0, "unchanged": 0},
|
|
"generation": "gen:abc",
|
|
"manifest_id": "manifest-1",
|
|
"operation": "preprocess_evidence",
|
|
"published": True,
|
|
"removed": [],
|
|
"resumed_from": None,
|
|
"run_id": "a" * 32,
|
|
"schemaVersion": 1,
|
|
"status": "succeeded",
|
|
"unchanged": [],
|
|
"workspaceId": "psd-clinical",
|
|
"workspaceRevision": "a" * 40,
|
|
}
|
|
|
|
|
|
def test_preprocess_failure_is_structured_and_nonzero(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
monkeypatch.setattr(
|
|
command,
|
|
"run_from_config",
|
|
lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail")),
|
|
)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code != 0
|
|
payload = json.loads(response.stdout)
|
|
assert payload == {
|
|
"code": "preprocessing_failed",
|
|
"error": "preprocessing failed",
|
|
"operation": "preprocess_evidence",
|
|
"schemaVersion": 1,
|
|
"status": "failed",
|
|
"workspaceId": "psd-clinical",
|
|
"workspaceRevision": "a" * 40,
|
|
}
|
|
assert "secret detail" not in response.output
|
|
|
|
|
|
def test_preprocess_failed_job_report_is_sanitized_json_and_nonzero(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "failed",
|
|
"run_id": "a" * 32,
|
|
"published": False,
|
|
"generation": "gen:" + "b" * 32,
|
|
"changed": ["fs:one"],
|
|
"unchanged": [],
|
|
"removed": [],
|
|
"counts": {"changed": 1, "unchanged": 0, "removed": 0, "documents": 1, "chunks": 1},
|
|
"manifest_id": "manifest-1",
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 1
|
|
payload = json.loads(response.stdout)
|
|
assert payload["status"] == "failed"
|
|
assert payload["error"] == "preprocessing job failed"
|
|
assert payload["workspaceId"] == "psd-clinical"
|
|
assert "traceback" not in response.output.lower()
|
|
|
|
|
|
def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path):
|
|
from test_corpus_pipeline import Source, item, pipeline
|
|
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = pipeline(
|
|
tmp_path,
|
|
Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))]),
|
|
).run_as_job(
|
|
workspace_id="demo",
|
|
workspace_root=tmp_path,
|
|
config_fingerprint="sha256:" + "1" * 64,
|
|
input_fingerprint="sha256:" + "2" * 64,
|
|
)
|
|
assert result.status == "failed"
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 1
|
|
assert json.loads(response.stdout)["status"] == "failed"
|
|
assert "SENSITIVE EVIDENCE" not in response.output
|
|
assert "secret" not in response.output
|
|
|
|
|
|
def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "succeeded",
|
|
"run_id": "a" * 32,
|
|
"generation": "gen:" + "b" * 64,
|
|
"published": True,
|
|
"changed": ["fs:item"] * 100,
|
|
"unchanged": ["fs:item"] * 100,
|
|
"removed": ["fs:item"] * 100,
|
|
"counts": {"changed": 1001, "unchanged": 902, "removed": 803},
|
|
"manifest_id": "manifest-1",
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "-c", str(config)])
|
|
|
|
assert response.exit_code == 0, response.output
|
|
assert "changed=1001 unchanged=902 removed=803" in response.output
|
|
|
|
|
|
def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
called = False
|
|
|
|
def forbidden(*args, **kwargs):
|
|
nonlocal called
|
|
called = True
|
|
|
|
monkeypatch.setattr(command, "run_from_config", forbidden)
|
|
response = CliRunner().invoke(
|
|
app,
|
|
[
|
|
"preprocess",
|
|
"evidence",
|
|
"--resume",
|
|
"gen:" + "a" * 32,
|
|
"--json",
|
|
"-c",
|
|
str(config),
|
|
],
|
|
)
|
|
assert response.exit_code != 0
|
|
assert json.loads(response.output) == {
|
|
"code": "invalid_resume",
|
|
"error": "resume requires a preprocessing run id",
|
|
"operation": "preprocess_evidence",
|
|
"schemaVersion": 1,
|
|
"status": "failed",
|
|
}
|
|
assert called is False
|
|
|
|
|
|
def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path, name="3")
|
|
calls = {}
|
|
|
|
class FakePipeline:
|
|
def __init__(
|
|
self,
|
|
*,
|
|
store,
|
|
sources,
|
|
embedder,
|
|
vector_store,
|
|
embedding_model,
|
|
embedding_dimensions,
|
|
chunk_policy,
|
|
pipeline_version,
|
|
retain_published_generations,
|
|
sparse_language,
|
|
candidate_evaluator,
|
|
):
|
|
calls["init"] = {
|
|
"embedding_model": embedding_model,
|
|
"embedding_dimensions": embedding_dimensions,
|
|
"pipeline_version": pipeline_version,
|
|
"sparse_language": sparse_language,
|
|
"candidate_evaluator": candidate_evaluator,
|
|
}
|
|
|
|
def run_as_job(self, **kwargs):
|
|
calls["run_as_job"] = kwargs
|
|
return SimpleNamespace(model_dump=lambda mode=None: {"status": "succeeded"})
|
|
|
|
monkeypatch.setattr("tht.evidence.build_sources", lambda cfg: [])
|
|
monkeypatch.setattr(
|
|
"tht.evidence.build_preprocessing_pipeline",
|
|
lambda **kwargs: FakePipeline(**kwargs),
|
|
)
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: object())
|
|
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
|
|
|
command.run_from_config(config)
|
|
|
|
assert calls["init"]["sparse_language"] == "english"
|
|
assert calls["init"]["candidate_evaluator"] is None
|
|
assert calls["run_as_job"]["workspace_id"] == "psd-clinical"
|
|
assert calls["run_as_job"]["input_fingerprint"] != calls["run_as_job"]["config_fingerprint"]
|
|
|
|
|
|
@pytest.mark.parametrize("schema_version, source_type", [
|
|
(1, "filesystem"),
|
|
(1, "http"),
|
|
(1, "s3"),
|
|
(2, "http"),
|
|
(2, "s3"),
|
|
])
|
|
def test_candidate_evaluation_is_not_required_outside_v2_filesystem_corpora(
|
|
schema_version, source_type,
|
|
):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
cfg = SimpleNamespace(
|
|
evidence=SimpleNamespace(
|
|
schema_version=schema_version,
|
|
source_root=None,
|
|
sources=[SimpleNamespace(type=source_type)],
|
|
),
|
|
language="it",
|
|
)
|
|
|
|
assert command._candidate_evaluator(cfg, vector_store=object(), embedder=object()) is None
|
|
|
|
|
|
def test_preprocess_evidence_gc_json_is_pristine(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
monkeypatch.setattr(command, "gc_from_config", lambda *a, **k: {
|
|
"status": "succeeded",
|
|
"dry_run": True,
|
|
"evicted": [],
|
|
"failures": [],
|
|
})
|
|
response = CliRunner().invoke(
|
|
app,
|
|
["preprocess", "evidence", "gc", "--dry-run", "--json", "-c", str(config)],
|
|
)
|
|
assert response.exit_code == 0, response.output
|
|
assert json.loads(response.output)["dry_run"] is True
|