import json from pathlib import Path from types import SimpleNamespace import pytest from typer.testing import CliRunner from tht.cli import app def _runtime_config( tmp_path: Path, name: str = "workspace.yaml", evidence_schema_version: int | None = None, ) -> Path: path = tmp_path / name (tmp_path / "evidence").mkdir(exist_ok=True) evidence_version = "" if evidence_schema_version is None else f"\n schema_version: {evidence_schema_version}" path.write_text( f""" runtime_identity: workspace_id: psd-clinical workspace_revision: {'a' * 40} dwh: type: postgres_direct connection: {{database: analytics, schema: mart, user: reader, password: secret}} vectors: type: qdrant base_url: http://qdrant:6333 collection: psd-clinical embeddings: provider: ollama_internal base_url: http://embedding:11434 model: qwen3-embedding:0.6b dim: 1024 evidence: {evidence_version} sources: - type: filesystem root: {tmp_path / 'evidence'} roots: sessions: {tmp_path / 'sessions'} artifacts: {tmp_path / 'artifacts'} indexes: {tmp_path / 'indexes'} """ ) return path def test_v2_invalid_materialized_corpus_stops_before_vector_store_construction(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command from tht.evidence import ValidationFinding, ValidationReport config = _runtime_config(tmp_path, evidence_schema_version=2) vector_store_constructed = False monkeypatch.setattr( "tht.evidence.validate_workspace_evidence", lambda root: ValidationReport((ValidationFinding( "error", "manifest_missing", "manifest.yaml", "missing", ),)), ) def forbidden_vector_store(*args, **kwargs): nonlocal vector_store_constructed vector_store_constructed = True raise AssertionError("invalid corpus must not reach vector upsert setup") monkeypatch.setattr("tht.adapters.factory.build_vector_store", forbidden_vector_store) with pytest.raises(RuntimeError, match="curated Evidence corpus is invalid"): command.run_from_config(config) assert vector_store_constructed is False def test_v2_valid_materialized_corpus_is_validated_before_preprocessing(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command from tht.evidence import ValidationReport config = _runtime_config(tmp_path, evidence_schema_version=2) calls = [] class FakePipeline: def run_as_job(self, **kwargs): calls.append(("run", kwargs)) return "completed" monkeypatch.setattr("tht.evidence.validate_workspace_evidence", lambda root: calls.append(("validate", root)) or ValidationReport(())) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: calls.append(("vector", require_write)) or object()) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object()) monkeypatch.setattr("tht.evidence.build_sources", lambda evidence: []) monkeypatch.setattr("tht.evidence.build_preprocessing_pipeline", lambda **kwargs: FakePipeline()) assert command.run_from_config(config) == "completed" assert calls[:2] == [("validate", tmp_path), ("vector", True)] def test_clear_removes_only_derived_paths_and_preserves_memory(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command monkeypatch.setenv("THT_PROFILE", "server") config = _runtime_config(tmp_path) for path in ( tmp_path / ".tht-dwh", tmp_path / ".tht-jobs", tmp_path / "corpus", tmp_path / "indexes" / "lsh", ): path.mkdir(parents=True, exist_ok=True) (path / "derived").write_text("generated", encoding="utf-8") physical = tmp_path / "artifacts" / "mschema" / "physical.yaml" physical.parent.mkdir(parents=True) physical.write_text("generated", encoding="utf-8") memory = tmp_path / "memory" / "registry.jsonl" memory.parent.mkdir() memory.write_text("durable", encoding="utf-8") class Store: def clear_reference(self): return True monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *_args, **_kwargs: Store()) counts = command.clear_from_config(config) assert counts == {"referenceCollections": 1, "derivedPaths": 5} assert memory.read_text(encoding="utf-8") == "durable" assert not (tmp_path / ".tht-dwh").exists() assert not (tmp_path / "indexes" / "lsh").exists() def test_preprocess_evidence_json_is_pristine(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) result = SimpleNamespace( model_dump=lambda mode=None: { "status": "succeeded", "generation": "gen:abc", "published": True, "counts": {"changed": 0, "unchanged": 0, "removed": 0, "documents": 0, "chunks": 0}, "changed": [], "unchanged": [], "removed": [], "manifest_id": "manifest-1", "run_id": "a" * 32, "resumed_from": None, } ) monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)]) assert response.exit_code == 0, response.output assert response.stderr == "" assert json.loads(response.stdout) == { "changed": [], "code": "ok", "counts": {"changed": 0, "chunks": 0, "documents": 0, "removed": 0, "unchanged": 0}, "generation": "gen:abc", "manifest_id": "manifest-1", "operation": "preprocess_evidence", "published": True, "removed": [], "resumed_from": None, "run_id": "a" * 32, "schemaVersion": 1, "status": "succeeded", "unchanged": [], "workspaceId": "psd-clinical", "workspaceRevision": "a" * 40, } def test_preprocess_failure_is_structured_and_nonzero(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) monkeypatch.setattr( command, "run_from_config", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail")), ) response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)]) assert response.exit_code != 0 payload = json.loads(response.stdout) assert payload == { "code": "preprocessing_failed", "error": "preprocessing failed", "operation": "preprocess_evidence", "schemaVersion": 1, "status": "failed", "workspaceId": "psd-clinical", "workspaceRevision": "a" * 40, } assert "secret detail" not in response.output def test_preprocess_failed_job_report_is_sanitized_json_and_nonzero(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) result = SimpleNamespace( model_dump=lambda mode=None: { "status": "failed", "run_id": "a" * 32, "published": False, "generation": "gen:" + "b" * 32, "changed": ["fs:one"], "unchanged": [], "removed": [], "counts": {"changed": 1, "unchanged": 0, "removed": 0, "documents": 1, "chunks": 1}, "manifest_id": "manifest-1", "resumed_from": None, } ) monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)]) assert response.exit_code == 1 payload = json.loads(response.stdout) assert payload["status"] == "failed" assert payload["error"] == "preprocessing job failed" assert payload["workspaceId"] == "psd-clinical" assert "traceback" not in response.output.lower() def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path): from test_corpus_pipeline import Source, item, pipeline import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) result = pipeline( tmp_path, Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))]), ).run_as_job( workspace_id="demo", workspace_root=tmp_path, config_fingerprint="sha256:" + "1" * 64, input_fingerprint="sha256:" + "2" * 64, ) assert result.status == "failed" monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)]) assert response.exit_code == 1 assert json.loads(response.stdout)["status"] == "failed" assert "SENSITIVE EVIDENCE" not in response.output assert "secret" not in response.output def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) result = SimpleNamespace( model_dump=lambda mode=None: { "status": "succeeded", "run_id": "a" * 32, "generation": "gen:" + "b" * 64, "published": True, "changed": ["fs:item"] * 100, "unchanged": ["fs:item"] * 100, "removed": ["fs:item"] * 100, "counts": {"changed": 1001, "unchanged": 902, "removed": 803}, "manifest_id": "manifest-1", "resumed_from": None, } ) monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) response = CliRunner().invoke(app, ["preprocess", "evidence", "-c", str(config)]) assert response.exit_code == 0, response.output assert "changed=1001 unchanged=902 removed=803" in response.output def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) called = False def forbidden(*args, **kwargs): nonlocal called called = True monkeypatch.setattr(command, "run_from_config", forbidden) response = CliRunner().invoke( app, [ "preprocess", "evidence", "--resume", "gen:" + "a" * 32, "--json", "-c", str(config), ], ) assert response.exit_code != 0 assert json.loads(response.output) == { "code": "invalid_resume", "error": "resume requires a preprocessing run id", "operation": "preprocess_evidence", "schemaVersion": 1, "status": "failed", } assert called is False def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path, name="3") calls = {} class FakePipeline: def __init__( self, *, store, sources, embedder, vector_store, embedding_id, embedding_model, embedding_dimensions, chunk_policy, pipeline_version, retain_published_generations, sparse_language, candidate_evaluator, ): calls["init"] = { "embedding_id": embedding_id, "embedding_model": embedding_model, "embedding_dimensions": embedding_dimensions, "pipeline_version": pipeline_version, "sparse_language": sparse_language, "candidate_evaluator": candidate_evaluator, } def run_as_job(self, **kwargs): calls["run_as_job"] = kwargs return SimpleNamespace(model_dump=lambda mode=None: {"status": "succeeded"}) monkeypatch.setattr("tht.evidence.build_sources", lambda cfg: []) monkeypatch.setattr( "tht.evidence.build_preprocessing_pipeline", lambda **kwargs: FakePipeline(**kwargs), ) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: object()) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object()) command.run_from_config(config) assert calls["init"]["embedding_id"] == "ollama/qwen3-embedding:0.6b" assert calls["init"]["sparse_language"] == "english" assert calls["init"]["candidate_evaluator"] is None assert calls["run_as_job"]["workspace_id"] == "psd-clinical" assert calls["run_as_job"]["input_fingerprint"] != calls["run_as_job"]["config_fingerprint"] @pytest.mark.parametrize("schema_version, source_type", [ (1, "filesystem"), (1, "http"), (1, "s3"), (2, "http"), (2, "s3"), ]) def test_candidate_evaluation_is_not_required_outside_v2_filesystem_corpora( schema_version, source_type, ): import tht.cli.preprocess_cmd as command cfg = SimpleNamespace( evidence=SimpleNamespace( schema_version=schema_version, source_root=None, sources=[SimpleNamespace(type=source_type)], ), language="it", ) assert command._candidate_evaluator(cfg, vector_store=object(), embedder=object()) is None def test_preprocess_evidence_gc_json_is_pristine(monkeypatch, tmp_path): import tht.cli.preprocess_cmd as command config = _runtime_config(tmp_path) monkeypatch.setattr(command, "gc_from_config", lambda *a, **k: { "status": "succeeded", "dry_run": True, "evicted": [], "failures": [], }) response = CliRunner().invoke( app, ["preprocess", "evidence", "gc", "--dry-run", "--json", "-c", str(config)], ) assert response.exit_code == 0, response.output assert json.loads(response.output)["dry_run"] is True