import os import shutil from pathlib import Path from uuid import uuid4 import pytest import requests from testcontainers.core.container import DockerContainer from testcontainers.core.waiting_utils import wait_for_logs from tht.adapters.vector.qdrant import QdrantVectorStore from tht.evidence.adapters import FilesystemEvidenceSource from tht.evidence.canonical import CuratedEvidence, dump_curated_markdown from tht.evidence.corpus.chunk import ChunkPolicy from tht.evidence.corpus.pipeline import CorpusPipeline from tht.evidence.corpus.store import CorpusStore from tht.evidence.local_archive import LocalEvidenceArchive from tht.evidence.search import ActiveEvidenceSearcher, EvidenceSearchContext, search_evidence from tht.ports.vector import VectorWriteRecord from tht.vectorstore.records import VectorRecord pytestmark = pytest.mark.l0 class Embeddings: def embed_documents(self, texts): return [[1.0, 0.0, 0.0] for _ in texts] def embed_query(self, query): return [1.0, 0.0, 0.0] def test_editable_archive_reindexes_visible_edits_for_core_and_preserves_other_indexes(tmp_path, monkeypatch): with DockerContainer("qdrant/qdrant:v1.18.2").with_exposed_ports(6333) as container: wait_for_logs(container, "Qdrant HTTP listening on 6333") url = f"http://{container.get_container_host_ip()}:{container.get_exposed_port(6333)}" workspace = "editable-" + uuid4().hex collections = {key: workspace + "-" + key for key in ("reference", "memory")} for collection in collections.values(): requests.put( f"{url}/collections/{collection}", json={ "vectors": {"size": 3, "distance": "Cosine"}, "sparse_vectors": {"bm25": {"modifier": "idf"}}, }, timeout=10, ).raise_for_status() vectors = QdrantVectorStore( base_url=url, collections=collections, workspace_id=workspace, expected_dimension=3 ) for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]: vectors.upsert( collection, [ VectorWriteRecord( record=VectorRecord( id=kind, ref=kind, kind=kind, title="Preserve", content="Unrelated content", ), embedding=[1, 0, 0], content_hash="sha256:" + "a" * 64, ) ], ) corpus = CorpusStore(tmp_path / "corpus") archive = LocalEvidenceArchive(tmp_path / "workspace") path = archive.evidence / "curated/domain/order-key.md" path.parent.mkdir(parents=True) unit = CuratedEvidence.model_validate( { "schema_version": 4, "id": "evidence:order-key", "title": "Order key", "kind": "domain", "purposes": ["sql_generation"], "language": "en", "provenance": {"kind": "manual", "declared_by": "Alice"}, "payload": {"rule": "Join orders using the complete business key."}, } ) path.write_text(dump_curated_markdown(unit)) def activate(snapshot): result = CorpusPipeline( store=corpus, sources=[FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))], embedder=Embeddings(), vector_store=vectors, embedding_model="fixture", embedding_dimensions=3, chunk_policy=ChunkPolicy(version="semantic:v1", max_chars=5000), pipeline_version="editable-v4", workspace_id=workspace, sparse_language="english", ).run() assert result.status == "succeeded", (result, result.review_items) class Delegate: def search(self, embedding, top_n=10, kinds=None, **kwargs): return vectors.search(["evidence"], embedding, limit=top_n, kinds=kinds, **kwargs) searcher = ActiveEvidenceSearcher(corpus, Delegate(), workspace, "english") def lookup(): result = search_evidence( "order key", "sql_generation", EvidenceSearchContext(), searcher=searcher, embedder=Embeddings(), ) assert result.status == "available" return result.results archive.consolidate(actor="Alice", activate=activate) assert lookup()[0].evidence_id == unit.id assert lookup()[0].provenance["kind"] == "manual" path.write_text( path.read_text().replace( "complete business key", "order number, financial year and company" ) ) assert "financial year" not in " ".join(lookup()[0].excerpts) archive.consolidate(actor="Bob", activate=activate) assert "financial year" in " ".join(lookup()[0].excerpts) assert lookup()[0].provenance["declared_by"] == "Bob" previous_snapshot = archive.active_snapshot() valid = path.read_text() path.write_text( valid.replace( "Join orders using the order number, financial year and company.", "x" * 6000 ) ) with pytest.raises(AssertionError, match="blocked"): archive.consolidate(actor="Bob", activate=activate) assert archive.active_snapshot() == previous_snapshot assert "financial year" in " ".join(lookup()[0].excerpts) path.write_text(valid) archive.consolidate(actor="Bob", activate=activate) path.unlink() archive.consolidate(actor="Bob", activate=activate) assert lookup() == () # Optional real corpus probe, always copied into the test's private archive. if supplied := os.environ.get("THT_E1_PSD_COPY"): from tht.evidence.canonical import load_curated_tree source_root = Path(supplied) / "evidence" expected = {u.id for u in load_curated_tree(source_root / "curated")} assert len(expected) == 35 shutil.copytree( source_root / "curated", archive.evidence / "curated", dirs_exist_ok=True ) shutil.copytree(source_root / "source", archive.evidence / "source", dirs_exist_ok=True) archive.consolidate(actor="Validation", activate=activate) actual = { d.metadata["curated_evidence"]["id"] for d in corpus.active_manifest().documents } assert actual == expected print("PSD v4: all 35 converted Evidence units indexed with unchanged identities") # Exercise the installed harness entry point against the same real Qdrant. import json import yaml from typer.testing import CliRunner from tht.cli import app from tht.cli.preprocess_cmd import clear_from_config, run_from_config runtime = tmp_path / "runtime.yaml" runtime.write_text(yaml.safe_dump({ "runtime_identity": {"workspace_id": workspace, "workspace_revision": "a" * 40}, "dwh": {"type": "postgres_direct", "connection": {"database": "unused", "schema": "public", "user": "unused", "password": "unused"}}, "vectors": {"type": "qdrant", "base_url": "http://qdrant:6333", "collection": workspace}, "embeddings": {"provider": "ollama_internal", "base_url": "http://embedding:11434", "model": "fixture", "dim": 3}, "evidence": {"schema_version": 2, "local_archive_root": str(archive.root), "sources": [{"type": "http", "urls": ["https://must-not-be-fetched.invalid/source.md"]}]}, "vector": {"max_chunk_chars": 5000}, "roots": {"sessions": str(tmp_path / "sessions"), "artifacts": str(tmp_path / "artifacts"), "indexes": str(tmp_path / "indexes")}, })) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *a, **kw: vectors) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: Embeddings()) monkeypatch.setenv("THT_PRINCIPAL_SUBJECT", "Installed curator") path.write_text(dump_curated_markdown(unit)) result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)]) assert result.exit_code == 0, result.output assert json.loads(result.stdout)["status"] == "succeeded" assert any(hit.provenance.get("declared_by") == "Installed curator" for hit in lookup()) for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]: assert vectors.existing_hashes(collection, [kind]) primary = path.read_bytes() monkeypatch.setenv("THT_PROFILE", "server") clear_from_config(runtime) assert path.read_bytes() == primary assert archive.active_snapshot().is_dir() # Full preprocessing must recreate Reference after Clear; Evidence alone # cannot claim the missing Schema derivatives are ready. requests.put(f"{url}/collections/{collections['reference']}", json={ "vectors": {"size": 3, "distance": "Cosine"}, "sparse_vectors": {"bm25": {"modifier": "idf"}}, }, timeout=10).raise_for_status() assert run_from_config(runtime).status == "succeeded" assert any(hit.evidence_id == unit.id for hit in lookup()) # E3 traverses the public source commands and the same real index activation. from test_evidence_imports import Refiner, Remote from tht.evidence.imports import reviews vectors.upsert("schema_records", [VectorWriteRecord(record=VectorRecord( id="schema_table", ref="schema_table", kind="schema_table", title="Preserve", content="Unrelated schema"), embedding=[1, 0, 0], content_hash="sha256:" + "a" * 64)]) remote, refiner = Remote(), Refiner() monkeypatch.setattr("tht.evidence.imports.acquisition_sources", lambda cfg: [remote]) monkeypatch.setattr("tht.evidence.authoring.PiEvidenceRestructurer", lambda *a, **kw: refiner) def source_command(*args): outcome = CliRunner().invoke(app, ["evidence", "sources", *args, "--json", "-c", str(runtime)]) assert outcome.exit_code == 0, outcome.output return json.loads(outcome.stdout) source_command("refresh") row = reviews(archive)[0] imported_id = row["proposed"][0]["id"] assert not any(hit.evidence_id == imported_id for hit in lookup()) source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace") assert any(hit.evidence_id == imported_id for hit in lookup()) imported = archive.evidence / f"curated/domain/{imported_id[9:]}.md" imported.write_text(imported.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nKeep the curator's company key.")) result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)]) assert result.exit_code == 0, result.output remote.text = "The refreshed source has another key." source_command("refresh") assert any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id) row = reviews(archive)[0] source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace") assert any("another key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id) assert not any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id) imported.unlink() result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)]) assert result.exit_code == 0, result.output remote.text = "Do not regenerate the retired source rule." source_command("refresh") row = reviews(archive)[0] assert not row["proposed"] source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace") assert not any(hit.evidence_id == imported_id for hit in lookup()) for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]: assert vectors.existing_hashes(collection, [kind]) assert vectors.existing_hashes("memory", ["memory"])