Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
250 lines
12 KiB
Python
250 lines
12 KiB
Python
import os
|
|
import shutil
|
|
from pathlib import Path
|
|
from uuid import uuid4
|
|
|
|
import pytest
|
|
import requests
|
|
from testcontainers.core.container import DockerContainer
|
|
from testcontainers.core.waiting_utils import wait_for_logs
|
|
|
|
from tht.adapters.vector.qdrant import QdrantVectorStore
|
|
from tht.evidence.adapters import FilesystemEvidenceSource
|
|
from tht.evidence.canonical import CuratedEvidence, dump_curated_markdown
|
|
from tht.evidence.corpus.chunk import ChunkPolicy
|
|
from tht.evidence.corpus.pipeline import CorpusPipeline
|
|
from tht.evidence.corpus.store import CorpusStore
|
|
from tht.evidence.local_archive import LocalEvidenceArchive
|
|
from tht.evidence.search import ActiveEvidenceSearcher, EvidenceSearchContext, search_evidence
|
|
from tht.ports.vector import VectorWriteRecord
|
|
from tht.vectorstore.records import VectorRecord
|
|
|
|
pytestmark = pytest.mark.l0
|
|
|
|
|
|
class Embeddings:
|
|
def embed_documents(self, texts):
|
|
return [[1.0, 0.0, 0.0] for _ in texts]
|
|
|
|
def embed_query(self, query):
|
|
return [1.0, 0.0, 0.0]
|
|
|
|
|
|
def test_editable_archive_reindexes_visible_edits_for_core_and_preserves_other_indexes(tmp_path, monkeypatch):
|
|
with DockerContainer("qdrant/qdrant:v1.18.2").with_exposed_ports(6333) as container:
|
|
wait_for_logs(container, "Qdrant HTTP listening on 6333")
|
|
url = f"http://{container.get_container_host_ip()}:{container.get_exposed_port(6333)}"
|
|
workspace = "editable-" + uuid4().hex
|
|
collections = {key: workspace + "-" + key for key in ("reference", "memory")}
|
|
for collection in collections.values():
|
|
requests.put(
|
|
f"{url}/collections/{collection}",
|
|
json={
|
|
"vectors": {"size": 3, "distance": "Cosine"},
|
|
"sparse_vectors": {"bm25": {"modifier": "idf"}},
|
|
},
|
|
timeout=10,
|
|
).raise_for_status()
|
|
vectors = QdrantVectorStore(
|
|
base_url=url, collections=collections, workspace_id=workspace, expected_dimension=3
|
|
)
|
|
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
|
vectors.upsert(
|
|
collection,
|
|
[
|
|
VectorWriteRecord(
|
|
record=VectorRecord(
|
|
id=kind,
|
|
ref=kind,
|
|
kind=kind,
|
|
title="Preserve",
|
|
content="Unrelated content",
|
|
),
|
|
embedding=[1, 0, 0],
|
|
content_hash="sha256:" + "a" * 64,
|
|
)
|
|
],
|
|
)
|
|
corpus = CorpusStore(tmp_path / "corpus")
|
|
archive = LocalEvidenceArchive(tmp_path / "workspace")
|
|
path = archive.evidence / "curated/domain/order-key.md"
|
|
path.parent.mkdir(parents=True)
|
|
unit = CuratedEvidence.model_validate(
|
|
{
|
|
"schema_version": 4,
|
|
"id": "evidence:order-key",
|
|
"title": "Order key",
|
|
"kind": "domain",
|
|
"purposes": ["sql_generation"],
|
|
"language": "en",
|
|
"provenance": {"kind": "manual", "declared_by": "Alice"},
|
|
"payload": {"rule": "Join orders using the complete business key."},
|
|
}
|
|
)
|
|
path.write_text(dump_curated_markdown(unit))
|
|
|
|
def activate(snapshot):
|
|
result = CorpusPipeline(
|
|
store=corpus,
|
|
sources=[FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))],
|
|
embedder=Embeddings(),
|
|
vector_store=vectors,
|
|
embedding_model="fixture",
|
|
embedding_dimensions=3,
|
|
chunk_policy=ChunkPolicy(version="semantic:v1", max_chars=5000),
|
|
pipeline_version="editable-v4",
|
|
workspace_id=workspace,
|
|
sparse_language="english",
|
|
).run()
|
|
assert result.status == "succeeded", (result, result.review_items)
|
|
|
|
class Delegate:
|
|
def search(self, embedding, top_n=10, kinds=None, **kwargs):
|
|
return vectors.search(["evidence"], embedding, limit=top_n, kinds=kinds, **kwargs)
|
|
|
|
searcher = ActiveEvidenceSearcher(corpus, Delegate(), workspace, "english")
|
|
|
|
def lookup():
|
|
result = search_evidence(
|
|
"order key",
|
|
"sql_generation",
|
|
EvidenceSearchContext(),
|
|
searcher=searcher,
|
|
embedder=Embeddings(),
|
|
)
|
|
assert result.status == "available"
|
|
return result.results
|
|
|
|
archive.consolidate(actor="Alice", activate=activate)
|
|
assert lookup()[0].evidence_id == unit.id
|
|
assert lookup()[0].provenance["kind"] == "manual"
|
|
path.write_text(
|
|
path.read_text().replace(
|
|
"complete business key", "order number, financial year and company"
|
|
)
|
|
)
|
|
assert "financial year" not in " ".join(lookup()[0].excerpts)
|
|
archive.consolidate(actor="Bob", activate=activate)
|
|
assert "financial year" in " ".join(lookup()[0].excerpts)
|
|
assert lookup()[0].provenance["declared_by"] == "Bob"
|
|
previous_snapshot = archive.active_snapshot()
|
|
valid = path.read_text()
|
|
path.write_text(
|
|
valid.replace(
|
|
"Join orders using the order number, financial year and company.", "x" * 6000
|
|
)
|
|
)
|
|
with pytest.raises(AssertionError, match="blocked"):
|
|
archive.consolidate(actor="Bob", activate=activate)
|
|
assert archive.active_snapshot() == previous_snapshot
|
|
assert "financial year" in " ".join(lookup()[0].excerpts)
|
|
path.write_text(valid)
|
|
archive.consolidate(actor="Bob", activate=activate)
|
|
path.unlink()
|
|
archive.consolidate(actor="Bob", activate=activate)
|
|
assert lookup() == ()
|
|
# Optional real corpus probe, always copied into the test's private archive.
|
|
if supplied := os.environ.get("THT_E1_PSD_COPY"):
|
|
from tht.evidence.canonical import load_curated_tree
|
|
|
|
source_root = Path(supplied) / "evidence"
|
|
expected = {u.id for u in load_curated_tree(source_root / "curated")}
|
|
assert len(expected) == 35
|
|
shutil.copytree(
|
|
source_root / "curated", archive.evidence / "curated", dirs_exist_ok=True
|
|
)
|
|
shutil.copytree(source_root / "source", archive.evidence / "source", dirs_exist_ok=True)
|
|
archive.consolidate(actor="Validation", activate=activate)
|
|
actual = {
|
|
d.metadata["curated_evidence"]["id"] for d in corpus.active_manifest().documents
|
|
}
|
|
assert actual == expected
|
|
print("PSD v4: all 35 converted Evidence units indexed with unchanged identities")
|
|
# Exercise the installed harness entry point against the same real Qdrant.
|
|
import json
|
|
|
|
import yaml
|
|
from typer.testing import CliRunner
|
|
|
|
from tht.cli import app
|
|
from tht.cli.preprocess_cmd import clear_from_config, run_from_config
|
|
runtime = tmp_path / "runtime.yaml"
|
|
runtime.write_text(yaml.safe_dump({
|
|
"runtime_identity": {"workspace_id": workspace, "workspace_revision": "a" * 40},
|
|
"dwh": {"type": "postgres_direct", "connection": {"database": "unused", "schema": "public", "user": "unused", "password": "unused"}},
|
|
"vectors": {"type": "qdrant", "base_url": "http://qdrant:6333", "collection": workspace},
|
|
"embeddings": {"provider": "ollama_internal", "base_url": "http://embedding:11434", "model": "fixture", "dim": 3},
|
|
"evidence": {"schema_version": 2, "local_archive_root": str(archive.root), "sources": [{"type": "http", "urls": ["https://must-not-be-fetched.invalid/source.md"]}]},
|
|
"vector": {"max_chunk_chars": 5000},
|
|
"roots": {"sessions": str(tmp_path / "sessions"), "artifacts": str(tmp_path / "artifacts"), "indexes": str(tmp_path / "indexes")},
|
|
}))
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *a, **kw: vectors)
|
|
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: Embeddings())
|
|
monkeypatch.setenv("THT_PRINCIPAL_SUBJECT", "Installed curator")
|
|
path.write_text(dump_curated_markdown(unit))
|
|
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
|
assert result.exit_code == 0, result.output
|
|
assert json.loads(result.stdout)["status"] == "succeeded"
|
|
assert any(hit.provenance.get("declared_by") == "Installed curator" for hit in lookup())
|
|
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
|
assert vectors.existing_hashes(collection, [kind])
|
|
primary = path.read_bytes()
|
|
monkeypatch.setenv("THT_PROFILE", "server")
|
|
clear_from_config(runtime)
|
|
assert path.read_bytes() == primary
|
|
assert archive.active_snapshot().is_dir()
|
|
# Full preprocessing must recreate Reference after Clear; Evidence alone
|
|
# cannot claim the missing Schema derivatives are ready.
|
|
requests.put(f"{url}/collections/{collections['reference']}", json={
|
|
"vectors": {"size": 3, "distance": "Cosine"},
|
|
"sparse_vectors": {"bm25": {"modifier": "idf"}},
|
|
}, timeout=10).raise_for_status()
|
|
assert run_from_config(runtime).status == "succeeded"
|
|
assert any(hit.evidence_id == unit.id for hit in lookup())
|
|
|
|
# E3 traverses the public source commands and the same real index activation.
|
|
from test_evidence_imports import Refiner, Remote
|
|
|
|
from tht.evidence.imports import reviews
|
|
vectors.upsert("schema_records", [VectorWriteRecord(record=VectorRecord(
|
|
id="schema_table", ref="schema_table", kind="schema_table", title="Preserve",
|
|
content="Unrelated schema"), embedding=[1, 0, 0], content_hash="sha256:" + "a" * 64)])
|
|
remote, refiner = Remote(), Refiner()
|
|
monkeypatch.setattr("tht.evidence.imports.acquisition_sources", lambda cfg: [remote])
|
|
monkeypatch.setattr("tht.evidence.authoring.PiEvidenceRestructurer", lambda *a, **kw: refiner)
|
|
|
|
def source_command(*args):
|
|
outcome = CliRunner().invoke(app, ["evidence", "sources", *args, "--json", "-c", str(runtime)])
|
|
assert outcome.exit_code == 0, outcome.output
|
|
return json.loads(outcome.stdout)
|
|
|
|
source_command("refresh")
|
|
row = reviews(archive)[0]
|
|
imported_id = row["proposed"][0]["id"]
|
|
assert not any(hit.evidence_id == imported_id for hit in lookup())
|
|
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
|
assert any(hit.evidence_id == imported_id for hit in lookup())
|
|
imported = archive.evidence / f"curated/domain/{imported_id[9:]}.md"
|
|
imported.write_text(imported.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nKeep the curator's company key."))
|
|
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
|
assert result.exit_code == 0, result.output
|
|
remote.text = "The refreshed source has another key."
|
|
source_command("refresh")
|
|
assert any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
|
row = reviews(archive)[0]
|
|
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
|
assert any("another key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
|
assert not any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
|
imported.unlink()
|
|
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
|
assert result.exit_code == 0, result.output
|
|
remote.text = "Do not regenerate the retired source rule."
|
|
source_command("refresh")
|
|
row = reviews(archive)[0]
|
|
assert not row["proposed"]
|
|
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
|
assert not any(hit.evidence_id == imported_id for hit in lookup())
|
|
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
|
assert vectors.existing_hashes(collection, [kind])
|
|
assert vectors.existing_hashes("memory", ["memory"])
|