refactor: remove pgvector runtime

This commit is contained in:
2026-08-08 21:38:03 +02:00
parent 5c12d9bb79
commit 8826f8ac6b
34 changed files with 200 additions and 3362 deletions
@@ -1,258 +0,0 @@
"""L0 gate for the complete durable Evidence/pgvector lifecycle."""
import hashlib
import json
import pytest
from sqlalchemy import create_engine, text
from testcontainers.postgres import PostgresContainer
from tht.adapters.evidence import FilesystemEvidenceSource
from tht.adapters.vector.pgvector import PgVectorStore
from tht.cli.vector_migrate_cmd import migrate
from tht.config import DatabaseConfig
from tht.corpus.chunk import ChunkPolicy
from tht.corpus.pipeline import CorpusPipeline
from tht.corpus.store import CorpusStore
from tht.ports.vector import VectorRecord, VectorWriteRecord
from tht.search import combined_search
from tht.search.evidence import ActiveEvidenceSearcher, resolve_evidence_file
DIMENSIONS = 768
class DeterministicEmbedder:
def embed_documents(self, texts):
return [self.embed_query(text) for text in texts]
def embed_query(self, text):
vector = [0.0] * DIMENSIONS
vector[0] = 0.8
vector[1] = 0.6
return vector
class EvidenceDelegate:
"""Adapt the real multi-collection port to the runtime search protocol."""
def __init__(self, store):
self.store = store
def search(self, embedding, top_n=10, kinds=None, metadata_filter=None):
return self.store.search(
["evidence"], embedding, limit=top_n, kinds=kinds,
metadata_filter=metadata_filter,
)
class InterruptAfterRealPartialUpsert:
"""Crash after a committed real row, as a process death would."""
def __init__(self, store):
self.store = store
self.interrupt = True
def __getattr__(self, name):
return getattr(self.store, name)
def upsert(self, collection, records):
if self.interrupt and len(records) > 1:
self.interrupt = False
self.store.upsert(collection, records[:1])
raise KeyboardInterrupt("injected process death after committed vector row")
return self.store.upsert(collection, records)
@pytest.fixture(scope="module")
def persistent_pgvector():
with PostgresContainer("pgvector/pgvector:pg16") as postgres:
migrate(postgres.get_connection_url())
admin = create_engine(postgres.get_connection_url())
with admin.begin() as connection:
connection.exec_driver_sql(
"ALTER ROLE vector_reader LOGIN PASSWORD 'reader-lifecycle'"
)
connection.exec_driver_sql(
"ALTER ROLE vector_writer LOGIN PASSWORD 'writer-lifecycle'"
)
url = admin.url
common = dict(
host=url.host, port=url.port, database=url.database, schema="vectors"
)
reader = DatabaseConfig(
**common, user="vector_reader", password="reader-lifecycle"
)
writer = DatabaseConfig(
**common, user="vector_writer", password="writer-lifecycle"
)
yield postgres, admin, reader, writer
admin.dispose()
def _pipeline(root, source_root, vectors):
return CorpusPipeline(
store=CorpusStore(root / "corpus"),
sources=[FilesystemEvidenceSource(source_root)],
embedder=DeterministicEmbedder(),
vector_store=vectors,
embedding_model="deterministic-l0",
embedding_dimensions=DIMENSIONS,
chunk_policy=ChunkPolicy(version="lifecycle-v1", max_chars=48),
pipeline_version="evidence-v1",
retain_published_generations=2,
)
def _publish(pipeline, root, serial):
return pipeline.run_as_job(
workspace_id="pgvector-lifecycle",
workspace_root=root,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + f"{serial:x}" * 64,
)
@pytest.mark.l0
def test_real_pgvector_corpus_job_lifecycle(tmp_path, persistent_pgvector):
postgres, admin, reader_config, writer_config = persistent_pgvector
source_root = tmp_path / "sources"
source_root.mkdir()
kept = source_root / "kept.md"
stable = source_root / "stable.md"
stable.write_text("unchanged dependency evidence", encoding="utf-8")
removed = source_root / "removed.md"
removed.write_text("removed evidence generation zero", encoding="utf-8")
vectors = PgVectorStore(reader_config, writer_config, expected_dimension=DIMENSIONS)
generations = []
for serial in range(3):
kept.write_text(f"active evidence generation {serial}", encoding="utf-8")
result = _publish(_pipeline(tmp_path, source_root, vectors), tmp_path, serial + 1)
assert result.status == "succeeded"
generations.append(result.generation)
removed_document = next(
doc for doc in CorpusStore(tmp_path / "corpus").active_manifest().documents
if "removed.md" in doc.source_uri
)
removed_document_id = removed_document.document_id
removed_ref = removed_document.document_id
# A stale, closer row must not consume LIMIT before ACTIVE filtering.
stale_generation = generations[-2]
stale = VectorWriteRecord(
record=VectorRecord(
id="chunk:stale-perfect-match", kind="evidence", ref="doc:stale",
title="stale forbidden", content="stale forbidden",
metadata={
"document_id": "doc:stale",
"vector_generation": stale_generation,
},
),
embedding=[1.0] + [0.0] * (DIMENSIONS - 1),
content_hash="sha256:" + "a" * 64,
)
vectors.upsert("evidence", [stale])
runtime = ActiveEvidenceSearcher(CorpusStore(tmp_path / "corpus"), EvidenceDelegate(vectors))
query = DeterministicEmbedder().embed_query("active")
hits = runtime.search(query, top_n=1, kinds=["evidence"])
assert len(hits) == 1 and hits[0].title != "stale forbidden"
packed = combined_search(
"active", lsh_hits=None, store=runtime, embedder=DeterministicEmbedder(),
top=1, rrf_k=60, kinds=["evidence"], query_vec=query,
)
assert len(packed) == 1 and packed[0].label != "stale forbidden"
# Fourth publication removes a document and creates multiple chunks for crash recovery.
removed.unlink()
kept.write_text("active fourth generation " * 8, encoding="utf-8")
crashing = InterruptAfterRealPartialUpsert(vectors)
candidate = _pipeline(tmp_path, source_root, crashing)
with pytest.raises(KeyboardInterrupt, match="injected process death"):
_publish(candidate, tmp_path, 4)
runs = tmp_path / ".tht-jobs" / "evidence" / "runs"
crashed_run = max(runs.iterdir(), key=lambda path: path.stat().st_mtime_ns).name
before = vectors.existing_hashes("evidence", ["evidence"])
intent = json.loads(
(runs / crashed_run / "artifacts" / "vector-intent.json").read_text()
)["records"]
already_present = set(intent) & set(before)
assert len(already_present) == 1
resumed = candidate.run_as_job(
workspace_id="pgvector-lifecycle", workspace_root=tmp_path,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + "4" * 64,
resume_run_id=crashed_run,
)
assert resumed.status == "succeeded" and resumed.resumed_from == crashed_run
generations.append(resumed.generation)
after = vectors.existing_hashes("evidence", ["evidence"])
assert {key: after[key] for key in already_present} == {
key: before[key] for key in already_present
}
assert set(intent).issubset(after)
with admin.connect() as connection:
duplicate_count = connection.execute(text(
"SELECT count(*) - count(DISTINCT record_key) FROM vectors.evidence"
)).scalar_one()
assert duplicate_count == 0
runtime = ActiveEvidenceSearcher(CorpusStore(tmp_path / "corpus"), EvidenceDelegate(vectors))
active_hits = runtime.search(query, top_n=20, kinds=["evidence"])
assert active_hits
assert any("active fourth generation" in hit.content for hit in active_hits)
assert all(hit.ref not in {"doc:stale", removed_ref} for hit in active_hits)
assert all(hit.metadata.get("document_id") != removed_document_id for hit in active_hits)
active_pack = combined_search(
"active", lsh_hits=None, store=runtime, embedder=DeterministicEmbedder(),
top=20, rrf_k=60, kinds=["evidence"], query_vec=query,
)
assert active_pack
assert any("active fourth generation" in result.content for result in active_pack)
assert all(removed_document.content not in result.content for result in active_pack)
assert any("unchanged dependency evidence" in result.content for result in active_pack)
manifest = CorpusStore(tmp_path / "corpus").active_manifest()
assert resolve_evidence_file(
CorpusStore(tmp_path / "corpus"), removed_document_id,
materialized_root=tmp_path / "session",
) == ""
active_document = manifest.documents[0]
owned = CorpusStore(tmp_path / "corpus").materialize_document(
active_document.document_id, tmp_path / "session" / "active-evidence.md"
)
assert owned.read_bytes() == active_document.content.encode()
assert hashlib.sha256(owned.read_bytes()).hexdigest() == active_document.content_hash[7:]
# Recreate engines and stores against the same persisted database.
vectors._reader.dispose()
vectors._writer.dispose()
recreated = PgVectorStore(reader_config, writer_config, expected_dimension=DIMENSIONS)
assert recreated.health().ok is True
recreated_hits = ActiveEvidenceSearcher(
CorpusStore(tmp_path / "corpus"), EvidenceDelegate(recreated)
).search(query, top_n=2, kinds=["evidence"])
active_dependencies = set(manifest.metadata["document_generations"].values())
assert recreated_hits
assert all(hit.metadata["vector_generation"] in active_dependencies for hit in recreated_hits)
orphan = "gen:" + "f" * 32
recreated.upsert("evidence", [VectorWriteRecord(
record=VectorRecord(
id="chunk:exact-vector-orphan", kind="evidence", ref="doc:orphan",
title="orphan", content="orphan",
metadata={"document_id": "doc:orphan", "vector_generation": orphan,
"workspace_id": "pgvector-lifecycle"},
),
embedding=query, content_hash="sha256:" + "f" * 64,
)])
final_pipeline = _pipeline(tmp_path, source_root, recreated)
report = final_pipeline.gc(workspace_root=tmp_path)
assert report["evicted"] == [orphan]
expected_fs = set(generations[-2:])
assert set(CorpusStore(tmp_path / "corpus").list_generations()) == expected_fs
expected_vectors = expected_fs | {generations[0]}
assert set(recreated.list_evidence_generations(
"evidence", "pgvector-lifecycle"
)) == expected_vectors
assert final_pipeline.gc(workspace_root=tmp_path)["evicted"] == []
-407
View File
@@ -1,407 +0,0 @@
import pytest
from psycopg2.errors import InsufficientPrivilege
from sqlalchemy import create_engine, text
from sqlalchemy.exc import ProgrammingError
from testcontainers.postgres import PostgresContainer
from tht.adapters.vector.thoth_http import ThothHttpVectorStore
from tht.config import DatabaseConfig
from tht.ports.vector import (
VectorReadUnavailable,
VectorRecord,
VectorStoreError,
VectorWriteRecord,
VectorWriteUnavailable,
)
def _record(content_hash: str, embedding: list[float], *, kind: str = "memory"):
return VectorWriteRecord(
record=VectorRecord(
id=f"record:{content_hash}",
kind=kind,
ref="session:test",
title=content_hash,
content=f"content {content_hash}",
metadata={"content_hash": content_hash},
),
embedding=embedding,
content_hash=content_hash,
)
@pytest.fixture(scope="module")
def vector_configs():
with PostgresContainer("pgvector/pgvector:pg16") as pg:
host = pg.get_container_host_ip()
port = int(pg.get_exposed_port(5432))
admin_config = DatabaseConfig(
host=host,
port=port,
database=pg.dbname,
schema="vectors",
user=pg.username,
password=pg.password,
)
engine = create_engine(pg.get_connection_url())
with engine.begin() as connection:
connection.exec_driver_sql("CREATE SCHEMA vectors")
# Match the co-located Supabase deployment: tables are in vectors, extension in public.
connection.exec_driver_sql("CREATE EXTENSION vector WITH SCHEMA public")
for table in ("schema_records", "evidence", "memory"):
connection.exec_driver_sql(f"""
CREATE TABLE vectors.{table} (
id bigserial PRIMARY KEY,
record_key text UNIQUE NOT NULL,
kind text NOT NULL,
content_hash text NOT NULL,
metadata jsonb NOT NULL,
embedding public.vector(2) NOT NULL,
indexed_at timestamptz NOT NULL DEFAULT now()
)
""")
connection.exec_driver_sql("CREATE ROLE vector_l0_reader LOGIN PASSWORD 'reader'")
connection.exec_driver_sql("CREATE ROLE vector_l0_writer LOGIN PASSWORD 'writer'")
connection.exec_driver_sql(
"CREATE ROLE vector_l0_no_sequence LOGIN PASSWORD 'no_sequence'"
)
connection.exec_driver_sql(
"GRANT USAGE ON SCHEMA vectors TO vector_l0_reader, vector_l0_writer, "
"vector_l0_no_sequence"
)
connection.exec_driver_sql(
"GRANT SELECT ON ALL TABLES IN SCHEMA vectors TO vector_l0_reader"
)
connection.exec_driver_sql(
"GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA vectors TO vector_l0_writer"
)
for table in ("schema_records", "evidence", "memory"):
connection.exec_driver_sql(
f"GRANT INSERT, UPDATE ON vectors.{table} "
"TO vector_l0_writer, vector_l0_no_sequence"
)
if table == "evidence":
connection.exec_driver_sql(
"GRANT DELETE ON vectors.evidence TO vector_l0_writer"
)
connection.exec_driver_sql(
"GRANT SELECT (kind, metadata) ON vectors.evidence TO vector_l0_writer"
)
connection.exec_driver_sql(
f"GRANT SELECT (record_key, kind, content_hash) "
f"ON vectors.{table} TO vector_l0_writer, vector_l0_no_sequence"
)
engine.dispose()
reader_config = admin_config.model_copy(
update={"user": "vector_l0_reader", "password": "reader"}
)
writer_config = admin_config.model_copy(
update={"user": "vector_l0_writer", "password": "writer"}
)
no_sequence_config = admin_config.model_copy(
update={"user": "vector_l0_no_sequence", "password": "no_sequence"}
)
yield admin_config, reader_config, writer_config, no_sequence_config
@pytest.fixture
def store(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
_, reader_config, writer_config, _ = vector_configs
store = PgVectorStore(reader_config, writer_config, expected_dimension=2)
store.upsert("memory", [_record("reset", [0.0, 1.0])])
yield store
def test_pgvector_round_trip_hash_and_upsert(store):
assert store.upsert("memory", [_record("a", [1.0, 0.0])]) == 1
assert store.existing_hashes("memory", ["memory"])["record:a"] == "a"
hits = store.search(["memory"], [1.0, 0.0], limit=5, kinds=["memory"])
assert hits[0].metadata["content_hash"] == "a"
assert hits[0].id == "record:a"
assert store.upsert("memory", [_record("a", [0.8, 0.2])]) == 1
assert store.search(["memory"], [0.8, 0.2], limit=1)[0].id == "record:a"
def test_pgvector_lists_and_deletes_exact_evidence_generation(store):
generation = "gen:" + "a" * 32
value = VectorWriteRecord(
record=VectorRecord(
id="evidence-generation-a", kind="evidence", ref="doc:a", title="a",
content="content", metadata={"vector_generation": generation, "workspace_id": "default"},
),
embedding=[1.0, 0.0], content_hash="sha256:" + "a" * 64,
)
store.upsert("evidence", [value])
assert generation in store.list_evidence_generations("evidence", "default")
assert store.delete_generation("evidence", generation, "default") == 1
assert generation not in store.list_evidence_generations("evidence", "default")
def test_pgvector_generation_cleanup_isolated_between_workspaces(store):
generation = "gen:" + "b" * 32
records = [VectorWriteRecord(
record=VectorRecord(
id=f"evidence-{workspace}", kind="evidence", ref=f"doc:{workspace}",
title=workspace, content=workspace,
metadata={"vector_generation": generation, "workspace_id": workspace},
), embedding=[1.0, 0.0], content_hash="sha256:" + key * 64,
) for workspace, key in (("workspace-a", "b"), ("workspace-b", "c"))]
store.upsert("evidence", records)
assert store.delete_generation("evidence", generation, "workspace-a") == 1
assert generation not in store.list_evidence_generations("evidence", "workspace-a")
assert generation in store.list_evidence_generations("evidence", "workspace-b")
def test_pgvector_search_filters_kinds_before_limit(store):
store.upsert("memory", [_record("solved", [1.0, 0.0], kind="solved_question")])
hits = store.search("memory".split(), [1.0, 0.0], limit=1, kinds=["memory"])
assert len(hits) == 1
assert hits[0].kind == "memory"
def test_pgvector_multi_collection_search_skips_collections_unrelated_to_kinds(store):
store.upsert("evidence", [_record("evidence", [1.0, 0.0], kind="evidence")])
hits = store.search(["evidence", "memory"], [1.0, 0.0], limit=3, kinds=["memory"])
assert hits
assert {hit.kind for hit in hits} == {"memory"}
def test_pgvector_multi_collection_kind_filter_matches_http_adapter(store):
class Reader:
def search_similar(self, collection, embedding, limit, kinds=None):
if collection != "memory" or "memory" not in (kinds or []):
return []
return [
{
"similarity": 1.0,
"metadata": {
"record_key": "record:a",
"kind": "memory",
"ref": "session:test",
"title": "a",
"content": "content a",
"content_hash": "a",
},
}
]
direct = store.search(["evidence", "memory"], [1.0, 0.0], limit=1, kinds=["memory"])
http = ThothHttpVectorStore(Reader(), None).search(
["evidence", "memory"], [1.0, 0.0], limit=1, kinds=["memory"]
)
assert [(hit.id, hit.kind) for hit in direct] == [(hit.id, hit.kind) for hit in http]
def test_pgvector_search_rejects_unknown_kind_globally(store):
with pytest.raises(VectorStoreError, match="Kind not allowed"):
store.search(["memory"], [1.0, 0.0], limit=1, kinds=["unknown"])
@pytest.mark.parametrize("limit", [True, False, 1.0, 0, -1])
def test_pgvector_search_requires_strict_positive_limit(store, limit):
with pytest.raises(ValueError, match="positive integer"):
store.search(["memory"], [1.0, 0.0], limit=limit)
def test_pgvector_allowlists_collections(store):
with pytest.raises(VectorStoreError, match="Collection not allowed"):
store.search(["memory; DROP SCHEMA vectors"], [1.0, 0.0], limit=1)
with pytest.raises(VectorStoreError, match="Collection not allowed"):
store.upsert("unknown", [])
def test_pgvector_rejects_kinds_not_belonging_to_collection(store):
with pytest.raises(VectorStoreError, match="Kind not allowed"):
store.existing_hashes("evidence", ["memory"])
with pytest.raises(VectorStoreError, match="Kind not allowed"):
store.upsert("evidence", [_record("wrong", [1.0, 0.0])])
def test_pgvector_separates_read_and_write_credentials(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
_, reader_config, writer_config, _ = vector_configs
reader = PgVectorStore(reader_config, expected_dimension=2)
assert reader.capabilities.search is True
assert reader.capabilities.upsert is False
with pytest.raises(VectorWriteUnavailable):
reader.upsert("memory", [])
writer = PgVectorStore(None, writer_config, expected_dimension=2)
assert writer.capabilities.search is False
assert writer.capabilities.upsert is True
with pytest.raises(VectorReadUnavailable):
writer.search(["memory"], [1.0, 0.0], limit=1)
def test_pgvector_database_roles_are_least_privilege(vector_configs):
_, reader_config, writer_config, _ = vector_configs
reader_engine = create_engine(
f"postgresql+psycopg2://{reader_config.user}:{reader_config.password}"
f"@{reader_config.host}:{reader_config.port}/{reader_config.database}"
)
writer_engine = create_engine(
f"postgresql+psycopg2://{writer_config.user}:{writer_config.password}"
f"@{writer_config.host}:{writer_config.port}/{writer_config.database}"
)
with pytest.raises(ProgrammingError):
with reader_engine.begin() as connection:
connection.execute(
text(
"INSERT INTO vectors.memory "
"(record_key, kind, content_hash, metadata, embedding) "
"VALUES ('forbidden', 'memory', 'x', '{}', '[1,0]')"
)
)
with pytest.raises(ProgrammingError):
with writer_engine.connect() as connection:
connection.execute(
text(
"SELECT metadata, 1 - (embedding <=> '[1,0]'::vector) AS similarity "
"FROM vectors.memory ORDER BY embedding <=> '[1,0]'::vector LIMIT 1"
)
)
reader_engine.dispose()
writer_engine.dispose()
def test_pgvector_writer_health_requires_sequence_usage(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
admin_config, _, _, no_sequence_config = vector_configs
store = PgVectorStore(None, no_sequence_config, expected_dimension=2)
health = store.health()
assert health.ok is False
assert health.write_reachable is False
assert health.write_detail == (
"vector schema incomplete: missing sequence privileges evidence, memory, schema_records"
)
with pytest.raises(VectorWriteUnavailable) as error:
store.upsert("memory", [_record("needs-sequence", [1.0, 0.0])])
assert isinstance(error.value.__cause__, InsufficientPrivilege)
admin_engine = create_engine(
f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}"
f"@{admin_config.host}:{admin_config.port}/{admin_config.database}"
)
with admin_engine.begin() as connection:
connection.exec_driver_sql(
"GRANT USAGE ON ALL SEQUENCES IN SCHEMA vectors TO vector_l0_no_sequence"
)
admin_engine.dispose()
assert store.health().ok is True
assert store.upsert("memory", [_record("has-sequence", [1.0, 0.0])]) == 1
def test_pgvector_health_requires_schema_usage_for_reader_and_writer(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
admin_config, reader_config, writer_config, _ = vector_configs
admin_engine = create_engine(
f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}"
f"@{admin_config.host}:{admin_config.port}/{admin_config.database}"
)
store = PgVectorStore(reader_config, writer_config, expected_dimension=2)
with admin_engine.begin() as connection:
connection.exec_driver_sql(
f"REVOKE USAGE ON SCHEMA vectors FROM {reader_config.user}, {writer_config.user}"
)
health = store.health()
assert health.read_reachable is False and health.write_reachable is False
assert "missing schema usage" in health.read_detail
assert "missing schema usage" in health.write_detail
with pytest.raises(VectorReadUnavailable, match="Vector read operation unavailable"):
store.search(["memory"], [1.0, 0.0], limit=1)
with pytest.raises(VectorWriteUnavailable, match="Vector write operation unavailable"):
store.upsert("memory", [_record("blocked", [1.0, 0.0])])
with admin_engine.begin() as connection:
connection.exec_driver_sql(
f"GRANT USAGE ON SCHEMA vectors TO {reader_config.user}, {writer_config.user}"
)
admin_engine.dispose()
assert store.health().ok is True
def test_pgvector_maps_unavailable_connections_without_leaking_password(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
_, reader_config, writer_config, _ = vector_configs
password = "never-leak-this"
reader = reader_config.model_copy(update={"port": 1, "password": password})
writer = writer_config.model_copy(update={"port": 1, "password": password})
with pytest.raises(VectorReadUnavailable) as read_error:
PgVectorStore(reader, None).search(["memory"], [1.0, 0.0], limit=1)
with pytest.raises(VectorWriteUnavailable) as hash_error:
PgVectorStore(None, writer).existing_hashes("memory", ["memory"])
with pytest.raises(VectorWriteUnavailable) as write_error:
PgVectorStore(None, writer).upsert("memory", [_record("x", [1.0, 0.0])])
assert password not in str(read_error.value)
assert password not in str(hash_error.value)
assert password not in str(write_error.value)
def test_pgvector_health_reports_dimension_and_each_connection(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
_, reader_config, writer_config, _ = vector_configs
health = PgVectorStore(reader_config, writer_config, expected_dimension=2).health()
assert health.ok is True
assert health.read_reachable is True
assert health.write_reachable is True
assert health.observed_dimensions == (2,)
assert health.dimension_compatible is True
mismatch = PgVectorStore(reader_config, None, expected_dimension=3).health()
assert mismatch.ok is False
assert mismatch.read_reachable is False
assert mismatch.read_detail == (
"embedding dimension mismatch: evidence=2, memory=2, schema_records=2"
)
assert mismatch.dimension_compatible is False
def test_pgvector_health_rejects_clean_and_partial_schemas(vector_configs):
from tht.adapters.vector.pgvector import PgVectorStore
admin_config, _, _, _ = vector_configs
engine = create_engine(
f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}"
f"@{admin_config.host}:{admin_config.port}/{admin_config.database}"
)
with engine.begin() as connection:
connection.exec_driver_sql("CREATE SCHEMA clean_vectors")
connection.exec_driver_sql("CREATE SCHEMA partial_vectors")
connection.exec_driver_sql(
"CREATE TABLE partial_vectors.memory "
"(record_key text, kind text, content_hash text, metadata jsonb)"
)
engine.dispose()
clean = PgVectorStore(
admin_config.model_copy(update={"db_schema": "clean_vectors"}),
expected_dimension=2,
).health()
assert clean.ok is False
assert clean.read_reachable is False
assert clean.read_detail == (
"vector schema incomplete: missing tables evidence, memory, schema_records"
)
partial = PgVectorStore(
admin_config.model_copy(update={"db_schema": "partial_vectors"}),
expected_dimension=2,
).health()
assert partial.ok is False
assert partial.read_reachable is False
assert partial.read_detail == (
"vector schema incomplete: missing tables evidence, schema_records; "
"missing embedding columns memory"
)
@@ -1,294 +0,0 @@
import math
import pytest
from sqlalchemy import create_engine
from testcontainers.postgres import PostgresContainer
from tht.adapters.vector.pgvector import PgVectorStore
from tht.adapters.vector.thoth_http import ThothHttpVectorStore
from tht.config import DatabaseConfig, RestConfig
from tht.ports.vector import VectorRecord, VectorStoreError, VectorWriteRecord
from tht.vectorstore.rest_client import VectorRestClient, VectorRestError
def _write(record_id, kind, embedding, content_hash):
return VectorWriteRecord(
VectorRecord(
id=record_id,
kind=kind,
ref="fixture",
title=record_id,
content=f"content {record_id}",
metadata={"fixture": True},
),
embedding,
content_hash,
)
FIXTURE = [
_write("memory:a", "memory", [1.0, 0.0], "hash-a"),
_write("memory:b", "memory", [1.0, 0.0], "hash-b"),
_write("solved:a", "solved_question", [0.8, 0.2], "hash-solved"),
]
class Response:
def __init__(self, payload=None, status=200):
self.status_code = status
self.payload = payload
self.text = "" if payload is None else "json"
@property
def ok(self):
return self.status_code < 400
def json(self):
return self.payload
class FixtureHttpTransport:
def __init__(self):
self.rows = {}
self.calls = []
def post(self, url, json, headers, **kwargs):
assert headers == {"X-API-Key": "parity-key"}
self.calls.append((url.rsplit("/", 1)[-1], json))
function = self.calls[-1][0]
if function == "list_tables":
return Response([{"table_name": "memory", "vector_dimensions": 2}])
if function == "upsert_vector_records":
for row in json["rows"]:
self.rows[(json["table_name"], row["record_key"])] = row
return Response({"upserted": len(json["rows"])})
if function == "existing_vector_hashes":
return Response([
{"record_key": row["record_key"], "content_hash": row["content_hash"]}
for (table, _), row in self.rows.items()
if table == json["table_name"] and row["kind"] in json["kinds"]
])
assert function == "search_similar"
table_name = json["table_name"]
embedding = json["query_embedding"]
kinds = json.get("kinds")
def similarity(row):
left, right = row["embedding"], embedding
return sum(a * b for a, b in zip(left, right)) / (
math.sqrt(sum(a * a for a in left))
* math.sqrt(sum(b * b for b in right))
)
rows = [
{"metadata": row["metadata"], "similarity": similarity(row)}
for (table, _), row in self.rows.items()
if table == table_name and (not kinds or row["kind"] in kinds)
]
payload = sorted(
rows,
key=lambda row: (-row["similarity"], row["metadata"]["record_key"]),
)[: json["limit_count"]]
return Response(payload)
@pytest.fixture
def direct_store():
with PostgresContainer("pgvector/pgvector:pg16") as postgres:
config = DatabaseConfig(
host=postgres.get_container_host_ip(),
port=int(postgres.get_exposed_port(5432)),
database=postgres.dbname,
schema="vectors",
user=postgres.username,
password=postgres.password,
)
engine = create_engine(postgres.get_connection_url())
with engine.begin() as connection:
connection.exec_driver_sql("CREATE SCHEMA vectors")
connection.exec_driver_sql("CREATE EXTENSION vector WITH SCHEMA vectors")
connection.exec_driver_sql(
"CREATE TABLE vectors.memory ("
"id bigserial PRIMARY KEY, record_key text UNIQUE NOT NULL, "
"kind text NOT NULL, content_hash text NOT NULL, metadata jsonb NOT NULL, "
"embedding vectors.vector(2) NOT NULL, indexed_at timestamptz NOT NULL "
"DEFAULT now())"
)
engine.dispose()
reader, writer = config, config
store = PgVectorStore(reader, writer, expected_dimension=2)
store.upsert("memory", FIXTURE)
yield store
@pytest.fixture
def http_store(monkeypatch):
transport = FixtureHttpTransport()
monkeypatch.setattr("tht.vectorstore.rest_client.requests.post", transport.post)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key"))
store = ThothHttpVectorStore(client, client, expected_dimension=2)
store.upsert("memory", FIXTURE)
store.transport = transport
return store
@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"])
def test_kind_filtered_search_has_identical_order(request, store_fixture):
store = request.getfixturevalue(store_fixture)
hits = store.search(["memory"], [1.0, 0.0], limit=3, kinds=["memory"])
assert [(hit.id, hit.kind, round(hit.similarity, 6)) for hit in hits] == [
("memory:a", "memory", 1.0),
("memory:b", "memory", 1.0),
]
@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"])
def test_hash_and_upsert_parity(request, store_fixture):
store = request.getfixturevalue(store_fixture)
assert store.existing_hashes("memory", ["memory"]) == {
"memory:a": "hash-a",
"memory:b": "hash-b",
}
replacement = _write("memory:a", "memory", [0.0, 1.0], "hash-a-2")
assert store.upsert("memory", [replacement]) == 1
assert store.existing_hashes("memory", ["memory"])["memory:a"] == "hash-a-2"
assert store.search(["memory"], [0.0, 1.0], limit=1, kinds=["memory"])[0].id == "memory:a"
@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"])
def test_validation_error_parity(request, store_fixture):
store = request.getfixturevalue(store_fixture)
with pytest.raises(VectorStoreError, match="Collection not allowed"):
store.search(["not_allowed"], [1.0, 0.0], limit=1)
with pytest.raises(VectorStoreError, match="Kind not allowed"):
store.search(["memory"], [1.0, 0.0], limit=1, kinds=["not_allowed"])
@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"])
def test_dimension_error_parity(request, store_fixture):
store = request.getfixturevalue(store_fixture)
with pytest.raises(VectorStoreError, match="Query embedding dimension"):
store.search(["memory"], [1.0], limit=1)
with pytest.raises(VectorStoreError, match="Embedding dimension"):
store.upsert("memory", [_write("bad", "memory", [1.0], "bad")])
def test_http_parity_exercises_rpc_kinds_payload(http_store):
http_store.search(["memory"], [1.0, 0.0], limit=2, kinds=["memory"])
search_calls = [payload for function, payload in http_store.transport.calls if function == "search_similar"]
assert search_calls[-1] == {
"query_embedding": [1.0, 0.0],
"limit_count": 2,
"table_name": "memory",
"kinds": ["memory"],
}
def test_http_adapter_maps_transport_error(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response({"message": "server broke"}, status=500),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key"))
store = ThothHttpVectorStore(client, client, expected_dimension=2)
with pytest.raises(VectorStoreError, match="HTTP 500"):
store.search(["memory"], [1.0, 0.0], limit=1, kinds=["memory"])
def test_http_adapter_tolerates_malformed_metadata(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response([{"similarity": 0.5, "metadata": None}]),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key"))
hit = ThothHttpVectorStore(client, None, expected_dimension=2).search(
["memory"], [1.0, 0.0], limit=1
)[0]
assert (hit.id, hit.kind, hit.metadata) == ("", "", {})
def test_http_adapter_legacy_fallback_preserves_kind_semantics(monkeypatch):
calls = []
def post(url, json, **kwargs):
calls.append(json)
if "kinds" in json:
return Response({"message": "function not found"}, status=404)
return Response([
{"similarity": 1.0, "metadata": {"record_key": "wrong", "kind": "solved_question"}},
{"similarity": 0.9, "metadata": {"record_key": "right", "kind": "memory"}},
])
monkeypatch.setattr("tht.vectorstore.rest_client.requests.post", post)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key"))
hits = ThothHttpVectorStore(client, None, expected_dimension=2).search(
["memory"], [1.0, 0.0], limit=2, kinds=["memory"]
)
assert [hit.id for hit in hits] == ["right"]
assert "kinds" in calls[0] and "kinds" not in calls[1]
def test_http_delete_generation_uses_exact_allowlisted_rpc_payload(monkeypatch):
calls = []
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda url, json, **kwargs: calls.append((url, json)) or Response({"deleted": 2}),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
assert client.delete_generation("evidence", "gen:" + "a" * 32, "default") == 2
assert calls == [("https://vectors.test/rpc/delete_vector_generation", {
"table_name": "evidence", "kind": "evidence", "generation": "gen:" + "a" * 32,
"workspace_id": "default",
})]
def test_http_delete_generation_legacy_404_fails_closed_without_body_leak(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response({"message": "secret legacy endpoint detail"}, status=404),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(VectorRestError, match="delete_vector_generation RPC is unavailable") as error:
client.delete_generation("evidence", "gen:" + "a" * 32, "default")
assert "secret" not in str(error.value)
def test_http_rest_client_does_not_advertise_nonexistent_delete_kinds_rpc():
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
assert hasattr(client, "delete_kinds") is False
def test_http_list_evidence_generations_exact_rpc_and_legacy_fail_closed(monkeypatch):
calls = []
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda url, json, **kwargs: calls.append((url, json)) or Response([
{"generation": "gen:" + "a" * 32}
]),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
assert client.list_evidence_generations("evidence", "default") == ["gen:" + "a" * 32]
assert calls[0][0].endswith("/rpc/list_evidence_generations")
assert calls[0][1] == {"table_name": "evidence", "kind": "evidence", "workspace_id": "default"}
@pytest.mark.parametrize("generation", ["gen:a", "gen:" + "A" * 32, "gen:" + "a" * 33])
def test_http_generation_operations_reject_noncanonical_values(monkeypatch, generation):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: pytest.fail("invalid generation reached transport"),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(ValueError, match="canonical"):
client.delete_generation("evidence", generation, "default")
def test_http_inventory_rejects_malformed_rpc_output(monkeypatch):
monkeypatch.setattr(
"tht.vectorstore.rest_client.requests.post",
lambda *args, **kwargs: Response([{"generation": "gen:../escape"}]),
)
client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer"))
with pytest.raises(VectorRestError, match="malformed"):
client.list_evidence_generations("evidence", "default")
-303
View File
@@ -1,303 +0,0 @@
import json
from pathlib import Path
import pytest
from sqlalchemy import create_engine, text
from sqlalchemy.exc import ProgrammingError
from testcontainers.postgres import PostgresContainer
from typer.testing import CliRunner
from tht.cli import app
from tht.config import DatabaseConfig
from tht.ports.vector import VectorRecord, VectorWriteRecord
@pytest.fixture(scope="module")
def database_url():
with PostgresContainer("pgvector/pgvector:pg16") as postgres:
yield postgres.get_connection_url()
def test_migrations_are_clean_and_idempotent(database_url):
from tht.cli.vector_migrate_cmd import migrate, migration_status
before = migration_status(database_url)
assert [item.version for item in before.pending] == ["001", "002", "003", "004"]
migrate(database_url)
migrate(database_url)
status = migration_status(database_url)
assert status.pending == ()
assert status.drifted == ()
assert [item.version for item in status.applied] == ["001", "002", "003", "004"]
def test_schema_matches_direct_adapter_contract(database_url):
engine = create_engine(database_url)
with engine.connect() as connection:
rows = connection.execute(
text(
"SELECT table_name, column_name, data_type, udt_name "
"FROM information_schema.columns WHERE table_schema = 'vectors' "
"ORDER BY table_name, ordinal_position"
)
).all()
vector_types = connection.execute(
text(
"SELECT c.relname, format_type(a.atttypid, a.atttypmod) "
"FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace "
"JOIN pg_attribute a ON a.attrelid = c.oid AND a.attname = 'embedding' "
"WHERE n.nspname = 'vectors' ORDER BY c.relname"
)
).all()
engine.dispose()
tables = {row.table_name for row in rows}
assert tables == {"evidence", "memory", "schema_records"}
required = {"id", "record_key", "kind", "content_hash", "metadata", "embedding", "indexed_at"}
for table in tables:
assert {row.column_name for row in rows if row.table_name == table} == required
assert vector_types == [
("evidence", "vectors.vector(768)"),
("memory", "vectors.vector(768)"),
("schema_records", "vectors.vector(768)"),
]
def test_roles_have_runtime_privileges_only(database_url):
from tht.cli.vector_migrate_cmd import migrate
migrate(database_url)
admin = create_engine(database_url)
with admin.begin() as connection:
connection.exec_driver_sql("ALTER ROLE vector_reader LOGIN PASSWORD 'reader-test-only'")
connection.exec_driver_sql("ALTER ROLE vector_writer LOGIN PASSWORD 'writer-test-only'")
url = admin.url
reader = create_engine(url.set(username="vector_reader", password="reader-test-only"))
writer = create_engine(url.set(username="vector_writer", password="writer-test-only"))
from tht.adapters.vector.pgvector import PgVectorStore
common = {
"host": url.host,
"port": url.port,
"database": url.database,
"schema": "vectors",
}
reader_config = DatabaseConfig(
**common, user="vector_reader", password="reader-test-only"
)
writer_config = DatabaseConfig(
**common, user="vector_writer", password="writer-test-only"
)
store = PgVectorStore(reader_config, writer_config, expected_dimension=768)
assert store.health().ok is True
assert store.upsert(
"memory",
[
VectorWriteRecord(
record=VectorRecord(
id="adapter-write",
kind="memory",
ref="session:test",
title="test",
content="test",
),
embedding=[0.0] * 768,
content_hash="adapter-hash",
)
],
) == 1
with reader.connect() as connection:
connection.execute(text("SELECT metadata, embedding FROM vectors.memory")).all()
with pytest.raises(ProgrammingError):
with reader.begin() as connection:
connection.execute(
text(
"INSERT INTO vectors.memory "
"(record_key, kind, content_hash, metadata, embedding) "
"VALUES ('reader-write', 'memory', 'x', '{}', "
"array_fill(0, ARRAY[768])::vectors.vector)"
)
)
with writer.begin() as connection:
connection.execute(
text(
"INSERT INTO vectors.memory "
"(record_key, kind, content_hash, metadata, embedding) "
"VALUES ('writer-ok', 'memory', 'x', '{}', "
"array_fill(0, ARRAY[768])::vectors.vector)"
)
)
assert connection.execute(
text("SELECT content_hash FROM vectors.memory WHERE record_key = 'writer-ok'")
).scalar_one() == "x"
connection.execute(
text("UPDATE vectors.memory SET content_hash = 'y' WHERE record_key = 'writer-ok'")
)
with pytest.raises(ProgrammingError):
with writer.connect() as connection:
connection.execute(text("SELECT metadata FROM vectors.memory")).all()
with pytest.raises(ProgrammingError):
with writer.begin() as connection:
connection.execute(text("DELETE FROM vectors.memory WHERE record_key = 'writer-ok'"))
reader.dispose()
writer.dispose()
admin.dispose()
def test_status_json_is_pristine(database_url, monkeypatch):
monkeypatch.setenv("THT_VECTOR_ADMIN_URL", database_url)
result = CliRunner().invoke(app, ["vector", "migrate", "--status", "--json"])
assert result.exit_code == 0, result.output
assert json.loads(result.stdout) == {
"applied": ["001", "002", "003", "004"],
"drifted": [],
"pending": [],
}
assert result.stderr == ""
def test_checksum_drift_is_reported_and_refused(database_url, tmp_path):
from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status
migrations = _copy_migrations(tmp_path)
migrate(database_url, migrations)
(migrations / "002_schema_tables.sql").write_text("SELECT 2;\n")
assert [item.version for item in migration_status(database_url, migrations).drifted] == [
"002"
]
with pytest.raises(MigrationError, match="checksum drift"):
migrate(database_url, migrations)
def test_unknown_applied_version_is_downgrade_drift(database_url):
from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status
migrate(database_url)
engine = create_engine(database_url)
with engine.begin() as connection:
connection.execute(
text(
"INSERT INTO public.tht_vector_migrations (version, name, checksum) "
"VALUES ('999', 'future', 'future-checksum'), "
"('future_x', 'future_named', 'future-checksum')"
)
)
try:
with pytest.raises(
MigrationError, match="absent from local manifest: 999, future_x"
):
migration_status(database_url)
with pytest.raises(
MigrationError, match="absent from local manifest: 999, future_x"
):
migrate(database_url)
finally:
with engine.begin() as connection:
connection.execute(
text(
"DELETE FROM public.tht_vector_migrations "
"WHERE version IN ('999', 'future_x')"
)
)
engine.dispose()
def test_migration_versions_sort_numerically_and_reject_numeric_duplicates(tmp_path):
from tht.cli.vector_migrate_cmd import MigrationError, _discover
migrations = tmp_path / "ordered"
migrations.mkdir()
(migrations / "10_tenth.sql").write_text("SELECT 10;\n")
(migrations / "2_second.sql").write_text("SELECT 2;\n")
assert [item.version for item in _discover(migrations)] == ["2", "10"]
(migrations / "02_duplicate.sql").write_text("SELECT 2;\n")
with pytest.raises(MigrationError, match="Duplicate migration version: 2"):
_discover(migrations)
def test_hostile_admin_search_path_cannot_shadow_migration_objects(database_url):
from tht.cli.vector_migrate_cmd import migrate
admin = create_engine(database_url, isolation_level="AUTOCOMMIT")
with admin.connect() as connection:
connection.exec_driver_sql("DROP DATABASE IF EXISTS vector_hostile")
connection.exec_driver_sql("CREATE DATABASE vector_hostile")
hostile_url = admin.url.set(database="vector_hostile")
hostile = create_engine(hostile_url)
try:
with hostile.begin() as connection:
connection.exec_driver_sql("CREATE SCHEMA shadow")
connection.exec_driver_sql(
"CREATE TABLE shadow.tht_vector_migrations "
"(version text, checksum text, poisoned boolean DEFAULT true)"
)
connection.exec_driver_sql("ALTER ROLE test SET search_path = shadow, public")
hostile.dispose()
migrate(hostile_url.render_as_string(hide_password=False))
verification = create_engine(hostile_url)
with verification.connect() as connection:
assert connection.execute(
text("SELECT count(*) FROM public.tht_vector_migrations")
).scalar_one() == 4
assert connection.execute(
text("SELECT count(*) FROM shadow.tht_vector_migrations")
).scalar_one() == 0
assert connection.execute(
text(
"SELECT format_type(a.atttypid, a.atttypmod) "
"FROM pg_catalog.pg_attribute a "
"WHERE a.attrelid = 'vectors.memory'::pg_catalog.regclass "
"AND a.attname = 'embedding'"
)
).scalar_one() == "vectors.vector(768)"
verification.dispose()
finally:
cleanup = create_engine(database_url, isolation_level="AUTOCOMMIT")
with cleanup.connect() as connection:
connection.exec_driver_sql("ALTER ROLE test RESET search_path")
connection.exec_driver_sql(
"SELECT pg_catalog.pg_terminate_backend(pid) FROM pg_catalog.pg_stat_activity "
"WHERE datname = 'vector_hostile' AND pid <> pg_catalog.pg_backend_pid()"
)
connection.exec_driver_sql("DROP DATABASE IF EXISTS vector_hostile")
cleanup.dispose()
admin.dispose()
def test_failed_batch_rolls_back_schema_and_ledger(database_url, tmp_path):
from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status
migrations = _copy_migrations(tmp_path)
(migrations / "005_first.sql").write_text("CREATE TABLE public.must_rollback (id int);\n")
(migrations / "006_broken.sql").write_text("THIS IS NOT SQL;\n")
with pytest.raises(MigrationError, match="006_broken.sql"):
migrate(database_url, migrations)
engine = create_engine(database_url)
with engine.connect() as connection:
assert connection.execute(text("SELECT to_regclass('public.must_rollback')")).scalar() is None
engine.dispose()
status = migration_status(database_url, migrations)
assert [item.version for item in status.applied] == ["001", "002", "003", "004"]
assert [item.version for item in status.pending] == ["005", "006"]
def _copy_migrations(tmp_path: Path) -> Path:
source = Path(__file__).parents[2] / "tht" / "migrations" / "vector"
target = tmp_path / "migrations"
target.mkdir()
for migration in source.glob("*.sql"):
(target / migration.name).write_bytes(migration.read_bytes())
return target