from __future__ import annotations import json from datetime import UTC, datetime from pathlib import Path from types import SimpleNamespace from typer.testing import CliRunner from tht.cli import app from tht.memory import MemoryRecord, save_registry class _FakeEmbedder: def embed_documents(self, documents): return [[0.1] * 4 for _ in documents] class _FakeVectorStore: def __init__(self): self.upserts = [] self.deleted = [] def existing_hashes(self, collection, kinds): return {} def upsert(self, collection, records): self.upserts.append((collection, records)) return len(records) def delete_kinds(self, collection, kinds): self.deleted.append((collection, list(kinds))) return 3 def _qdrant_runtime_config(tmp_path: Path) -> Path: cfg = tmp_path / "workspace.yaml" cfg.write_text( f""" runtime_identity: workspace_id: psd-clinical workspace_revision: {'a' * 40} dwh: type: postgres_direct connection: {{database: analytics, schema: mart, user: reader, password: secret}} vectors: type: qdrant base_url: http://qdrant:6333 collection: psd-clinical roots: sessions: {tmp_path / 'sessions'} artifacts: {tmp_path / 'artifacts'} indexes: {tmp_path / 'indexes'} embeddings: provider: ollama_internal base_url: http://embedding:11434 model: qwen3-embedding:0.6b dim: 1024 """ ) return cfg def _legacy_qdrant_runtime_config(tmp_path: Path) -> Path: cfg = _qdrant_runtime_config(tmp_path) text = cfg.read_text() text = text.replace( "dwh:\n type: postgres_direct\n connection: {database: analytics, schema: mart, user: reader, password: secret}\n", "database: {database: analytics, schema: mart, user: reader, password: secret}\n", ) text = text.replace("roots:\n", "paths:\n") cfg.write_text(text) return cfg def _write_schema_artifacts(tmp_path: Path) -> None: (tmp_path / "artifacts" / "mschema").mkdir(parents=True, exist_ok=True) (tmp_path / "artifacts" / "mschema" / "physical.yaml").write_text( """ database: analytics schema: mart introspected_at: 2026-01-01T00:00:00+00:00 tables: fact_patient: comment: Patients columns: id: type: bigint """ ) (tmp_path / "artifacts" / "mschema" / "annotations.yaml").write_text( "tables: {}\n" ) def _memory_record() -> MemoryRecord: return MemoryRecord( id="mem-0001", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1", decision_seq=7, type="concept_clarified", subject="paziente attivo", detail="flag_attivo = TRUE", rationale="r", question_context="dammi i pazienti attivi", tables=[], concepts=["paziente attivo"], ) def test_vector_index_schema_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch): cfg = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) store = _FakeVectorStore() monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder()) res = CliRunner().invoke(app, ["vector", "index-schema", "-c", str(cfg)]) assert res.exit_code == 0, res.output assert store.upserts def test_memory_promote_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch): cfg = _qdrant_runtime_config(tmp_path) store = _FakeVectorStore() promoted = [_memory_record()] snapshot = SimpleNamespace(manifest=SimpleNamespace(id="s1")) monkeypatch.setattr("tht.cli.memory_cmd.load_snapshot_or_exit", lambda cfg, session: snapshot) monkeypatch.setattr("tht.memory.promote_snapshot", lambda *args, **kwargs: promoted) monkeypatch.setattr("tht.memory.load_registry", lambda path: promoted) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder()) res = CliRunner().invoke( app, ["memory", "promote", "--session", "s1", "--decision", "7", "--json", "-c", str(cfg)], ) assert res.exit_code == 0, res.output assert json.loads(res.stdout)["indexed"] is True assert store.upserts def test_memory_index_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch): cfg = _qdrant_runtime_config(tmp_path) store = _FakeVectorStore() records = [_memory_record()] save_registry(records, tmp_path / "artifacts" / "memory" / "registry.jsonl") monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder()) res = CliRunner().invoke(app, ["memory", "index", "-c", str(cfg)]) assert res.exit_code == 0, res.output assert "OK:" in res.output assert store.upserts def test_memory_clear_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch): cfg = _qdrant_runtime_config(tmp_path) store = _FakeVectorStore() records = [_memory_record()] registry = tmp_path / "artifacts" / "memory" / "registry.jsonl" save_registry(records, registry) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) res = CliRunner().invoke(app, ["memory", "clear", "--yes", "-c", str(cfg)]) assert res.exit_code == 0, res.output assert store.deleted == [("memory", ["memory"])] assert not registry.exists() def test_vector_help_does_not_expose_migrate_and_keeps_qdrant_commands(): res = CliRunner().invoke(app, ["vector", "--help"]) assert res.exit_code == 0, res.output assert "migrate" not in res.output assert "init" in res.output assert "index-schema" in res.output def test_vector_migrate_command_is_absent(): res = CliRunner().invoke(app, ["vector", "migrate", "--help"]) assert res.exit_code != 0 assert "No such command 'migrate'" in res.output def test_memory_solved_index_help_uses_semantic_store_wording(): res = CliRunner().invoke(app, ["memory", "solved-index", "--help"]) assert res.exit_code == 0, res.output assert "semantic" in res.output.lower() or "qdrant" in res.output.lower() assert "vectordb" not in res.output.lower() def test_vector_index_schema_json_is_single_document(monkeypatch, tmp_path): import json cfg = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) store = _FakeVectorStore() monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder()) response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code == 0, response.output assert response.stdout.count("\n") == 1 payload = json.loads(response.stdout) assert payload["status"] == "succeeded" assert payload["code"] == "ok" assert payload["counts"]["added"] == 2 def test_vector_index_schema_json_failure_is_safe(monkeypatch, tmp_path): import json cfg = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) monkeypatch.setattr( "tht.adapters.factory.build_vector_store", lambda cfg, require_write: (_ for _ in ()).throw(Exception("secret qdrant endpoint")), ) response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code != 0 assert response.stdout.count("\n") == 1 payload = json.loads(response.stdout) assert payload == {"status": "failed", "code": "schema_index_failed"} assert "secret qdrant" not in response.stdout assert response.stderr == "" def test_vector_index_schema_human_missing_physical_has_original_error(tmp_path): cfg = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) physical = tmp_path / "artifacts" / "mschema" / "physical.yaml" physical.unlink() response = CliRunner().invoke(app, ["vector", "index-schema", "-c", str(cfg)]) assert response.exit_code == 1 assert "physical.yaml non trovato. Esegui prima `tht schema introspect`." in response.output def test_vector_index_schema_json_missing_physical_has_no_stderr_prose(tmp_path): import json cfg = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) (tmp_path / "artifacts" / "mschema" / "physical.yaml").unlink() response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code == 1 assert response.stderr == "" assert json.loads(response.stdout) == {"status": "failed", "code": "physical_schema_missing"} def test_vector_index_schema_human_missing_config_has_original_error(tmp_path): cfg = tmp_path / "missing.yaml" response = CliRunner().invoke(app, ["vector", "index-schema", "-c", str(cfg)]) assert response.exit_code == 1 assert response.output assert "ERRORE:" in response.output def test_vector_index_schema_json_legacy_config_has_no_stderr_on_success(monkeypatch, tmp_path): cfg = _legacy_qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) store = _FakeVectorStore() monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder()) response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code == 0, response.output assert response.stderr == "" assert response.stdout.count("\n") == 1 assert json.loads(response.stdout)["status"] == "succeeded" def test_vector_index_schema_json_legacy_config_has_no_stderr_on_failure(tmp_path): cfg = _legacy_qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) (tmp_path / "artifacts" / "mschema" / "physical.yaml").unlink() response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code == 1 assert response.stderr == "" assert response.stdout.count("\n") == 1 assert json.loads(response.stdout) == { "status": "failed", "code": "physical_schema_missing" } def test_vector_index_schema_guards_before_artifact_access(tmp_path, monkeypatch): import tht.cli.vector_cmd as command cfg = _legacy_qdrant_runtime_config(tmp_path) text = cfg.read_text() text = text.replace("vectors:\n type: qdrant\n base_url: http://qdrant:6333\n collection: psd-clinical\n", "") text = text.replace("embeddings:\n provider: ollama_internal\n base_url: http://embedding:11434\n model: qwen3-embedding:0.6b\n dim: 1024\n", "") cfg.write_text(text) monkeypatch.setattr(command, "_load_schema_artifacts", lambda cfg: (_ for _ in ()).throw(AssertionError("artifact access"))) response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)]) assert response.exit_code == 1 assert json.loads(response.stdout) == {"status": "failed", "code": "vector_configuration_missing"} assert response.stderr == "" def test_vector_index_schema_core_reuses_injected_artifacts_without_path_resolution(tmp_path, monkeypatch): import tht.cli.vector_cmd as command from tht.config import load_config from tht.mschema.models import Annotations, PhysicalSchema cfg_path = _qdrant_runtime_config(tmp_path) _write_schema_artifacts(tmp_path) physical = PhysicalSchema.from_yaml(tmp_path / "artifacts" / "mschema" / "physical.yaml") annotations = Annotations.from_yaml(tmp_path / "artifacts" / "mschema" / "annotations.yaml") store = _FakeVectorStore() monkeypatch.setattr(command, "physical_path", lambda cfg: (_ for _ in ()).throw(AssertionError("physical path"))) monkeypatch.setattr(command, "annotations_path", lambda cfg: (_ for _ in ()).throw(AssertionError("annotations path"))) monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) monkeypatch.setattr(command, "make_embedder", lambda _: _FakeEmbedder()) payload = command.index_schema_data(load_config(cfg_path), physical=physical, annotations=annotations) assert payload["status"] == "succeeded"