from pathlib import Path import typer from tht.cli._guards import require_server_profile, require_vector_write_allowed from tht.cli.config_cmd import CONFIG_OPT from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path from tht.ports.vector import VectorWriteRecord from tht.vectorstore.store import SyncStats, content_hash vector_app = typer.Typer(help="Indice semantico Qdrant (derivato, rigenerabile)") def make_embedder(embeddings_cfg): """Factory del client embeddings (monkeypatchabile nei test).""" from tht.vectorstore.embeddings import OllamaEmbeddings return OllamaEmbeddings(embeddings_cfg) def require_vector_cfg(cfg): missing = [] if cfg.embeddings is None: missing.append("embeddings") if cfg.vectors is None: missing.append("vectors") if missing: typer.secho( f"ERRORE: sezioni mancanti nel workspace yaml: {', '.join(missing)}.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) def open_searcher(cfg): """Searcher per la lettura semantic search sul runtime vettoriale attivo.""" from tht.adapters.factory import build_vector_store from tht.vectorstore.reader import tables_for_kinds store = build_vector_store(cfg) class AdapterSearcher: def search(self, query_vec, top_n=10, kinds=None, metadata_filter=None): return store.search( tables_for_kinds(kinds), query_vec, limit=top_n, kinds=kinds, metadata_filter=metadata_filter, ) return AdapterSearcher() def sync_canonical_records(collection, records, *, store, embedder): kinds = sorted({record.kind for record in records}) existing = store.existing_hashes(collection, kinds) pending = [] stats = SyncStats() changed = [] for record in records: hashed = content_hash(record.content) current = existing.get(record.id) if current == hashed: stats.unchanged += 1 continue changed.append((record, hashed, current is None)) if changed: embeddings = embedder.embed_documents([record.content for record, *_ in changed]) for (record, hashed, is_added), embedding in zip(changed, embeddings, strict=True): pending.append(VectorWriteRecord(record=record, embedding=embedding, content_hash=hashed)) if is_added: stats.added += 1 else: stats.updated += 1 store.upsert(collection, pending) return stats def _print_stats(stats) -> None: typer.secho( f"OK: {stats.added} nuovi, {stats.updated} aggiornati, " f"{stats.deleted} rimossi, {stats.unchanged} invariati", fg=typer.colors.GREEN, ) @vector_app.command("init") def init_cmd( config: Path = CONFIG_OPT, skip_ollama_check: bool = typer.Option( False, "--skip-ollama-check", help="Non verificare la raggiungibilita' di Ollama." ), ) -> None: """Verifica il runtime Qdrant e la raggiungibilita' dell'embedder configurato.""" from tht.adapters.factory import build_vector_store from tht.vectorstore.embeddings import EmbeddingsError cfg = _load_config_or_exit(config) require_server_profile(cfg, "vector init") require_vector_cfg(cfg) health = build_vector_store(cfg, require_write=True).health() if not health.ok: typer.secho( f"ERRORE runtime vettoriale: {health.detail or 'Qdrant non raggiungibile o incompatibile'}", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) if not skip_ollama_check: try: make_embedder(cfg.embeddings).embed_query("ping") except EmbeddingsError as e: typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) typer.secho( f"OK: runtime Qdrant pronto per la collezione {cfg.vectors.collection}", fg=typer.colors.GREEN, ) @vector_app.command("index-schema") def index_schema_cmd(config: Path = CONFIG_OPT) -> None: """Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store.""" from tht.mschema.models import Annotations, PhysicalSchema from tht.vectorstore.records import schema_records cfg = _load_config_or_exit(config) require_vector_write_allowed(cfg, "vector index-schema") require_vector_cfg(cfg) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) annotations = Annotations.from_yaml(annotations_path(cfg)) records = schema_records(physical, annotations) from tht.adapters.factory import build_vector_store stats = sync_canonical_records( "schema_records", records, store=build_vector_store(cfg, require_write=True), embedder=make_embedder(cfg.embeddings), ) _print_stats(stats)