147 lines
5.0 KiB
Python
147 lines
5.0 KiB
Python
from pathlib import Path
|
|
|
|
import typer
|
|
|
|
from tht.cli._guards import require_server_profile, require_vector_write_allowed
|
|
from tht.cli.config_cmd import CONFIG_OPT
|
|
from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path
|
|
from tht.ports.vector import VectorWriteRecord
|
|
from tht.vectorstore.store import SyncStats, content_hash
|
|
|
|
vector_app = typer.Typer(help="Indice semantico Qdrant (derivato, rigenerabile)")
|
|
|
|
|
|
def make_embedder(embeddings_cfg):
|
|
"""Factory del client embeddings (monkeypatchabile nei test)."""
|
|
from tht.vectorstore.embeddings import OllamaEmbeddings
|
|
|
|
return OllamaEmbeddings(embeddings_cfg)
|
|
|
|
|
|
def require_vector_cfg(cfg):
|
|
missing = []
|
|
if cfg.embeddings is None:
|
|
missing.append("embeddings")
|
|
if cfg.vectors is None:
|
|
missing.append("vectors")
|
|
if missing:
|
|
typer.secho(
|
|
f"ERRORE: sezioni mancanti nel workspace yaml: {', '.join(missing)}.",
|
|
fg=typer.colors.RED, err=True,
|
|
)
|
|
raise typer.Exit(code=1)
|
|
|
|
|
|
def open_searcher(cfg):
|
|
"""Searcher per la lettura semantic search sul runtime vettoriale attivo."""
|
|
from tht.adapters.factory import build_vector_store
|
|
from tht.vectorstore.reader import tables_for_kinds
|
|
|
|
store = build_vector_store(cfg)
|
|
|
|
class AdapterSearcher:
|
|
def search(self, query_vec, top_n=10, kinds=None, metadata_filter=None):
|
|
return store.search(
|
|
tables_for_kinds(kinds), query_vec, limit=top_n, kinds=kinds,
|
|
metadata_filter=metadata_filter,
|
|
)
|
|
|
|
return AdapterSearcher()
|
|
|
|
|
|
def sync_canonical_records(collection, records, *, store, embedder):
|
|
kinds = sorted({record.kind for record in records})
|
|
existing = store.existing_hashes(collection, kinds)
|
|
pending = []
|
|
stats = SyncStats()
|
|
changed = []
|
|
for record in records:
|
|
hashed = content_hash(record.content)
|
|
current = existing.get(record.id)
|
|
if current == hashed:
|
|
stats.unchanged += 1
|
|
continue
|
|
changed.append((record, hashed, current is None))
|
|
if changed:
|
|
embeddings = embedder.embed_documents([record.content for record, *_ in changed])
|
|
for (record, hashed, is_added), embedding in zip(changed, embeddings, strict=True):
|
|
pending.append(VectorWriteRecord(record=record, embedding=embedding, content_hash=hashed))
|
|
if is_added:
|
|
stats.added += 1
|
|
else:
|
|
stats.updated += 1
|
|
store.upsert(collection, pending)
|
|
return stats
|
|
|
|
|
|
def _print_stats(stats) -> None:
|
|
typer.secho(
|
|
f"OK: {stats.added} nuovi, {stats.updated} aggiornati, "
|
|
f"{stats.deleted} rimossi, {stats.unchanged} invariati",
|
|
fg=typer.colors.GREEN,
|
|
)
|
|
|
|
|
|
@vector_app.command("init")
|
|
def init_cmd(
|
|
config: Path = CONFIG_OPT,
|
|
skip_ollama_check: bool = typer.Option(
|
|
False, "--skip-ollama-check", help="Non verificare la raggiungibilita' di Ollama."
|
|
),
|
|
) -> None:
|
|
"""Verifica il runtime Qdrant e la raggiungibilita' dell'embedder configurato."""
|
|
from tht.adapters.factory import build_vector_store
|
|
from tht.vectorstore.embeddings import EmbeddingsError
|
|
|
|
cfg = _load_config_or_exit(config)
|
|
require_server_profile(cfg, "vector init")
|
|
require_vector_cfg(cfg)
|
|
health = build_vector_store(cfg, require_write=True).health()
|
|
if not health.ok:
|
|
typer.secho(
|
|
f"ERRORE runtime vettoriale: {health.detail or 'Qdrant non raggiungibile o incompatibile'}",
|
|
fg=typer.colors.RED,
|
|
err=True,
|
|
)
|
|
raise typer.Exit(code=1)
|
|
if not skip_ollama_check:
|
|
try:
|
|
make_embedder(cfg.embeddings).embed_query("ping")
|
|
except EmbeddingsError as e:
|
|
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
|
|
raise typer.Exit(code=1)
|
|
typer.secho(
|
|
f"OK: runtime Qdrant pronto per la collezione {cfg.vectors.collection}",
|
|
fg=typer.colors.GREEN,
|
|
)
|
|
|
|
|
|
@vector_app.command("index-schema")
|
|
def index_schema_cmd(config: Path = CONFIG_OPT) -> None:
|
|
"""Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store."""
|
|
from tht.mschema.models import Annotations, PhysicalSchema
|
|
from tht.vectorstore.records import schema_records
|
|
|
|
cfg = _load_config_or_exit(config)
|
|
require_vector_write_allowed(cfg, "vector index-schema")
|
|
require_vector_cfg(cfg)
|
|
phys_file = physical_path(cfg)
|
|
if not phys_file.exists():
|
|
typer.secho(
|
|
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
|
fg=typer.colors.RED, err=True,
|
|
)
|
|
raise typer.Exit(code=1)
|
|
physical = PhysicalSchema.from_yaml(phys_file)
|
|
annotations = Annotations.from_yaml(annotations_path(cfg))
|
|
records = schema_records(physical, annotations)
|
|
from tht.adapters.factory import build_vector_store
|
|
|
|
stats = sync_canonical_records(
|
|
"schema_records",
|
|
records,
|
|
store=build_vector_store(cfg, require_write=True),
|
|
embedder=make_embedder(cfg.embeddings),
|
|
)
|
|
_print_stats(stats)
|