feat(harness): port memory/search/evidence/db/decision cmd (Onda 3.2)
5 cmd foglia portati con rename + drift fix: - memory_cmd: portato col modello registry INTATTO (TODO marker per il drop registry decisione spec 5 — task separato, richiede L2 per validare il rewrite su vectordb) - search_cmd: creata search_app sub-app (era funzione standalone in ChironeWp3), registrata come 'tht search find' - evidence_cmd, db_cmd, decision_cmd: port verbatim Drift fix decision_cmd: DECISION_MIN_PHASE.get(type,1) -> load_workflow().decision_min_phase(type). Check grep-per-file: ~15 residui nsp/PSD_SSL_CA nei messaggi utente fixati (nsp <cmd> -> tht <cmd>, nsp.yaml -> workspace yaml, PSD_SSL_CA -> THT_SSL_CA). Suite: 165 passed. tht --help ora mostra 10 sottocomandi.
This commit is contained in:
@@ -0,0 +1,182 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
|
||||
KIND_MAP = {
|
||||
"evidence": ["evidence"],
|
||||
"schema": ["schema_table", "schema_column"],
|
||||
"values": [], # solo LSH
|
||||
}
|
||||
|
||||
# Default di `--top` per le famiglie diverse da `schema` (numero di risultati). Per `schema`
|
||||
# `--top` indica il numero di TABELLE candidate ed e' configurabile via `search.top_schema_tables`
|
||||
# (recupero ancorato alle tabelle: di ognuna si rendono tutte le colonne + FK).
|
||||
DEFAULT_TOP_FALLBACK = 10
|
||||
|
||||
search_app = typer.Typer(help="Ricerca semantica (evidence/schema/values) nel vectorstore")
|
||||
|
||||
|
||||
@search_app.command("find")
|
||||
def search_cmd(
|
||||
keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."),
|
||||
config: Path = CONFIG_OPT,
|
||||
top: int | None = typer.Option(
|
||||
None, "--top",
|
||||
help="Max risultati; con --kind schema indica il numero di tabelle "
|
||||
"(default: 12 tabelle per schema, 10 altrimenti).",
|
||||
),
|
||||
kind: str = typer.Option(
|
||||
None, "--kind", help="Filtra per famiglia: evidence | schema | values."
|
||||
),
|
||||
explain: bool = typer.Option(False, "--explain", help="Mostra anche il testo matchato."),
|
||||
json_out: bool = typer.Option(
|
||||
False, "--json", help="Output JSON machine-readable per Pi (sopprime le tabelle a video)."
|
||||
),
|
||||
) -> None:
|
||||
"""Ricerca combinata LSH + pgvector con ranking RRF spiegabile."""
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
|
||||
from tht.lshindex import LshIndexError, load_index, query_index
|
||||
from tht.search import combined_search
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
if kind is not None and kind not in KIND_MAP:
|
||||
typer.secho(
|
||||
f"ERRORE: --kind sconosciuto: {kind} (validi: {', '.join(KIND_MAP)})",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
if top is None:
|
||||
top = cfg.search.top_schema_tables if kind == "schema" else DEFAULT_TOP_FALLBACK
|
||||
|
||||
lsh_hits = None
|
||||
try:
|
||||
lsh, minhashes, meta = load_index(
|
||||
cfg.paths.indexes / "lsh", name=cfg.database.db_schema
|
||||
)
|
||||
hits = query_index(lsh, minhashes, keyword, meta, top_n=top * 3)
|
||||
lsh_hits = [(h.table, h.column, h.value, h.score) for h in hits]
|
||||
except LshIndexError:
|
||||
if not json_out: # in JSON mode lo stdout resta puro: niente warning umano
|
||||
typer.secho(
|
||||
"ATTENZIONE: indice LSH assente, ricerca solo vettoriale "
|
||||
"(esegui `tht lsh build`).", fg=typer.colors.YELLOW,
|
||||
)
|
||||
|
||||
if kind == "schema":
|
||||
from tht.cli.schema_cmd import annotations_path, physical_path
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
from tht.mschema.render import to_mschema_text
|
||||
from tht.search import schema_tables
|
||||
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
candidates = combined_search(
|
||||
keyword=keyword, lsh_hits=lsh_hits,
|
||||
store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings),
|
||||
top=cfg.search.schema_chunk_pool, rrf_k=cfg.search.rrf_k,
|
||||
kinds=KIND_MAP["schema"],
|
||||
)
|
||||
ranked = schema_tables(candidates, top_tables=top)
|
||||
if not ranked:
|
||||
if json_out:
|
||||
typer.echo(json.dumps({"tables": [], "mschema": ""}, ensure_ascii=False))
|
||||
return
|
||||
typer.secho(f"Nessuna tabella candidata per '{keyword}'.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations = Annotations.from_yaml(annotations_path(cfg))
|
||||
selected = [t for t, _ in ranked]
|
||||
mschema = to_mschema_text(physical, annotations, tables=selected)
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(
|
||||
{
|
||||
"tables": [{"name": n, "rrf": round(s, 6)} for n, s in ranked],
|
||||
"mschema": mschema,
|
||||
},
|
||||
ensure_ascii=False, indent=2,
|
||||
))
|
||||
return
|
||||
|
||||
reviewer = Table(title=f"Tabelle candidate per '{keyword}' (top {top}, RRF)")
|
||||
reviewer.add_column("#", justify="right")
|
||||
reviewer.add_column("Tabella")
|
||||
reviewer.add_column("RRF", justify="right")
|
||||
for i, (name, score) in enumerate(ranked, start=1):
|
||||
reviewer.add_row(str(i), name, f"{score:.4f}")
|
||||
Console().print(reviewer)
|
||||
Console().print(mschema)
|
||||
return
|
||||
|
||||
if kind == "values":
|
||||
results = []
|
||||
kinds = None
|
||||
else:
|
||||
kinds = KIND_MAP.get(kind) if kind else None
|
||||
results = combined_search(
|
||||
keyword=keyword, lsh_hits=lsh_hits if kind != "evidence" else None,
|
||||
store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings),
|
||||
top=top, rrf_k=cfg.search.rrf_k, kinds=kinds,
|
||||
)
|
||||
|
||||
if kind == "values":
|
||||
if json_out:
|
||||
typer.echo(json.dumps(
|
||||
[{"table": t, "column": c, "value": v, "score": round(s, 6)}
|
||||
for t, c, v, s in (lsh_hits or [])[:top]],
|
||||
ensure_ascii=False, indent=2,
|
||||
))
|
||||
return
|
||||
if not lsh_hits:
|
||||
typer.secho("Nessun match LSH.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Match LSH per '{keyword}'")
|
||||
table.add_column("Tabella.Colonna")
|
||||
table.add_column("Valore")
|
||||
table.add_column("Score", justify="right")
|
||||
for t, c, v, s in lsh_hits[:top]:
|
||||
table.add_row(f"{t}.{c}", v, f"{s:.3f}")
|
||||
Console().print(table)
|
||||
return
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(
|
||||
[r.model_dump() for r in results], ensure_ascii=False, indent=2
|
||||
))
|
||||
return
|
||||
if not results:
|
||||
typer.secho(f"Nessun candidato per '{keyword}'.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Candidati per '{keyword}' (RRF, k={cfg.search.rrf_k})")
|
||||
table.add_column("Candidato")
|
||||
table.add_column("Tipo")
|
||||
table.add_column("Segnali")
|
||||
table.add_column("RRF", justify="right")
|
||||
table.add_column("Status")
|
||||
if explain:
|
||||
table.add_column("Testo")
|
||||
for r in results:
|
||||
signals = " · ".join(
|
||||
f"{name} #{s['rank']} ({s['score']})" for name, s in r.signals.items()
|
||||
)
|
||||
row = [r.label, r.kind, signals, f"{r.rrf:.4f}", r.status]
|
||||
if explain:
|
||||
row.append((r.content[:120] + "…") if len(r.content) > 120 else r.content)
|
||||
table.add_row(*row)
|
||||
Console().print(table)
|
||||
Reference in New Issue
Block a user