Files
ThothII/harness/tht/cli/search_cmd.py
T
marcopanandClaude Opus 4.8 c4d130828f fix(harness): remediation difetti review — gate↔CLI, D15, D7/D6, D14, robustezza
Implementazione del piano di remediation progressiva sui difetti emersi
dall'analisi dell'harness. Tutto verificato: 214 test Python (incl. L0 su
Postgres reale), 14 test JS del gate, ruff pulito.

Blocco 1 (CRITICA, integrazione gate↔CLI):
- phase advance: gate usa --auto + exit 6; reviewer_confirm kind:phase fa
  advance esplicito che applica i prerequisiti (prima non avanzava per le
  fasi a conferma umana).
- cte plan riceve i --name dal gate (param names); set-question con id
  posizionale; skill `tht search find`; nuovo comando `tht memory save-one`
  con dedup hash client-side in save_one_memory.

Blocco 2 (D15, stato post-rollback):
- campo `phase` su DecisionRecord + effective_decisions phase-aware per i
  subject "a nome" (cte_approved ecc.); _compute_promotions e finalize sulla
  vista effective; finalize confronta col piano CTE effettivo, non glob;
  `decision add --retracts` + comando `decision retract`.

Blocco 3 (D7 read-only + D6 manifest):
- assert_read_only su tutti e quattro i codepath (direct + REST);
- manifest author/summary/updated_at/updated_by/schema_version popolati +
  helper touch_manifest sulle mutazioni.

Blocco 4-5 (D14a/D14b):
- decision_min_phase data-driven via `emits:` in workflow.yaml;
- formula evidence: status auto, search_formulas, gruppo CLI `tht formula`,
  `search find --kind formula`, load_evidence_dir salta i .sql.md.

Blocco 6 (robustezza):
- taskdoc slice promoted_tables + bound enforced; report escaping/bound +
  rsplit note; filtro kind reader REST/direct; conteggio upserted robusto;
  guard REST run_query non-list; LSH disallineato -> LshIndexError.

Blocco 7 (pulizia):
- dead code gate e KIND_TO_TABLE morto rimossi; doc Postgres-only
  (README + connection.py).

Blocco 0 (parziale): test di compatibilità firma gate↔CLI
(tests/integration). Rinviati: fake-Pi runtime completo, artifact-gate da
disco (#23), parità eligibility REST/direct (#28), unificazione
reserved-labels (#30), memory_rejected da deselezione (#33).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-27 17:16:51 +02:00

208 lines
7.9 KiB
Python

import json
from pathlib import Path
import typer
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit
KIND_MAP = {
"evidence": ["evidence"],
"schema": ["schema_table", "schema_column"],
"values": [], # solo LSH
"formula": [], # solo formula store (D14b), niente LSH/vector
}
# Default di `--top` per le famiglie diverse da `schema` (numero di risultati). Per `schema`
# `--top` indica il numero di TABELLE candidate ed e' configurabile via `search.top_schema_tables`
# (recupero ancorato alle tabelle: di ognuna si rendono tutte le colonne + FK).
DEFAULT_TOP_FALLBACK = 10
search_app = typer.Typer(help="Ricerca semantica (evidence/schema/values) nel vectorstore")
@search_app.command("find")
def search_cmd(
keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."),
config: Path = CONFIG_OPT,
top: int | None = typer.Option(
None, "--top",
help="Max risultati; con --kind schema indica il numero di tabelle "
"(default: 12 tabelle per schema, 10 altrimenti).",
),
kind: str = typer.Option(
None, "--kind", help="Filtra per famiglia: evidence | schema | values | formula."
),
explain: bool = typer.Option(False, "--explain", help="Mostra anche il testo matchato."),
json_out: bool = typer.Option(
False, "--json", help="Output JSON machine-readable per Pi (sopprime le tabelle a video)."
),
) -> None:
"""Ricerca combinata LSH + pgvector con ranking RRF spiegabile."""
from rich.console import Console
from rich.table import Table
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
from tht.lshindex import LshIndexError, load_index, query_index
from tht.search import combined_search
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
if kind is not None and kind not in KIND_MAP:
typer.secho(
f"ERRORE: --kind sconosciuto: {kind} (validi: {', '.join(KIND_MAP)})",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
if top is None:
top = cfg.search.top_schema_tables if kind == "schema" else DEFAULT_TOP_FALLBACK
if kind == "formula":
# D14b: recupero formule di concetto dallo store locale (niente LSH/vector).
from tht.cli.evidence_cmd import evidence_root
from tht.evidence.formula_store import search_formulas
formulas = search_formulas(evidence_root(cfg), keyword)[:top]
if json_out:
typer.echo(json.dumps(
[f.model_dump(mode="json") for f in formulas], ensure_ascii=False, indent=2))
return
if not formulas:
typer.secho(f"Nessuna formula per '{keyword}'.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Formule per '{keyword}'")
table.add_column("Concetto")
table.add_column("Status")
table.add_column("Colonne")
table.add_column("SQL")
for f in formulas:
sql_preview = (f.sql[:80] + "…") if len(f.sql) > 80 else f.sql
table.add_row(f.concept, f.status, ", ".join(f.columns), sql_preview)
Console().print(table)
return
lsh_hits = None
try:
lsh, minhashes, meta = load_index(
cfg.paths.indexes / "lsh", name=cfg.database.db_schema
)
hits = query_index(lsh, minhashes, keyword, meta, top_n=top * 3)
lsh_hits = [(h.table, h.column, h.value, h.score) for h in hits]
except LshIndexError:
if not json_out: # in JSON mode lo stdout resta puro: niente warning umano
typer.secho(
"ATTENZIONE: indice LSH assente, ricerca solo vettoriale "
"(esegui `tht lsh build`).", fg=typer.colors.YELLOW,
)
if kind == "schema":
from tht.cli.schema_cmd import annotations_path, physical_path
from tht.mschema.models import Annotations, PhysicalSchema
from tht.mschema.render import to_mschema_text
from tht.search import schema_tables
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
candidates = combined_search(
keyword=keyword, lsh_hits=lsh_hits,
store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings),
top=cfg.search.schema_chunk_pool, rrf_k=cfg.search.rrf_k,
kinds=KIND_MAP["schema"],
)
ranked = schema_tables(candidates, top_tables=top)
if not ranked:
if json_out:
typer.echo(json.dumps({"tables": [], "mschema": ""}, ensure_ascii=False))
return
typer.secho(f"Nessuna tabella candidata per '{keyword}'.", fg=typer.colors.YELLOW)
return
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
selected = [t for t, _ in ranked]
mschema = to_mschema_text(physical, annotations, tables=selected)
if json_out:
typer.echo(json.dumps(
{
"tables": [{"name": n, "rrf": round(s, 6)} for n, s in ranked],
"mschema": mschema,
},
ensure_ascii=False, indent=2,
))
return
reviewer = Table(title=f"Tabelle candidate per '{keyword}' (top {top}, RRF)")
reviewer.add_column("#", justify="right")
reviewer.add_column("Tabella")
reviewer.add_column("RRF", justify="right")
for i, (name, score) in enumerate(ranked, start=1):
reviewer.add_row(str(i), name, f"{score:.4f}")
Console().print(reviewer)
Console().print(mschema)
return
if kind == "values":
results = []
kinds = None
else:
kinds = KIND_MAP.get(kind) if kind else None
results = combined_search(
keyword=keyword, lsh_hits=lsh_hits if kind != "evidence" else None,
store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings),
top=top, rrf_k=cfg.search.rrf_k, kinds=kinds,
)
if kind == "values":
if json_out:
typer.echo(json.dumps(
[{"table": t, "column": c, "value": v, "score": round(s, 6)}
for t, c, v, s in (lsh_hits or [])[:top]],
ensure_ascii=False, indent=2,
))
return
if not lsh_hits:
typer.secho("Nessun match LSH.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Match LSH per '{keyword}'")
table.add_column("Tabella.Colonna")
table.add_column("Valore")
table.add_column("Score", justify="right")
for t, c, v, s in lsh_hits[:top]:
table.add_row(f"{t}.{c}", v, f"{s:.3f}")
Console().print(table)
return
if json_out:
typer.echo(json.dumps(
[r.model_dump() for r in results], ensure_ascii=False, indent=2
))
return
if not results:
typer.secho(f"Nessun candidato per '{keyword}'.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Candidati per '{keyword}' (RRF, k={cfg.search.rrf_k})")
table.add_column("Candidato")
table.add_column("Tipo")
table.add_column("Segnali")
table.add_column("RRF", justify="right")
table.add_column("Status")
if explain:
table.add_column("Testo")
for r in results:
signals = " · ".join(
f"{name} #{s['rank']} ({s['score']})" for name, s in r.signals.items()
)
row = [r.label, r.kind, signals, f"{r.rrf:.4f}", r.status]
if explain:
row.append((r.content[:120] + "…") if len(r.content) > 120 else r.content)
table.add_row(*row)
Console().print(table)