feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s

Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation.

Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
Codex
2026-09-10 10:31:34 +02:00
parent 8fe526dd6e
commit 82e2c91f42
168 changed files with 11914 additions and 1772 deletions
+43 -2
View File
@@ -23,6 +23,46 @@ from tht.evidence import (
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
@evidence_app.command("sources", hidden=True)
def sources_cmd(action: str, config: Path = CONFIG_OPT,
source_id: str | None = typer.Option(None), revision: str | None = typer.Option(None),
decision: str | None = typer.Option(None), actor: str = typer.Option("installation operator"),
json_output: bool = typer.Option(False, "--json")):
from tht.evidence.administration import ConsolidationError, source_action
try:
if action not in {"refresh", "decide"} or len(actor) > 256 or not actor.strip():
raise ValueError("Invalid source action")
if action == "refresh" and any(v is not None for v in (source_id, revision, decision)):
raise ValueError("Refresh does not accept decision options")
if action == "decide" and (decision not in {"keep", "replace"} or not source_id or not revision):
raise ValueError("A source decision requires source-id, revision and keep or replace")
payload = source_action(config, action=action, source_id=source_id, revision=revision,
decision=decision, actor=actor)
except Exception as error: # noqa: BLE001 - never expose connector/provider exception details
safe = isinstance(error, (ValueError, EvidencePreparationError, ConsolidationError))
_emit({"status": "failed", "code": "evidence_source_failed",
"error": str(error)[:1500] if safe else "Source acquisition or refinement failed; existing Evidence is preserved.",
"saved": isinstance(error, ConsolidationError) and error.saved}, json_output)
raise typer.Exit(1) from None
_emit(payload, json_output)
@evidence_app.command("admin", hidden=True)
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
from tht.evidence.administration import browse
try:
if os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() not in {"true", "1"}:
raise ValueError("Evidence administration requires an administrator")
payload = json.loads(config.read_text())
root = Path(payload["root"])
if not root.is_absolute() or root.name != workspace:
raise ValueError("Invalid workspace archive identity")
_emit(browse(root, payload.get("query", {})), True)
except (ValueError, OSError) as error:
_emit({"code": "evidence_unavailable", "message": str(error)[:1500]}, True)
raise typer.Exit(1) from None
def _canonical_worktree(workspace_root: Path) -> Path:
requested = workspace_root.absolute()
root = workspace_root.resolve()
@@ -123,7 +163,8 @@ def prepare_cmd(
) -> None:
"""Prepare changed Source Evidence without committing or publishing it."""
root = _canonical_worktree(workspace_root)
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
from tht.evidence.authoring import authoring_skill_path
skill_path = authoring_skill_path()
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
try:
try:
@@ -180,7 +221,7 @@ def migrate_cmd(
workspace_root: Path,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Rewrite legacy Curated units as table-free v3 Markdown without model calls."""
"""Convert legacy units to editable v4 Markdown and establish a local baseline."""
root = _canonical_worktree(workspace_root)
try:
report = migrate_workspace_evidence(root)
+307 -454
View File
@@ -1,502 +1,355 @@
# TODO (drop registry, decisione spec 5): questo modulo e' portato col modello
# registry intatto (load_registry/promote/update_record/delete_record). Le memory
# dovrebbero vivere SOLO nel vectordb (metadata arricchito con subject/detail/rationale
# in Onda 3.1). Riscrivere: promote -> upsert batch vectordb; list/show -> scan
# vectordb; delete -> metadata.status="superseded"; index/clear -> droppati.
# Task separato: la validazione richiede L2 (vectordb reale).
"""Thin command adapters for the authoritative Memory service."""
import json
import re
from contextlib import contextmanager
from pathlib import Path
import typer
from sqlalchemy.exc import OperationalError, ProgrammingError
from pydantic import ValidationError
from tht.cli._guards import (
require_server_profile,
require_vector_write_allowed,
)
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.session_cmd import load_snapshot_or_exit
from tht.cli.vector_cmd import require_vector_cfg
from tht.memory.models import CardInput, CardQuery, MemoryError
from tht.memory.runtime import admin_service, memory_service
from tht.ports.vector import VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError
memory_app = typer.Typer(help="Review memory (registro canonico + indice semantico)")
DECISION_OPT = typer.Option(None, "--decision", help="Seq da promuovere (ripetibile).")
memory_app = typer.Typer(help="Authoritative Memory cards and verified recall")
DECISION_OPT = typer.Option(None, "--decision")
def registry_path(cfg) -> Path:
if getattr(cfg.paths, "memory", None) is not None:
return cfg.paths.memory / "registry.jsonl"
# Legacy location remains readable while old workspaces are retired.
return cfg.paths.artifacts / "memory" / "registry.jsonl"
def _output(value):
typer.echo(json.dumps(value, ensure_ascii=False, default=str))
def _resync_memory(cfg):
"""Risincronizza l'indice semantico col registro corrente (incrementale)."""
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder, sync_canonical_records
from tht.memory import load_registry, memory_vector_records
records = memory_vector_records(load_registry(registry_path(cfg)))
return sync_canonical_records(
"memory",
records,
store=build_vector_store(cfg, require_write=True),
embedder=make_embedder(cfg.embeddings),
)
@memory_app.command("promote")
def promote_cmd(
session: str = typer.Option(..., "--session"),
decision: list[int] = DECISION_OPT,
preview: bool = typer.Option(False, "--preview", help="Mostra i candidati in JSON, non scrive."),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Promuove le decisioni SCELTE nel registro globale. Usa --preview per vedere i candidati."""
import json as _json
from tht.memory import promote_snapshot
cfg = _load_config_or_exit(config)
snapshot = load_snapshot_or_exit(cfg, session)
if preview:
from tht.memory import (
MAX_PROMOTION_CANDIDATES,
preview_promotions_snapshot,
reusable_promotions_snapshot,
)
cand = preview_promotions_snapshot(snapshot, registry_path(cfg))
extra = len(reusable_promotions_snapshot(snapshot, registry_path(cfg))) - len(cand)
payload = [
{"decision_seq": c.decision_seq, "type": c.type, "subject": c.subject,
"detail": c.detail, "rationale": c.rationale,
"question_context": c.question_context,
"tables": c.tables, "concepts": c.concepts}
for c in cand
]
if json_out:
typer.echo(_json.dumps(payload, ensure_ascii=False, indent=2))
elif not payload:
typer.secho("Nessun candidato da promuovere.", fg=typer.colors.YELLOW)
else:
for c in payload:
typer.echo(f" [{c['decision_seq']}] {c['type']}: {c['subject']}")
if extra > 0:
typer.secho(
f"NOTA: mostrati {len(cand)} candidati su {len(cand) + extra} riusabili "
f"(cap {MAX_PROMOTION_CANDIDATES}); gli altri non sono proposti.",
fg=typer.colors.YELLOW, err=True,
)
return
if not decision:
typer.secho("ERRORE: indica le decisioni con --decision <seq> (vedi `--preview`).",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
require_server_profile(cfg, "memory promote")
require_vector_cfg(cfg)
promoted = promote_snapshot(snapshot, seqs=list(decision), registry_path=registry_path(cfg))
if not promoted:
msg = "Nessuna nuova promozione (gia' presenti o seq inesistenti)."
if json_out:
typer.echo(_json.dumps(
{"promoted": [], "indexed": False, "message": msg}, ensure_ascii=False))
else:
typer.secho(msg, fg=typer.colors.YELLOW)
return
# Promozione nel registro: riuscita. L'indicizzazione semantica puo' fallire
# (runtime non pronto o vectordb irraggiungibile da questa postazione): in quel
# caso le memorie restano nel registro ma NON sono trovate da `tht memory
# search` finche' non si reindicizza sul server. `indexed` rende lo stato
# leggibile da Pi, cosi' il reviewer lo vede invece di perderlo nello stderr.
ids = [{"id": r.id, "type": r.type, "subject": r.subject} for r in promoted]
indexed = True
warning = None
@contextmanager
def _service(config):
service = None
try:
_resync_memory(cfg)
except (ProgrammingError, OperationalError):
indexed = False
warning = (
f"{len(promoted)} memorie promosse nel registro, ma l'indice vettoriale "
"NON e' stato sincronizzato (runtime vettoriale mancante o irraggiungibile): "
"NON saranno trovate da `tht memory search` finche' non reindicizzi sul "
"server (`tht memory index` quando il runtime vettoriale è disponibile)."
)
if json_out:
typer.echo(_json.dumps(
{"promoted": ids, "indexed": indexed,
"message": warning or f"{len(promoted)} memorie promosse e indicizzate."},
ensure_ascii=False, indent=2))
return
for r in promoted:
typer.echo(f" {r.id}: {r.type} {r.subject}")
if indexed:
typer.secho(f"OK: {len(promoted)} memorie promosse e indicizzate.",
fg=typer.colors.GREEN)
else:
typer.secho(f"ATTENZIONE: {warning}", fg=typer.colors.YELLOW)
cfg = _load_config_or_exit(config)
service = memory_service(cfg)
yield cfg, service
except MemoryError as error:
_output({"code": error.code, "message": str(error), "status": error.status})
raise typer.Exit(1) from None
except (ValidationError, ValueError):
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
raise typer.Exit(1) from None
except (VectorStoreError, EmbeddingsError, OSError):
_output({"code": "memory_unavailable", "message": "Memory operation is unavailable",
"status": 503})
raise typer.Exit(1) from None
finally:
if service:
service.close()
@memory_app.command("save-one")
def save_one_cmd(
session: str = typer.Option(..., "--session"),
decision: int = typer.Option(
..., "--decision", help="decision_seq della decisione da salvare come memoria."
),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Upsert mirato (una riga) della memoria di una decisione nel semantic store (D11).
Promuove la decisione nel registro locale (idempotente) e fa un singolo upsert
con dedup hash client-side -- niente full-resync. Il factory seleziona il writer
del runtime vettoriale attivo.
"""
import json as _json
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder
from tht.memory import load_registry, promote_snapshot, save_one_memory
cfg = _load_config_or_exit(config)
snapshot = load_snapshot_or_exit(cfg, session)
require_vector_write_allowed(cfg, "memory save-one")
store = build_vector_store(cfg, require_write=True)
# Promuove la decisione scelta nel registro locale (idempotente: salta se gia' presente
# o se stale post-rollback, perche' _compute_promotions usa la vista effective).
promote_snapshot(snapshot, seqs=[decision], registry_path=registry_path(cfg))
records = [r for r in load_registry(registry_path(cfg)) if r.session_id == snapshot.manifest.id]
embedder = make_embedder(cfg.embeddings)
count = save_one_memory(records, decision, store=store, embedder=embedder)
msg = (
f"{count} memoria salvata nell'indice semantico (decision_seq {decision})."
if count
else f"Nessun upsert (decisione {decision} assente/stale o memoria gia' aggiornata)."
)
if json_out:
typer.echo(_json.dumps(
{"upserted": count, "decision_seq": decision, "message": msg}, ensure_ascii=False))
return
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN if count else typer.colors.YELLOW)
@memory_app.command("index")
def index_cmd(config: Path = CONFIG_OPT) -> None:
"""Sincronizza il registro memory nell'indice semantico (full-resync)."""
from tht.cli.vector_cmd import _print_stats
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "memory index")
require_vector_cfg(cfg)
_print_stats(_resync_memory(cfg))
@memory_app.command("admin")
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
"""Backend-owned request snapshot. No DWH or session configuration is required."""
service = None
try:
if not re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace):
raise ValueError("Invalid workspace")
payload = json.loads(config.read_text())
service = admin_service(workspace, payload["runtime"])
request = payload.get("request", {})
action = payload["action"]
if action == "cleanup":
from tht.memory.cleanup import CleanupRequest, cleanup
result = cleanup(service, CleanupRequest.model_validate(request))
elif action == "list":
result = service.list(CardQuery.model_validate(request))
elif action == "show":
result = service.get(request["id"])
elif action in {"create", "update"}:
result = service.save(CardInput.model_validate(request["card"]),
request.get("id") if action == "update" else None)
elif action == "delete":
result = service.delete(request["id"])
elif action == "pending":
result = service.pending()
elif action == "retry":
result = service.retry(request["id"])
else:
raise ValueError("Unknown Memory action")
_output(result)
except MemoryError as error:
_output({"code": error.code, "message": str(error), "status": error.status})
raise typer.Exit(1) from None
except (ValueError, KeyError, TypeError, OSError):
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
raise typer.Exit(1) from None
finally:
if service:
service.close()
@memory_app.command("list")
def list_cmd(
type_: str = typer.Option(None, "--type", help="Filtra per tipo decisione."),
session: str = typer.Option(None, "--session", help="Filtra per sessione."),
table: str = typer.Option(None, "--table", help="Filtra per tabella coinvolta."),
concept: str = typer.Option(None, "--concept", help="Filtra per concetto."),
json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT,
) -> None:
"""Elenca le memorie del registro (filtri combinati in AND)."""
import json as _json
from rich.console import Console
from rich.table import Table
from tht.memory import load_registry
cfg = _load_config_or_exit(config)
recs = load_registry(registry_path(cfg))
if type_:
recs = [r for r in recs if r.type == type_]
if session:
recs = [r for r in recs if r.session_id == session]
if table:
recs = [r for r in recs if table in r.tables]
if concept:
recs = [r for r in recs if concept in r.concepts]
if json_out:
typer.echo(_json.dumps([r.model_dump(mode="json") for r in recs],
ensure_ascii=False, indent=2))
return
if not recs:
typer.secho("Nessuna memoria nel registro.", fg=typer.colors.YELLOW)
return
t = Table(title="Review memory")
for col in ("Id", "Tipo", "Soggetto", "Sessione"):
t.add_column(col)
for r in recs:
t.add_row(r.id, r.type, r.subject, r.session_id)
Console().print(t)
def list_cmd(filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.list(CardQuery.model_validate_json(filters)))
@memory_app.command("show")
def show_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT,
) -> None:
"""Mostra una singola memoria."""
import json as _json
def show_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.get(mem_id))
from tht.memory import load_registry
cfg = _load_config_or_exit(config)
rec = {r.id: r for r in load_registry(registry_path(cfg))}.get(mem_id)
if rec is None:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
if json_out:
typer.echo(_json.dumps(rec.model_dump(mode="json"), ensure_ascii=False, indent=2))
return
for k, v in rec.model_dump(mode="json").items():
typer.echo(f"{k}: {v}")
@memory_app.command("create")
def create_cmd(data: Path = typer.Option(..., "--data"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.save(CardInput.model_validate_json(data.read_text())))
@memory_app.command("update")
def update_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
subject: str = typer.Option(None, "--subject"),
type_: str = typer.Option(None, "--type"),
detail: str = typer.Option(None, "--detail"),
rationale: str = typer.Option(None, "--rationale"),
question_context: str = typer.Option(None, "--question-context"),
tables: str = typer.Option(None, "--tables", help="CSV; \"\" per azzerare."),
concepts: str = typer.Option(None, "--concepts", help="CSV; \"\" per azzerare."),
config: Path = CONFIG_OPT,
) -> None:
"""Modifica i campi di merito di una memoria (provenienza immutabile)."""
from typing import get_args
from tht.decisions import DecisionType
from tht.memory import MemoryNotFound, update_record
cfg = _load_config_or_exit(config)
fields: dict = {}
for name, val in (("subject", subject), ("type", type_), ("detail", detail),
("rationale", rationale), ("question_context", question_context)):
if val is not None:
fields[name] = val
if tables is not None:
fields["tables"] = [t.strip() for t in tables.split(",") if t.strip()]
if concepts is not None:
fields["concepts"] = [c.strip() for c in concepts.split(",") if c.strip()]
if not fields:
typer.secho("ERRORE: nessun campo da modificare indicato.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
if "type" in fields and fields["type"] not in get_args(DecisionType):
typer.secho(f"ERRORE: tipo '{fields['type']}' non valido.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
require_server_profile(cfg, "memory update")
require_vector_cfg(cfg)
try:
rec = update_record(registry_path(cfg), mem_id, fields)
except MemoryNotFound:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
_resync_memory(cfg)
typer.secho(f"OK: {rec.id} aggiornata e reindicizzata.", fg=typer.colors.GREEN)
def update_cmd(mem_id: str, data: Path = typer.Option(..., "--data"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.save(CardInput.model_validate_json(data.read_text()), mem_id))
@memory_app.command("delete")
def delete_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
yes: bool = typer.Option(False, "--yes", "-y", help="Salta la conferma."),
config: Path = CONFIG_OPT,
) -> None:
"""Cancella una singola memoria (registro + indice)."""
from tht.memory import MemoryNotFound, delete_record
def delete_cmd(mem_id: str, yes: bool = typer.Option(False, "--yes", "-y"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
service._admin()
if not yes and not typer.confirm(f"Delete Memory card {mem_id} and its links?"):
raise typer.Exit(1)
_output(service.delete(mem_id))
cfg = _load_config_or_exit(config)
require_server_profile(cfg, "memory delete")
require_vector_cfg(cfg)
if not yes and not typer.confirm(f"Cancellare definitivamente la memoria '{mem_id}'?"):
typer.secho("Annullato.", fg=typer.colors.YELLOW)
raise typer.Exit(code=1)
try:
delete_record(registry_path(cfg), mem_id)
except MemoryNotFound:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
_resync_memory(cfg)
typer.secho(f"OK: {mem_id} cancellata e deindicizzata.", fg=typer.colors.GREEN)
@memory_app.command("pending")
def pending_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.pending())
@memory_app.command("retry")
def retry_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.retry(mem_id))
@memory_app.command("index")
def index_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.rebuild())
@memory_app.command("promote")
def promote_cmd(session: str = typer.Option(..., "--session"), decision: list[int] = DECISION_OPT,
preview: bool = typer.Option(False, "--preview"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
if preview:
_output(service.promotions(snapshot)[:5])
elif not decision:
raise ValueError("Explicit decisions are required")
else:
results = service.promote(snapshot, decision)
_output({"promoted": [r.get("card") for r in results],
"indexed": all(r["indexed"] for r in results), "results": results})
@memory_app.command("propose")
def propose_cmd(session: str = typer.Option(..., "--session"),
data: Path = typer.Option(..., "--data"), config: Path = CONFIG_OPT):
"""Persist reviewer-grounded proposals without changing the Memory archive."""
from tht.cli.session_cmd import session_repository
from tht.memory.review import validate_proposals
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
service._session(snapshot)
proposals = validate_proposals(snapshot, json.loads(data.read_text()))
session_repository(cfg).write_artifact(session, "memory_proposals",
json.dumps([p.model_dump(mode="json") for p in proposals], ensure_ascii=False))
_output({"proposals": len(proposals), "saved_to_archive": False})
@memory_app.command("summary")
def summary_cmd(session: str = typer.Option(..., "--session"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.memory.review import prepare
with _service(config) as (cfg, service):
_output(prepare(service, load_snapshot_or_exit(cfg, session)))
@memory_app.command("repair-prepare")
def repair_prepare_cmd(session: str = typer.Option(..., "--session"),
proposal_json: str = typer.Option(..., "--proposal-json"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import RepairProposal, prepare
if len(proposal_json.encode()) > 1_000_000:
_output({"code": "memory_invalid", "message": "Repair proposal is too large", "status": 400})
raise typer.Exit(1)
with _service(config) as (cfg, service):
_output(prepare(service, load_snapshot_or_exit(cfg, session), cfg,
RepairProposal.model_validate_json(proposal_json)))
@memory_app.command("repair-target")
def repair_target_cmd(session: str = typer.Option(..., "--session"),
archive: str = typer.Option(..., "--archive"),
target_id: str = typer.Option(..., "--target-id"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import target
with _service(config) as (cfg, service):
_output(target(service, load_snapshot_or_exit(cfg, session), cfg, archive, target_id))
@memory_app.command("repairs")
def repairs_cmd(session: str = typer.Option(..., "--session"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import list_repairs
with _service(config) as (cfg, service):
_output(list_repairs(service, load_snapshot_or_exit(cfg, session)))
@memory_app.command("repair-show")
def repair_show_cmd(session: str = typer.Option(..., "--session"),
repair_id: str = typer.Option(..., "--repair-id"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import show
with _service(config) as (cfg, service):
_output(show(service, load_snapshot_or_exit(cfg, session), cfg, repair_id))
@memory_app.command("repair-apply")
def repair_apply_cmd(session: str = typer.Option(..., "--session"),
repair_id: str = typer.Option(..., "--repair-id"),
choice: str = typer.Option(..., "--choice"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import apply
from tht.cli.preprocess_cmd import run_from_config
def activate(snapshot):
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise RuntimeError("Evidence activation did not complete")
with _service(config) as (cfg, service):
_output(apply(service, load_snapshot_or_exit(cfg, session), cfg, repair_id, choice,
activate=activate))
@memory_app.command("review-apply")
def review_apply_cmd(session: str = typer.Option(..., "--session"),
review_json: str = typer.Option(..., "--review-json"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.memory.review import ReviewResponse, apply
with _service(config) as (cfg, service):
_output(apply(service, load_snapshot_or_exit(cfg, session),
ReviewResponse.model_validate_json(review_json)))
@memory_app.command("save-one")
def save_one_cmd(session: str = typer.Option(..., "--session"),
decision: int = typer.Option(..., "--decision"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (cfg, service):
results = service.promote(load_snapshot_or_exit(cfg, session), [decision])
_output({"upserted": sum(r["indexed"] for r in results), "decision_seq": decision,
"indexed": all(r["indexed"] for r in results), "results": results})
@memory_app.command("search")
def search_cmd(
question: str = typer.Argument(..., help="Domanda o termini di ricerca."),
top: int = typer.Option(5, "--top"),
session: str = typer.Option(
None, "--session",
help="Esclude le memorie gia' decise (applicate o rifiutate) in questa sessione.",
),
json_out: bool = typer.Option(False, "--json", help="Output JSON per Pi."),
config: Path = CONFIG_OPT,
) -> None:
"""Cerca memorie riapplicabili, ordinate per similarita'. Mai applicate in automatico.
def search_cmd(question: str, top: int = typer.Option(5, "--top"),
session: str | None = typer.Option(None, "--session"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.cli.vector_cmd import make_embedder, open_searcher
with _service(config) as (cfg, service):
decisions = load_snapshot_or_exit(cfg, session).decisions if session else []
_output(service.recall(question, searcher=open_searcher(cfg),
embedder=make_embedder(cfg.embeddings), top=top, decisions=decisions,
scope=_recall_scope(cfg, filters)))
Con `--session` non ripropone le memorie gia' decise in quella sessione (fix:
memorie scartate riproposte): rifiutate via `memory_rejected` o gia' applicate."""
from rich.console import Console
from rich.table import Table
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
from tht.memory import load_registry, recall_memories
def _recall_scope(cfg, filters):
from tht.memory.retrieval import RecallScope
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
decisions = load_snapshot_or_exit(cfg, session).decisions if session is not None else []
searcher = open_searcher(cfg)
embedder = make_embedder(cfg.embeddings)
results = recall_memories(
question,
records=load_registry(registry_path(cfg)),
decisions=decisions,
searcher=searcher,
embedder=embedder,
top=top,
)
context = {"database": cfg.database.database, "schema_name": cfg.database.db_schema}
supplied = json.loads(filters)
if not isinstance(supplied, dict) or any(
key in supplied and supplied[key] != value for key, value in context.items()
):
raise ValueError("Recall cannot override the configured database/schema context")
return RecallScope.model_validate({**supplied, **context})
if json_out:
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
return
if not results:
typer.secho("Nessuna memoria candidata.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Memorie candidate per: {question}")
table.add_column("Id")
table.add_column("Tipo")
table.add_column("Soggetto")
table.add_column("Contesto originale")
table.add_column("Score", justify="right")
for r in results:
table.add_row(r["id"], r["type"], r["subject"],
r["question_context"][:60], f"{r['score']:.3f}")
Console().print(table)
@memory_app.command("rules")
def rules_cmd(question: str, session: str = typer.Option(..., "--session"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
"""Consult SQL rules and explained errors in schema linking and SQL construction."""
from types import SimpleNamespace
from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.phase import current_phase
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
service._session(snapshot)
if current_phase(snapshot) not in {4, 6, 7}:
raise ValueError("Memory rules are consulted in schema linking or SQL construction")
scope = _recall_scope(cfg, filters)
vector = make_embedder(cfg.embeddings).embed_query(question)
embedder = SimpleNamespace(embed_query=lambda _: vector)
searcher = open_searcher(cfg)
candidates = []
for family in ("sql_rule", "explained_error"):
candidates.extend(service.retrieve(question, searcher=searcher, embedder=embedder,
scope=scope, family=family, top=5))
_output([{**candidate.card.model_dump(mode="json"), "score": candidate.score,
"retrieval_path": candidate.path, "consultative": True}
for candidate in sorted(candidates, key=lambda c: (-c.score, c.card.id))[:10]])
def index_solved_session(cfg, session_id: str) -> int:
"""Indicizza la coppia domanda->SQL della sessione (kind solved_question).
Solleva SolvedIndexError se mancano gli artefatti: il finalize lo degrada a warning,
il comando CLI lo converte in errore esplicito."""
from tht.adapters.factory import build_vector_store
from tht.cli.sql_cmd import promoted_tables_for
from tht.cli.vector_cmd import make_embedder
from tht.memory import index_solved_question
store = build_vector_store(cfg, require_write=True)
return index_solved_question(
load_snapshot_or_exit(cfg, session_id),
promoted_tables_for(cfg, session_id),
store=store,
embedder=make_embedder(cfg.embeddings),
)
"""Recovery only: never recreate a deleted card from historical session artifacts."""
service = memory_service(cfg)
try:
result = service.retry_solved(load_snapshot_or_exit(cfg, session_id))
if not result["indexed"]:
raise RuntimeError(result["error"])
return int(result["action"] == "upsert")
finally:
service.close()
@memory_app.command("solved-index")
def solved_index_cmd(
session_id: str = typer.Argument(..., help="Id sessione con sql_final.sql approvato."),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Indicizza la coppia domanda->SQL nel semantic store (backfill; il finalize lo fa da solo)."""
import json as _json
from tht.memory import SolvedIndexError
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "memory solved-index")
try:
count = index_solved_session(cfg, session_id)
except RuntimeError as e:
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=4)
except SolvedIndexError as e:
typer.secho(f"ERRORE: sessione {session_id} non indicizzabile: {e}",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=3)
msg = (
f"1 coppia domanda->SQL indicizzata (solved:{session_id})."
if count else "Nessun upsert: coppia gia' aggiornata."
)
if json_out:
typer.echo(_json.dumps({"upserted": count, "id": f"solved:{session_id}"},
ensure_ascii=False))
return
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN)
def solved_index_cmd(session_id: str, json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT):
"""Retry an existing authoritative exemplar's Qdrant projection; never import a session."""
with _service(config) as (cfg, service):
_output(service.retry_solved(load_snapshot_or_exit(cfg, session_id)))
@memory_app.command("solved-search")
def solved_search_cmd(
question: str = typer.Argument(..., help="Domanda da confrontare con quelle risolte."),
top: int = typer.Option(3, "--top"),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Domande gia' risolte simili (kind solved_question): domanda, SQL e tabelle."""
from rich.console import Console
from rich.table import Table
def solved_search_cmd(question: str, top: int = typer.Option(3, "--top"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.memory import search_solved_questions
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
# Degrado gentile: SKILL.md prescrive solved-search in F4/F6/F7 di ogni sessione,
# quindi vectordb/Ollama irraggiungibili non devono produrre un traceback grezzo
# nel transcript: avviso di una riga su stderr, stdout puro ([] in --json), exit 0.
try:
searcher = open_searcher(cfg)
embedder = make_embedder(cfg.embeddings)
results = search_solved_questions(
question,
searcher=searcher,
embedder=embedder,
top=top,
)
except (VectorStoreError, VectorReadUnavailable, EmbeddingsError, OperationalError) as e:
typer.secho(
f"ATTENZIONE: exemplar non disponibili ({e}). Prosegui senza.",
fg=typer.colors.YELLOW, err=True,
)
if json_out:
typer.echo("[]")
return
if json_out:
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
return
if not results:
typer.secho("Nessuna domanda risolta simile.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Domande risolte simili a: {question}")
table.add_column("Sessione")
table.add_column("Domanda")
table.add_column("Tabelle")
table.add_column("Score", justify="right")
for r in results:
table.add_row(r["session_id"], r["question"][:60],
", ".join(r["tables"]), f"{r['score']:.3f}")
Console().print(table)
with _service(config) as (cfg, service):
try:
result = service.recall(question, searcher=open_searcher(cfg),
embedder=make_embedder(cfg.embeddings), top=top, solved=True,
scope=_recall_scope(cfg, filters))
except (VectorStoreError, EmbeddingsError):
typer.echo("Avviso: exemplar non disponibili; ricerca semantica non riuscita.", err=True)
result = []
_output(result)
+30 -6
View File
@@ -50,6 +50,10 @@ def _evaluation_workspace_root(cfg) -> Path:
def _requires_candidate_evaluation(cfg) -> bool:
evidence = cfg.evidence
if evidence and evidence.local_archive_root and (
evidence.local_archive_root / "evidence/.local/state.yaml"
).exists():
return False
if evidence is None or evidence.schema_version != 2:
return False
return evidence.source_root is not None or any(
@@ -224,7 +228,8 @@ def _parse_dwh_steps(value: str) -> tuple[str, ...]:
return steps
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None):
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None,
local_snapshot: Path | None = None):
from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder
@@ -239,8 +244,11 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
corpus_root = cfg.paths.artifacts.parent / "corpus"
vector_store = build_vector_store(cfg, require_write=True)
embedder = make_embedder(cfg.embeddings)
from tht.evidence.adapters import FilesystemEvidenceSource
sources = [FilesystemEvidenceSource(local_snapshot, patterns=("curated/**/*.md",))] \
if local_snapshot is not None else build_sources(cfg.evidence)
pipeline = build_preprocessing_pipeline(
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
store=CorpusStore(corpus_root), sources=sources,
embedder=embedder,
vector_store=vector_store,
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
@@ -376,9 +384,17 @@ def evidence_cmd(
dry_run: bool = typer.Option(False, "--dry-run"),
resume: str | None = typer.Option(None, "--resume"),
json_output: bool = typer.Option(False, "--json"),
consolidate: bool = typer.Option(False, "--consolidate"),
) -> None:
if action is not None and action != "gc":
raise typer.BadParameter("only the optional 'gc' action is supported")
if consolidate and (dry_run or resume is not None or action is not None):
message = "Consolidation cannot be combined with dry-run, resume or gc"
if json_output:
typer.echo(json.dumps({"status": "failed", "code": "invalid_consolidation", "error": message}))
else:
typer.secho(message, fg=typer.colors.RED, err=True)
raise typer.Exit(code=2)
if action == "gc":
try:
payload = gc_from_config(config, dry_run=dry_run)
@@ -406,21 +422,29 @@ def evidence_cmd(
typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True)
raise typer.Exit(code=2)
try:
result = run_from_config(config, dry_run=dry_run, resume=resume)
except Exception: # noqa: BLE001
if consolidate:
from tht.evidence.administration import consolidate_from_config
result = consolidate_from_config(config)
else:
result = run_from_config(config, dry_run=dry_run, resume=resume)
except Exception as error: # noqa: BLE001
from tht.evidence.administration import ConsolidationError
detail = str(error) if isinstance(error, ConsolidationError) else "preprocessing failed"
payload = {"status": "failed"}
if isinstance(error, ConsolidationError):
payload["saved"] = error.saved
if json_output:
typer.echo(json.dumps(
_evidence_json_payload(
cfg,
payload,
code="preprocessing_failed",
error="preprocessing failed",
error=detail,
),
sort_keys=True,
))
else:
typer.secho("ERRORE: preprocessing failed", fg=typer.colors.RED, err=True)
typer.secho(f"ERRORE: {detail}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = result.model_dump(mode="json")
if payload.get("status") != "succeeded":
+1 -38
View File
@@ -624,44 +624,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
repository, session_id, validation_report=report, evidence=evidence
)
# --- memoria attiva (parte B): indicizza la coppia domanda->SQL, best-effort ---
# Qualunque errore (writer key assente, VPN giu', Ollama spento) NON deve
# bloccare il finalize: l'indice e' derivato e recuperabile con
# `tht memory solved-index <id>`. Memory owns this best-effort policy; core
# has already committed the authoritative finalized snapshot above.
try:
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder
from tht.memory import index_solved_question_best_effort
finalized_snapshot = repository.get(session_id)
outcome = index_solved_question_best_effort(
finalized_snapshot,
promoted_tables,
store_factory=lambda: build_vector_store(cfg, require_write=True),
embedder_factory=lambda: make_embedder(cfg.embeddings),
)
if outcome.error is not None:
typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({outcome.error}). "
f"Recupera con `tht memory solved-index {session_id}`.",
fg=typer.colors.YELLOW, err=True,
)
elif outcome.upserted:
typer.secho(
"OK: coppia domanda->SQL indicizzata nel vectordb (solved_question).",
fg=typer.colors.GREEN,
)
else:
typer.secho(
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
fg=typer.colors.CYAN,
)
except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
f"Recupera con `tht memory solved-index {session_id}`.",
fg=typer.colors.YELLOW, err=True,
)
# Memory cards, including exemplars, are saved only by the explicit final review.
typer.secho(f"OK: sessione {session_id} finalizzata. Artefatti:", fg=typer.colors.GREEN)
for name in ARTIFACT_FILES:
state = "presente" if name not in {"session_manifest.yaml", "review_decisions.jsonl"} else "persistito"