feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s

Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation.

Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
Codex
2026-09-10 10:31:34 +02:00
parent 8fe526dd6e
commit 82e2c91f42
168 changed files with 11914 additions and 1772 deletions
+78 -67
View File
@@ -80,9 +80,6 @@ class QdrantVectorStore:
if not legacy_constructor and collections["reference"] == collections["memory"]:
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
self._collections = dict(collections)
self._split_collections = collections["reference"] != collections["memory"]
self._legacy_collection = workspace_id
self._legacy_checked = False
self._workspace_id = workspace_id
self._workspace_revision = None
self._workspace_revision = workspace_revision
@@ -177,7 +174,13 @@ class QdrantVectorStore:
filter_must.extend(self._revision_filter(allowed_record_kinds))
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
filter_must.append({"key": "record_kind", "match": {"any": allowed_record_kinds}})
if metadata_filter is not None:
if metadata_filter is not None and "memory" in metadata_filter:
if set(metadata_filter) != {"memory"} or not set(allowed_record_kinds) <= {
"memory", "solved_question",
}:
raise VectorStoreError("Unsupported Memory metadata filter")
filter_must.extend(self._memory_filter(metadata_filter["memory"]))
elif metadata_filter is not None:
allowed_filters = {
"vector_generation", "document_ids", "workspace_id", "purpose",
"required_kinds", "required_concepts", "required_tables", "required_columns",
@@ -221,9 +224,9 @@ class QdrantVectorStore:
raise VectorStoreError("Invalid vector metadata filter")
filter_must.extend({"key": payload_key, "match": {"value": item}} for item in values)
if retrieval_mode not in {"fused", "dense", "bm25"}:
raise VectorStoreError("Evidence retrieval mode is invalid")
raise VectorStoreError("Vector retrieval mode is invalid")
if retrieval_mode == "dense":
if allowed_record_kinds != ["evidence"]:
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
response = self._call(
"POST",
@@ -236,7 +239,7 @@ class QdrantVectorStore:
},
)
elif retrieval_mode == "bm25":
if allowed_record_kinds != ["evidence"]:
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
@@ -267,8 +270,8 @@ class QdrantVectorStore:
},
)
else:
if allowed_record_kinds != ["evidence"]:
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
raise VectorStoreError("Hybrid BM25 is only available for Evidence and Memory")
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
@@ -323,16 +326,13 @@ class QdrantVectorStore:
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
validate_collection(collection)
# The first write with the split configuration is also the upgrade cutover. This keeps
# existing runtime memory reachable even when the operator reruns preprocessing without
# invoking the explicit clear operation first.
if self._split_collections:
self._migrate_legacy_memory()
# Memory is projected only from its authoritative archive. Never import legacy payloads.
physical_collection = self._physical_collection_for_logical(collection)
self._ensure_collection(
physical_collection,
strict=True,
require_bm25=any(record.sparse_text is not None for record in records),
maintain_bm25=collection == "memory",
)
points = []
for write_record in records:
@@ -341,8 +341,8 @@ class QdrantVectorStore:
semantic_kind = qdrant_semantic_kind(write_record.record.kind)
vector: list[float] | dict = write_record.embedding
if write_record.sparse_text is not None:
if semantic_kind != "evidence" or write_record.sparse_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 document is invalid")
if semantic_kind not in {"evidence", "memory"} or write_record.sparse_language not in _BM25_LANGUAGES:
raise VectorStoreError("BM25 document is invalid")
vector = {
"": write_record.embedding,
"bm25": self._bm25_document(write_record.sparse_text, write_record.sparse_language),
@@ -391,6 +391,26 @@ class QdrantVectorStore:
)
return before
def prepare_memory_index(self) -> None:
"""Explicit rebuild may recreate a lost collection; reads never do so."""
self._ensure_collection(self._collections["memory"], strict=True,
require_bm25=True, maintain_bm25=True, allow_create=True)
def delete_memory_records(self, record_keys: list[str]) -> None:
"""Delete exact authoritative Memory projections, never reference vectors."""
if not record_keys or any(not key.startswith("card:mem-") for key in record_keys):
raise VectorStoreError("Exact Memory card keys are required")
collection = self._collections["memory"]
if self._call("GET", f"/collections/{collection}", None, allow_missing=True) is None:
return
self._call("POST", f"/collections/{collection}/points/delete?wait=true", {
"filter": {"must": [
*self._workspace_filter(),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
{"key": "record_key", "match": {"any": record_keys}},
]},
})
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int:
if collection != "evidence" or _GENERATION.fullmatch(generation) is None:
raise VectorStoreError("Only exact Evidence generations may be deleted")
@@ -442,60 +462,14 @@ class QdrantVectorStore:
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
def clear_reference(self) -> bool:
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
legacy_deleted = self._migrate_legacy_memory()
"""Drop only replaceable schema/Evidence vectors; do not import legacy Memory."""
collection = self._collections["reference"]
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
return legacy_deleted
return False
self._call("DELETE", f"/collections/{collection}", None)
return True
def _migrate_legacy_memory(self) -> bool:
"""Preserve memory from the pre-split collection before retiring it."""
legacy = self._legacy_collection
if (
self._legacy_checked
or not self._split_collections
or legacy in self._collections.values()
):
return False
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
if response is None:
self._legacy_checked = True
return False
memory = self._collections["memory"]
self._ensure_collection(memory, strict=True, allow_create=True)
points = self._scroll(
legacy,
[
*self._workspace_filter(),
self._semantic_kind_filter(["memory", "solved_question"]),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
],
with_vector=True,
)
migrated = []
for point in points:
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
raise VectorStoreError("Qdrant returned malformed legacy memory response")
if not isinstance(point.get("payload"), dict):
raise VectorStoreError("Qdrant returned malformed legacy memory response")
migrated.append({
"id": point["id"],
"vector": point["vector"],
"payload": point["payload"],
})
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{memory}/points?wait=true",
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
)
self._call("DELETE", f"/collections/{legacy}", None)
self._legacy_checked = True
return True
def _physical_collection_for_logical(self, collection: str) -> str:
validate_collection(collection)
return self._collections["memory" if collection == "memory" else "reference"]
@@ -551,6 +525,34 @@ class QdrantVectorStore:
else "Embedding dimension does not match configured dimension"
)
@staticmethod
def _memory_filter(value: object) -> list[dict]:
fields = {"scope", "database", "schema_name", "table", "column"}
if not isinstance(value, dict) or set(value) != fields | {"family", "concepts", "format"}:
raise VectorStoreError("Invalid Memory metadata filter")
if (any(not isinstance(value[key], str) for key in fields)
or value["family"] is not None and not isinstance(value["family"], str)
or value["family"] not in {None, "domain_clarification", "sql_rule",
"solved_question", "explained_error"}
or type(value["format"]) is not int or value["format"] != 2
or not isinstance(value["concepts"], list)
or not all(isinstance(c, str) and c for c in value["concepts"])):
raise VectorStoreError("Invalid Memory metadata filter")
must = [{"key": "memory_format", "match": {"value": value["format"]}}]
for key in ("family", "scope"):
if value[key]:
must.append({"key": f"memory_{key}", "match": {"value": value[key]}})
must.extend({"key": "memory_concepts", "match": {"value": c}} for c in value["concepts"])
if value["database"]:
dependency = [{"key": "database", "match": {"value": value["database"]}}]
dependency.extend({"key": key, "match": {"any": ["", value[key]]}}
for key in ("schema_name", "table", "column") if value[key])
must.append({"should": [
{"is_empty": {"key": "memory_dependencies"}},
{"nested": {"key": "memory_dependencies", "filter": {"must": dependency}}},
]})
return must
@staticmethod
def _bm25_compatible(info: dict) -> bool:
sparse_vectors = info.get("config", {}).get("params", {}).get("sparse_vectors")
@@ -561,7 +563,7 @@ class QdrantVectorStore:
def _ensure_collection(
self, collection: str, *, strict: bool, require_bm25: bool = False,
allow_create: bool = False,
allow_create: bool = False, maintain_bm25: bool = False,
) -> dict | None:
created = False
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
@@ -573,7 +575,9 @@ class QdrantVectorStore:
self._call(
"PUT",
f"/collections/{collection}",
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"},
**({"sparse_vectors": {"bm25": {"modifier": "idf"}}}
if require_bm25 and maintain_bm25 else {})},
)
for field_name in _KEYWORD_INDEXES:
self._call(
@@ -614,7 +618,14 @@ class QdrantVectorStore:
{"field_name": field_name, "field_schema": "keyword"},
)
if require_bm25 and not self._bm25_compatible(result):
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
sparse = result.get("config", {}).get("params", {}).get("sparse_vectors")
if maintain_bm25 and (sparse is None or isinstance(sparse, dict) and "bm25" not in sparse):
# Explicit Memory writes may add the missing sparse vector without
# touching dense points or the separately managed Reference collection.
self._call("PUT", f"/collections/{collection}/vectors/bm25",
{"sparse": {"modifier": "idf"}})
else:
raise VectorStoreError("BM25 collection configuration mismatch")
return result
def _scroll(
+260
View File
@@ -0,0 +1,260 @@
"""Closed session repair choices, durable receipts and authoritative archive activation.
Memory commits its receipt with the card. Evidence records the choice before writing
files and recovers by comparing the approved result, never by repeating a stale write.
"""
import json
from pathlib import Path
from typing import Literal
from pydantic import BaseModel, ConfigDict, Field
from sqlalchemy import text
from tht.evidence.canonical import CuratedEvidence
from tht.evidence.local_archive import LocalEvidenceArchive, _content, _digest
from tht.memory.models import CardInput, MemoryConflict, MemoryNotFound
from tht.memory.review import digest
from tht.phase import current_phase, effective_decisions
class RepairOption(BaseModel):
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
label: str = Field(min_length=1, max_length=1000)
archive: Literal["memory", "evidence"]
target_id: str = Field(min_length=1, max_length=100)
revision: str = Field(min_length=1, max_length=100)
content: dict
class RepairProposal(BaseModel):
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
reason: str = Field(min_length=1, max_length=10000)
options: list[RepairOption] = Field(min_length=1, max_length=5)
def _context(snapshot):
return digest({"phase": current_phase(snapshot), "question": snapshot.artifacts.get("question"),
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)]})
def _authorize(service, snapshot):
service._session(snapshot)
if snapshot.manifest.status in {"finalized", "archived"}:
raise MemoryConflict("Archive repair requires an open session")
def _archive(cfg, workspace):
root = cfg.evidence.local_archive_root if cfg.evidence else None
if not root or Path(root).name != workspace:
raise MemoryConflict("Local Evidence is unavailable in this workspace")
return LocalEvidenceArchive(root)
def _params(repo, snapshot, repair_id):
return {"w": repo.workspace_id, "s": snapshot.manifest.id, "r": repair_id}
def target(service, snapshot, cfg, archive, identity):
"""Read complete current content for a proposal, bound to the source session."""
_authorize(service, snapshot)
if archive == "memory":
value = service.repository.get(identity)
return {"revision": value.revision, "content": CardInput.model_validate(
value.model_dump(include=set(CardInput.model_fields))).model_dump(mode="json")}
if archive != "evidence":
raise ValueError("Unknown archive")
try:
value = _archive(cfg, service.repository.workspace_id).get(identity)
except KeyError:
raise MemoryNotFound("Evidence was not found in this workspace") from None
return {"revision": value["revision"], "content": value["unit"].model_dump(mode="json")}
def _read(repo, snapshot, repair_id):
with repo.transaction() as connection:
value = connection.execute(text("SELECT data FROM thoth_memory.archive_repairs "
"WHERE workspace_id=:w AND session_id=:s AND repair_id=:r"),
_params(repo, snapshot, repair_id)).scalar_one_or_none()
if value is None:
raise MemoryNotFound("Repair was not found in this session and workspace")
return value
def _write(repo, snapshot, repair_id, value):
with repo.transaction() as connection:
connection.execute(text("INSERT INTO thoth_memory.archive_repairs "
"(workspace_id,session_id,repair_id,data) VALUES (:w,:s,:r,CAST(:data AS jsonb)) "
"ON CONFLICT (workspace_id,session_id,repair_id) DO UPDATE SET data=EXCLUDED.data"),
{**_params(repo, snapshot, repair_id), "data": json.dumps(value)})
def prepare(service, snapshot, cfg, proposal: RepairProposal):
_authorize(service, snapshot)
if (len({o.id for o in proposal.options}) != len(proposal.options)
or any(o.id in {"reject", "continue"} for o in proposal.options)):
raise ValueError("Repair choices must have distinct identities")
items = []
for option in proposal.options:
item = option.model_dump(mode="json")
if option.archive == "memory":
before = service.repository.get(option.target_id)
if before.revision != option.revision:
raise MemoryConflict("Memory changed; prepare new choices")
item["before"] = before.model_dump(mode="json")
item["content"] = CardInput.model_validate(option.content).model_dump(mode="json")
else:
archive = _archive(cfg, service.repository.workspace_id)
with archive.operation():
state = archive._state()
if not state["active"] or state["pending"] or state.get("import_writes"):
raise MemoryConflict("Consolidate Evidence before proposing a session repair")
files = archive._files(archive.evidence)
if files != archive._files(archive._snapshot(state["active"])):
raise MemoryConflict("Consolidate external Evidence edits before session repair")
existing = archive._units(files).get(option.target_id)
if existing is None:
raise MemoryNotFound("Evidence was not found in this workspace")
relative, before = existing
if _content(before) != option.revision:
raise MemoryConflict("Evidence changed; prepare new choices")
value = CuratedEvidence.model_validate(option.content)
if (value.schema_version != 4 or value.id != before.id
or value.kind != before.kind or value.review_items):
raise ValueError("Repair must preserve Evidence identity/kind and resolve review items")
# Curators change knowledge, not the source history supplied by the archive.
value = value.model_copy(update={"provenance": before.provenance})
item.update(content=value.model_dump(mode="json"),
before=before.model_dump(mode="json"), file=relative,
other_files={p: _digest(v) for p, v in files.items() if p != relative})
items.append(item)
value = {"reason": proposal.reason, "options": items, "context": _context(snapshot),
"choice": None, "status": "proposed", "saved": False, "indexed": False}
repair_id = digest({"workspace": service.repository.workspace_id,
"session": snapshot.manifest.id, **value})
with service.repository.operation() as repo:
try:
_read(repo, snapshot, repair_id)
except MemoryNotFound:
_write(repo, snapshot, repair_id, value)
return show(service, snapshot, cfg, repair_id)
def show(service, snapshot, cfg, repair_id):
_authorize(service, snapshot)
value = _read(service.repository, snapshot, repair_id)
# A completed receipt describes history; current eligibility is checked afresh.
if value.get("saved"):
option = next(o for o in value["options"] if o["id"] == value["choice"])
try:
if option["archive"] == "memory":
current = service.repository.get(option["target_id"])
same = current.revision == value["saved_revision"]
indexed = same and current.indexed
else:
archive = _archive(cfg, service.repository.workspace_id)
with archive.operation():
units = archive._units(archive._files(archive.evidence))
current = units[option["target_id"]][1]
same = _content(current) == _content(
CuratedEvidence.model_validate(option["content"]))
state = archive._state()
active = archive._units(archive._files(archive._snapshot(state["active"]))) \
if state["active"] else {}
indexed = same and option["target_id"] in active and \
active[option["target_id"]][1] == current
value.update(indexed=bool(indexed), status="superseded" if not same else
"active" if indexed else "pending_activation")
except (MemoryNotFound, KeyError, ValueError):
value.update(indexed=False, status="superseded")
return {**value, "repair_id": repair_id, "can_apply": service.principal.is_admin}
def list_repairs(service, snapshot):
_authorize(service, snapshot)
with service.repository.transaction() as connection:
rows = connection.execute(text("SELECT repair_id,data->>'status' AS recorded_status, "
"data->>'reason' AS reason FROM thoth_memory.archive_repairs "
"WHERE workspace_id=:w AND session_id=:s ORDER BY created_at"),
{"w": service.repository.workspace_id, "s": snapshot.manifest.id}).mappings().all()
return {"repairs": [dict(row) for row in rows]}
def apply(service, snapshot, cfg, repair_id, choice, *, activate=None):
_authorize(service, snapshot)
if choice != "reject":
service._admin()
with service.repository.operation() as repo:
value = _read(repo, snapshot, repair_id)
if value["choice"] is not None and value["choice"] != choice:
raise MemoryConflict("This repair already has a different recorded choice")
if value["choice"] is None and value["context"] != _context(snapshot):
raise MemoryConflict("Session decisions changed; reformulate the repair")
if choice == "reject":
value.update(choice=choice, status="rejected", actor=service.principal.subject)
_write(repo, snapshot, repair_id, value)
else:
option = next((o for o in value["options"] if o["id"] == choice), None)
if option is None:
raise ValueError("Select one of the reviewed repair choices")
if option["archive"] == "memory":
with repo.transaction():
if not value["saved"]:
current = repo.get(option["target_id"])
if current.revision != option["revision"]:
raise MemoryConflict("Memory changed; reformulate the repair")
repo.save(CardInput.model_validate(option["content"]),
card_id=option["target_id"])
value.update(choice=choice, saved=True, status="pending_activation",
actor=service.principal.subject,
saved_revision=repo.get(option["target_id"]).revision)
_write(repo, snapshot, repair_id, value)
elif repo.get(option["target_id"]).revision != value["saved_revision"]:
raise MemoryConflict("The repaired Memory was changed again; do not replay it")
result = service._propagate(repo, option["target_id"])
value.update(indexed=result["indexed"], status="active" if result["indexed"] else
"pending_activation")
_write(repo, snapshot, repair_id, value)
else:
_apply_evidence(service, repo, snapshot, cfg, repair_id, choice, value, option,
activate)
return show(service, snapshot, cfg, repair_id)
def _apply_evidence(service, repo, snapshot, cfg, repair_id, choice, receipt, option, activate):
archive = _archive(cfg, repo.workspace_id)
proposed = CuratedEvidence.model_validate(option["content"])
with archive.operation():
state = archive._state()
if state.get("import_writes") or (receipt["choice"] is None and state["pending"]):
raise MemoryConflict("Finish the pending Evidence consolidation before this repair")
files = archive._files(archive.evidence)
if {p: _digest(v) for p, v in files.items() if p != option["file"]} != option["other_files"]:
raise MemoryConflict("Other Evidence files changed; reconcile them before retrying")
existing = archive._units(files).get(option["target_id"])
if existing is None:
raise MemoryConflict("Evidence was removed after review")
_, current = existing
already_written = _content(current) == _content(proposed) and receipt["choice"] == choice
if not already_written and (receipt["saved"] or _content(current) != option["revision"]
or current.provenance != proposed.provenance):
raise MemoryConflict("Evidence changed; the approved correction cannot overwrite it")
# Commit approval before touching the filesystem; a restart can recover only this choice.
receipt.update(choice=choice, actor=service.principal.subject, status="applying")
_write(repo, snapshot, repair_id, receipt)
try:
if not already_written:
archive._save(proposed, expected_revision=option["revision"],
actor=service.principal.subject)
receipt.update(saved=True, status="pending_activation")
_write(repo, snapshot, repair_id, receipt)
archive._consolidate(service.principal.subject, activate)
except Exception: # noqa: BLE001 - durable approval covers file/index interruption.
receipt.update(status="pending_activation" if receipt["saved"] else "applying",
indexed=False)
_write(repo, snapshot, repair_id, receipt)
return
receipt.update(indexed=activate is not None,
status="active" if activate else "pending_activation")
_write(repo, snapshot, repair_id, receipt)
+43 -2
View File
@@ -23,6 +23,46 @@ from tht.evidence import (
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
@evidence_app.command("sources", hidden=True)
def sources_cmd(action: str, config: Path = CONFIG_OPT,
source_id: str | None = typer.Option(None), revision: str | None = typer.Option(None),
decision: str | None = typer.Option(None), actor: str = typer.Option("installation operator"),
json_output: bool = typer.Option(False, "--json")):
from tht.evidence.administration import ConsolidationError, source_action
try:
if action not in {"refresh", "decide"} or len(actor) > 256 or not actor.strip():
raise ValueError("Invalid source action")
if action == "refresh" and any(v is not None for v in (source_id, revision, decision)):
raise ValueError("Refresh does not accept decision options")
if action == "decide" and (decision not in {"keep", "replace"} or not source_id or not revision):
raise ValueError("A source decision requires source-id, revision and keep or replace")
payload = source_action(config, action=action, source_id=source_id, revision=revision,
decision=decision, actor=actor)
except Exception as error: # noqa: BLE001 - never expose connector/provider exception details
safe = isinstance(error, (ValueError, EvidencePreparationError, ConsolidationError))
_emit({"status": "failed", "code": "evidence_source_failed",
"error": str(error)[:1500] if safe else "Source acquisition or refinement failed; existing Evidence is preserved.",
"saved": isinstance(error, ConsolidationError) and error.saved}, json_output)
raise typer.Exit(1) from None
_emit(payload, json_output)
@evidence_app.command("admin", hidden=True)
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
from tht.evidence.administration import browse
try:
if os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() not in {"true", "1"}:
raise ValueError("Evidence administration requires an administrator")
payload = json.loads(config.read_text())
root = Path(payload["root"])
if not root.is_absolute() or root.name != workspace:
raise ValueError("Invalid workspace archive identity")
_emit(browse(root, payload.get("query", {})), True)
except (ValueError, OSError) as error:
_emit({"code": "evidence_unavailable", "message": str(error)[:1500]}, True)
raise typer.Exit(1) from None
def _canonical_worktree(workspace_root: Path) -> Path:
requested = workspace_root.absolute()
root = workspace_root.resolve()
@@ -123,7 +163,8 @@ def prepare_cmd(
) -> None:
"""Prepare changed Source Evidence without committing or publishing it."""
root = _canonical_worktree(workspace_root)
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
from tht.evidence.authoring import authoring_skill_path
skill_path = authoring_skill_path()
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
try:
try:
@@ -180,7 +221,7 @@ def migrate_cmd(
workspace_root: Path,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Rewrite legacy Curated units as table-free v3 Markdown without model calls."""
"""Convert legacy units to editable v4 Markdown and establish a local baseline."""
root = _canonical_worktree(workspace_root)
try:
report = migrate_workspace_evidence(root)
+307 -454
View File
@@ -1,502 +1,355 @@
# TODO (drop registry, decisione spec 5): questo modulo e' portato col modello
# registry intatto (load_registry/promote/update_record/delete_record). Le memory
# dovrebbero vivere SOLO nel vectordb (metadata arricchito con subject/detail/rationale
# in Onda 3.1). Riscrivere: promote -> upsert batch vectordb; list/show -> scan
# vectordb; delete -> metadata.status="superseded"; index/clear -> droppati.
# Task separato: la validazione richiede L2 (vectordb reale).
"""Thin command adapters for the authoritative Memory service."""
import json
import re
from contextlib import contextmanager
from pathlib import Path
import typer
from sqlalchemy.exc import OperationalError, ProgrammingError
from pydantic import ValidationError
from tht.cli._guards import (
require_server_profile,
require_vector_write_allowed,
)
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.session_cmd import load_snapshot_or_exit
from tht.cli.vector_cmd import require_vector_cfg
from tht.memory.models import CardInput, CardQuery, MemoryError
from tht.memory.runtime import admin_service, memory_service
from tht.ports.vector import VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError
memory_app = typer.Typer(help="Review memory (registro canonico + indice semantico)")
DECISION_OPT = typer.Option(None, "--decision", help="Seq da promuovere (ripetibile).")
memory_app = typer.Typer(help="Authoritative Memory cards and verified recall")
DECISION_OPT = typer.Option(None, "--decision")
def registry_path(cfg) -> Path:
if getattr(cfg.paths, "memory", None) is not None:
return cfg.paths.memory / "registry.jsonl"
# Legacy location remains readable while old workspaces are retired.
return cfg.paths.artifacts / "memory" / "registry.jsonl"
def _output(value):
typer.echo(json.dumps(value, ensure_ascii=False, default=str))
def _resync_memory(cfg):
"""Risincronizza l'indice semantico col registro corrente (incrementale)."""
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder, sync_canonical_records
from tht.memory import load_registry, memory_vector_records
records = memory_vector_records(load_registry(registry_path(cfg)))
return sync_canonical_records(
"memory",
records,
store=build_vector_store(cfg, require_write=True),
embedder=make_embedder(cfg.embeddings),
)
@memory_app.command("promote")
def promote_cmd(
session: str = typer.Option(..., "--session"),
decision: list[int] = DECISION_OPT,
preview: bool = typer.Option(False, "--preview", help="Mostra i candidati in JSON, non scrive."),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Promuove le decisioni SCELTE nel registro globale. Usa --preview per vedere i candidati."""
import json as _json
from tht.memory import promote_snapshot
cfg = _load_config_or_exit(config)
snapshot = load_snapshot_or_exit(cfg, session)
if preview:
from tht.memory import (
MAX_PROMOTION_CANDIDATES,
preview_promotions_snapshot,
reusable_promotions_snapshot,
)
cand = preview_promotions_snapshot(snapshot, registry_path(cfg))
extra = len(reusable_promotions_snapshot(snapshot, registry_path(cfg))) - len(cand)
payload = [
{"decision_seq": c.decision_seq, "type": c.type, "subject": c.subject,
"detail": c.detail, "rationale": c.rationale,
"question_context": c.question_context,
"tables": c.tables, "concepts": c.concepts}
for c in cand
]
if json_out:
typer.echo(_json.dumps(payload, ensure_ascii=False, indent=2))
elif not payload:
typer.secho("Nessun candidato da promuovere.", fg=typer.colors.YELLOW)
else:
for c in payload:
typer.echo(f" [{c['decision_seq']}] {c['type']}: {c['subject']}")
if extra > 0:
typer.secho(
f"NOTA: mostrati {len(cand)} candidati su {len(cand) + extra} riusabili "
f"(cap {MAX_PROMOTION_CANDIDATES}); gli altri non sono proposti.",
fg=typer.colors.YELLOW, err=True,
)
return
if not decision:
typer.secho("ERRORE: indica le decisioni con --decision <seq> (vedi `--preview`).",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
require_server_profile(cfg, "memory promote")
require_vector_cfg(cfg)
promoted = promote_snapshot(snapshot, seqs=list(decision), registry_path=registry_path(cfg))
if not promoted:
msg = "Nessuna nuova promozione (gia' presenti o seq inesistenti)."
if json_out:
typer.echo(_json.dumps(
{"promoted": [], "indexed": False, "message": msg}, ensure_ascii=False))
else:
typer.secho(msg, fg=typer.colors.YELLOW)
return
# Promozione nel registro: riuscita. L'indicizzazione semantica puo' fallire
# (runtime non pronto o vectordb irraggiungibile da questa postazione): in quel
# caso le memorie restano nel registro ma NON sono trovate da `tht memory
# search` finche' non si reindicizza sul server. `indexed` rende lo stato
# leggibile da Pi, cosi' il reviewer lo vede invece di perderlo nello stderr.
ids = [{"id": r.id, "type": r.type, "subject": r.subject} for r in promoted]
indexed = True
warning = None
@contextmanager
def _service(config):
service = None
try:
_resync_memory(cfg)
except (ProgrammingError, OperationalError):
indexed = False
warning = (
f"{len(promoted)} memorie promosse nel registro, ma l'indice vettoriale "
"NON e' stato sincronizzato (runtime vettoriale mancante o irraggiungibile): "
"NON saranno trovate da `tht memory search` finche' non reindicizzi sul "
"server (`tht memory index` quando il runtime vettoriale è disponibile)."
)
if json_out:
typer.echo(_json.dumps(
{"promoted": ids, "indexed": indexed,
"message": warning or f"{len(promoted)} memorie promosse e indicizzate."},
ensure_ascii=False, indent=2))
return
for r in promoted:
typer.echo(f" {r.id}: {r.type} {r.subject}")
if indexed:
typer.secho(f"OK: {len(promoted)} memorie promosse e indicizzate.",
fg=typer.colors.GREEN)
else:
typer.secho(f"ATTENZIONE: {warning}", fg=typer.colors.YELLOW)
cfg = _load_config_or_exit(config)
service = memory_service(cfg)
yield cfg, service
except MemoryError as error:
_output({"code": error.code, "message": str(error), "status": error.status})
raise typer.Exit(1) from None
except (ValidationError, ValueError):
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
raise typer.Exit(1) from None
except (VectorStoreError, EmbeddingsError, OSError):
_output({"code": "memory_unavailable", "message": "Memory operation is unavailable",
"status": 503})
raise typer.Exit(1) from None
finally:
if service:
service.close()
@memory_app.command("save-one")
def save_one_cmd(
session: str = typer.Option(..., "--session"),
decision: int = typer.Option(
..., "--decision", help="decision_seq della decisione da salvare come memoria."
),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Upsert mirato (una riga) della memoria di una decisione nel semantic store (D11).
Promuove la decisione nel registro locale (idempotente) e fa un singolo upsert
con dedup hash client-side -- niente full-resync. Il factory seleziona il writer
del runtime vettoriale attivo.
"""
import json as _json
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder
from tht.memory import load_registry, promote_snapshot, save_one_memory
cfg = _load_config_or_exit(config)
snapshot = load_snapshot_or_exit(cfg, session)
require_vector_write_allowed(cfg, "memory save-one")
store = build_vector_store(cfg, require_write=True)
# Promuove la decisione scelta nel registro locale (idempotente: salta se gia' presente
# o se stale post-rollback, perche' _compute_promotions usa la vista effective).
promote_snapshot(snapshot, seqs=[decision], registry_path=registry_path(cfg))
records = [r for r in load_registry(registry_path(cfg)) if r.session_id == snapshot.manifest.id]
embedder = make_embedder(cfg.embeddings)
count = save_one_memory(records, decision, store=store, embedder=embedder)
msg = (
f"{count} memoria salvata nell'indice semantico (decision_seq {decision})."
if count
else f"Nessun upsert (decisione {decision} assente/stale o memoria gia' aggiornata)."
)
if json_out:
typer.echo(_json.dumps(
{"upserted": count, "decision_seq": decision, "message": msg}, ensure_ascii=False))
return
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN if count else typer.colors.YELLOW)
@memory_app.command("index")
def index_cmd(config: Path = CONFIG_OPT) -> None:
"""Sincronizza il registro memory nell'indice semantico (full-resync)."""
from tht.cli.vector_cmd import _print_stats
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "memory index")
require_vector_cfg(cfg)
_print_stats(_resync_memory(cfg))
@memory_app.command("admin")
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
"""Backend-owned request snapshot. No DWH or session configuration is required."""
service = None
try:
if not re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace):
raise ValueError("Invalid workspace")
payload = json.loads(config.read_text())
service = admin_service(workspace, payload["runtime"])
request = payload.get("request", {})
action = payload["action"]
if action == "cleanup":
from tht.memory.cleanup import CleanupRequest, cleanup
result = cleanup(service, CleanupRequest.model_validate(request))
elif action == "list":
result = service.list(CardQuery.model_validate(request))
elif action == "show":
result = service.get(request["id"])
elif action in {"create", "update"}:
result = service.save(CardInput.model_validate(request["card"]),
request.get("id") if action == "update" else None)
elif action == "delete":
result = service.delete(request["id"])
elif action == "pending":
result = service.pending()
elif action == "retry":
result = service.retry(request["id"])
else:
raise ValueError("Unknown Memory action")
_output(result)
except MemoryError as error:
_output({"code": error.code, "message": str(error), "status": error.status})
raise typer.Exit(1) from None
except (ValueError, KeyError, TypeError, OSError):
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
raise typer.Exit(1) from None
finally:
if service:
service.close()
@memory_app.command("list")
def list_cmd(
type_: str = typer.Option(None, "--type", help="Filtra per tipo decisione."),
session: str = typer.Option(None, "--session", help="Filtra per sessione."),
table: str = typer.Option(None, "--table", help="Filtra per tabella coinvolta."),
concept: str = typer.Option(None, "--concept", help="Filtra per concetto."),
json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT,
) -> None:
"""Elenca le memorie del registro (filtri combinati in AND)."""
import json as _json
from rich.console import Console
from rich.table import Table
from tht.memory import load_registry
cfg = _load_config_or_exit(config)
recs = load_registry(registry_path(cfg))
if type_:
recs = [r for r in recs if r.type == type_]
if session:
recs = [r for r in recs if r.session_id == session]
if table:
recs = [r for r in recs if table in r.tables]
if concept:
recs = [r for r in recs if concept in r.concepts]
if json_out:
typer.echo(_json.dumps([r.model_dump(mode="json") for r in recs],
ensure_ascii=False, indent=2))
return
if not recs:
typer.secho("Nessuna memoria nel registro.", fg=typer.colors.YELLOW)
return
t = Table(title="Review memory")
for col in ("Id", "Tipo", "Soggetto", "Sessione"):
t.add_column(col)
for r in recs:
t.add_row(r.id, r.type, r.subject, r.session_id)
Console().print(t)
def list_cmd(filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.list(CardQuery.model_validate_json(filters)))
@memory_app.command("show")
def show_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT,
) -> None:
"""Mostra una singola memoria."""
import json as _json
def show_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.get(mem_id))
from tht.memory import load_registry
cfg = _load_config_or_exit(config)
rec = {r.id: r for r in load_registry(registry_path(cfg))}.get(mem_id)
if rec is None:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
if json_out:
typer.echo(_json.dumps(rec.model_dump(mode="json"), ensure_ascii=False, indent=2))
return
for k, v in rec.model_dump(mode="json").items():
typer.echo(f"{k}: {v}")
@memory_app.command("create")
def create_cmd(data: Path = typer.Option(..., "--data"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.save(CardInput.model_validate_json(data.read_text())))
@memory_app.command("update")
def update_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
subject: str = typer.Option(None, "--subject"),
type_: str = typer.Option(None, "--type"),
detail: str = typer.Option(None, "--detail"),
rationale: str = typer.Option(None, "--rationale"),
question_context: str = typer.Option(None, "--question-context"),
tables: str = typer.Option(None, "--tables", help="CSV; \"\" per azzerare."),
concepts: str = typer.Option(None, "--concepts", help="CSV; \"\" per azzerare."),
config: Path = CONFIG_OPT,
) -> None:
"""Modifica i campi di merito di una memoria (provenienza immutabile)."""
from typing import get_args
from tht.decisions import DecisionType
from tht.memory import MemoryNotFound, update_record
cfg = _load_config_or_exit(config)
fields: dict = {}
for name, val in (("subject", subject), ("type", type_), ("detail", detail),
("rationale", rationale), ("question_context", question_context)):
if val is not None:
fields[name] = val
if tables is not None:
fields["tables"] = [t.strip() for t in tables.split(",") if t.strip()]
if concepts is not None:
fields["concepts"] = [c.strip() for c in concepts.split(",") if c.strip()]
if not fields:
typer.secho("ERRORE: nessun campo da modificare indicato.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
if "type" in fields and fields["type"] not in get_args(DecisionType):
typer.secho(f"ERRORE: tipo '{fields['type']}' non valido.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
require_server_profile(cfg, "memory update")
require_vector_cfg(cfg)
try:
rec = update_record(registry_path(cfg), mem_id, fields)
except MemoryNotFound:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
_resync_memory(cfg)
typer.secho(f"OK: {rec.id} aggiornata e reindicizzata.", fg=typer.colors.GREEN)
def update_cmd(mem_id: str, data: Path = typer.Option(..., "--data"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.save(CardInput.model_validate_json(data.read_text()), mem_id))
@memory_app.command("delete")
def delete_cmd(
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
yes: bool = typer.Option(False, "--yes", "-y", help="Salta la conferma."),
config: Path = CONFIG_OPT,
) -> None:
"""Cancella una singola memoria (registro + indice)."""
from tht.memory import MemoryNotFound, delete_record
def delete_cmd(mem_id: str, yes: bool = typer.Option(False, "--yes", "-y"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
service._admin()
if not yes and not typer.confirm(f"Delete Memory card {mem_id} and its links?"):
raise typer.Exit(1)
_output(service.delete(mem_id))
cfg = _load_config_or_exit(config)
require_server_profile(cfg, "memory delete")
require_vector_cfg(cfg)
if not yes and not typer.confirm(f"Cancellare definitivamente la memoria '{mem_id}'?"):
typer.secho("Annullato.", fg=typer.colors.YELLOW)
raise typer.Exit(code=1)
try:
delete_record(registry_path(cfg), mem_id)
except MemoryNotFound:
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=6)
_resync_memory(cfg)
typer.secho(f"OK: {mem_id} cancellata e deindicizzata.", fg=typer.colors.GREEN)
@memory_app.command("pending")
def pending_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.pending())
@memory_app.command("retry")
def retry_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.retry(mem_id))
@memory_app.command("index")
def index_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (_, service):
_output(service.rebuild())
@memory_app.command("promote")
def promote_cmd(session: str = typer.Option(..., "--session"), decision: list[int] = DECISION_OPT,
preview: bool = typer.Option(False, "--preview"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
if preview:
_output(service.promotions(snapshot)[:5])
elif not decision:
raise ValueError("Explicit decisions are required")
else:
results = service.promote(snapshot, decision)
_output({"promoted": [r.get("card") for r in results],
"indexed": all(r["indexed"] for r in results), "results": results})
@memory_app.command("propose")
def propose_cmd(session: str = typer.Option(..., "--session"),
data: Path = typer.Option(..., "--data"), config: Path = CONFIG_OPT):
"""Persist reviewer-grounded proposals without changing the Memory archive."""
from tht.cli.session_cmd import session_repository
from tht.memory.review import validate_proposals
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
service._session(snapshot)
proposals = validate_proposals(snapshot, json.loads(data.read_text()))
session_repository(cfg).write_artifact(session, "memory_proposals",
json.dumps([p.model_dump(mode="json") for p in proposals], ensure_ascii=False))
_output({"proposals": len(proposals), "saved_to_archive": False})
@memory_app.command("summary")
def summary_cmd(session: str = typer.Option(..., "--session"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.memory.review import prepare
with _service(config) as (cfg, service):
_output(prepare(service, load_snapshot_or_exit(cfg, session)))
@memory_app.command("repair-prepare")
def repair_prepare_cmd(session: str = typer.Option(..., "--session"),
proposal_json: str = typer.Option(..., "--proposal-json"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import RepairProposal, prepare
if len(proposal_json.encode()) > 1_000_000:
_output({"code": "memory_invalid", "message": "Repair proposal is too large", "status": 400})
raise typer.Exit(1)
with _service(config) as (cfg, service):
_output(prepare(service, load_snapshot_or_exit(cfg, session), cfg,
RepairProposal.model_validate_json(proposal_json)))
@memory_app.command("repair-target")
def repair_target_cmd(session: str = typer.Option(..., "--session"),
archive: str = typer.Option(..., "--archive"),
target_id: str = typer.Option(..., "--target-id"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import target
with _service(config) as (cfg, service):
_output(target(service, load_snapshot_or_exit(cfg, session), cfg, archive, target_id))
@memory_app.command("repairs")
def repairs_cmd(session: str = typer.Option(..., "--session"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import list_repairs
with _service(config) as (cfg, service):
_output(list_repairs(service, load_snapshot_or_exit(cfg, session)))
@memory_app.command("repair-show")
def repair_show_cmd(session: str = typer.Option(..., "--session"),
repair_id: str = typer.Option(..., "--repair-id"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import show
with _service(config) as (cfg, service):
_output(show(service, load_snapshot_or_exit(cfg, session), cfg, repair_id))
@memory_app.command("repair-apply")
def repair_apply_cmd(session: str = typer.Option(..., "--session"),
repair_id: str = typer.Option(..., "--repair-id"),
choice: str = typer.Option(..., "--choice"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.archive_repair import apply
from tht.cli.preprocess_cmd import run_from_config
def activate(snapshot):
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise RuntimeError("Evidence activation did not complete")
with _service(config) as (cfg, service):
_output(apply(service, load_snapshot_or_exit(cfg, session), cfg, repair_id, choice,
activate=activate))
@memory_app.command("review-apply")
def review_apply_cmd(session: str = typer.Option(..., "--session"),
review_json: str = typer.Option(..., "--review-json"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.memory.review import ReviewResponse, apply
with _service(config) as (cfg, service):
_output(apply(service, load_snapshot_or_exit(cfg, session),
ReviewResponse.model_validate_json(review_json)))
@memory_app.command("save-one")
def save_one_cmd(session: str = typer.Option(..., "--session"),
decision: int = typer.Option(..., "--decision"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
with _service(config) as (cfg, service):
results = service.promote(load_snapshot_or_exit(cfg, session), [decision])
_output({"upserted": sum(r["indexed"] for r in results), "decision_seq": decision,
"indexed": all(r["indexed"] for r in results), "results": results})
@memory_app.command("search")
def search_cmd(
question: str = typer.Argument(..., help="Domanda o termini di ricerca."),
top: int = typer.Option(5, "--top"),
session: str = typer.Option(
None, "--session",
help="Esclude le memorie gia' decise (applicate o rifiutate) in questa sessione.",
),
json_out: bool = typer.Option(False, "--json", help="Output JSON per Pi."),
config: Path = CONFIG_OPT,
) -> None:
"""Cerca memorie riapplicabili, ordinate per similarita'. Mai applicate in automatico.
def search_cmd(question: str, top: int = typer.Option(5, "--top"),
session: str | None = typer.Option(None, "--session"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.cli.vector_cmd import make_embedder, open_searcher
with _service(config) as (cfg, service):
decisions = load_snapshot_or_exit(cfg, session).decisions if session else []
_output(service.recall(question, searcher=open_searcher(cfg),
embedder=make_embedder(cfg.embeddings), top=top, decisions=decisions,
scope=_recall_scope(cfg, filters)))
Con `--session` non ripropone le memorie gia' decise in quella sessione (fix:
memorie scartate riproposte): rifiutate via `memory_rejected` o gia' applicate."""
from rich.console import Console
from rich.table import Table
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
from tht.memory import load_registry, recall_memories
def _recall_scope(cfg, filters):
from tht.memory.retrieval import RecallScope
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
decisions = load_snapshot_or_exit(cfg, session).decisions if session is not None else []
searcher = open_searcher(cfg)
embedder = make_embedder(cfg.embeddings)
results = recall_memories(
question,
records=load_registry(registry_path(cfg)),
decisions=decisions,
searcher=searcher,
embedder=embedder,
top=top,
)
context = {"database": cfg.database.database, "schema_name": cfg.database.db_schema}
supplied = json.loads(filters)
if not isinstance(supplied, dict) or any(
key in supplied and supplied[key] != value for key, value in context.items()
):
raise ValueError("Recall cannot override the configured database/schema context")
return RecallScope.model_validate({**supplied, **context})
if json_out:
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
return
if not results:
typer.secho("Nessuna memoria candidata.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Memorie candidate per: {question}")
table.add_column("Id")
table.add_column("Tipo")
table.add_column("Soggetto")
table.add_column("Contesto originale")
table.add_column("Score", justify="right")
for r in results:
table.add_row(r["id"], r["type"], r["subject"],
r["question_context"][:60], f"{r['score']:.3f}")
Console().print(table)
@memory_app.command("rules")
def rules_cmd(question: str, session: str = typer.Option(..., "--session"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
"""Consult SQL rules and explained errors in schema linking and SQL construction."""
from types import SimpleNamespace
from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.phase import current_phase
with _service(config) as (cfg, service):
snapshot = load_snapshot_or_exit(cfg, session)
service._session(snapshot)
if current_phase(snapshot) not in {4, 6, 7}:
raise ValueError("Memory rules are consulted in schema linking or SQL construction")
scope = _recall_scope(cfg, filters)
vector = make_embedder(cfg.embeddings).embed_query(question)
embedder = SimpleNamespace(embed_query=lambda _: vector)
searcher = open_searcher(cfg)
candidates = []
for family in ("sql_rule", "explained_error"):
candidates.extend(service.retrieve(question, searcher=searcher, embedder=embedder,
scope=scope, family=family, top=5))
_output([{**candidate.card.model_dump(mode="json"), "score": candidate.score,
"retrieval_path": candidate.path, "consultative": True}
for candidate in sorted(candidates, key=lambda c: (-c.score, c.card.id))[:10]])
def index_solved_session(cfg, session_id: str) -> int:
"""Indicizza la coppia domanda->SQL della sessione (kind solved_question).
Solleva SolvedIndexError se mancano gli artefatti: il finalize lo degrada a warning,
il comando CLI lo converte in errore esplicito."""
from tht.adapters.factory import build_vector_store
from tht.cli.sql_cmd import promoted_tables_for
from tht.cli.vector_cmd import make_embedder
from tht.memory import index_solved_question
store = build_vector_store(cfg, require_write=True)
return index_solved_question(
load_snapshot_or_exit(cfg, session_id),
promoted_tables_for(cfg, session_id),
store=store,
embedder=make_embedder(cfg.embeddings),
)
"""Recovery only: never recreate a deleted card from historical session artifacts."""
service = memory_service(cfg)
try:
result = service.retry_solved(load_snapshot_or_exit(cfg, session_id))
if not result["indexed"]:
raise RuntimeError(result["error"])
return int(result["action"] == "upsert")
finally:
service.close()
@memory_app.command("solved-index")
def solved_index_cmd(
session_id: str = typer.Argument(..., help="Id sessione con sql_final.sql approvato."),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Indicizza la coppia domanda->SQL nel semantic store (backfill; il finalize lo fa da solo)."""
import json as _json
from tht.memory import SolvedIndexError
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "memory solved-index")
try:
count = index_solved_session(cfg, session_id)
except RuntimeError as e:
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=4)
except SolvedIndexError as e:
typer.secho(f"ERRORE: sessione {session_id} non indicizzabile: {e}",
fg=typer.colors.RED, err=True)
raise typer.Exit(code=3)
msg = (
f"1 coppia domanda->SQL indicizzata (solved:{session_id})."
if count else "Nessun upsert: coppia gia' aggiornata."
)
if json_out:
typer.echo(_json.dumps({"upserted": count, "id": f"solved:{session_id}"},
ensure_ascii=False))
return
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN)
def solved_index_cmd(session_id: str, json_out: bool = typer.Option(False, "--json"),
config: Path = CONFIG_OPT):
"""Retry an existing authoritative exemplar's Qdrant projection; never import a session."""
with _service(config) as (cfg, service):
_output(service.retry_solved(load_snapshot_or_exit(cfg, session_id)))
@memory_app.command("solved-search")
def solved_search_cmd(
question: str = typer.Argument(..., help="Domanda da confrontare con quelle risolte."),
top: int = typer.Option(3, "--top"),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
config: Path = CONFIG_OPT,
) -> None:
"""Domande gia' risolte simili (kind solved_question): domanda, SQL e tabelle."""
from rich.console import Console
from rich.table import Table
def solved_search_cmd(question: str, top: int = typer.Option(3, "--top"),
filters: str = typer.Option("{}", "--filters"),
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.memory import search_solved_questions
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
# Degrado gentile: SKILL.md prescrive solved-search in F4/F6/F7 di ogni sessione,
# quindi vectordb/Ollama irraggiungibili non devono produrre un traceback grezzo
# nel transcript: avviso di una riga su stderr, stdout puro ([] in --json), exit 0.
try:
searcher = open_searcher(cfg)
embedder = make_embedder(cfg.embeddings)
results = search_solved_questions(
question,
searcher=searcher,
embedder=embedder,
top=top,
)
except (VectorStoreError, VectorReadUnavailable, EmbeddingsError, OperationalError) as e:
typer.secho(
f"ATTENZIONE: exemplar non disponibili ({e}). Prosegui senza.",
fg=typer.colors.YELLOW, err=True,
)
if json_out:
typer.echo("[]")
return
if json_out:
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
return
if not results:
typer.secho("Nessuna domanda risolta simile.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Domande risolte simili a: {question}")
table.add_column("Sessione")
table.add_column("Domanda")
table.add_column("Tabelle")
table.add_column("Score", justify="right")
for r in results:
table.add_row(r["session_id"], r["question"][:60],
", ".join(r["tables"]), f"{r['score']:.3f}")
Console().print(table)
with _service(config) as (cfg, service):
try:
result = service.recall(question, searcher=open_searcher(cfg),
embedder=make_embedder(cfg.embeddings), top=top, solved=True,
scope=_recall_scope(cfg, filters))
except (VectorStoreError, EmbeddingsError):
typer.echo("Avviso: exemplar non disponibili; ricerca semantica non riuscita.", err=True)
result = []
_output(result)
+30 -6
View File
@@ -50,6 +50,10 @@ def _evaluation_workspace_root(cfg) -> Path:
def _requires_candidate_evaluation(cfg) -> bool:
evidence = cfg.evidence
if evidence and evidence.local_archive_root and (
evidence.local_archive_root / "evidence/.local/state.yaml"
).exists():
return False
if evidence is None or evidence.schema_version != 2:
return False
return evidence.source_root is not None or any(
@@ -224,7 +228,8 @@ def _parse_dwh_steps(value: str) -> tuple[str, ...]:
return steps
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None):
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None,
local_snapshot: Path | None = None):
from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder
@@ -239,8 +244,11 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
corpus_root = cfg.paths.artifacts.parent / "corpus"
vector_store = build_vector_store(cfg, require_write=True)
embedder = make_embedder(cfg.embeddings)
from tht.evidence.adapters import FilesystemEvidenceSource
sources = [FilesystemEvidenceSource(local_snapshot, patterns=("curated/**/*.md",))] \
if local_snapshot is not None else build_sources(cfg.evidence)
pipeline = build_preprocessing_pipeline(
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
store=CorpusStore(corpus_root), sources=sources,
embedder=embedder,
vector_store=vector_store,
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
@@ -376,9 +384,17 @@ def evidence_cmd(
dry_run: bool = typer.Option(False, "--dry-run"),
resume: str | None = typer.Option(None, "--resume"),
json_output: bool = typer.Option(False, "--json"),
consolidate: bool = typer.Option(False, "--consolidate"),
) -> None:
if action is not None and action != "gc":
raise typer.BadParameter("only the optional 'gc' action is supported")
if consolidate and (dry_run or resume is not None or action is not None):
message = "Consolidation cannot be combined with dry-run, resume or gc"
if json_output:
typer.echo(json.dumps({"status": "failed", "code": "invalid_consolidation", "error": message}))
else:
typer.secho(message, fg=typer.colors.RED, err=True)
raise typer.Exit(code=2)
if action == "gc":
try:
payload = gc_from_config(config, dry_run=dry_run)
@@ -406,21 +422,29 @@ def evidence_cmd(
typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True)
raise typer.Exit(code=2)
try:
result = run_from_config(config, dry_run=dry_run, resume=resume)
except Exception: # noqa: BLE001
if consolidate:
from tht.evidence.administration import consolidate_from_config
result = consolidate_from_config(config)
else:
result = run_from_config(config, dry_run=dry_run, resume=resume)
except Exception as error: # noqa: BLE001
from tht.evidence.administration import ConsolidationError
detail = str(error) if isinstance(error, ConsolidationError) else "preprocessing failed"
payload = {"status": "failed"}
if isinstance(error, ConsolidationError):
payload["saved"] = error.saved
if json_output:
typer.echo(json.dumps(
_evidence_json_payload(
cfg,
payload,
code="preprocessing_failed",
error="preprocessing failed",
error=detail,
),
sort_keys=True,
))
else:
typer.secho("ERRORE: preprocessing failed", fg=typer.colors.RED, err=True)
typer.secho(f"ERRORE: {detail}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = result.model_dump(mode="json")
if payload.get("status") != "succeeded":
+1 -38
View File
@@ -624,44 +624,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
repository, session_id, validation_report=report, evidence=evidence
)
# --- memoria attiva (parte B): indicizza la coppia domanda->SQL, best-effort ---
# Qualunque errore (writer key assente, VPN giu', Ollama spento) NON deve
# bloccare il finalize: l'indice e' derivato e recuperabile con
# `tht memory solved-index <id>`. Memory owns this best-effort policy; core
# has already committed the authoritative finalized snapshot above.
try:
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder
from tht.memory import index_solved_question_best_effort
finalized_snapshot = repository.get(session_id)
outcome = index_solved_question_best_effort(
finalized_snapshot,
promoted_tables,
store_factory=lambda: build_vector_store(cfg, require_write=True),
embedder_factory=lambda: make_embedder(cfg.embeddings),
)
if outcome.error is not None:
typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({outcome.error}). "
f"Recupera con `tht memory solved-index {session_id}`.",
fg=typer.colors.YELLOW, err=True,
)
elif outcome.upserted:
typer.secho(
"OK: coppia domanda->SQL indicizzata nel vectordb (solved_question).",
fg=typer.colors.GREEN,
)
else:
typer.secho(
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
fg=typer.colors.CYAN,
)
except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
f"Recupera con `tht memory solved-index {session_id}`.",
fg=typer.colors.YELLOW, err=True,
)
# Memory cards, including exemplars, are saved only by the explicit final review.
typer.secho(f"OK: sessione {session_id} finalizzata. Artefatti:", fg=typer.colors.GREEN)
for name in ARTIFACT_FILES:
state = "presente" if name not in {"session_manifest.yaml", "review_decisions.jsonl"} else "persistito"
+2
View File
@@ -512,6 +512,8 @@ EvidenceSourceConfig = Annotated[
class EvidenceSourcesConfig(BaseModel):
# Persistent curator checkout; only its activated snapshot is used by preprocessing.
local_archive_root: Path | None = None
# Version 2 is the materialized source/curated authoring layout.
schema_version: Literal[1, 2] = 1
# Legacy curated-tree configuration remains accepted during migration.
+1
View File
@@ -43,6 +43,7 @@ DecisionType = Literal[
# declined_promotion_seqs per non riproporre i candidati rifiutati).
"memory_promoted",
"memory_promotion_declined",
"memory_summary_reviewed",
# D15: marker di ritrazione. subject = "phase:N", retracts = decision_seq ritirata.
# Resta nel log di audit (append-only); effective_decisions() la esclude dalla vista.
"decision_retracted",
+5
View File
@@ -22,6 +22,7 @@ from tht.evidence.authoring import (
)
from tht.evidence.canonical import (
CuratedEvidence,
ManualEvidenceProvenance,
dump_curated_markdown,
load_curated_tree,
parse_curated_markdown,
@@ -37,6 +38,7 @@ from tht.evidence.contracts import (
validate_namespaced_value,
validate_safe_metadata,
)
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive
from tht.evidence.preprocessing import EvidenceEmbedder, build_preprocessing_pipeline
from tht.evidence.search import (
ActiveEvidenceSearcher,
@@ -58,6 +60,7 @@ from tht.evidence.sources import build_sources
__all__ = [
"AcquiredDocument",
"ActiveEvidenceSearcher",
"ArchiveConflict",
"CorpusWorkspaceMismatchError",
"CuratedEvidence",
"EvidenceEmbedder",
@@ -75,6 +78,8 @@ __all__ = [
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"LocalEvidenceArchive",
"ManualEvidenceProvenance",
"PiEvidenceRestructurer",
"RestructureCandidate",
"RestructureRequest",
+152
View File
@@ -0,0 +1,152 @@
"""Local Evidence browsing and explicit consolidation, independent of DWH access."""
import os
from pathlib import Path
from .canonical import parse_curated_markdown
from .local_archive import LocalEvidenceArchive, _content
class ConsolidationError(RuntimeError):
def __init__(self, message, *, saved=False):
super().__init__(message)
self.saved = saved
def consolidate_from_config(config: Path):
from tht.cli.preprocess_cmd import run_from_config
from tht.config import load_config
from .authoring import EvidencePreparationError, migrate_workspace_evidence
cfg = load_config(config)
if not cfg.evidence or not cfg.evidence.local_archive_root:
raise ConsolidationError("Local Evidence is not configured for this workspace")
root = cfg.evidence.local_archive_root
archive = LocalEvidenceArchive(root)
if not (archive.metadata / "state.yaml").exists():
try:
# Explicit first consolidation performs the one-time legacy conversion.
if (archive.evidence / "manifest.yaml").is_file():
migrate_workspace_evidence(root)
else:
archive.initialize()
except (ValueError, OSError, EvidencePreparationError) as error:
raise ConsolidationError(str(error)) from error
result = None
def activate(snapshot):
nonlocal result
try:
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise ConsolidationError("Evidence indexing is blocked; check unit size and review items", saved=True)
except ConsolidationError:
raise
except Exception as error:
raise ConsolidationError("Evidence files were saved, but indexing failed. Retry consolidation.", saved=True) from error
try:
archive.consolidate(actor=os.environ.get("THT_PRINCIPAL_SUBJECT") or "installation operator",
activate=activate)
except (ValueError, OSError) as error:
raise ConsolidationError(str(error)) from error
return result
def browse(root: Path, query: dict):
"""Read complete working units and their active status without opening an index."""
archive = LocalEvidenceArchive(root)
with archive.operation():
state = archive._state()
active = archive._snapshot(state["active"]) if state["active"] else None
active_units = archive._units(archive._files(active), allow_review=True) if active else {}
files = archive._files(archive.evidence)
items, errors = [], []
seen = set()
for relative, data in files.items():
try:
unit = parse_curated_markdown(data.decode(), path=Path(relative))
if unit.id in seen:
raise ValueError("Duplicate Evidence identifier")
seen.add(unit.id)
old = active_units.get(unit.id)
status = "review_required" if unit.review_items else "legacy" if unit.schema_version != 4 \
else "active" if old and _content(old[1]) == _content(unit) and old[1].provenance == unit.provenance else "modified" if old else "new"
items.append({**unit.model_dump(mode="json"), "file": relative, "status": status,
"revision": _content(unit)})
except (ValueError, UnicodeError) as error:
errors.append({"file": relative, "message": str(error)[:1500]})
for identity, (relative, unit) in active_units.items():
if identity not in seen:
items.append({**unit.model_dump(mode="json"), "file": relative,
"status": "invalid" if relative in files else "removed", "revision": _content(unit)})
def matches(item):
for field in ("kind", "status", "language"):
if query.get(field) and item[field] != query[field]:
return False
if query.get("purpose") and query["purpose"] not in item["purposes"]:
return False
for key, field in (("concept", "concepts"), ("table", "tables"), ("column", "columns")):
if query.get(key) and not any(query[key].casefold() in v.casefold() for v in item["applies_to"][field]):
return False
provenance = item["provenance"]
if query.get("source") and query["source"].casefold() not in str(provenance).casefold():
return False
return not query.get("q") or query["q"].casefold() in str(item).casefold()
selected = [item for item in items if matches(item)]
selected.sort(key=lambda item: (str(item.get(query.get("sort", "title"), "")).casefold(), item["id"]),
reverse=query.get("direction") == "desc")
page, size = int(query.get("page", 1)), int(query.get("page_size", 25))
if page < 1 or not 1 <= size <= 100:
raise ValueError("Invalid Evidence page")
result = {"items": selected[(page-1)*size:page*size], "total": len(selected), "page": page,
"page_size": size, "errors": errors, "active_revision": state["active"],
"pending_revision": state["pending"], "initialized": bool(state.get("baseline") or state["pending"] or state["active"])}
if query.get("id"):
result["item"] = next((item for item in items if item["id"] == query["id"]), None)
from .imports import reviews
result["source_reviews"] = reviews(archive)
return result
def source_action(config, *, action, source_id=None, revision=None, decision=None, actor="installation operator"):
from tht.config import load_config
from .authoring import PiEvidenceRestructurer, authoring_skill_path, migrate_workspace_evidence
from .imports import acquisition_sources, decide, refresh
cfg = load_config(config)
if not cfg.evidence or not cfg.evidence.local_archive_root:
raise ValueError("Local Evidence is not configured")
archive = LocalEvidenceArchive(cfg.evidence.local_archive_root)
if not (archive.metadata / "state.yaml").exists():
if (archive.evidence / "manifest.yaml").is_file():
migrate_workspace_evidence(archive.root)
else:
(archive.evidence / "curated").mkdir(parents=True, exist_ok=True)
archive.initialize()
if action == "refresh":
skill = authoring_skill_path()
return refresh(archive, acquisition_sources(cfg), PiEvidenceRestructurer(
os.environ.get("THT_PI_EXECUTABLE", "pi"), skill))
if action != "decide":
raise ValueError("Unknown source action")
def activate(snapshot):
from tht.cli.preprocess_cmd import run_from_config
try:
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise RuntimeError("Indexing did not succeed")
except Exception as error:
raise ConsolidationError("Source decision saved, but indexing failed. Retry the same decision.", saved=True) from error
try:
return decide(archive, source_id=source_id, revision=revision, decision=decision,
actor=actor, activate=activate)
except (ValueError, OSError) as error:
from .imports import reviews
if any(r["id"] == source_id and r["status"] == "applying" for r in reviews(archive)):
raise ConsolidationError(f"Source decision saved. {str(error)[:1200]}. Retry the same decision.", saved=True) from error
raise
+30 -5
View File
@@ -25,6 +25,7 @@ from tht.evidence.canonical import (
EvidenceKind,
EvidencePurpose,
EvidenceScope,
ManualEvidenceProvenance,
ReviewItem,
StrictModel,
dump_curated_markdown,
@@ -189,6 +190,12 @@ def _restore_exact_source_excerpts(
})
def authoring_skill_path() -> Path:
"""The installed wheel and the deployment's Pi resources live in different roots."""
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
@@ -389,6 +396,14 @@ def dump_manifest(manifest: EvidenceManifest) -> str:
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
"""Validate the curated corpus without writing the workspace."""
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").is_file():
from .local_archive import LocalEvidenceArchive
try:
LocalEvidenceArchive(workspace_root).validate()
return ValidationReport(())
except (OSError, ValueError) as error:
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
"evidence/curated", str(error)),))
findings: list[ValidationFinding] = []
manifest_path = evidence_root / "manifest.yaml"
if not manifest_path.is_file():
@@ -485,6 +500,9 @@ def _validate_manifest_source(
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
if isinstance(evidence.provenance, ManualEvidenceProvenance):
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
for item in evidence.review_items]
if evidence.id in manifest.orphans:
return []
path = evidence.provenance.source_file
@@ -560,6 +578,8 @@ def prepare_workspace_evidence(
raise EvidencePreparationError("authoring_workers_invalid")
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
manifest_path = evidence_root / "manifest.yaml"
try:
@@ -695,10 +715,9 @@ def migrate_workspace_evidence(
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
@@ -709,7 +728,7 @@ def migrate_workspace_evidence(
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 3})
evidence_id: document.model_copy(update={"schema_version": 4})
for evidence_id, document in documents_by_id.items()
}
migrated_ids: list[str] = []
@@ -728,12 +747,16 @@ def migrate_workspace_evidence(
migrated = tuple(sorted(migrated_ids))
unchanged = tuple(sorted(unchanged_ids))
if not migrated:
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=migrated,
unchanged=unchanged,
@@ -762,6 +785,8 @@ def resolve_workspace_evidence(
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local/state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
_reject_dirty_worktree(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
@@ -1017,7 +1042,7 @@ def _candidate_to_evidence(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 3
data["schema_version"] = 4
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
@@ -1038,7 +1063,7 @@ def _unsupported_unit(
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 3,
"schema_version": 4,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
+29 -3
View File
@@ -98,6 +98,19 @@ class ReviewItem(StrictModel):
field: str | None = None
class ManualEvidenceProvenance(StrictModel):
"""The curator supports the current content; an earlier document is only its origin."""
model_config = ConfigDict(extra="forbid", frozen=True)
kind: Literal["manual"] = "manual"
declared_by: str = Field(min_length=1)
original: EvidenceProvenance | None = None
@property
def source_file(self) -> str:
return f"Manual declaration: {self.declared_by}"
class FormulaPayload(StrictModel):
concept: str
columns: tuple[str, ...]
@@ -229,19 +242,23 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1, 2, 3]
schema_version: Literal[1, 2, 3, 4]
id: str
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
provenance: EvidenceProvenance
provenance: EvidenceProvenance | ManualEvidenceProvenance
review_items: tuple[ReviewItem, ...] = ()
payload: EvidencePayload
@model_validator(mode="after")
def _validate_kind_payload(self) -> CuratedEvidence:
if isinstance(self.provenance, ManualEvidenceProvenance) and self.schema_version != 4:
raise ValueError("manual declarations require Curated unit schema v4")
if self.schema_version == 4 and (not self.purposes or not self.language.strip()):
raise ValueError("editable Evidence requires a language and at least one purpose")
if not is_evidence_id(self.id):
raise ValueError("id must use the evidence:<slug> form")
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
@@ -1056,13 +1073,19 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
try:
raw = yaml.safe_load(frontmatter)
except yaml.YAMLError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
try:
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
elif data.get("schema_version") == 4:
from tht.evidence.editable import parse_document, parse_metadata
data = parse_document(parse_metadata(frontmatter), body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
@@ -1081,6 +1104,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render one canonical Curated Evidence Markdown document."""
if value.schema_version == 4:
from tht.evidence.editable import render_document
return render_document(value)
if value.schema_version == 3:
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
if value.schema_version == 2:
+10 -5
View File
@@ -8,7 +8,12 @@ from typing import Self
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
from tht.evidence.canonical import EVIDENCE_KINDS, EVIDENCE_PURPOSES
from tht.evidence.canonical import (
EVIDENCE_KINDS,
EVIDENCE_PURPOSES,
EvidenceProvenance,
ManualEvidenceProvenance,
)
from tht.evidence.contracts import (
canonical_provenance_uri,
normalize_aware_datetime,
@@ -76,10 +81,10 @@ def _validate_evidence_metadata(metadata: Mapping[str, JsonValue]) -> None:
if not isinstance(metadata["language"], str) or not metadata["language"]:
raise ValueError("typed Evidence metadata must contain language")
provenance = metadata["provenance"]
if not isinstance(provenance, dict) or set(provenance) != {
"source_file", "source_sha256", "supporting_excerpts",
}:
raise ValueError("typed Evidence metadata must contain canonical provenance")
if not isinstance(provenance, dict):
raise ValueError("typed Evidence metadata must contain canonical provenance") # noqa: TRY004
model = ManualEvidenceProvenance if provenance.get("kind") == "manual" else EvidenceProvenance
model.model_validate(provenance)
class _CanonicalValue(BaseModel):
+177
View File
@@ -0,0 +1,177 @@
"""Curated unit v4: visible Markdown fields are the sole human-content authority."""
from __future__ import annotations
import json
import re
import yaml
from .canonical import (
_PAYLOAD_TYPE_BY_KIND,
CuratedEvidence,
ManualEvidenceProvenance,
_v2_labels,
)
_LIST_FIELDS = {"synonyms", "variants", "columns", "tables"}
def parse_metadata(text: str) -> dict:
class UniqueKeysLoader(yaml.SafeLoader):
pass
def mapping(loader, node):
pairs = loader.construct_pairs(node, deep=True)
result = {}
for key, value in pairs:
if key in result:
raise ValueError(f"Duplicate metadata key: {key}")
result[key] = value
return result
UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping)
try:
return yaml.load(text, Loader=UniqueKeysLoader)
except (yaml.YAMLError, TypeError) as error:
raise ValueError("Evidence metadata is malformed") from error
def _sections(body: str, headings: dict[str, str]) -> dict[str, str]:
"""Recognize structural H2s outside code fences; all other Markdown is content."""
sections: dict[str, list[str]] = {}
field = None
fence = None
for line in body.splitlines():
match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line)
if match:
marker = match[1]
if fence is None:
fence = marker
elif marker[0] == fence[0] and len(marker) >= len(fence):
fence = None
heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None
if heading:
if heading in sections:
raise ValueError(f"Duplicate section: {line[3:]}")
field = heading
sections[field] = []
elif field is not None:
sections[field].append(line)
elif line.strip():
raise ValueError("Content must follow a documented section heading")
if fence:
raise ValueError("Unclosed Markdown code fence")
return {key: "\n".join(lines).strip() for key, lines in sections.items()}
def _list(text: str) -> list[str]:
if not text:
return []
values = []
for line in text.splitlines():
if not line.startswith("- ") or not line[2:].strip():
raise ValueError("List entries must use '- value', one per line")
value = line[2:]
# Quoted strings preserve multiline and unusual values during migration.
values.append(json.loads(value) if value.startswith('"') else value)
return values
def _values(text: str) -> dict[str, str]:
values = {}
key = None
lines = []
for line in text.splitlines():
if line.startswith("### "):
if key is not None:
values[key] = "\n".join(lines).strip()
label = line[4:]
key = json.loads(label) if label.startswith('"') else label
if key in values:
raise ValueError("Duplicate enum value")
lines = []
elif key is None:
if line.strip():
raise ValueError("Enum values require '### value' headings")
else:
lines.append(line)
if key is not None:
values[key] = "\n".join(lines).strip()
return values
def parse_document(metadata: dict, body: str) -> dict:
data = dict(metadata)
if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data):
raise ValueError("Title and payload must be edited only in the Markdown body")
lines = body.strip().splitlines()
if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip():
raise ValueError("A title starting with '# ' is required")
data["title"] = lines[0][2:].strip()
kind = data.get("kind")
if kind not in _PAYLOAD_TYPE_BY_KIND:
raise ValueError("Unknown Evidence kind")
labels = _v2_labels(str(data.get("language", "")))
fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields
sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields})
required = {name for name, field in fields.items() if field.is_required()}
if not required <= sections.keys():
raise ValueError(
"Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys()))
)
payload = {}
for key, content in sections.items():
if key in _LIST_FIELDS:
payload[key] = _list(content)
elif key == "values":
payload[key] = _values(content)
elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"):
payload[key] = content[7:-4]
else:
payload[key] = content
if fields[key].is_required() and not payload[key] and key != "values":
raise ValueError(f"Section {labels[key]} must not be empty")
data["payload"] = payload
data.setdefault(
"provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump()
)
return data
def render_document(value: CuratedEvidence) -> str:
metadata = value.model_dump(mode="json", exclude={"title", "payload"})
labels = _v2_labels(value.language)
parts = [
f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}"
]
for key, content in value.payload.model_dump(mode="json").items():
if key in _LIST_FIELDS:
rendered = "\n".join(
"- "
+ (
json.dumps(v, ensure_ascii=False)
if "\n" in v or v.startswith('"') or v != v.strip()
else v
)
for v in content
)
elif key == "values":
rendered = "\n\n".join(
f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items()
)
elif key == "sql":
rendered = f"```sql\n{content}\n```"
else:
rendered = content
parts.append(f"## {labels[key]}\n\n{rendered}")
rendered = "\n\n".join(parts) + "\n"
# Migration must fail explicitly rather than silently changing unrepresentable content.
restored = CuratedEvidence.model_validate(
parse_document(metadata, rendered.split("---\n", 2)[2])
)
if restored != value:
raise ValueError(
f"{value.id}: content cannot be represented losslessly in editable Markdown"
)
return rendered
+235
View File
@@ -0,0 +1,235 @@
"""Explicit acquisition and durable source comparisons; never implicit runtime refresh."""
import base64
import json
from .authoring import (
RestructureRequest,
_allocate_evidence_id,
_candidate_to_evidence,
normalize_source_text,
)
from .canonical import (
CuratedEvidence,
EvidenceProvenance,
ManualEvidenceProvenance,
dump_curated_markdown,
)
from .corpus.normalize import _decode
from .local_archive import ArchiveConflict, LocalEvidenceArchive, _atomic, _digest
MAX_DOCUMENTS = 200
MAX_TOTAL_BYTES = 100 * 1024 * 1024
def _origin(unit):
return unit.provenance.original if isinstance(unit.provenance, ManualEvidenceProvenance) else unit.provenance
def _records(archive):
path = archive.metadata / "sources.json"
if path.is_symlink():
raise ValueError("Source metadata must not use symlinks")
return json.loads(path.read_text()) if path.exists() else {}
def _write_records(archive, records):
_atomic(archive.metadata / "sources.json", json.dumps(records, ensure_ascii=False, sort_keys=True))
def reviews(archive):
return [{k: v for k, v in row.items() if k not in {"expected", "text", "writes", "approved"}}
for row in _records(archive).values()]
def acquisition_sources(cfg):
"""Local drafts and original files, plus configured read-only remote connectors."""
from .adapters import FilesystemEvidenceSource
from .sources import build_sources
root = cfg.evidence.local_archive_root / "evidence"
# Canonical acquired versions are immutable lineage, never new input documents.
patterns = [str(p.relative_to(root)) for folder in ("incoming", "source")
for p in sorted((root / folder).rglob("*.md"))
if not p.is_relative_to(root / "source/acquired")]
result = [FilesystemEvidenceSource(root, patterns=patterns)] if patterns else []
# Filesystem descriptors select the installation's local authoring tree after E2.
remote = cfg.evidence.model_copy(update={"source_root": None,
"sources": [s for s in cfg.evidence.sources if s.type != "filesystem"]})
result.extend(build_sources(remote, acquisition=True))
return result
def refresh(archive: LocalEvidenceArchive, sources, restructurer):
"""Acquire everything successfully before recording proposals. Missing is never deletion."""
with archive.operation():
state = archive._state()
if state.get("pending") or state.get("import_writes"):
raise ArchiveConflict("Complete the pending consolidation before refreshing sources")
records = _records(archive)
if any(r["status"] == "applying" for r in records.values()):
raise ArchiveConflict("Retry the pending source decision before refreshing")
files = archive._files(archive.evidence)
units = archive._units(files, allow_review=True)
documents, total = {}, 0
for adapter in sources:
for item in adapter.discover():
document = adapter.acquire(item)
total += len(document.content)
if len(documents) >= MAX_DOCUMENTS or total > MAX_TOTAL_BYTES:
raise ValueError("Source refresh exceeds the local acquisition limit")
relative = item.metadata.get("relative_path") if item.uri.startswith("file:") else None
identity = f"local:{relative}" if relative else item.uri
key = _digest(identity.encode())
if key in documents:
raise ValueError("Duplicate acquisition identity")
documents[key] = (document, relative)
reserved = set(units) | set(state["deleted_ids"])
for record in records.values():
reserved.update(u["id"] for u in record.get("proposed", []))
changed, unchanged = 0, 0
acquired = {}
for key, (document, relative) in documents.items():
text = normalize_source_text(_decode(document))
sha = "sha256:" + _digest(text.encode())
old = records.get(key)
stale = old and old["status"] == "review" and any(
p not in files or _digest(files[p]) != h for p, h in old["expected"].items())
if old and old["sha256"] == sha and not stale:
old["availability"] = "available"
unchanged += 1
continue
current = {i: pair for i, pair in units.items()
if (old and i in old["unit_ids"]) or
(_origin(pair[1]) and _origin(pair[1]).source_file == relative)}
# Seed imported E2 document identity without asking the model to recurate unchanged text.
if old is None and current and all(_origin(u).source_sha256 == sha for _, u in current.values()):
records[key] = {"id": key, "uri": document.source.uri, "legacy_file": relative,
"sha256": sha, "unit_ids": sorted(current), "status": "accepted",
"availability": "available", "revision": sha[7:], "proposed": []}
unchanged += 1
continue
source_file = f"source/acquired/{key}/{sha[7:]}.md"
request = RestructureRequest(source_file=source_file, source_sha256=sha,
normalized_text=text, previous_units=tuple(u for _, u in current.values()))
proposed = []
seen = set()
suppressed = relative in state["suppressed_sources"] or any(
p.startswith(f"source/acquired/{key}/") for p in state["suppressed_sources"]) or (old and (
old.get("suppressed", False) or any(i in state["deleted_ids"] for i in old["unit_ids"])))
for candidate in restructurer.restructure(request):
identity = candidate.existing_id
if identity in state["deleted_ids"]:
continue
if identity is not None and identity not in current:
raise ValueError("Source proposal refers to an unrelated Evidence identity")
if identity is None:
if suppressed:
continue # A model-created identifier cannot bypass a curated deletion.
identity = _allocate_evidence_id(candidate.title, reserved)
reserved.add(identity)
if identity in seen:
raise ValueError("Source proposal repeats an Evidence identity")
seen.add(identity)
unit = _candidate_to_evidence(candidate, identity, source_file, sha)
dump_curated_markdown(unit) # Refuse an uneditable proposal before saving any review.
if any(excerpt not in text for excerpt in unit.provenance.supporting_excerpts):
raise ValueError("Source proposal contains an excerpt absent from the acquired document")
proposed.append(unit.model_dump(mode="json"))
expected = {p: _digest(files[p]) for p, _ in current.values()}
row = {"id": key, "uri": document.source.uri, "legacy_file": relative,
"sha256": sha, "source_file": source_file, "text": text,
"status": "review", "availability": "available", "suppressed": bool(suppressed),
"unit_ids": sorted(current), "expected": expected,
"current": [u.model_dump(mode="json") for _, u in current.values()], "proposed": proposed,
"removed_ids": sorted(set(current) - seen)}
row["revision"] = _digest(json.dumps(row, sort_keys=True).encode())
records[key] = row
acquired[key] = {"source": document.source.model_dump(mode="json"),
"media_type": document.media_type, "raw_base64": base64.b64encode(document.content).decode()}
changed += 1
for key, row in records.items():
if key not in documents:
row["availability"] = "missing"
# No writes above: an access/model failure preserves every previous review and active unit.
for key, value in acquired.items():
directory = archive.metadata / "acquisitions" / key
if any(p.is_symlink() for p in [directory, directory.parent]):
raise ValueError("Acquisition metadata must not use symlinks")
_atomic(directory / f"{records[key]['sha256'][7:]}.json", json.dumps(value, ensure_ascii=False))
_write_records(archive, records)
return {"status": "succeeded", "counts": {"changed": changed, "unchanged": unchanged,
"review": sum(r["status"] == "review" for r in records.values())}}
def decide(archive, *, source_id, revision, decision, actor, activate):
if decision not in {"keep", "replace"} or not actor.strip():
raise ValueError("Choose keep or replace and supply a curator")
with archive.operation():
records = _records(archive)
row = records.get(source_id)
if row is None or row["revision"] != revision:
raise ArchiveConflict("Source comparison changed; refresh the page")
if row["status"] == "applying":
if row["decision"] != decision:
raise ArchiveConflict("Retry the saved source decision before changing it")
state = archive._state()
state.update(import_writes=row["writes"], approved_imports=row["approved"])
archive._write_state(state)
result = archive._consolidate(actor, activate)
else:
if row["status"] != "review":
raise ArchiveConflict("This source comparison was already decided")
state = archive._state()
if state.get("pending") or state.get("import_writes"):
raise ArchiveConflict("Complete the pending consolidation first")
files = archive._files(archive.evidence)
units = archive._units(files, allow_review=True)
if any(_digest(files[p]) != h if p in files else True for p, h in row["expected"].items()):
raise ArchiveConflict("Curated files changed since source review; refresh the source comparison")
selected = [CuratedEvidence.model_validate(v) for v in row["proposed"]] if decision == "replace" else [units[i][1] for i in row["unit_ids"]]
writes, approved = {}, {}
def write(path, content):
current = archive.evidence / path
writes[path] = {"before": _digest(current.read_bytes()) if current.exists() else None, "after": content}
if decision == "replace":
write(row["source_file"], row["text"])
for identity in row["unit_ids"]:
write(units[identity][0], None)
for value in selected:
if value.review_items:
raise ValueError("The proposal needs review; correct the input draft and refresh, or keep local content")
if decision == "keep":
origin = _origin(value)
if origin:
old = archive._snapshot(state["active"] or state["baseline"])
candidate = old / origin.source_file
if not candidate.is_file():
candidate = archive.evidence / origin.source_file
text = normalize_source_text(candidate.read_text())
if "sha256:" + _digest(text.encode()) != origin.source_sha256:
raise ValueError("The original source version is unavailable")
path = f"source/acquired/{source_id}/{origin.source_sha256[7:]}.md"
write(path, text)
origin = EvidenceProvenance(source_file=path, source_sha256=origin.source_sha256,
supporting_excerpts=origin.supporting_excerpts)
value = value.model_copy(update={"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)})
if value.id in units and value.id not in row["unit_ids"]:
raise ArchiveConflict("A proposed Evidence identity was created elsewhere")
path = units[value.id][0] if value.id in units and units[value.id][1].kind == value.kind else f"curated/{value.kind}/{value.id[9:]}.md"
if path in files and value.id not in units:
raise ArchiveConflict("The proposed file path is occupied")
write(path, dump_curated_markdown(value))
approved[value.id] = _digest(dump_curated_markdown(value).encode())
# Journal before applying files; retry never silently clobbers an external edit.
row.update(status="applying", decision=decision, decided_by=actor,
next_unit_ids=[u.id for u in selected], writes=writes, approved=approved)
_write_records(archive, records)
state.update(import_writes=writes, approved_imports=approved)
archive._write_state(state)
result = archive._consolidate(actor, activate)
if result["status"] != "active":
raise ValueError("Source decision was saved but not activated; retry")
row.update(status="accepted" if decision == "replace" else "kept", unit_ids=row["next_unit_ids"])
_write_records(archive, records)
return {"status": "succeeded", "counts": {"units": result["units"]}}
+472
View File
@@ -0,0 +1,472 @@
"""Persistent curated files, immutable consolidation candidates and explicit activation.
The working tree is primary data. Core consumers use only active_snapshot(); an
editor save or failed activation never switches that pointer. No Git/network/DWH I/O.
"""
from __future__ import annotations
import fcntl
import hashlib
import json
import os
import tempfile
from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
import yaml
from .canonical import (
MAX_CURATED_FILE_BYTES,
CuratedEvidence,
EvidenceProvenance,
ManualEvidenceProvenance,
dump_curated_markdown,
parse_curated_markdown,
)
class ArchiveConflict(ValueError):
"""The curator must reconcile a concurrent change before replacing it."""
def _digest(value: bytes) -> str:
return hashlib.sha256(value).hexdigest()
def _content(unit: CuratedEvidence) -> str:
return _digest(
json.dumps(
unit.model_dump(mode="json", exclude={"schema_version", "provenance"}),
sort_keys=True,
ensure_ascii=False,
).encode()
)
def _atomic(path: Path, data: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent)
try:
with os.fdopen(fd, "w") as handle:
handle.write(data)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary, path)
finally:
if os.path.exists(temporary):
os.unlink(temporary)
class LocalEvidenceArchive:
def __init__(self, workspace_root: Path):
self.root = workspace_root.resolve()
self.evidence = self.root / "evidence"
self.metadata = self.evidence / ".local"
if self.evidence.is_symlink() or self.metadata.is_symlink():
raise ValueError("The local Evidence archive must use persistent regular directories")
@contextmanager
def operation(self):
if any(
path.is_symlink()
for path in (
self.evidence,
self.metadata,
self.metadata / "snapshots",
self.metadata / "state.yaml",
)
):
raise ValueError("Evidence archive metadata must not use symlinks")
self.root.mkdir(parents=True, exist_ok=True)
lock = self.root / ".evidence-archive.lock"
if lock.is_symlink():
raise ValueError("Evidence lock must not be a symlink")
with lock.open("a") as handle:
fcntl.flock(handle, fcntl.LOCK_EX)
try:
yield
finally:
fcntl.flock(handle, fcntl.LOCK_UN)
def _state(self):
path = self.metadata / "state.yaml"
if not path.exists():
return {
"schema_version": 1,
"active": None,
"pending": None,
"baseline": None,
"deleted_ids": [],
"suppressed_sources": [],
}
value = yaml.safe_load(path.read_text())
if not isinstance(value, dict) or value.get("schema_version") != 1:
raise ValueError("Unsupported local Evidence archive state")
return value
def _write_state(self, state):
_atomic(self.metadata / "state.yaml", yaml.safe_dump(state, sort_keys=True))
def _snapshot(self, revision: str) -> Path:
if (
not isinstance(revision, str)
or len(revision) != 64
or any(c not in "0123456789abcdef" for c in revision)
):
raise ValueError("Invalid Evidence snapshot identity")
path = self.metadata / "snapshots" / revision
if not path.is_dir() or path.is_symlink():
raise ValueError("Consolidated Evidence snapshot is missing")
return path
def active_snapshot(self) -> Path | None:
"""Only this immutable source is eligible for core consumption."""
with self.operation():
revision = self._state()["active"]
return self._snapshot(revision) if revision else None
def validate(self):
"""Check the working files without modifying them or changing active content."""
with self.operation():
units = self._units(self._files(self.evidence))
state = self._state()
revision = state["pending"] or state["active"] or state.get("baseline")
previous = (
self._units(self._files(self._snapshot(revision)), allow_review=True)
if revision
else {}
)
for identity, (_, unit) in units.items():
old = previous.get(identity)
if isinstance(unit.provenance, EvidenceProvenance) and (
old is None or _content(old[1]) == _content(unit)
):
self._validate_document_source(unit.provenance)
return len(units)
def _files(self, root: Path):
result = {}
curated = root / "curated"
if curated.is_symlink():
raise ValueError("Curated directory must not be a symlink")
if not curated.is_dir():
raise ValueError(f"{curated}: curated archive is unavailable; absence is not deletion")
for path in sorted(curated.rglob("*")):
if path.is_symlink():
raise ValueError(f"{path}: symlinks are not supported in the curated archive")
if path.suffix != ".md" or path.name.upper().startswith("README"):
continue
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{path}: Evidence file exceeds the size limit")
result[path.relative_to(root).as_posix()] = path.read_bytes()
return result
def _units(self, files, *, allow_review=False):
units = {}
for relative, data in files.items():
try:
unit = parse_curated_markdown(data.decode("utf-8"), path=Path(relative))
if unit.schema_version != 4:
raise ValueError("Run evidence migrate before consolidating legacy units")
if unit.id in units:
raise ValueError(f"Duplicate Evidence identity {unit.id}")
if unit.review_items and not allow_review:
raise ValueError("Resolve review items before consolidation")
units[unit.id] = (relative, unit)
except ValueError as error:
raise ValueError(f"{relative}: {error}") from error
return units
def initialize(self):
"""Capture migrated/refined content before edits, without activating review items."""
with self.operation():
state = self._state()
if state.get("baseline") or state["active"] or state["pending"]:
return
files = self._files(self.evidence)
self._units(files, allow_review=True)
source = self.evidence / "source"
if source.is_symlink():
raise ValueError("Source symlinks are not supported")
if source.exists():
for path in sorted(source.rglob("*")):
if path.is_symlink():
raise ValueError("Source symlinks are not supported")
if path.is_file():
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{path}: source exceeds the size limit")
files[path.relative_to(self.evidence).as_posix()] = path.read_bytes()
revision = _digest(b"".join(k.encode() + b"\0" + v for k, v in sorted(files.items())))
snapshots = self.metadata / "snapshots"
snapshots.mkdir(parents=True, exist_ok=True)
destination = snapshots / revision
if not destination.exists():
with tempfile.TemporaryDirectory(prefix=".baseline-", dir=snapshots) as tmp:
candidate = Path(tmp) / "snapshot"
(candidate / "curated").mkdir(parents=True)
for relative, data in files.items():
path = candidate / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
os.replace(candidate, destination)
state["baseline"] = revision
self._write_state(state)
def consolidate(self, *, actor: str, activate: Callable[[Path], None] | None = None):
"""Persist one valid candidate; optionally activate it through the existing index stage.
A missing callback deliberately reports pending_activation, never success.
A retry of unchanged files reuses the same candidate after an index failure.
"""
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
return self._consolidate(actor, activate)
def _consolidate(self, actor, activate):
state = self._state()
self._finish_import(state)
self._finish_normalization(state)
original_files = self._files(self.evidence)
units = self._units(original_files)
previous_revision = state["pending"] or state["active"] or state.get("baseline")
previous_root = self._snapshot(previous_revision) if previous_revision else None
previous = (
self._units(self._files(previous_root), allow_review=True) if previous_root else {}
)
deleted = set(state["deleted_ids"]) | (previous.keys() - units.keys())
suppressed = set(state["suppressed_sources"])
for identity in previous.keys() - units.keys():
provenance = previous[identity][1].provenance
source = (
provenance.original
if isinstance(provenance, ManualEvidenceProvenance)
else provenance
)
if source:
suppressed.add(source.source_file)
files = {}
for identity, (relative, unit) in units.items():
old = previous.get(identity)
changed = old is not None and _content(old[1]) != _content(unit)
approved = state.get("approved_imports", {}).get(identity) == _digest(
dump_curated_markdown(unit).encode()
)
if approved:
pass # An explicit source decision authorized this exact content and provenance.
elif changed or (
isinstance(unit.provenance, ManualEvidenceProvenance)
and (old is None or unit.provenance.declared_by == "local curator")
):
origin = old[1].provenance if old else unit.provenance
origin = origin.original if isinstance(origin, ManualEvidenceProvenance) else origin
unit = unit.model_copy(
update={
"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)
}
)
elif old and unit.provenance != old[1].provenance:
raise ArchiveConflict(
f"{relative}: provenance is managed; edit the content instead"
)
if isinstance(unit.provenance, EvidenceProvenance):
self._validate_document_source(unit.provenance)
files[relative] = dump_curated_markdown(unit).encode()
origin = (
unit.provenance.original
if isinstance(unit.provenance, ManualEvidenceProvenance)
else unit.provenance
)
if origin:
candidates = [previous_root / origin.source_file] if previous_root else []
candidates.append(self.evidence / origin.source_file)
from .authoring import normalize_source_text
for source in candidates:
if source.is_file() and not source.is_symlink():
if source.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{source}: source exceeds the size limit")
raw = source.read_bytes()
if (
"sha256:" + _digest(normalize_source_text(raw.decode()).encode())
== origin.source_sha256
):
if origin.source_file in files and files[origin.source_file] != raw:
raise ArchiveConflict(
"Different source revisions require explicit source resolution"
)
files[origin.source_file] = raw
break
else:
raise ValueError(
f"{origin.source_file}: the recorded original document is unavailable"
)
# Record current declarations and original document lineage distinctly.
manifest = {
"schema_version": 1,
"units": {
identity: {
"file": relative,
"content_hash": _content(parse_curated_markdown(files[relative].decode())),
"provenance": parse_curated_markdown(
files[relative].decode()
).provenance.model_dump(mode="json"),
}
for identity, (relative, _) in sorted(units.items())
},
"deleted_ids": sorted(deleted),
"suppressed_sources": sorted(suppressed),
}
files["local-manifest.yaml"] = yaml.safe_dump(
manifest, allow_unicode=True, sort_keys=True
).encode()
revision = _digest(
b"".join(path.encode() + b"\0" + data + b"\0" for path, data in sorted(files.items()))
)
snapshots = self.metadata / "snapshots"
snapshots.mkdir(parents=True, exist_ok=True)
destination = snapshots / revision
if not destination.exists():
with tempfile.TemporaryDirectory(prefix=".candidate-", dir=snapshots) as tmp:
candidate = Path(tmp) / "snapshot"
(candidate / "curated").mkdir(parents=True)
for relative, data in files.items():
path = candidate / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
os.replace(candidate, destination)
if self._files(self.evidence) != original_files:
raise ArchiveConflict(
"Evidence files changed during consolidation; retry with the current files"
)
# The candidate exists first. Pending state makes every subsequent interruption recoverable.
state.update(
pending=revision,
deleted_ids=sorted(deleted),
suppressed_sources=sorted(suppressed),
normalization={relative: _digest(data) for relative, data in original_files.items()},
)
state.pop("approved_imports", None)
self._write_state(state)
self._finish_normalization(state)
if activate is None:
return {
"status": "pending_activation",
"revision": revision,
"snapshot": str(destination),
}
activate(destination)
state.update(active=revision, pending=None)
self._write_state(state)
return {
"status": "active",
"revision": revision,
"snapshot": str(destination),
"units": len(units),
"deleted": len(previous.keys() - units.keys()),
}
def _finish_import(self, state):
"""Replay an explicit source decision, rejecting intervening external edits."""
writes = state.get("import_writes")
if writes is None:
return
for relative, change in writes.items():
path = self.evidence / relative
if not relative.startswith(("curated/", "source/acquired/")) or ".." in Path(relative).parts:
raise ValueError("Invalid import destination")
if any(p.is_symlink() for p in [path, *path.parents] if p != self.root.parent):
raise ValueError("Import destinations must not use symlinks")
current = _digest(path.read_bytes()) if path.exists() else None
after = _digest(change["after"].encode()) if change["after"] is not None else None
if current not in (change["before"], after):
raise ArchiveConflict("Evidence changed during a source decision; restore or review the file")
for relative, change in writes.items():
path = self.evidence / relative
if change["after"] is None:
path.unlink(missing_ok=True)
else:
_atomic(path, change["after"])
del state["import_writes"]
self._write_state(state)
def _finish_normalization(self, state):
"""Replay interrupted managed writes only where the user's bytes are unchanged."""
if "normalization" not in state:
return
snapshot = self._snapshot(state["pending"])
current = self._files(self.evidence)
for relative, original_hash in state["normalization"].items():
if relative in current and _digest(current[relative]) == original_hash:
_atomic(self.evidence / relative, (snapshot / relative).read_text())
_atomic(
self.evidence / "local-manifest.yaml", (snapshot / "local-manifest.yaml").read_text()
)
del state["normalization"]
self._write_state(state)
def _validate_document_source(self, provenance):
from .authoring import _normalize, normalize_source_text
path = self.evidence / provenance.source_file
if (
path.is_symlink()
or not path.is_file()
or not path.resolve().is_relative_to(self.evidence.resolve())
):
raise ValueError(f"{provenance.source_file}: source document is unavailable")
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{provenance.source_file}: source exceeds the size limit")
source = normalize_source_text(path.read_text())
if "sha256:" + _digest(source.encode()) != provenance.source_sha256:
raise ArchiveConflict(
f"{provenance.source_file}: source changed; explicit source refresh is required"
)
if any(_normalize(excerpt) not in source for excerpt in provenance.supporting_excerpts):
raise ValueError(f"{provenance.source_file}: supporting excerpt is missing")
def save(self, value: CuratedEvidence, *, expected_revision: str | None, actor: str):
"""Explicit workflow correction with optimistic concurrency; activation is a separate boundary."""
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
return self._save(value, expected_revision=expected_revision, actor=actor)
def _save(self, value, *, expected_revision, actor):
"""Save while the caller holds operation(), including workflow receipt recovery."""
self._finish_normalization(self._state())
units = self._units(self._files(self.evidence))
existing = units.get(value.id)
if (existing is None) != (expected_revision is None):
raise ArchiveConflict("Evidence was created or removed since review")
if existing and _content(existing[1]) != expected_revision:
raise ArchiveConflict("Evidence changed since review")
relative = (
existing[0]
if existing
else f"curated/{value.kind}/{value.id.removeprefix('evidence:')}.md"
)
if existing and existing[1].kind != value.kind:
raise ValueError("An update cannot change the Evidence kind")
if existing:
value = value.model_copy(update={"provenance": existing[1].provenance})
_atomic(self.evidence / relative, dump_curated_markdown(value))
return self._consolidate(actor, None)
def get(self, identity: str):
with self.operation():
relative, unit = self._units(self._files(self.evidence))[identity]
return {"unit": unit, "revision": _content(unit), "path": str(self.evidence / relative)}
def remove(self, identity: str, *, expected_revision: str, actor: str):
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
self._finish_normalization(self._state())
relative, unit = self._units(self._files(self.evidence))[identity]
if _content(unit) != expected_revision:
raise ArchiveConflict("Evidence changed since review")
(self.evidence / relative).unlink()
return self._consolidate(actor, None)
+9 -1
View File
@@ -12,10 +12,18 @@ if TYPE_CHECKING:
from tht.config import EvidenceSourcesConfig
def build_sources(evidence: EvidenceSourcesConfig | None) -> list[EvidenceSource]:
def build_sources(evidence: EvidenceSourcesConfig | None, *, acquisition: bool = False) -> list[EvidenceSource]:
"""Build configured Evidence adapters in the existing deterministic order."""
if evidence is None:
return []
if evidence.local_archive_root is not None and not acquisition:
from tht.evidence.local_archive import LocalEvidenceArchive
archive = LocalEvidenceArchive(evidence.local_archive_root)
if (archive.metadata / "state.yaml").exists():
snapshot = archive.active_snapshot()
if snapshot is None:
raise ValueError("Local Evidence has not been activated; run Evidence consolidation")
return [FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))]
sources: list[EvidenceSource] = []
if evidence.source_root is not None:
legacy_root = evidence.source_root / evidence.evidence_dir
+82
View File
@@ -0,0 +1,82 @@
"""Apply only removals confirmed by a successful Catalog physical synchronization."""
import json
from pydantic import BaseModel, ConfigDict, Field
from sqlalchemy import text
from .models import MemoryConflict
from .review import digest
class RemovedColumn(BaseModel):
model_config = ConfigDict(extra="forbid")
table: str = Field(min_length=1, max_length=200)
column: str = Field(min_length=1, max_length=200)
class CleanupRequest(BaseModel):
model_config = ConfigDict(extra="forbid")
sync_id: str = Field(min_length=1, max_length=200)
database: str = Field(min_length=1, max_length=200)
schema_name: str = Field(min_length=1, max_length=200)
removed_tables: list[str] = Field(default_factory=list, max_length=10000)
removed_columns: list[RemovedColumn] = Field(default_factory=list, max_length=100000)
def cleanup(service, request: CleanupRequest):
service._admin()
fingerprint = digest(request.model_dump(mode="json"))
with service.repository.operation() as repo:
with repo.transaction() as connection:
params = {"w": repo.workspace_id, "sync": request.sync_id}
receipt = (
connection.execute(
text(
"SELECT request_hash,card_ids "
"FROM thoth_memory.cleanup_receipts WHERE workspace_id=:w AND sync_id=:sync"
),
params,
)
.mappings()
.first()
)
if receipt:
if receipt["request_hash"] != fingerprint:
raise MemoryConflict(
"Physical cleanup identity was reused with different removals"
)
identities = receipt["card_ids"]
else:
identities = list(
connection.execute(
text(
"SELECT DISTINCT card_id "
"FROM thoth_memory.dependencies d WHERE workspace_id=:w "
"AND database_id=:db AND schema_name=:schema AND (table_name=ANY(:tables) "
"OR EXISTS (SELECT 1 FROM jsonb_to_recordset(CAST(:columns AS jsonb)) "
'AS removed("table" text, "column" text) WHERE '
'removed."table"=d.table_name AND removed."column"=d.column_name))'
),
{
**params,
"db": request.database,
"schema": request.schema_name,
"tables": request.removed_tables,
"columns": json.dumps(
[c.model_dump() for c in request.removed_columns]
),
},
).scalars()
)
for identity in identities:
repo.delete(identity)
connection.execute(
text(
"INSERT INTO thoth_memory.cleanup_receipts "
"VALUES (:w,:sync,:hash,CAST(:ids AS jsonb))"
),
{**params, "hash": fingerprint, "ids": json.dumps(identities)},
)
results = [service._propagate(repo, identity) for identity in identities]
return {"deleted": len(identities), "indexed": all(r["indexed"] for r in results)}
+1 -1
View File
@@ -25,7 +25,7 @@ class MemoryRecord(BaseModel):
concepts: list[str] = []
_MEM_ID_RE = re.compile(r"\bmem-\d{4,}\b")
_MEM_ID_RE = re.compile(r"\bmem-(?:[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}|\d{4,})\b")
def decided_memory_ids(decisions: list[DecisionRecord]) -> set[str]:
+74
View File
@@ -0,0 +1,74 @@
"""Versioned Memory migration pack, run only by installation preparation."""
import hashlib
import os
from functools import lru_cache
from importlib.resources import files
from pathlib import Path
from sqlalchemy import URL, create_engine, text
from sqlalchemy.pool import NullPool
def installation_url(*, migrator: bool = False) -> str:
role = "MIGRATOR" if migrator else "RUNTIME"
direct = os.environ.get(f"THT_CATALOG_{role}_DATABASE_URL")
if not migrator:
direct = direct or os.environ.get("THT_CATALOG_DATABASE_URL")
if direct:
return direct.replace("postgresql://", "postgresql+psycopg2://", 1)
prefix = "THT_CATALOG_"
try:
password = Path(os.environ[prefix + role + "_PASSWORD_FILE"]).read_text().strip()
url = URL.create(
"postgresql+psycopg2", host=os.environ[prefix + "DB_HOST"],
port=int(os.environ.get(prefix + "DB_PORT", "5432")),
database=os.environ[prefix + "DB_NAME"],
username=os.environ[prefix + role + "_USER"], password=password,
)
return url.render_as_string(hide_password=False)
except (KeyError, OSError, ValueError):
raise ValueError("Memory PostgreSQL installation configuration is unavailable") from None
@lru_cache(maxsize=1)
def expected_migrations() -> dict[str, str]:
return {p.name: hashlib.sha256(p.read_text().encode()).hexdigest()
for p in files("tht").joinpath("migrations/memory").iterdir()
if p.name.endswith(".sql")}
def migrate(database_url: str) -> None:
engine = create_engine(database_url, poolclass=NullPool)
try:
with engine.begin() as connection:
connection.execute(text("SELECT pg_advisory_xact_lock(792114203)"))
connection.execute(text("CREATE SCHEMA IF NOT EXISTS thoth_memory"))
connection.execute(text("CREATE TABLE IF NOT EXISTS thoth_memory.migrations "
"(version text PRIMARY KEY, checksum text NOT NULL)"))
applied = dict(connection.execute(text(
"SELECT version, checksum FROM thoth_memory.migrations"
)).all())
pack = sorted(files("tht").joinpath("migrations/memory").iterdir(), key=lambda p: p.name)
known = {p.name for p in pack if p.name.endswith(".sql")}
if set(applied) - known:
raise ValueError("Memory schema is newer than this application")
for path in pack:
if path.name not in known:
continue
sql = path.read_text()
digest = hashlib.sha256(sql.encode()).hexdigest()
if path.name in applied:
if applied[path.name] != digest:
raise ValueError("Memory migration checksum mismatch")
continue
connection.execute(text(sql))
connection.execute(text("INSERT INTO thoth_memory.migrations VALUES (:v, :c)"),
{"v": path.name, "c": digest})
finally:
engine.dispose()
if __name__ == "__main__":
migrate(installation_url(migrator=True))
print("Memory migrations: ready")
+112
View File
@@ -0,0 +1,112 @@
"""Authoritative Memory contracts, independent of workflow decision kinds."""
from datetime import datetime
from typing import Literal, Self
from pydantic import BaseModel, ConfigDict, Field, model_validator
Family = Literal["domain_clarification", "sql_rule", "solved_question", "explained_error"]
class Dependency(BaseModel):
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
database: str = Field(min_length=1, max_length=200)
schema_name: str = Field(default="", max_length=200)
table: str = Field(default="", max_length=200)
column: str = Field(default="", max_length=200)
@model_validator(mode="after")
def structured(self) -> Self:
if self.column and not self.table:
raise ValueError("A column dependency requires a table")
if self.table and not self.schema_name:
raise ValueError("A table dependency requires a schema")
return self
class LinkInput(BaseModel):
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
target_id: str = Field(min_length=1, max_length=100)
meaning: str = Field(min_length=1, max_length=1000)
class CardInput(BaseModel):
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
family: Family
subject: str = Field(min_length=1, max_length=1000)
detail: str = Field(default="", max_length=50000)
scope: str = Field(min_length=1, max_length=10000)
rationale: str = Field(default="", max_length=10000)
question: str = Field(default="", max_length=10000)
sql: str = Field(default="", max_length=100000)
concepts: list[str] = Field(default_factory=list, max_length=100)
dependencies: list[Dependency] = Field(default_factory=list, max_length=200)
links: list[LinkInput] = Field(default_factory=list, max_length=200)
@model_validator(mode="after")
def valid_family(self) -> Self:
if self.family == "solved_question" and (not self.question or not self.sql):
raise ValueError("A solved question requires its question and approved SQL")
if self.family == "explained_error" and (not self.detail or not self.rationale):
raise ValueError("An explained error requires a correction and rationale")
if self.family != "solved_question" and self.sql:
raise ValueError("Only solved questions carry exemplar SQL")
if any(not c.strip() or len(c) > 200 for c in self.concepts):
raise ValueError("Concepts must contain between 1 and 200 characters")
if len({link.target_id for link in self.links}) != len(self.links):
raise ValueError("Each linked destination must be unique")
return self
class Card(CardInput):
id: str
workspace_id: str
origin: Literal["manual", "workflow"]
session_id: str | None = None
decision_seq: int | None = None
created_at: datetime
updated_at: datetime
revision: str
indexed: bool = False
class CardQuery(BaseModel):
model_config = ConfigDict(extra="forbid")
q: str = Field(default="", max_length=1000)
family: Family | None = None
concept: str = Field(default="", max_length=200)
database: str = Field(default="", max_length=200)
table: str = Field(default="", max_length=200)
column: str = Field(default="", max_length=200)
origin: Literal["manual", "workflow"] | None = None
updated_after: datetime | None = None
updated_before: datetime | None = None
page: int = Field(default=1, ge=1)
page_size: int = Field(default=25, ge=1, le=100)
sort: Literal["updated_at", "created_at", "subject", "family"] = "updated_at"
direction: Literal["asc", "desc"] = "desc"
class MemoryError(Exception):
code = "memory_operation_failed"
status = 500
class MemoryUnavailable(MemoryError):
code = "memory_unavailable"
status = 503
class MemoryNotFound(MemoryError):
code = "memory_not_found"
status = 404
class MemoryForbidden(MemoryError):
code = "memory_forbidden"
status = 403
class MemoryConflict(MemoryError):
code = "memory_conflict"
status = 409
+245
View File
@@ -0,0 +1,245 @@
"""Workspace-scoped PostgreSQL persistence and durable projection work."""
import json
from contextlib import contextmanager
from uuid import uuid4
from sqlalchemy import create_engine, text
from sqlalchemy.exc import IntegrityError, SQLAlchemyError
from sqlalchemy.pool import NullPool
from .migrate import expected_migrations
from .models import Card, CardInput, CardQuery, MemoryConflict, MemoryNotFound, MemoryUnavailable
class MemoryRepository:
def __init__(self, database_url: str, workspace_id: str, *, engine=None, connection=None):
self.workspace_id = workspace_id
self.engine = engine or create_engine(
database_url, poolclass=NullPool, connect_args={"connect_timeout": 5},
)
self.connection = connection
def close(self):
self.engine.dispose()
@contextmanager
def transaction(self):
connection = self.connection
try:
connection = connection or self.engine.connect()
with (connection.begin_nested() if connection.in_transaction() else connection.begin()):
connection.execute(text("SET LOCAL ROLE thoth_memory_runtime"))
connection.execute(text("SELECT set_config('thoth.memory_workspace', :w, true)"),
{"w": self.workspace_id})
connection.execute(text("SET LOCAL statement_timeout = '15s'"))
installed = dict(connection.execute(text(
"SELECT version, checksum FROM thoth_memory.migrations"
)).all())
if installed != expected_migrations():
raise MemoryUnavailable("Memory schema is incompatible; run installation migrations")
yield connection
except IntegrityError:
raise MemoryConflict("Memory references conflict with the current archive") from None
except SQLAlchemyError:
raise MemoryUnavailable("Memory archive is unavailable; check its migrations and access") \
from None
finally:
if self.connection is None and connection is not None:
connection.close()
@contextmanager
def operation(self):
"""Serialize each workspace across SQL commits and the bounded vector call."""
try:
connection = self.engine.connect()
connection.execute(text("SET statement_timeout = '15s'"))
connection.execute(text("SELECT pg_advisory_lock(hashtextextended(:w, 792114204))"),
{"w": self.workspace_id})
connection.commit()
except SQLAlchemyError:
if 'connection' in locals():
connection.close()
raise MemoryUnavailable("Memory archive is busy or unavailable") from None
try:
yield MemoryRepository("", self.workspace_id, engine=self.engine, connection=connection)
finally:
# NullPool closes the physical connection, releasing the session advisory lock.
connection.close()
def _card(self, connection, row) -> Card:
params = {"w": self.workspace_id, "id": row["id"]}
links = connection.execute(text(
"SELECT target_id, meaning FROM thoth_memory.links "
"WHERE workspace_id=:w AND source_id=:id ORDER BY target_id"
), params).mappings().all()
dependencies = connection.execute(text(
'SELECT database_id AS database, schema_name, table_name AS "table", '
'column_name AS "column" FROM thoth_memory.dependencies '
"WHERE workspace_id=:w AND card_id=:id "
"ORDER BY database_id, schema_name, table_name, column_name"
), params).mappings().all()
return Card.model_validate({
**row["data"], "id": row["id"], "workspace_id": self.workspace_id,
"family": row["family"], "subject": row["subject"], "origin": row["origin"],
"created_at": row["created_at"], "updated_at": row["updated_at"],
"revision": row["revision"], "indexed": row["indexed"],
"links": [dict(v) for v in links], "dependencies": [dict(v) for v in dependencies],
})
@staticmethod
def _selection():
return ("SELECT c.*, (p.revision=c.revision AND NOT p.pending "
"AND p.action='upsert' AND p.format=2) AS indexed FROM thoth_memory.cards c "
"JOIN thoth_memory.projections p ON p.workspace_id=c.workspace_id "
"AND p.card_id=c.id ")
def get(self, card_id: str) -> Card:
with self.transaction() as c:
row = c.execute(text(self._selection()+"WHERE c.workspace_id=:w AND c.id=:id"),
{"w": self.workspace_id, "id": card_id}).mappings().first()
if row is None:
raise MemoryNotFound("Memory card was not found in this workspace")
return self._card(c, row)
def list(self, query: CardQuery) -> dict:
conditions = ["c.workspace_id=:w"]
params = {"w": self.workspace_id}
if query.q:
conditions.append("(c.id ILIKE :q OR c.subject ILIKE :q OR c.data::text ILIKE :q)")
params["q"] = "%" + query.q.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + "%"
for key in ("family", "origin"):
if value := getattr(query, key):
conditions.append(f"c.{key}=:{key}")
params[key] = value
if query.concept:
conditions.append("c.data->'concepts' @> CAST(:concept AS jsonb)")
params["concept"] = json.dumps([query.concept])
refs = []
for key, column in [("database", "database_id"), ("table", "table_name"),
("column", "column_name")]:
if value := getattr(query, key):
refs.append(f"d.{column}=:{key}")
params[key] = value
if refs:
conditions.append("EXISTS (SELECT 1 FROM thoth_memory.dependencies d WHERE "
"d.workspace_id=c.workspace_id AND d.card_id=c.id AND "
+ " AND ".join(refs) + ")")
for key, comparison in [("updated_after", ">="), ("updated_before", "<=")]:
if value := getattr(query, key):
conditions.append(f"c.updated_at {comparison} :{key}")
params[key] = value
where = " WHERE " + " AND ".join(conditions)
with self.transaction() as c:
total = c.execute(text("SELECT count(*) FROM thoth_memory.cards c" + where),
params).scalar_one()
rows = c.execute(text(self._selection() + where
+ f" ORDER BY c.{query.sort} {query.direction}, c.id ASC LIMIT :limit OFFSET :offset"),
{**params, "limit": query.page_size, "offset": (query.page - 1) * query.page_size},
).mappings().all()
return {"items": [self._card(c, row).model_dump(mode="json") for row in rows],
"total": total, "page": query.page, "page_size": query.page_size}
def exact_match(self, value: CardInput) -> Card | None:
"""Match authored content only; provenance and index state do not create new knowledge."""
data = value.model_dump(mode="json", exclude={"dependencies", "links"})
with self.transaction() as c:
rows = c.execute(text(self._selection() +
"WHERE c.workspace_id=:w AND c.family=:family AND c.subject=:subject "
"AND c.data - 'session_id' - 'decision_seq'=CAST(:data AS jsonb) ORDER BY c.id"),
{"w": self.workspace_id, "family": value.family, "subject": value.subject,
"data": json.dumps(data)}).mappings()
for row in rows:
candidate = self._card(c, row)
def ordered(values):
return sorted(json.dumps(v.model_dump(), sort_keys=True) for v in values)
if (ordered(candidate.dependencies) == ordered(value.dependencies)
and ordered(candidate.links) == ordered(value.links)):
return candidate
return None
def source(self, source_key: str):
with self.transaction() as c:
row = c.execute(text("SELECT card_id, action FROM thoth_memory.projections "
"WHERE workspace_id=:w AND source_key=:s"),
{"w": self.workspace_id, "s": source_key}).mappings().first()
return dict(row) if row else None
def save(self, value: CardInput, *, card_id: str | None = None, source_key: str | None = None,
session_id: str | None = None, decision_seq: int | None = None,
new_id: str | None = None) -> str:
creating = card_id is None
card_id = card_id or new_id or "mem-" + str(uuid4())
revision = str(uuid4())
data = value.model_dump(mode="json", exclude={"links", "dependencies"})
with self.transaction() as c:
if not creating:
old = c.execute(text("SELECT data FROM thoth_memory.cards "
"WHERE workspace_id=:w AND id=:id"),
{"w": self.workspace_id, "id": card_id}).scalar_one_or_none()
if old is None:
raise MemoryNotFound("Memory card was not found in this workspace")
data.update({k: old.get(k) for k in ("session_id", "decision_seq")})
else:
if c.execute(text("SELECT 1 FROM thoth_memory.projections "
"WHERE workspace_id=:w AND card_id=:id"),
{"w": self.workspace_id, "id": card_id}).first():
raise MemoryConflict("A proposed card identity was already used")
data.update(session_id=session_id, decision_seq=decision_seq)
params = {"w": self.workspace_id, "id": card_id, "r": revision,
"data": json.dumps(data), "family": value.family, "subject": value.subject,
"origin": "workflow" if source_key else "manual", "source": source_key}
c.execute(text("INSERT INTO thoth_memory.cards "
"(workspace_id,id,family,subject,origin,data,revision) "
"VALUES (:w,:id,:family,:subject,:origin,CAST(:data AS jsonb),:r) "
"ON CONFLICT (workspace_id,id) DO UPDATE SET family=EXCLUDED.family, "
"subject=EXCLUDED.subject,data=EXCLUDED.data,revision=EXCLUDED.revision, "
"updated_at=clock_timestamp()"), params)
c.execute(text("DELETE FROM thoth_memory.links WHERE workspace_id=:w AND source_id=:id"), params)
for link in value.links:
c.execute(text("INSERT INTO thoth_memory.links VALUES (:w,:id,:target,:meaning)"),
{**params, "target": link.target_id, "meaning": link.meaning})
c.execute(text("DELETE FROM thoth_memory.dependencies "
"WHERE workspace_id=:w AND card_id=:id"), params)
for dep in {tuple(d.model_dump().values()) for d in value.dependencies}:
c.execute(text("INSERT INTO thoth_memory.dependencies VALUES "
"(:w,:id,:database,:schema,:table,:column)"),
{**params, **dict(zip(("database", "schema", "table", "column"), dep))})
c.execute(text("INSERT INTO thoth_memory.projections "
"(workspace_id,card_id,revision,action,source_key) VALUES (:w,:id,:r,'upsert',:source) "
"ON CONFLICT (workspace_id,card_id) DO UPDATE SET revision=EXCLUDED.revision, "
"action='upsert',pending=true,error=NULL,updated_at=clock_timestamp()"), params)
return card_id
def delete(self, card_id: str):
with self.transaction() as c:
params = {"w": self.workspace_id, "id": card_id, "r": str(uuid4())}
deleted = c.execute(text("DELETE FROM thoth_memory.cards "
"WHERE workspace_id=:w AND id=:id"), params).rowcount
if not deleted:
raise MemoryNotFound("Memory card was not found in this workspace")
c.execute(text("UPDATE thoth_memory.projections SET action='delete',revision=:r,"
"pending=true,error=NULL,updated_at=clock_timestamp() "
"WHERE workspace_id=:w AND card_id=:id"), params)
def projections(self, *, pending: bool = True):
with self.transaction() as c:
needs_update = "(pending OR (action='upsert' AND format<>2))"
rows = c.execute(text("SELECT card_id,revision,action,"+needs_update+" AS pending,"
"error,updated_at "
"FROM thoth_memory.projections WHERE workspace_id=:w "
+ ("AND "+needs_update+" " if pending else "") + "ORDER BY updated_at,card_id"),
{"w": self.workspace_id}).mappings().all()
return [dict(row) for row in rows]
def projection_result(self, card_id: str, revision: str, error: str | None):
with self.transaction() as c:
c.execute(text("UPDATE thoth_memory.projections SET pending=:p,error=:error,format=2,"
"updated_at=clock_timestamp() WHERE workspace_id=:w AND card_id=:id AND revision=:r"),
{"p": error is not None, "error": error, "w": self.workspace_id,
"id": card_id, "r": revision})
def invalidate_all(self):
with self.transaction() as c:
c.execute(text("UPDATE thoth_memory.projections SET pending=true,error=NULL "
"WHERE workspace_id=:w"), {"w": self.workspace_id})
+131
View File
@@ -0,0 +1,131 @@
"""Bounded graph recall over current Memory cards; no model or approval side effects."""
from dataclasses import dataclass
from pydantic import BaseModel, ConfigDict, Field, model_validator
from .models import Card, Family, MemoryNotFound
PROJECTION_FORMAT = 2
MAX_SEEDS = 100
MAX_DEPTH = 2
MAX_LINKS_PER_CARD = 20
MAX_VISITED = 200
MAX_EDGES = 400
RRF_K = 60
class RecallScope(BaseModel):
"""Exact business scope/concepts and hierarchical physical context.
Cards without dependencies are workspace-wide. A dependency applies to its
database and every descendant of the schema/table/column it names.
All physical fields must match the SAME dependency, never separate entries.
"""
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
scope: str = Field(default="", max_length=10000)
database: str = Field(default="", max_length=200)
schema_name: str = Field(default="", max_length=200)
table: str = Field(default="", max_length=200)
column: str = Field(default="", max_length=200)
concepts: list[str] = Field(default_factory=list, max_length=100)
@model_validator(mode="after")
def validate_context(self):
if ((self.schema_name and not self.database) or (self.table and not self.schema_name)
or (self.column and not self.table)):
raise ValueError("Physical recall scope requires its database/schema/table ancestors")
if any(not value.strip() or len(value) > 200 for value in self.concepts):
raise ValueError("Recall concepts must contain between 1 and 200 characters")
return self
def matches(self, card: Card) -> bool:
if self.scope and self.scope != card.scope:
return False
if not set(self.concepts) <= set(card.concepts):
return False
if not self.database or not card.dependencies:
return True
return any(all(not getattr(self, key) or getattr(dep, key) in ("", getattr(self, key))
for key in ("database", "schema_name", "table", "column"))
for dep in card.dependencies)
def vector_filter(self, family: Family | None) -> dict:
return {"memory": {**self.model_dump(), "family": family,
"format": PROJECTION_FORMAT}}
@dataclass(frozen=True)
class RecalledCard:
card: Card
score: float
path: tuple[str, ...]
def expand_and_rank(repo, hits, *, scope: RecallScope, family: Family | None,
excluded: set[str], top: int) -> list[RecalledCard]:
"""RRF direct rank + strongest link path, decayed by 0.5 per outgoing hop.
Roots, nodes, fan-out and depth are all bounded. Repeated paths do not add
votes: cycles and highly connected cards cannot amplify their own relevance.
The caller holds the workspace operation lock while resolving authority.
"""
cache: dict[str, Card | None] = {}
def current(identity):
if identity not in cache:
if len(cache) >= MAX_VISITED:
return None
try:
card = repo.get(identity)
except MemoryNotFound:
card = None
if card is not None and (not card.indexed or card.id in excluded
or (family and card.family != family) or not scope.matches(card)):
card = None
cache[identity] = card
return cache[identity]
direct: dict[str, float] = {}
graph: dict[str, tuple[float, tuple[str, ...]]] = {}
seeds = []
for rank, hit in enumerate(hits[:MAX_SEEDS], 1):
card = current(hit.ref)
if (card is None or hit.metadata.get("memory_revision") != card.revision
or hit.metadata.get("memory_format") != PROJECTION_FORMAT
or card.id in direct):
continue
direct[card.id] = 1 / (RRF_K + rank)
seeds.append(card)
traversed = 0
for seed in seeds:
frontier = [(seed, (seed.id,))]
visited = {seed.id}
for depth in range(1, MAX_DEPTH + 1):
next_frontier = []
for source, path in frontier:
for link in sorted(source.links, key=lambda link: link.target_id)[:MAX_LINKS_PER_CARD]:
if traversed >= MAX_EDGES:
break
traversed += 1
if link.target_id in visited:
continue
visited.add(link.target_id)
target = current(link.target_id)
if target is None:
continue
target_path = (*path, target.id)
score = direct[seed.id] * 0.5 ** depth
previous = graph.get(target.id)
if previous is None or (-score, target_path) < (-previous[0], previous[1]):
graph[target.id] = (score, target_path)
next_frontier.append((target, target_path))
frontier = next_frontier
ranked = []
for identity in direct.keys() | graph.keys():
graph_score, path = graph.get(identity, (0, (identity,)))
ranked.append(RecalledCard(cache[identity], direct.get(identity, 0) + graph_score, path))
return sorted(ranked, key=lambda result: (-result.score, result.card.id))[:top]
+313
View File
@@ -0,0 +1,313 @@
"""Reviewer-edited additions/updates, grounded in effective approved session decisions."""
import hashlib
import json
from uuid import NAMESPACE_URL, uuid5
from pydantic import BaseModel, ConfigDict, Field
from sqlalchemy import text
from tht.phase import current_phase, effective_decisions
from .models import CardInput, MemoryConflict
from .solved import _build_solved_snapshot
APPROVED_SOURCES = {
"concept_clarified",
"join_modified",
"column_corrected",
"concept_formula_approved",
"cte_corrected",
"cte_approved",
"sql_approved",
}
class Proposal(BaseModel):
model_config = ConfigDict(extra="forbid")
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
source_seqs: list[int] = Field(min_length=1, max_length=20)
card: CardInput
target_id: str | None = None
target_revision: str | None = None
reason: str = Field(min_length=1, max_length=10000)
class Selection(BaseModel):
model_config = ConfigDict(extra="forbid")
id: str
card: CardInput
class ReviewResponse(BaseModel):
model_config = ConfigDict(extra="forbid")
summary_id: str
items: list[Selection] = Field(max_length=20)
def digest(value):
return hashlib.sha256(
json.dumps(value, sort_keys=True, ensure_ascii=False).encode()
).hexdigest()
def context_hash(snapshot):
return digest(
{
"decisions": [
d.model_dump(mode="json")
for d in effective_decisions(snapshot)
if d.type != "memory_summary_reviewed"
],
"proposals": snapshot.artifacts.get("memory_proposals"),
"sql": snapshot.artifacts.get("sql_final"),
}
)
def solved_snapshot(snapshot):
linking = json.loads(snapshot.artifacts.get("schema_linking", "{}"))
tables = {
c["name"]
for c in linking.get("candidates", [])
if c.get("kind") == "table" and c.get("decision") == "promoted"
}
return _build_solved_snapshot(snapshot, tables)
def validate_proposals(snapshot, raw):
if not isinstance(raw, list) or len(raw) > 20:
raise ValueError("Memory proposals must be a list of at most 20 cards")
proposals = [Proposal.model_validate(p) for p in raw]
effective = {d.seq: d for d in effective_decisions(snapshot)}
if len({p.id for p in proposals}) != len(proposals):
raise ValueError("Memory proposal identities must be unique")
for p in proposals:
sources = [effective.get(seq) for seq in p.source_seqs]
if any(d is None or d.type not in APPROVED_SOURCES for d in sources):
raise MemoryConflict("Memory proposals require effective approved source decisions")
if p.card.family == "explained_error" and not any(d.rationale.strip() for d in sources):
raise MemoryConflict("Explained errors require an approved explanation")
if p.card.family == "solved_question":
solved = solved_snapshot(snapshot)
if p.card.sql != solved.metadata["sql"]:
raise MemoryConflict("Exemplar SQL must match the current approved solution")
if bool(p.target_id) != bool(p.target_revision):
raise ValueError("Updates require the identity and revision of the card being replaced")
return proposals
def prepare(service, snapshot):
service._session(snapshot)
if current_phase(snapshot) != 8 or snapshot.manifest.status in {"finalized", "archived"}:
raise MemoryConflict("The Memory summary is reviewed at the end of F8")
with service.repository.transaction() as connection:
receipt = (
connection.execute(
text(
"SELECT summary_id,result FROM thoth_memory.reviews "
"WHERE workspace_id=:w AND session_id=:s AND result->>'context_hash'=:h "
"ORDER BY created_at DESC LIMIT 1"
),
{
"w": service.repository.workspace_id,
"s": snapshot.manifest.id,
"h": context_hash(snapshot),
},
)
.mappings()
.first()
)
if receipt:
return {
"reviewed": True,
"summary_id": receipt["summary_id"],
"saved": len(receipt["result"]["saved"]),
}
raw = json.loads(snapshot.artifacts.get("memory_proposals", "[]"))
proposals = validate_proposals(snapshot, raw)
covered = {seq for p in proposals for seq in p.source_seqs}
for d in service.promotions(snapshot):
if d["decision_seq"] not in covered:
proposals.append(
Proposal(
id=f"decision-{d['decision_seq']}",
source_seqs=[d["decision_seq"]],
reason="Reusable domain clarification",
card=CardInput(
family="domain_clarification",
subject=d["subject"],
detail=d["detail"],
rationale=d["rationale"],
question=d["question_context"],
scope=service.repository.workspace_id,
concepts=[d["subject"]],
),
)
)
if not any(p.card.family == "solved_question" for p in proposals):
solved = solved_snapshot(snapshot)
approved = [d for d in effective_decisions(snapshot) if d.type == "sql_approved"][-1]
proposals.append(
Proposal(
id="solved-question",
source_seqs=[approved.seq],
reason="Approved solution, for consultation in future questions",
card=CardInput(
family="solved_question",
subject=solved.title,
question=solved.content,
sql=solved.metadata["sql"],
scope=service.repository.workspace_id,
dependencies=[
{
"database": snapshot.manifest.database,
"schema_name": snapshot.manifest.db_schema,
"table": table,
}
for table in solved.metadata["tables"]
],
),
)
)
if len(proposals) > 20:
raise MemoryConflict("Reduce the final Memory summary to at most 20 cards")
items = []
targets = set()
content = {}
aliases = {}
for p in proposals:
before = None
if p.target_id:
old = service.repository.get(p.target_id)
if old.revision != p.target_revision:
raise MemoryConflict(
"A Memory card changed; refresh the proposed update before review"
)
if p.target_id in targets:
raise MemoryConflict("Propose only one update to each Memory card")
targets.add(p.target_id)
before = old.model_dump(mode="json")
key = digest(p.card.model_dump(mode="json"))
duplicate = service.repository.exact_match(p.card)
if duplicate and (not p.target_id or duplicate.id == p.target_id):
aliases["proposal:" + p.id] = duplicate.id
continue
if not p.target_id and key in content:
aliases["proposal:" + p.id] = "proposal:" + content[key]
continue
content[key] = p.id
items.append({**p.model_dump(mode="json"), "before": before})
for item in items:
for link in item["card"]["links"]:
link["target_id"] = aliases.get(link["target_id"], link["target_id"])
return {
"summary_id": digest(
{
"items": items,
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)],
}
),
"items": items,
}
def apply(service, snapshot, response: ReviewResponse):
service._session(snapshot)
request_hash = digest(response.model_dump(mode="json"))
session_id = snapshot.manifest.id
with service.repository.operation() as repo:
with repo.transaction() as connection:
params = {"w": repo.workspace_id, "s": session_id, "id": response.summary_id}
receipt = (
connection.execute(
text(
"SELECT request_hash,result FROM thoth_memory.reviews "
"WHERE workspace_id=:w AND session_id=:s AND summary_id=:id"
),
params,
)
.mappings()
.first()
)
if receipt:
if receipt["request_hash"] != request_hash:
raise MemoryConflict(
"This summary was already reviewed with different selections"
)
saved = receipt["result"]["saved"]
else:
# Resolve the same locked repository for preview and optimistic update checks.
original = service.repository
service.repository = repo
try:
summary = prepare(service, snapshot)
finally:
service.repository = original
if summary.get("reviewed") or summary["summary_id"] != response.summary_id:
raise MemoryConflict("The Memory summary changed; review it again")
choices = {choice.id: choice for choice in response.items}
candidates = {p["id"]: p for p in summary["items"]}
if len(choices) != len(response.items) or choices.keys() - candidates.keys():
raise ValueError("Memory review contains duplicate or unknown choices")
selected = validate_proposals(
snapshot,
[
{
**{k: v for k, v in candidates[key].items() if k != "before"},
"card": choice.card.model_dump(mode="json"),
}
for key, choice in choices.items()
],
)
identities = {
p.id: p.target_id
or "mem-"
+ str(uuid5(NAMESPACE_URL, f"thothii:{repo.workspace_id}:{session_id}:{p.id}"))
for p in selected
}
saved = []
for p in selected:
# Create/update every selected card before inserting links among new cards.
identity = repo.save(
p.card.model_copy(update={"links": []}),
card_id=p.target_id,
new_id=identities[p.id],
source_key=f"review:{session_id}:{p.id}",
session_id=session_id,
decision_seq=p.source_seqs[0],
)
saved.append({"id": identity, "proposal_id": p.id})
for p in selected:
value = p.card.model_dump(mode="json")
for link in value["links"]:
if link["target_id"].startswith("proposal:"):
target = link["target_id"].removeprefix("proposal:")
if target not in identities:
raise ValueError("Select the linked card or remove its link")
link["target_id"] = identities[target]
repo.save(CardInput.model_validate(value), card_id=identities[p.id])
connection.execute(
text(
"INSERT INTO thoth_memory.reviews "
"(workspace_id,session_id,summary_id,request_hash,result) "
"VALUES (:w,:s,:id,:hash,CAST(:result AS jsonb))"
),
{
**params,
"hash": request_hash,
"result": json.dumps(
{
"saved": saved,
"context_hash": context_hash(snapshot),
}
),
},
)
results = [service._propagate(repo, item["id"]) for item in saved]
return {
"saved": len(saved),
"declined": len(summary["items"]) - len(saved) if not receipt else None,
"indexed": all(r["indexed"] for r in results),
"results": results,
}
+63
View File
@@ -0,0 +1,63 @@
"""Installation binding for Memory; vector dependencies are opened only when needed."""
import os
from tht.session.models import PrincipalContext
from .migrate import installation_url
from .models import MemoryForbidden, MemoryUnavailable
from .repository import MemoryRepository
from .service import MemoryService
def _principal():
issuer = os.environ.get("THT_PRINCIPAL_ISSUER", "").strip()
subject = os.environ.get("THT_PRINCIPAL_SUBJECT", "").strip()
if not issuer or not subject:
raise MemoryForbidden("A trusted runtime principal is required for Memory")
return PrincipalContext(
issuer=issuer, subject=subject,
is_admin=os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() in {"1", "true"},
)
def _repository(workspace_id):
try:
url = installation_url()
except ValueError:
raise MemoryUnavailable("Memory PostgreSQL installation configuration is unavailable") \
from None
return MemoryRepository(url, workspace_id)
def memory_service(cfg):
from tht.adapters.factory import build_vector_store
from tht.cli.vector_cmd import make_embedder
principal = _principal()
return MemoryService(_repository(cfg._workspace_id), principal,
language=cfg.language,
store_factory=lambda: build_vector_store(cfg, require_write=True),
embedder_factory=lambda: make_embedder(cfg.embeddings))
def admin_service(workspace_id, runtime):
"""Admin access needs no DWH binding, active session or Evidence materialization."""
from tht.adapters.vector.qdrant import QdrantVectorStore
from tht.config import EmbeddingsConfig
from tht.vectorstore.embeddings import OllamaEmbeddings
principal = _principal()
if not principal.is_admin:
raise MemoryForbidden("Memory administration requires an administrator")
return MemoryService(_repository(workspace_id), principal,
language=runtime.get("memoryLanguage", "en"),
store_factory=lambda: QdrantVectorStore(
base_url=runtime["internalQdrantUrl"], workspace_id=workspace_id,
collections={"reference": workspace_id+"-reference", "memory": workspace_id+"-memory"},
expected_dimension=runtime["internalEmbeddingDimensions"],
),
embedder_factory=lambda: OllamaEmbeddings(EmbeddingsConfig(
base_url=runtime["internalEmbeddingUrl"], model=runtime["internalEmbeddingModel"],
dimensions=runtime["internalEmbeddingDimensions"], timeout=30,
)))
+240
View File
@@ -0,0 +1,240 @@
"""Memory operations: SQL authority, explicit projection recovery, verified recall."""
import hashlib
from tht.phase import effective_decisions
from tht.ports.vector import VectorWriteRecord
from tht.session.models import PrincipalContext
from tht.vectorstore.records import VectorRecord
from .core import decided_memory_ids, declined_promotion_seqs, question_context
from .models import (
CardInput,
CardQuery,
Dependency,
MemoryConflict,
MemoryForbidden,
MemoryNotFound,
)
from .repository import MemoryRepository
from .retrieval import MAX_SEEDS, PROJECTION_FORMAT, RecallScope, expand_and_rank
from .solved import _build_solved_snapshot
class MemoryService:
def __init__(self, repository: MemoryRepository, principal: PrincipalContext,
*, store_factory, embedder_factory, language: str = "en"):
self.repository = repository
self.principal = principal
self.store_factory = store_factory
self.embedder_factory = embedder_factory
if language not in {"en", "it"}:
raise ValueError("Memory language must be en or it")
self.query_language = {"en": "english", "it": "italian"}[language]
def close(self):
self.repository.close()
def _admin(self):
if not self.principal.is_admin:
raise MemoryForbidden("Memory administration requires an administrator")
def list(self, query: CardQuery):
self._admin()
return self.repository.list(query)
def get(self, card_id: str):
self._admin()
return self.repository.get(card_id).model_dump(mode="json")
def pending(self):
self._admin()
return self.repository.projections()
def save(self, value: CardInput, card_id: str | None = None):
self._admin()
with self.repository.operation() as repo:
card_id = repo.save(value, card_id=card_id)
return self._propagate(repo, card_id)
def delete(self, card_id: str):
self._admin()
with self.repository.operation() as repo:
repo.delete(card_id)
return self._propagate(repo, card_id)
def retry(self, card_id: str):
self._admin()
with self.repository.operation() as repo:
return self._propagate(repo, card_id)
def _propagate(self, repo, card_id):
operation = next((p for p in repo.projections(pending=False)
if p["card_id"] == card_id), None)
if operation is None:
raise MemoryNotFound("Memory operation was not found in this workspace")
error = None
if operation["pending"]:
try:
store = self.store_factory()
if operation["action"] == "delete":
store.delete_memory_records([f"card:{card_id}"])
else:
card = repo.get(card_id)
kind = "solved_question" if card.family == "solved_question" else "memory"
content = "\n".join(filter(None, [card.subject, card.detail, card.scope,
card.rationale, card.question, card.sql,
" ".join(card.concepts),
"\n".join(".".join(filter(None, [d.database,
d.schema_name, d.table, d.column]))
for d in card.dependencies)]))
record = VectorRecord(
id=f"card:{card.id}", kind=kind, ref=card.id, title=card.subject,
content=content, metadata={"memory_revision": card.revision,
"memory_format": PROJECTION_FORMAT, "memory_family": card.family,
"memory_scope": card.scope, "memory_concepts": card.concepts,
"memory_dependencies": [d.model_dump() for d in card.dependencies]},
)
embedding = self.embedder_factory().embed_documents([content])[0]
# A family change may change the vector kind (and point identity).
store.delete_memory_records([f"card:{card_id}"])
store.upsert("memory", [VectorWriteRecord(
record=record, embedding=embedding,
content_hash=hashlib.sha256(card.revision.encode()).hexdigest(),
sparse_text=content, sparse_language=self.query_language,
)])
except Exception: # noqa: BLE001 - durable pending state covers adapter/factory failures.
error = "Memory change is saved; index update is incomplete. Retry the index update."
repo.projection_result(card_id, operation["revision"], error)
result = {"id": card_id, "saved": True, "indexed": error is None,
"action": operation["action"], "error": error}
if operation["action"] != "delete":
result["card"] = repo.get(card_id).model_dump(mode="json")
return result
def rebuild(self):
self._admin()
with self.repository.operation() as repo:
# Persist invalidation before deleting anything. A crash remains recoverable.
repo.invalidate_all()
store = self.store_factory()
store.prepare_memory_index()
store.delete_kinds("memory", ["memory", "solved_question"])
return [self._propagate(repo, p["card_id"]) for p in repo.projections()]
def retrieve(self, question: str, *, searcher, embedder, top: int = 5,
family=None, scope: RecallScope | None = None, excluded=()):
if type(top) is not int or not 1 <= top <= 100:
raise ValueError("Recall limit must be between 1 and 100")
if not question.strip():
raise ValueError("Recall question must not be empty")
scope = scope or RecallScope()
self.repository.list(CardQuery(page_size=1))
kinds = (["solved_question"] if family == "solved_question" else
["memory"] if family else ["memory", "solved_question"])
hits = searcher.search(embedder.embed_query(question),
top_n=min(MAX_SEEDS, max(20, top * 3)), kinds=kinds,
query_text=question, query_language=self.query_language,
metadata_filter=scope.vector_filter(family))
# Mutations use the same lock: links, eligibility and payload are resolved
# together against current authority, after the potentially slow vector call.
with self.repository.operation() as repo:
return expand_and_rank(repo, hits, scope=scope, family=family,
excluded=set(excluded), top=top)
def recall(self, question: str, *, searcher, embedder, top: int = 5,
solved: bool = False, decisions=(), scope: RecallScope | None = None):
family = "solved_question" if solved else "domain_clarification"
candidates = self.retrieve(question, searcher=searcher, embedder=embedder, top=top,
family=family, scope=scope, excluded=decided_memory_ids(list(decisions)))
result = []
for candidate in candidates:
card = candidate.card
common = {"id": card.id, "session_id": card.session_id,
"revision": card.revision, "family": card.family, "scope": card.scope,
"dependencies": [d.model_dump() for d in card.dependencies],
"tables": sorted({d.table for d in card.dependencies if d.table}),
"score": round(candidate.score, 6),
"retrieval": {"path": list(candidate.path), "method": "hybrid_links"}}
if solved:
result.append({**common, "question": card.question, "sql": card.sql})
else:
# New workflow category consumption belongs to M3.
result.append({**common, "type": "concept_clarified",
"subject": card.subject, "detail": card.detail, "rationale": card.rationale,
"question_context": card.question, "scope": card.scope,
"concepts": card.concepts})
return result
def _session(self, snapshot):
if (snapshot.manifest.workspace_id != self.repository.workspace_id
or (not self.principal.is_admin
and snapshot.manifest.author != self.principal.subject)):
raise MemoryForbidden("Memory source session is outside the authorized context")
def promotions(self, snapshot):
self._session(snapshot)
decisions = effective_decisions(snapshot)
declined = declined_promotion_seqs(decisions)
result = []
seen = set()
for d in decisions:
if d.type != "concept_clarified" or d.seq in declined:
continue
source_key = f"decision:{snapshot.manifest.id}:{d.seq}"
if self.repository.source(source_key) is not None:
continue
content = (d.subject, d.detail, d.rationale)
if content in seen:
continue
seen.add(content)
result.append({"decision_seq": d.seq, "type": d.type, "subject": d.subject,
"detail": d.detail, "rationale": d.rationale,
"question_context": question_context(decisions, snapshot.manifest)})
return result
def promote(self, snapshot, seqs):
self._session(snapshot)
decisions = effective_decisions(snapshot)
selected = [d for d in decisions if d.seq in seqs and d.type == "concept_clarified"]
results = []
with self.repository.operation() as repo:
for d in selected:
key = f"decision:{snapshot.manifest.id}:{d.seq}"
existing = repo.source(key)
if existing and existing["action"] == "delete":
continue
card_id = existing["card_id"] if existing else repo.save(CardInput(
family="domain_clarification", subject=d.subject, detail=d.detail,
rationale=d.rationale, scope=self.repository.workspace_id,
question=question_context(decisions, snapshot.manifest), concepts=[d.subject],
), source_key=key, session_id=snapshot.manifest.id, decision_seq=d.seq)
results.append(self._propagate(repo, card_id))
return results
def save_solved(self, snapshot, promoted_tables=None):
self._session(snapshot)
if snapshot.manifest.status != "finalized":
raise MemoryConflict("Only a finalized session can produce a solved-question card")
record = _build_solved_snapshot(snapshot, promoted_tables)
with self.repository.operation() as repo:
existing = repo.source(f"solved:{snapshot.manifest.id}")
if existing and existing["action"] == "delete":
return {"saved": True, "indexed": True, "action": "delete", "error": None}
card_id = existing["card_id"] if existing else repo.save(CardInput(
family="solved_question", subject=record.title,
scope=self.repository.workspace_id, question=record.content,
sql=record.metadata["sql"],
dependencies=[Dependency(database=snapshot.manifest.database,
schema_name=snapshot.manifest.db_schema, table=table)
for table in record.metadata["tables"]],
), source_key=f"solved:{snapshot.manifest.id}", session_id=snapshot.manifest.id)
return self._propagate(repo, card_id)
def retry_solved(self, snapshot):
self._session(snapshot)
with self.repository.operation() as repo:
existing = repo.source(f"solved:{snapshot.manifest.id}")
if existing is None:
raise MemoryNotFound("No authoritative exemplar exists for this session")
return self._propagate(repo, existing["card_id"])
@@ -0,0 +1,72 @@
CREATE SCHEMA IF NOT EXISTS thoth_memory;
DO $$ BEGIN
IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'thoth_memory_runtime') THEN
CREATE ROLE thoth_memory_runtime NOLOGIN;
END IF;
IF EXISTS (SELECT FROM pg_roles WHERE rolname = 'thothii_catalog_runtime') THEN
GRANT thoth_memory_runtime TO thothii_catalog_runtime;
END IF;
END $$;
CREATE TABLE thoth_memory.cards (
workspace_id text NOT NULL,
id text NOT NULL,
family text NOT NULL,
subject text NOT NULL,
origin text NOT NULL CHECK (origin IN ('manual', 'workflow')),
data jsonb NOT NULL,
created_at timestamptz NOT NULL DEFAULT now(),
updated_at timestamptz NOT NULL DEFAULT now(),
revision text NOT NULL,
PRIMARY KEY (workspace_id, id)
);
CREATE INDEX ON thoth_memory.cards (workspace_id, updated_at, id);
CREATE INDEX ON thoth_memory.cards (workspace_id, family);
CREATE TABLE thoth_memory.links (
workspace_id text NOT NULL,
source_id text NOT NULL,
target_id text NOT NULL,
meaning text NOT NULL,
PRIMARY KEY (workspace_id, source_id, target_id),
CHECK (source_id <> target_id),
FOREIGN KEY (workspace_id, source_id) REFERENCES thoth_memory.cards ON DELETE CASCADE,
FOREIGN KEY (workspace_id, target_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
);
CREATE TABLE thoth_memory.dependencies (
workspace_id text NOT NULL,
card_id text NOT NULL,
database_id text NOT NULL,
schema_name text NOT NULL,
table_name text NOT NULL,
column_name text NOT NULL,
PRIMARY KEY (workspace_id, card_id, database_id, schema_name, table_name, column_name),
FOREIGN KEY (workspace_id, card_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
);
-- No FK to cards: deletion recovery and source receipts survive removal of the card.
CREATE TABLE thoth_memory.projections (
workspace_id text NOT NULL,
card_id text NOT NULL,
revision text NOT NULL,
action text NOT NULL CHECK (action IN ('upsert', 'delete')),
pending boolean NOT NULL DEFAULT true,
error text,
source_key text,
updated_at timestamptz NOT NULL DEFAULT now(),
PRIMARY KEY (workspace_id, card_id),
UNIQUE (workspace_id, source_key)
);
DO $$ DECLARE t text; BEGIN
FOREACH t IN ARRAY ARRAY['cards', 'links', 'dependencies', 'projections'] LOOP
EXECUTE format('ALTER TABLE thoth_memory.%I ENABLE ROW LEVEL SECURITY', t);
EXECUTE format('ALTER TABLE thoth_memory.%I FORCE ROW LEVEL SECURITY', t);
EXECUTE format(
'CREATE POLICY workspace_isolation ON thoth_memory.%I USING '
'(workspace_id = current_setting(''thoth.memory_workspace'', true)) '
'WITH CHECK (workspace_id = current_setting(''thoth.memory_workspace'', true))', t);
EXECUTE format('GRANT SELECT, INSERT, UPDATE, DELETE ON thoth_memory.%I '
'TO thoth_memory_runtime', t);
END LOOP;
END $$;
GRANT USAGE ON SCHEMA thoth_memory TO thoth_memory_runtime;
GRANT SELECT ON thoth_memory.migrations TO thoth_memory_runtime;
REVOKE ALL ON SCHEMA thoth_memory FROM PUBLIC;
@@ -0,0 +1,3 @@
-- Existing dense projections remain recoverable, but are not hybrid-ready.
-- No cross-workspace data update or weakening of FORCE RLS is necessary.
ALTER TABLE thoth_memory.projections ADD COLUMN format integer NOT NULL DEFAULT 1;
@@ -0,0 +1,29 @@
CREATE TABLE thoth_memory.reviews (
workspace_id text NOT NULL,
session_id text NOT NULL,
summary_id text NOT NULL,
request_hash text NOT NULL,
result jsonb NOT NULL,
created_at timestamptz NOT NULL DEFAULT now(),
PRIMARY KEY (workspace_id, session_id, summary_id)
);
ALTER TABLE thoth_memory.reviews ENABLE ROW LEVEL SECURITY;
ALTER TABLE thoth_memory.reviews FORCE ROW LEVEL SECURITY;
CREATE POLICY workspace_isolation ON thoth_memory.reviews
USING (workspace_id = current_setting('thoth.memory_workspace', true))
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
GRANT SELECT, INSERT ON thoth_memory.reviews TO thoth_memory_runtime;
CREATE TABLE thoth_memory.cleanup_receipts (
workspace_id text NOT NULL,
sync_id text NOT NULL,
request_hash text NOT NULL,
card_ids jsonb NOT NULL,
PRIMARY KEY (workspace_id, sync_id)
);
ALTER TABLE thoth_memory.cleanup_receipts ENABLE ROW LEVEL SECURITY;
ALTER TABLE thoth_memory.cleanup_receipts FORCE ROW LEVEL SECURITY;
CREATE POLICY workspace_isolation ON thoth_memory.cleanup_receipts
USING (workspace_id = current_setting('thoth.memory_workspace', true))
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
GRANT SELECT, INSERT ON thoth_memory.cleanup_receipts TO thoth_memory_runtime;
@@ -0,0 +1,14 @@
CREATE TABLE thoth_memory.archive_repairs (
workspace_id text NOT NULL,
session_id text NOT NULL,
repair_id text NOT NULL,
data jsonb NOT NULL,
created_at timestamptz NOT NULL DEFAULT now(),
PRIMARY KEY (workspace_id, session_id, repair_id)
);
ALTER TABLE thoth_memory.archive_repairs ENABLE ROW LEVEL SECURITY;
ALTER TABLE thoth_memory.archive_repairs FORCE ROW LEVEL SECURITY;
CREATE POLICY workspace_isolation ON thoth_memory.archive_repairs
USING (workspace_id = current_setting('thoth.memory_workspace', true))
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
GRANT SELECT, INSERT, UPDATE ON thoth_memory.archive_repairs TO thoth_memory_runtime;
+2
View File
@@ -230,6 +230,8 @@ def advance_problems(source: Path | SessionSnapshot, phase: int) -> list[str]:
problems.append(f"CTE non ancora approvato: {nc} (Fase 6)")
if phase == 7 and not _has_decision(source, "sql_approved"):
problems.append("manca la decisione sql_approved (Fase 7)")
if phase == 8 and not _has_decision(source, "memory_summary_reviewed"):
problems.append("Fase 8: il riepilogo Memory deve essere revisionato prima della chiusura")
if phase == 8 and not any(
d.type in ("datamart_requested", "datamart_declined")
for d in effective_decisions(source)
+4
View File
@@ -88,6 +88,10 @@ class VectorStore(Protocol):
def delete_kinds(self, collection: str, kinds: list[str]) -> int: ...
def prepare_memory_index(self) -> None: ...
def delete_memory_records(self, record_keys: list[str]) -> None: ...
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ...
def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ...
@@ -30,6 +30,7 @@ _ARTIFACT_FILES = {
"cte_tests": "cte_tests.json",
"cte_plan": "cte_plan.json",
"cte_plan_doc": "cte_plan_doc.json",
"memory_proposals": "memory_proposals.json",
}
_ARTIFACT_KEYS = {filename: key for key, filename in _ARTIFACT_FILES.items()}
_SAFE_CTE_NAME = re.compile(r"[A-Za-z0-9_-]+\Z")
@@ -31,6 +31,7 @@ _ARTIFACT_KEYS = {
"evidence",
"evidence_receipts",
"sql_final",
"memory_proposals",
"validation_report",
"retrieval_pack",
"cte_tests",