feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s
Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
@@ -80,9 +80,6 @@ class QdrantVectorStore:
|
||||
if not legacy_constructor and collections["reference"] == collections["memory"]:
|
||||
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
|
||||
self._collections = dict(collections)
|
||||
self._split_collections = collections["reference"] != collections["memory"]
|
||||
self._legacy_collection = workspace_id
|
||||
self._legacy_checked = False
|
||||
self._workspace_id = workspace_id
|
||||
self._workspace_revision = None
|
||||
self._workspace_revision = workspace_revision
|
||||
@@ -177,7 +174,13 @@ class QdrantVectorStore:
|
||||
filter_must.extend(self._revision_filter(allowed_record_kinds))
|
||||
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
|
||||
filter_must.append({"key": "record_kind", "match": {"any": allowed_record_kinds}})
|
||||
if metadata_filter is not None:
|
||||
if metadata_filter is not None and "memory" in metadata_filter:
|
||||
if set(metadata_filter) != {"memory"} or not set(allowed_record_kinds) <= {
|
||||
"memory", "solved_question",
|
||||
}:
|
||||
raise VectorStoreError("Unsupported Memory metadata filter")
|
||||
filter_must.extend(self._memory_filter(metadata_filter["memory"]))
|
||||
elif metadata_filter is not None:
|
||||
allowed_filters = {
|
||||
"vector_generation", "document_ids", "workspace_id", "purpose",
|
||||
"required_kinds", "required_concepts", "required_tables", "required_columns",
|
||||
@@ -221,9 +224,9 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Invalid vector metadata filter")
|
||||
filter_must.extend({"key": payload_key, "match": {"value": item}} for item in values)
|
||||
if retrieval_mode not in {"fused", "dense", "bm25"}:
|
||||
raise VectorStoreError("Evidence retrieval mode is invalid")
|
||||
raise VectorStoreError("Vector retrieval mode is invalid")
|
||||
if retrieval_mode == "dense":
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
response = self._call(
|
||||
"POST",
|
||||
@@ -236,7 +239,7 @@ class QdrantVectorStore:
|
||||
},
|
||||
)
|
||||
elif retrieval_mode == "bm25":
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
@@ -267,8 +270,8 @@ class QdrantVectorStore:
|
||||
},
|
||||
)
|
||||
else:
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence and Memory")
|
||||
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
@@ -323,16 +326,13 @@ class QdrantVectorStore:
|
||||
|
||||
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
|
||||
validate_collection(collection)
|
||||
# The first write with the split configuration is also the upgrade cutover. This keeps
|
||||
# existing runtime memory reachable even when the operator reruns preprocessing without
|
||||
# invoking the explicit clear operation first.
|
||||
if self._split_collections:
|
||||
self._migrate_legacy_memory()
|
||||
# Memory is projected only from its authoritative archive. Never import legacy payloads.
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
self._ensure_collection(
|
||||
physical_collection,
|
||||
strict=True,
|
||||
require_bm25=any(record.sparse_text is not None for record in records),
|
||||
maintain_bm25=collection == "memory",
|
||||
)
|
||||
points = []
|
||||
for write_record in records:
|
||||
@@ -341,8 +341,8 @@ class QdrantVectorStore:
|
||||
semantic_kind = qdrant_semantic_kind(write_record.record.kind)
|
||||
vector: list[float] | dict = write_record.embedding
|
||||
if write_record.sparse_text is not None:
|
||||
if semantic_kind != "evidence" or write_record.sparse_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 document is invalid")
|
||||
if semantic_kind not in {"evidence", "memory"} or write_record.sparse_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("BM25 document is invalid")
|
||||
vector = {
|
||||
"": write_record.embedding,
|
||||
"bm25": self._bm25_document(write_record.sparse_text, write_record.sparse_language),
|
||||
@@ -391,6 +391,26 @@ class QdrantVectorStore:
|
||||
)
|
||||
return before
|
||||
|
||||
def prepare_memory_index(self) -> None:
|
||||
"""Explicit rebuild may recreate a lost collection; reads never do so."""
|
||||
self._ensure_collection(self._collections["memory"], strict=True,
|
||||
require_bm25=True, maintain_bm25=True, allow_create=True)
|
||||
|
||||
def delete_memory_records(self, record_keys: list[str]) -> None:
|
||||
"""Delete exact authoritative Memory projections, never reference vectors."""
|
||||
if not record_keys or any(not key.startswith("card:mem-") for key in record_keys):
|
||||
raise VectorStoreError("Exact Memory card keys are required")
|
||||
collection = self._collections["memory"]
|
||||
if self._call("GET", f"/collections/{collection}", None, allow_missing=True) is None:
|
||||
return
|
||||
self._call("POST", f"/collections/{collection}/points/delete?wait=true", {
|
||||
"filter": {"must": [
|
||||
*self._workspace_filter(),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
{"key": "record_key", "match": {"any": record_keys}},
|
||||
]},
|
||||
})
|
||||
|
||||
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int:
|
||||
if collection != "evidence" or _GENERATION.fullmatch(generation) is None:
|
||||
raise VectorStoreError("Only exact Evidence generations may be deleted")
|
||||
@@ -442,60 +462,14 @@ class QdrantVectorStore:
|
||||
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
|
||||
|
||||
def clear_reference(self) -> bool:
|
||||
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
|
||||
legacy_deleted = self._migrate_legacy_memory()
|
||||
"""Drop only replaceable schema/Evidence vectors; do not import legacy Memory."""
|
||||
collection = self._collections["reference"]
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
return legacy_deleted
|
||||
return False
|
||||
self._call("DELETE", f"/collections/{collection}", None)
|
||||
return True
|
||||
|
||||
def _migrate_legacy_memory(self) -> bool:
|
||||
"""Preserve memory from the pre-split collection before retiring it."""
|
||||
legacy = self._legacy_collection
|
||||
if (
|
||||
self._legacy_checked
|
||||
or not self._split_collections
|
||||
or legacy in self._collections.values()
|
||||
):
|
||||
return False
|
||||
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
|
||||
if response is None:
|
||||
self._legacy_checked = True
|
||||
return False
|
||||
memory = self._collections["memory"]
|
||||
self._ensure_collection(memory, strict=True, allow_create=True)
|
||||
points = self._scroll(
|
||||
legacy,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(["memory", "solved_question"]),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
],
|
||||
with_vector=True,
|
||||
)
|
||||
migrated = []
|
||||
for point in points:
|
||||
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
if not isinstance(point.get("payload"), dict):
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
migrated.append({
|
||||
"id": point["id"],
|
||||
"vector": point["vector"],
|
||||
"payload": point["payload"],
|
||||
})
|
||||
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{memory}/points?wait=true",
|
||||
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
self._call("DELETE", f"/collections/{legacy}", None)
|
||||
self._legacy_checked = True
|
||||
return True
|
||||
|
||||
def _physical_collection_for_logical(self, collection: str) -> str:
|
||||
validate_collection(collection)
|
||||
return self._collections["memory" if collection == "memory" else "reference"]
|
||||
@@ -551,6 +525,34 @@ class QdrantVectorStore:
|
||||
else "Embedding dimension does not match configured dimension"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _memory_filter(value: object) -> list[dict]:
|
||||
fields = {"scope", "database", "schema_name", "table", "column"}
|
||||
if not isinstance(value, dict) or set(value) != fields | {"family", "concepts", "format"}:
|
||||
raise VectorStoreError("Invalid Memory metadata filter")
|
||||
if (any(not isinstance(value[key], str) for key in fields)
|
||||
or value["family"] is not None and not isinstance(value["family"], str)
|
||||
or value["family"] not in {None, "domain_clarification", "sql_rule",
|
||||
"solved_question", "explained_error"}
|
||||
or type(value["format"]) is not int or value["format"] != 2
|
||||
or not isinstance(value["concepts"], list)
|
||||
or not all(isinstance(c, str) and c for c in value["concepts"])):
|
||||
raise VectorStoreError("Invalid Memory metadata filter")
|
||||
must = [{"key": "memory_format", "match": {"value": value["format"]}}]
|
||||
for key in ("family", "scope"):
|
||||
if value[key]:
|
||||
must.append({"key": f"memory_{key}", "match": {"value": value[key]}})
|
||||
must.extend({"key": "memory_concepts", "match": {"value": c}} for c in value["concepts"])
|
||||
if value["database"]:
|
||||
dependency = [{"key": "database", "match": {"value": value["database"]}}]
|
||||
dependency.extend({"key": key, "match": {"any": ["", value[key]]}}
|
||||
for key in ("schema_name", "table", "column") if value[key])
|
||||
must.append({"should": [
|
||||
{"is_empty": {"key": "memory_dependencies"}},
|
||||
{"nested": {"key": "memory_dependencies", "filter": {"must": dependency}}},
|
||||
]})
|
||||
return must
|
||||
|
||||
@staticmethod
|
||||
def _bm25_compatible(info: dict) -> bool:
|
||||
sparse_vectors = info.get("config", {}).get("params", {}).get("sparse_vectors")
|
||||
@@ -561,7 +563,7 @@ class QdrantVectorStore:
|
||||
|
||||
def _ensure_collection(
|
||||
self, collection: str, *, strict: bool, require_bm25: bool = False,
|
||||
allow_create: bool = False,
|
||||
allow_create: bool = False, maintain_bm25: bool = False,
|
||||
) -> dict | None:
|
||||
created = False
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
@@ -573,7 +575,9 @@ class QdrantVectorStore:
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{collection}",
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"},
|
||||
**({"sparse_vectors": {"bm25": {"modifier": "idf"}}}
|
||||
if require_bm25 and maintain_bm25 else {})},
|
||||
)
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
self._call(
|
||||
@@ -614,7 +618,14 @@ class QdrantVectorStore:
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
if require_bm25 and not self._bm25_compatible(result):
|
||||
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
|
||||
sparse = result.get("config", {}).get("params", {}).get("sparse_vectors")
|
||||
if maintain_bm25 and (sparse is None or isinstance(sparse, dict) and "bm25" not in sparse):
|
||||
# Explicit Memory writes may add the missing sparse vector without
|
||||
# touching dense points or the separately managed Reference collection.
|
||||
self._call("PUT", f"/collections/{collection}/vectors/bm25",
|
||||
{"sparse": {"modifier": "idf"}})
|
||||
else:
|
||||
raise VectorStoreError("BM25 collection configuration mismatch")
|
||||
return result
|
||||
|
||||
def _scroll(
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
"""Closed session repair choices, durable receipts and authoritative archive activation.
|
||||
|
||||
Memory commits its receipt with the card. Evidence records the choice before writing
|
||||
files and recovers by comparing the approved result, never by repeating a stale write.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from tht.evidence.canonical import CuratedEvidence
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive, _content, _digest
|
||||
from tht.memory.models import CardInput, MemoryConflict, MemoryNotFound
|
||||
from tht.memory.review import digest
|
||||
from tht.phase import current_phase, effective_decisions
|
||||
|
||||
|
||||
class RepairOption(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
|
||||
label: str = Field(min_length=1, max_length=1000)
|
||||
archive: Literal["memory", "evidence"]
|
||||
target_id: str = Field(min_length=1, max_length=100)
|
||||
revision: str = Field(min_length=1, max_length=100)
|
||||
content: dict
|
||||
|
||||
|
||||
class RepairProposal(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
reason: str = Field(min_length=1, max_length=10000)
|
||||
options: list[RepairOption] = Field(min_length=1, max_length=5)
|
||||
|
||||
|
||||
def _context(snapshot):
|
||||
return digest({"phase": current_phase(snapshot), "question": snapshot.artifacts.get("question"),
|
||||
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)]})
|
||||
|
||||
|
||||
def _authorize(service, snapshot):
|
||||
service._session(snapshot)
|
||||
if snapshot.manifest.status in {"finalized", "archived"}:
|
||||
raise MemoryConflict("Archive repair requires an open session")
|
||||
|
||||
|
||||
def _archive(cfg, workspace):
|
||||
root = cfg.evidence.local_archive_root if cfg.evidence else None
|
||||
if not root or Path(root).name != workspace:
|
||||
raise MemoryConflict("Local Evidence is unavailable in this workspace")
|
||||
return LocalEvidenceArchive(root)
|
||||
|
||||
|
||||
def _params(repo, snapshot, repair_id):
|
||||
return {"w": repo.workspace_id, "s": snapshot.manifest.id, "r": repair_id}
|
||||
|
||||
|
||||
def target(service, snapshot, cfg, archive, identity):
|
||||
"""Read complete current content for a proposal, bound to the source session."""
|
||||
_authorize(service, snapshot)
|
||||
if archive == "memory":
|
||||
value = service.repository.get(identity)
|
||||
return {"revision": value.revision, "content": CardInput.model_validate(
|
||||
value.model_dump(include=set(CardInput.model_fields))).model_dump(mode="json")}
|
||||
if archive != "evidence":
|
||||
raise ValueError("Unknown archive")
|
||||
try:
|
||||
value = _archive(cfg, service.repository.workspace_id).get(identity)
|
||||
except KeyError:
|
||||
raise MemoryNotFound("Evidence was not found in this workspace") from None
|
||||
return {"revision": value["revision"], "content": value["unit"].model_dump(mode="json")}
|
||||
|
||||
|
||||
def _read(repo, snapshot, repair_id):
|
||||
with repo.transaction() as connection:
|
||||
value = connection.execute(text("SELECT data FROM thoth_memory.archive_repairs "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND repair_id=:r"),
|
||||
_params(repo, snapshot, repair_id)).scalar_one_or_none()
|
||||
if value is None:
|
||||
raise MemoryNotFound("Repair was not found in this session and workspace")
|
||||
return value
|
||||
|
||||
|
||||
def _write(repo, snapshot, repair_id, value):
|
||||
with repo.transaction() as connection:
|
||||
connection.execute(text("INSERT INTO thoth_memory.archive_repairs "
|
||||
"(workspace_id,session_id,repair_id,data) VALUES (:w,:s,:r,CAST(:data AS jsonb)) "
|
||||
"ON CONFLICT (workspace_id,session_id,repair_id) DO UPDATE SET data=EXCLUDED.data"),
|
||||
{**_params(repo, snapshot, repair_id), "data": json.dumps(value)})
|
||||
|
||||
|
||||
def prepare(service, snapshot, cfg, proposal: RepairProposal):
|
||||
_authorize(service, snapshot)
|
||||
if (len({o.id for o in proposal.options}) != len(proposal.options)
|
||||
or any(o.id in {"reject", "continue"} for o in proposal.options)):
|
||||
raise ValueError("Repair choices must have distinct identities")
|
||||
items = []
|
||||
for option in proposal.options:
|
||||
item = option.model_dump(mode="json")
|
||||
if option.archive == "memory":
|
||||
before = service.repository.get(option.target_id)
|
||||
if before.revision != option.revision:
|
||||
raise MemoryConflict("Memory changed; prepare new choices")
|
||||
item["before"] = before.model_dump(mode="json")
|
||||
item["content"] = CardInput.model_validate(option.content).model_dump(mode="json")
|
||||
else:
|
||||
archive = _archive(cfg, service.repository.workspace_id)
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if not state["active"] or state["pending"] or state.get("import_writes"):
|
||||
raise MemoryConflict("Consolidate Evidence before proposing a session repair")
|
||||
files = archive._files(archive.evidence)
|
||||
if files != archive._files(archive._snapshot(state["active"])):
|
||||
raise MemoryConflict("Consolidate external Evidence edits before session repair")
|
||||
existing = archive._units(files).get(option.target_id)
|
||||
if existing is None:
|
||||
raise MemoryNotFound("Evidence was not found in this workspace")
|
||||
relative, before = existing
|
||||
if _content(before) != option.revision:
|
||||
raise MemoryConflict("Evidence changed; prepare new choices")
|
||||
value = CuratedEvidence.model_validate(option.content)
|
||||
if (value.schema_version != 4 or value.id != before.id
|
||||
or value.kind != before.kind or value.review_items):
|
||||
raise ValueError("Repair must preserve Evidence identity/kind and resolve review items")
|
||||
# Curators change knowledge, not the source history supplied by the archive.
|
||||
value = value.model_copy(update={"provenance": before.provenance})
|
||||
item.update(content=value.model_dump(mode="json"),
|
||||
before=before.model_dump(mode="json"), file=relative,
|
||||
other_files={p: _digest(v) for p, v in files.items() if p != relative})
|
||||
items.append(item)
|
||||
value = {"reason": proposal.reason, "options": items, "context": _context(snapshot),
|
||||
"choice": None, "status": "proposed", "saved": False, "indexed": False}
|
||||
repair_id = digest({"workspace": service.repository.workspace_id,
|
||||
"session": snapshot.manifest.id, **value})
|
||||
with service.repository.operation() as repo:
|
||||
try:
|
||||
_read(repo, snapshot, repair_id)
|
||||
except MemoryNotFound:
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
return show(service, snapshot, cfg, repair_id)
|
||||
|
||||
|
||||
def show(service, snapshot, cfg, repair_id):
|
||||
_authorize(service, snapshot)
|
||||
value = _read(service.repository, snapshot, repair_id)
|
||||
# A completed receipt describes history; current eligibility is checked afresh.
|
||||
if value.get("saved"):
|
||||
option = next(o for o in value["options"] if o["id"] == value["choice"])
|
||||
try:
|
||||
if option["archive"] == "memory":
|
||||
current = service.repository.get(option["target_id"])
|
||||
same = current.revision == value["saved_revision"]
|
||||
indexed = same and current.indexed
|
||||
else:
|
||||
archive = _archive(cfg, service.repository.workspace_id)
|
||||
with archive.operation():
|
||||
units = archive._units(archive._files(archive.evidence))
|
||||
current = units[option["target_id"]][1]
|
||||
same = _content(current) == _content(
|
||||
CuratedEvidence.model_validate(option["content"]))
|
||||
state = archive._state()
|
||||
active = archive._units(archive._files(archive._snapshot(state["active"]))) \
|
||||
if state["active"] else {}
|
||||
indexed = same and option["target_id"] in active and \
|
||||
active[option["target_id"]][1] == current
|
||||
value.update(indexed=bool(indexed), status="superseded" if not same else
|
||||
"active" if indexed else "pending_activation")
|
||||
except (MemoryNotFound, KeyError, ValueError):
|
||||
value.update(indexed=False, status="superseded")
|
||||
return {**value, "repair_id": repair_id, "can_apply": service.principal.is_admin}
|
||||
|
||||
|
||||
def list_repairs(service, snapshot):
|
||||
_authorize(service, snapshot)
|
||||
with service.repository.transaction() as connection:
|
||||
rows = connection.execute(text("SELECT repair_id,data->>'status' AS recorded_status, "
|
||||
"data->>'reason' AS reason FROM thoth_memory.archive_repairs "
|
||||
"WHERE workspace_id=:w AND session_id=:s ORDER BY created_at"),
|
||||
{"w": service.repository.workspace_id, "s": snapshot.manifest.id}).mappings().all()
|
||||
return {"repairs": [dict(row) for row in rows]}
|
||||
|
||||
|
||||
def apply(service, snapshot, cfg, repair_id, choice, *, activate=None):
|
||||
_authorize(service, snapshot)
|
||||
if choice != "reject":
|
||||
service._admin()
|
||||
with service.repository.operation() as repo:
|
||||
value = _read(repo, snapshot, repair_id)
|
||||
if value["choice"] is not None and value["choice"] != choice:
|
||||
raise MemoryConflict("This repair already has a different recorded choice")
|
||||
if value["choice"] is None and value["context"] != _context(snapshot):
|
||||
raise MemoryConflict("Session decisions changed; reformulate the repair")
|
||||
if choice == "reject":
|
||||
value.update(choice=choice, status="rejected", actor=service.principal.subject)
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
else:
|
||||
option = next((o for o in value["options"] if o["id"] == choice), None)
|
||||
if option is None:
|
||||
raise ValueError("Select one of the reviewed repair choices")
|
||||
if option["archive"] == "memory":
|
||||
with repo.transaction():
|
||||
if not value["saved"]:
|
||||
current = repo.get(option["target_id"])
|
||||
if current.revision != option["revision"]:
|
||||
raise MemoryConflict("Memory changed; reformulate the repair")
|
||||
repo.save(CardInput.model_validate(option["content"]),
|
||||
card_id=option["target_id"])
|
||||
value.update(choice=choice, saved=True, status="pending_activation",
|
||||
actor=service.principal.subject,
|
||||
saved_revision=repo.get(option["target_id"]).revision)
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
elif repo.get(option["target_id"]).revision != value["saved_revision"]:
|
||||
raise MemoryConflict("The repaired Memory was changed again; do not replay it")
|
||||
result = service._propagate(repo, option["target_id"])
|
||||
value.update(indexed=result["indexed"], status="active" if result["indexed"] else
|
||||
"pending_activation")
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
else:
|
||||
_apply_evidence(service, repo, snapshot, cfg, repair_id, choice, value, option,
|
||||
activate)
|
||||
return show(service, snapshot, cfg, repair_id)
|
||||
|
||||
|
||||
def _apply_evidence(service, repo, snapshot, cfg, repair_id, choice, receipt, option, activate):
|
||||
archive = _archive(cfg, repo.workspace_id)
|
||||
proposed = CuratedEvidence.model_validate(option["content"])
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if state.get("import_writes") or (receipt["choice"] is None and state["pending"]):
|
||||
raise MemoryConflict("Finish the pending Evidence consolidation before this repair")
|
||||
files = archive._files(archive.evidence)
|
||||
if {p: _digest(v) for p, v in files.items() if p != option["file"]} != option["other_files"]:
|
||||
raise MemoryConflict("Other Evidence files changed; reconcile them before retrying")
|
||||
existing = archive._units(files).get(option["target_id"])
|
||||
if existing is None:
|
||||
raise MemoryConflict("Evidence was removed after review")
|
||||
_, current = existing
|
||||
already_written = _content(current) == _content(proposed) and receipt["choice"] == choice
|
||||
if not already_written and (receipt["saved"] or _content(current) != option["revision"]
|
||||
or current.provenance != proposed.provenance):
|
||||
raise MemoryConflict("Evidence changed; the approved correction cannot overwrite it")
|
||||
# Commit approval before touching the filesystem; a restart can recover only this choice.
|
||||
receipt.update(choice=choice, actor=service.principal.subject, status="applying")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
try:
|
||||
if not already_written:
|
||||
archive._save(proposed, expected_revision=option["revision"],
|
||||
actor=service.principal.subject)
|
||||
receipt.update(saved=True, status="pending_activation")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
archive._consolidate(service.principal.subject, activate)
|
||||
except Exception: # noqa: BLE001 - durable approval covers file/index interruption.
|
||||
receipt.update(status="pending_activation" if receipt["saved"] else "applying",
|
||||
indexed=False)
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
return
|
||||
receipt.update(indexed=activate is not None,
|
||||
status="active" if activate else "pending_activation")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
@@ -23,6 +23,46 @@ from tht.evidence import (
|
||||
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
|
||||
|
||||
|
||||
@evidence_app.command("sources", hidden=True)
|
||||
def sources_cmd(action: str, config: Path = CONFIG_OPT,
|
||||
source_id: str | None = typer.Option(None), revision: str | None = typer.Option(None),
|
||||
decision: str | None = typer.Option(None), actor: str = typer.Option("installation operator"),
|
||||
json_output: bool = typer.Option(False, "--json")):
|
||||
from tht.evidence.administration import ConsolidationError, source_action
|
||||
try:
|
||||
if action not in {"refresh", "decide"} or len(actor) > 256 or not actor.strip():
|
||||
raise ValueError("Invalid source action")
|
||||
if action == "refresh" and any(v is not None for v in (source_id, revision, decision)):
|
||||
raise ValueError("Refresh does not accept decision options")
|
||||
if action == "decide" and (decision not in {"keep", "replace"} or not source_id or not revision):
|
||||
raise ValueError("A source decision requires source-id, revision and keep or replace")
|
||||
payload = source_action(config, action=action, source_id=source_id, revision=revision,
|
||||
decision=decision, actor=actor)
|
||||
except Exception as error: # noqa: BLE001 - never expose connector/provider exception details
|
||||
safe = isinstance(error, (ValueError, EvidencePreparationError, ConsolidationError))
|
||||
_emit({"status": "failed", "code": "evidence_source_failed",
|
||||
"error": str(error)[:1500] if safe else "Source acquisition or refinement failed; existing Evidence is preserved.",
|
||||
"saved": isinstance(error, ConsolidationError) and error.saved}, json_output)
|
||||
raise typer.Exit(1) from None
|
||||
_emit(payload, json_output)
|
||||
|
||||
|
||||
@evidence_app.command("admin", hidden=True)
|
||||
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
|
||||
from tht.evidence.administration import browse
|
||||
try:
|
||||
if os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() not in {"true", "1"}:
|
||||
raise ValueError("Evidence administration requires an administrator")
|
||||
payload = json.loads(config.read_text())
|
||||
root = Path(payload["root"])
|
||||
if not root.is_absolute() or root.name != workspace:
|
||||
raise ValueError("Invalid workspace archive identity")
|
||||
_emit(browse(root, payload.get("query", {})), True)
|
||||
except (ValueError, OSError) as error:
|
||||
_emit({"code": "evidence_unavailable", "message": str(error)[:1500]}, True)
|
||||
raise typer.Exit(1) from None
|
||||
|
||||
|
||||
def _canonical_worktree(workspace_root: Path) -> Path:
|
||||
requested = workspace_root.absolute()
|
||||
root = workspace_root.resolve()
|
||||
@@ -123,7 +163,8 @@ def prepare_cmd(
|
||||
) -> None:
|
||||
"""Prepare changed Source Evidence without committing or publishing it."""
|
||||
root = _canonical_worktree(workspace_root)
|
||||
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
from tht.evidence.authoring import authoring_skill_path
|
||||
skill_path = authoring_skill_path()
|
||||
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||
try:
|
||||
try:
|
||||
@@ -180,7 +221,7 @@ def migrate_cmd(
|
||||
workspace_root: Path,
|
||||
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
|
||||
) -> None:
|
||||
"""Rewrite legacy Curated units as table-free v3 Markdown without model calls."""
|
||||
"""Convert legacy units to editable v4 Markdown and establish a local baseline."""
|
||||
root = _canonical_worktree(workspace_root)
|
||||
try:
|
||||
report = migrate_workspace_evidence(root)
|
||||
|
||||
+307
-454
@@ -1,502 +1,355 @@
|
||||
# TODO (drop registry, decisione spec 5): questo modulo e' portato col modello
|
||||
# registry intatto (load_registry/promote/update_record/delete_record). Le memory
|
||||
# dovrebbero vivere SOLO nel vectordb (metadata arricchito con subject/detail/rationale
|
||||
# in Onda 3.1). Riscrivere: promote -> upsert batch vectordb; list/show -> scan
|
||||
# vectordb; delete -> metadata.status="superseded"; index/clear -> droppati.
|
||||
# Task separato: la validazione richiede L2 (vectordb reale).
|
||||
"""Thin command adapters for the authoritative Memory service."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
from sqlalchemy.exc import OperationalError, ProgrammingError
|
||||
from pydantic import ValidationError
|
||||
|
||||
from tht.cli._guards import (
|
||||
require_server_profile,
|
||||
require_vector_write_allowed,
|
||||
)
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.session_cmd import load_snapshot_or_exit
|
||||
from tht.cli.vector_cmd import require_vector_cfg
|
||||
from tht.memory.models import CardInput, CardQuery, MemoryError
|
||||
from tht.memory.runtime import admin_service, memory_service
|
||||
from tht.ports.vector import VectorStoreError
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
|
||||
memory_app = typer.Typer(help="Review memory (registro canonico + indice semantico)")
|
||||
DECISION_OPT = typer.Option(None, "--decision", help="Seq da promuovere (ripetibile).")
|
||||
memory_app = typer.Typer(help="Authoritative Memory cards and verified recall")
|
||||
DECISION_OPT = typer.Option(None, "--decision")
|
||||
|
||||
|
||||
def registry_path(cfg) -> Path:
|
||||
if getattr(cfg.paths, "memory", None) is not None:
|
||||
return cfg.paths.memory / "registry.jsonl"
|
||||
# Legacy location remains readable while old workspaces are retired.
|
||||
return cfg.paths.artifacts / "memory" / "registry.jsonl"
|
||||
def _output(value):
|
||||
typer.echo(json.dumps(value, ensure_ascii=False, default=str))
|
||||
|
||||
|
||||
def _resync_memory(cfg):
|
||||
"""Risincronizza l'indice semantico col registro corrente (incrementale)."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder, sync_canonical_records
|
||||
from tht.memory import load_registry, memory_vector_records
|
||||
|
||||
records = memory_vector_records(load_registry(registry_path(cfg)))
|
||||
return sync_canonical_records(
|
||||
"memory",
|
||||
records,
|
||||
store=build_vector_store(cfg, require_write=True),
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
|
||||
|
||||
@memory_app.command("promote")
|
||||
def promote_cmd(
|
||||
session: str = typer.Option(..., "--session"),
|
||||
decision: list[int] = DECISION_OPT,
|
||||
preview: bool = typer.Option(False, "--preview", help="Mostra i candidati in JSON, non scrive."),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Promuove le decisioni SCELTE nel registro globale. Usa --preview per vedere i candidati."""
|
||||
import json as _json
|
||||
|
||||
from tht.memory import promote_snapshot
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
|
||||
if preview:
|
||||
from tht.memory import (
|
||||
MAX_PROMOTION_CANDIDATES,
|
||||
preview_promotions_snapshot,
|
||||
reusable_promotions_snapshot,
|
||||
)
|
||||
cand = preview_promotions_snapshot(snapshot, registry_path(cfg))
|
||||
extra = len(reusable_promotions_snapshot(snapshot, registry_path(cfg))) - len(cand)
|
||||
payload = [
|
||||
{"decision_seq": c.decision_seq, "type": c.type, "subject": c.subject,
|
||||
"detail": c.detail, "rationale": c.rationale,
|
||||
"question_context": c.question_context,
|
||||
"tables": c.tables, "concepts": c.concepts}
|
||||
for c in cand
|
||||
]
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(payload, ensure_ascii=False, indent=2))
|
||||
elif not payload:
|
||||
typer.secho("Nessun candidato da promuovere.", fg=typer.colors.YELLOW)
|
||||
else:
|
||||
for c in payload:
|
||||
typer.echo(f" [{c['decision_seq']}] {c['type']}: {c['subject']}")
|
||||
if extra > 0:
|
||||
typer.secho(
|
||||
f"NOTA: mostrati {len(cand)} candidati su {len(cand) + extra} riusabili "
|
||||
f"(cap {MAX_PROMOTION_CANDIDATES}); gli altri non sono proposti.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
return
|
||||
|
||||
if not decision:
|
||||
typer.secho("ERRORE: indica le decisioni con --decision <seq> (vedi `--preview`).",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
require_server_profile(cfg, "memory promote")
|
||||
require_vector_cfg(cfg)
|
||||
promoted = promote_snapshot(snapshot, seqs=list(decision), registry_path=registry_path(cfg))
|
||||
if not promoted:
|
||||
msg = "Nessuna nuova promozione (gia' presenti o seq inesistenti)."
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"promoted": [], "indexed": False, "message": msg}, ensure_ascii=False))
|
||||
else:
|
||||
typer.secho(msg, fg=typer.colors.YELLOW)
|
||||
return
|
||||
|
||||
# Promozione nel registro: riuscita. L'indicizzazione semantica puo' fallire
|
||||
# (runtime non pronto o vectordb irraggiungibile da questa postazione): in quel
|
||||
# caso le memorie restano nel registro ma NON sono trovate da `tht memory
|
||||
# search` finche' non si reindicizza sul server. `indexed` rende lo stato
|
||||
# leggibile da Pi, cosi' il reviewer lo vede invece di perderlo nello stderr.
|
||||
ids = [{"id": r.id, "type": r.type, "subject": r.subject} for r in promoted]
|
||||
indexed = True
|
||||
warning = None
|
||||
@contextmanager
|
||||
def _service(config):
|
||||
service = None
|
||||
try:
|
||||
_resync_memory(cfg)
|
||||
except (ProgrammingError, OperationalError):
|
||||
indexed = False
|
||||
warning = (
|
||||
f"{len(promoted)} memorie promosse nel registro, ma l'indice vettoriale "
|
||||
"NON e' stato sincronizzato (runtime vettoriale mancante o irraggiungibile): "
|
||||
"NON saranno trovate da `tht memory search` finche' non reindicizzi sul "
|
||||
"server (`tht memory index` quando il runtime vettoriale è disponibile)."
|
||||
)
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"promoted": ids, "indexed": indexed,
|
||||
"message": warning or f"{len(promoted)} memorie promosse e indicizzate."},
|
||||
ensure_ascii=False, indent=2))
|
||||
return
|
||||
for r in promoted:
|
||||
typer.echo(f" {r.id}: {r.type} {r.subject}")
|
||||
if indexed:
|
||||
typer.secho(f"OK: {len(promoted)} memorie promosse e indicizzate.",
|
||||
fg=typer.colors.GREEN)
|
||||
else:
|
||||
typer.secho(f"ATTENZIONE: {warning}", fg=typer.colors.YELLOW)
|
||||
cfg = _load_config_or_exit(config)
|
||||
service = memory_service(cfg)
|
||||
yield cfg, service
|
||||
except MemoryError as error:
|
||||
_output({"code": error.code, "message": str(error), "status": error.status})
|
||||
raise typer.Exit(1) from None
|
||||
except (ValidationError, ValueError):
|
||||
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
|
||||
raise typer.Exit(1) from None
|
||||
except (VectorStoreError, EmbeddingsError, OSError):
|
||||
_output({"code": "memory_unavailable", "message": "Memory operation is unavailable",
|
||||
"status": 503})
|
||||
raise typer.Exit(1) from None
|
||||
finally:
|
||||
if service:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("save-one")
|
||||
def save_one_cmd(
|
||||
session: str = typer.Option(..., "--session"),
|
||||
decision: int = typer.Option(
|
||||
..., "--decision", help="decision_seq della decisione da salvare come memoria."
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Upsert mirato (una riga) della memoria di una decisione nel semantic store (D11).
|
||||
|
||||
Promuove la decisione nel registro locale (idempotente) e fa un singolo upsert
|
||||
con dedup hash client-side -- niente full-resync. Il factory seleziona il writer
|
||||
del runtime vettoriale attivo.
|
||||
"""
|
||||
import json as _json
|
||||
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import load_registry, promote_snapshot, save_one_memory
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
require_vector_write_allowed(cfg, "memory save-one")
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
|
||||
# Promuove la decisione scelta nel registro locale (idempotente: salta se gia' presente
|
||||
# o se stale post-rollback, perche' _compute_promotions usa la vista effective).
|
||||
promote_snapshot(snapshot, seqs=[decision], registry_path=registry_path(cfg))
|
||||
records = [r for r in load_registry(registry_path(cfg)) if r.session_id == snapshot.manifest.id]
|
||||
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
count = save_one_memory(records, decision, store=store, embedder=embedder)
|
||||
|
||||
msg = (
|
||||
f"{count} memoria salvata nell'indice semantico (decision_seq {decision})."
|
||||
if count
|
||||
else f"Nessun upsert (decisione {decision} assente/stale o memoria gia' aggiornata)."
|
||||
)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"upserted": count, "decision_seq": decision, "message": msg}, ensure_ascii=False))
|
||||
return
|
||||
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN if count else typer.colors.YELLOW)
|
||||
|
||||
|
||||
@memory_app.command("index")
|
||||
def index_cmd(config: Path = CONFIG_OPT) -> None:
|
||||
"""Sincronizza il registro memory nell'indice semantico (full-resync)."""
|
||||
from tht.cli.vector_cmd import _print_stats
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "memory index")
|
||||
require_vector_cfg(cfg)
|
||||
_print_stats(_resync_memory(cfg))
|
||||
@memory_app.command("admin")
|
||||
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
|
||||
"""Backend-owned request snapshot. No DWH or session configuration is required."""
|
||||
service = None
|
||||
try:
|
||||
if not re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace):
|
||||
raise ValueError("Invalid workspace")
|
||||
payload = json.loads(config.read_text())
|
||||
service = admin_service(workspace, payload["runtime"])
|
||||
request = payload.get("request", {})
|
||||
action = payload["action"]
|
||||
if action == "cleanup":
|
||||
from tht.memory.cleanup import CleanupRequest, cleanup
|
||||
result = cleanup(service, CleanupRequest.model_validate(request))
|
||||
elif action == "list":
|
||||
result = service.list(CardQuery.model_validate(request))
|
||||
elif action == "show":
|
||||
result = service.get(request["id"])
|
||||
elif action in {"create", "update"}:
|
||||
result = service.save(CardInput.model_validate(request["card"]),
|
||||
request.get("id") if action == "update" else None)
|
||||
elif action == "delete":
|
||||
result = service.delete(request["id"])
|
||||
elif action == "pending":
|
||||
result = service.pending()
|
||||
elif action == "retry":
|
||||
result = service.retry(request["id"])
|
||||
else:
|
||||
raise ValueError("Unknown Memory action")
|
||||
_output(result)
|
||||
except MemoryError as error:
|
||||
_output({"code": error.code, "message": str(error), "status": error.status})
|
||||
raise typer.Exit(1) from None
|
||||
except (ValueError, KeyError, TypeError, OSError):
|
||||
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
|
||||
raise typer.Exit(1) from None
|
||||
finally:
|
||||
if service:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("list")
|
||||
def list_cmd(
|
||||
type_: str = typer.Option(None, "--type", help="Filtra per tipo decisione."),
|
||||
session: str = typer.Option(None, "--session", help="Filtra per sessione."),
|
||||
table: str = typer.Option(None, "--table", help="Filtra per tabella coinvolta."),
|
||||
concept: str = typer.Option(None, "--concept", help="Filtra per concetto."),
|
||||
json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Elenca le memorie del registro (filtri combinati in AND)."""
|
||||
import json as _json
|
||||
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
from tht.memory import load_registry
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
recs = load_registry(registry_path(cfg))
|
||||
if type_:
|
||||
recs = [r for r in recs if r.type == type_]
|
||||
if session:
|
||||
recs = [r for r in recs if r.session_id == session]
|
||||
if table:
|
||||
recs = [r for r in recs if table in r.tables]
|
||||
if concept:
|
||||
recs = [r for r in recs if concept in r.concepts]
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps([r.model_dump(mode="json") for r in recs],
|
||||
ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not recs:
|
||||
typer.secho("Nessuna memoria nel registro.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
t = Table(title="Review memory")
|
||||
for col in ("Id", "Tipo", "Soggetto", "Sessione"):
|
||||
t.add_column(col)
|
||||
for r in recs:
|
||||
t.add_row(r.id, r.type, r.subject, r.session_id)
|
||||
Console().print(t)
|
||||
def list_cmd(filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.list(CardQuery.model_validate_json(filters)))
|
||||
|
||||
|
||||
@memory_app.command("show")
|
||||
def show_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Mostra una singola memoria."""
|
||||
import json as _json
|
||||
def show_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.get(mem_id))
|
||||
|
||||
from tht.memory import load_registry
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
rec = {r.id: r for r in load_registry(registry_path(cfg))}.get(mem_id)
|
||||
if rec is None:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(rec.model_dump(mode="json"), ensure_ascii=False, indent=2))
|
||||
return
|
||||
for k, v in rec.model_dump(mode="json").items():
|
||||
typer.echo(f"{k}: {v}")
|
||||
@memory_app.command("create")
|
||||
def create_cmd(data: Path = typer.Option(..., "--data"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.save(CardInput.model_validate_json(data.read_text())))
|
||||
|
||||
|
||||
@memory_app.command("update")
|
||||
def update_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
subject: str = typer.Option(None, "--subject"),
|
||||
type_: str = typer.Option(None, "--type"),
|
||||
detail: str = typer.Option(None, "--detail"),
|
||||
rationale: str = typer.Option(None, "--rationale"),
|
||||
question_context: str = typer.Option(None, "--question-context"),
|
||||
tables: str = typer.Option(None, "--tables", help="CSV; \"\" per azzerare."),
|
||||
concepts: str = typer.Option(None, "--concepts", help="CSV; \"\" per azzerare."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Modifica i campi di merito di una memoria (provenienza immutabile)."""
|
||||
from typing import get_args
|
||||
|
||||
from tht.decisions import DecisionType
|
||||
from tht.memory import MemoryNotFound, update_record
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
|
||||
fields: dict = {}
|
||||
for name, val in (("subject", subject), ("type", type_), ("detail", detail),
|
||||
("rationale", rationale), ("question_context", question_context)):
|
||||
if val is not None:
|
||||
fields[name] = val
|
||||
if tables is not None:
|
||||
fields["tables"] = [t.strip() for t in tables.split(",") if t.strip()]
|
||||
if concepts is not None:
|
||||
fields["concepts"] = [c.strip() for c in concepts.split(",") if c.strip()]
|
||||
|
||||
if not fields:
|
||||
typer.secho("ERRORE: nessun campo da modificare indicato.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
if "type" in fields and fields["type"] not in get_args(DecisionType):
|
||||
typer.secho(f"ERRORE: tipo '{fields['type']}' non valido.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
require_server_profile(cfg, "memory update")
|
||||
require_vector_cfg(cfg)
|
||||
try:
|
||||
rec = update_record(registry_path(cfg), mem_id, fields)
|
||||
except MemoryNotFound:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
_resync_memory(cfg)
|
||||
typer.secho(f"OK: {rec.id} aggiornata e reindicizzata.", fg=typer.colors.GREEN)
|
||||
def update_cmd(mem_id: str, data: Path = typer.Option(..., "--data"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.save(CardInput.model_validate_json(data.read_text()), mem_id))
|
||||
|
||||
|
||||
@memory_app.command("delete")
|
||||
def delete_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
yes: bool = typer.Option(False, "--yes", "-y", help="Salta la conferma."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Cancella una singola memoria (registro + indice)."""
|
||||
from tht.memory import MemoryNotFound, delete_record
|
||||
def delete_cmd(mem_id: str, yes: bool = typer.Option(False, "--yes", "-y"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
service._admin()
|
||||
if not yes and not typer.confirm(f"Delete Memory card {mem_id} and its links?"):
|
||||
raise typer.Exit(1)
|
||||
_output(service.delete(mem_id))
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_server_profile(cfg, "memory delete")
|
||||
require_vector_cfg(cfg)
|
||||
if not yes and not typer.confirm(f"Cancellare definitivamente la memoria '{mem_id}'?"):
|
||||
typer.secho("Annullato.", fg=typer.colors.YELLOW)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
delete_record(registry_path(cfg), mem_id)
|
||||
except MemoryNotFound:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
_resync_memory(cfg)
|
||||
typer.secho(f"OK: {mem_id} cancellata e deindicizzata.", fg=typer.colors.GREEN)
|
||||
|
||||
@memory_app.command("pending")
|
||||
def pending_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.pending())
|
||||
|
||||
|
||||
@memory_app.command("retry")
|
||||
def retry_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.retry(mem_id))
|
||||
|
||||
|
||||
@memory_app.command("index")
|
||||
def index_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.rebuild())
|
||||
|
||||
|
||||
@memory_app.command("promote")
|
||||
def promote_cmd(session: str = typer.Option(..., "--session"), decision: list[int] = DECISION_OPT,
|
||||
preview: bool = typer.Option(False, "--preview"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
if preview:
|
||||
_output(service.promotions(snapshot)[:5])
|
||||
elif not decision:
|
||||
raise ValueError("Explicit decisions are required")
|
||||
else:
|
||||
results = service.promote(snapshot, decision)
|
||||
_output({"promoted": [r.get("card") for r in results],
|
||||
"indexed": all(r["indexed"] for r in results), "results": results})
|
||||
|
||||
|
||||
@memory_app.command("propose")
|
||||
def propose_cmd(session: str = typer.Option(..., "--session"),
|
||||
data: Path = typer.Option(..., "--data"), config: Path = CONFIG_OPT):
|
||||
"""Persist reviewer-grounded proposals without changing the Memory archive."""
|
||||
from tht.cli.session_cmd import session_repository
|
||||
from tht.memory.review import validate_proposals
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
service._session(snapshot)
|
||||
proposals = validate_proposals(snapshot, json.loads(data.read_text()))
|
||||
session_repository(cfg).write_artifact(session, "memory_proposals",
|
||||
json.dumps([p.model_dump(mode="json") for p in proposals], ensure_ascii=False))
|
||||
_output({"proposals": len(proposals), "saved_to_archive": False})
|
||||
|
||||
|
||||
@memory_app.command("summary")
|
||||
def summary_cmd(session: str = typer.Option(..., "--session"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.memory.review import prepare
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(prepare(service, load_snapshot_or_exit(cfg, session)))
|
||||
|
||||
|
||||
@memory_app.command("repair-prepare")
|
||||
def repair_prepare_cmd(session: str = typer.Option(..., "--session"),
|
||||
proposal_json: str = typer.Option(..., "--proposal-json"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import RepairProposal, prepare
|
||||
|
||||
if len(proposal_json.encode()) > 1_000_000:
|
||||
_output({"code": "memory_invalid", "message": "Repair proposal is too large", "status": 400})
|
||||
raise typer.Exit(1)
|
||||
with _service(config) as (cfg, service):
|
||||
_output(prepare(service, load_snapshot_or_exit(cfg, session), cfg,
|
||||
RepairProposal.model_validate_json(proposal_json)))
|
||||
|
||||
|
||||
@memory_app.command("repair-target")
|
||||
def repair_target_cmd(session: str = typer.Option(..., "--session"),
|
||||
archive: str = typer.Option(..., "--archive"),
|
||||
target_id: str = typer.Option(..., "--target-id"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import target
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(target(service, load_snapshot_or_exit(cfg, session), cfg, archive, target_id))
|
||||
|
||||
|
||||
@memory_app.command("repairs")
|
||||
def repairs_cmd(session: str = typer.Option(..., "--session"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import list_repairs
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(list_repairs(service, load_snapshot_or_exit(cfg, session)))
|
||||
|
||||
|
||||
@memory_app.command("repair-show")
|
||||
def repair_show_cmd(session: str = typer.Option(..., "--session"),
|
||||
repair_id: str = typer.Option(..., "--repair-id"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import show
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(show(service, load_snapshot_or_exit(cfg, session), cfg, repair_id))
|
||||
|
||||
|
||||
@memory_app.command("repair-apply")
|
||||
def repair_apply_cmd(session: str = typer.Option(..., "--session"),
|
||||
repair_id: str = typer.Option(..., "--repair-id"),
|
||||
choice: str = typer.Option(..., "--choice"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import apply
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
|
||||
def activate(snapshot):
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise RuntimeError("Evidence activation did not complete")
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(apply(service, load_snapshot_or_exit(cfg, session), cfg, repair_id, choice,
|
||||
activate=activate))
|
||||
|
||||
|
||||
@memory_app.command("review-apply")
|
||||
def review_apply_cmd(session: str = typer.Option(..., "--session"),
|
||||
review_json: str = typer.Option(..., "--review-json"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.memory.review import ReviewResponse, apply
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(apply(service, load_snapshot_or_exit(cfg, session),
|
||||
ReviewResponse.model_validate_json(review_json)))
|
||||
|
||||
|
||||
@memory_app.command("save-one")
|
||||
def save_one_cmd(session: str = typer.Option(..., "--session"),
|
||||
decision: int = typer.Option(..., "--decision"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (cfg, service):
|
||||
results = service.promote(load_snapshot_or_exit(cfg, session), [decision])
|
||||
_output({"upserted": sum(r["indexed"] for r in results), "decision_seq": decision,
|
||||
"indexed": all(r["indexed"] for r in results), "results": results})
|
||||
|
||||
|
||||
@memory_app.command("search")
|
||||
def search_cmd(
|
||||
question: str = typer.Argument(..., help="Domanda o termini di ricerca."),
|
||||
top: int = typer.Option(5, "--top"),
|
||||
session: str = typer.Option(
|
||||
None, "--session",
|
||||
help="Esclude le memorie gia' decise (applicate o rifiutate) in questa sessione.",
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON per Pi."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Cerca memorie riapplicabili, ordinate per similarita'. Mai applicate in automatico.
|
||||
def search_cmd(question: str, top: int = typer.Option(5, "--top"),
|
||||
session: str | None = typer.Option(None, "--session"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
with _service(config) as (cfg, service):
|
||||
decisions = load_snapshot_or_exit(cfg, session).decisions if session else []
|
||||
_output(service.recall(question, searcher=open_searcher(cfg),
|
||||
embedder=make_embedder(cfg.embeddings), top=top, decisions=decisions,
|
||||
scope=_recall_scope(cfg, filters)))
|
||||
|
||||
Con `--session` non ripropone le memorie gia' decise in quella sessione (fix:
|
||||
memorie scartate riproposte): rifiutate via `memory_rejected` o gia' applicate."""
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
|
||||
from tht.memory import load_registry, recall_memories
|
||||
def _recall_scope(cfg, filters):
|
||||
from tht.memory.retrieval import RecallScope
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
decisions = load_snapshot_or_exit(cfg, session).decisions if session is not None else []
|
||||
searcher = open_searcher(cfg)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
results = recall_memories(
|
||||
question,
|
||||
records=load_registry(registry_path(cfg)),
|
||||
decisions=decisions,
|
||||
searcher=searcher,
|
||||
embedder=embedder,
|
||||
top=top,
|
||||
)
|
||||
context = {"database": cfg.database.database, "schema_name": cfg.database.db_schema}
|
||||
supplied = json.loads(filters)
|
||||
if not isinstance(supplied, dict) or any(
|
||||
key in supplied and supplied[key] != value for key, value in context.items()
|
||||
):
|
||||
raise ValueError("Recall cannot override the configured database/schema context")
|
||||
return RecallScope.model_validate({**supplied, **context})
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not results:
|
||||
typer.secho("Nessuna memoria candidata.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Memorie candidate per: {question}")
|
||||
table.add_column("Id")
|
||||
table.add_column("Tipo")
|
||||
table.add_column("Soggetto")
|
||||
table.add_column("Contesto originale")
|
||||
table.add_column("Score", justify="right")
|
||||
for r in results:
|
||||
table.add_row(r["id"], r["type"], r["subject"],
|
||||
r["question_context"][:60], f"{r['score']:.3f}")
|
||||
Console().print(table)
|
||||
|
||||
@memory_app.command("rules")
|
||||
def rules_cmd(question: str, session: str = typer.Option(..., "--session"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
"""Consult SQL rules and explained errors in schema linking and SQL construction."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
from tht.phase import current_phase
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
service._session(snapshot)
|
||||
if current_phase(snapshot) not in {4, 6, 7}:
|
||||
raise ValueError("Memory rules are consulted in schema linking or SQL construction")
|
||||
scope = _recall_scope(cfg, filters)
|
||||
vector = make_embedder(cfg.embeddings).embed_query(question)
|
||||
embedder = SimpleNamespace(embed_query=lambda _: vector)
|
||||
searcher = open_searcher(cfg)
|
||||
candidates = []
|
||||
for family in ("sql_rule", "explained_error"):
|
||||
candidates.extend(service.retrieve(question, searcher=searcher, embedder=embedder,
|
||||
scope=scope, family=family, top=5))
|
||||
_output([{**candidate.card.model_dump(mode="json"), "score": candidate.score,
|
||||
"retrieval_path": candidate.path, "consultative": True}
|
||||
for candidate in sorted(candidates, key=lambda c: (-c.score, c.card.id))[:10]])
|
||||
|
||||
|
||||
def index_solved_session(cfg, session_id: str) -> int:
|
||||
"""Indicizza la coppia domanda->SQL della sessione (kind solved_question).
|
||||
|
||||
Solleva SolvedIndexError se mancano gli artefatti: il finalize lo degrada a warning,
|
||||
il comando CLI lo converte in errore esplicito."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.sql_cmd import promoted_tables_for
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import index_solved_question
|
||||
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
return index_solved_question(
|
||||
load_snapshot_or_exit(cfg, session_id),
|
||||
promoted_tables_for(cfg, session_id),
|
||||
store=store,
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
"""Recovery only: never recreate a deleted card from historical session artifacts."""
|
||||
service = memory_service(cfg)
|
||||
try:
|
||||
result = service.retry_solved(load_snapshot_or_exit(cfg, session_id))
|
||||
if not result["indexed"]:
|
||||
raise RuntimeError(result["error"])
|
||||
return int(result["action"] == "upsert")
|
||||
finally:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("solved-index")
|
||||
def solved_index_cmd(
|
||||
session_id: str = typer.Argument(..., help="Id sessione con sql_final.sql approvato."),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Indicizza la coppia domanda->SQL nel semantic store (backfill; il finalize lo fa da solo)."""
|
||||
import json as _json
|
||||
|
||||
from tht.memory import SolvedIndexError
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "memory solved-index")
|
||||
try:
|
||||
count = index_solved_session(cfg, session_id)
|
||||
except RuntimeError as e:
|
||||
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=4)
|
||||
except SolvedIndexError as e:
|
||||
typer.secho(f"ERRORE: sessione {session_id} non indicizzabile: {e}",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=3)
|
||||
msg = (
|
||||
f"1 coppia domanda->SQL indicizzata (solved:{session_id})."
|
||||
if count else "Nessun upsert: coppia gia' aggiornata."
|
||||
)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps({"upserted": count, "id": f"solved:{session_id}"},
|
||||
ensure_ascii=False))
|
||||
return
|
||||
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN)
|
||||
def solved_index_cmd(session_id: str, json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT):
|
||||
"""Retry an existing authoritative exemplar's Qdrant projection; never import a session."""
|
||||
with _service(config) as (cfg, service):
|
||||
_output(service.retry_solved(load_snapshot_or_exit(cfg, session_id)))
|
||||
|
||||
|
||||
@memory_app.command("solved-search")
|
||||
def solved_search_cmd(
|
||||
question: str = typer.Argument(..., help="Domanda da confrontare con quelle risolte."),
|
||||
top: int = typer.Option(3, "--top"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Domande gia' risolte simili (kind solved_question): domanda, SQL e tabelle."""
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
def solved_search_cmd(question: str, top: int = typer.Option(3, "--top"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
from tht.memory import search_solved_questions
|
||||
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
# Degrado gentile: SKILL.md prescrive solved-search in F4/F6/F7 di ogni sessione,
|
||||
# quindi vectordb/Ollama irraggiungibili non devono produrre un traceback grezzo
|
||||
# nel transcript: avviso di una riga su stderr, stdout puro ([] in --json), exit 0.
|
||||
try:
|
||||
searcher = open_searcher(cfg)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
results = search_solved_questions(
|
||||
question,
|
||||
searcher=searcher,
|
||||
embedder=embedder,
|
||||
top=top,
|
||||
)
|
||||
except (VectorStoreError, VectorReadUnavailable, EmbeddingsError, OperationalError) as e:
|
||||
typer.secho(
|
||||
f"ATTENZIONE: exemplar non disponibili ({e}). Prosegui senza.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
if json_out:
|
||||
typer.echo("[]")
|
||||
return
|
||||
if json_out:
|
||||
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not results:
|
||||
typer.secho("Nessuna domanda risolta simile.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Domande risolte simili a: {question}")
|
||||
table.add_column("Sessione")
|
||||
table.add_column("Domanda")
|
||||
table.add_column("Tabelle")
|
||||
table.add_column("Score", justify="right")
|
||||
for r in results:
|
||||
table.add_row(r["session_id"], r["question"][:60],
|
||||
", ".join(r["tables"]), f"{r['score']:.3f}")
|
||||
Console().print(table)
|
||||
with _service(config) as (cfg, service):
|
||||
try:
|
||||
result = service.recall(question, searcher=open_searcher(cfg),
|
||||
embedder=make_embedder(cfg.embeddings), top=top, solved=True,
|
||||
scope=_recall_scope(cfg, filters))
|
||||
except (VectorStoreError, EmbeddingsError):
|
||||
typer.echo("Avviso: exemplar non disponibili; ricerca semantica non riuscita.", err=True)
|
||||
result = []
|
||||
_output(result)
|
||||
|
||||
@@ -50,6 +50,10 @@ def _evaluation_workspace_root(cfg) -> Path:
|
||||
|
||||
def _requires_candidate_evaluation(cfg) -> bool:
|
||||
evidence = cfg.evidence
|
||||
if evidence and evidence.local_archive_root and (
|
||||
evidence.local_archive_root / "evidence/.local/state.yaml"
|
||||
).exists():
|
||||
return False
|
||||
if evidence is None or evidence.schema_version != 2:
|
||||
return False
|
||||
return evidence.source_root is not None or any(
|
||||
@@ -224,7 +228,8 @@ def _parse_dwh_steps(value: str) -> tuple[str, ...]:
|
||||
return steps
|
||||
|
||||
|
||||
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None):
|
||||
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None,
|
||||
local_snapshot: Path | None = None):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
@@ -239,8 +244,11 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
|
||||
corpus_root = cfg.paths.artifacts.parent / "corpus"
|
||||
vector_store = build_vector_store(cfg, require_write=True)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
from tht.evidence.adapters import FilesystemEvidenceSource
|
||||
sources = [FilesystemEvidenceSource(local_snapshot, patterns=("curated/**/*.md",))] \
|
||||
if local_snapshot is not None else build_sources(cfg.evidence)
|
||||
pipeline = build_preprocessing_pipeline(
|
||||
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
|
||||
store=CorpusStore(corpus_root), sources=sources,
|
||||
embedder=embedder,
|
||||
vector_store=vector_store,
|
||||
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
|
||||
@@ -376,9 +384,17 @@ def evidence_cmd(
|
||||
dry_run: bool = typer.Option(False, "--dry-run"),
|
||||
resume: str | None = typer.Option(None, "--resume"),
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
consolidate: bool = typer.Option(False, "--consolidate"),
|
||||
) -> None:
|
||||
if action is not None and action != "gc":
|
||||
raise typer.BadParameter("only the optional 'gc' action is supported")
|
||||
if consolidate and (dry_run or resume is not None or action is not None):
|
||||
message = "Consolidation cannot be combined with dry-run, resume or gc"
|
||||
if json_output:
|
||||
typer.echo(json.dumps({"status": "failed", "code": "invalid_consolidation", "error": message}))
|
||||
else:
|
||||
typer.secho(message, fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=2)
|
||||
if action == "gc":
|
||||
try:
|
||||
payload = gc_from_config(config, dry_run=dry_run)
|
||||
@@ -406,21 +422,29 @@ def evidence_cmd(
|
||||
typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=2)
|
||||
try:
|
||||
result = run_from_config(config, dry_run=dry_run, resume=resume)
|
||||
except Exception: # noqa: BLE001
|
||||
if consolidate:
|
||||
from tht.evidence.administration import consolidate_from_config
|
||||
result = consolidate_from_config(config)
|
||||
else:
|
||||
result = run_from_config(config, dry_run=dry_run, resume=resume)
|
||||
except Exception as error: # noqa: BLE001
|
||||
from tht.evidence.administration import ConsolidationError
|
||||
detail = str(error) if isinstance(error, ConsolidationError) else "preprocessing failed"
|
||||
payload = {"status": "failed"}
|
||||
if isinstance(error, ConsolidationError):
|
||||
payload["saved"] = error.saved
|
||||
if json_output:
|
||||
typer.echo(json.dumps(
|
||||
_evidence_json_payload(
|
||||
cfg,
|
||||
payload,
|
||||
code="preprocessing_failed",
|
||||
error="preprocessing failed",
|
||||
error=detail,
|
||||
),
|
||||
sort_keys=True,
|
||||
))
|
||||
else:
|
||||
typer.secho("ERRORE: preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
typer.secho(f"ERRORE: {detail}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
payload = result.model_dump(mode="json")
|
||||
if payload.get("status") != "succeeded":
|
||||
|
||||
@@ -624,44 +624,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
|
||||
repository, session_id, validation_report=report, evidence=evidence
|
||||
)
|
||||
# --- memoria attiva (parte B): indicizza la coppia domanda->SQL, best-effort ---
|
||||
# Qualunque errore (writer key assente, VPN giu', Ollama spento) NON deve
|
||||
# bloccare il finalize: l'indice e' derivato e recuperabile con
|
||||
# `tht memory solved-index <id>`. Memory owns this best-effort policy; core
|
||||
# has already committed the authoritative finalized snapshot above.
|
||||
try:
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import index_solved_question_best_effort
|
||||
|
||||
finalized_snapshot = repository.get(session_id)
|
||||
outcome = index_solved_question_best_effort(
|
||||
finalized_snapshot,
|
||||
promoted_tables,
|
||||
store_factory=lambda: build_vector_store(cfg, require_write=True),
|
||||
embedder_factory=lambda: make_embedder(cfg.embeddings),
|
||||
)
|
||||
if outcome.error is not None:
|
||||
typer.secho(
|
||||
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({outcome.error}). "
|
||||
f"Recupera con `tht memory solved-index {session_id}`.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
elif outcome.upserted:
|
||||
typer.secho(
|
||||
"OK: coppia domanda->SQL indicizzata nel vectordb (solved_question).",
|
||||
fg=typer.colors.GREEN,
|
||||
)
|
||||
else:
|
||||
typer.secho(
|
||||
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
|
||||
fg=typer.colors.CYAN,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
|
||||
typer.secho(
|
||||
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
|
||||
f"Recupera con `tht memory solved-index {session_id}`.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
# Memory cards, including exemplars, are saved only by the explicit final review.
|
||||
typer.secho(f"OK: sessione {session_id} finalizzata. Artefatti:", fg=typer.colors.GREEN)
|
||||
for name in ARTIFACT_FILES:
|
||||
state = "presente" if name not in {"session_manifest.yaml", "review_decisions.jsonl"} else "persistito"
|
||||
|
||||
@@ -512,6 +512,8 @@ EvidenceSourceConfig = Annotated[
|
||||
|
||||
|
||||
class EvidenceSourcesConfig(BaseModel):
|
||||
# Persistent curator checkout; only its activated snapshot is used by preprocessing.
|
||||
local_archive_root: Path | None = None
|
||||
# Version 2 is the materialized source/curated authoring layout.
|
||||
schema_version: Literal[1, 2] = 1
|
||||
# Legacy curated-tree configuration remains accepted during migration.
|
||||
|
||||
@@ -43,6 +43,7 @@ DecisionType = Literal[
|
||||
# declined_promotion_seqs per non riproporre i candidati rifiutati).
|
||||
"memory_promoted",
|
||||
"memory_promotion_declined",
|
||||
"memory_summary_reviewed",
|
||||
# D15: marker di ritrazione. subject = "phase:N", retracts = decision_seq ritirata.
|
||||
# Resta nel log di audit (append-only); effective_decisions() la esclude dalla vista.
|
||||
"decision_retracted",
|
||||
|
||||
@@ -22,6 +22,7 @@ from tht.evidence.authoring import (
|
||||
)
|
||||
from tht.evidence.canonical import (
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
load_curated_tree,
|
||||
parse_curated_markdown,
|
||||
@@ -37,6 +38,7 @@ from tht.evidence.contracts import (
|
||||
validate_namespaced_value,
|
||||
validate_safe_metadata,
|
||||
)
|
||||
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive
|
||||
from tht.evidence.preprocessing import EvidenceEmbedder, build_preprocessing_pipeline
|
||||
from tht.evidence.search import (
|
||||
ActiveEvidenceSearcher,
|
||||
@@ -58,6 +60,7 @@ from tht.evidence.sources import build_sources
|
||||
__all__ = [
|
||||
"AcquiredDocument",
|
||||
"ActiveEvidenceSearcher",
|
||||
"ArchiveConflict",
|
||||
"CorpusWorkspaceMismatchError",
|
||||
"CuratedEvidence",
|
||||
"EvidenceEmbedder",
|
||||
@@ -75,6 +78,8 @@ __all__ = [
|
||||
"EvidenceSource",
|
||||
"EvidenceSourceError",
|
||||
"EvidenceSourceErrorCategory",
|
||||
"LocalEvidenceArchive",
|
||||
"ManualEvidenceProvenance",
|
||||
"PiEvidenceRestructurer",
|
||||
"RestructureCandidate",
|
||||
"RestructureRequest",
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Local Evidence browsing and explicit consolidation, independent of DWH access."""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from .canonical import parse_curated_markdown
|
||||
from .local_archive import LocalEvidenceArchive, _content
|
||||
|
||||
|
||||
class ConsolidationError(RuntimeError):
|
||||
def __init__(self, message, *, saved=False):
|
||||
super().__init__(message)
|
||||
self.saved = saved
|
||||
|
||||
|
||||
def consolidate_from_config(config: Path):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import EvidencePreparationError, migrate_workspace_evidence
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ConsolidationError("Local Evidence is not configured for this workspace")
|
||||
root = cfg.evidence.local_archive_root
|
||||
archive = LocalEvidenceArchive(root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
try:
|
||||
# Explicit first consolidation performs the one-time legacy conversion.
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(root)
|
||||
else:
|
||||
archive.initialize()
|
||||
except (ValueError, OSError, EvidencePreparationError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
result = None
|
||||
|
||||
def activate(snapshot):
|
||||
nonlocal result
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise ConsolidationError("Evidence indexing is blocked; check unit size and review items", saved=True)
|
||||
except ConsolidationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Evidence files were saved, but indexing failed. Retry consolidation.", saved=True) from error
|
||||
|
||||
try:
|
||||
archive.consolidate(actor=os.environ.get("THT_PRINCIPAL_SUBJECT") or "installation operator",
|
||||
activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
return result
|
||||
|
||||
|
||||
def browse(root: Path, query: dict):
|
||||
"""Read complete working units and their active status without opening an index."""
|
||||
archive = LocalEvidenceArchive(root)
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
active = archive._snapshot(state["active"]) if state["active"] else None
|
||||
active_units = archive._units(archive._files(active), allow_review=True) if active else {}
|
||||
files = archive._files(archive.evidence)
|
||||
items, errors = [], []
|
||||
seen = set()
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode(), path=Path(relative))
|
||||
if unit.id in seen:
|
||||
raise ValueError("Duplicate Evidence identifier")
|
||||
seen.add(unit.id)
|
||||
old = active_units.get(unit.id)
|
||||
status = "review_required" if unit.review_items else "legacy" if unit.schema_version != 4 \
|
||||
else "active" if old and _content(old[1]) == _content(unit) and old[1].provenance == unit.provenance else "modified" if old else "new"
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative, "status": status,
|
||||
"revision": _content(unit)})
|
||||
except (ValueError, UnicodeError) as error:
|
||||
errors.append({"file": relative, "message": str(error)[:1500]})
|
||||
for identity, (relative, unit) in active_units.items():
|
||||
if identity not in seen:
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative,
|
||||
"status": "invalid" if relative in files else "removed", "revision": _content(unit)})
|
||||
def matches(item):
|
||||
for field in ("kind", "status", "language"):
|
||||
if query.get(field) and item[field] != query[field]:
|
||||
return False
|
||||
if query.get("purpose") and query["purpose"] not in item["purposes"]:
|
||||
return False
|
||||
for key, field in (("concept", "concepts"), ("table", "tables"), ("column", "columns")):
|
||||
if query.get(key) and not any(query[key].casefold() in v.casefold() for v in item["applies_to"][field]):
|
||||
return False
|
||||
provenance = item["provenance"]
|
||||
if query.get("source") and query["source"].casefold() not in str(provenance).casefold():
|
||||
return False
|
||||
return not query.get("q") or query["q"].casefold() in str(item).casefold()
|
||||
selected = [item for item in items if matches(item)]
|
||||
selected.sort(key=lambda item: (str(item.get(query.get("sort", "title"), "")).casefold(), item["id"]),
|
||||
reverse=query.get("direction") == "desc")
|
||||
page, size = int(query.get("page", 1)), int(query.get("page_size", 25))
|
||||
if page < 1 or not 1 <= size <= 100:
|
||||
raise ValueError("Invalid Evidence page")
|
||||
result = {"items": selected[(page-1)*size:page*size], "total": len(selected), "page": page,
|
||||
"page_size": size, "errors": errors, "active_revision": state["active"],
|
||||
"pending_revision": state["pending"], "initialized": bool(state.get("baseline") or state["pending"] or state["active"])}
|
||||
if query.get("id"):
|
||||
result["item"] = next((item for item in items if item["id"] == query["id"]), None)
|
||||
from .imports import reviews
|
||||
result["source_reviews"] = reviews(archive)
|
||||
return result
|
||||
|
||||
|
||||
def source_action(config, *, action, source_id=None, revision=None, decision=None, actor="installation operator"):
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import PiEvidenceRestructurer, authoring_skill_path, migrate_workspace_evidence
|
||||
from .imports import acquisition_sources, decide, refresh
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ValueError("Local Evidence is not configured")
|
||||
archive = LocalEvidenceArchive(cfg.evidence.local_archive_root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(archive.root)
|
||||
else:
|
||||
(archive.evidence / "curated").mkdir(parents=True, exist_ok=True)
|
||||
archive.initialize()
|
||||
if action == "refresh":
|
||||
skill = authoring_skill_path()
|
||||
return refresh(archive, acquisition_sources(cfg), PiEvidenceRestructurer(
|
||||
os.environ.get("THT_PI_EXECUTABLE", "pi"), skill))
|
||||
if action != "decide":
|
||||
raise ValueError("Unknown source action")
|
||||
|
||||
def activate(snapshot):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise RuntimeError("Indexing did not succeed")
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Source decision saved, but indexing failed. Retry the same decision.", saved=True) from error
|
||||
|
||||
try:
|
||||
return decide(archive, source_id=source_id, revision=revision, decision=decision,
|
||||
actor=actor, activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
from .imports import reviews
|
||||
if any(r["id"] == source_id and r["status"] == "applying" for r in reviews(archive)):
|
||||
raise ConsolidationError(f"Source decision saved. {str(error)[:1200]}. Retry the same decision.", saved=True) from error
|
||||
raise
|
||||
@@ -25,6 +25,7 @@ from tht.evidence.canonical import (
|
||||
EvidenceKind,
|
||||
EvidencePurpose,
|
||||
EvidenceScope,
|
||||
ManualEvidenceProvenance,
|
||||
ReviewItem,
|
||||
StrictModel,
|
||||
dump_curated_markdown,
|
||||
@@ -189,6 +190,12 @@ def _restore_exact_source_excerpts(
|
||||
})
|
||||
|
||||
|
||||
def authoring_skill_path() -> Path:
|
||||
"""The installed wheel and the deployment's Pi resources live in different roots."""
|
||||
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
|
||||
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
|
||||
|
||||
class PiEvidenceRestructurer:
|
||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||
|
||||
@@ -389,6 +396,14 @@ def dump_manifest(manifest: EvidenceManifest) -> str:
|
||||
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
|
||||
"""Validate the curated corpus without writing the workspace."""
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").is_file():
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
try:
|
||||
LocalEvidenceArchive(workspace_root).validate()
|
||||
return ValidationReport(())
|
||||
except (OSError, ValueError) as error:
|
||||
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
|
||||
"evidence/curated", str(error)),))
|
||||
findings: list[ValidationFinding] = []
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
if not manifest_path.is_file():
|
||||
@@ -485,6 +500,9 @@ def _validate_manifest_source(
|
||||
def _validate_unit(
|
||||
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
||||
) -> list[ValidationFinding]:
|
||||
if isinstance(evidence.provenance, ManualEvidenceProvenance):
|
||||
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
|
||||
for item in evidence.review_items]
|
||||
if evidence.id in manifest.orphans:
|
||||
return []
|
||||
path = evidence.provenance.source_file
|
||||
@@ -560,6 +578,8 @@ def prepare_workspace_evidence(
|
||||
raise EvidencePreparationError("authoring_workers_invalid")
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
try:
|
||||
@@ -695,10 +715,9 @@ def migrate_workspace_evidence(
|
||||
*,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
) -> EvidenceMigrationReport:
|
||||
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
|
||||
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
documents = load_curated_tree(evidence_root / "curated")
|
||||
@@ -709,7 +728,7 @@ def migrate_workspace_evidence(
|
||||
if len(documents_by_id) != len(documents):
|
||||
raise EvidencePreparationError("duplicate_evidence_id")
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 3})
|
||||
evidence_id: document.model_copy(update={"schema_version": 4})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
migrated_ids: list[str] = []
|
||||
@@ -728,12 +747,16 @@ def migrate_workspace_evidence(
|
||||
migrated = tuple(sorted(migrated_ids))
|
||||
unchanged = tuple(sorted(unchanged_ids))
|
||||
if not migrated:
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=(),
|
||||
unchanged=unchanged,
|
||||
findings=validate_workspace_evidence(workspace_root).findings,
|
||||
)
|
||||
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=migrated,
|
||||
unchanged=unchanged,
|
||||
@@ -762,6 +785,8 @@ def resolve_workspace_evidence(
|
||||
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local/state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
|
||||
_reject_dirty_worktree(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
@@ -1017,7 +1042,7 @@ def _candidate_to_evidence(
|
||||
mode="json",
|
||||
exclude={"schema_version", "existing_id", "supporting_excerpts"},
|
||||
)
|
||||
data["schema_version"] = 3
|
||||
data["schema_version"] = 4
|
||||
data["id"] = evidence_id
|
||||
data["provenance"] = {
|
||||
"source_file": source_file,
|
||||
@@ -1038,7 +1063,7 @@ def _unsupported_unit(
|
||||
message="The current source no longer supports this Evidence unit.",
|
||||
),)
|
||||
return evidence.model_copy(update={
|
||||
"schema_version": 3,
|
||||
"schema_version": 4,
|
||||
"provenance": evidence.provenance.model_copy(update={
|
||||
"source_file": source_file,
|
||||
"source_sha256": source_hash,
|
||||
|
||||
@@ -98,6 +98,19 @@ class ReviewItem(StrictModel):
|
||||
field: str | None = None
|
||||
|
||||
|
||||
class ManualEvidenceProvenance(StrictModel):
|
||||
"""The curator supports the current content; an earlier document is only its origin."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
kind: Literal["manual"] = "manual"
|
||||
declared_by: str = Field(min_length=1)
|
||||
original: EvidenceProvenance | None = None
|
||||
|
||||
@property
|
||||
def source_file(self) -> str:
|
||||
return f"Manual declaration: {self.declared_by}"
|
||||
|
||||
|
||||
class FormulaPayload(StrictModel):
|
||||
concept: str
|
||||
columns: tuple[str, ...]
|
||||
@@ -229,19 +242,23 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
|
||||
|
||||
class CuratedEvidence(StrictModel):
|
||||
schema_version: Literal[1, 2, 3]
|
||||
schema_version: Literal[1, 2, 3, 4]
|
||||
id: str
|
||||
title: str
|
||||
kind: EvidenceKind
|
||||
purposes: tuple[EvidencePurpose, ...]
|
||||
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
||||
language: str
|
||||
provenance: EvidenceProvenance
|
||||
provenance: EvidenceProvenance | ManualEvidenceProvenance
|
||||
review_items: tuple[ReviewItem, ...] = ()
|
||||
payload: EvidencePayload
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _validate_kind_payload(self) -> CuratedEvidence:
|
||||
if isinstance(self.provenance, ManualEvidenceProvenance) and self.schema_version != 4:
|
||||
raise ValueError("manual declarations require Curated unit schema v4")
|
||||
if self.schema_version == 4 and (not self.purposes or not self.language.strip()):
|
||||
raise ValueError("editable Evidence requires a language and at least one purpose")
|
||||
if not is_evidence_id(self.id):
|
||||
raise ValueError("id must use the evidence:<slug> form")
|
||||
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
|
||||
@@ -1056,13 +1073,19 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
_, frontmatter, body = text.split("---\n", 2)
|
||||
except ValueError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
try:
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
except yaml.YAMLError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
try:
|
||||
data = dict(raw)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
||||
if data.get("schema_version") == 2:
|
||||
data = _parse_v2_body(data, body)
|
||||
elif data.get("schema_version") == 4:
|
||||
from tht.evidence.editable import parse_document, parse_metadata
|
||||
data = parse_document(parse_metadata(frontmatter), body)
|
||||
else:
|
||||
if body.strip():
|
||||
raise ValueError("curated evidence must not contain an ignored body")
|
||||
@@ -1081,6 +1104,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
|
||||
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
||||
"""Render one canonical Curated Evidence Markdown document."""
|
||||
if value.schema_version == 4:
|
||||
from tht.evidence.editable import render_document
|
||||
return render_document(value)
|
||||
if value.schema_version == 3:
|
||||
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
|
||||
if value.schema_version == 2:
|
||||
|
||||
@@ -8,7 +8,12 @@ from typing import Self
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
|
||||
|
||||
from tht.evidence.canonical import EVIDENCE_KINDS, EVIDENCE_PURPOSES
|
||||
from tht.evidence.canonical import (
|
||||
EVIDENCE_KINDS,
|
||||
EVIDENCE_PURPOSES,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
)
|
||||
from tht.evidence.contracts import (
|
||||
canonical_provenance_uri,
|
||||
normalize_aware_datetime,
|
||||
@@ -76,10 +81,10 @@ def _validate_evidence_metadata(metadata: Mapping[str, JsonValue]) -> None:
|
||||
if not isinstance(metadata["language"], str) or not metadata["language"]:
|
||||
raise ValueError("typed Evidence metadata must contain language")
|
||||
provenance = metadata["provenance"]
|
||||
if not isinstance(provenance, dict) or set(provenance) != {
|
||||
"source_file", "source_sha256", "supporting_excerpts",
|
||||
}:
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance")
|
||||
if not isinstance(provenance, dict):
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance") # noqa: TRY004
|
||||
model = ManualEvidenceProvenance if provenance.get("kind") == "manual" else EvidenceProvenance
|
||||
model.model_validate(provenance)
|
||||
|
||||
|
||||
class _CanonicalValue(BaseModel):
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
"""Curated unit v4: visible Markdown fields are the sole human-content authority."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
_PAYLOAD_TYPE_BY_KIND,
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
_v2_labels,
|
||||
)
|
||||
|
||||
_LIST_FIELDS = {"synonyms", "variants", "columns", "tables"}
|
||||
|
||||
|
||||
def parse_metadata(text: str) -> dict:
|
||||
class UniqueKeysLoader(yaml.SafeLoader):
|
||||
pass
|
||||
|
||||
def mapping(loader, node):
|
||||
pairs = loader.construct_pairs(node, deep=True)
|
||||
result = {}
|
||||
for key, value in pairs:
|
||||
if key in result:
|
||||
raise ValueError(f"Duplicate metadata key: {key}")
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping)
|
||||
try:
|
||||
return yaml.load(text, Loader=UniqueKeysLoader)
|
||||
except (yaml.YAMLError, TypeError) as error:
|
||||
raise ValueError("Evidence metadata is malformed") from error
|
||||
|
||||
|
||||
def _sections(body: str, headings: dict[str, str]) -> dict[str, str]:
|
||||
"""Recognize structural H2s outside code fences; all other Markdown is content."""
|
||||
sections: dict[str, list[str]] = {}
|
||||
field = None
|
||||
fence = None
|
||||
for line in body.splitlines():
|
||||
match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line)
|
||||
if match:
|
||||
marker = match[1]
|
||||
if fence is None:
|
||||
fence = marker
|
||||
elif marker[0] == fence[0] and len(marker) >= len(fence):
|
||||
fence = None
|
||||
heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None
|
||||
if heading:
|
||||
if heading in sections:
|
||||
raise ValueError(f"Duplicate section: {line[3:]}")
|
||||
field = heading
|
||||
sections[field] = []
|
||||
elif field is not None:
|
||||
sections[field].append(line)
|
||||
elif line.strip():
|
||||
raise ValueError("Content must follow a documented section heading")
|
||||
if fence:
|
||||
raise ValueError("Unclosed Markdown code fence")
|
||||
return {key: "\n".join(lines).strip() for key, lines in sections.items()}
|
||||
|
||||
|
||||
def _list(text: str) -> list[str]:
|
||||
if not text:
|
||||
return []
|
||||
values = []
|
||||
for line in text.splitlines():
|
||||
if not line.startswith("- ") or not line[2:].strip():
|
||||
raise ValueError("List entries must use '- value', one per line")
|
||||
value = line[2:]
|
||||
# Quoted strings preserve multiline and unusual values during migration.
|
||||
values.append(json.loads(value) if value.startswith('"') else value)
|
||||
return values
|
||||
|
||||
|
||||
def _values(text: str) -> dict[str, str]:
|
||||
values = {}
|
||||
key = None
|
||||
lines = []
|
||||
for line in text.splitlines():
|
||||
if line.startswith("### "):
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
label = line[4:]
|
||||
key = json.loads(label) if label.startswith('"') else label
|
||||
if key in values:
|
||||
raise ValueError("Duplicate enum value")
|
||||
lines = []
|
||||
elif key is None:
|
||||
if line.strip():
|
||||
raise ValueError("Enum values require '### value' headings")
|
||||
else:
|
||||
lines.append(line)
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
return values
|
||||
|
||||
|
||||
def parse_document(metadata: dict, body: str) -> dict:
|
||||
data = dict(metadata)
|
||||
if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data):
|
||||
raise ValueError("Title and payload must be edited only in the Markdown body")
|
||||
lines = body.strip().splitlines()
|
||||
if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip():
|
||||
raise ValueError("A title starting with '# ' is required")
|
||||
data["title"] = lines[0][2:].strip()
|
||||
kind = data.get("kind")
|
||||
if kind not in _PAYLOAD_TYPE_BY_KIND:
|
||||
raise ValueError("Unknown Evidence kind")
|
||||
labels = _v2_labels(str(data.get("language", "")))
|
||||
fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields
|
||||
sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields})
|
||||
required = {name for name, field in fields.items() if field.is_required()}
|
||||
if not required <= sections.keys():
|
||||
raise ValueError(
|
||||
"Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys()))
|
||||
)
|
||||
payload = {}
|
||||
for key, content in sections.items():
|
||||
if key in _LIST_FIELDS:
|
||||
payload[key] = _list(content)
|
||||
elif key == "values":
|
||||
payload[key] = _values(content)
|
||||
elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"):
|
||||
payload[key] = content[7:-4]
|
||||
else:
|
||||
payload[key] = content
|
||||
if fields[key].is_required() and not payload[key] and key != "values":
|
||||
raise ValueError(f"Section {labels[key]} must not be empty")
|
||||
data["payload"] = payload
|
||||
data.setdefault(
|
||||
"provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump()
|
||||
)
|
||||
return data
|
||||
|
||||
|
||||
def render_document(value: CuratedEvidence) -> str:
|
||||
metadata = value.model_dump(mode="json", exclude={"title", "payload"})
|
||||
labels = _v2_labels(value.language)
|
||||
parts = [
|
||||
f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}"
|
||||
]
|
||||
for key, content in value.payload.model_dump(mode="json").items():
|
||||
if key in _LIST_FIELDS:
|
||||
rendered = "\n".join(
|
||||
"- "
|
||||
+ (
|
||||
json.dumps(v, ensure_ascii=False)
|
||||
if "\n" in v or v.startswith('"') or v != v.strip()
|
||||
else v
|
||||
)
|
||||
for v in content
|
||||
)
|
||||
elif key == "values":
|
||||
rendered = "\n\n".join(
|
||||
f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items()
|
||||
)
|
||||
elif key == "sql":
|
||||
rendered = f"```sql\n{content}\n```"
|
||||
else:
|
||||
rendered = content
|
||||
parts.append(f"## {labels[key]}\n\n{rendered}")
|
||||
rendered = "\n\n".join(parts) + "\n"
|
||||
# Migration must fail explicitly rather than silently changing unrepresentable content.
|
||||
restored = CuratedEvidence.model_validate(
|
||||
parse_document(metadata, rendered.split("---\n", 2)[2])
|
||||
)
|
||||
if restored != value:
|
||||
raise ValueError(
|
||||
f"{value.id}: content cannot be represented losslessly in editable Markdown"
|
||||
)
|
||||
return rendered
|
||||
@@ -0,0 +1,235 @@
|
||||
"""Explicit acquisition and durable source comparisons; never implicit runtime refresh."""
|
||||
|
||||
import base64
|
||||
import json
|
||||
|
||||
from .authoring import (
|
||||
RestructureRequest,
|
||||
_allocate_evidence_id,
|
||||
_candidate_to_evidence,
|
||||
normalize_source_text,
|
||||
)
|
||||
from .canonical import (
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
)
|
||||
from .corpus.normalize import _decode
|
||||
from .local_archive import ArchiveConflict, LocalEvidenceArchive, _atomic, _digest
|
||||
|
||||
MAX_DOCUMENTS = 200
|
||||
MAX_TOTAL_BYTES = 100 * 1024 * 1024
|
||||
|
||||
|
||||
def _origin(unit):
|
||||
return unit.provenance.original if isinstance(unit.provenance, ManualEvidenceProvenance) else unit.provenance
|
||||
|
||||
|
||||
def _records(archive):
|
||||
path = archive.metadata / "sources.json"
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source metadata must not use symlinks")
|
||||
return json.loads(path.read_text()) if path.exists() else {}
|
||||
|
||||
|
||||
def _write_records(archive, records):
|
||||
_atomic(archive.metadata / "sources.json", json.dumps(records, ensure_ascii=False, sort_keys=True))
|
||||
|
||||
|
||||
def reviews(archive):
|
||||
return [{k: v for k, v in row.items() if k not in {"expected", "text", "writes", "approved"}}
|
||||
for row in _records(archive).values()]
|
||||
|
||||
|
||||
def acquisition_sources(cfg):
|
||||
"""Local drafts and original files, plus configured read-only remote connectors."""
|
||||
from .adapters import FilesystemEvidenceSource
|
||||
from .sources import build_sources
|
||||
|
||||
root = cfg.evidence.local_archive_root / "evidence"
|
||||
# Canonical acquired versions are immutable lineage, never new input documents.
|
||||
patterns = [str(p.relative_to(root)) for folder in ("incoming", "source")
|
||||
for p in sorted((root / folder).rglob("*.md"))
|
||||
if not p.is_relative_to(root / "source/acquired")]
|
||||
result = [FilesystemEvidenceSource(root, patterns=patterns)] if patterns else []
|
||||
# Filesystem descriptors select the installation's local authoring tree after E2.
|
||||
remote = cfg.evidence.model_copy(update={"source_root": None,
|
||||
"sources": [s for s in cfg.evidence.sources if s.type != "filesystem"]})
|
||||
result.extend(build_sources(remote, acquisition=True))
|
||||
return result
|
||||
|
||||
|
||||
def refresh(archive: LocalEvidenceArchive, sources, restructurer):
|
||||
"""Acquire everything successfully before recording proposals. Missing is never deletion."""
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation before refreshing sources")
|
||||
records = _records(archive)
|
||||
if any(r["status"] == "applying" for r in records.values()):
|
||||
raise ArchiveConflict("Retry the pending source decision before refreshing")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
documents, total = {}, 0
|
||||
for adapter in sources:
|
||||
for item in adapter.discover():
|
||||
document = adapter.acquire(item)
|
||||
total += len(document.content)
|
||||
if len(documents) >= MAX_DOCUMENTS or total > MAX_TOTAL_BYTES:
|
||||
raise ValueError("Source refresh exceeds the local acquisition limit")
|
||||
relative = item.metadata.get("relative_path") if item.uri.startswith("file:") else None
|
||||
identity = f"local:{relative}" if relative else item.uri
|
||||
key = _digest(identity.encode())
|
||||
if key in documents:
|
||||
raise ValueError("Duplicate acquisition identity")
|
||||
documents[key] = (document, relative)
|
||||
reserved = set(units) | set(state["deleted_ids"])
|
||||
for record in records.values():
|
||||
reserved.update(u["id"] for u in record.get("proposed", []))
|
||||
changed, unchanged = 0, 0
|
||||
acquired = {}
|
||||
for key, (document, relative) in documents.items():
|
||||
text = normalize_source_text(_decode(document))
|
||||
sha = "sha256:" + _digest(text.encode())
|
||||
old = records.get(key)
|
||||
stale = old and old["status"] == "review" and any(
|
||||
p not in files or _digest(files[p]) != h for p, h in old["expected"].items())
|
||||
if old and old["sha256"] == sha and not stale:
|
||||
old["availability"] = "available"
|
||||
unchanged += 1
|
||||
continue
|
||||
current = {i: pair for i, pair in units.items()
|
||||
if (old and i in old["unit_ids"]) or
|
||||
(_origin(pair[1]) and _origin(pair[1]).source_file == relative)}
|
||||
# Seed imported E2 document identity without asking the model to recurate unchanged text.
|
||||
if old is None and current and all(_origin(u).source_sha256 == sha for _, u in current.values()):
|
||||
records[key] = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "unit_ids": sorted(current), "status": "accepted",
|
||||
"availability": "available", "revision": sha[7:], "proposed": []}
|
||||
unchanged += 1
|
||||
continue
|
||||
source_file = f"source/acquired/{key}/{sha[7:]}.md"
|
||||
request = RestructureRequest(source_file=source_file, source_sha256=sha,
|
||||
normalized_text=text, previous_units=tuple(u for _, u in current.values()))
|
||||
proposed = []
|
||||
seen = set()
|
||||
suppressed = relative in state["suppressed_sources"] or any(
|
||||
p.startswith(f"source/acquired/{key}/") for p in state["suppressed_sources"]) or (old and (
|
||||
old.get("suppressed", False) or any(i in state["deleted_ids"] for i in old["unit_ids"])))
|
||||
for candidate in restructurer.restructure(request):
|
||||
identity = candidate.existing_id
|
||||
if identity in state["deleted_ids"]:
|
||||
continue
|
||||
if identity is not None and identity not in current:
|
||||
raise ValueError("Source proposal refers to an unrelated Evidence identity")
|
||||
if identity is None:
|
||||
if suppressed:
|
||||
continue # A model-created identifier cannot bypass a curated deletion.
|
||||
identity = _allocate_evidence_id(candidate.title, reserved)
|
||||
reserved.add(identity)
|
||||
if identity in seen:
|
||||
raise ValueError("Source proposal repeats an Evidence identity")
|
||||
seen.add(identity)
|
||||
unit = _candidate_to_evidence(candidate, identity, source_file, sha)
|
||||
dump_curated_markdown(unit) # Refuse an uneditable proposal before saving any review.
|
||||
if any(excerpt not in text for excerpt in unit.provenance.supporting_excerpts):
|
||||
raise ValueError("Source proposal contains an excerpt absent from the acquired document")
|
||||
proposed.append(unit.model_dump(mode="json"))
|
||||
expected = {p: _digest(files[p]) for p, _ in current.values()}
|
||||
row = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "source_file": source_file, "text": text,
|
||||
"status": "review", "availability": "available", "suppressed": bool(suppressed),
|
||||
"unit_ids": sorted(current), "expected": expected,
|
||||
"current": [u.model_dump(mode="json") for _, u in current.values()], "proposed": proposed,
|
||||
"removed_ids": sorted(set(current) - seen)}
|
||||
row["revision"] = _digest(json.dumps(row, sort_keys=True).encode())
|
||||
records[key] = row
|
||||
acquired[key] = {"source": document.source.model_dump(mode="json"),
|
||||
"media_type": document.media_type, "raw_base64": base64.b64encode(document.content).decode()}
|
||||
changed += 1
|
||||
for key, row in records.items():
|
||||
if key not in documents:
|
||||
row["availability"] = "missing"
|
||||
# No writes above: an access/model failure preserves every previous review and active unit.
|
||||
for key, value in acquired.items():
|
||||
directory = archive.metadata / "acquisitions" / key
|
||||
if any(p.is_symlink() for p in [directory, directory.parent]):
|
||||
raise ValueError("Acquisition metadata must not use symlinks")
|
||||
_atomic(directory / f"{records[key]['sha256'][7:]}.json", json.dumps(value, ensure_ascii=False))
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"changed": changed, "unchanged": unchanged,
|
||||
"review": sum(r["status"] == "review" for r in records.values())}}
|
||||
|
||||
|
||||
def decide(archive, *, source_id, revision, decision, actor, activate):
|
||||
if decision not in {"keep", "replace"} or not actor.strip():
|
||||
raise ValueError("Choose keep or replace and supply a curator")
|
||||
with archive.operation():
|
||||
records = _records(archive)
|
||||
row = records.get(source_id)
|
||||
if row is None or row["revision"] != revision:
|
||||
raise ArchiveConflict("Source comparison changed; refresh the page")
|
||||
if row["status"] == "applying":
|
||||
if row["decision"] != decision:
|
||||
raise ArchiveConflict("Retry the saved source decision before changing it")
|
||||
state = archive._state()
|
||||
state.update(import_writes=row["writes"], approved_imports=row["approved"])
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
else:
|
||||
if row["status"] != "review":
|
||||
raise ArchiveConflict("This source comparison was already decided")
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation first")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
if any(_digest(files[p]) != h if p in files else True for p, h in row["expected"].items()):
|
||||
raise ArchiveConflict("Curated files changed since source review; refresh the source comparison")
|
||||
selected = [CuratedEvidence.model_validate(v) for v in row["proposed"]] if decision == "replace" else [units[i][1] for i in row["unit_ids"]]
|
||||
writes, approved = {}, {}
|
||||
def write(path, content):
|
||||
current = archive.evidence / path
|
||||
writes[path] = {"before": _digest(current.read_bytes()) if current.exists() else None, "after": content}
|
||||
if decision == "replace":
|
||||
write(row["source_file"], row["text"])
|
||||
for identity in row["unit_ids"]:
|
||||
write(units[identity][0], None)
|
||||
for value in selected:
|
||||
if value.review_items:
|
||||
raise ValueError("The proposal needs review; correct the input draft and refresh, or keep local content")
|
||||
if decision == "keep":
|
||||
origin = _origin(value)
|
||||
if origin:
|
||||
old = archive._snapshot(state["active"] or state["baseline"])
|
||||
candidate = old / origin.source_file
|
||||
if not candidate.is_file():
|
||||
candidate = archive.evidence / origin.source_file
|
||||
text = normalize_source_text(candidate.read_text())
|
||||
if "sha256:" + _digest(text.encode()) != origin.source_sha256:
|
||||
raise ValueError("The original source version is unavailable")
|
||||
path = f"source/acquired/{source_id}/{origin.source_sha256[7:]}.md"
|
||||
write(path, text)
|
||||
origin = EvidenceProvenance(source_file=path, source_sha256=origin.source_sha256,
|
||||
supporting_excerpts=origin.supporting_excerpts)
|
||||
value = value.model_copy(update={"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)})
|
||||
if value.id in units and value.id not in row["unit_ids"]:
|
||||
raise ArchiveConflict("A proposed Evidence identity was created elsewhere")
|
||||
path = units[value.id][0] if value.id in units and units[value.id][1].kind == value.kind else f"curated/{value.kind}/{value.id[9:]}.md"
|
||||
if path in files and value.id not in units:
|
||||
raise ArchiveConflict("The proposed file path is occupied")
|
||||
write(path, dump_curated_markdown(value))
|
||||
approved[value.id] = _digest(dump_curated_markdown(value).encode())
|
||||
# Journal before applying files; retry never silently clobbers an external edit.
|
||||
row.update(status="applying", decision=decision, decided_by=actor,
|
||||
next_unit_ids=[u.id for u in selected], writes=writes, approved=approved)
|
||||
_write_records(archive, records)
|
||||
state.update(import_writes=writes, approved_imports=approved)
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
if result["status"] != "active":
|
||||
raise ValueError("Source decision was saved but not activated; retry")
|
||||
row.update(status="accepted" if decision == "replace" else "kept", unit_ids=row["next_unit_ids"])
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"units": result["units"]}}
|
||||
@@ -0,0 +1,472 @@
|
||||
"""Persistent curated files, immutable consolidation candidates and explicit activation.
|
||||
|
||||
The working tree is primary data. Core consumers use only active_snapshot(); an
|
||||
editor save or failed activation never switches that pointer. No Git/network/DWH I/O.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import fcntl
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
MAX_CURATED_FILE_BYTES,
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
parse_curated_markdown,
|
||||
)
|
||||
|
||||
|
||||
class ArchiveConflict(ValueError):
|
||||
"""The curator must reconcile a concurrent change before replacing it."""
|
||||
|
||||
|
||||
def _digest(value: bytes) -> str:
|
||||
return hashlib.sha256(value).hexdigest()
|
||||
|
||||
|
||||
def _content(unit: CuratedEvidence) -> str:
|
||||
return _digest(
|
||||
json.dumps(
|
||||
unit.model_dump(mode="json", exclude={"schema_version", "provenance"}),
|
||||
sort_keys=True,
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
)
|
||||
|
||||
|
||||
def _atomic(path: Path, data: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w") as handle:
|
||||
handle.write(data)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
os.replace(temporary, path)
|
||||
finally:
|
||||
if os.path.exists(temporary):
|
||||
os.unlink(temporary)
|
||||
|
||||
|
||||
class LocalEvidenceArchive:
|
||||
def __init__(self, workspace_root: Path):
|
||||
self.root = workspace_root.resolve()
|
||||
self.evidence = self.root / "evidence"
|
||||
self.metadata = self.evidence / ".local"
|
||||
if self.evidence.is_symlink() or self.metadata.is_symlink():
|
||||
raise ValueError("The local Evidence archive must use persistent regular directories")
|
||||
|
||||
@contextmanager
|
||||
def operation(self):
|
||||
if any(
|
||||
path.is_symlink()
|
||||
for path in (
|
||||
self.evidence,
|
||||
self.metadata,
|
||||
self.metadata / "snapshots",
|
||||
self.metadata / "state.yaml",
|
||||
)
|
||||
):
|
||||
raise ValueError("Evidence archive metadata must not use symlinks")
|
||||
self.root.mkdir(parents=True, exist_ok=True)
|
||||
lock = self.root / ".evidence-archive.lock"
|
||||
if lock.is_symlink():
|
||||
raise ValueError("Evidence lock must not be a symlink")
|
||||
with lock.open("a") as handle:
|
||||
fcntl.flock(handle, fcntl.LOCK_EX)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
fcntl.flock(handle, fcntl.LOCK_UN)
|
||||
|
||||
def _state(self):
|
||||
path = self.metadata / "state.yaml"
|
||||
if not path.exists():
|
||||
return {
|
||||
"schema_version": 1,
|
||||
"active": None,
|
||||
"pending": None,
|
||||
"baseline": None,
|
||||
"deleted_ids": [],
|
||||
"suppressed_sources": [],
|
||||
}
|
||||
value = yaml.safe_load(path.read_text())
|
||||
if not isinstance(value, dict) or value.get("schema_version") != 1:
|
||||
raise ValueError("Unsupported local Evidence archive state")
|
||||
return value
|
||||
|
||||
def _write_state(self, state):
|
||||
_atomic(self.metadata / "state.yaml", yaml.safe_dump(state, sort_keys=True))
|
||||
|
||||
def _snapshot(self, revision: str) -> Path:
|
||||
if (
|
||||
not isinstance(revision, str)
|
||||
or len(revision) != 64
|
||||
or any(c not in "0123456789abcdef" for c in revision)
|
||||
):
|
||||
raise ValueError("Invalid Evidence snapshot identity")
|
||||
path = self.metadata / "snapshots" / revision
|
||||
if not path.is_dir() or path.is_symlink():
|
||||
raise ValueError("Consolidated Evidence snapshot is missing")
|
||||
return path
|
||||
|
||||
def active_snapshot(self) -> Path | None:
|
||||
"""Only this immutable source is eligible for core consumption."""
|
||||
with self.operation():
|
||||
revision = self._state()["active"]
|
||||
return self._snapshot(revision) if revision else None
|
||||
|
||||
def validate(self):
|
||||
"""Check the working files without modifying them or changing active content."""
|
||||
with self.operation():
|
||||
units = self._units(self._files(self.evidence))
|
||||
state = self._state()
|
||||
revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous = (
|
||||
self._units(self._files(self._snapshot(revision)), allow_review=True)
|
||||
if revision
|
||||
else {}
|
||||
)
|
||||
for identity, (_, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
if isinstance(unit.provenance, EvidenceProvenance) and (
|
||||
old is None or _content(old[1]) == _content(unit)
|
||||
):
|
||||
self._validate_document_source(unit.provenance)
|
||||
return len(units)
|
||||
|
||||
def _files(self, root: Path):
|
||||
result = {}
|
||||
curated = root / "curated"
|
||||
if curated.is_symlink():
|
||||
raise ValueError("Curated directory must not be a symlink")
|
||||
if not curated.is_dir():
|
||||
raise ValueError(f"{curated}: curated archive is unavailable; absence is not deletion")
|
||||
for path in sorted(curated.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError(f"{path}: symlinks are not supported in the curated archive")
|
||||
if path.suffix != ".md" or path.name.upper().startswith("README"):
|
||||
continue
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: Evidence file exceeds the size limit")
|
||||
result[path.relative_to(root).as_posix()] = path.read_bytes()
|
||||
return result
|
||||
|
||||
def _units(self, files, *, allow_review=False):
|
||||
units = {}
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode("utf-8"), path=Path(relative))
|
||||
if unit.schema_version != 4:
|
||||
raise ValueError("Run evidence migrate before consolidating legacy units")
|
||||
if unit.id in units:
|
||||
raise ValueError(f"Duplicate Evidence identity {unit.id}")
|
||||
if unit.review_items and not allow_review:
|
||||
raise ValueError("Resolve review items before consolidation")
|
||||
units[unit.id] = (relative, unit)
|
||||
except ValueError as error:
|
||||
raise ValueError(f"{relative}: {error}") from error
|
||||
return units
|
||||
|
||||
def initialize(self):
|
||||
"""Capture migrated/refined content before edits, without activating review items."""
|
||||
with self.operation():
|
||||
state = self._state()
|
||||
if state.get("baseline") or state["active"] or state["pending"]:
|
||||
return
|
||||
files = self._files(self.evidence)
|
||||
self._units(files, allow_review=True)
|
||||
source = self.evidence / "source"
|
||||
if source.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if source.exists():
|
||||
for path in sorted(source.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if path.is_file():
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: source exceeds the size limit")
|
||||
files[path.relative_to(self.evidence).as_posix()] = path.read_bytes()
|
||||
revision = _digest(b"".join(k.encode() + b"\0" + v for k, v in sorted(files.items())))
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".baseline-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
state["baseline"] = revision
|
||||
self._write_state(state)
|
||||
|
||||
def consolidate(self, *, actor: str, activate: Callable[[Path], None] | None = None):
|
||||
"""Persist one valid candidate; optionally activate it through the existing index stage.
|
||||
|
||||
A missing callback deliberately reports pending_activation, never success.
|
||||
A retry of unchanged files reuses the same candidate after an index failure.
|
||||
"""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._consolidate(actor, activate)
|
||||
|
||||
def _consolidate(self, actor, activate):
|
||||
state = self._state()
|
||||
self._finish_import(state)
|
||||
self._finish_normalization(state)
|
||||
original_files = self._files(self.evidence)
|
||||
units = self._units(original_files)
|
||||
previous_revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous_root = self._snapshot(previous_revision) if previous_revision else None
|
||||
previous = (
|
||||
self._units(self._files(previous_root), allow_review=True) if previous_root else {}
|
||||
)
|
||||
deleted = set(state["deleted_ids"]) | (previous.keys() - units.keys())
|
||||
suppressed = set(state["suppressed_sources"])
|
||||
for identity in previous.keys() - units.keys():
|
||||
provenance = previous[identity][1].provenance
|
||||
source = (
|
||||
provenance.original
|
||||
if isinstance(provenance, ManualEvidenceProvenance)
|
||||
else provenance
|
||||
)
|
||||
if source:
|
||||
suppressed.add(source.source_file)
|
||||
files = {}
|
||||
for identity, (relative, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
changed = old is not None and _content(old[1]) != _content(unit)
|
||||
approved = state.get("approved_imports", {}).get(identity) == _digest(
|
||||
dump_curated_markdown(unit).encode()
|
||||
)
|
||||
if approved:
|
||||
pass # An explicit source decision authorized this exact content and provenance.
|
||||
elif changed or (
|
||||
isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
and (old is None or unit.provenance.declared_by == "local curator")
|
||||
):
|
||||
origin = old[1].provenance if old else unit.provenance
|
||||
origin = origin.original if isinstance(origin, ManualEvidenceProvenance) else origin
|
||||
unit = unit.model_copy(
|
||||
update={
|
||||
"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)
|
||||
}
|
||||
)
|
||||
elif old and unit.provenance != old[1].provenance:
|
||||
raise ArchiveConflict(
|
||||
f"{relative}: provenance is managed; edit the content instead"
|
||||
)
|
||||
if isinstance(unit.provenance, EvidenceProvenance):
|
||||
self._validate_document_source(unit.provenance)
|
||||
files[relative] = dump_curated_markdown(unit).encode()
|
||||
origin = (
|
||||
unit.provenance.original
|
||||
if isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
else unit.provenance
|
||||
)
|
||||
if origin:
|
||||
candidates = [previous_root / origin.source_file] if previous_root else []
|
||||
candidates.append(self.evidence / origin.source_file)
|
||||
from .authoring import normalize_source_text
|
||||
|
||||
for source in candidates:
|
||||
if source.is_file() and not source.is_symlink():
|
||||
if source.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{source}: source exceeds the size limit")
|
||||
raw = source.read_bytes()
|
||||
if (
|
||||
"sha256:" + _digest(normalize_source_text(raw.decode()).encode())
|
||||
== origin.source_sha256
|
||||
):
|
||||
if origin.source_file in files and files[origin.source_file] != raw:
|
||||
raise ArchiveConflict(
|
||||
"Different source revisions require explicit source resolution"
|
||||
)
|
||||
files[origin.source_file] = raw
|
||||
break
|
||||
else:
|
||||
raise ValueError(
|
||||
f"{origin.source_file}: the recorded original document is unavailable"
|
||||
)
|
||||
# Record current declarations and original document lineage distinctly.
|
||||
manifest = {
|
||||
"schema_version": 1,
|
||||
"units": {
|
||||
identity: {
|
||||
"file": relative,
|
||||
"content_hash": _content(parse_curated_markdown(files[relative].decode())),
|
||||
"provenance": parse_curated_markdown(
|
||||
files[relative].decode()
|
||||
).provenance.model_dump(mode="json"),
|
||||
}
|
||||
for identity, (relative, _) in sorted(units.items())
|
||||
},
|
||||
"deleted_ids": sorted(deleted),
|
||||
"suppressed_sources": sorted(suppressed),
|
||||
}
|
||||
files["local-manifest.yaml"] = yaml.safe_dump(
|
||||
manifest, allow_unicode=True, sort_keys=True
|
||||
).encode()
|
||||
revision = _digest(
|
||||
b"".join(path.encode() + b"\0" + data + b"\0" for path, data in sorted(files.items()))
|
||||
)
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".candidate-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
if self._files(self.evidence) != original_files:
|
||||
raise ArchiveConflict(
|
||||
"Evidence files changed during consolidation; retry with the current files"
|
||||
)
|
||||
# The candidate exists first. Pending state makes every subsequent interruption recoverable.
|
||||
state.update(
|
||||
pending=revision,
|
||||
deleted_ids=sorted(deleted),
|
||||
suppressed_sources=sorted(suppressed),
|
||||
normalization={relative: _digest(data) for relative, data in original_files.items()},
|
||||
)
|
||||
state.pop("approved_imports", None)
|
||||
self._write_state(state)
|
||||
self._finish_normalization(state)
|
||||
if activate is None:
|
||||
return {
|
||||
"status": "pending_activation",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
}
|
||||
activate(destination)
|
||||
state.update(active=revision, pending=None)
|
||||
self._write_state(state)
|
||||
return {
|
||||
"status": "active",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
"units": len(units),
|
||||
"deleted": len(previous.keys() - units.keys()),
|
||||
}
|
||||
|
||||
def _finish_import(self, state):
|
||||
"""Replay an explicit source decision, rejecting intervening external edits."""
|
||||
writes = state.get("import_writes")
|
||||
if writes is None:
|
||||
return
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if not relative.startswith(("curated/", "source/acquired/")) or ".." in Path(relative).parts:
|
||||
raise ValueError("Invalid import destination")
|
||||
if any(p.is_symlink() for p in [path, *path.parents] if p != self.root.parent):
|
||||
raise ValueError("Import destinations must not use symlinks")
|
||||
current = _digest(path.read_bytes()) if path.exists() else None
|
||||
after = _digest(change["after"].encode()) if change["after"] is not None else None
|
||||
if current not in (change["before"], after):
|
||||
raise ArchiveConflict("Evidence changed during a source decision; restore or review the file")
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if change["after"] is None:
|
||||
path.unlink(missing_ok=True)
|
||||
else:
|
||||
_atomic(path, change["after"])
|
||||
del state["import_writes"]
|
||||
self._write_state(state)
|
||||
|
||||
def _finish_normalization(self, state):
|
||||
"""Replay interrupted managed writes only where the user's bytes are unchanged."""
|
||||
if "normalization" not in state:
|
||||
return
|
||||
snapshot = self._snapshot(state["pending"])
|
||||
current = self._files(self.evidence)
|
||||
for relative, original_hash in state["normalization"].items():
|
||||
if relative in current and _digest(current[relative]) == original_hash:
|
||||
_atomic(self.evidence / relative, (snapshot / relative).read_text())
|
||||
_atomic(
|
||||
self.evidence / "local-manifest.yaml", (snapshot / "local-manifest.yaml").read_text()
|
||||
)
|
||||
del state["normalization"]
|
||||
self._write_state(state)
|
||||
|
||||
def _validate_document_source(self, provenance):
|
||||
from .authoring import _normalize, normalize_source_text
|
||||
|
||||
path = self.evidence / provenance.source_file
|
||||
if (
|
||||
path.is_symlink()
|
||||
or not path.is_file()
|
||||
or not path.resolve().is_relative_to(self.evidence.resolve())
|
||||
):
|
||||
raise ValueError(f"{provenance.source_file}: source document is unavailable")
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{provenance.source_file}: source exceeds the size limit")
|
||||
source = normalize_source_text(path.read_text())
|
||||
if "sha256:" + _digest(source.encode()) != provenance.source_sha256:
|
||||
raise ArchiveConflict(
|
||||
f"{provenance.source_file}: source changed; explicit source refresh is required"
|
||||
)
|
||||
if any(_normalize(excerpt) not in source for excerpt in provenance.supporting_excerpts):
|
||||
raise ValueError(f"{provenance.source_file}: supporting excerpt is missing")
|
||||
|
||||
def save(self, value: CuratedEvidence, *, expected_revision: str | None, actor: str):
|
||||
"""Explicit workflow correction with optimistic concurrency; activation is a separate boundary."""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._save(value, expected_revision=expected_revision, actor=actor)
|
||||
|
||||
def _save(self, value, *, expected_revision, actor):
|
||||
"""Save while the caller holds operation(), including workflow receipt recovery."""
|
||||
self._finish_normalization(self._state())
|
||||
units = self._units(self._files(self.evidence))
|
||||
existing = units.get(value.id)
|
||||
if (existing is None) != (expected_revision is None):
|
||||
raise ArchiveConflict("Evidence was created or removed since review")
|
||||
if existing and _content(existing[1]) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
relative = (
|
||||
existing[0]
|
||||
if existing
|
||||
else f"curated/{value.kind}/{value.id.removeprefix('evidence:')}.md"
|
||||
)
|
||||
if existing and existing[1].kind != value.kind:
|
||||
raise ValueError("An update cannot change the Evidence kind")
|
||||
if existing:
|
||||
value = value.model_copy(update={"provenance": existing[1].provenance})
|
||||
_atomic(self.evidence / relative, dump_curated_markdown(value))
|
||||
return self._consolidate(actor, None)
|
||||
|
||||
def get(self, identity: str):
|
||||
with self.operation():
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
return {"unit": unit, "revision": _content(unit), "path": str(self.evidence / relative)}
|
||||
|
||||
def remove(self, identity: str, *, expected_revision: str, actor: str):
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
self._finish_normalization(self._state())
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
if _content(unit) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
(self.evidence / relative).unlink()
|
||||
return self._consolidate(actor, None)
|
||||
@@ -12,10 +12,18 @@ if TYPE_CHECKING:
|
||||
from tht.config import EvidenceSourcesConfig
|
||||
|
||||
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None) -> list[EvidenceSource]:
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None, *, acquisition: bool = False) -> list[EvidenceSource]:
|
||||
"""Build configured Evidence adapters in the existing deterministic order."""
|
||||
if evidence is None:
|
||||
return []
|
||||
if evidence.local_archive_root is not None and not acquisition:
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive
|
||||
archive = LocalEvidenceArchive(evidence.local_archive_root)
|
||||
if (archive.metadata / "state.yaml").exists():
|
||||
snapshot = archive.active_snapshot()
|
||||
if snapshot is None:
|
||||
raise ValueError("Local Evidence has not been activated; run Evidence consolidation")
|
||||
return [FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))]
|
||||
sources: list[EvidenceSource] = []
|
||||
if evidence.source_root is not None:
|
||||
legacy_root = evidence.source_root / evidence.evidence_dir
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Apply only removals confirmed by a successful Catalog physical synchronization."""
|
||||
|
||||
import json
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from .models import MemoryConflict
|
||||
from .review import digest
|
||||
|
||||
|
||||
class RemovedColumn(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
table: str = Field(min_length=1, max_length=200)
|
||||
column: str = Field(min_length=1, max_length=200)
|
||||
|
||||
|
||||
class CleanupRequest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
sync_id: str = Field(min_length=1, max_length=200)
|
||||
database: str = Field(min_length=1, max_length=200)
|
||||
schema_name: str = Field(min_length=1, max_length=200)
|
||||
removed_tables: list[str] = Field(default_factory=list, max_length=10000)
|
||||
removed_columns: list[RemovedColumn] = Field(default_factory=list, max_length=100000)
|
||||
|
||||
|
||||
def cleanup(service, request: CleanupRequest):
|
||||
service._admin()
|
||||
fingerprint = digest(request.model_dump(mode="json"))
|
||||
with service.repository.operation() as repo:
|
||||
with repo.transaction() as connection:
|
||||
params = {"w": repo.workspace_id, "sync": request.sync_id}
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT request_hash,card_ids "
|
||||
"FROM thoth_memory.cleanup_receipts WHERE workspace_id=:w AND sync_id=:sync"
|
||||
),
|
||||
params,
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
if receipt["request_hash"] != fingerprint:
|
||||
raise MemoryConflict(
|
||||
"Physical cleanup identity was reused with different removals"
|
||||
)
|
||||
identities = receipt["card_ids"]
|
||||
else:
|
||||
identities = list(
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT DISTINCT card_id "
|
||||
"FROM thoth_memory.dependencies d WHERE workspace_id=:w "
|
||||
"AND database_id=:db AND schema_name=:schema AND (table_name=ANY(:tables) "
|
||||
"OR EXISTS (SELECT 1 FROM jsonb_to_recordset(CAST(:columns AS jsonb)) "
|
||||
'AS removed("table" text, "column" text) WHERE '
|
||||
'removed."table"=d.table_name AND removed."column"=d.column_name))'
|
||||
),
|
||||
{
|
||||
**params,
|
||||
"db": request.database,
|
||||
"schema": request.schema_name,
|
||||
"tables": request.removed_tables,
|
||||
"columns": json.dumps(
|
||||
[c.model_dump() for c in request.removed_columns]
|
||||
),
|
||||
},
|
||||
).scalars()
|
||||
)
|
||||
for identity in identities:
|
||||
repo.delete(identity)
|
||||
connection.execute(
|
||||
text(
|
||||
"INSERT INTO thoth_memory.cleanup_receipts "
|
||||
"VALUES (:w,:sync,:hash,CAST(:ids AS jsonb))"
|
||||
),
|
||||
{**params, "hash": fingerprint, "ids": json.dumps(identities)},
|
||||
)
|
||||
results = [service._propagate(repo, identity) for identity in identities]
|
||||
return {"deleted": len(identities), "indexed": all(r["indexed"] for r in results)}
|
||||
@@ -25,7 +25,7 @@ class MemoryRecord(BaseModel):
|
||||
concepts: list[str] = []
|
||||
|
||||
|
||||
_MEM_ID_RE = re.compile(r"\bmem-\d{4,}\b")
|
||||
_MEM_ID_RE = re.compile(r"\bmem-(?:[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}|\d{4,})\b")
|
||||
|
||||
|
||||
def decided_memory_ids(decisions: list[DecisionRecord]) -> set[str]:
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Versioned Memory migration pack, run only by installation preparation."""
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
from functools import lru_cache
|
||||
from importlib.resources import files
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import URL, create_engine, text
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
|
||||
def installation_url(*, migrator: bool = False) -> str:
|
||||
role = "MIGRATOR" if migrator else "RUNTIME"
|
||||
direct = os.environ.get(f"THT_CATALOG_{role}_DATABASE_URL")
|
||||
if not migrator:
|
||||
direct = direct or os.environ.get("THT_CATALOG_DATABASE_URL")
|
||||
if direct:
|
||||
return direct.replace("postgresql://", "postgresql+psycopg2://", 1)
|
||||
prefix = "THT_CATALOG_"
|
||||
try:
|
||||
password = Path(os.environ[prefix + role + "_PASSWORD_FILE"]).read_text().strip()
|
||||
url = URL.create(
|
||||
"postgresql+psycopg2", host=os.environ[prefix + "DB_HOST"],
|
||||
port=int(os.environ.get(prefix + "DB_PORT", "5432")),
|
||||
database=os.environ[prefix + "DB_NAME"],
|
||||
username=os.environ[prefix + role + "_USER"], password=password,
|
||||
)
|
||||
return url.render_as_string(hide_password=False)
|
||||
except (KeyError, OSError, ValueError):
|
||||
raise ValueError("Memory PostgreSQL installation configuration is unavailable") from None
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def expected_migrations() -> dict[str, str]:
|
||||
return {p.name: hashlib.sha256(p.read_text().encode()).hexdigest()
|
||||
for p in files("tht").joinpath("migrations/memory").iterdir()
|
||||
if p.name.endswith(".sql")}
|
||||
|
||||
|
||||
def migrate(database_url: str) -> None:
|
||||
engine = create_engine(database_url, poolclass=NullPool)
|
||||
try:
|
||||
with engine.begin() as connection:
|
||||
connection.execute(text("SELECT pg_advisory_xact_lock(792114203)"))
|
||||
connection.execute(text("CREATE SCHEMA IF NOT EXISTS thoth_memory"))
|
||||
connection.execute(text("CREATE TABLE IF NOT EXISTS thoth_memory.migrations "
|
||||
"(version text PRIMARY KEY, checksum text NOT NULL)"))
|
||||
applied = dict(connection.execute(text(
|
||||
"SELECT version, checksum FROM thoth_memory.migrations"
|
||||
)).all())
|
||||
pack = sorted(files("tht").joinpath("migrations/memory").iterdir(), key=lambda p: p.name)
|
||||
known = {p.name for p in pack if p.name.endswith(".sql")}
|
||||
if set(applied) - known:
|
||||
raise ValueError("Memory schema is newer than this application")
|
||||
for path in pack:
|
||||
if path.name not in known:
|
||||
continue
|
||||
sql = path.read_text()
|
||||
digest = hashlib.sha256(sql.encode()).hexdigest()
|
||||
if path.name in applied:
|
||||
if applied[path.name] != digest:
|
||||
raise ValueError("Memory migration checksum mismatch")
|
||||
continue
|
||||
connection.execute(text(sql))
|
||||
connection.execute(text("INSERT INTO thoth_memory.migrations VALUES (:v, :c)"),
|
||||
{"v": path.name, "c": digest})
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
migrate(installation_url(migrator=True))
|
||||
print("Memory migrations: ready")
|
||||
@@ -0,0 +1,112 @@
|
||||
"""Authoritative Memory contracts, independent of workflow decision kinds."""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Literal, Self
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
Family = Literal["domain_clarification", "sql_rule", "solved_question", "explained_error"]
|
||||
|
||||
|
||||
class Dependency(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
database: str = Field(min_length=1, max_length=200)
|
||||
schema_name: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def structured(self) -> Self:
|
||||
if self.column and not self.table:
|
||||
raise ValueError("A column dependency requires a table")
|
||||
if self.table and not self.schema_name:
|
||||
raise ValueError("A table dependency requires a schema")
|
||||
return self
|
||||
|
||||
|
||||
class LinkInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
target_id: str = Field(min_length=1, max_length=100)
|
||||
meaning: str = Field(min_length=1, max_length=1000)
|
||||
|
||||
|
||||
class CardInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
family: Family
|
||||
subject: str = Field(min_length=1, max_length=1000)
|
||||
detail: str = Field(default="", max_length=50000)
|
||||
scope: str = Field(min_length=1, max_length=10000)
|
||||
rationale: str = Field(default="", max_length=10000)
|
||||
question: str = Field(default="", max_length=10000)
|
||||
sql: str = Field(default="", max_length=100000)
|
||||
concepts: list[str] = Field(default_factory=list, max_length=100)
|
||||
dependencies: list[Dependency] = Field(default_factory=list, max_length=200)
|
||||
links: list[LinkInput] = Field(default_factory=list, max_length=200)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def valid_family(self) -> Self:
|
||||
if self.family == "solved_question" and (not self.question or not self.sql):
|
||||
raise ValueError("A solved question requires its question and approved SQL")
|
||||
if self.family == "explained_error" and (not self.detail or not self.rationale):
|
||||
raise ValueError("An explained error requires a correction and rationale")
|
||||
if self.family != "solved_question" and self.sql:
|
||||
raise ValueError("Only solved questions carry exemplar SQL")
|
||||
if any(not c.strip() or len(c) > 200 for c in self.concepts):
|
||||
raise ValueError("Concepts must contain between 1 and 200 characters")
|
||||
if len({link.target_id for link in self.links}) != len(self.links):
|
||||
raise ValueError("Each linked destination must be unique")
|
||||
return self
|
||||
|
||||
|
||||
class Card(CardInput):
|
||||
id: str
|
||||
workspace_id: str
|
||||
origin: Literal["manual", "workflow"]
|
||||
session_id: str | None = None
|
||||
decision_seq: int | None = None
|
||||
created_at: datetime
|
||||
updated_at: datetime
|
||||
revision: str
|
||||
indexed: bool = False
|
||||
|
||||
|
||||
class CardQuery(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
q: str = Field(default="", max_length=1000)
|
||||
family: Family | None = None
|
||||
concept: str = Field(default="", max_length=200)
|
||||
database: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
origin: Literal["manual", "workflow"] | None = None
|
||||
updated_after: datetime | None = None
|
||||
updated_before: datetime | None = None
|
||||
page: int = Field(default=1, ge=1)
|
||||
page_size: int = Field(default=25, ge=1, le=100)
|
||||
sort: Literal["updated_at", "created_at", "subject", "family"] = "updated_at"
|
||||
direction: Literal["asc", "desc"] = "desc"
|
||||
|
||||
|
||||
class MemoryError(Exception):
|
||||
code = "memory_operation_failed"
|
||||
status = 500
|
||||
|
||||
|
||||
class MemoryUnavailable(MemoryError):
|
||||
code = "memory_unavailable"
|
||||
status = 503
|
||||
|
||||
|
||||
class MemoryNotFound(MemoryError):
|
||||
code = "memory_not_found"
|
||||
status = 404
|
||||
|
||||
|
||||
class MemoryForbidden(MemoryError):
|
||||
code = "memory_forbidden"
|
||||
status = 403
|
||||
|
||||
|
||||
class MemoryConflict(MemoryError):
|
||||
code = "memory_conflict"
|
||||
status = 409
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Workspace-scoped PostgreSQL persistence and durable projection work."""
|
||||
|
||||
import json
|
||||
from contextlib import contextmanager
|
||||
from uuid import uuid4
|
||||
|
||||
from sqlalchemy import create_engine, text
|
||||
from sqlalchemy.exc import IntegrityError, SQLAlchemyError
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
from .migrate import expected_migrations
|
||||
from .models import Card, CardInput, CardQuery, MemoryConflict, MemoryNotFound, MemoryUnavailable
|
||||
|
||||
|
||||
class MemoryRepository:
|
||||
def __init__(self, database_url: str, workspace_id: str, *, engine=None, connection=None):
|
||||
self.workspace_id = workspace_id
|
||||
self.engine = engine or create_engine(
|
||||
database_url, poolclass=NullPool, connect_args={"connect_timeout": 5},
|
||||
)
|
||||
self.connection = connection
|
||||
|
||||
def close(self):
|
||||
self.engine.dispose()
|
||||
|
||||
@contextmanager
|
||||
def transaction(self):
|
||||
connection = self.connection
|
||||
try:
|
||||
connection = connection or self.engine.connect()
|
||||
with (connection.begin_nested() if connection.in_transaction() else connection.begin()):
|
||||
connection.execute(text("SET LOCAL ROLE thoth_memory_runtime"))
|
||||
connection.execute(text("SELECT set_config('thoth.memory_workspace', :w, true)"),
|
||||
{"w": self.workspace_id})
|
||||
connection.execute(text("SET LOCAL statement_timeout = '15s'"))
|
||||
installed = dict(connection.execute(text(
|
||||
"SELECT version, checksum FROM thoth_memory.migrations"
|
||||
)).all())
|
||||
if installed != expected_migrations():
|
||||
raise MemoryUnavailable("Memory schema is incompatible; run installation migrations")
|
||||
yield connection
|
||||
except IntegrityError:
|
||||
raise MemoryConflict("Memory references conflict with the current archive") from None
|
||||
except SQLAlchemyError:
|
||||
raise MemoryUnavailable("Memory archive is unavailable; check its migrations and access") \
|
||||
from None
|
||||
finally:
|
||||
if self.connection is None and connection is not None:
|
||||
connection.close()
|
||||
|
||||
@contextmanager
|
||||
def operation(self):
|
||||
"""Serialize each workspace across SQL commits and the bounded vector call."""
|
||||
try:
|
||||
connection = self.engine.connect()
|
||||
connection.execute(text("SET statement_timeout = '15s'"))
|
||||
connection.execute(text("SELECT pg_advisory_lock(hashtextextended(:w, 792114204))"),
|
||||
{"w": self.workspace_id})
|
||||
connection.commit()
|
||||
except SQLAlchemyError:
|
||||
if 'connection' in locals():
|
||||
connection.close()
|
||||
raise MemoryUnavailable("Memory archive is busy or unavailable") from None
|
||||
try:
|
||||
yield MemoryRepository("", self.workspace_id, engine=self.engine, connection=connection)
|
||||
finally:
|
||||
# NullPool closes the physical connection, releasing the session advisory lock.
|
||||
connection.close()
|
||||
|
||||
def _card(self, connection, row) -> Card:
|
||||
params = {"w": self.workspace_id, "id": row["id"]}
|
||||
links = connection.execute(text(
|
||||
"SELECT target_id, meaning FROM thoth_memory.links "
|
||||
"WHERE workspace_id=:w AND source_id=:id ORDER BY target_id"
|
||||
), params).mappings().all()
|
||||
dependencies = connection.execute(text(
|
||||
'SELECT database_id AS database, schema_name, table_name AS "table", '
|
||||
'column_name AS "column" FROM thoth_memory.dependencies '
|
||||
"WHERE workspace_id=:w AND card_id=:id "
|
||||
"ORDER BY database_id, schema_name, table_name, column_name"
|
||||
), params).mappings().all()
|
||||
return Card.model_validate({
|
||||
**row["data"], "id": row["id"], "workspace_id": self.workspace_id,
|
||||
"family": row["family"], "subject": row["subject"], "origin": row["origin"],
|
||||
"created_at": row["created_at"], "updated_at": row["updated_at"],
|
||||
"revision": row["revision"], "indexed": row["indexed"],
|
||||
"links": [dict(v) for v in links], "dependencies": [dict(v) for v in dependencies],
|
||||
})
|
||||
|
||||
@staticmethod
|
||||
def _selection():
|
||||
return ("SELECT c.*, (p.revision=c.revision AND NOT p.pending "
|
||||
"AND p.action='upsert' AND p.format=2) AS indexed FROM thoth_memory.cards c "
|
||||
"JOIN thoth_memory.projections p ON p.workspace_id=c.workspace_id "
|
||||
"AND p.card_id=c.id ")
|
||||
|
||||
def get(self, card_id: str) -> Card:
|
||||
with self.transaction() as c:
|
||||
row = c.execute(text(self._selection()+"WHERE c.workspace_id=:w AND c.id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).mappings().first()
|
||||
if row is None:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
return self._card(c, row)
|
||||
|
||||
def list(self, query: CardQuery) -> dict:
|
||||
conditions = ["c.workspace_id=:w"]
|
||||
params = {"w": self.workspace_id}
|
||||
if query.q:
|
||||
conditions.append("(c.id ILIKE :q OR c.subject ILIKE :q OR c.data::text ILIKE :q)")
|
||||
params["q"] = "%" + query.q.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + "%"
|
||||
for key in ("family", "origin"):
|
||||
if value := getattr(query, key):
|
||||
conditions.append(f"c.{key}=:{key}")
|
||||
params[key] = value
|
||||
if query.concept:
|
||||
conditions.append("c.data->'concepts' @> CAST(:concept AS jsonb)")
|
||||
params["concept"] = json.dumps([query.concept])
|
||||
refs = []
|
||||
for key, column in [("database", "database_id"), ("table", "table_name"),
|
||||
("column", "column_name")]:
|
||||
if value := getattr(query, key):
|
||||
refs.append(f"d.{column}=:{key}")
|
||||
params[key] = value
|
||||
if refs:
|
||||
conditions.append("EXISTS (SELECT 1 FROM thoth_memory.dependencies d WHERE "
|
||||
"d.workspace_id=c.workspace_id AND d.card_id=c.id AND "
|
||||
+ " AND ".join(refs) + ")")
|
||||
for key, comparison in [("updated_after", ">="), ("updated_before", "<=")]:
|
||||
if value := getattr(query, key):
|
||||
conditions.append(f"c.updated_at {comparison} :{key}")
|
||||
params[key] = value
|
||||
where = " WHERE " + " AND ".join(conditions)
|
||||
with self.transaction() as c:
|
||||
total = c.execute(text("SELECT count(*) FROM thoth_memory.cards c" + where),
|
||||
params).scalar_one()
|
||||
rows = c.execute(text(self._selection() + where
|
||||
+ f" ORDER BY c.{query.sort} {query.direction}, c.id ASC LIMIT :limit OFFSET :offset"),
|
||||
{**params, "limit": query.page_size, "offset": (query.page - 1) * query.page_size},
|
||||
).mappings().all()
|
||||
return {"items": [self._card(c, row).model_dump(mode="json") for row in rows],
|
||||
"total": total, "page": query.page, "page_size": query.page_size}
|
||||
|
||||
def exact_match(self, value: CardInput) -> Card | None:
|
||||
"""Match authored content only; provenance and index state do not create new knowledge."""
|
||||
data = value.model_dump(mode="json", exclude={"dependencies", "links"})
|
||||
with self.transaction() as c:
|
||||
rows = c.execute(text(self._selection() +
|
||||
"WHERE c.workspace_id=:w AND c.family=:family AND c.subject=:subject "
|
||||
"AND c.data - 'session_id' - 'decision_seq'=CAST(:data AS jsonb) ORDER BY c.id"),
|
||||
{"w": self.workspace_id, "family": value.family, "subject": value.subject,
|
||||
"data": json.dumps(data)}).mappings()
|
||||
for row in rows:
|
||||
candidate = self._card(c, row)
|
||||
def ordered(values):
|
||||
return sorted(json.dumps(v.model_dump(), sort_keys=True) for v in values)
|
||||
if (ordered(candidate.dependencies) == ordered(value.dependencies)
|
||||
and ordered(candidate.links) == ordered(value.links)):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
def source(self, source_key: str):
|
||||
with self.transaction() as c:
|
||||
row = c.execute(text("SELECT card_id, action FROM thoth_memory.projections "
|
||||
"WHERE workspace_id=:w AND source_key=:s"),
|
||||
{"w": self.workspace_id, "s": source_key}).mappings().first()
|
||||
return dict(row) if row else None
|
||||
|
||||
def save(self, value: CardInput, *, card_id: str | None = None, source_key: str | None = None,
|
||||
session_id: str | None = None, decision_seq: int | None = None,
|
||||
new_id: str | None = None) -> str:
|
||||
creating = card_id is None
|
||||
card_id = card_id or new_id or "mem-" + str(uuid4())
|
||||
revision = str(uuid4())
|
||||
data = value.model_dump(mode="json", exclude={"links", "dependencies"})
|
||||
with self.transaction() as c:
|
||||
if not creating:
|
||||
old = c.execute(text("SELECT data FROM thoth_memory.cards "
|
||||
"WHERE workspace_id=:w AND id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).scalar_one_or_none()
|
||||
if old is None:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
data.update({k: old.get(k) for k in ("session_id", "decision_seq")})
|
||||
else:
|
||||
if c.execute(text("SELECT 1 FROM thoth_memory.projections "
|
||||
"WHERE workspace_id=:w AND card_id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).first():
|
||||
raise MemoryConflict("A proposed card identity was already used")
|
||||
data.update(session_id=session_id, decision_seq=decision_seq)
|
||||
params = {"w": self.workspace_id, "id": card_id, "r": revision,
|
||||
"data": json.dumps(data), "family": value.family, "subject": value.subject,
|
||||
"origin": "workflow" if source_key else "manual", "source": source_key}
|
||||
c.execute(text("INSERT INTO thoth_memory.cards "
|
||||
"(workspace_id,id,family,subject,origin,data,revision) "
|
||||
"VALUES (:w,:id,:family,:subject,:origin,CAST(:data AS jsonb),:r) "
|
||||
"ON CONFLICT (workspace_id,id) DO UPDATE SET family=EXCLUDED.family, "
|
||||
"subject=EXCLUDED.subject,data=EXCLUDED.data,revision=EXCLUDED.revision, "
|
||||
"updated_at=clock_timestamp()"), params)
|
||||
c.execute(text("DELETE FROM thoth_memory.links WHERE workspace_id=:w AND source_id=:id"), params)
|
||||
for link in value.links:
|
||||
c.execute(text("INSERT INTO thoth_memory.links VALUES (:w,:id,:target,:meaning)"),
|
||||
{**params, "target": link.target_id, "meaning": link.meaning})
|
||||
c.execute(text("DELETE FROM thoth_memory.dependencies "
|
||||
"WHERE workspace_id=:w AND card_id=:id"), params)
|
||||
for dep in {tuple(d.model_dump().values()) for d in value.dependencies}:
|
||||
c.execute(text("INSERT INTO thoth_memory.dependencies VALUES "
|
||||
"(:w,:id,:database,:schema,:table,:column)"),
|
||||
{**params, **dict(zip(("database", "schema", "table", "column"), dep))})
|
||||
c.execute(text("INSERT INTO thoth_memory.projections "
|
||||
"(workspace_id,card_id,revision,action,source_key) VALUES (:w,:id,:r,'upsert',:source) "
|
||||
"ON CONFLICT (workspace_id,card_id) DO UPDATE SET revision=EXCLUDED.revision, "
|
||||
"action='upsert',pending=true,error=NULL,updated_at=clock_timestamp()"), params)
|
||||
return card_id
|
||||
|
||||
def delete(self, card_id: str):
|
||||
with self.transaction() as c:
|
||||
params = {"w": self.workspace_id, "id": card_id, "r": str(uuid4())}
|
||||
deleted = c.execute(text("DELETE FROM thoth_memory.cards "
|
||||
"WHERE workspace_id=:w AND id=:id"), params).rowcount
|
||||
if not deleted:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
c.execute(text("UPDATE thoth_memory.projections SET action='delete',revision=:r,"
|
||||
"pending=true,error=NULL,updated_at=clock_timestamp() "
|
||||
"WHERE workspace_id=:w AND card_id=:id"), params)
|
||||
|
||||
def projections(self, *, pending: bool = True):
|
||||
with self.transaction() as c:
|
||||
needs_update = "(pending OR (action='upsert' AND format<>2))"
|
||||
rows = c.execute(text("SELECT card_id,revision,action,"+needs_update+" AS pending,"
|
||||
"error,updated_at "
|
||||
"FROM thoth_memory.projections WHERE workspace_id=:w "
|
||||
+ ("AND "+needs_update+" " if pending else "") + "ORDER BY updated_at,card_id"),
|
||||
{"w": self.workspace_id}).mappings().all()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
def projection_result(self, card_id: str, revision: str, error: str | None):
|
||||
with self.transaction() as c:
|
||||
c.execute(text("UPDATE thoth_memory.projections SET pending=:p,error=:error,format=2,"
|
||||
"updated_at=clock_timestamp() WHERE workspace_id=:w AND card_id=:id AND revision=:r"),
|
||||
{"p": error is not None, "error": error, "w": self.workspace_id,
|
||||
"id": card_id, "r": revision})
|
||||
|
||||
def invalidate_all(self):
|
||||
with self.transaction() as c:
|
||||
c.execute(text("UPDATE thoth_memory.projections SET pending=true,error=NULL "
|
||||
"WHERE workspace_id=:w"), {"w": self.workspace_id})
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Bounded graph recall over current Memory cards; no model or approval side effects."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
from .models import Card, Family, MemoryNotFound
|
||||
|
||||
PROJECTION_FORMAT = 2
|
||||
MAX_SEEDS = 100
|
||||
MAX_DEPTH = 2
|
||||
MAX_LINKS_PER_CARD = 20
|
||||
MAX_VISITED = 200
|
||||
MAX_EDGES = 400
|
||||
RRF_K = 60
|
||||
|
||||
|
||||
class RecallScope(BaseModel):
|
||||
"""Exact business scope/concepts and hierarchical physical context.
|
||||
|
||||
Cards without dependencies are workspace-wide. A dependency applies to its
|
||||
database and every descendant of the schema/table/column it names.
|
||||
All physical fields must match the SAME dependency, never separate entries.
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
scope: str = Field(default="", max_length=10000)
|
||||
database: str = Field(default="", max_length=200)
|
||||
schema_name: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
concepts: list[str] = Field(default_factory=list, max_length=100)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_context(self):
|
||||
if ((self.schema_name and not self.database) or (self.table and not self.schema_name)
|
||||
or (self.column and not self.table)):
|
||||
raise ValueError("Physical recall scope requires its database/schema/table ancestors")
|
||||
if any(not value.strip() or len(value) > 200 for value in self.concepts):
|
||||
raise ValueError("Recall concepts must contain between 1 and 200 characters")
|
||||
return self
|
||||
|
||||
def matches(self, card: Card) -> bool:
|
||||
if self.scope and self.scope != card.scope:
|
||||
return False
|
||||
if not set(self.concepts) <= set(card.concepts):
|
||||
return False
|
||||
if not self.database or not card.dependencies:
|
||||
return True
|
||||
return any(all(not getattr(self, key) or getattr(dep, key) in ("", getattr(self, key))
|
||||
for key in ("database", "schema_name", "table", "column"))
|
||||
for dep in card.dependencies)
|
||||
|
||||
def vector_filter(self, family: Family | None) -> dict:
|
||||
return {"memory": {**self.model_dump(), "family": family,
|
||||
"format": PROJECTION_FORMAT}}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecalledCard:
|
||||
card: Card
|
||||
score: float
|
||||
path: tuple[str, ...]
|
||||
|
||||
|
||||
def expand_and_rank(repo, hits, *, scope: RecallScope, family: Family | None,
|
||||
excluded: set[str], top: int) -> list[RecalledCard]:
|
||||
"""RRF direct rank + strongest link path, decayed by 0.5 per outgoing hop.
|
||||
|
||||
Roots, nodes, fan-out and depth are all bounded. Repeated paths do not add
|
||||
votes: cycles and highly connected cards cannot amplify their own relevance.
|
||||
The caller holds the workspace operation lock while resolving authority.
|
||||
"""
|
||||
cache: dict[str, Card | None] = {}
|
||||
|
||||
def current(identity):
|
||||
if identity not in cache:
|
||||
if len(cache) >= MAX_VISITED:
|
||||
return None
|
||||
try:
|
||||
card = repo.get(identity)
|
||||
except MemoryNotFound:
|
||||
card = None
|
||||
if card is not None and (not card.indexed or card.id in excluded
|
||||
or (family and card.family != family) or not scope.matches(card)):
|
||||
card = None
|
||||
cache[identity] = card
|
||||
return cache[identity]
|
||||
|
||||
direct: dict[str, float] = {}
|
||||
graph: dict[str, tuple[float, tuple[str, ...]]] = {}
|
||||
seeds = []
|
||||
for rank, hit in enumerate(hits[:MAX_SEEDS], 1):
|
||||
card = current(hit.ref)
|
||||
if (card is None or hit.metadata.get("memory_revision") != card.revision
|
||||
or hit.metadata.get("memory_format") != PROJECTION_FORMAT
|
||||
or card.id in direct):
|
||||
continue
|
||||
direct[card.id] = 1 / (RRF_K + rank)
|
||||
seeds.append(card)
|
||||
|
||||
traversed = 0
|
||||
for seed in seeds:
|
||||
frontier = [(seed, (seed.id,))]
|
||||
visited = {seed.id}
|
||||
for depth in range(1, MAX_DEPTH + 1):
|
||||
next_frontier = []
|
||||
for source, path in frontier:
|
||||
for link in sorted(source.links, key=lambda link: link.target_id)[:MAX_LINKS_PER_CARD]:
|
||||
if traversed >= MAX_EDGES:
|
||||
break
|
||||
traversed += 1
|
||||
if link.target_id in visited:
|
||||
continue
|
||||
visited.add(link.target_id)
|
||||
target = current(link.target_id)
|
||||
if target is None:
|
||||
continue
|
||||
target_path = (*path, target.id)
|
||||
score = direct[seed.id] * 0.5 ** depth
|
||||
previous = graph.get(target.id)
|
||||
if previous is None or (-score, target_path) < (-previous[0], previous[1]):
|
||||
graph[target.id] = (score, target_path)
|
||||
next_frontier.append((target, target_path))
|
||||
frontier = next_frontier
|
||||
|
||||
ranked = []
|
||||
for identity in direct.keys() | graph.keys():
|
||||
graph_score, path = graph.get(identity, (0, (identity,)))
|
||||
ranked.append(RecalledCard(cache[identity], direct.get(identity, 0) + graph_score, path))
|
||||
return sorted(ranked, key=lambda result: (-result.score, result.card.id))[:top]
|
||||
@@ -0,0 +1,313 @@
|
||||
"""Reviewer-edited additions/updates, grounded in effective approved session decisions."""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from uuid import NAMESPACE_URL, uuid5
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from tht.phase import current_phase, effective_decisions
|
||||
|
||||
from .models import CardInput, MemoryConflict
|
||||
from .solved import _build_solved_snapshot
|
||||
|
||||
APPROVED_SOURCES = {
|
||||
"concept_clarified",
|
||||
"join_modified",
|
||||
"column_corrected",
|
||||
"concept_formula_approved",
|
||||
"cte_corrected",
|
||||
"cte_approved",
|
||||
"sql_approved",
|
||||
}
|
||||
|
||||
|
||||
class Proposal(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
|
||||
source_seqs: list[int] = Field(min_length=1, max_length=20)
|
||||
card: CardInput
|
||||
target_id: str | None = None
|
||||
target_revision: str | None = None
|
||||
reason: str = Field(min_length=1, max_length=10000)
|
||||
|
||||
|
||||
class Selection(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
id: str
|
||||
card: CardInput
|
||||
|
||||
|
||||
class ReviewResponse(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
summary_id: str
|
||||
items: list[Selection] = Field(max_length=20)
|
||||
|
||||
|
||||
def digest(value):
|
||||
return hashlib.sha256(
|
||||
json.dumps(value, sort_keys=True, ensure_ascii=False).encode()
|
||||
).hexdigest()
|
||||
|
||||
|
||||
def context_hash(snapshot):
|
||||
return digest(
|
||||
{
|
||||
"decisions": [
|
||||
d.model_dump(mode="json")
|
||||
for d in effective_decisions(snapshot)
|
||||
if d.type != "memory_summary_reviewed"
|
||||
],
|
||||
"proposals": snapshot.artifacts.get("memory_proposals"),
|
||||
"sql": snapshot.artifacts.get("sql_final"),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def solved_snapshot(snapshot):
|
||||
linking = json.loads(snapshot.artifacts.get("schema_linking", "{}"))
|
||||
tables = {
|
||||
c["name"]
|
||||
for c in linking.get("candidates", [])
|
||||
if c.get("kind") == "table" and c.get("decision") == "promoted"
|
||||
}
|
||||
return _build_solved_snapshot(snapshot, tables)
|
||||
|
||||
|
||||
def validate_proposals(snapshot, raw):
|
||||
if not isinstance(raw, list) or len(raw) > 20:
|
||||
raise ValueError("Memory proposals must be a list of at most 20 cards")
|
||||
proposals = [Proposal.model_validate(p) for p in raw]
|
||||
effective = {d.seq: d for d in effective_decisions(snapshot)}
|
||||
if len({p.id for p in proposals}) != len(proposals):
|
||||
raise ValueError("Memory proposal identities must be unique")
|
||||
for p in proposals:
|
||||
sources = [effective.get(seq) for seq in p.source_seqs]
|
||||
if any(d is None or d.type not in APPROVED_SOURCES for d in sources):
|
||||
raise MemoryConflict("Memory proposals require effective approved source decisions")
|
||||
if p.card.family == "explained_error" and not any(d.rationale.strip() for d in sources):
|
||||
raise MemoryConflict("Explained errors require an approved explanation")
|
||||
if p.card.family == "solved_question":
|
||||
solved = solved_snapshot(snapshot)
|
||||
if p.card.sql != solved.metadata["sql"]:
|
||||
raise MemoryConflict("Exemplar SQL must match the current approved solution")
|
||||
if bool(p.target_id) != bool(p.target_revision):
|
||||
raise ValueError("Updates require the identity and revision of the card being replaced")
|
||||
return proposals
|
||||
|
||||
|
||||
def prepare(service, snapshot):
|
||||
service._session(snapshot)
|
||||
if current_phase(snapshot) != 8 or snapshot.manifest.status in {"finalized", "archived"}:
|
||||
raise MemoryConflict("The Memory summary is reviewed at the end of F8")
|
||||
with service.repository.transaction() as connection:
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT summary_id,result FROM thoth_memory.reviews "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND result->>'context_hash'=:h "
|
||||
"ORDER BY created_at DESC LIMIT 1"
|
||||
),
|
||||
{
|
||||
"w": service.repository.workspace_id,
|
||||
"s": snapshot.manifest.id,
|
||||
"h": context_hash(snapshot),
|
||||
},
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
return {
|
||||
"reviewed": True,
|
||||
"summary_id": receipt["summary_id"],
|
||||
"saved": len(receipt["result"]["saved"]),
|
||||
}
|
||||
raw = json.loads(snapshot.artifacts.get("memory_proposals", "[]"))
|
||||
proposals = validate_proposals(snapshot, raw)
|
||||
covered = {seq for p in proposals for seq in p.source_seqs}
|
||||
for d in service.promotions(snapshot):
|
||||
if d["decision_seq"] not in covered:
|
||||
proposals.append(
|
||||
Proposal(
|
||||
id=f"decision-{d['decision_seq']}",
|
||||
source_seqs=[d["decision_seq"]],
|
||||
reason="Reusable domain clarification",
|
||||
card=CardInput(
|
||||
family="domain_clarification",
|
||||
subject=d["subject"],
|
||||
detail=d["detail"],
|
||||
rationale=d["rationale"],
|
||||
question=d["question_context"],
|
||||
scope=service.repository.workspace_id,
|
||||
concepts=[d["subject"]],
|
||||
),
|
||||
)
|
||||
)
|
||||
if not any(p.card.family == "solved_question" for p in proposals):
|
||||
solved = solved_snapshot(snapshot)
|
||||
approved = [d for d in effective_decisions(snapshot) if d.type == "sql_approved"][-1]
|
||||
proposals.append(
|
||||
Proposal(
|
||||
id="solved-question",
|
||||
source_seqs=[approved.seq],
|
||||
reason="Approved solution, for consultation in future questions",
|
||||
card=CardInput(
|
||||
family="solved_question",
|
||||
subject=solved.title,
|
||||
question=solved.content,
|
||||
sql=solved.metadata["sql"],
|
||||
scope=service.repository.workspace_id,
|
||||
dependencies=[
|
||||
{
|
||||
"database": snapshot.manifest.database,
|
||||
"schema_name": snapshot.manifest.db_schema,
|
||||
"table": table,
|
||||
}
|
||||
for table in solved.metadata["tables"]
|
||||
],
|
||||
),
|
||||
)
|
||||
)
|
||||
if len(proposals) > 20:
|
||||
raise MemoryConflict("Reduce the final Memory summary to at most 20 cards")
|
||||
items = []
|
||||
targets = set()
|
||||
content = {}
|
||||
aliases = {}
|
||||
for p in proposals:
|
||||
before = None
|
||||
if p.target_id:
|
||||
old = service.repository.get(p.target_id)
|
||||
if old.revision != p.target_revision:
|
||||
raise MemoryConflict(
|
||||
"A Memory card changed; refresh the proposed update before review"
|
||||
)
|
||||
if p.target_id in targets:
|
||||
raise MemoryConflict("Propose only one update to each Memory card")
|
||||
targets.add(p.target_id)
|
||||
before = old.model_dump(mode="json")
|
||||
key = digest(p.card.model_dump(mode="json"))
|
||||
duplicate = service.repository.exact_match(p.card)
|
||||
if duplicate and (not p.target_id or duplicate.id == p.target_id):
|
||||
aliases["proposal:" + p.id] = duplicate.id
|
||||
continue
|
||||
if not p.target_id and key in content:
|
||||
aliases["proposal:" + p.id] = "proposal:" + content[key]
|
||||
continue
|
||||
content[key] = p.id
|
||||
items.append({**p.model_dump(mode="json"), "before": before})
|
||||
for item in items:
|
||||
for link in item["card"]["links"]:
|
||||
link["target_id"] = aliases.get(link["target_id"], link["target_id"])
|
||||
return {
|
||||
"summary_id": digest(
|
||||
{
|
||||
"items": items,
|
||||
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)],
|
||||
}
|
||||
),
|
||||
"items": items,
|
||||
}
|
||||
|
||||
|
||||
def apply(service, snapshot, response: ReviewResponse):
|
||||
service._session(snapshot)
|
||||
request_hash = digest(response.model_dump(mode="json"))
|
||||
session_id = snapshot.manifest.id
|
||||
with service.repository.operation() as repo:
|
||||
with repo.transaction() as connection:
|
||||
params = {"w": repo.workspace_id, "s": session_id, "id": response.summary_id}
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT request_hash,result FROM thoth_memory.reviews "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND summary_id=:id"
|
||||
),
|
||||
params,
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
if receipt["request_hash"] != request_hash:
|
||||
raise MemoryConflict(
|
||||
"This summary was already reviewed with different selections"
|
||||
)
|
||||
saved = receipt["result"]["saved"]
|
||||
else:
|
||||
# Resolve the same locked repository for preview and optimistic update checks.
|
||||
original = service.repository
|
||||
service.repository = repo
|
||||
try:
|
||||
summary = prepare(service, snapshot)
|
||||
finally:
|
||||
service.repository = original
|
||||
if summary.get("reviewed") or summary["summary_id"] != response.summary_id:
|
||||
raise MemoryConflict("The Memory summary changed; review it again")
|
||||
choices = {choice.id: choice for choice in response.items}
|
||||
candidates = {p["id"]: p for p in summary["items"]}
|
||||
if len(choices) != len(response.items) or choices.keys() - candidates.keys():
|
||||
raise ValueError("Memory review contains duplicate or unknown choices")
|
||||
selected = validate_proposals(
|
||||
snapshot,
|
||||
[
|
||||
{
|
||||
**{k: v for k, v in candidates[key].items() if k != "before"},
|
||||
"card": choice.card.model_dump(mode="json"),
|
||||
}
|
||||
for key, choice in choices.items()
|
||||
],
|
||||
)
|
||||
identities = {
|
||||
p.id: p.target_id
|
||||
or "mem-"
|
||||
+ str(uuid5(NAMESPACE_URL, f"thothii:{repo.workspace_id}:{session_id}:{p.id}"))
|
||||
for p in selected
|
||||
}
|
||||
saved = []
|
||||
for p in selected:
|
||||
# Create/update every selected card before inserting links among new cards.
|
||||
identity = repo.save(
|
||||
p.card.model_copy(update={"links": []}),
|
||||
card_id=p.target_id,
|
||||
new_id=identities[p.id],
|
||||
source_key=f"review:{session_id}:{p.id}",
|
||||
session_id=session_id,
|
||||
decision_seq=p.source_seqs[0],
|
||||
)
|
||||
saved.append({"id": identity, "proposal_id": p.id})
|
||||
for p in selected:
|
||||
value = p.card.model_dump(mode="json")
|
||||
for link in value["links"]:
|
||||
if link["target_id"].startswith("proposal:"):
|
||||
target = link["target_id"].removeprefix("proposal:")
|
||||
if target not in identities:
|
||||
raise ValueError("Select the linked card or remove its link")
|
||||
link["target_id"] = identities[target]
|
||||
repo.save(CardInput.model_validate(value), card_id=identities[p.id])
|
||||
connection.execute(
|
||||
text(
|
||||
"INSERT INTO thoth_memory.reviews "
|
||||
"(workspace_id,session_id,summary_id,request_hash,result) "
|
||||
"VALUES (:w,:s,:id,:hash,CAST(:result AS jsonb))"
|
||||
),
|
||||
{
|
||||
**params,
|
||||
"hash": request_hash,
|
||||
"result": json.dumps(
|
||||
{
|
||||
"saved": saved,
|
||||
"context_hash": context_hash(snapshot),
|
||||
}
|
||||
),
|
||||
},
|
||||
)
|
||||
results = [service._propagate(repo, item["id"]) for item in saved]
|
||||
return {
|
||||
"saved": len(saved),
|
||||
"declined": len(summary["items"]) - len(saved) if not receipt else None,
|
||||
"indexed": all(r["indexed"] for r in results),
|
||||
"results": results,
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Installation binding for Memory; vector dependencies are opened only when needed."""
|
||||
|
||||
import os
|
||||
|
||||
from tht.session.models import PrincipalContext
|
||||
|
||||
from .migrate import installation_url
|
||||
from .models import MemoryForbidden, MemoryUnavailable
|
||||
from .repository import MemoryRepository
|
||||
from .service import MemoryService
|
||||
|
||||
|
||||
def _principal():
|
||||
issuer = os.environ.get("THT_PRINCIPAL_ISSUER", "").strip()
|
||||
subject = os.environ.get("THT_PRINCIPAL_SUBJECT", "").strip()
|
||||
if not issuer or not subject:
|
||||
raise MemoryForbidden("A trusted runtime principal is required for Memory")
|
||||
return PrincipalContext(
|
||||
issuer=issuer, subject=subject,
|
||||
is_admin=os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() in {"1", "true"},
|
||||
)
|
||||
|
||||
|
||||
def _repository(workspace_id):
|
||||
try:
|
||||
url = installation_url()
|
||||
except ValueError:
|
||||
raise MemoryUnavailable("Memory PostgreSQL installation configuration is unavailable") \
|
||||
from None
|
||||
return MemoryRepository(url, workspace_id)
|
||||
|
||||
|
||||
def memory_service(cfg):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
|
||||
principal = _principal()
|
||||
return MemoryService(_repository(cfg._workspace_id), principal,
|
||||
language=cfg.language,
|
||||
store_factory=lambda: build_vector_store(cfg, require_write=True),
|
||||
embedder_factory=lambda: make_embedder(cfg.embeddings))
|
||||
|
||||
|
||||
def admin_service(workspace_id, runtime):
|
||||
"""Admin access needs no DWH binding, active session or Evidence materialization."""
|
||||
from tht.adapters.vector.qdrant import QdrantVectorStore
|
||||
from tht.config import EmbeddingsConfig
|
||||
from tht.vectorstore.embeddings import OllamaEmbeddings
|
||||
|
||||
principal = _principal()
|
||||
if not principal.is_admin:
|
||||
raise MemoryForbidden("Memory administration requires an administrator")
|
||||
return MemoryService(_repository(workspace_id), principal,
|
||||
language=runtime.get("memoryLanguage", "en"),
|
||||
store_factory=lambda: QdrantVectorStore(
|
||||
base_url=runtime["internalQdrantUrl"], workspace_id=workspace_id,
|
||||
collections={"reference": workspace_id+"-reference", "memory": workspace_id+"-memory"},
|
||||
expected_dimension=runtime["internalEmbeddingDimensions"],
|
||||
),
|
||||
embedder_factory=lambda: OllamaEmbeddings(EmbeddingsConfig(
|
||||
base_url=runtime["internalEmbeddingUrl"], model=runtime["internalEmbeddingModel"],
|
||||
dimensions=runtime["internalEmbeddingDimensions"], timeout=30,
|
||||
)))
|
||||
@@ -0,0 +1,240 @@
|
||||
"""Memory operations: SQL authority, explicit projection recovery, verified recall."""
|
||||
|
||||
import hashlib
|
||||
|
||||
from tht.phase import effective_decisions
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.session.models import PrincipalContext
|
||||
from tht.vectorstore.records import VectorRecord
|
||||
|
||||
from .core import decided_memory_ids, declined_promotion_seqs, question_context
|
||||
from .models import (
|
||||
CardInput,
|
||||
CardQuery,
|
||||
Dependency,
|
||||
MemoryConflict,
|
||||
MemoryForbidden,
|
||||
MemoryNotFound,
|
||||
)
|
||||
from .repository import MemoryRepository
|
||||
from .retrieval import MAX_SEEDS, PROJECTION_FORMAT, RecallScope, expand_and_rank
|
||||
from .solved import _build_solved_snapshot
|
||||
|
||||
|
||||
class MemoryService:
|
||||
def __init__(self, repository: MemoryRepository, principal: PrincipalContext,
|
||||
*, store_factory, embedder_factory, language: str = "en"):
|
||||
self.repository = repository
|
||||
self.principal = principal
|
||||
self.store_factory = store_factory
|
||||
self.embedder_factory = embedder_factory
|
||||
if language not in {"en", "it"}:
|
||||
raise ValueError("Memory language must be en or it")
|
||||
self.query_language = {"en": "english", "it": "italian"}[language]
|
||||
|
||||
def close(self):
|
||||
self.repository.close()
|
||||
|
||||
def _admin(self):
|
||||
if not self.principal.is_admin:
|
||||
raise MemoryForbidden("Memory administration requires an administrator")
|
||||
|
||||
def list(self, query: CardQuery):
|
||||
self._admin()
|
||||
return self.repository.list(query)
|
||||
|
||||
def get(self, card_id: str):
|
||||
self._admin()
|
||||
return self.repository.get(card_id).model_dump(mode="json")
|
||||
|
||||
def pending(self):
|
||||
self._admin()
|
||||
return self.repository.projections()
|
||||
|
||||
def save(self, value: CardInput, card_id: str | None = None):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
card_id = repo.save(value, card_id=card_id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def delete(self, card_id: str):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
repo.delete(card_id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def retry(self, card_id: str):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def _propagate(self, repo, card_id):
|
||||
operation = next((p for p in repo.projections(pending=False)
|
||||
if p["card_id"] == card_id), None)
|
||||
if operation is None:
|
||||
raise MemoryNotFound("Memory operation was not found in this workspace")
|
||||
error = None
|
||||
if operation["pending"]:
|
||||
try:
|
||||
store = self.store_factory()
|
||||
if operation["action"] == "delete":
|
||||
store.delete_memory_records([f"card:{card_id}"])
|
||||
else:
|
||||
card = repo.get(card_id)
|
||||
kind = "solved_question" if card.family == "solved_question" else "memory"
|
||||
content = "\n".join(filter(None, [card.subject, card.detail, card.scope,
|
||||
card.rationale, card.question, card.sql,
|
||||
" ".join(card.concepts),
|
||||
"\n".join(".".join(filter(None, [d.database,
|
||||
d.schema_name, d.table, d.column]))
|
||||
for d in card.dependencies)]))
|
||||
record = VectorRecord(
|
||||
id=f"card:{card.id}", kind=kind, ref=card.id, title=card.subject,
|
||||
content=content, metadata={"memory_revision": card.revision,
|
||||
"memory_format": PROJECTION_FORMAT, "memory_family": card.family,
|
||||
"memory_scope": card.scope, "memory_concepts": card.concepts,
|
||||
"memory_dependencies": [d.model_dump() for d in card.dependencies]},
|
||||
)
|
||||
embedding = self.embedder_factory().embed_documents([content])[0]
|
||||
# A family change may change the vector kind (and point identity).
|
||||
store.delete_memory_records([f"card:{card_id}"])
|
||||
store.upsert("memory", [VectorWriteRecord(
|
||||
record=record, embedding=embedding,
|
||||
content_hash=hashlib.sha256(card.revision.encode()).hexdigest(),
|
||||
sparse_text=content, sparse_language=self.query_language,
|
||||
)])
|
||||
except Exception: # noqa: BLE001 - durable pending state covers adapter/factory failures.
|
||||
error = "Memory change is saved; index update is incomplete. Retry the index update."
|
||||
repo.projection_result(card_id, operation["revision"], error)
|
||||
result = {"id": card_id, "saved": True, "indexed": error is None,
|
||||
"action": operation["action"], "error": error}
|
||||
if operation["action"] != "delete":
|
||||
result["card"] = repo.get(card_id).model_dump(mode="json")
|
||||
return result
|
||||
|
||||
def rebuild(self):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
# Persist invalidation before deleting anything. A crash remains recoverable.
|
||||
repo.invalidate_all()
|
||||
store = self.store_factory()
|
||||
store.prepare_memory_index()
|
||||
store.delete_kinds("memory", ["memory", "solved_question"])
|
||||
return [self._propagate(repo, p["card_id"]) for p in repo.projections()]
|
||||
|
||||
def retrieve(self, question: str, *, searcher, embedder, top: int = 5,
|
||||
family=None, scope: RecallScope | None = None, excluded=()):
|
||||
if type(top) is not int or not 1 <= top <= 100:
|
||||
raise ValueError("Recall limit must be between 1 and 100")
|
||||
if not question.strip():
|
||||
raise ValueError("Recall question must not be empty")
|
||||
scope = scope or RecallScope()
|
||||
self.repository.list(CardQuery(page_size=1))
|
||||
kinds = (["solved_question"] if family == "solved_question" else
|
||||
["memory"] if family else ["memory", "solved_question"])
|
||||
hits = searcher.search(embedder.embed_query(question),
|
||||
top_n=min(MAX_SEEDS, max(20, top * 3)), kinds=kinds,
|
||||
query_text=question, query_language=self.query_language,
|
||||
metadata_filter=scope.vector_filter(family))
|
||||
# Mutations use the same lock: links, eligibility and payload are resolved
|
||||
# together against current authority, after the potentially slow vector call.
|
||||
with self.repository.operation() as repo:
|
||||
return expand_and_rank(repo, hits, scope=scope, family=family,
|
||||
excluded=set(excluded), top=top)
|
||||
|
||||
def recall(self, question: str, *, searcher, embedder, top: int = 5,
|
||||
solved: bool = False, decisions=(), scope: RecallScope | None = None):
|
||||
family = "solved_question" if solved else "domain_clarification"
|
||||
candidates = self.retrieve(question, searcher=searcher, embedder=embedder, top=top,
|
||||
family=family, scope=scope, excluded=decided_memory_ids(list(decisions)))
|
||||
result = []
|
||||
for candidate in candidates:
|
||||
card = candidate.card
|
||||
common = {"id": card.id, "session_id": card.session_id,
|
||||
"revision": card.revision, "family": card.family, "scope": card.scope,
|
||||
"dependencies": [d.model_dump() for d in card.dependencies],
|
||||
"tables": sorted({d.table for d in card.dependencies if d.table}),
|
||||
"score": round(candidate.score, 6),
|
||||
"retrieval": {"path": list(candidate.path), "method": "hybrid_links"}}
|
||||
if solved:
|
||||
result.append({**common, "question": card.question, "sql": card.sql})
|
||||
else:
|
||||
# New workflow category consumption belongs to M3.
|
||||
result.append({**common, "type": "concept_clarified",
|
||||
"subject": card.subject, "detail": card.detail, "rationale": card.rationale,
|
||||
"question_context": card.question, "scope": card.scope,
|
||||
"concepts": card.concepts})
|
||||
return result
|
||||
|
||||
def _session(self, snapshot):
|
||||
if (snapshot.manifest.workspace_id != self.repository.workspace_id
|
||||
or (not self.principal.is_admin
|
||||
and snapshot.manifest.author != self.principal.subject)):
|
||||
raise MemoryForbidden("Memory source session is outside the authorized context")
|
||||
|
||||
def promotions(self, snapshot):
|
||||
self._session(snapshot)
|
||||
decisions = effective_decisions(snapshot)
|
||||
declined = declined_promotion_seqs(decisions)
|
||||
result = []
|
||||
seen = set()
|
||||
for d in decisions:
|
||||
if d.type != "concept_clarified" or d.seq in declined:
|
||||
continue
|
||||
source_key = f"decision:{snapshot.manifest.id}:{d.seq}"
|
||||
if self.repository.source(source_key) is not None:
|
||||
continue
|
||||
content = (d.subject, d.detail, d.rationale)
|
||||
if content in seen:
|
||||
continue
|
||||
seen.add(content)
|
||||
result.append({"decision_seq": d.seq, "type": d.type, "subject": d.subject,
|
||||
"detail": d.detail, "rationale": d.rationale,
|
||||
"question_context": question_context(decisions, snapshot.manifest)})
|
||||
return result
|
||||
|
||||
def promote(self, snapshot, seqs):
|
||||
self._session(snapshot)
|
||||
decisions = effective_decisions(snapshot)
|
||||
selected = [d for d in decisions if d.seq in seqs and d.type == "concept_clarified"]
|
||||
results = []
|
||||
with self.repository.operation() as repo:
|
||||
for d in selected:
|
||||
key = f"decision:{snapshot.manifest.id}:{d.seq}"
|
||||
existing = repo.source(key)
|
||||
if existing and existing["action"] == "delete":
|
||||
continue
|
||||
card_id = existing["card_id"] if existing else repo.save(CardInput(
|
||||
family="domain_clarification", subject=d.subject, detail=d.detail,
|
||||
rationale=d.rationale, scope=self.repository.workspace_id,
|
||||
question=question_context(decisions, snapshot.manifest), concepts=[d.subject],
|
||||
), source_key=key, session_id=snapshot.manifest.id, decision_seq=d.seq)
|
||||
results.append(self._propagate(repo, card_id))
|
||||
return results
|
||||
|
||||
def save_solved(self, snapshot, promoted_tables=None):
|
||||
self._session(snapshot)
|
||||
if snapshot.manifest.status != "finalized":
|
||||
raise MemoryConflict("Only a finalized session can produce a solved-question card")
|
||||
record = _build_solved_snapshot(snapshot, promoted_tables)
|
||||
with self.repository.operation() as repo:
|
||||
existing = repo.source(f"solved:{snapshot.manifest.id}")
|
||||
if existing and existing["action"] == "delete":
|
||||
return {"saved": True, "indexed": True, "action": "delete", "error": None}
|
||||
card_id = existing["card_id"] if existing else repo.save(CardInput(
|
||||
family="solved_question", subject=record.title,
|
||||
scope=self.repository.workspace_id, question=record.content,
|
||||
sql=record.metadata["sql"],
|
||||
dependencies=[Dependency(database=snapshot.manifest.database,
|
||||
schema_name=snapshot.manifest.db_schema, table=table)
|
||||
for table in record.metadata["tables"]],
|
||||
), source_key=f"solved:{snapshot.manifest.id}", session_id=snapshot.manifest.id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def retry_solved(self, snapshot):
|
||||
self._session(snapshot)
|
||||
with self.repository.operation() as repo:
|
||||
existing = repo.source(f"solved:{snapshot.manifest.id}")
|
||||
if existing is None:
|
||||
raise MemoryNotFound("No authoritative exemplar exists for this session")
|
||||
return self._propagate(repo, existing["card_id"])
|
||||
@@ -0,0 +1,72 @@
|
||||
CREATE SCHEMA IF NOT EXISTS thoth_memory;
|
||||
DO $$ BEGIN
|
||||
IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'thoth_memory_runtime') THEN
|
||||
CREATE ROLE thoth_memory_runtime NOLOGIN;
|
||||
END IF;
|
||||
IF EXISTS (SELECT FROM pg_roles WHERE rolname = 'thothii_catalog_runtime') THEN
|
||||
GRANT thoth_memory_runtime TO thothii_catalog_runtime;
|
||||
END IF;
|
||||
END $$;
|
||||
|
||||
CREATE TABLE thoth_memory.cards (
|
||||
workspace_id text NOT NULL,
|
||||
id text NOT NULL,
|
||||
family text NOT NULL,
|
||||
subject text NOT NULL,
|
||||
origin text NOT NULL CHECK (origin IN ('manual', 'workflow')),
|
||||
data jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
updated_at timestamptz NOT NULL DEFAULT now(),
|
||||
revision text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, id)
|
||||
);
|
||||
CREATE INDEX ON thoth_memory.cards (workspace_id, updated_at, id);
|
||||
CREATE INDEX ON thoth_memory.cards (workspace_id, family);
|
||||
CREATE TABLE thoth_memory.links (
|
||||
workspace_id text NOT NULL,
|
||||
source_id text NOT NULL,
|
||||
target_id text NOT NULL,
|
||||
meaning text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, source_id, target_id),
|
||||
CHECK (source_id <> target_id),
|
||||
FOREIGN KEY (workspace_id, source_id) REFERENCES thoth_memory.cards ON DELETE CASCADE,
|
||||
FOREIGN KEY (workspace_id, target_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
|
||||
);
|
||||
CREATE TABLE thoth_memory.dependencies (
|
||||
workspace_id text NOT NULL,
|
||||
card_id text NOT NULL,
|
||||
database_id text NOT NULL,
|
||||
schema_name text NOT NULL,
|
||||
table_name text NOT NULL,
|
||||
column_name text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, card_id, database_id, schema_name, table_name, column_name),
|
||||
FOREIGN KEY (workspace_id, card_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
|
||||
);
|
||||
-- No FK to cards: deletion recovery and source receipts survive removal of the card.
|
||||
CREATE TABLE thoth_memory.projections (
|
||||
workspace_id text NOT NULL,
|
||||
card_id text NOT NULL,
|
||||
revision text NOT NULL,
|
||||
action text NOT NULL CHECK (action IN ('upsert', 'delete')),
|
||||
pending boolean NOT NULL DEFAULT true,
|
||||
error text,
|
||||
source_key text,
|
||||
updated_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, card_id),
|
||||
UNIQUE (workspace_id, source_key)
|
||||
);
|
||||
DO $$ DECLARE t text; BEGIN
|
||||
FOREACH t IN ARRAY ARRAY['cards', 'links', 'dependencies', 'projections'] LOOP
|
||||
EXECUTE format('ALTER TABLE thoth_memory.%I ENABLE ROW LEVEL SECURITY', t);
|
||||
EXECUTE format('ALTER TABLE thoth_memory.%I FORCE ROW LEVEL SECURITY', t);
|
||||
EXECUTE format(
|
||||
'CREATE POLICY workspace_isolation ON thoth_memory.%I USING '
|
||||
'(workspace_id = current_setting(''thoth.memory_workspace'', true)) '
|
||||
'WITH CHECK (workspace_id = current_setting(''thoth.memory_workspace'', true))', t);
|
||||
EXECUTE format('GRANT SELECT, INSERT, UPDATE, DELETE ON thoth_memory.%I '
|
||||
'TO thoth_memory_runtime', t);
|
||||
END LOOP;
|
||||
END $$;
|
||||
GRANT USAGE ON SCHEMA thoth_memory TO thoth_memory_runtime;
|
||||
GRANT SELECT ON thoth_memory.migrations TO thoth_memory_runtime;
|
||||
REVOKE ALL ON SCHEMA thoth_memory FROM PUBLIC;
|
||||
@@ -0,0 +1,3 @@
|
||||
-- Existing dense projections remain recoverable, but are not hybrid-ready.
|
||||
-- No cross-workspace data update or weakening of FORCE RLS is necessary.
|
||||
ALTER TABLE thoth_memory.projections ADD COLUMN format integer NOT NULL DEFAULT 1;
|
||||
@@ -0,0 +1,29 @@
|
||||
CREATE TABLE thoth_memory.reviews (
|
||||
workspace_id text NOT NULL,
|
||||
session_id text NOT NULL,
|
||||
summary_id text NOT NULL,
|
||||
request_hash text NOT NULL,
|
||||
result jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, session_id, summary_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.reviews ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.reviews FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.reviews
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT ON thoth_memory.reviews TO thoth_memory_runtime;
|
||||
|
||||
CREATE TABLE thoth_memory.cleanup_receipts (
|
||||
workspace_id text NOT NULL,
|
||||
sync_id text NOT NULL,
|
||||
request_hash text NOT NULL,
|
||||
card_ids jsonb NOT NULL,
|
||||
PRIMARY KEY (workspace_id, sync_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.cleanup_receipts ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.cleanup_receipts FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.cleanup_receipts
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT ON thoth_memory.cleanup_receipts TO thoth_memory_runtime;
|
||||
@@ -0,0 +1,14 @@
|
||||
CREATE TABLE thoth_memory.archive_repairs (
|
||||
workspace_id text NOT NULL,
|
||||
session_id text NOT NULL,
|
||||
repair_id text NOT NULL,
|
||||
data jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, session_id, repair_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.archive_repairs ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.archive_repairs FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.archive_repairs
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT, UPDATE ON thoth_memory.archive_repairs TO thoth_memory_runtime;
|
||||
@@ -230,6 +230,8 @@ def advance_problems(source: Path | SessionSnapshot, phase: int) -> list[str]:
|
||||
problems.append(f"CTE non ancora approvato: {nc} (Fase 6)")
|
||||
if phase == 7 and not _has_decision(source, "sql_approved"):
|
||||
problems.append("manca la decisione sql_approved (Fase 7)")
|
||||
if phase == 8 and not _has_decision(source, "memory_summary_reviewed"):
|
||||
problems.append("Fase 8: il riepilogo Memory deve essere revisionato prima della chiusura")
|
||||
if phase == 8 and not any(
|
||||
d.type in ("datamart_requested", "datamart_declined")
|
||||
for d in effective_decisions(source)
|
||||
|
||||
@@ -88,6 +88,10 @@ class VectorStore(Protocol):
|
||||
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int: ...
|
||||
|
||||
def prepare_memory_index(self) -> None: ...
|
||||
|
||||
def delete_memory_records(self, record_keys: list[str]) -> None: ...
|
||||
|
||||
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ...
|
||||
|
||||
def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ...
|
||||
|
||||
@@ -30,6 +30,7 @@ _ARTIFACT_FILES = {
|
||||
"cte_tests": "cte_tests.json",
|
||||
"cte_plan": "cte_plan.json",
|
||||
"cte_plan_doc": "cte_plan_doc.json",
|
||||
"memory_proposals": "memory_proposals.json",
|
||||
}
|
||||
_ARTIFACT_KEYS = {filename: key for key, filename in _ARTIFACT_FILES.items()}
|
||||
_SAFE_CTE_NAME = re.compile(r"[A-Za-z0-9_-]+\Z")
|
||||
|
||||
@@ -31,6 +31,7 @@ _ARTIFACT_KEYS = {
|
||||
"evidence",
|
||||
"evidence_receipts",
|
||||
"sql_final",
|
||||
"memory_proposals",
|
||||
"validation_report",
|
||||
"retrieval_pack",
|
||||
"cte_tests",
|
||||
|
||||
Reference in New Issue
Block a user