Files
Codex 82e2c91f42
Publish documentation / publish (push) Successful in 1m27s
feat: implement memory and evidence administration with guided repairs
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation.

Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
2026-09-10 10:31:34 +02:00

132 lines
5.2 KiB
Python

"""Bounded graph recall over current Memory cards; no model or approval side effects."""
from dataclasses import dataclass
from pydantic import BaseModel, ConfigDict, Field, model_validator
from .models import Card, Family, MemoryNotFound
PROJECTION_FORMAT = 2
MAX_SEEDS = 100
MAX_DEPTH = 2
MAX_LINKS_PER_CARD = 20
MAX_VISITED = 200
MAX_EDGES = 400
RRF_K = 60
class RecallScope(BaseModel):
"""Exact business scope/concepts and hierarchical physical context.
Cards without dependencies are workspace-wide. A dependency applies to its
database and every descendant of the schema/table/column it names.
All physical fields must match the SAME dependency, never separate entries.
"""
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
scope: str = Field(default="", max_length=10000)
database: str = Field(default="", max_length=200)
schema_name: str = Field(default="", max_length=200)
table: str = Field(default="", max_length=200)
column: str = Field(default="", max_length=200)
concepts: list[str] = Field(default_factory=list, max_length=100)
@model_validator(mode="after")
def validate_context(self):
if ((self.schema_name and not self.database) or (self.table and not self.schema_name)
or (self.column and not self.table)):
raise ValueError("Physical recall scope requires its database/schema/table ancestors")
if any(not value.strip() or len(value) > 200 for value in self.concepts):
raise ValueError("Recall concepts must contain between 1 and 200 characters")
return self
def matches(self, card: Card) -> bool:
if self.scope and self.scope != card.scope:
return False
if not set(self.concepts) <= set(card.concepts):
return False
if not self.database or not card.dependencies:
return True
return any(all(not getattr(self, key) or getattr(dep, key) in ("", getattr(self, key))
for key in ("database", "schema_name", "table", "column"))
for dep in card.dependencies)
def vector_filter(self, family: Family | None) -> dict:
return {"memory": {**self.model_dump(), "family": family,
"format": PROJECTION_FORMAT}}
@dataclass(frozen=True)
class RecalledCard:
card: Card
score: float
path: tuple[str, ...]
def expand_and_rank(repo, hits, *, scope: RecallScope, family: Family | None,
excluded: set[str], top: int) -> list[RecalledCard]:
"""RRF direct rank + strongest link path, decayed by 0.5 per outgoing hop.
Roots, nodes, fan-out and depth are all bounded. Repeated paths do not add
votes: cycles and highly connected cards cannot amplify their own relevance.
The caller holds the workspace operation lock while resolving authority.
"""
cache: dict[str, Card | None] = {}
def current(identity):
if identity not in cache:
if len(cache) >= MAX_VISITED:
return None
try:
card = repo.get(identity)
except MemoryNotFound:
card = None
if card is not None and (not card.indexed or card.id in excluded
or (family and card.family != family) or not scope.matches(card)):
card = None
cache[identity] = card
return cache[identity]
direct: dict[str, float] = {}
graph: dict[str, tuple[float, tuple[str, ...]]] = {}
seeds = []
for rank, hit in enumerate(hits[:MAX_SEEDS], 1):
card = current(hit.ref)
if (card is None or hit.metadata.get("memory_revision") != card.revision
or hit.metadata.get("memory_format") != PROJECTION_FORMAT
or card.id in direct):
continue
direct[card.id] = 1 / (RRF_K + rank)
seeds.append(card)
traversed = 0
for seed in seeds:
frontier = [(seed, (seed.id,))]
visited = {seed.id}
for depth in range(1, MAX_DEPTH + 1):
next_frontier = []
for source, path in frontier:
for link in sorted(source.links, key=lambda link: link.target_id)[:MAX_LINKS_PER_CARD]:
if traversed >= MAX_EDGES:
break
traversed += 1
if link.target_id in visited:
continue
visited.add(link.target_id)
target = current(link.target_id)
if target is None:
continue
target_path = (*path, target.id)
score = direct[seed.id] * 0.5 ** depth
previous = graph.get(target.id)
if previous is None or (-score, target_path) < (-previous[0], previous[1]):
graph[target.id] = (score, target_path)
next_frontier.append((target, target_path))
frontier = next_frontier
ranked = []
for identity in direct.keys() | graph.keys():
graph_score, path = graph.get(identity, (0, (identity,)))
ranked.append(RecalledCard(cache[identity], direct.get(identity, 0) + graph_score, path))
return sorted(ranked, key=lambda result: (-result.score, result.card.id))[:top]