Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
132 lines
5.2 KiB
Python
132 lines
5.2 KiB
Python
"""Bounded graph recall over current Memory cards; no model or approval side effects."""
|
|
|
|
from dataclasses import dataclass
|
|
|
|
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
|
|
from .models import Card, Family, MemoryNotFound
|
|
|
|
PROJECTION_FORMAT = 2
|
|
MAX_SEEDS = 100
|
|
MAX_DEPTH = 2
|
|
MAX_LINKS_PER_CARD = 20
|
|
MAX_VISITED = 200
|
|
MAX_EDGES = 400
|
|
RRF_K = 60
|
|
|
|
|
|
class RecallScope(BaseModel):
|
|
"""Exact business scope/concepts and hierarchical physical context.
|
|
|
|
Cards without dependencies are workspace-wide. A dependency applies to its
|
|
database and every descendant of the schema/table/column it names.
|
|
All physical fields must match the SAME dependency, never separate entries.
|
|
"""
|
|
|
|
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
|
scope: str = Field(default="", max_length=10000)
|
|
database: str = Field(default="", max_length=200)
|
|
schema_name: str = Field(default="", max_length=200)
|
|
table: str = Field(default="", max_length=200)
|
|
column: str = Field(default="", max_length=200)
|
|
concepts: list[str] = Field(default_factory=list, max_length=100)
|
|
|
|
@model_validator(mode="after")
|
|
def validate_context(self):
|
|
if ((self.schema_name and not self.database) or (self.table and not self.schema_name)
|
|
or (self.column and not self.table)):
|
|
raise ValueError("Physical recall scope requires its database/schema/table ancestors")
|
|
if any(not value.strip() or len(value) > 200 for value in self.concepts):
|
|
raise ValueError("Recall concepts must contain between 1 and 200 characters")
|
|
return self
|
|
|
|
def matches(self, card: Card) -> bool:
|
|
if self.scope and self.scope != card.scope:
|
|
return False
|
|
if not set(self.concepts) <= set(card.concepts):
|
|
return False
|
|
if not self.database or not card.dependencies:
|
|
return True
|
|
return any(all(not getattr(self, key) or getattr(dep, key) in ("", getattr(self, key))
|
|
for key in ("database", "schema_name", "table", "column"))
|
|
for dep in card.dependencies)
|
|
|
|
def vector_filter(self, family: Family | None) -> dict:
|
|
return {"memory": {**self.model_dump(), "family": family,
|
|
"format": PROJECTION_FORMAT}}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class RecalledCard:
|
|
card: Card
|
|
score: float
|
|
path: tuple[str, ...]
|
|
|
|
|
|
def expand_and_rank(repo, hits, *, scope: RecallScope, family: Family | None,
|
|
excluded: set[str], top: int) -> list[RecalledCard]:
|
|
"""RRF direct rank + strongest link path, decayed by 0.5 per outgoing hop.
|
|
|
|
Roots, nodes, fan-out and depth are all bounded. Repeated paths do not add
|
|
votes: cycles and highly connected cards cannot amplify their own relevance.
|
|
The caller holds the workspace operation lock while resolving authority.
|
|
"""
|
|
cache: dict[str, Card | None] = {}
|
|
|
|
def current(identity):
|
|
if identity not in cache:
|
|
if len(cache) >= MAX_VISITED:
|
|
return None
|
|
try:
|
|
card = repo.get(identity)
|
|
except MemoryNotFound:
|
|
card = None
|
|
if card is not None and (not card.indexed or card.id in excluded
|
|
or (family and card.family != family) or not scope.matches(card)):
|
|
card = None
|
|
cache[identity] = card
|
|
return cache[identity]
|
|
|
|
direct: dict[str, float] = {}
|
|
graph: dict[str, tuple[float, tuple[str, ...]]] = {}
|
|
seeds = []
|
|
for rank, hit in enumerate(hits[:MAX_SEEDS], 1):
|
|
card = current(hit.ref)
|
|
if (card is None or hit.metadata.get("memory_revision") != card.revision
|
|
or hit.metadata.get("memory_format") != PROJECTION_FORMAT
|
|
or card.id in direct):
|
|
continue
|
|
direct[card.id] = 1 / (RRF_K + rank)
|
|
seeds.append(card)
|
|
|
|
traversed = 0
|
|
for seed in seeds:
|
|
frontier = [(seed, (seed.id,))]
|
|
visited = {seed.id}
|
|
for depth in range(1, MAX_DEPTH + 1):
|
|
next_frontier = []
|
|
for source, path in frontier:
|
|
for link in sorted(source.links, key=lambda link: link.target_id)[:MAX_LINKS_PER_CARD]:
|
|
if traversed >= MAX_EDGES:
|
|
break
|
|
traversed += 1
|
|
if link.target_id in visited:
|
|
continue
|
|
visited.add(link.target_id)
|
|
target = current(link.target_id)
|
|
if target is None:
|
|
continue
|
|
target_path = (*path, target.id)
|
|
score = direct[seed.id] * 0.5 ** depth
|
|
previous = graph.get(target.id)
|
|
if previous is None or (-score, target_path) < (-previous[0], previous[1]):
|
|
graph[target.id] = (score, target_path)
|
|
next_frontier.append((target, target_path))
|
|
frontier = next_frontier
|
|
|
|
ranked = []
|
|
for identity in direct.keys() | graph.keys():
|
|
graph_score, path = graph.get(identity, (0, (identity,)))
|
|
ranked.append(RecalledCard(cache[identity], direct.get(identity, 0) + graph_score, path))
|
|
return sorted(ranked, key=lambda result: (-result.score, result.card.id))[:top]
|