Files
ThothII/harness/nsp/vectorstore/records.py
T
marcopan 796a39d893 feat(harness): port vectorstore dual-key + reader RPC (D11, §5.4)
Ports vectorstore/{rest_client,rest_writer,store,reader,embeddings,records},
evidence/model (leaf dep of records), and cli/_guards (require_vector_write_allowed
workstation write-guard). Renamed psdwp3->nsp, verbatim.

VectorRestClient gains an api_key property so reader/writer clients carry their
distinct keys visibly (spec D11: vector_reader / vector_writer on the same endpoint).

scripts/create_vector_reader_rpc.sql is NEW: the reader RPCs (search_similar,
list_tables) lived server-side in Supabase and were never versioned. Authored now
mirroring the writer allowlist pattern (table allowlist, security definer, revoke
from anon/authenticated, grant to vector_reader only). Writer RPC ported verbatim.

L1: test_vector_dual_key (7 tests) pins the dual-key construction + the workstation
write-guard (exit 4 without writer key).
2026-06-26 22:55:40 +02:00

102 lines
3.9 KiB
Python

import re
from pydantic import BaseModel
from nsp.evidence.model import EvidenceDoc
from nsp.mschema.models import Annotations, PhysicalSchema
MAX_EXAMPLES_IN_RECORD = 5
class VectorRecord(BaseModel):
id: str
kind: str # evidence | schema_table | schema_column
ref: str # file/chiave canonica di provenienza
title: str
content: str
metadata: dict = {}
def split_markdown(text: str, max_chars: int) -> list[str]:
"""Spezza un markdown: intero se sta nel limite, altrimenti per heading '##',
e in ultima istanza per accumulo greedy di righe."""
if len(text) <= max_chars:
return [text]
parts = re.split(r"(?=^## )", text, flags=re.MULTILINE)
chunks: list[str] = []
for part in parts:
part = part.strip("\n")
if not part:
continue
if len(part) <= max_chars:
chunks.append(part)
continue
current: list[str] = []
size = 0
for line in part.splitlines():
if size + len(line) > max_chars and current:
chunks.append("\n".join(current))
current, size = [], 0
current.append(line)
size += len(line) + 1
if current:
chunks.append("\n".join(current))
return chunks
def evidence_records(docs: list[EvidenceDoc], max_chunk_chars: int) -> list[VectorRecord]:
"""Record per tutte le evidence presenti: la sola presenza basta a indicizzarle."""
records: list[VectorRecord] = []
for doc in docs:
content = f"{doc.title}\n\n{doc.body}"
for i, chunk in enumerate(split_markdown(content, max_chunk_chars)):
records.append(
VectorRecord(
id=f"evidence:{doc.id}:{i}",
kind="evidence",
ref=str(doc.path) if doc.path else doc.id,
title=doc.title,
content=chunk,
metadata={
"status": doc.status, "tier": doc.tier,
"tables": doc.tables, "concepts": doc.concepts,
},
)
)
return records
def schema_records(physical: PhysicalSchema, annotations: Annotations) -> list[VectorRecord]:
"""Un record per tabella e uno per colonna, da mschema (physical + annotations)."""
records: list[VectorRecord] = []
for table_name, table in physical.tables.items():
ann_t = annotations.tables.get(table_name)
t_desc = (ann_t.description if ann_t and ann_t.description else table.comment)
t_concepts = ann_t.concepts if ann_t else []
lines = [f"Tabella {table_name}", t_desc]
if t_concepts:
lines.append("Concetti: " + ", ".join(t_concepts))
lines.append("Colonne: " + ", ".join(table.columns))
records.append(
VectorRecord(
id=f"schema_table:{table_name}", kind="schema_table", ref=table_name,
title=table_name, content="\n".join(filter(None, lines)),
)
)
for column_name, column in table.columns.items():
ann_c = ann_t.columns.get(column_name) if ann_t else None
c_desc = (ann_c.description if ann_c and ann_c.description else column.comment)
lines = [f"Colonna {table_name}.{column_name} ({column.type})", c_desc]
if ann_c and ann_c.synonyms:
lines.append("Sinonimi: " + ", ".join(ann_c.synonyms))
if column.examples:
lines.append("Esempi: " + ", ".join(column.examples[:MAX_EXAMPLES_IN_RECORD]))
records.append(
VectorRecord(
id=f"schema_column:{table_name}.{column_name}", kind="schema_column",
ref=f"{table_name}.{column_name}", title=f"{table_name}.{column_name}",
content="\n".join(filter(None, lines)),
)
)
return records