refactor(harness): renaming prodotto tht (Onda -1)
Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.
Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
implementation', 'the DWH')
Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).
.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).
Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
This commit is contained in:
@@ -0,0 +1,101 @@
|
||||
import re
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from tht.evidence.model import EvidenceDoc
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
MAX_EXAMPLES_IN_RECORD = 5
|
||||
|
||||
|
||||
class VectorRecord(BaseModel):
|
||||
id: str
|
||||
kind: str # evidence | schema_table | schema_column
|
||||
ref: str # file/chiave canonica di provenienza
|
||||
title: str
|
||||
content: str
|
||||
metadata: dict = {}
|
||||
|
||||
|
||||
def split_markdown(text: str, max_chars: int) -> list[str]:
|
||||
"""Spezza un markdown: intero se sta nel limite, altrimenti per heading '##',
|
||||
e in ultima istanza per accumulo greedy di righe."""
|
||||
if len(text) <= max_chars:
|
||||
return [text]
|
||||
parts = re.split(r"(?=^## )", text, flags=re.MULTILINE)
|
||||
chunks: list[str] = []
|
||||
for part in parts:
|
||||
part = part.strip("\n")
|
||||
if not part:
|
||||
continue
|
||||
if len(part) <= max_chars:
|
||||
chunks.append(part)
|
||||
continue
|
||||
current: list[str] = []
|
||||
size = 0
|
||||
for line in part.splitlines():
|
||||
if size + len(line) > max_chars and current:
|
||||
chunks.append("\n".join(current))
|
||||
current, size = [], 0
|
||||
current.append(line)
|
||||
size += len(line) + 1
|
||||
if current:
|
||||
chunks.append("\n".join(current))
|
||||
return chunks
|
||||
|
||||
|
||||
def evidence_records(docs: list[EvidenceDoc], max_chunk_chars: int) -> list[VectorRecord]:
|
||||
"""Record per tutte le evidence presenti: la sola presenza basta a indicizzarle."""
|
||||
records: list[VectorRecord] = []
|
||||
for doc in docs:
|
||||
content = f"{doc.title}\n\n{doc.body}"
|
||||
for i, chunk in enumerate(split_markdown(content, max_chunk_chars)):
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"evidence:{doc.id}:{i}",
|
||||
kind="evidence",
|
||||
ref=str(doc.path) if doc.path else doc.id,
|
||||
title=doc.title,
|
||||
content=chunk,
|
||||
metadata={
|
||||
"status": doc.status, "tier": doc.tier,
|
||||
"tables": doc.tables, "concepts": doc.concepts,
|
||||
},
|
||||
)
|
||||
)
|
||||
return records
|
||||
|
||||
|
||||
def schema_records(physical: PhysicalSchema, annotations: Annotations) -> list[VectorRecord]:
|
||||
"""Un record per tabella e uno per colonna, da mschema (physical + annotations)."""
|
||||
records: list[VectorRecord] = []
|
||||
for table_name, table in physical.tables.items():
|
||||
ann_t = annotations.tables.get(table_name)
|
||||
t_desc = (ann_t.description if ann_t and ann_t.description else table.comment)
|
||||
t_concepts = ann_t.concepts if ann_t else []
|
||||
lines = [f"Tabella {table_name}", t_desc]
|
||||
if t_concepts:
|
||||
lines.append("Concetti: " + ", ".join(t_concepts))
|
||||
lines.append("Colonne: " + ", ".join(table.columns))
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"schema_table:{table_name}", kind="schema_table", ref=table_name,
|
||||
title=table_name, content="\n".join(filter(None, lines)),
|
||||
)
|
||||
)
|
||||
for column_name, column in table.columns.items():
|
||||
ann_c = ann_t.columns.get(column_name) if ann_t else None
|
||||
c_desc = (ann_c.description if ann_c and ann_c.description else column.comment)
|
||||
lines = [f"Colonna {table_name}.{column_name} ({column.type})", c_desc]
|
||||
if ann_c and ann_c.synonyms:
|
||||
lines.append("Sinonimi: " + ", ".join(ann_c.synonyms))
|
||||
if column.examples:
|
||||
lines.append("Esempi: " + ", ".join(column.examples[:MAX_EXAMPLES_IN_RECORD]))
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"schema_column:{table_name}.{column_name}", kind="schema_column",
|
||||
ref=f"{table_name}.{column_name}", title=f"{table_name}.{column_name}",
|
||||
content="\n".join(filter(None, lines)),
|
||||
)
|
||||
)
|
||||
return records
|
||||
Reference in New Issue
Block a user