feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s
Publish documentation / publish (push) Successful in 2m12s
This commit is contained in:
@@ -32,7 +32,7 @@ def build_vector_store(cfg: Config, *, require_write: bool = False) -> VectorSto
|
||||
case "qdrant":
|
||||
return QdrantVectorStore(
|
||||
base_url=resource.base_url,
|
||||
collection=resource.collection,
|
||||
collections=resource.collections,
|
||||
workspace_id=cfg._workspace_id,
|
||||
workspace_revision=cfg._workspace_revision,
|
||||
expected_dimension=cfg.embeddings.dim if cfg.embeddings is not None else None,
|
||||
|
||||
@@ -5,7 +5,7 @@ from __future__ import annotations
|
||||
from tht.ports.vector import VectorStoreError
|
||||
|
||||
COLLECTION_KINDS = {
|
||||
"schema_records": {"schema_table", "schema_column"},
|
||||
"schema_records": {"schema_table", "schema_column", "schema_relationship"},
|
||||
"evidence": {"evidence"},
|
||||
"memory": {"memory", "solved_question"},
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
"""Qdrant-backed vector store for one workspace-owned semantic collection."""
|
||||
"""Qdrant-backed vector store with separate reference and memory lifecycles."""
|
||||
|
||||
import re
|
||||
from collections.abc import Callable
|
||||
@@ -25,6 +25,9 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
|
||||
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
|
||||
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
|
||||
_BM25_LANGUAGES = frozenset({"english", "italian"})
|
||||
_REFERENCE_KINDS = frozenset({
|
||||
"schema_table", "schema_column", "schema_relationship", "evidence",
|
||||
})
|
||||
_KEYWORD_INDEXES = (
|
||||
|
||||
"content_hash",
|
||||
@@ -58,7 +61,8 @@ class QdrantVectorStore:
|
||||
self,
|
||||
*,
|
||||
base_url: str,
|
||||
collection: str,
|
||||
collections: dict[str, str] | None = None,
|
||||
collection: str | None = None,
|
||||
workspace_id: str,
|
||||
workspace_revision: str | None = None,
|
||||
expected_dimension: int | None = None,
|
||||
@@ -68,7 +72,17 @@ class QdrantVectorStore:
|
||||
read_timeout: float = 10.0,
|
||||
):
|
||||
self._base_url = base_url.rstrip("/")
|
||||
self._collection = collection
|
||||
legacy_constructor = collections is None and collection is not None
|
||||
if legacy_constructor:
|
||||
collections = {"reference": collection, "memory": collection}
|
||||
if collections is None or set(collections) != {"reference", "memory"}:
|
||||
raise VectorStoreError("Qdrant collections must define reference and memory")
|
||||
if not legacy_constructor and collections["reference"] == collections["memory"]:
|
||||
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
|
||||
self._collections = dict(collections)
|
||||
self._split_collections = collections["reference"] != collections["memory"]
|
||||
self._legacy_collection = workspace_id
|
||||
self._legacy_checked = False
|
||||
self._workspace_id = workspace_id
|
||||
self._workspace_revision = None
|
||||
self._workspace_revision = workspace_revision
|
||||
@@ -90,7 +104,10 @@ class QdrantVectorStore:
|
||||
|
||||
def health(self) -> VectorHealth:
|
||||
try:
|
||||
info = self._ensure_collection(strict=False)
|
||||
infos = {
|
||||
name: self._ensure_collection(collection, strict=False)
|
||||
for name, collection in self._collections.items()
|
||||
}
|
||||
except VectorStoreError as exc:
|
||||
return VectorHealth(
|
||||
ok=False,
|
||||
@@ -105,8 +122,10 @@ class QdrantVectorStore:
|
||||
bm25_compatible=None,
|
||||
)
|
||||
|
||||
dimension = info["config"]["params"]["vectors"]["size"]
|
||||
dimensions = (dimension,)
|
||||
dimensions = tuple(sorted({
|
||||
info["config"]["params"]["vectors"]["size"]
|
||||
for info in infos.values()
|
||||
}))
|
||||
compatible = (
|
||||
None if self._expected_dimension is None else dimensions == (self._expected_dimension,)
|
||||
)
|
||||
@@ -119,7 +138,7 @@ class QdrantVectorStore:
|
||||
expected_dimension=self._expected_dimension,
|
||||
observed_dimensions=dimensions,
|
||||
dimension_compatible=compatible,
|
||||
bm25_compatible=self._bm25_compatible(info),
|
||||
bm25_compatible=self._bm25_compatible(infos["reference"]),
|
||||
)
|
||||
|
||||
def search(
|
||||
@@ -139,6 +158,21 @@ class QdrantVectorStore:
|
||||
allowed_record_kinds = self._allowed_record_kinds(collections, kinds)
|
||||
if not allowed_record_kinds:
|
||||
return []
|
||||
physical_collections = {
|
||||
self._physical_collection_for_kind(kind) for kind in allowed_record_kinds
|
||||
}
|
||||
if len(physical_collections) != 1:
|
||||
hits: list[VectorHit] = []
|
||||
for collection in collections:
|
||||
nested_kinds = sorted(set(allowed_record_kinds) & COLLECTION_KINDS[collection])
|
||||
if nested_kinds:
|
||||
hits.extend(self.search(
|
||||
[collection], embedding, limit=limit, kinds=nested_kinds,
|
||||
metadata_filter=metadata_filter, query_text=query_text,
|
||||
query_language=query_language, retrieval_mode=retrieval_mode,
|
||||
))
|
||||
return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit]
|
||||
physical_collection = physical_collections.pop()
|
||||
filter_must = self._workspace_filter()
|
||||
filter_must.extend(self._revision_filter(allowed_record_kinds))
|
||||
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
|
||||
@@ -193,7 +227,7 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"vector": embedding,
|
||||
"limit": limit,
|
||||
@@ -206,11 +240,11 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(strict=False, require_bm25=True)
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
shared_filter = {"must": filter_must}
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"query": self._bm25_document(query_text, query_language),
|
||||
"using": "bm25",
|
||||
@@ -224,7 +258,7 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Evidence hybrid query text is required")
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"vector": embedding,
|
||||
"limit": limit,
|
||||
@@ -237,11 +271,11 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
|
||||
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(strict=False, require_bm25=True)
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
shared_filter = {"must": filter_must}
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/query",
|
||||
f"/collections/{physical_collection}/points/query",
|
||||
{
|
||||
"prefetch": [
|
||||
{"query": embedding, "limit": limit * 2, "filter": shared_filter},
|
||||
@@ -266,7 +300,9 @@ class QdrantVectorStore:
|
||||
def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]:
|
||||
validate_collection(collection)
|
||||
validate_collection_kinds(collection, kinds)
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
points = self._scroll(
|
||||
physical_collection,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(kinds),
|
||||
@@ -287,7 +323,14 @@ class QdrantVectorStore:
|
||||
|
||||
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
|
||||
validate_collection(collection)
|
||||
# The first write with the split configuration is also the upgrade cutover. This keeps
|
||||
# existing runtime memory reachable even when the operator reruns preprocessing without
|
||||
# invoking the explicit clear operation first.
|
||||
if self._split_collections:
|
||||
self._migrate_legacy_memory()
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
self._ensure_collection(
|
||||
physical_collection,
|
||||
strict=True,
|
||||
require_bm25=any(record.sparse_text is not None for record in records),
|
||||
)
|
||||
@@ -310,7 +353,7 @@ class QdrantVectorStore:
|
||||
self._workspace_id,
|
||||
semantic_kind,
|
||||
write_record.record.id,
|
||||
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "evidence") else None,
|
||||
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "schema_relationship", "evidence") else None,
|
||||
),
|
||||
"vector": vector,
|
||||
"payload": qdrant_payload(
|
||||
@@ -326,7 +369,7 @@ class QdrantVectorStore:
|
||||
for start in range(0, len(points), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/points?wait=true",
|
||||
f"/collections/{physical_collection}/points?wait=true",
|
||||
{"points": points[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
return len(records)
|
||||
@@ -334,15 +377,16 @@ class QdrantVectorStore:
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int:
|
||||
validate_collection(collection)
|
||||
validate_collection_kinds(collection, kinds)
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
must = [
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(kinds),
|
||||
{"key": "record_kind", "match": {"any": sorted(kinds)}},
|
||||
]
|
||||
before = len(self._scroll(must))
|
||||
before = len(self._scroll(physical_collection, must))
|
||||
self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/delete?wait=true",
|
||||
f"/collections/{physical_collection}/points/delete?wait=true",
|
||||
{"filter": {"must": must}},
|
||||
)
|
||||
return before
|
||||
@@ -353,6 +397,7 @@ class QdrantVectorStore:
|
||||
if _WORKSPACE.fullmatch(workspace_id) is None:
|
||||
raise VectorStoreError("Invalid Evidence workspace namespace")
|
||||
self._require_bound_workspace(workspace_id)
|
||||
physical_collection = self._collections["reference"]
|
||||
must = [
|
||||
*self._workspace_filter(),
|
||||
{"key": "kind", "match": {"value": "evidence"}},
|
||||
@@ -360,11 +405,11 @@ class QdrantVectorStore:
|
||||
{"key": "vector_generation", "match": {"value": generation}},
|
||||
]
|
||||
before = len(
|
||||
self._scroll(must)
|
||||
self._scroll(physical_collection, must)
|
||||
)
|
||||
self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/delete?wait=true",
|
||||
f"/collections/{physical_collection}/points/delete?wait=true",
|
||||
{"filter": {"must": must}},
|
||||
)
|
||||
return before
|
||||
@@ -375,7 +420,9 @@ class QdrantVectorStore:
|
||||
if _WORKSPACE.fullmatch(workspace_id) is None:
|
||||
raise VectorStoreError("Invalid Evidence workspace namespace")
|
||||
self._require_bound_workspace(workspace_id)
|
||||
physical_collection = self._collections["reference"]
|
||||
points = self._scroll(
|
||||
physical_collection,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
{"key": "kind", "match": {"value": "evidence"}},
|
||||
@@ -394,6 +441,68 @@ class QdrantVectorStore:
|
||||
def _workspace_filter(self) -> list[dict]:
|
||||
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
|
||||
|
||||
def clear_reference(self) -> bool:
|
||||
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
|
||||
legacy_deleted = self._migrate_legacy_memory()
|
||||
collection = self._collections["reference"]
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
return legacy_deleted
|
||||
self._call("DELETE", f"/collections/{collection}", None)
|
||||
return True
|
||||
|
||||
def _migrate_legacy_memory(self) -> bool:
|
||||
"""Preserve memory from the pre-split collection before retiring it."""
|
||||
legacy = self._legacy_collection
|
||||
if (
|
||||
self._legacy_checked
|
||||
or not self._split_collections
|
||||
or legacy in self._collections.values()
|
||||
):
|
||||
return False
|
||||
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
|
||||
if response is None:
|
||||
self._legacy_checked = True
|
||||
return False
|
||||
memory = self._collections["memory"]
|
||||
self._ensure_collection(memory, strict=True, allow_create=True)
|
||||
points = self._scroll(
|
||||
legacy,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(["memory", "solved_question"]),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
],
|
||||
with_vector=True,
|
||||
)
|
||||
migrated = []
|
||||
for point in points:
|
||||
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
if not isinstance(point.get("payload"), dict):
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
migrated.append({
|
||||
"id": point["id"],
|
||||
"vector": point["vector"],
|
||||
"payload": point["payload"],
|
||||
})
|
||||
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{memory}/points?wait=true",
|
||||
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
self._call("DELETE", f"/collections/{legacy}", None)
|
||||
self._legacy_checked = True
|
||||
return True
|
||||
|
||||
def _physical_collection_for_logical(self, collection: str) -> str:
|
||||
validate_collection(collection)
|
||||
return self._collections["memory" if collection == "memory" else "reference"]
|
||||
|
||||
def _physical_collection_for_kind(self, kind: str) -> str:
|
||||
return self._collections["reference" if kind in _REFERENCE_KINDS else "memory"]
|
||||
|
||||
@staticmethod
|
||||
def _bm25_document(text: str, language: str) -> dict:
|
||||
return {
|
||||
@@ -405,7 +514,7 @@ class QdrantVectorStore:
|
||||
def _revision_filter(self, kinds: list[str]) -> list[dict]:
|
||||
if self._workspace_revision is None:
|
||||
return []
|
||||
if not any(kind in ("schema_table", "schema_column", "evidence") for kind in kinds):
|
||||
if not any(kind in ("schema_table", "schema_column", "schema_relationship", "evidence") for kind in kinds):
|
||||
return []
|
||||
return [{"key": "workspace_revision", "match": {"value": self._workspace_revision}}]
|
||||
|
||||
@@ -450,25 +559,30 @@ class QdrantVectorStore:
|
||||
bm25 = sparse_vectors.get("bm25")
|
||||
return isinstance(bm25, dict) and bm25.get("modifier") == "idf"
|
||||
|
||||
def _ensure_collection(self, *, strict: bool, require_bm25: bool = False) -> dict | None:
|
||||
response = self._call("GET", f"/collections/{self._collection}", None, allow_missing=True)
|
||||
def _ensure_collection(
|
||||
self, collection: str, *, strict: bool, require_bm25: bool = False,
|
||||
allow_create: bool = False,
|
||||
) -> dict | None:
|
||||
created = False
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
if not strict:
|
||||
raise VectorStoreError("Qdrant collection is missing")
|
||||
if self._collection_lifecycle == "require_existing":
|
||||
if self._collection_lifecycle == "require_existing" and not allow_create:
|
||||
raise VectorStoreError("semantic_index_incompatible")
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}",
|
||||
f"/collections/{collection}",
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
|
||||
)
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/index",
|
||||
f"/collections/{collection}/index",
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
response = self._call("GET", f"/collections/{self._collection}", None)
|
||||
created = True
|
||||
response = self._call("GET", f"/collections/{collection}", None)
|
||||
result = response.get("result") if isinstance(response, dict) else None
|
||||
config = result.get("config", {}).get("params", {}).get("vectors") if isinstance(result, dict) else None
|
||||
if not isinstance(config, dict):
|
||||
@@ -484,29 +598,38 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Qdrant collection configuration mismatch")
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
if field_name not in result.get("payload_schema", {}):
|
||||
# A successful index-creation response can precede visibility in the
|
||||
# collection-info payload. The newly-created collection is already safe to
|
||||
# use; later readiness checks will validate the asynchronously published
|
||||
# indexes. Existing collections still follow the strict lifecycle policy.
|
||||
if created:
|
||||
continue
|
||||
if strict and self._collection_lifecycle == "require_existing":
|
||||
raise VectorStoreError("semantic_index_incompatible")
|
||||
if not strict:
|
||||
raise VectorStoreError("Qdrant collection payload indexes mismatch")
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{self._collection}/index",
|
||||
f"/collections/{collection}/index",
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
if require_bm25 and not self._bm25_compatible(result):
|
||||
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
|
||||
return result
|
||||
|
||||
def _scroll(self, must: list[dict]) -> list[dict]:
|
||||
def _scroll(
|
||||
self, collection: str, must: list[dict], *, with_vector: bool = False
|
||||
) -> list[dict]:
|
||||
points: list[dict] = []
|
||||
offset = None
|
||||
seen_offsets = set()
|
||||
while True:
|
||||
response = self._call(
|
||||
"POST",
|
||||
f"/collections/{self._collection}/points/scroll",
|
||||
f"/collections/{collection}/points/scroll",
|
||||
{
|
||||
"with_payload": True,
|
||||
"with_vector": with_vector,
|
||||
"limit": 10000,
|
||||
"filter": {"must": must},
|
||||
"offset": offset,
|
||||
|
||||
+41
-10
@@ -3,7 +3,7 @@ from pathlib import Path
|
||||
from tht.cli.schema_cmd import physical_path
|
||||
|
||||
|
||||
def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
def _extract_lsh_values(dwh, physical, annotations, limit, eligibility_cfg=None):
|
||||
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
|
||||
from tht.mschema.eligibility import effective_eligibility
|
||||
|
||||
@@ -12,7 +12,16 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
table_ann = annotations.tables.get(table_name)
|
||||
for column_name, column in table.columns.items():
|
||||
ann_col = table_ann.columns.get(column_name) if table_ann else None
|
||||
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
|
||||
from_catalog = column.eligibility_reason in {"catalog", "sensitive"}
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
if from_catalog and column.eligibility_reason == "sensitive":
|
||||
continue
|
||||
if from_catalog and eligibility_cfg is not None and (
|
||||
column_name.lower() in {name.lower() for name in eligibility_cfg.ignore_columns}
|
||||
):
|
||||
continue
|
||||
if not from_catalog and not effective_eligibility(column, ann_col)[0]:
|
||||
continue
|
||||
try:
|
||||
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
|
||||
@@ -20,6 +29,21 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
|
||||
continue
|
||||
vals = [str(value) for value in distinct.values if value not in (None, "")]
|
||||
if from_catalog and eligibility_cfg is not None:
|
||||
from tht.mschema.eligibility import classify_column
|
||||
|
||||
lengths = [len(value) for value in vals]
|
||||
eligible, reason = classify_column(
|
||||
column.type,
|
||||
column.is_enum,
|
||||
(sum(lengths) / len(lengths)) if lengths else None,
|
||||
max(lengths) if lengths else None,
|
||||
eligibility_cfg,
|
||||
)
|
||||
column.eligible = eligible
|
||||
column.eligibility_reason = reason
|
||||
if not eligible:
|
||||
continue
|
||||
if vals:
|
||||
values.setdefault(table_name, {})[column_name] = vals
|
||||
if distinct.truncated:
|
||||
@@ -33,18 +57,25 @@ def build_lsh_artifacts(
|
||||
):
|
||||
"""Run the existing LSH extraction/build algorithm and persist its outputs."""
|
||||
from tht.adapters.factory import build_dwh
|
||||
from tht.cli.schema_cmd import annotations_path
|
||||
from tht.lshindex import build_index, save_index
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
phys_file = physical_file or physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations = Annotations.from_yaml(annotations_path(cfg))
|
||||
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
|
||||
from tht.mschema.context import load_schema_context
|
||||
|
||||
context = load_schema_context(cfg)
|
||||
physical, annotations = context.physical, context.annotations
|
||||
else:
|
||||
from tht.cli.schema_cmd import annotations_path
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
phys_file = physical_file or physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations = Annotations.from_yaml(annotations_path(cfg))
|
||||
target = dwh if dwh is not None else build_dwh(cfg)
|
||||
values, skipped, truncated = _extract_lsh_values(
|
||||
target, physical, annotations, cfg.lsh.max_values_per_column
|
||||
target, physical, annotations, cfg.lsh.max_values_per_column, cfg.eligibility
|
||||
)
|
||||
lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose)
|
||||
save_index(
|
||||
|
||||
@@ -4,7 +4,10 @@ from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import stat
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
@@ -144,6 +147,8 @@ def run_dwh_from_config(
|
||||
workspace_root=workspace_root,
|
||||
config_fingerprint=binding["config_fingerprint"],
|
||||
input_fingerprint=binding["input_fingerprint"],
|
||||
catalog_database_id=binding.get("catalog_database_id"),
|
||||
metadata_content_revision=binding.get("metadata_content_revision"),
|
||||
introspect=lambda output: refresh_catalog(cfg, output_path=output),
|
||||
build_lsh=lambda physical, output: build_lsh_artifacts(
|
||||
cfg, physical_file=physical, output_dir=output
|
||||
@@ -155,6 +160,57 @@ def run_dwh_from_config(
|
||||
return pipeline.run(steps, resume_run_id=resume)
|
||||
|
||||
|
||||
def run_catalog_dwh_from_config(cfg):
|
||||
"""Publish Catalog-derived physical schema and LSH as one bound generation."""
|
||||
from tht.cli.lsh_cmd import build_lsh_artifacts
|
||||
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
|
||||
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
|
||||
|
||||
snapshot = load_catalog_metadata_snapshot(
|
||||
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
|
||||
)
|
||||
binding = config_dwh_binding(cfg)
|
||||
workspace_root = cfg.paths.artifacts.parent
|
||||
# Catalog preprocessing is a replace-in-place operation. Its DWH/LSH output is wholly
|
||||
# derived, there is no supported concurrent runtime, and a failed Clear may have left an
|
||||
# older binding behind. Start from an empty owned generation root so a retry can always
|
||||
# rebuild the current Catalog revision instead of deadlocking on the stale OWNER marker.
|
||||
_remove_owned_derived_path(workspace_root, workspace_root / ".tht-dwh")
|
||||
_remove_owned_derived_path(workspace_root, cfg.paths.artifacts / "mschema" / "physical.yaml")
|
||||
_remove_owned_derived_path(workspace_root, cfg.paths.indexes / "lsh")
|
||||
observed: dict[str, object] = {}
|
||||
|
||||
def materialize_physical(output: Path):
|
||||
physical, _annotations, _relationships = snapshot.to_schema_inputs()
|
||||
physical.to_yaml(output)
|
||||
return physical
|
||||
|
||||
def materialize_lsh(physical: Path, output: Path):
|
||||
result = build_lsh_artifacts(cfg, physical_file=physical, output_dir=output)
|
||||
observed["result"] = result
|
||||
return result
|
||||
|
||||
report = DwhPreprocessPipeline(
|
||||
workspace_id=str(binding["workspace_id"]),
|
||||
workspace_root=workspace_root,
|
||||
config_fingerprint=str(binding["config_fingerprint"]),
|
||||
input_fingerprint=str(binding["input_fingerprint"]),
|
||||
catalog_database_id=str(binding["catalog_database_id"]),
|
||||
metadata_content_revision=int(binding["metadata_content_revision"]),
|
||||
introspect=materialize_physical,
|
||||
build_lsh=materialize_lsh,
|
||||
lsh_filenames=(
|
||||
f"{cfg.database.db_schema}_lsh.pkl",
|
||||
f"{cfg.database.db_schema}_minhashes.pkl",
|
||||
f"{cfg.database.db_schema}_meta.json",
|
||||
),
|
||||
).run(("introspect", "lsh"))
|
||||
result = observed.get("result")
|
||||
if not isinstance(result, tuple) or len(result) != 4:
|
||||
raise RuntimeError("Catalog LSH publication did not complete")
|
||||
return report, result
|
||||
|
||||
|
||||
def _parse_dwh_steps(value: str) -> tuple[str, ...]:
|
||||
allowed = ("introspect", "lsh")
|
||||
steps = tuple(part.strip() for part in value.split(",") if part.strip())
|
||||
@@ -234,6 +290,85 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
|
||||
return pipeline.gc(workspace_root=corpus_root.parent, dry_run=dry_run)
|
||||
|
||||
|
||||
def _remove_owned_derived_path(workspace_root: Path, target: Path) -> bool:
|
||||
"""Remove one generated path without following links or escaping the workspace."""
|
||||
root = workspace_root.resolve()
|
||||
resolved = target.resolve(strict=False)
|
||||
if not resolved.is_relative_to(root):
|
||||
raise RuntimeError("derived cleanup target escapes the workspace")
|
||||
try:
|
||||
info = target.lstat()
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
if stat.S_ISLNK(info.st_mode) or info.st_uid != os.getuid():
|
||||
raise RuntimeError("derived cleanup target is unsafe")
|
||||
if stat.S_ISDIR(info.st_mode):
|
||||
shutil.rmtree(target)
|
||||
elif stat.S_ISREG(info.st_mode):
|
||||
target.unlink()
|
||||
else:
|
||||
raise RuntimeError("derived cleanup target is unsafe")
|
||||
return True
|
||||
|
||||
|
||||
def clear_from_config(config: Path) -> dict[str, int]:
|
||||
"""Clear workspace reference vectors and local preprocessing derivatives."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "preprocess clear")
|
||||
workspace_root = cfg.paths.artifacts.parent
|
||||
vector_store = build_vector_store(cfg, require_write=True)
|
||||
reference_deleted = int(vector_store.clear_reference())
|
||||
paths = [
|
||||
workspace_root / ".tht-dwh",
|
||||
workspace_root / ".tht-jobs",
|
||||
workspace_root / "corpus",
|
||||
cfg.paths.artifacts / "mschema" / "physical.yaml",
|
||||
cfg.paths.indexes / "lsh",
|
||||
]
|
||||
if cfg.paths.catalog_metadata_snapshot is not None:
|
||||
paths.append(cfg.paths.catalog_metadata_snapshot)
|
||||
removed = sum(int(_remove_owned_derived_path(workspace_root, path)) for path in paths)
|
||||
return {"referenceCollections": reference_deleted, "derivedPaths": removed}
|
||||
|
||||
|
||||
@preprocess_app.command("clear", hidden=True)
|
||||
def clear_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Clear replaceable preprocessing output while preserving workspace memory."""
|
||||
try:
|
||||
counts = clear_from_config(config)
|
||||
except Exception: # noqa: BLE001 - do not disclose paths, endpoints, or credentials
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"code": "preprocessing_clear_failed",
|
||||
"operation": "preprocess_clear",
|
||||
"error": "Preprocessing clear failed",
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.secho("ERRORE: preprocessing clear failed", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "succeeded",
|
||||
"code": "ok",
|
||||
"operation": "preprocess_clear",
|
||||
"counts": counts,
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.echo("OK: reference vectors and LSH cleared; memory preserved")
|
||||
|
||||
|
||||
@preprocess_app.command("evidence")
|
||||
def evidence_cmd(
|
||||
action: str | None = typer.Argument(None),
|
||||
@@ -359,3 +494,71 @@ def dwh_cmd(
|
||||
typer.secho(f"ERRORE: run={result.run_id} DWH preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
if result.status != "succeeded":
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@preprocess_app.command("catalog", hidden=True)
|
||||
def catalog_cmd(
|
||||
catalog_metadata: Path = typer.Option(..., "--catalog-metadata"),
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Build current LSH and schema vectors from one immutable Catalog snapshot."""
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.vector_cmd import index_catalog_schema, require_vector_cfg
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
cfg = cfg.model_copy(
|
||||
update={
|
||||
"paths": cfg.paths.model_copy(
|
||||
update={"catalog_metadata_snapshot": catalog_metadata}
|
||||
)
|
||||
}
|
||||
)
|
||||
require_vector_write_allowed(cfg, "preprocess catalog")
|
||||
require_vector_cfg(cfg)
|
||||
try:
|
||||
report, (minhashes, skipped, truncated, _values) = run_catalog_dwh_from_config(cfg)
|
||||
if report.status != "succeeded":
|
||||
raise RuntimeError("Catalog DWH preprocessing failed")
|
||||
stats, counts, snapshot_path = index_catalog_schema(cfg)
|
||||
except Exception: # noqa: BLE001 - public output must never disclose endpoints or SQL
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"code": "catalog_preprocessing_failed",
|
||||
"operation": "preprocess_catalog",
|
||||
"error": "Catalog preprocessing failed",
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.secho("ERRORE: Catalog preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "succeeded",
|
||||
"code": "ok",
|
||||
"operation": "preprocess_catalog",
|
||||
"artifactIdentities": [
|
||||
{
|
||||
"kind": "catalog_metadata_snapshot",
|
||||
"digest": "sha256:" + hashlib.sha256(snapshot_path.read_bytes()).hexdigest(),
|
||||
}
|
||||
],
|
||||
"counts": {
|
||||
**counts,
|
||||
"lshEntries": len(minhashes),
|
||||
"lshSkippedColumns": len(skipped),
|
||||
"lshTruncatedColumns": len(truncated),
|
||||
"added": stats.added,
|
||||
"deleted": stats.deleted,
|
||||
"unchanged": stats.unchanged,
|
||||
"updated": stats.updated,
|
||||
},
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
|
||||
else:
|
||||
typer.echo("OK: Catalog metadata, LSH and schema vectors rebuilt")
|
||||
|
||||
@@ -560,22 +560,14 @@ def render_cmd(
|
||||
),
|
||||
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
|
||||
) -> None:
|
||||
"""Serializza mschema (physical + annotations) nel formato richiesto."""
|
||||
"""Serialize the current PostgreSQL Catalog projection."""
|
||||
import json
|
||||
|
||||
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED,
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
context = load_schema_context(cfg, physical_file=phys_file)
|
||||
context = load_schema_context(cfg)
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
@@ -625,17 +617,12 @@ def columns_cmd(
|
||||
"""Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo."""
|
||||
import json as _json
|
||||
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
try:
|
||||
physical = load_schema_context(cfg).physical
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
|
||||
def _payload(name, tbl):
|
||||
return {
|
||||
|
||||
@@ -9,7 +9,7 @@ from tht.config import workspace_id_for_config
|
||||
|
||||
KIND_MAP = {
|
||||
"evidence": ["evidence"],
|
||||
"schema": ["schema_table", "schema_column"],
|
||||
"schema": ["schema_table", "schema_column", "schema_relationship"],
|
||||
"values": [], # solo LSH
|
||||
"formula": ["evidence"],
|
||||
}
|
||||
@@ -235,15 +235,8 @@ def search_cmd(
|
||||
from tht.mschema.render import to_mschema_text
|
||||
from tht.search import schema_tables
|
||||
|
||||
phys_file = dwh_snapshot.physical
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
schema_context = load_schema_context(cfg, physical_file=phys_file)
|
||||
schema_context = load_schema_context(cfg)
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
@@ -390,7 +383,7 @@ def pack_cmd(
|
||||
|
||||
workspace_id = workspace_id_for_config(cfg, config)
|
||||
validate_corpus_workspace(cfg, workspace_id)
|
||||
dwh_snapshot = _leased_dwh_snapshot(cfg, ctx)
|
||||
_leased_dwh_snapshot(cfg, ctx)
|
||||
require_vector_cfg(cfg)
|
||||
|
||||
tables: list[dict] = []
|
||||
@@ -414,12 +407,13 @@ def pack_cmd(
|
||||
|
||||
if vec is not None:
|
||||
descriptions: dict[str, str] = {}
|
||||
phys_file = dwh_snapshot.physical
|
||||
if phys_file.exists():
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
from tht.mschema.context import SchemaContextError, load_schema_context
|
||||
|
||||
phys = PhysicalSchema.from_yaml(phys_file)
|
||||
try:
|
||||
phys = load_schema_context(cfg).physical
|
||||
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
|
||||
except SchemaContextError:
|
||||
pass
|
||||
try:
|
||||
cand = combined_search(
|
||||
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
|
||||
|
||||
@@ -4,7 +4,7 @@ from pathlib import Path
|
||||
import typer
|
||||
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
|
||||
sql_app = typer.Typer(help="Validazione ed esecuzione controllata di SQL (read-only)")
|
||||
|
||||
@@ -17,16 +17,16 @@ def _read_sql(file: Path) -> str:
|
||||
|
||||
|
||||
def _load_physical_or_exit(cfg):
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
from tht.mschema.context import SchemaContextError, load_schema_context
|
||||
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
try:
|
||||
return load_schema_context(cfg).physical
|
||||
except SchemaContextError:
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
"ERRORE: Catalog Metadata Snapshot non disponibile. Esegui il preprocessing.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
return PhysicalSchema.from_yaml(phys_file)
|
||||
|
||||
|
||||
def require_action(cfg, action: str) -> None:
|
||||
|
||||
@@ -6,7 +6,7 @@ import typer
|
||||
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.vectorstore.store import SyncStats, content_hash
|
||||
|
||||
@@ -112,42 +112,14 @@ def index_schema_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
"""Replace the schema slice from the PostgreSQL Catalog projection."""
|
||||
from tht.ports.vector import VectorStoreError
|
||||
from tht.vectorstore.records import schema_records
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "vector index-schema")
|
||||
require_vector_cfg(cfg)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
message = f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`."
|
||||
if json_output:
|
||||
_emit_json({
|
||||
"code": "schema_missing",
|
||||
"error": "physical schema is missing",
|
||||
"operation": "index_schema",
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"workspaceId": cfg._workspace_id,
|
||||
"workspaceRevision": cfg._workspace_revision,
|
||||
})
|
||||
else:
|
||||
typer.secho(message, fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations_file = annotations_path(cfg)
|
||||
annotations = Annotations.from_yaml(annotations_file)
|
||||
records = schema_records(physical, annotations)
|
||||
try:
|
||||
stats = sync_canonical_records(
|
||||
"schema_records",
|
||||
records,
|
||||
store=build_vector_store(cfg, require_write=True),
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
stats, counts, snapshot_path = index_catalog_schema(cfg)
|
||||
except VectorStoreError as exc:
|
||||
code = str(exc)
|
||||
error = "semantic index incompatible" if code == "semantic_index_incompatible" else "schema indexing failed"
|
||||
@@ -167,17 +139,17 @@ def index_schema_cmd(
|
||||
if json_output:
|
||||
_emit_json({
|
||||
"artifactIdentities": [
|
||||
{"digest": _artifact_digest(annotations_file), "kind": "schema_annotations"},
|
||||
{"digest": _artifact_digest(phys_file), "kind": "physical_schema"},
|
||||
{"digest": _artifact_digest(snapshot_path), "kind": "catalog_metadata_snapshot"},
|
||||
],
|
||||
"code": "ok",
|
||||
"collection": cfg.vectors.collection,
|
||||
"collection": cfg.vectors.collections["reference"],
|
||||
"counts": {
|
||||
"added": stats.added,
|
||||
"columns": sum(len(table.columns) for table in physical.tables.values()),
|
||||
"columns": counts["columns"],
|
||||
"deleted": stats.deleted,
|
||||
"records": len(records),
|
||||
"tables": len(physical.tables),
|
||||
"records": counts["records"],
|
||||
"relationships": counts["relationships"],
|
||||
"tables": counts["tables"],
|
||||
"unchanged": stats.unchanged,
|
||||
"updated": stats.updated,
|
||||
},
|
||||
@@ -189,3 +161,32 @@ def index_schema_cmd(
|
||||
})
|
||||
return
|
||||
_print_stats(stats)
|
||||
|
||||
|
||||
def index_catalog_schema(cfg):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
|
||||
from tht.vectorstore.records import catalog_schema_records
|
||||
|
||||
snapshot_path = cfg.paths.catalog_metadata_snapshot
|
||||
if snapshot_path is None:
|
||||
raise ValueError("catalog metadata snapshot is not configured")
|
||||
snapshot = load_catalog_metadata_snapshot(snapshot_path, cfg._workspace_id)
|
||||
records = catalog_schema_records(snapshot)
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
deleted = store.delete_kinds(
|
||||
"schema_records", ["schema_table", "schema_column", "schema_relationship"]
|
||||
)
|
||||
stats = sync_canonical_records(
|
||||
"schema_records",
|
||||
records,
|
||||
store=store,
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
stats.deleted = deleted
|
||||
return stats, {
|
||||
"tables": len(snapshot.tables),
|
||||
"columns": sum(len(table.columns) for table in snapshot.tables),
|
||||
"relationships": len(snapshot.relationships),
|
||||
"records": len(records),
|
||||
}, snapshot_path
|
||||
|
||||
+48
-8
@@ -49,8 +49,8 @@ def canonical_effective_config_document(cfg) -> dict:
|
||||
else:
|
||||
raise ConfigError("unsupported DWH transport in canonical effective config")
|
||||
vectors = getattr(cfg, "vectors", None)
|
||||
collection = getattr(vectors, "collection", None) if vectors is not None else None
|
||||
if not collection:
|
||||
collections = getattr(vectors, "collections", None) if vectors is not None else None
|
||||
if not collections or set(collections) != {"reference", "memory"}:
|
||||
raise ConfigError("vector configuration is unavailable; cannot canonicalize effective config")
|
||||
embeddings = getattr(cfg, "embeddings", None)
|
||||
model = getattr(embeddings, "model", None) if embeddings is not None else None
|
||||
@@ -58,9 +58,16 @@ def canonical_effective_config_document(cfg) -> dict:
|
||||
if not model or not embed_dim:
|
||||
raise ConfigError("embedding configuration is unavailable; cannot canonicalize effective config")
|
||||
return {
|
||||
"schemaVersion": 1,
|
||||
"schemaVersion": 3,
|
||||
"dwh": dwh,
|
||||
"vector": {"collection": collection, "dimensions": int(embed_dim), "distance": "cosine"},
|
||||
"vector": {
|
||||
"collections": {
|
||||
"reference": collections["reference"],
|
||||
"memory": collections["memory"],
|
||||
},
|
||||
"dimensions": int(embed_dim),
|
||||
"distance": "cosine",
|
||||
},
|
||||
"embedding": {
|
||||
"id": getattr(embeddings, "id", None) or f"ollama/{model}",
|
||||
"model": model,
|
||||
@@ -311,9 +318,32 @@ class ThothVectorHttpConfig(BaseModel):
|
||||
class QdrantConfig(BaseModel):
|
||||
type: Literal["qdrant"]
|
||||
base_url: str
|
||||
collection: str = Field(min_length=1)
|
||||
collections: dict[Literal["reference", "memory"], str]
|
||||
collection_lifecycle: Literal["self_heal", "require_existing"] = "self_heal"
|
||||
|
||||
@model_validator(mode="before")
|
||||
@classmethod
|
||||
def accept_single_collection_test_fixture(cls, value: Any) -> Any:
|
||||
if isinstance(value, dict) and "collections" not in value and "collection" in value:
|
||||
translated = dict(value)
|
||||
collection = translated.pop("collection")
|
||||
translated["collections"] = {
|
||||
"reference": collection,
|
||||
"memory": f"{collection}-memory",
|
||||
}
|
||||
return translated
|
||||
return value
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_collections(self):
|
||||
if set(self.collections) != {"reference", "memory"}:
|
||||
raise ValueError("qdrant collections must define reference and memory")
|
||||
if any(not value for value in self.collections.values()):
|
||||
raise ValueError("qdrant collection names must not be empty")
|
||||
if self.collections["reference"] == self.collections["memory"]:
|
||||
raise ValueError("qdrant reference and memory collections must be distinct")
|
||||
return self
|
||||
|
||||
|
||||
VectorResourceConfig = Annotated[
|
||||
PgvectorDirectConfig | ThothVectorHttpConfig | QdrantConfig,
|
||||
@@ -334,6 +364,8 @@ class PathsConfig(BaseModel):
|
||||
# Runtime-only, backend-derived effective relationship snapshot. When present,
|
||||
# it is the exclusive FK source; it is not an authored workspace artifact.
|
||||
effective_relationships: Path | None = None
|
||||
# Runtime-only, backend-produced projection of the PostgreSQL Metadata Catalog.
|
||||
catalog_metadata_snapshot: Path | None = None
|
||||
|
||||
|
||||
class RuntimeIdentityConfig(BaseModel):
|
||||
@@ -686,6 +718,7 @@ def load_config(path: Path) -> Config:
|
||||
memory=cfg.paths.memory,
|
||||
annotations_root=cfg.paths.annotations_root,
|
||||
effective_relationships=cfg.paths.effective_relationships,
|
||||
catalog_metadata_snapshot=cfg.paths.catalog_metadata_snapshot,
|
||||
)
|
||||
}
|
||||
)
|
||||
@@ -829,8 +862,9 @@ def _validate_internal_vector_contract(raw: dict[str, Any], path: Path) -> None:
|
||||
engine = vector.get("engine")
|
||||
base_url = vector.get("base_url")
|
||||
collection = vector.get("collection")
|
||||
collections = vector.get("collections")
|
||||
lifecycle = vector.get("collection_lifecycle")
|
||||
allowed = {"engine", "base_url", "collection", "collection_lifecycle"}
|
||||
allowed = {"engine", "base_url", "collection", "collections", "collection_lifecycle"}
|
||||
if lifecycle is not None and lifecycle not in ("self_heal", "require_existing"):
|
||||
raise ConfigError(
|
||||
f"Configurazione non valida in {path}:\n"
|
||||
@@ -847,10 +881,16 @@ def _validate_internal_vector_contract(raw: dict[str, Any], path: Path) -> None:
|
||||
f"Configurazione non valida in {path}:\n"
|
||||
"resources.vector.engine deve essere 'qdrant'"
|
||||
)
|
||||
if not isinstance(collection, str) or not collection:
|
||||
valid_collections = (
|
||||
isinstance(collections, dict)
|
||||
and set(collections) == {"reference", "memory"}
|
||||
and all(isinstance(value, str) and value for value in collections.values())
|
||||
and collections["reference"] != collections["memory"]
|
||||
)
|
||||
if not valid_collections and (not isinstance(collection, str) or not collection):
|
||||
raise ConfigError(
|
||||
f"Configurazione non valida in {path}:\n"
|
||||
"resources.vector.collection deve essere valorizzato"
|
||||
"resources.vector.collections deve definire reference e memory"
|
||||
)
|
||||
if not _is_allowed_internal_qdrant_url(base_url):
|
||||
raise ConfigError(
|
||||
|
||||
@@ -39,8 +39,11 @@ def translate_legacy_config(raw: dict[str, Any]) -> tuple[dict[str, Any], bool]:
|
||||
translated_vectors: dict[str, Any] = {
|
||||
"type": "qdrant",
|
||||
"base_url": vector.get("base_url"),
|
||||
"collection": vector.get("collection"),
|
||||
}
|
||||
if isinstance(vector.get("collections"), dict):
|
||||
translated_vectors["collections"] = vector["collections"]
|
||||
else:
|
||||
translated_vectors["collection"] = vector.get("collection")
|
||||
if vector.get("collection_lifecycle") in ("self_heal", "require_existing"):
|
||||
translated_vectors["collection_lifecycle"] = vector["collection_lifecycle"]
|
||||
translated["vectors"] = translated_vectors
|
||||
|
||||
@@ -195,7 +195,7 @@ class ActiveEvidenceSearcher:
|
||||
query_language=None,
|
||||
):
|
||||
requested = set(kinds) if kinds is not None else {
|
||||
"schema_table", "schema_column", "evidence", "memory", "solved_question",
|
||||
"schema_table", "schema_column", "schema_relationship", "evidence", "memory", "solved_question",
|
||||
}
|
||||
include_evidence = "evidence" in requested
|
||||
other_kinds = sorted(requested - {"evidence"})
|
||||
|
||||
@@ -42,7 +42,7 @@ def _cleanup_snapshot_dirs() -> None:
|
||||
atexit.register(_cleanup_snapshot_dirs)
|
||||
|
||||
|
||||
def config_dwh_binding(cfg) -> dict[str, str]:
|
||||
def config_dwh_binding(cfg) -> dict[str, str | int]:
|
||||
workspace_id = getattr(cfg, "_workspace_id", None)
|
||||
config_source = getattr(cfg, "_config_source", None)
|
||||
if not isinstance(workspace_id, str) or not isinstance(config_source, str):
|
||||
@@ -72,14 +72,24 @@ def config_dwh_binding(cfg) -> dict[str, str]:
|
||||
else:
|
||||
config_fingerprint = fingerprint(cfg.model_dump_json())
|
||||
input_fingerprint = fingerprint(config_source)
|
||||
return {
|
||||
binding: dict[str, str | int] = {
|
||||
"workspace_id": workspace_id,
|
||||
"config_fingerprint": config_fingerprint,
|
||||
"input_fingerprint": input_fingerprint,
|
||||
}
|
||||
snapshot_path = getattr(getattr(cfg, "paths", None), "catalog_metadata_snapshot", None)
|
||||
if snapshot_path is not None:
|
||||
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
|
||||
|
||||
snapshot = load_catalog_metadata_snapshot(snapshot_path, workspace_id)
|
||||
binding.update({
|
||||
"catalog_database_id": snapshot.database_id,
|
||||
"metadata_content_revision": snapshot.metadata_content_revision,
|
||||
})
|
||||
return binding
|
||||
|
||||
|
||||
def _binding_digest(binding: dict[str, str]) -> str:
|
||||
def _binding_digest(binding: dict[str, str | int]) -> str:
|
||||
payload = json.dumps(binding, sort_keys=True, separators=(",", ":"))
|
||||
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
||||
|
||||
@@ -103,10 +113,19 @@ def _read_root_binding_fd(root_fd: int) -> dict[str, str]:
|
||||
os.close(fd)
|
||||
payload = json.loads(b"".join(chunks).decode("utf-8"))
|
||||
binding = payload["binding"]
|
||||
keys = set(binding) if isinstance(binding, dict) else set()
|
||||
base_keys = {"workspace_id", "config_fingerprint", "input_fingerprint"}
|
||||
catalog_keys = {"catalog_database_id", "metadata_content_revision"}
|
||||
if (
|
||||
payload.get("schema_version") != 1
|
||||
payload.get("schema_version") not in {1, 2}
|
||||
or not isinstance(binding, dict)
|
||||
or set(binding) != {"workspace_id", "config_fingerprint", "input_fingerprint"}
|
||||
or frozenset(keys) not in {frozenset(base_keys), frozenset(base_keys | catalog_keys)}
|
||||
or (keys == base_keys | catalog_keys and (
|
||||
not isinstance(binding["catalog_database_id"], str)
|
||||
or not binding["catalog_database_id"]
|
||||
or not isinstance(binding["metadata_content_revision"], int)
|
||||
or binding["metadata_content_revision"] < 0
|
||||
))
|
||||
or payload.get("binding_sha256") != _binding_digest(binding)
|
||||
):
|
||||
raise ValueError
|
||||
@@ -168,7 +187,7 @@ def _claim_or_validate_root_binding(
|
||||
finally:
|
||||
os.close(generations_fd)
|
||||
payload = {
|
||||
"schema_version": 1,
|
||||
"schema_version": 2 if "catalog_database_id" in binding else 1,
|
||||
"binding": binding,
|
||||
"binding_sha256": _binding_digest(binding),
|
||||
}
|
||||
@@ -557,6 +576,8 @@ class DwhPreprocessPipeline:
|
||||
workspace_root: Path,
|
||||
config_fingerprint: str,
|
||||
input_fingerprint: str,
|
||||
catalog_database_id: str | None = None,
|
||||
metadata_content_revision: int | None = None,
|
||||
introspect: Callable[[Path], object],
|
||||
build_lsh: Callable[[Path, Path], object],
|
||||
lsh_filenames: tuple[str, str, str] | None = None,
|
||||
@@ -569,6 +590,8 @@ class DwhPreprocessPipeline:
|
||||
self.workspace_root = workspace_root
|
||||
self.config_fingerprint = config_fingerprint
|
||||
self.input_fingerprint = input_fingerprint
|
||||
self.catalog_database_id = catalog_database_id
|
||||
self.metadata_content_revision = metadata_content_revision
|
||||
self.introspect = introspect
|
||||
self.build_lsh = build_lsh
|
||||
self.lsh_filenames = lsh_filenames or (
|
||||
@@ -588,12 +611,20 @@ class DwhPreprocessPipeline:
|
||||
raise ValueError("LSH filenames must be unique flat safe names")
|
||||
|
||||
@property
|
||||
def binding(self) -> dict[str, str]:
|
||||
return {
|
||||
def binding(self) -> dict[str, str | int]:
|
||||
binding: dict[str, str | int] = {
|
||||
"workspace_id": self.workspace_id,
|
||||
"config_fingerprint": self.config_fingerprint,
|
||||
"input_fingerprint": self.input_fingerprint,
|
||||
}
|
||||
if self.catalog_database_id is not None:
|
||||
if self.metadata_content_revision is None:
|
||||
raise ValueError("Catalog metadata revision is required for DWH binding")
|
||||
binding.update({
|
||||
"catalog_database_id": self.catalog_database_id,
|
||||
"metadata_content_revision": self.metadata_content_revision,
|
||||
})
|
||||
return binding
|
||||
|
||||
def _assert_active_binding(self, root_fd: int) -> None:
|
||||
_validate_root_binding_fd(root_fd, self.binding)
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, ValidationError, model_validator
|
||||
|
||||
from tht.mschema.models import (
|
||||
Annotations,
|
||||
ColumnPhysical,
|
||||
ForeignKey,
|
||||
PhysicalSchema,
|
||||
TablePhysical,
|
||||
)
|
||||
|
||||
|
||||
class CatalogSnapshotError(ValueError):
|
||||
"""The backend-produced Catalog Metadata Snapshot is absent or invalid."""
|
||||
|
||||
|
||||
class CatalogSnapshotColumn(BaseModel):
|
||||
id: str = Field(min_length=1)
|
||||
name: str = Field(min_length=1)
|
||||
ordinal_position: int = Field(alias="ordinalPosition", ge=1)
|
||||
data_type: str = Field(alias="dataType", min_length=1)
|
||||
is_nullable: bool = Field(alias="isNullable")
|
||||
default_expression: str | None = Field(alias="defaultExpression")
|
||||
primary_key_position: int | None = Field(alias="primaryKeyPosition", ge=1)
|
||||
sensitive: bool
|
||||
description: str | None
|
||||
description_source: Literal["curated", "generated", "source_comment"] | None = Field(
|
||||
alias="descriptionSource"
|
||||
)
|
||||
|
||||
model_config = {"extra": "forbid", "populate_by_name": True}
|
||||
|
||||
|
||||
class CatalogSnapshotTable(BaseModel):
|
||||
id: str = Field(min_length=1)
|
||||
name: str = Field(min_length=1)
|
||||
description: str | None
|
||||
description_source: Literal["curated", "generated", "source_comment"] | None = Field(
|
||||
alias="descriptionSource"
|
||||
)
|
||||
columns: list[CatalogSnapshotColumn]
|
||||
|
||||
model_config = {"extra": "forbid", "populate_by_name": True}
|
||||
|
||||
@model_validator(mode="after")
|
||||
def unique_columns(self):
|
||||
names = [column.name for column in self.columns]
|
||||
if len(names) != len(set(names)):
|
||||
raise ValueError("catalog snapshot table contains duplicate columns")
|
||||
return self
|
||||
|
||||
|
||||
class CatalogSnapshotRelationship(BaseModel):
|
||||
id: str = Field(min_length=1)
|
||||
origin: Literal["physical", "generated", "manual"]
|
||||
source_table: str = Field(alias="sourceTable", min_length=1)
|
||||
source_columns: list[str] = Field(alias="sourceColumns", min_length=1)
|
||||
target_table: str = Field(alias="targetTable", min_length=1)
|
||||
target_columns: list[str] = Field(alias="targetColumns", min_length=1)
|
||||
|
||||
model_config = {"extra": "forbid", "populate_by_name": True}
|
||||
|
||||
@model_validator(mode="after")
|
||||
def paired_columns(self):
|
||||
if len(self.source_columns) != len(self.target_columns):
|
||||
raise ValueError("catalog snapshot relationship columns are not paired")
|
||||
return self
|
||||
|
||||
|
||||
class CatalogMetadataSnapshot(BaseModel):
|
||||
schema_version: Literal[1] = Field(alias="schemaVersion")
|
||||
workspace_id: str = Field(alias="workspaceId", min_length=1)
|
||||
database_id: str = Field(alias="databaseId", min_length=1)
|
||||
database_name: str = Field(alias="databaseName", min_length=1)
|
||||
schema_name: str = Field(alias="schemaName", min_length=1)
|
||||
metadata_content_revision: int = Field(alias="metadataContentRevision", ge=0)
|
||||
tables: list[CatalogSnapshotTable]
|
||||
relationships: list[CatalogSnapshotRelationship]
|
||||
|
||||
model_config = {"extra": "forbid", "populate_by_name": True}
|
||||
|
||||
@model_validator(mode="after")
|
||||
def valid_graph(self):
|
||||
tables = {table.name: {column.name for column in table.columns} for table in self.tables}
|
||||
if len(tables) != len(self.tables):
|
||||
raise ValueError("catalog snapshot contains duplicate tables")
|
||||
for relationship in self.relationships:
|
||||
if relationship.source_table not in tables or relationship.target_table not in tables:
|
||||
raise ValueError("catalog snapshot relationship references an unknown table")
|
||||
if any(name not in tables[relationship.source_table] for name in relationship.source_columns):
|
||||
raise ValueError("catalog snapshot relationship references an unknown source column")
|
||||
if any(name not in tables[relationship.target_table] for name in relationship.target_columns):
|
||||
raise ValueError("catalog snapshot relationship references an unknown target column")
|
||||
return self
|
||||
|
||||
def to_schema_inputs(
|
||||
self,
|
||||
) -> tuple[PhysicalSchema, Annotations, dict[str, list[ForeignKey]]]:
|
||||
relationships: dict[str, list[ForeignKey]] = {}
|
||||
for relationship in self.relationships:
|
||||
relationships.setdefault(relationship.source_table, []).append(
|
||||
ForeignKey(
|
||||
name=relationship.id,
|
||||
columns=relationship.source_columns,
|
||||
ref_table=relationship.target_table,
|
||||
ref_columns=relationship.target_columns,
|
||||
)
|
||||
)
|
||||
tables = {
|
||||
table.name: TablePhysical(
|
||||
comment=table.description or "",
|
||||
columns={
|
||||
column.name: ColumnPhysical(
|
||||
type=column.data_type,
|
||||
nullable=column.is_nullable,
|
||||
pk=column.primary_key_position is not None,
|
||||
default=column.default_expression,
|
||||
comment=column.description or "",
|
||||
eligible=not column.sensitive,
|
||||
eligibility_reason="sensitive" if column.sensitive else "catalog",
|
||||
)
|
||||
for column in table.columns
|
||||
},
|
||||
foreign_keys=relationships.get(table.name, []),
|
||||
)
|
||||
for table in self.tables
|
||||
}
|
||||
return (
|
||||
PhysicalSchema(
|
||||
database=self.database_name,
|
||||
schema=self.schema_name,
|
||||
introspected_at=datetime.now(timezone.utc),
|
||||
tables=tables,
|
||||
),
|
||||
Annotations(),
|
||||
relationships,
|
||||
)
|
||||
|
||||
|
||||
def load_catalog_metadata_snapshot(path: Path, workspace_id: str | None) -> CatalogMetadataSnapshot:
|
||||
if not path.is_file():
|
||||
raise CatalogSnapshotError(f"catalog metadata snapshot is missing: {path}")
|
||||
try:
|
||||
snapshot = CatalogMetadataSnapshot.model_validate(json.loads(path.read_text()))
|
||||
except (OSError, json.JSONDecodeError, ValidationError) as exc:
|
||||
raise CatalogSnapshotError("catalog metadata snapshot is invalid") from exc
|
||||
if workspace_id is not None and snapshot.workspace_id != workspace_id:
|
||||
raise CatalogSnapshotError("catalog metadata snapshot belongs to another workspace")
|
||||
return snapshot
|
||||
@@ -118,6 +118,24 @@ def _load_effective_relationships(cfg, physical: PhysicalSchema) -> dict[str, li
|
||||
|
||||
|
||||
def load_schema_context(cfg, *, physical_file: Path | None = None) -> SchemaContext:
|
||||
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
|
||||
from tht.mschema.catalog_snapshot import (
|
||||
CatalogSnapshotError,
|
||||
load_catalog_metadata_snapshot,
|
||||
)
|
||||
|
||||
try:
|
||||
snapshot = load_catalog_metadata_snapshot(
|
||||
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
|
||||
)
|
||||
physical, annotations, relationships = snapshot.to_schema_inputs()
|
||||
except CatalogSnapshotError as exc:
|
||||
raise SchemaContextError(str(exc)) from exc
|
||||
return SchemaContext(
|
||||
physical=physical,
|
||||
annotations=annotations,
|
||||
effective_relationships=relationships,
|
||||
)
|
||||
physical_file = physical_file or physical_path(cfg)
|
||||
if not physical_file.is_file():
|
||||
raise SchemaContextError(f"physical schema is missing: {physical_file}")
|
||||
|
||||
@@ -86,10 +86,14 @@ class VectorStore(Protocol):
|
||||
|
||||
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: ...
|
||||
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int: ...
|
||||
|
||||
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ...
|
||||
|
||||
def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ...
|
||||
|
||||
def clear_reference(self) -> bool: ...
|
||||
|
||||
|
||||
__all__ = [
|
||||
"VectorCapabilities",
|
||||
|
||||
@@ -82,6 +82,8 @@ def _vector_key(hit) -> str:
|
||||
return f"column:{hit.ref}"
|
||||
if hit.kind == "schema_table":
|
||||
return f"table:{hit.ref}"
|
||||
if hit.kind == "schema_relationship":
|
||||
return f"relationship:{hit.ref}"
|
||||
return f"evidence:{hit.id}"
|
||||
|
||||
|
||||
@@ -92,7 +94,13 @@ def schema_tables(results: list["SearchResult"], top_tables: int) -> list[tuple[
|
||||
ignorati."""
|
||||
best: dict[str, float] = {}
|
||||
for r in results:
|
||||
if r.kind not in ("schema_table", "schema_column"):
|
||||
if r.kind not in ("schema_table", "schema_column", "schema_relationship"):
|
||||
continue
|
||||
if r.kind == "schema_relationship":
|
||||
endpoints = r.key.split(":", 1)[1].split("->", 1)
|
||||
for table in endpoints:
|
||||
if table not in best or r.rrf > best[table]:
|
||||
best[table] = r.rrf
|
||||
continue
|
||||
table = r.key.split(":", 1)[1].split(".", 1)[0]
|
||||
if table not in best or r.rrf > best[table]:
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
KIND_TO_TABLE = {
|
||||
"schema_table": "schema_records",
|
||||
"schema_column": "schema_records",
|
||||
"schema_relationship": "schema_records",
|
||||
"evidence": "evidence",
|
||||
"memory": "memory",
|
||||
"solved_question": "memory", # coppie domanda->SQL: stessa tabella, kind dedicato
|
||||
|
||||
@@ -15,7 +15,7 @@ MAX_EXAMPLES_IN_RECORD = 5
|
||||
|
||||
class VectorRecord(BaseModel):
|
||||
id: str
|
||||
kind: str # evidence | schema_table | schema_column
|
||||
kind: str # evidence | schema_table | schema_column | schema_relationship
|
||||
ref: str # file/chiave canonica di provenienza
|
||||
title: str
|
||||
content: str
|
||||
@@ -23,7 +23,7 @@ class VectorRecord(BaseModel):
|
||||
|
||||
|
||||
def qdrant_semantic_kind(kind: str) -> str:
|
||||
if kind in {"schema_table", "schema_column"}:
|
||||
if kind in {"schema_table", "schema_column", "schema_relationship"}:
|
||||
return "schema"
|
||||
if kind in {"memory", "solved_question"}:
|
||||
return "memory"
|
||||
@@ -136,3 +136,65 @@ def schema_records(physical: PhysicalSchema, annotations: Annotations) -> list[V
|
||||
)
|
||||
)
|
||||
return records
|
||||
|
||||
|
||||
def catalog_schema_records(snapshot) -> list[VectorRecord]:
|
||||
"""Build the complete schema slice from a Catalog Metadata Snapshot."""
|
||||
records: list[VectorRecord] = []
|
||||
for table in snapshot.tables:
|
||||
column_names = [column.name for column in table.columns]
|
||||
lines = [f"Tabella {table.name}"]
|
||||
if table.description:
|
||||
lines.append(table.description)
|
||||
lines.append("Colonne: " + ", ".join(column_names))
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"schema_table:{table.id}",
|
||||
kind="schema_table",
|
||||
ref=table.name,
|
||||
title=table.name,
|
||||
content="\n".join(lines),
|
||||
metadata={"tables": [table.name]},
|
||||
)
|
||||
)
|
||||
for column in table.columns:
|
||||
lines = [f"Colonna {table.name}.{column.name}", f"Tipo: {column.data_type}"]
|
||||
if column.description:
|
||||
lines.append(column.description)
|
||||
if column.sensitive:
|
||||
lines.append("Dato sensibile")
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"schema_column:{column.id}",
|
||||
kind="schema_column",
|
||||
ref=f"{table.name}.{column.name}",
|
||||
title=f"{table.name}.{column.name}",
|
||||
content="\n".join(lines),
|
||||
metadata={"tables": [table.name], "sensitive": column.sensitive},
|
||||
)
|
||||
)
|
||||
for relationship in snapshot.relationships:
|
||||
pairs = ", ".join(
|
||||
f"{relationship.source_table}.{source} -> "
|
||||
f"{relationship.target_table}.{target}"
|
||||
for source, target in zip(
|
||||
relationship.source_columns, relationship.target_columns, strict=True
|
||||
)
|
||||
)
|
||||
ref = f"{relationship.source_table}->{relationship.target_table}"
|
||||
records.append(
|
||||
VectorRecord(
|
||||
id=f"schema_relationship:{relationship.id}",
|
||||
kind="schema_relationship",
|
||||
ref=ref,
|
||||
title=ref,
|
||||
content=f"Relazione {relationship.origin}: {pairs}",
|
||||
metadata={
|
||||
"tables": [relationship.source_table, relationship.target_table],
|
||||
"origin": relationship.origin,
|
||||
"source_table": relationship.source_table,
|
||||
"target_table": relationship.target_table,
|
||||
},
|
||||
)
|
||||
)
|
||||
return records
|
||||
|
||||
Reference in New Issue
Block a user