feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s

This commit is contained in:
Codex
2026-09-06 17:49:35 +02:00
parent 8707ae1d46
commit cffa60772e
141 changed files with 5898 additions and 3015 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ def build_vector_store(cfg: Config, *, require_write: bool = False) -> VectorSto
case "qdrant":
return QdrantVectorStore(
base_url=resource.base_url,
collection=resource.collection,
collections=resource.collections,
workspace_id=cfg._workspace_id,
workspace_revision=cfg._workspace_revision,
expected_dimension=cfg.embeddings.dim if cfg.embeddings is not None else None,
+1 -1
View File
@@ -5,7 +5,7 @@ from __future__ import annotations
from tht.ports.vector import VectorStoreError
COLLECTION_KINDS = {
"schema_records": {"schema_table", "schema_column"},
"schema_records": {"schema_table", "schema_column", "schema_relationship"},
"evidence": {"evidence"},
"memory": {"memory", "solved_question"},
}
+152 -29
View File
@@ -1,4 +1,4 @@
"""Qdrant-backed vector store for one workspace-owned semantic collection."""
"""Qdrant-backed vector store with separate reference and memory lifecycles."""
import re
from collections.abc import Callable
@@ -25,6 +25,9 @@ from tht.vectorstore.store import VectorHit, hit_from_metadata
_GENERATION = re.compile(r"gen:[0-9a-f]{32}")
_WORKSPACE = re.compile(r"[a-z][a-z0-9_-]{0,63}")
_BM25_LANGUAGES = frozenset({"english", "italian"})
_REFERENCE_KINDS = frozenset({
"schema_table", "schema_column", "schema_relationship", "evidence",
})
_KEYWORD_INDEXES = (
"content_hash",
@@ -58,7 +61,8 @@ class QdrantVectorStore:
self,
*,
base_url: str,
collection: str,
collections: dict[str, str] | None = None,
collection: str | None = None,
workspace_id: str,
workspace_revision: str | None = None,
expected_dimension: int | None = None,
@@ -68,7 +72,17 @@ class QdrantVectorStore:
read_timeout: float = 10.0,
):
self._base_url = base_url.rstrip("/")
self._collection = collection
legacy_constructor = collections is None and collection is not None
if legacy_constructor:
collections = {"reference": collection, "memory": collection}
if collections is None or set(collections) != {"reference", "memory"}:
raise VectorStoreError("Qdrant collections must define reference and memory")
if not legacy_constructor and collections["reference"] == collections["memory"]:
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
self._collections = dict(collections)
self._split_collections = collections["reference"] != collections["memory"]
self._legacy_collection = workspace_id
self._legacy_checked = False
self._workspace_id = workspace_id
self._workspace_revision = None
self._workspace_revision = workspace_revision
@@ -90,7 +104,10 @@ class QdrantVectorStore:
def health(self) -> VectorHealth:
try:
info = self._ensure_collection(strict=False)
infos = {
name: self._ensure_collection(collection, strict=False)
for name, collection in self._collections.items()
}
except VectorStoreError as exc:
return VectorHealth(
ok=False,
@@ -105,8 +122,10 @@ class QdrantVectorStore:
bm25_compatible=None,
)
dimension = info["config"]["params"]["vectors"]["size"]
dimensions = (dimension,)
dimensions = tuple(sorted({
info["config"]["params"]["vectors"]["size"]
for info in infos.values()
}))
compatible = (
None if self._expected_dimension is None else dimensions == (self._expected_dimension,)
)
@@ -119,7 +138,7 @@ class QdrantVectorStore:
expected_dimension=self._expected_dimension,
observed_dimensions=dimensions,
dimension_compatible=compatible,
bm25_compatible=self._bm25_compatible(info),
bm25_compatible=self._bm25_compatible(infos["reference"]),
)
def search(
@@ -139,6 +158,21 @@ class QdrantVectorStore:
allowed_record_kinds = self._allowed_record_kinds(collections, kinds)
if not allowed_record_kinds:
return []
physical_collections = {
self._physical_collection_for_kind(kind) for kind in allowed_record_kinds
}
if len(physical_collections) != 1:
hits: list[VectorHit] = []
for collection in collections:
nested_kinds = sorted(set(allowed_record_kinds) & COLLECTION_KINDS[collection])
if nested_kinds:
hits.extend(self.search(
[collection], embedding, limit=limit, kinds=nested_kinds,
metadata_filter=metadata_filter, query_text=query_text,
query_language=query_language, retrieval_mode=retrieval_mode,
))
return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit]
physical_collection = physical_collections.pop()
filter_must = self._workspace_filter()
filter_must.extend(self._revision_filter(allowed_record_kinds))
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
@@ -193,7 +227,7 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"vector": embedding,
"limit": limit,
@@ -206,11 +240,11 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(strict=False, require_bm25=True)
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
shared_filter = {"must": filter_must}
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"query": self._bm25_document(query_text, query_language),
"using": "bm25",
@@ -224,7 +258,7 @@ class QdrantVectorStore:
raise VectorStoreError("Evidence hybrid query text is required")
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"vector": embedding,
"limit": limit,
@@ -237,11 +271,11 @@ class QdrantVectorStore:
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
raise VectorStoreError("Evidence BM25 query is invalid")
self._ensure_collection(strict=False, require_bm25=True)
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
shared_filter = {"must": filter_must}
response = self._call(
"POST",
f"/collections/{self._collection}/points/query",
f"/collections/{physical_collection}/points/query",
{
"prefetch": [
{"query": embedding, "limit": limit * 2, "filter": shared_filter},
@@ -266,7 +300,9 @@ class QdrantVectorStore:
def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]:
validate_collection(collection)
validate_collection_kinds(collection, kinds)
physical_collection = self._physical_collection_for_logical(collection)
points = self._scroll(
physical_collection,
[
*self._workspace_filter(),
self._semantic_kind_filter(kinds),
@@ -287,7 +323,14 @@ class QdrantVectorStore:
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
validate_collection(collection)
# The first write with the split configuration is also the upgrade cutover. This keeps
# existing runtime memory reachable even when the operator reruns preprocessing without
# invoking the explicit clear operation first.
if self._split_collections:
self._migrate_legacy_memory()
physical_collection = self._physical_collection_for_logical(collection)
self._ensure_collection(
physical_collection,
strict=True,
require_bm25=any(record.sparse_text is not None for record in records),
)
@@ -310,7 +353,7 @@ class QdrantVectorStore:
self._workspace_id,
semantic_kind,
write_record.record.id,
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "evidence") else None,
self._workspace_revision if semantic_kind in ("schema_table", "schema_column", "schema_relationship", "evidence") else None,
),
"vector": vector,
"payload": qdrant_payload(
@@ -326,7 +369,7 @@ class QdrantVectorStore:
for start in range(0, len(points), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{self._collection}/points?wait=true",
f"/collections/{physical_collection}/points?wait=true",
{"points": points[start:start + UPSERT_BATCH_SIZE]},
)
return len(records)
@@ -334,15 +377,16 @@ class QdrantVectorStore:
def delete_kinds(self, collection: str, kinds: list[str]) -> int:
validate_collection(collection)
validate_collection_kinds(collection, kinds)
physical_collection = self._physical_collection_for_logical(collection)
must = [
*self._workspace_filter(),
self._semantic_kind_filter(kinds),
{"key": "record_kind", "match": {"any": sorted(kinds)}},
]
before = len(self._scroll(must))
before = len(self._scroll(physical_collection, must))
self._call(
"POST",
f"/collections/{self._collection}/points/delete?wait=true",
f"/collections/{physical_collection}/points/delete?wait=true",
{"filter": {"must": must}},
)
return before
@@ -353,6 +397,7 @@ class QdrantVectorStore:
if _WORKSPACE.fullmatch(workspace_id) is None:
raise VectorStoreError("Invalid Evidence workspace namespace")
self._require_bound_workspace(workspace_id)
physical_collection = self._collections["reference"]
must = [
*self._workspace_filter(),
{"key": "kind", "match": {"value": "evidence"}},
@@ -360,11 +405,11 @@ class QdrantVectorStore:
{"key": "vector_generation", "match": {"value": generation}},
]
before = len(
self._scroll(must)
self._scroll(physical_collection, must)
)
self._call(
"POST",
f"/collections/{self._collection}/points/delete?wait=true",
f"/collections/{physical_collection}/points/delete?wait=true",
{"filter": {"must": must}},
)
return before
@@ -375,7 +420,9 @@ class QdrantVectorStore:
if _WORKSPACE.fullmatch(workspace_id) is None:
raise VectorStoreError("Invalid Evidence workspace namespace")
self._require_bound_workspace(workspace_id)
physical_collection = self._collections["reference"]
points = self._scroll(
physical_collection,
[
*self._workspace_filter(),
{"key": "kind", "match": {"value": "evidence"}},
@@ -394,6 +441,68 @@ class QdrantVectorStore:
def _workspace_filter(self) -> list[dict]:
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
def clear_reference(self) -> bool:
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
legacy_deleted = self._migrate_legacy_memory()
collection = self._collections["reference"]
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
return legacy_deleted
self._call("DELETE", f"/collections/{collection}", None)
return True
def _migrate_legacy_memory(self) -> bool:
"""Preserve memory from the pre-split collection before retiring it."""
legacy = self._legacy_collection
if (
self._legacy_checked
or not self._split_collections
or legacy in self._collections.values()
):
return False
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
if response is None:
self._legacy_checked = True
return False
memory = self._collections["memory"]
self._ensure_collection(memory, strict=True, allow_create=True)
points = self._scroll(
legacy,
[
*self._workspace_filter(),
self._semantic_kind_filter(["memory", "solved_question"]),
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
],
with_vector=True,
)
migrated = []
for point in points:
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
raise VectorStoreError("Qdrant returned malformed legacy memory response")
if not isinstance(point.get("payload"), dict):
raise VectorStoreError("Qdrant returned malformed legacy memory response")
migrated.append({
"id": point["id"],
"vector": point["vector"],
"payload": point["payload"],
})
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
self._call(
"PUT",
f"/collections/{memory}/points?wait=true",
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
)
self._call("DELETE", f"/collections/{legacy}", None)
self._legacy_checked = True
return True
def _physical_collection_for_logical(self, collection: str) -> str:
validate_collection(collection)
return self._collections["memory" if collection == "memory" else "reference"]
def _physical_collection_for_kind(self, kind: str) -> str:
return self._collections["reference" if kind in _REFERENCE_KINDS else "memory"]
@staticmethod
def _bm25_document(text: str, language: str) -> dict:
return {
@@ -405,7 +514,7 @@ class QdrantVectorStore:
def _revision_filter(self, kinds: list[str]) -> list[dict]:
if self._workspace_revision is None:
return []
if not any(kind in ("schema_table", "schema_column", "evidence") for kind in kinds):
if not any(kind in ("schema_table", "schema_column", "schema_relationship", "evidence") for kind in kinds):
return []
return [{"key": "workspace_revision", "match": {"value": self._workspace_revision}}]
@@ -450,25 +559,30 @@ class QdrantVectorStore:
bm25 = sparse_vectors.get("bm25")
return isinstance(bm25, dict) and bm25.get("modifier") == "idf"
def _ensure_collection(self, *, strict: bool, require_bm25: bool = False) -> dict | None:
response = self._call("GET", f"/collections/{self._collection}", None, allow_missing=True)
def _ensure_collection(
self, collection: str, *, strict: bool, require_bm25: bool = False,
allow_create: bool = False,
) -> dict | None:
created = False
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
if response is None:
if not strict:
raise VectorStoreError("Qdrant collection is missing")
if self._collection_lifecycle == "require_existing":
if self._collection_lifecycle == "require_existing" and not allow_create:
raise VectorStoreError("semantic_index_incompatible")
self._call(
"PUT",
f"/collections/{self._collection}",
f"/collections/{collection}",
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
)
for field_name in _KEYWORD_INDEXES:
self._call(
"PUT",
f"/collections/{self._collection}/index",
f"/collections/{collection}/index",
{"field_name": field_name, "field_schema": "keyword"},
)
response = self._call("GET", f"/collections/{self._collection}", None)
created = True
response = self._call("GET", f"/collections/{collection}", None)
result = response.get("result") if isinstance(response, dict) else None
config = result.get("config", {}).get("params", {}).get("vectors") if isinstance(result, dict) else None
if not isinstance(config, dict):
@@ -484,29 +598,38 @@ class QdrantVectorStore:
raise VectorStoreError("Qdrant collection configuration mismatch")
for field_name in _KEYWORD_INDEXES:
if field_name not in result.get("payload_schema", {}):
# A successful index-creation response can precede visibility in the
# collection-info payload. The newly-created collection is already safe to
# use; later readiness checks will validate the asynchronously published
# indexes. Existing collections still follow the strict lifecycle policy.
if created:
continue
if strict and self._collection_lifecycle == "require_existing":
raise VectorStoreError("semantic_index_incompatible")
if not strict:
raise VectorStoreError("Qdrant collection payload indexes mismatch")
self._call(
"PUT",
f"/collections/{self._collection}/index",
f"/collections/{collection}/index",
{"field_name": field_name, "field_schema": "keyword"},
)
if require_bm25 and not self._bm25_compatible(result):
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
return result
def _scroll(self, must: list[dict]) -> list[dict]:
def _scroll(
self, collection: str, must: list[dict], *, with_vector: bool = False
) -> list[dict]:
points: list[dict] = []
offset = None
seen_offsets = set()
while True:
response = self._call(
"POST",
f"/collections/{self._collection}/points/scroll",
f"/collections/{collection}/points/scroll",
{
"with_payload": True,
"with_vector": with_vector,
"limit": 10000,
"filter": {"must": must},
"offset": offset,
+41 -10
View File
@@ -3,7 +3,7 @@ from pathlib import Path
from tht.cli.schema_cmd import physical_path
def _extract_lsh_values(dwh, physical, annotations, limit):
def _extract_lsh_values(dwh, physical, annotations, limit, eligibility_cfg=None):
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
from tht.mschema.eligibility import effective_eligibility
@@ -12,7 +12,16 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
table_ann = annotations.tables.get(table_name)
for column_name, column in table.columns.items():
ann_col = table_ann.columns.get(column_name) if table_ann else None
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
from_catalog = column.eligibility_reason in {"catalog", "sensitive"}
if not is_text_type(column.type):
continue
if from_catalog and column.eligibility_reason == "sensitive":
continue
if from_catalog and eligibility_cfg is not None and (
column_name.lower() in {name.lower() for name in eligibility_cfg.ignore_columns}
):
continue
if not from_catalog and not effective_eligibility(column, ann_col)[0]:
continue
try:
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
@@ -20,6 +29,21 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
continue
vals = [str(value) for value in distinct.values if value not in (None, "")]
if from_catalog and eligibility_cfg is not None:
from tht.mschema.eligibility import classify_column
lengths = [len(value) for value in vals]
eligible, reason = classify_column(
column.type,
column.is_enum,
(sum(lengths) / len(lengths)) if lengths else None,
max(lengths) if lengths else None,
eligibility_cfg,
)
column.eligible = eligible
column.eligibility_reason = reason
if not eligible:
continue
if vals:
values.setdefault(table_name, {})[column_name] = vals
if distinct.truncated:
@@ -33,18 +57,25 @@ def build_lsh_artifacts(
):
"""Run the existing LSH extraction/build algorithm and persist its outputs."""
from tht.adapters.factory import build_dwh
from tht.cli.schema_cmd import annotations_path
from tht.lshindex import build_index, save_index
from tht.mschema.models import Annotations, PhysicalSchema
phys_file = physical_file or physical_path(cfg)
if not phys_file.exists():
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
from tht.mschema.context import load_schema_context
context = load_schema_context(cfg)
physical, annotations = context.physical, context.annotations
else:
from tht.cli.schema_cmd import annotations_path
from tht.mschema.models import Annotations, PhysicalSchema
phys_file = physical_file or physical_path(cfg)
if not phys_file.exists():
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
target = dwh if dwh is not None else build_dwh(cfg)
values, skipped, truncated = _extract_lsh_values(
target, physical, annotations, cfg.lsh.max_values_per_column
target, physical, annotations, cfg.lsh.max_values_per_column, cfg.eligibility
)
lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose)
save_index(
+203
View File
@@ -4,7 +4,10 @@ from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import stat
from pathlib import Path
import typer
@@ -144,6 +147,8 @@ def run_dwh_from_config(
workspace_root=workspace_root,
config_fingerprint=binding["config_fingerprint"],
input_fingerprint=binding["input_fingerprint"],
catalog_database_id=binding.get("catalog_database_id"),
metadata_content_revision=binding.get("metadata_content_revision"),
introspect=lambda output: refresh_catalog(cfg, output_path=output),
build_lsh=lambda physical, output: build_lsh_artifacts(
cfg, physical_file=physical, output_dir=output
@@ -155,6 +160,57 @@ def run_dwh_from_config(
return pipeline.run(steps, resume_run_id=resume)
def run_catalog_dwh_from_config(cfg):
"""Publish Catalog-derived physical schema and LSH as one bound generation."""
from tht.cli.lsh_cmd import build_lsh_artifacts
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
snapshot = load_catalog_metadata_snapshot(
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
)
binding = config_dwh_binding(cfg)
workspace_root = cfg.paths.artifacts.parent
# Catalog preprocessing is a replace-in-place operation. Its DWH/LSH output is wholly
# derived, there is no supported concurrent runtime, and a failed Clear may have left an
# older binding behind. Start from an empty owned generation root so a retry can always
# rebuild the current Catalog revision instead of deadlocking on the stale OWNER marker.
_remove_owned_derived_path(workspace_root, workspace_root / ".tht-dwh")
_remove_owned_derived_path(workspace_root, cfg.paths.artifacts / "mschema" / "physical.yaml")
_remove_owned_derived_path(workspace_root, cfg.paths.indexes / "lsh")
observed: dict[str, object] = {}
def materialize_physical(output: Path):
physical, _annotations, _relationships = snapshot.to_schema_inputs()
physical.to_yaml(output)
return physical
def materialize_lsh(physical: Path, output: Path):
result = build_lsh_artifacts(cfg, physical_file=physical, output_dir=output)
observed["result"] = result
return result
report = DwhPreprocessPipeline(
workspace_id=str(binding["workspace_id"]),
workspace_root=workspace_root,
config_fingerprint=str(binding["config_fingerprint"]),
input_fingerprint=str(binding["input_fingerprint"]),
catalog_database_id=str(binding["catalog_database_id"]),
metadata_content_revision=int(binding["metadata_content_revision"]),
introspect=materialize_physical,
build_lsh=materialize_lsh,
lsh_filenames=(
f"{cfg.database.db_schema}_lsh.pkl",
f"{cfg.database.db_schema}_minhashes.pkl",
f"{cfg.database.db_schema}_meta.json",
),
).run(("introspect", "lsh"))
result = observed.get("result")
if not isinstance(result, tuple) or len(result) != 4:
raise RuntimeError("Catalog LSH publication did not complete")
return report, result
def _parse_dwh_steps(value: str) -> tuple[str, ...]:
allowed = ("introspect", "lsh")
steps = tuple(part.strip() for part in value.split(",") if part.strip())
@@ -234,6 +290,85 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
return pipeline.gc(workspace_root=corpus_root.parent, dry_run=dry_run)
def _remove_owned_derived_path(workspace_root: Path, target: Path) -> bool:
"""Remove one generated path without following links or escaping the workspace."""
root = workspace_root.resolve()
resolved = target.resolve(strict=False)
if not resolved.is_relative_to(root):
raise RuntimeError("derived cleanup target escapes the workspace")
try:
info = target.lstat()
except FileNotFoundError:
return False
if stat.S_ISLNK(info.st_mode) or info.st_uid != os.getuid():
raise RuntimeError("derived cleanup target is unsafe")
if stat.S_ISDIR(info.st_mode):
shutil.rmtree(target)
elif stat.S_ISREG(info.st_mode):
target.unlink()
else:
raise RuntimeError("derived cleanup target is unsafe")
return True
def clear_from_config(config: Path) -> dict[str, int]:
"""Clear workspace reference vectors and local preprocessing derivatives."""
from tht.adapters.factory import build_vector_store
from tht.cli._guards import require_vector_write_allowed
from tht.cli.schema_cmd import _load_config_or_exit
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "preprocess clear")
workspace_root = cfg.paths.artifacts.parent
vector_store = build_vector_store(cfg, require_write=True)
reference_deleted = int(vector_store.clear_reference())
paths = [
workspace_root / ".tht-dwh",
workspace_root / ".tht-jobs",
workspace_root / "corpus",
cfg.paths.artifacts / "mschema" / "physical.yaml",
cfg.paths.indexes / "lsh",
]
if cfg.paths.catalog_metadata_snapshot is not None:
paths.append(cfg.paths.catalog_metadata_snapshot)
removed = sum(int(_remove_owned_derived_path(workspace_root, path)) for path in paths)
return {"referenceCollections": reference_deleted, "derivedPaths": removed}
@preprocess_app.command("clear", hidden=True)
def clear_cmd(
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Clear replaceable preprocessing output while preserving workspace memory."""
try:
counts = clear_from_config(config)
except Exception: # noqa: BLE001 - do not disclose paths, endpoints, or credentials
payload = {
"schemaVersion": 1,
"status": "failed",
"code": "preprocessing_clear_failed",
"operation": "preprocess_clear",
"error": "Preprocessing clear failed",
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.secho("ERRORE: preprocessing clear failed", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = {
"schemaVersion": 1,
"status": "succeeded",
"code": "ok",
"operation": "preprocess_clear",
"counts": counts,
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.echo("OK: reference vectors and LSH cleared; memory preserved")
@preprocess_app.command("evidence")
def evidence_cmd(
action: str | None = typer.Argument(None),
@@ -359,3 +494,71 @@ def dwh_cmd(
typer.secho(f"ERRORE: run={result.run_id} DWH preprocessing failed", fg=typer.colors.RED, err=True)
if result.status != "succeeded":
raise typer.Exit(code=1)
@preprocess_app.command("catalog", hidden=True)
def catalog_cmd(
catalog_metadata: Path = typer.Option(..., "--catalog-metadata"),
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Build current LSH and schema vectors from one immutable Catalog snapshot."""
from tht.cli._guards import require_vector_write_allowed
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import index_catalog_schema, require_vector_cfg
cfg = _load_config_or_exit(config)
cfg = cfg.model_copy(
update={
"paths": cfg.paths.model_copy(
update={"catalog_metadata_snapshot": catalog_metadata}
)
}
)
require_vector_write_allowed(cfg, "preprocess catalog")
require_vector_cfg(cfg)
try:
report, (minhashes, skipped, truncated, _values) = run_catalog_dwh_from_config(cfg)
if report.status != "succeeded":
raise RuntimeError("Catalog DWH preprocessing failed")
stats, counts, snapshot_path = index_catalog_schema(cfg)
except Exception: # noqa: BLE001 - public output must never disclose endpoints or SQL
payload = {
"schemaVersion": 1,
"status": "failed",
"code": "catalog_preprocessing_failed",
"operation": "preprocess_catalog",
"error": "Catalog preprocessing failed",
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.secho("ERRORE: Catalog preprocessing failed", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = {
"schemaVersion": 1,
"status": "succeeded",
"code": "ok",
"operation": "preprocess_catalog",
"artifactIdentities": [
{
"kind": "catalog_metadata_snapshot",
"digest": "sha256:" + hashlib.sha256(snapshot_path.read_bytes()).hexdigest(),
}
],
"counts": {
**counts,
"lshEntries": len(minhashes),
"lshSkippedColumns": len(skipped),
"lshTruncatedColumns": len(truncated),
"added": stats.added,
"deleted": stats.deleted,
"unchanged": stats.unchanged,
"updated": stats.updated,
},
}
if json_output:
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
else:
typer.echo("OK: Catalog metadata, LSH and schema vectors rebuilt")
+6 -19
View File
@@ -560,22 +560,14 @@ def render_cmd(
),
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
) -> None:
"""Serializza mschema (physical + annotations) nel formato richiesto."""
"""Serialize the current PostgreSQL Catalog projection."""
import json
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED,
err=True,
)
raise typer.Exit(code=1)
try:
context = load_schema_context(cfg, physical_file=phys_file)
context = load_schema_context(cfg)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
@@ -625,17 +617,12 @@ def columns_cmd(
"""Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo."""
import json as _json
from tht.mschema.models import PhysicalSchema
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
try:
physical = load_schema_context(cfg).physical
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
def _payload(name, tbl):
return {
+8 -14
View File
@@ -9,7 +9,7 @@ from tht.config import workspace_id_for_config
KIND_MAP = {
"evidence": ["evidence"],
"schema": ["schema_table", "schema_column"],
"schema": ["schema_table", "schema_column", "schema_relationship"],
"values": [], # solo LSH
"formula": ["evidence"],
}
@@ -235,15 +235,8 @@ def search_cmd(
from tht.mschema.render import to_mschema_text
from tht.search import schema_tables
phys_file = dwh_snapshot.physical
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
try:
schema_context = load_schema_context(cfg, physical_file=phys_file)
schema_context = load_schema_context(cfg)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
@@ -390,7 +383,7 @@ def pack_cmd(
workspace_id = workspace_id_for_config(cfg, config)
validate_corpus_workspace(cfg, workspace_id)
dwh_snapshot = _leased_dwh_snapshot(cfg, ctx)
_leased_dwh_snapshot(cfg, ctx)
require_vector_cfg(cfg)
tables: list[dict] = []
@@ -414,12 +407,13 @@ def pack_cmd(
if vec is not None:
descriptions: dict[str, str] = {}
phys_file = dwh_snapshot.physical
if phys_file.exists():
from tht.mschema.models import PhysicalSchema
from tht.mschema.context import SchemaContextError, load_schema_context
phys = PhysicalSchema.from_yaml(phys_file)
try:
phys = load_schema_context(cfg).physical
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
except SchemaContextError:
pass
try:
cand = combined_search(
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
+6 -6
View File
@@ -4,7 +4,7 @@ from pathlib import Path
import typer
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
from tht.cli.schema_cmd import _load_config_or_exit
sql_app = typer.Typer(help="Validazione ed esecuzione controllata di SQL (read-only)")
@@ -17,16 +17,16 @@ def _read_sql(file: Path) -> str:
def _load_physical_or_exit(cfg):
from tht.mschema.models import PhysicalSchema
from tht.mschema.context import SchemaContextError, load_schema_context
phys_file = physical_path(cfg)
if not phys_file.exists():
try:
return load_schema_context(cfg).physical
except SchemaContextError:
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
"ERRORE: Catalog Metadata Snapshot non disponibile. Esegui il preprocessing.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
return PhysicalSchema.from_yaml(phys_file)
def require_action(cfg, action: str) -> None:
+38 -37
View File
@@ -6,7 +6,7 @@ import typer
from tht.cli._guards import require_vector_write_allowed
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path
from tht.cli.schema_cmd import _load_config_or_exit
from tht.ports.vector import VectorWriteRecord
from tht.vectorstore.store import SyncStats, content_hash
@@ -112,42 +112,14 @@ def index_schema_cmd(
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store."""
from tht.adapters.factory import build_vector_store
from tht.mschema.models import Annotations, PhysicalSchema
"""Replace the schema slice from the PostgreSQL Catalog projection."""
from tht.ports.vector import VectorStoreError
from tht.vectorstore.records import schema_records
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "vector index-schema")
require_vector_cfg(cfg)
phys_file = physical_path(cfg)
if not phys_file.exists():
message = f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`."
if json_output:
_emit_json({
"code": "schema_missing",
"error": "physical schema is missing",
"operation": "index_schema",
"schemaVersion": 1,
"status": "failed",
"workspaceId": cfg._workspace_id,
"workspaceRevision": cfg._workspace_revision,
})
else:
typer.secho(message, fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
annotations_file = annotations_path(cfg)
annotations = Annotations.from_yaml(annotations_file)
records = schema_records(physical, annotations)
try:
stats = sync_canonical_records(
"schema_records",
records,
store=build_vector_store(cfg, require_write=True),
embedder=make_embedder(cfg.embeddings),
)
stats, counts, snapshot_path = index_catalog_schema(cfg)
except VectorStoreError as exc:
code = str(exc)
error = "semantic index incompatible" if code == "semantic_index_incompatible" else "schema indexing failed"
@@ -167,17 +139,17 @@ def index_schema_cmd(
if json_output:
_emit_json({
"artifactIdentities": [
{"digest": _artifact_digest(annotations_file), "kind": "schema_annotations"},
{"digest": _artifact_digest(phys_file), "kind": "physical_schema"},
{"digest": _artifact_digest(snapshot_path), "kind": "catalog_metadata_snapshot"},
],
"code": "ok",
"collection": cfg.vectors.collection,
"collection": cfg.vectors.collections["reference"],
"counts": {
"added": stats.added,
"columns": sum(len(table.columns) for table in physical.tables.values()),
"columns": counts["columns"],
"deleted": stats.deleted,
"records": len(records),
"tables": len(physical.tables),
"records": counts["records"],
"relationships": counts["relationships"],
"tables": counts["tables"],
"unchanged": stats.unchanged,
"updated": stats.updated,
},
@@ -189,3 +161,32 @@ def index_schema_cmd(
})
return
_print_stats(stats)
def index_catalog_schema(cfg):
from tht.adapters.factory import build_vector_store
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
from tht.vectorstore.records import catalog_schema_records
snapshot_path = cfg.paths.catalog_metadata_snapshot
if snapshot_path is None:
raise ValueError("catalog metadata snapshot is not configured")
snapshot = load_catalog_metadata_snapshot(snapshot_path, cfg._workspace_id)
records = catalog_schema_records(snapshot)
store = build_vector_store(cfg, require_write=True)
deleted = store.delete_kinds(
"schema_records", ["schema_table", "schema_column", "schema_relationship"]
)
stats = sync_canonical_records(
"schema_records",
records,
store=store,
embedder=make_embedder(cfg.embeddings),
)
stats.deleted = deleted
return stats, {
"tables": len(snapshot.tables),
"columns": sum(len(table.columns) for table in snapshot.tables),
"relationships": len(snapshot.relationships),
"records": len(records),
}, snapshot_path
+48 -8
View File
@@ -49,8 +49,8 @@ def canonical_effective_config_document(cfg) -> dict:
else:
raise ConfigError("unsupported DWH transport in canonical effective config")
vectors = getattr(cfg, "vectors", None)
collection = getattr(vectors, "collection", None) if vectors is not None else None
if not collection:
collections = getattr(vectors, "collections", None) if vectors is not None else None
if not collections or set(collections) != {"reference", "memory"}:
raise ConfigError("vector configuration is unavailable; cannot canonicalize effective config")
embeddings = getattr(cfg, "embeddings", None)
model = getattr(embeddings, "model", None) if embeddings is not None else None
@@ -58,9 +58,16 @@ def canonical_effective_config_document(cfg) -> dict:
if not model or not embed_dim:
raise ConfigError("embedding configuration is unavailable; cannot canonicalize effective config")
return {
"schemaVersion": 1,
"schemaVersion": 3,
"dwh": dwh,
"vector": {"collection": collection, "dimensions": int(embed_dim), "distance": "cosine"},
"vector": {
"collections": {
"reference": collections["reference"],
"memory": collections["memory"],
},
"dimensions": int(embed_dim),
"distance": "cosine",
},
"embedding": {
"id": getattr(embeddings, "id", None) or f"ollama/{model}",
"model": model,
@@ -311,9 +318,32 @@ class ThothVectorHttpConfig(BaseModel):
class QdrantConfig(BaseModel):
type: Literal["qdrant"]
base_url: str
collection: str = Field(min_length=1)
collections: dict[Literal["reference", "memory"], str]
collection_lifecycle: Literal["self_heal", "require_existing"] = "self_heal"
@model_validator(mode="before")
@classmethod
def accept_single_collection_test_fixture(cls, value: Any) -> Any:
if isinstance(value, dict) and "collections" not in value and "collection" in value:
translated = dict(value)
collection = translated.pop("collection")
translated["collections"] = {
"reference": collection,
"memory": f"{collection}-memory",
}
return translated
return value
@model_validator(mode="after")
def validate_collections(self):
if set(self.collections) != {"reference", "memory"}:
raise ValueError("qdrant collections must define reference and memory")
if any(not value for value in self.collections.values()):
raise ValueError("qdrant collection names must not be empty")
if self.collections["reference"] == self.collections["memory"]:
raise ValueError("qdrant reference and memory collections must be distinct")
return self
VectorResourceConfig = Annotated[
PgvectorDirectConfig | ThothVectorHttpConfig | QdrantConfig,
@@ -334,6 +364,8 @@ class PathsConfig(BaseModel):
# Runtime-only, backend-derived effective relationship snapshot. When present,
# it is the exclusive FK source; it is not an authored workspace artifact.
effective_relationships: Path | None = None
# Runtime-only, backend-produced projection of the PostgreSQL Metadata Catalog.
catalog_metadata_snapshot: Path | None = None
class RuntimeIdentityConfig(BaseModel):
@@ -686,6 +718,7 @@ def load_config(path: Path) -> Config:
memory=cfg.paths.memory,
annotations_root=cfg.paths.annotations_root,
effective_relationships=cfg.paths.effective_relationships,
catalog_metadata_snapshot=cfg.paths.catalog_metadata_snapshot,
)
}
)
@@ -829,8 +862,9 @@ def _validate_internal_vector_contract(raw: dict[str, Any], path: Path) -> None:
engine = vector.get("engine")
base_url = vector.get("base_url")
collection = vector.get("collection")
collections = vector.get("collections")
lifecycle = vector.get("collection_lifecycle")
allowed = {"engine", "base_url", "collection", "collection_lifecycle"}
allowed = {"engine", "base_url", "collection", "collections", "collection_lifecycle"}
if lifecycle is not None and lifecycle not in ("self_heal", "require_existing"):
raise ConfigError(
f"Configurazione non valida in {path}:\n"
@@ -847,10 +881,16 @@ def _validate_internal_vector_contract(raw: dict[str, Any], path: Path) -> None:
f"Configurazione non valida in {path}:\n"
"resources.vector.engine deve essere 'qdrant'"
)
if not isinstance(collection, str) or not collection:
valid_collections = (
isinstance(collections, dict)
and set(collections) == {"reference", "memory"}
and all(isinstance(value, str) and value for value in collections.values())
and collections["reference"] != collections["memory"]
)
if not valid_collections and (not isinstance(collection, str) or not collection):
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"resources.vector.collection deve essere valorizzato"
"resources.vector.collections deve definire reference e memory"
)
if not _is_allowed_internal_qdrant_url(base_url):
raise ConfigError(
+4 -1
View File
@@ -39,8 +39,11 @@ def translate_legacy_config(raw: dict[str, Any]) -> tuple[dict[str, Any], bool]:
translated_vectors: dict[str, Any] = {
"type": "qdrant",
"base_url": vector.get("base_url"),
"collection": vector.get("collection"),
}
if isinstance(vector.get("collections"), dict):
translated_vectors["collections"] = vector["collections"]
else:
translated_vectors["collection"] = vector.get("collection")
if vector.get("collection_lifecycle") in ("self_heal", "require_existing"):
translated_vectors["collection_lifecycle"] = vector["collection_lifecycle"]
translated["vectors"] = translated_vectors
+1 -1
View File
@@ -195,7 +195,7 @@ class ActiveEvidenceSearcher:
query_language=None,
):
requested = set(kinds) if kinds is not None else {
"schema_table", "schema_column", "evidence", "memory", "solved_question",
"schema_table", "schema_column", "schema_relationship", "evidence", "memory", "solved_question",
}
include_evidence = "evidence" in requested
other_kinds = sorted(requested - {"evidence"})
+39 -8
View File
@@ -42,7 +42,7 @@ def _cleanup_snapshot_dirs() -> None:
atexit.register(_cleanup_snapshot_dirs)
def config_dwh_binding(cfg) -> dict[str, str]:
def config_dwh_binding(cfg) -> dict[str, str | int]:
workspace_id = getattr(cfg, "_workspace_id", None)
config_source = getattr(cfg, "_config_source", None)
if not isinstance(workspace_id, str) or not isinstance(config_source, str):
@@ -72,14 +72,24 @@ def config_dwh_binding(cfg) -> dict[str, str]:
else:
config_fingerprint = fingerprint(cfg.model_dump_json())
input_fingerprint = fingerprint(config_source)
return {
binding: dict[str, str | int] = {
"workspace_id": workspace_id,
"config_fingerprint": config_fingerprint,
"input_fingerprint": input_fingerprint,
}
snapshot_path = getattr(getattr(cfg, "paths", None), "catalog_metadata_snapshot", None)
if snapshot_path is not None:
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
snapshot = load_catalog_metadata_snapshot(snapshot_path, workspace_id)
binding.update({
"catalog_database_id": snapshot.database_id,
"metadata_content_revision": snapshot.metadata_content_revision,
})
return binding
def _binding_digest(binding: dict[str, str]) -> str:
def _binding_digest(binding: dict[str, str | int]) -> str:
payload = json.dumps(binding, sort_keys=True, separators=(",", ":"))
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
@@ -103,10 +113,19 @@ def _read_root_binding_fd(root_fd: int) -> dict[str, str]:
os.close(fd)
payload = json.loads(b"".join(chunks).decode("utf-8"))
binding = payload["binding"]
keys = set(binding) if isinstance(binding, dict) else set()
base_keys = {"workspace_id", "config_fingerprint", "input_fingerprint"}
catalog_keys = {"catalog_database_id", "metadata_content_revision"}
if (
payload.get("schema_version") != 1
payload.get("schema_version") not in {1, 2}
or not isinstance(binding, dict)
or set(binding) != {"workspace_id", "config_fingerprint", "input_fingerprint"}
or frozenset(keys) not in {frozenset(base_keys), frozenset(base_keys | catalog_keys)}
or (keys == base_keys | catalog_keys and (
not isinstance(binding["catalog_database_id"], str)
or not binding["catalog_database_id"]
or not isinstance(binding["metadata_content_revision"], int)
or binding["metadata_content_revision"] < 0
))
or payload.get("binding_sha256") != _binding_digest(binding)
):
raise ValueError
@@ -168,7 +187,7 @@ def _claim_or_validate_root_binding(
finally:
os.close(generations_fd)
payload = {
"schema_version": 1,
"schema_version": 2 if "catalog_database_id" in binding else 1,
"binding": binding,
"binding_sha256": _binding_digest(binding),
}
@@ -557,6 +576,8 @@ class DwhPreprocessPipeline:
workspace_root: Path,
config_fingerprint: str,
input_fingerprint: str,
catalog_database_id: str | None = None,
metadata_content_revision: int | None = None,
introspect: Callable[[Path], object],
build_lsh: Callable[[Path, Path], object],
lsh_filenames: tuple[str, str, str] | None = None,
@@ -569,6 +590,8 @@ class DwhPreprocessPipeline:
self.workspace_root = workspace_root
self.config_fingerprint = config_fingerprint
self.input_fingerprint = input_fingerprint
self.catalog_database_id = catalog_database_id
self.metadata_content_revision = metadata_content_revision
self.introspect = introspect
self.build_lsh = build_lsh
self.lsh_filenames = lsh_filenames or (
@@ -588,12 +611,20 @@ class DwhPreprocessPipeline:
raise ValueError("LSH filenames must be unique flat safe names")
@property
def binding(self) -> dict[str, str]:
return {
def binding(self) -> dict[str, str | int]:
binding: dict[str, str | int] = {
"workspace_id": self.workspace_id,
"config_fingerprint": self.config_fingerprint,
"input_fingerprint": self.input_fingerprint,
}
if self.catalog_database_id is not None:
if self.metadata_content_revision is None:
raise ValueError("Catalog metadata revision is required for DWH binding")
binding.update({
"catalog_database_id": self.catalog_database_id,
"metadata_content_revision": self.metadata_content_revision,
})
return binding
def _assert_active_binding(self, root_fd: int) -> None:
_validate_root_binding_fd(root_fd, self.binding)
+155
View File
@@ -0,0 +1,155 @@
from __future__ import annotations
import json
from datetime import datetime, timezone
from pathlib import Path
from typing import Literal
from pydantic import BaseModel, Field, ValidationError, model_validator
from tht.mschema.models import (
Annotations,
ColumnPhysical,
ForeignKey,
PhysicalSchema,
TablePhysical,
)
class CatalogSnapshotError(ValueError):
"""The backend-produced Catalog Metadata Snapshot is absent or invalid."""
class CatalogSnapshotColumn(BaseModel):
id: str = Field(min_length=1)
name: str = Field(min_length=1)
ordinal_position: int = Field(alias="ordinalPosition", ge=1)
data_type: str = Field(alias="dataType", min_length=1)
is_nullable: bool = Field(alias="isNullable")
default_expression: str | None = Field(alias="defaultExpression")
primary_key_position: int | None = Field(alias="primaryKeyPosition", ge=1)
sensitive: bool
description: str | None
description_source: Literal["curated", "generated", "source_comment"] | None = Field(
alias="descriptionSource"
)
model_config = {"extra": "forbid", "populate_by_name": True}
class CatalogSnapshotTable(BaseModel):
id: str = Field(min_length=1)
name: str = Field(min_length=1)
description: str | None
description_source: Literal["curated", "generated", "source_comment"] | None = Field(
alias="descriptionSource"
)
columns: list[CatalogSnapshotColumn]
model_config = {"extra": "forbid", "populate_by_name": True}
@model_validator(mode="after")
def unique_columns(self):
names = [column.name for column in self.columns]
if len(names) != len(set(names)):
raise ValueError("catalog snapshot table contains duplicate columns")
return self
class CatalogSnapshotRelationship(BaseModel):
id: str = Field(min_length=1)
origin: Literal["physical", "generated", "manual"]
source_table: str = Field(alias="sourceTable", min_length=1)
source_columns: list[str] = Field(alias="sourceColumns", min_length=1)
target_table: str = Field(alias="targetTable", min_length=1)
target_columns: list[str] = Field(alias="targetColumns", min_length=1)
model_config = {"extra": "forbid", "populate_by_name": True}
@model_validator(mode="after")
def paired_columns(self):
if len(self.source_columns) != len(self.target_columns):
raise ValueError("catalog snapshot relationship columns are not paired")
return self
class CatalogMetadataSnapshot(BaseModel):
schema_version: Literal[1] = Field(alias="schemaVersion")
workspace_id: str = Field(alias="workspaceId", min_length=1)
database_id: str = Field(alias="databaseId", min_length=1)
database_name: str = Field(alias="databaseName", min_length=1)
schema_name: str = Field(alias="schemaName", min_length=1)
metadata_content_revision: int = Field(alias="metadataContentRevision", ge=0)
tables: list[CatalogSnapshotTable]
relationships: list[CatalogSnapshotRelationship]
model_config = {"extra": "forbid", "populate_by_name": True}
@model_validator(mode="after")
def valid_graph(self):
tables = {table.name: {column.name for column in table.columns} for table in self.tables}
if len(tables) != len(self.tables):
raise ValueError("catalog snapshot contains duplicate tables")
for relationship in self.relationships:
if relationship.source_table not in tables or relationship.target_table not in tables:
raise ValueError("catalog snapshot relationship references an unknown table")
if any(name not in tables[relationship.source_table] for name in relationship.source_columns):
raise ValueError("catalog snapshot relationship references an unknown source column")
if any(name not in tables[relationship.target_table] for name in relationship.target_columns):
raise ValueError("catalog snapshot relationship references an unknown target column")
return self
def to_schema_inputs(
self,
) -> tuple[PhysicalSchema, Annotations, dict[str, list[ForeignKey]]]:
relationships: dict[str, list[ForeignKey]] = {}
for relationship in self.relationships:
relationships.setdefault(relationship.source_table, []).append(
ForeignKey(
name=relationship.id,
columns=relationship.source_columns,
ref_table=relationship.target_table,
ref_columns=relationship.target_columns,
)
)
tables = {
table.name: TablePhysical(
comment=table.description or "",
columns={
column.name: ColumnPhysical(
type=column.data_type,
nullable=column.is_nullable,
pk=column.primary_key_position is not None,
default=column.default_expression,
comment=column.description or "",
eligible=not column.sensitive,
eligibility_reason="sensitive" if column.sensitive else "catalog",
)
for column in table.columns
},
foreign_keys=relationships.get(table.name, []),
)
for table in self.tables
}
return (
PhysicalSchema(
database=self.database_name,
schema=self.schema_name,
introspected_at=datetime.now(timezone.utc),
tables=tables,
),
Annotations(),
relationships,
)
def load_catalog_metadata_snapshot(path: Path, workspace_id: str | None) -> CatalogMetadataSnapshot:
if not path.is_file():
raise CatalogSnapshotError(f"catalog metadata snapshot is missing: {path}")
try:
snapshot = CatalogMetadataSnapshot.model_validate(json.loads(path.read_text()))
except (OSError, json.JSONDecodeError, ValidationError) as exc:
raise CatalogSnapshotError("catalog metadata snapshot is invalid") from exc
if workspace_id is not None and snapshot.workspace_id != workspace_id:
raise CatalogSnapshotError("catalog metadata snapshot belongs to another workspace")
return snapshot
+18
View File
@@ -118,6 +118,24 @@ def _load_effective_relationships(cfg, physical: PhysicalSchema) -> dict[str, li
def load_schema_context(cfg, *, physical_file: Path | None = None) -> SchemaContext:
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
from tht.mschema.catalog_snapshot import (
CatalogSnapshotError,
load_catalog_metadata_snapshot,
)
try:
snapshot = load_catalog_metadata_snapshot(
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
)
physical, annotations, relationships = snapshot.to_schema_inputs()
except CatalogSnapshotError as exc:
raise SchemaContextError(str(exc)) from exc
return SchemaContext(
physical=physical,
annotations=annotations,
effective_relationships=relationships,
)
physical_file = physical_file or physical_path(cfg)
if not physical_file.is_file():
raise SchemaContextError(f"physical schema is missing: {physical_file}")
+4
View File
@@ -86,10 +86,14 @@ class VectorStore(Protocol):
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: ...
def delete_kinds(self, collection: str, kinds: list[str]) -> int: ...
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ...
def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ...
def clear_reference(self) -> bool: ...
__all__ = [
"VectorCapabilities",
+9 -1
View File
@@ -82,6 +82,8 @@ def _vector_key(hit) -> str:
return f"column:{hit.ref}"
if hit.kind == "schema_table":
return f"table:{hit.ref}"
if hit.kind == "schema_relationship":
return f"relationship:{hit.ref}"
return f"evidence:{hit.id}"
@@ -92,7 +94,13 @@ def schema_tables(results: list["SearchResult"], top_tables: int) -> list[tuple[
ignorati."""
best: dict[str, float] = {}
for r in results:
if r.kind not in ("schema_table", "schema_column"):
if r.kind not in ("schema_table", "schema_column", "schema_relationship"):
continue
if r.kind == "schema_relationship":
endpoints = r.key.split(":", 1)[1].split("->", 1)
for table in endpoints:
if table not in best or r.rrf > best[table]:
best[table] = r.rrf
continue
table = r.key.split(":", 1)[1].split(".", 1)[0]
if table not in best or r.rrf > best[table]:
+1
View File
@@ -4,6 +4,7 @@
KIND_TO_TABLE = {
"schema_table": "schema_records",
"schema_column": "schema_records",
"schema_relationship": "schema_records",
"evidence": "evidence",
"memory": "memory",
"solved_question": "memory", # coppie domanda->SQL: stessa tabella, kind dedicato
+64 -2
View File
@@ -15,7 +15,7 @@ MAX_EXAMPLES_IN_RECORD = 5
class VectorRecord(BaseModel):
id: str
kind: str # evidence | schema_table | schema_column
kind: str # evidence | schema_table | schema_column | schema_relationship
ref: str # file/chiave canonica di provenienza
title: str
content: str
@@ -23,7 +23,7 @@ class VectorRecord(BaseModel):
def qdrant_semantic_kind(kind: str) -> str:
if kind in {"schema_table", "schema_column"}:
if kind in {"schema_table", "schema_column", "schema_relationship"}:
return "schema"
if kind in {"memory", "solved_question"}:
return "memory"
@@ -136,3 +136,65 @@ def schema_records(physical: PhysicalSchema, annotations: Annotations) -> list[V
)
)
return records
def catalog_schema_records(snapshot) -> list[VectorRecord]:
"""Build the complete schema slice from a Catalog Metadata Snapshot."""
records: list[VectorRecord] = []
for table in snapshot.tables:
column_names = [column.name for column in table.columns]
lines = [f"Tabella {table.name}"]
if table.description:
lines.append(table.description)
lines.append("Colonne: " + ", ".join(column_names))
records.append(
VectorRecord(
id=f"schema_table:{table.id}",
kind="schema_table",
ref=table.name,
title=table.name,
content="\n".join(lines),
metadata={"tables": [table.name]},
)
)
for column in table.columns:
lines = [f"Colonna {table.name}.{column.name}", f"Tipo: {column.data_type}"]
if column.description:
lines.append(column.description)
if column.sensitive:
lines.append("Dato sensibile")
records.append(
VectorRecord(
id=f"schema_column:{column.id}",
kind="schema_column",
ref=f"{table.name}.{column.name}",
title=f"{table.name}.{column.name}",
content="\n".join(lines),
metadata={"tables": [table.name], "sensitive": column.sensitive},
)
)
for relationship in snapshot.relationships:
pairs = ", ".join(
f"{relationship.source_table}.{source} -> "
f"{relationship.target_table}.{target}"
for source, target in zip(
relationship.source_columns, relationship.target_columns, strict=True
)
)
ref = f"{relationship.source_table}->{relationship.target_table}"
records.append(
VectorRecord(
id=f"schema_relationship:{relationship.id}",
kind="schema_relationship",
ref=ref,
title=ref,
content=f"Relazione {relationship.origin}: {pairs}",
metadata={
"tables": [relationship.source_table, relationship.target_table],
"origin": relationship.origin,
"source_table": relationship.source_table,
"target_table": relationship.target_table,
},
)
)
return records