feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s
Publish documentation / publish (push) Successful in 2m12s
This commit is contained in:
+41
-10
@@ -3,7 +3,7 @@ from pathlib import Path
|
||||
from tht.cli.schema_cmd import physical_path
|
||||
|
||||
|
||||
def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
def _extract_lsh_values(dwh, physical, annotations, limit, eligibility_cfg=None):
|
||||
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
|
||||
from tht.mschema.eligibility import effective_eligibility
|
||||
|
||||
@@ -12,7 +12,16 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
table_ann = annotations.tables.get(table_name)
|
||||
for column_name, column in table.columns.items():
|
||||
ann_col = table_ann.columns.get(column_name) if table_ann else None
|
||||
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
|
||||
from_catalog = column.eligibility_reason in {"catalog", "sensitive"}
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
if from_catalog and column.eligibility_reason == "sensitive":
|
||||
continue
|
||||
if from_catalog and eligibility_cfg is not None and (
|
||||
column_name.lower() in {name.lower() for name in eligibility_cfg.ignore_columns}
|
||||
):
|
||||
continue
|
||||
if not from_catalog and not effective_eligibility(column, ann_col)[0]:
|
||||
continue
|
||||
try:
|
||||
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
|
||||
@@ -20,6 +29,21 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
|
||||
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
|
||||
continue
|
||||
vals = [str(value) for value in distinct.values if value not in (None, "")]
|
||||
if from_catalog and eligibility_cfg is not None:
|
||||
from tht.mschema.eligibility import classify_column
|
||||
|
||||
lengths = [len(value) for value in vals]
|
||||
eligible, reason = classify_column(
|
||||
column.type,
|
||||
column.is_enum,
|
||||
(sum(lengths) / len(lengths)) if lengths else None,
|
||||
max(lengths) if lengths else None,
|
||||
eligibility_cfg,
|
||||
)
|
||||
column.eligible = eligible
|
||||
column.eligibility_reason = reason
|
||||
if not eligible:
|
||||
continue
|
||||
if vals:
|
||||
values.setdefault(table_name, {})[column_name] = vals
|
||||
if distinct.truncated:
|
||||
@@ -33,18 +57,25 @@ def build_lsh_artifacts(
|
||||
):
|
||||
"""Run the existing LSH extraction/build algorithm and persist its outputs."""
|
||||
from tht.adapters.factory import build_dwh
|
||||
from tht.cli.schema_cmd import annotations_path
|
||||
from tht.lshindex import build_index, save_index
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
phys_file = physical_file or physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations = Annotations.from_yaml(annotations_path(cfg))
|
||||
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
|
||||
from tht.mschema.context import load_schema_context
|
||||
|
||||
context = load_schema_context(cfg)
|
||||
physical, annotations = context.physical, context.annotations
|
||||
else:
|
||||
from tht.cli.schema_cmd import annotations_path
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
phys_file = physical_file or physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations = Annotations.from_yaml(annotations_path(cfg))
|
||||
target = dwh if dwh is not None else build_dwh(cfg)
|
||||
values, skipped, truncated = _extract_lsh_values(
|
||||
target, physical, annotations, cfg.lsh.max_values_per_column
|
||||
target, physical, annotations, cfg.lsh.max_values_per_column, cfg.eligibility
|
||||
)
|
||||
lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose)
|
||||
save_index(
|
||||
|
||||
@@ -4,7 +4,10 @@ from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import stat
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
@@ -144,6 +147,8 @@ def run_dwh_from_config(
|
||||
workspace_root=workspace_root,
|
||||
config_fingerprint=binding["config_fingerprint"],
|
||||
input_fingerprint=binding["input_fingerprint"],
|
||||
catalog_database_id=binding.get("catalog_database_id"),
|
||||
metadata_content_revision=binding.get("metadata_content_revision"),
|
||||
introspect=lambda output: refresh_catalog(cfg, output_path=output),
|
||||
build_lsh=lambda physical, output: build_lsh_artifacts(
|
||||
cfg, physical_file=physical, output_dir=output
|
||||
@@ -155,6 +160,57 @@ def run_dwh_from_config(
|
||||
return pipeline.run(steps, resume_run_id=resume)
|
||||
|
||||
|
||||
def run_catalog_dwh_from_config(cfg):
|
||||
"""Publish Catalog-derived physical schema and LSH as one bound generation."""
|
||||
from tht.cli.lsh_cmd import build_lsh_artifacts
|
||||
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
|
||||
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
|
||||
|
||||
snapshot = load_catalog_metadata_snapshot(
|
||||
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
|
||||
)
|
||||
binding = config_dwh_binding(cfg)
|
||||
workspace_root = cfg.paths.artifacts.parent
|
||||
# Catalog preprocessing is a replace-in-place operation. Its DWH/LSH output is wholly
|
||||
# derived, there is no supported concurrent runtime, and a failed Clear may have left an
|
||||
# older binding behind. Start from an empty owned generation root so a retry can always
|
||||
# rebuild the current Catalog revision instead of deadlocking on the stale OWNER marker.
|
||||
_remove_owned_derived_path(workspace_root, workspace_root / ".tht-dwh")
|
||||
_remove_owned_derived_path(workspace_root, cfg.paths.artifacts / "mschema" / "physical.yaml")
|
||||
_remove_owned_derived_path(workspace_root, cfg.paths.indexes / "lsh")
|
||||
observed: dict[str, object] = {}
|
||||
|
||||
def materialize_physical(output: Path):
|
||||
physical, _annotations, _relationships = snapshot.to_schema_inputs()
|
||||
physical.to_yaml(output)
|
||||
return physical
|
||||
|
||||
def materialize_lsh(physical: Path, output: Path):
|
||||
result = build_lsh_artifacts(cfg, physical_file=physical, output_dir=output)
|
||||
observed["result"] = result
|
||||
return result
|
||||
|
||||
report = DwhPreprocessPipeline(
|
||||
workspace_id=str(binding["workspace_id"]),
|
||||
workspace_root=workspace_root,
|
||||
config_fingerprint=str(binding["config_fingerprint"]),
|
||||
input_fingerprint=str(binding["input_fingerprint"]),
|
||||
catalog_database_id=str(binding["catalog_database_id"]),
|
||||
metadata_content_revision=int(binding["metadata_content_revision"]),
|
||||
introspect=materialize_physical,
|
||||
build_lsh=materialize_lsh,
|
||||
lsh_filenames=(
|
||||
f"{cfg.database.db_schema}_lsh.pkl",
|
||||
f"{cfg.database.db_schema}_minhashes.pkl",
|
||||
f"{cfg.database.db_schema}_meta.json",
|
||||
),
|
||||
).run(("introspect", "lsh"))
|
||||
result = observed.get("result")
|
||||
if not isinstance(result, tuple) or len(result) != 4:
|
||||
raise RuntimeError("Catalog LSH publication did not complete")
|
||||
return report, result
|
||||
|
||||
|
||||
def _parse_dwh_steps(value: str) -> tuple[str, ...]:
|
||||
allowed = ("introspect", "lsh")
|
||||
steps = tuple(part.strip() for part in value.split(",") if part.strip())
|
||||
@@ -234,6 +290,85 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
|
||||
return pipeline.gc(workspace_root=corpus_root.parent, dry_run=dry_run)
|
||||
|
||||
|
||||
def _remove_owned_derived_path(workspace_root: Path, target: Path) -> bool:
|
||||
"""Remove one generated path without following links or escaping the workspace."""
|
||||
root = workspace_root.resolve()
|
||||
resolved = target.resolve(strict=False)
|
||||
if not resolved.is_relative_to(root):
|
||||
raise RuntimeError("derived cleanup target escapes the workspace")
|
||||
try:
|
||||
info = target.lstat()
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
if stat.S_ISLNK(info.st_mode) or info.st_uid != os.getuid():
|
||||
raise RuntimeError("derived cleanup target is unsafe")
|
||||
if stat.S_ISDIR(info.st_mode):
|
||||
shutil.rmtree(target)
|
||||
elif stat.S_ISREG(info.st_mode):
|
||||
target.unlink()
|
||||
else:
|
||||
raise RuntimeError("derived cleanup target is unsafe")
|
||||
return True
|
||||
|
||||
|
||||
def clear_from_config(config: Path) -> dict[str, int]:
|
||||
"""Clear workspace reference vectors and local preprocessing derivatives."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "preprocess clear")
|
||||
workspace_root = cfg.paths.artifacts.parent
|
||||
vector_store = build_vector_store(cfg, require_write=True)
|
||||
reference_deleted = int(vector_store.clear_reference())
|
||||
paths = [
|
||||
workspace_root / ".tht-dwh",
|
||||
workspace_root / ".tht-jobs",
|
||||
workspace_root / "corpus",
|
||||
cfg.paths.artifacts / "mschema" / "physical.yaml",
|
||||
cfg.paths.indexes / "lsh",
|
||||
]
|
||||
if cfg.paths.catalog_metadata_snapshot is not None:
|
||||
paths.append(cfg.paths.catalog_metadata_snapshot)
|
||||
removed = sum(int(_remove_owned_derived_path(workspace_root, path)) for path in paths)
|
||||
return {"referenceCollections": reference_deleted, "derivedPaths": removed}
|
||||
|
||||
|
||||
@preprocess_app.command("clear", hidden=True)
|
||||
def clear_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Clear replaceable preprocessing output while preserving workspace memory."""
|
||||
try:
|
||||
counts = clear_from_config(config)
|
||||
except Exception: # noqa: BLE001 - do not disclose paths, endpoints, or credentials
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"code": "preprocessing_clear_failed",
|
||||
"operation": "preprocess_clear",
|
||||
"error": "Preprocessing clear failed",
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.secho("ERRORE: preprocessing clear failed", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "succeeded",
|
||||
"code": "ok",
|
||||
"operation": "preprocess_clear",
|
||||
"counts": counts,
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.echo("OK: reference vectors and LSH cleared; memory preserved")
|
||||
|
||||
|
||||
@preprocess_app.command("evidence")
|
||||
def evidence_cmd(
|
||||
action: str | None = typer.Argument(None),
|
||||
@@ -359,3 +494,71 @@ def dwh_cmd(
|
||||
typer.secho(f"ERRORE: run={result.run_id} DWH preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
if result.status != "succeeded":
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@preprocess_app.command("catalog", hidden=True)
|
||||
def catalog_cmd(
|
||||
catalog_metadata: Path = typer.Option(..., "--catalog-metadata"),
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Build current LSH and schema vectors from one immutable Catalog snapshot."""
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.vector_cmd import index_catalog_schema, require_vector_cfg
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
cfg = cfg.model_copy(
|
||||
update={
|
||||
"paths": cfg.paths.model_copy(
|
||||
update={"catalog_metadata_snapshot": catalog_metadata}
|
||||
)
|
||||
}
|
||||
)
|
||||
require_vector_write_allowed(cfg, "preprocess catalog")
|
||||
require_vector_cfg(cfg)
|
||||
try:
|
||||
report, (minhashes, skipped, truncated, _values) = run_catalog_dwh_from_config(cfg)
|
||||
if report.status != "succeeded":
|
||||
raise RuntimeError("Catalog DWH preprocessing failed")
|
||||
stats, counts, snapshot_path = index_catalog_schema(cfg)
|
||||
except Exception: # noqa: BLE001 - public output must never disclose endpoints or SQL
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"code": "catalog_preprocessing_failed",
|
||||
"operation": "preprocess_catalog",
|
||||
"error": "Catalog preprocessing failed",
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, sort_keys=True))
|
||||
else:
|
||||
typer.secho("ERRORE: Catalog preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"status": "succeeded",
|
||||
"code": "ok",
|
||||
"operation": "preprocess_catalog",
|
||||
"artifactIdentities": [
|
||||
{
|
||||
"kind": "catalog_metadata_snapshot",
|
||||
"digest": "sha256:" + hashlib.sha256(snapshot_path.read_bytes()).hexdigest(),
|
||||
}
|
||||
],
|
||||
"counts": {
|
||||
**counts,
|
||||
"lshEntries": len(minhashes),
|
||||
"lshSkippedColumns": len(skipped),
|
||||
"lshTruncatedColumns": len(truncated),
|
||||
"added": stats.added,
|
||||
"deleted": stats.deleted,
|
||||
"unchanged": stats.unchanged,
|
||||
"updated": stats.updated,
|
||||
},
|
||||
}
|
||||
if json_output:
|
||||
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
|
||||
else:
|
||||
typer.echo("OK: Catalog metadata, LSH and schema vectors rebuilt")
|
||||
|
||||
@@ -560,22 +560,14 @@ def render_cmd(
|
||||
),
|
||||
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
|
||||
) -> None:
|
||||
"""Serializza mschema (physical + annotations) nel formato richiesto."""
|
||||
"""Serialize the current PostgreSQL Catalog projection."""
|
||||
import json
|
||||
|
||||
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED,
|
||||
err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
context = load_schema_context(cfg, physical_file=phys_file)
|
||||
context = load_schema_context(cfg)
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
@@ -625,17 +617,12 @@ def columns_cmd(
|
||||
"""Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo."""
|
||||
import json as _json
|
||||
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
try:
|
||||
physical = load_schema_context(cfg).physical
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
|
||||
def _payload(name, tbl):
|
||||
return {
|
||||
|
||||
@@ -9,7 +9,7 @@ from tht.config import workspace_id_for_config
|
||||
|
||||
KIND_MAP = {
|
||||
"evidence": ["evidence"],
|
||||
"schema": ["schema_table", "schema_column"],
|
||||
"schema": ["schema_table", "schema_column", "schema_relationship"],
|
||||
"values": [], # solo LSH
|
||||
"formula": ["evidence"],
|
||||
}
|
||||
@@ -235,15 +235,8 @@ def search_cmd(
|
||||
from tht.mschema.render import to_mschema_text
|
||||
from tht.search import schema_tables
|
||||
|
||||
phys_file = dwh_snapshot.physical
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
schema_context = load_schema_context(cfg, physical_file=phys_file)
|
||||
schema_context = load_schema_context(cfg)
|
||||
except SchemaContextError as exc:
|
||||
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
@@ -390,7 +383,7 @@ def pack_cmd(
|
||||
|
||||
workspace_id = workspace_id_for_config(cfg, config)
|
||||
validate_corpus_workspace(cfg, workspace_id)
|
||||
dwh_snapshot = _leased_dwh_snapshot(cfg, ctx)
|
||||
_leased_dwh_snapshot(cfg, ctx)
|
||||
require_vector_cfg(cfg)
|
||||
|
||||
tables: list[dict] = []
|
||||
@@ -414,12 +407,13 @@ def pack_cmd(
|
||||
|
||||
if vec is not None:
|
||||
descriptions: dict[str, str] = {}
|
||||
phys_file = dwh_snapshot.physical
|
||||
if phys_file.exists():
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
from tht.mschema.context import SchemaContextError, load_schema_context
|
||||
|
||||
phys = PhysicalSchema.from_yaml(phys_file)
|
||||
try:
|
||||
phys = load_schema_context(cfg).physical
|
||||
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
|
||||
except SchemaContextError:
|
||||
pass
|
||||
try:
|
||||
cand = combined_search(
|
||||
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
|
||||
|
||||
@@ -4,7 +4,7 @@ from pathlib import Path
|
||||
import typer
|
||||
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
|
||||
sql_app = typer.Typer(help="Validazione ed esecuzione controllata di SQL (read-only)")
|
||||
|
||||
@@ -17,16 +17,16 @@ def _read_sql(file: Path) -> str:
|
||||
|
||||
|
||||
def _load_physical_or_exit(cfg):
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
from tht.mschema.context import SchemaContextError, load_schema_context
|
||||
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
try:
|
||||
return load_schema_context(cfg).physical
|
||||
except SchemaContextError:
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
"ERRORE: Catalog Metadata Snapshot non disponibile. Esegui il preprocessing.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
return PhysicalSchema.from_yaml(phys_file)
|
||||
|
||||
|
||||
def require_action(cfg, action: str) -> None:
|
||||
|
||||
@@ -6,7 +6,7 @@ import typer
|
||||
|
||||
from tht.cli._guards import require_vector_write_allowed
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.vectorstore.store import SyncStats, content_hash
|
||||
|
||||
@@ -112,42 +112,14 @@ def index_schema_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
) -> None:
|
||||
"""Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
"""Replace the schema slice from the PostgreSQL Catalog projection."""
|
||||
from tht.ports.vector import VectorStoreError
|
||||
from tht.vectorstore.records import schema_records
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "vector index-schema")
|
||||
require_vector_cfg(cfg)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
message = f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`."
|
||||
if json_output:
|
||||
_emit_json({
|
||||
"code": "schema_missing",
|
||||
"error": "physical schema is missing",
|
||||
"operation": "index_schema",
|
||||
"schemaVersion": 1,
|
||||
"status": "failed",
|
||||
"workspaceId": cfg._workspace_id,
|
||||
"workspaceRevision": cfg._workspace_revision,
|
||||
})
|
||||
else:
|
||||
typer.secho(message, fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
annotations_file = annotations_path(cfg)
|
||||
annotations = Annotations.from_yaml(annotations_file)
|
||||
records = schema_records(physical, annotations)
|
||||
try:
|
||||
stats = sync_canonical_records(
|
||||
"schema_records",
|
||||
records,
|
||||
store=build_vector_store(cfg, require_write=True),
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
stats, counts, snapshot_path = index_catalog_schema(cfg)
|
||||
except VectorStoreError as exc:
|
||||
code = str(exc)
|
||||
error = "semantic index incompatible" if code == "semantic_index_incompatible" else "schema indexing failed"
|
||||
@@ -167,17 +139,17 @@ def index_schema_cmd(
|
||||
if json_output:
|
||||
_emit_json({
|
||||
"artifactIdentities": [
|
||||
{"digest": _artifact_digest(annotations_file), "kind": "schema_annotations"},
|
||||
{"digest": _artifact_digest(phys_file), "kind": "physical_schema"},
|
||||
{"digest": _artifact_digest(snapshot_path), "kind": "catalog_metadata_snapshot"},
|
||||
],
|
||||
"code": "ok",
|
||||
"collection": cfg.vectors.collection,
|
||||
"collection": cfg.vectors.collections["reference"],
|
||||
"counts": {
|
||||
"added": stats.added,
|
||||
"columns": sum(len(table.columns) for table in physical.tables.values()),
|
||||
"columns": counts["columns"],
|
||||
"deleted": stats.deleted,
|
||||
"records": len(records),
|
||||
"tables": len(physical.tables),
|
||||
"records": counts["records"],
|
||||
"relationships": counts["relationships"],
|
||||
"tables": counts["tables"],
|
||||
"unchanged": stats.unchanged,
|
||||
"updated": stats.updated,
|
||||
},
|
||||
@@ -189,3 +161,32 @@ def index_schema_cmd(
|
||||
})
|
||||
return
|
||||
_print_stats(stats)
|
||||
|
||||
|
||||
def index_catalog_schema(cfg):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
|
||||
from tht.vectorstore.records import catalog_schema_records
|
||||
|
||||
snapshot_path = cfg.paths.catalog_metadata_snapshot
|
||||
if snapshot_path is None:
|
||||
raise ValueError("catalog metadata snapshot is not configured")
|
||||
snapshot = load_catalog_metadata_snapshot(snapshot_path, cfg._workspace_id)
|
||||
records = catalog_schema_records(snapshot)
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
deleted = store.delete_kinds(
|
||||
"schema_records", ["schema_table", "schema_column", "schema_relationship"]
|
||||
)
|
||||
stats = sync_canonical_records(
|
||||
"schema_records",
|
||||
records,
|
||||
store=store,
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
stats.deleted = deleted
|
||||
return stats, {
|
||||
"tables": len(snapshot.tables),
|
||||
"columns": sum(len(table.columns) for table in snapshot.tables),
|
||||
"relationships": len(snapshot.relationships),
|
||||
"records": len(records),
|
||||
}, snapshot_path
|
||||
|
||||
Reference in New Issue
Block a user