feat: complete catalog-driven preprocessing
Publish documentation / publish (push) Successful in 2m12s

This commit is contained in:
Codex
2026-09-06 17:49:35 +02:00
parent 8707ae1d46
commit cffa60772e
141 changed files with 5898 additions and 3015 deletions
+41 -10
View File
@@ -3,7 +3,7 @@ from pathlib import Path
from tht.cli.schema_cmd import physical_path
def _extract_lsh_values(dwh, physical, annotations, limit):
def _extract_lsh_values(dwh, physical, annotations, limit, eligibility_cfg=None):
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
from tht.mschema.eligibility import effective_eligibility
@@ -12,7 +12,16 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
table_ann = annotations.tables.get(table_name)
for column_name, column in table.columns.items():
ann_col = table_ann.columns.get(column_name) if table_ann else None
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
from_catalog = column.eligibility_reason in {"catalog", "sensitive"}
if not is_text_type(column.type):
continue
if from_catalog and column.eligibility_reason == "sensitive":
continue
if from_catalog and eligibility_cfg is not None and (
column_name.lower() in {name.lower() for name in eligibility_cfg.ignore_columns}
):
continue
if not from_catalog and not effective_eligibility(column, ann_col)[0]:
continue
try:
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
@@ -20,6 +29,21 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
continue
vals = [str(value) for value in distinct.values if value not in (None, "")]
if from_catalog and eligibility_cfg is not None:
from tht.mschema.eligibility import classify_column
lengths = [len(value) for value in vals]
eligible, reason = classify_column(
column.type,
column.is_enum,
(sum(lengths) / len(lengths)) if lengths else None,
max(lengths) if lengths else None,
eligibility_cfg,
)
column.eligible = eligible
column.eligibility_reason = reason
if not eligible:
continue
if vals:
values.setdefault(table_name, {})[column_name] = vals
if distinct.truncated:
@@ -33,18 +57,25 @@ def build_lsh_artifacts(
):
"""Run the existing LSH extraction/build algorithm and persist its outputs."""
from tht.adapters.factory import build_dwh
from tht.cli.schema_cmd import annotations_path
from tht.lshindex import build_index, save_index
from tht.mschema.models import Annotations, PhysicalSchema
phys_file = physical_file or physical_path(cfg)
if not phys_file.exists():
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None:
from tht.mschema.context import load_schema_context
context = load_schema_context(cfg)
physical, annotations = context.physical, context.annotations
else:
from tht.cli.schema_cmd import annotations_path
from tht.mschema.models import Annotations, PhysicalSchema
phys_file = physical_file or physical_path(cfg)
if not phys_file.exists():
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
target = dwh if dwh is not None else build_dwh(cfg)
values, skipped, truncated = _extract_lsh_values(
target, physical, annotations, cfg.lsh.max_values_per_column
target, physical, annotations, cfg.lsh.max_values_per_column, cfg.eligibility
)
lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose)
save_index(
+203
View File
@@ -4,7 +4,10 @@ from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import stat
from pathlib import Path
import typer
@@ -144,6 +147,8 @@ def run_dwh_from_config(
workspace_root=workspace_root,
config_fingerprint=binding["config_fingerprint"],
input_fingerprint=binding["input_fingerprint"],
catalog_database_id=binding.get("catalog_database_id"),
metadata_content_revision=binding.get("metadata_content_revision"),
introspect=lambda output: refresh_catalog(cfg, output_path=output),
build_lsh=lambda physical, output: build_lsh_artifacts(
cfg, physical_file=physical, output_dir=output
@@ -155,6 +160,57 @@ def run_dwh_from_config(
return pipeline.run(steps, resume_run_id=resume)
def run_catalog_dwh_from_config(cfg):
"""Publish Catalog-derived physical schema and LSH as one bound generation."""
from tht.cli.lsh_cmd import build_lsh_artifacts
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
snapshot = load_catalog_metadata_snapshot(
cfg.paths.catalog_metadata_snapshot, cfg._workspace_id
)
binding = config_dwh_binding(cfg)
workspace_root = cfg.paths.artifacts.parent
# Catalog preprocessing is a replace-in-place operation. Its DWH/LSH output is wholly
# derived, there is no supported concurrent runtime, and a failed Clear may have left an
# older binding behind. Start from an empty owned generation root so a retry can always
# rebuild the current Catalog revision instead of deadlocking on the stale OWNER marker.
_remove_owned_derived_path(workspace_root, workspace_root / ".tht-dwh")
_remove_owned_derived_path(workspace_root, cfg.paths.artifacts / "mschema" / "physical.yaml")
_remove_owned_derived_path(workspace_root, cfg.paths.indexes / "lsh")
observed: dict[str, object] = {}
def materialize_physical(output: Path):
physical, _annotations, _relationships = snapshot.to_schema_inputs()
physical.to_yaml(output)
return physical
def materialize_lsh(physical: Path, output: Path):
result = build_lsh_artifacts(cfg, physical_file=physical, output_dir=output)
observed["result"] = result
return result
report = DwhPreprocessPipeline(
workspace_id=str(binding["workspace_id"]),
workspace_root=workspace_root,
config_fingerprint=str(binding["config_fingerprint"]),
input_fingerprint=str(binding["input_fingerprint"]),
catalog_database_id=str(binding["catalog_database_id"]),
metadata_content_revision=int(binding["metadata_content_revision"]),
introspect=materialize_physical,
build_lsh=materialize_lsh,
lsh_filenames=(
f"{cfg.database.db_schema}_lsh.pkl",
f"{cfg.database.db_schema}_minhashes.pkl",
f"{cfg.database.db_schema}_meta.json",
),
).run(("introspect", "lsh"))
result = observed.get("result")
if not isinstance(result, tuple) or len(result) != 4:
raise RuntimeError("Catalog LSH publication did not complete")
return report, result
def _parse_dwh_steps(value: str) -> tuple[str, ...]:
allowed = ("introspect", "lsh")
steps = tuple(part.strip() for part in value.split(",") if part.strip())
@@ -234,6 +290,85 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
return pipeline.gc(workspace_root=corpus_root.parent, dry_run=dry_run)
def _remove_owned_derived_path(workspace_root: Path, target: Path) -> bool:
"""Remove one generated path without following links or escaping the workspace."""
root = workspace_root.resolve()
resolved = target.resolve(strict=False)
if not resolved.is_relative_to(root):
raise RuntimeError("derived cleanup target escapes the workspace")
try:
info = target.lstat()
except FileNotFoundError:
return False
if stat.S_ISLNK(info.st_mode) or info.st_uid != os.getuid():
raise RuntimeError("derived cleanup target is unsafe")
if stat.S_ISDIR(info.st_mode):
shutil.rmtree(target)
elif stat.S_ISREG(info.st_mode):
target.unlink()
else:
raise RuntimeError("derived cleanup target is unsafe")
return True
def clear_from_config(config: Path) -> dict[str, int]:
"""Clear workspace reference vectors and local preprocessing derivatives."""
from tht.adapters.factory import build_vector_store
from tht.cli._guards import require_vector_write_allowed
from tht.cli.schema_cmd import _load_config_or_exit
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "preprocess clear")
workspace_root = cfg.paths.artifacts.parent
vector_store = build_vector_store(cfg, require_write=True)
reference_deleted = int(vector_store.clear_reference())
paths = [
workspace_root / ".tht-dwh",
workspace_root / ".tht-jobs",
workspace_root / "corpus",
cfg.paths.artifacts / "mschema" / "physical.yaml",
cfg.paths.indexes / "lsh",
]
if cfg.paths.catalog_metadata_snapshot is not None:
paths.append(cfg.paths.catalog_metadata_snapshot)
removed = sum(int(_remove_owned_derived_path(workspace_root, path)) for path in paths)
return {"referenceCollections": reference_deleted, "derivedPaths": removed}
@preprocess_app.command("clear", hidden=True)
def clear_cmd(
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Clear replaceable preprocessing output while preserving workspace memory."""
try:
counts = clear_from_config(config)
except Exception: # noqa: BLE001 - do not disclose paths, endpoints, or credentials
payload = {
"schemaVersion": 1,
"status": "failed",
"code": "preprocessing_clear_failed",
"operation": "preprocess_clear",
"error": "Preprocessing clear failed",
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.secho("ERRORE: preprocessing clear failed", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = {
"schemaVersion": 1,
"status": "succeeded",
"code": "ok",
"operation": "preprocess_clear",
"counts": counts,
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.echo("OK: reference vectors and LSH cleared; memory preserved")
@preprocess_app.command("evidence")
def evidence_cmd(
action: str | None = typer.Argument(None),
@@ -359,3 +494,71 @@ def dwh_cmd(
typer.secho(f"ERRORE: run={result.run_id} DWH preprocessing failed", fg=typer.colors.RED, err=True)
if result.status != "succeeded":
raise typer.Exit(code=1)
@preprocess_app.command("catalog", hidden=True)
def catalog_cmd(
catalog_metadata: Path = typer.Option(..., "--catalog-metadata"),
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Build current LSH and schema vectors from one immutable Catalog snapshot."""
from tht.cli._guards import require_vector_write_allowed
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import index_catalog_schema, require_vector_cfg
cfg = _load_config_or_exit(config)
cfg = cfg.model_copy(
update={
"paths": cfg.paths.model_copy(
update={"catalog_metadata_snapshot": catalog_metadata}
)
}
)
require_vector_write_allowed(cfg, "preprocess catalog")
require_vector_cfg(cfg)
try:
report, (minhashes, skipped, truncated, _values) = run_catalog_dwh_from_config(cfg)
if report.status != "succeeded":
raise RuntimeError("Catalog DWH preprocessing failed")
stats, counts, snapshot_path = index_catalog_schema(cfg)
except Exception: # noqa: BLE001 - public output must never disclose endpoints or SQL
payload = {
"schemaVersion": 1,
"status": "failed",
"code": "catalog_preprocessing_failed",
"operation": "preprocess_catalog",
"error": "Catalog preprocessing failed",
}
if json_output:
typer.echo(json.dumps(payload, sort_keys=True))
else:
typer.secho("ERRORE: Catalog preprocessing failed", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
payload = {
"schemaVersion": 1,
"status": "succeeded",
"code": "ok",
"operation": "preprocess_catalog",
"artifactIdentities": [
{
"kind": "catalog_metadata_snapshot",
"digest": "sha256:" + hashlib.sha256(snapshot_path.read_bytes()).hexdigest(),
}
],
"counts": {
**counts,
"lshEntries": len(minhashes),
"lshSkippedColumns": len(skipped),
"lshTruncatedColumns": len(truncated),
"added": stats.added,
"deleted": stats.deleted,
"unchanged": stats.unchanged,
"updated": stats.updated,
},
}
if json_output:
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
else:
typer.echo("OK: Catalog metadata, LSH and schema vectors rebuilt")
+6 -19
View File
@@ -560,22 +560,14 @@ def render_cmd(
),
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
) -> None:
"""Serializza mschema (physical + annotations) nel formato richiesto."""
"""Serialize the current PostgreSQL Catalog projection."""
import json
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED,
err=True,
)
raise typer.Exit(code=1)
try:
context = load_schema_context(cfg, physical_file=phys_file)
context = load_schema_context(cfg)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
@@ -625,17 +617,12 @@ def columns_cmd(
"""Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo."""
import json as _json
from tht.mschema.models import PhysicalSchema
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
try:
physical = load_schema_context(cfg).physical
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
def _payload(name, tbl):
return {
+8 -14
View File
@@ -9,7 +9,7 @@ from tht.config import workspace_id_for_config
KIND_MAP = {
"evidence": ["evidence"],
"schema": ["schema_table", "schema_column"],
"schema": ["schema_table", "schema_column", "schema_relationship"],
"values": [], # solo LSH
"formula": ["evidence"],
}
@@ -235,15 +235,8 @@ def search_cmd(
from tht.mschema.render import to_mschema_text
from tht.search import schema_tables
phys_file = dwh_snapshot.physical
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
try:
schema_context = load_schema_context(cfg, physical_file=phys_file)
schema_context = load_schema_context(cfg)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
@@ -390,7 +383,7 @@ def pack_cmd(
workspace_id = workspace_id_for_config(cfg, config)
validate_corpus_workspace(cfg, workspace_id)
dwh_snapshot = _leased_dwh_snapshot(cfg, ctx)
_leased_dwh_snapshot(cfg, ctx)
require_vector_cfg(cfg)
tables: list[dict] = []
@@ -414,12 +407,13 @@ def pack_cmd(
if vec is not None:
descriptions: dict[str, str] = {}
phys_file = dwh_snapshot.physical
if phys_file.exists():
from tht.mschema.models import PhysicalSchema
from tht.mschema.context import SchemaContextError, load_schema_context
phys = PhysicalSchema.from_yaml(phys_file)
try:
phys = load_schema_context(cfg).physical
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
except SchemaContextError:
pass
try:
cand = combined_search(
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
+6 -6
View File
@@ -4,7 +4,7 @@ from pathlib import Path
import typer
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
from tht.cli.schema_cmd import _load_config_or_exit
sql_app = typer.Typer(help="Validazione ed esecuzione controllata di SQL (read-only)")
@@ -17,16 +17,16 @@ def _read_sql(file: Path) -> str:
def _load_physical_or_exit(cfg):
from tht.mschema.models import PhysicalSchema
from tht.mschema.context import SchemaContextError, load_schema_context
phys_file = physical_path(cfg)
if not phys_file.exists():
try:
return load_schema_context(cfg).physical
except SchemaContextError:
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
"ERRORE: Catalog Metadata Snapshot non disponibile. Esegui il preprocessing.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
return PhysicalSchema.from_yaml(phys_file)
def require_action(cfg, action: str) -> None:
+38 -37
View File
@@ -6,7 +6,7 @@ import typer
from tht.cli._guards import require_vector_write_allowed
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit, annotations_path, physical_path
from tht.cli.schema_cmd import _load_config_or_exit
from tht.ports.vector import VectorWriteRecord
from tht.vectorstore.store import SyncStats, content_hash
@@ -112,42 +112,14 @@ def index_schema_cmd(
config: Path = CONFIG_OPT,
json_output: bool = typer.Option(False, "--json"),
) -> None:
"""Embedda e sincronizza i record schema (tabelle e colonne) nel semantic store."""
from tht.adapters.factory import build_vector_store
from tht.mschema.models import Annotations, PhysicalSchema
"""Replace the schema slice from the PostgreSQL Catalog projection."""
from tht.ports.vector import VectorStoreError
from tht.vectorstore.records import schema_records
cfg = _load_config_or_exit(config)
require_vector_write_allowed(cfg, "vector index-schema")
require_vector_cfg(cfg)
phys_file = physical_path(cfg)
if not phys_file.exists():
message = f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`."
if json_output:
_emit_json({
"code": "schema_missing",
"error": "physical schema is missing",
"operation": "index_schema",
"schemaVersion": 1,
"status": "failed",
"workspaceId": cfg._workspace_id,
"workspaceRevision": cfg._workspace_revision,
})
else:
typer.secho(message, fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
annotations_file = annotations_path(cfg)
annotations = Annotations.from_yaml(annotations_file)
records = schema_records(physical, annotations)
try:
stats = sync_canonical_records(
"schema_records",
records,
store=build_vector_store(cfg, require_write=True),
embedder=make_embedder(cfg.embeddings),
)
stats, counts, snapshot_path = index_catalog_schema(cfg)
except VectorStoreError as exc:
code = str(exc)
error = "semantic index incompatible" if code == "semantic_index_incompatible" else "schema indexing failed"
@@ -167,17 +139,17 @@ def index_schema_cmd(
if json_output:
_emit_json({
"artifactIdentities": [
{"digest": _artifact_digest(annotations_file), "kind": "schema_annotations"},
{"digest": _artifact_digest(phys_file), "kind": "physical_schema"},
{"digest": _artifact_digest(snapshot_path), "kind": "catalog_metadata_snapshot"},
],
"code": "ok",
"collection": cfg.vectors.collection,
"collection": cfg.vectors.collections["reference"],
"counts": {
"added": stats.added,
"columns": sum(len(table.columns) for table in physical.tables.values()),
"columns": counts["columns"],
"deleted": stats.deleted,
"records": len(records),
"tables": len(physical.tables),
"records": counts["records"],
"relationships": counts["relationships"],
"tables": counts["tables"],
"unchanged": stats.unchanged,
"updated": stats.updated,
},
@@ -189,3 +161,32 @@ def index_schema_cmd(
})
return
_print_stats(stats)
def index_catalog_schema(cfg):
from tht.adapters.factory import build_vector_store
from tht.mschema.catalog_snapshot import load_catalog_metadata_snapshot
from tht.vectorstore.records import catalog_schema_records
snapshot_path = cfg.paths.catalog_metadata_snapshot
if snapshot_path is None:
raise ValueError("catalog metadata snapshot is not configured")
snapshot = load_catalog_metadata_snapshot(snapshot_path, cfg._workspace_id)
records = catalog_schema_records(snapshot)
store = build_vector_store(cfg, require_write=True)
deleted = store.delete_kinds(
"schema_records", ["schema_table", "schema_column", "schema_relationship"]
)
stats = sync_canonical_records(
"schema_records",
records,
store=store,
embedder=make_embedder(cfg.embeddings),
)
stats.deleted = deleted
return stats, {
"tables": len(snapshot.tables),
"columns": sum(len(table.columns) for table in snapshot.tables),
"relationships": len(snapshot.relationships),
"records": len(records),
}, snapshot_path