feat: consolidate database management work

Add catalog-owned logical relationships and runtime snapshots, extend the database-management UI and validation coverage, and document the updated operational workflow.

Keep active sensitive-generation status in a tooltip and indicator, and update the layout E2E to follow the history action in its new database-scoped location.
This commit is contained in:
Codex
2026-09-01 14:46:55 +02:00
parent f586152636
commit 076c9742c5
73 changed files with 6966 additions and 610 deletions
+48 -21
View File
@@ -10,6 +10,12 @@ from tht.adapters.factory import build_dwh
from tht.cli.config_cmd import CONFIG_OPT
from tht.config import ConfigError, load_config
from tht.db.sampling import is_text_type
from tht.mschema.context import (
SchemaContextError,
annotations_path,
load_schema_context,
physical_path,
)
from tht.mschema.eligibility import classify_all
schema_app = typer.Typer(help="Gestione mschema (rappresentazione canonica dello schema)")
@@ -40,20 +46,6 @@ def _load_config_or_exit(config: Path):
raise typer.Exit(code=1)
def physical_path(cfg) -> Path:
from tht.jobs.dwh_pipeline import resolve_dwh_snapshot
if not (cfg.paths.artifacts.parent / ".tht-dwh").exists():
return cfg.paths.artifacts / "mschema" / "physical.yaml"
return resolve_dwh_snapshot(cfg).physical
def annotations_path(cfg) -> Path:
if cfg.paths.annotations_root is not None:
return cfg.paths.annotations_root / "mschema" / "annotations.yaml"
return cfg.paths.artifacts / "mschema" / "annotations.yaml"
def refresh_catalog(cfg, *, dwh=None, output_path: Path | None = None):
"""Run the existing catalog algorithm and persist its canonical output."""
target = dwh if dwh is not None else build_dwh(cfg)
@@ -432,6 +424,21 @@ def suggest_fks_cmd(
from tht.mschema.models import Annotations, PhysicalSchema, TableAnnotation
cfg = _load_config_or_exit(config)
if write and cfg.paths.effective_relationships is not None:
error = "effective relationships are managed by the catalog; --write is disabled"
if json_output:
_emit_json({
"code": "catalog_relationships_managed",
"error": error,
"operation": "schema_suggest_fks",
"schemaVersion": 1,
"status": "failed",
"workspaceId": cfg._workspace_id,
"workspaceRevision": cfg._workspace_revision,
})
else:
typer.secho(f"ERRORE: {error}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
phys_file = physical_path(cfg)
if not phys_file.exists():
if json_output:
@@ -556,7 +563,6 @@ def render_cmd(
"""Serializza mschema (physical + annotations) nel formato richiesto."""
import json
from tht.mschema.models import Annotations, PhysicalSchema
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
cfg = _load_config_or_exit(config)
@@ -564,19 +570,40 @@ def render_cmd(
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
fg=typer.colors.RED,
err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
try:
context = load_schema_context(cfg, physical_file=phys_file)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
table_filter = list(tables) if tables else None
if format == "markdown":
out = to_markdown(physical, annotations)
out = to_markdown(
context.physical,
context.annotations,
effective_relationships=context.effective_relationships,
)
elif format == "mschema-text":
out = to_mschema_text(physical, annotations, tables=table_filter)
out = to_mschema_text(
context.physical,
context.annotations,
tables=table_filter,
effective_relationships=context.effective_relationships,
)
elif format == "schema-dict":
out = json.dumps(to_schema_dict(physical, annotations), ensure_ascii=False, indent=2)
out = json.dumps(
to_schema_dict(
context.physical,
context.annotations,
effective_relationships=context.effective_relationships,
),
ensure_ascii=False,
indent=2,
)
else:
typer.secho(f"ERRORE: formato sconosciuto: {format}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
+12 -5
View File
@@ -231,8 +231,7 @@ def search_cmd(
)
if kind == "schema":
from tht.cli.schema_cmd import annotations_path
from tht.mschema.models import Annotations, PhysicalSchema
from tht.mschema.context import SchemaContextError, load_schema_context
from tht.mschema.render import to_mschema_text
from tht.search import schema_tables
@@ -243,6 +242,11 @@ def search_cmd(
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
try:
schema_context = load_schema_context(cfg, physical_file=phys_file)
except SchemaContextError as exc:
typer.secho(f"ERRORE: {exc}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1) from None
candidates = combined_search(
keyword=keyword, lsh_hits=lsh_hits,
@@ -258,10 +262,13 @@ def search_cmd(
typer.secho(f"Nessuna tabella candidata per '{keyword}'.", fg=typer.colors.YELLOW)
return
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
selected = [t for t, _ in ranked]
mschema = to_mschema_text(physical, annotations, tables=selected)
mschema = to_mschema_text(
schema_context.physical,
schema_context.annotations,
tables=selected,
effective_relationships=schema_context.effective_relationships,
)
if json_out:
typer.echo(json.dumps(
+4
View File
@@ -327,6 +327,9 @@ class PathsConfig(BaseModel):
# Revision-qualified curated FK annotations root (P5). When absent, legacy
# `artifacts/mschema/annotations.yaml` remains the annotations source.
annotations_root: Path | None = None
# Runtime-only, backend-derived effective relationship snapshot. When present,
# it is the exclusive FK source; it is not an authored workspace artifact.
effective_relationships: Path | None = None
class RuntimeIdentityConfig(BaseModel):
@@ -676,6 +679,7 @@ def load_config(path: Path) -> Config:
indexes=resolved.indexes,
memory=cfg.paths.memory,
annotations_root=cfg.paths.annotations_root,
effective_relationships=cfg.paths.effective_relationships,
)
}
)
+133
View File
@@ -0,0 +1,133 @@
import json
from dataclasses import dataclass
from pathlib import Path
from typing import Annotated, Literal
from pydantic import BaseModel, Field, ValidationError, model_validator
from tht.mschema.models import Annotations, ForeignKey, PhysicalSchema
class SchemaContextError(ValueError):
"""The effective schema inputs cannot be used safely."""
RelationshipName = Annotated[str, Field(min_length=1)]
class EffectiveRelationship(BaseModel):
source_table: RelationshipName = Field(alias="sourceTable")
source_columns: list[RelationshipName] = Field(alias="sourceColumns", min_length=1)
target_table: RelationshipName = Field(alias="targetTable")
target_columns: list[RelationshipName] = Field(alias="targetColumns", min_length=1)
origin: Literal["physical", "generated", "manual"]
model_config = {"extra": "forbid", "populate_by_name": True}
@model_validator(mode="after")
def columns_are_paired(self):
if len(self.source_columns) != len(self.target_columns):
raise ValueError("sourceColumns and targetColumns must have the same length")
return self
class EffectiveRelationshipSnapshot(BaseModel):
schema_version: Literal[1] = Field(alias="schemaVersion")
workspace_id: RelationshipName = Field(alias="workspaceId")
relationships: list[EffectiveRelationship]
model_config = {"extra": "forbid", "populate_by_name": True}
@dataclass(frozen=True)
class SchemaContext:
physical: PhysicalSchema
annotations: Annotations
# None means legacy physical + annotation FK merging. A dict, including an
# empty one, means the catalog snapshot is the exclusive relationship source.
effective_relationships: dict[str, list[ForeignKey]] | None
def physical_path(cfg) -> Path:
from tht.jobs.dwh_pipeline import resolve_dwh_snapshot
if not (cfg.paths.artifacts.parent / ".tht-dwh").exists():
return cfg.paths.artifacts / "mschema" / "physical.yaml"
return resolve_dwh_snapshot(cfg).physical
def annotations_path(cfg) -> Path:
if cfg.paths.annotations_root is not None:
return cfg.paths.annotations_root / "mschema" / "annotations.yaml"
return cfg.paths.artifacts / "mschema" / "annotations.yaml"
def _load_effective_relationships(cfg, physical: PhysicalSchema) -> dict[str, list[ForeignKey]] | None:
path = cfg.paths.effective_relationships
if path is None:
return None
if not path.is_file():
raise SchemaContextError(f"effective relationship snapshot is missing: {path}")
try:
raw = json.loads(path.read_text())
snapshot = EffectiveRelationshipSnapshot.model_validate(raw)
except (OSError, json.JSONDecodeError, ValidationError) as exc:
raise SchemaContextError(f"effective relationship snapshot is invalid: {path}") from exc
if snapshot.workspace_id != cfg._workspace_id:
raise SchemaContextError(
"effective relationship snapshot workspace does not match runtime workspace"
)
by_table: dict[str, list[ForeignKey]] = {}
seen: set[tuple[str, tuple[str, ...], str, tuple[str, ...]]] = set()
for relationship in snapshot.relationships:
source = physical.tables.get(relationship.source_table)
target = physical.tables.get(relationship.target_table)
if source is None or target is None:
raise SchemaContextError(
"effective relationship endpoint table is absent from physical schema: "
f"{relationship.source_table}->{relationship.target_table}"
)
missing_source = [name for name in relationship.source_columns if name not in source.columns]
missing_target = [name for name in relationship.target_columns if name not in target.columns]
if missing_source or missing_target:
missing = ", ".join(
[f"{relationship.source_table}.{name}" for name in missing_source]
+ [f"{relationship.target_table}.{name}" for name in missing_target]
)
raise SchemaContextError(
f"effective relationship endpoint column is absent from physical schema: {missing}"
)
key = (
relationship.source_table,
tuple(relationship.source_columns),
relationship.target_table,
tuple(relationship.target_columns),
)
if key in seen:
continue
seen.add(key)
by_table.setdefault(relationship.source_table, []).append(
ForeignKey(
columns=relationship.source_columns,
ref_table=relationship.target_table,
ref_columns=relationship.target_columns,
)
)
return by_table
def load_schema_context(cfg, *, physical_file: Path | None = None) -> SchemaContext:
physical_file = physical_file or physical_path(cfg)
if not physical_file.is_file():
raise SchemaContextError(f"physical schema is missing: {physical_file}")
try:
physical = PhysicalSchema.from_yaml(physical_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
except (OSError, ValidationError, ValueError) as exc:
raise SchemaContextError("physical schema or annotations are invalid") from exc
return SchemaContext(
physical=physical,
annotations=annotations,
effective_relationships=_load_effective_relationships(cfg, physical),
)
+30 -7
View File
@@ -7,16 +7,23 @@ MAX_EXAMPLES_IN_PROMPT = 5
def table_foreign_keys(
physical: PhysicalSchema, annotations: Annotations, table: str
physical: PhysicalSchema,
annotations: Annotations,
table: str,
effective_relationships: dict[str, list[ForeignKey]] | None = None,
) -> list[ForeignKey]:
"""FK fisiche + FK logiche dalle annotations (dedup su columns/ref)."""
if effective_relationships is not None:
return list(effective_relationships.get(table, []))
fks = list(physical.tables[table].foreign_keys)
ann = annotations.tables.get(table)
if ann:
seen = {(tuple(f.columns), f.ref_table, tuple(f.ref_columns)) for f in fks}
for fk in ann.foreign_keys:
if (tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)) not in seen:
key = (tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns))
if key not in seen:
fks.append(fk)
seen.add(key)
return fks
@@ -47,6 +54,8 @@ def to_mschema_text(
physical: PhysicalSchema,
annotations: Annotations | None = None,
tables: list[str] | None = None,
*,
effective_relationships: dict[str, list[ForeignKey]] | None = None,
) -> str:
"""Serializzazione testuale in stile ThothAI (【Schema】/【Foreign keys】)."""
annotations = annotations or Annotations()
@@ -73,7 +82,9 @@ def to_mschema_text(
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
lines.append(f" -- Examples: {shown}")
lines.append(");")
for fk in table_foreign_keys(physical, annotations, table_name):
for fk in table_foreign_keys(
physical, annotations, table_name, effective_relationships
):
for src, dst in zip(fk.columns, fk.ref_columns):
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
lines.extend(["", "【Foreign keys】", *fk_lines])
@@ -81,7 +92,10 @@ def to_mschema_text(
def to_schema_dict(
physical: PhysicalSchema, annotations: Annotations | None = None
physical: PhysicalSchema,
annotations: Annotations | None = None,
*,
effective_relationships: dict[str, list[ForeignKey]] | None = None,
) -> dict[str, Any]:
"""Vista compatibile con le logiche AV-SQL (schema_dict)."""
annotations = annotations or Annotations()
@@ -103,13 +117,20 @@ def to_schema_dict(
"primary_keys": [c for c in cols if table.columns[c].pk],
"foreign_keys": [
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
for fk in table_foreign_keys(physical, annotations, table_name)
for fk in table_foreign_keys(
physical, annotations, table_name, effective_relationships
)
],
}
return out
def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None) -> str:
def to_markdown(
physical: PhysicalSchema,
annotations: Annotations | None = None,
*,
effective_relationships: dict[str, list[ForeignKey]] | None = None,
) -> str:
"""Report leggibile per il reviewer."""
annotations = annotations or Annotations()
lines = [
@@ -146,7 +167,9 @@ def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
)
fks = table_foreign_keys(physical, annotations, table_name)
fks = table_foreign_keys(
physical, annotations, table_name, effective_relationships
)
if fks:
lines += ["", "Foreign keys:"]
for fk in fks: