Lever 1: Join-graph via FK logics in annotations + suggest-fks command
- TableAnnotation.foreign_keys field stores curated logical FKs (DWH has no FK constraints)
- tht schema suggest-fks: mine from approved SQL, heuristics (time_key → dim_time),
same-name discovery + explicit --assume flag for multi-owner PKs
- mschema renders 【Foreign keys】 section populated; validation in merge.py
- SKILL.md F4 now reads FKs from mschema-text, no custom data_time_key logic
Lever 2: Context-pack consolidation at kickoff (tht search pack)
- Single embedding of question, reused for schema + evidence + solved searches
- One command: tht search pack <question> --session <id> → retrieval_pack.md
- Graceful degradation when Ollama/vector store unreachable (exit 0, empty sections)
- SKILL.md F1 prescribes as first call; reduces model thinking turns via pre-retrieval
Lever 3: Phase-summary recap v2 auto-construction from session ledger
- tht session show --json includes full decisions ledger
- tht phase meta --json exports 'emits' (substantive decision types per phase)
- Gate appends deterministic 【Decisioni registrate in questa fase】 section (appendLedgerSection)
- Model authors only summary + checks; recap table comes from persisted state (exact by construction)
- SKILL.md Disciplina 6: brief model output, gate fills the rest
Tests: 358 Python (including 10 FK + 3 pack + 1 session-ledger tests) + 111 JS gate tests, all pass.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
157 lines
6.3 KiB
Python
157 lines
6.3 KiB
Python
from typing import Any
|
|
|
|
from tht.mschema.eligibility import effective_eligibility
|
|
from tht.mschema.models import Annotations, ColumnAnnotation, ForeignKey, PhysicalSchema
|
|
|
|
MAX_EXAMPLES_IN_PROMPT = 5
|
|
|
|
|
|
def table_foreign_keys(
|
|
physical: PhysicalSchema, annotations: Annotations, table: str
|
|
) -> list[ForeignKey]:
|
|
"""FK fisiche + FK logiche dalle annotations (dedup su columns/ref)."""
|
|
fks = list(physical.tables[table].foreign_keys)
|
|
ann = annotations.tables.get(table)
|
|
if ann:
|
|
seen = {(tuple(f.columns), f.ref_table, tuple(f.ref_columns)) for f in fks}
|
|
for fk in ann.foreign_keys:
|
|
if (tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)) not in seen:
|
|
fks.append(fk)
|
|
return fks
|
|
|
|
|
|
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
|
|
ann = annotations.tables.get(table)
|
|
if ann is None:
|
|
return None
|
|
return ann.columns.get(column)
|
|
|
|
|
|
def _table_description(physical: PhysicalSchema, annotations: Annotations, table: str) -> str:
|
|
ann = annotations.tables.get(table)
|
|
if ann and ann.description:
|
|
return ann.description
|
|
return physical.tables[table].comment
|
|
|
|
|
|
def _column_description(
|
|
physical: PhysicalSchema, annotations: Annotations, table: str, column: str
|
|
) -> str:
|
|
ann = annotations.tables.get(table)
|
|
if ann and column in ann.columns and ann.columns[column].description:
|
|
return ann.columns[column].description
|
|
return physical.tables[table].columns[column].comment
|
|
|
|
|
|
def to_mschema_text(
|
|
physical: PhysicalSchema,
|
|
annotations: Annotations | None = None,
|
|
tables: list[str] | None = None,
|
|
) -> str:
|
|
"""Serializzazione testuale in stile ThothAI (【Schema】/【Foreign keys】)."""
|
|
annotations = annotations or Annotations()
|
|
selected = [t for t in physical.tables if tables is None or t in tables]
|
|
lines: list[str] = ["【Schema】"]
|
|
fk_lines: list[str] = []
|
|
for table_name in selected:
|
|
table = physical.tables[table_name]
|
|
desc = _table_description(physical, annotations, table_name)
|
|
if desc:
|
|
lines.append(f"-- {desc}")
|
|
lines.append(f"CREATE TABLE {table_name} (")
|
|
for column_name, column in table.columns.items():
|
|
if not effective_eligibility(column, _ann_col(annotations, table_name, column_name))[0]:
|
|
continue
|
|
line = f" {column_name} {column.type.upper()}"
|
|
if column.pk:
|
|
line += " -- PRIMARY KEY"
|
|
lines.append(line)
|
|
cdesc = _column_description(physical, annotations, table_name, column_name)
|
|
if cdesc:
|
|
lines.append(f" -- {cdesc}")
|
|
if column.examples:
|
|
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
|
|
lines.append(f" -- Examples: {shown}")
|
|
lines.append(");")
|
|
for fk in table_foreign_keys(physical, annotations, table_name):
|
|
for src, dst in zip(fk.columns, fk.ref_columns):
|
|
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
|
|
lines.extend(["", "【Foreign keys】", *fk_lines])
|
|
return "\n".join(lines)
|
|
|
|
|
|
def to_schema_dict(
|
|
physical: PhysicalSchema, annotations: Annotations | None = None
|
|
) -> dict[str, Any]:
|
|
"""Vista compatibile con le logiche AV-SQL (schema_dict)."""
|
|
annotations = annotations or Annotations()
|
|
out: dict[str, Any] = {}
|
|
for table_name, table in physical.tables.items():
|
|
cols = [
|
|
c
|
|
for c in table.columns
|
|
if effective_eligibility(table.columns[c], _ann_col(annotations, table_name, c))[0]
|
|
]
|
|
out[table_name] = {
|
|
"columns_name": cols,
|
|
"columns_type": [table.columns[c].type for c in cols],
|
|
"columns_description": [
|
|
_column_description(physical, annotations, table_name, c) for c in cols
|
|
],
|
|
"example_values": [table.columns[c].examples for c in cols],
|
|
"table_to_tablefullname": f"{physical.db_schema}.{table_name}",
|
|
"primary_keys": [c for c in cols if table.columns[c].pk],
|
|
"foreign_keys": [
|
|
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
|
|
for fk in table_foreign_keys(physical, annotations, table_name)
|
|
],
|
|
}
|
|
return out
|
|
|
|
|
|
def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None) -> str:
|
|
"""Report leggibile per il reviewer."""
|
|
annotations = annotations or Annotations()
|
|
lines = [
|
|
f"# Schema {physical.db_schema} ({physical.database})",
|
|
"",
|
|
f"Introspezione: {physical.introspected_at.isoformat()} — "
|
|
f"{len(physical.tables)} tabelle",
|
|
]
|
|
for table_name, table in physical.tables.items():
|
|
lines += ["", f"## {table_name}", ""]
|
|
desc = _table_description(physical, annotations, table_name)
|
|
if desc:
|
|
lines += [desc, ""]
|
|
lines += [
|
|
f"Righe (stima): {table.row_count}",
|
|
"",
|
|
"| Colonna | Tipo | Null | PK | Descrizione | Esempi |",
|
|
"|---|---|---|---|---|---|",
|
|
]
|
|
for column_name, column in table.columns.items():
|
|
eligible, reason = effective_eligibility(
|
|
column, _ann_col(annotations, table_name, column_name)
|
|
)
|
|
cdesc = _column_description(physical, annotations, table_name, column_name)
|
|
if not eligible:
|
|
lines.append(
|
|
f"| ~~{column_name}~~ | {column.type} | "
|
|
f"{'sì' if column.nullable else 'no'} | "
|
|
f"{'sì' if column.pk else ''} | {cdesc} | _ignorata: {reason}_ |"
|
|
)
|
|
continue
|
|
examples = ", ".join(column.examples[:3])
|
|
lines.append(
|
|
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
|
|
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
|
|
)
|
|
fks = table_foreign_keys(physical, annotations, table_name)
|
|
if fks:
|
|
lines += ["", "Foreign keys:"]
|
|
for fk in fks:
|
|
lines.append(
|
|
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
|
|
)
|
|
return "\n".join(lines)
|