chore: commit remaining worktree changes

This commit is contained in:
2026-08-26 08:10:37 +02:00
parent ec061c42d4
commit f48196a57f
234 changed files with 146 additions and 61044 deletions
-1
View File
@@ -1,6 +1,5 @@
"""Classificazione di column eligibility (principio trasversale Thoth).
Vedi docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md.
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
provengono solo da numerici, enum, temporali, booleani e testo breve.
"""
-102
View File
@@ -1,102 +0,0 @@
"""Per-step task document generator (spec D16, §4.9).
Emette un singolo documento compatto per fase/step, derivato dagli artefatti precedenti
e dalla vista effective delle decisioni, con byte budget enforced (target <20k token
per un modello 35B/<200k). MAI incorpora physical.yaml (~190k token, fatale).
D15+D16 complementari: il task doc e' generato dalla vista effective_decisions, quindi
post-rollback riflette automaticamente lo stato corretto (le decisioni stale di fasi
> current_phase sono escluse).
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from tht.phase import effective_decisions
from tht.session.models import SessionSnapshot
from tht.workflow import load_workflow
MAX_BODY_BYTES = 80_000 # ~20k token (target per task document di una fase)
_TRUNCATION_MARKER = "\n\n[...troncato per il budget di contesto D16...]"
@dataclass
class TaskDoc:
phase: int
body: str
byte_budget_ok: bool
def _slice_schema_linking(raw: str, promoted_tables: list[str] | None) -> str:
"""Riduce schema_linking.json alle sole tabelle promosse (D16 §4.9: 'solo le
tabelle/colonne promosse, non tutto lo schema'). Senza promoted_tables passa il
contenuto invariato (e' gia' la superficie decisionale, non lo schema fisico).
Su JSON malformato ritorna il raw (il bound piu' sotto lo tronca se enorme)."""
if not promoted_tables:
return raw
try:
data = json.loads(raw)
except json.JSONDecodeError:
return raw
allow = set(promoted_tables)
def table_of(cand: dict) -> str:
return str(cand.get("name", "")).split(".")[0]
if isinstance(data, dict) and isinstance(data.get("candidates"), list):
data["candidates"] = [c for c in data["candidates"] if table_of(c) in allow]
return json.dumps(data, ensure_ascii=False, indent=2)
def generate_task_doc(
session_dir: Path | str | SessionSnapshot,
phase: int,
promoted_tables: list[str] | None = None,
) -> TaskDoc:
"""Genera il documento di task per la fase `phase`.
Contenuto (compatti, mai artefatti integrali fatali):
- Domanda (question.md) se presente.
- Schema linking (schema_linking.json) solo da fase >= 4.
- Brief delle decisioni effective (esclude stale post-rollback, esclude ritirate).
- Header del task con il numero/nome della fase.
"""
snapshot = session_dir if isinstance(session_dir, SessionSnapshot) else None
session_dir = None if snapshot is not None else Path(session_dir)
parts: list[str] = []
question = snapshot.artifacts.get("question") if snapshot else (session_dir / "question.md").read_text() if (session_dir / "question.md").exists() else None
if question:
parts.append("## Domanda\n" + question)
linking = snapshot.artifacts.get("schema_linking") if snapshot else (session_dir / "schema_linking.json").read_text() if (session_dir / "schema_linking.json").exists() else None
if linking and phase >= 4:
sliced = _slice_schema_linking(linking, promoted_tables)
parts.append("## Schema linking (deciso)\n```json\n" + sliced + "\n```")
# Brief decisioni effective (D15-aware)
eff = effective_decisions(snapshot or session_dir)
if eff:
lines = [f"- {d.type} | {d.subject} | {d.detail}" for d in eff]
parts.append("## Decisioni effettive (effective)\n" + "\n".join(lines))
# Header fase
try:
wf = load_workflow()
name = wf.phase_name(phase)
header = f"## Task: fase {phase} ({name})"
except Exception: # noqa: BLE001 - task documents retain a phase-only fallback
header = f"## Task: fase {phase}"
parts.append(header)
body = "\n\n".join(parts)
# Enforcement del bound (D16): non solo segnalare -- troncare. Un task doc oltre
# budget collasserebbe un 35B; meglio un documento troncato e marcato.
encoded = body.encode()
budget = MAX_BODY_BYTES - len(_TRUNCATION_MARKER.encode())
if len(encoded) > MAX_BODY_BYTES:
body = encoded[:budget].decode("utf-8", "ignore") + _TRUNCATION_MARKER
return TaskDoc(phase=phase, body=body, byte_budget_ok=False)
return TaskDoc(phase=phase, body=body, byte_budget_ok=True)
-8
View File
@@ -1,8 +0,0 @@
import re
import unicodedata
def slugify(text: str) -> str:
"""Slug ASCII minuscolo; preserva gli underscore (nomi tabella)."""
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-")
+2 -2
View File
@@ -19,8 +19,8 @@
dalla sola parola inglese `name`.
**Nota: `skip_column` e `NAME_LIKE_TOKENS` non sono più usati dal flusso Thoth.**
La selezione delle colonne da indicizzare è ora governata dal *principio di eleggibilità
delle colonne* (spec: `docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md`).
La selezione delle colonne da indicizzare è ora governata dal principio di eleggibilità
implementato in `tht/mschema/eligibility.py`.
L'esclusione dei testi larghi avviene a monte tramite il flag `eligible` persistito in
`physical.yaml` (impostato da `tht schema introspect`), non tramite l'euristica sulla
lunghezza del vendorizzato. Le funzioni restano nel file per fedeltà alla sorgente upstream;