chore: commit remaining worktree changes

This commit is contained in:
2026-08-26 08:10:37 +02:00
parent ec061c42d4
commit f48196a57f
234 changed files with 146 additions and 61044 deletions
+3 -2
View File
@@ -140,5 +140,6 @@ docs/ testing guide + workflow editing
## Reference
- Architecture spec: `docs/superpowers/specs/2026-06-25-thothii-architecture-design.md`
- Implementation plan: `docs/superpowers/plans/2026-06-25-harness-implementation.md`
- Repository architecture: `../docs/architecture/overview.md`
- Components and flows: `../docs/architecture/components.md`
- Runtime contracts: `../docs/contracts/`
+1 -1
View File
@@ -42,7 +42,7 @@ no network.
**Coverage (honest):**
- Python logic-pure: `workflow.yaml` loading, `effective_decisions`,
`teardown_to_phase`, `generate_task_doc`, `aggregate_lsh_multi` (on fake hits),
`teardown_to_phase`, `aggregate_lsh_multi` (on fake hits),
`formula_store` read/write, `decision_retracted`, `save_one_memory`, the
rationale-capture contract, the session-coherence smoke, CLI `phase meta --json`.
- **Gate builder functions** (pure, in JS, tested in JS): the widget-descriptor
@@ -21,39 +21,3 @@ def test_acceptance_runner_includes_the_integrated_candidate_publication_gate():
runner = Path(__file__).parents[2] / "scripts" / "evidence-restructuring-acceptance.sh"
assert "test_evidence_candidate_publication.py" in runner.read_text(encoding="utf-8")
def test_owner_gate_manual_records_the_narrow_inventory_and_qdrant_baseline_protocol():
manual = Path(__file__).parents[2] / "docs" / "testing" / "evidence-restructuring-manual.md"
text = manual.read_text(encoding="utf-8")
normalized = " ".join(text.split())
assert "35 moveable source documents" in text
assert "retained `psd-clinical/evidence/README.md`" in text
inventory = text.split("README is not a source document and must remain", 1)[1]
inventory = inventory.split("```text", 1)[1].split("```", 1)[0]
assert sum(line.strip().endswith(".md") for line in inventory.splitlines()) == 35
assert "move exactly those 35 listed source documents" in normalized
assert "with_payload:false" in text
assert "with_payload:true" not in text
assert "must be repeated through the successful `vector inspect` command" in normalized
def test_owner_gate_current_summaries_supersede_stale_pre_inspection_history():
root = Path(__file__).parents[2]
report = (
root / ".superpowers" / "sdd" / "2026-08-24-evidence-restructuring"
/ "task-12-report.md"
).read_text(encoding="utf-8")
state = (root / "PROJECT_STATE.md").read_text(encoding="utf-8")
current_report = report.split("## Historical record", 1)[0]
current_state = state.split("## Modular workflow refactor candidate", 1)[0]
assert "225 passed, 1 warning" in current_report
assert "230 passed, 1 warning" in current_report
assert "authorized read-only PSD" in current_report
assert "35 moveable source documents" in current_report
assert "**225 passed, 1 known pytest deprecation warning**" in current_state
assert "**230 passed, 1 known pytest deprecation warning**" in current_state
+1 -12
View File
@@ -4,13 +4,12 @@ The CI-runnable proxy for 'full session coherence': build a synthetic ledger by
hand (decisions appended directly, not via an LLM) representing a full F1->F8 walk,
then a rollback F6->F4 + re-derive, and assert the invariants the D1-the-L2-version
would check. This is the coherence net for the Phase-A substrate (phase fold,
effective view, teardown, taskdoc byte budget).
effective view, teardown).
"""
from pathlib import Path
from tht.decisions import append_decision
from tht.phase import current_phase, effective_decisions
from tht.taskdoc import generate_task_doc
from tht.teardown import teardown_to_phase
from tht.workflow import load_workflow
@@ -77,16 +76,6 @@ def test_re_approve_after_rollback_advances_correctly(tmp_path):
assert current_phase(s) == 4
def test_task_doc_stays_under_byte_budget_across_phases(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("# Domanda\ndammi i pazienti con ablazione nel 2025")
(s / "schema_linking.json").write_text('{"candidates":[{"kind":"table","name":"fct_ricoveri"}]}')
for ph in range(1, 9):
doc = generate_task_doc(session_dir=s, phase=ph, promoted_tables=["fct_ricoveri"])
assert doc.byte_budget_ok, f"phase {ph} task doc over byte budget"
def test_decision_retraction_excluded_from_effective(tmp_path):
s = tmp_path / "sess"
s.mkdir()
-104
View File
@@ -1,104 +0,0 @@
from tht.taskdoc import generate_task_doc
def test_task_doc_includes_question_and_schema_scope(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("# Domanda\nQuanti pazienti?\n## Assunzioni\n- a")
(s / "schema_linking.json").write_text(
'{"question":"q","candidates":[{"kind":"table","name":"pazienti"}],"joins":[],"excluded":[],"open_questions":[]}'
)
doc = generate_task_doc(session_dir=s, phase=7)
assert "Quanti pazienti?" in doc.body
assert "pazienti" in doc.body
assert doc.byte_budget_ok is True
def test_task_doc_never_embeds_full_physical_yaml(tmp_path):
"""physical.yaml e' fatale per un 35B/<200k (~190k token). Mai incorporarlo."""
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
# un physical.yaml enorme fuori dalla sessione (come in ChironeWp3: artifacts/mschema/)
(s.parent / "physical.yaml").write_text("x: " + "y" * 800_000)
doc = generate_task_doc(session_dir=s, phase=7)
assert "physical.yaml" not in doc.body
assert len(doc.body) < 100_000 # bounded
def test_task_doc_byte_budget_enforced_on_normal_input(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
doc = generate_task_doc(session_dir=s, phase=1)
assert doc.byte_budget_ok is True
def test_task_doc_byte_budget_violation_flagged(tmp_path):
"""Se un artefatto di sessione e' enorme (input perverso), byte_budget_ok diventa False."""
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
(s / "schema_linking.json").write_text("x: " + "y" * 400_000) # ~400KB -> over budget
doc = generate_task_doc(session_dir=s, phase=7)
assert doc.byte_budget_ok is False
def test_taskdoc_truncates_over_budget(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("# Domanda\n" + "x" * 200_000)
doc = generate_task_doc(session_dir=s, phase=1)
assert doc.byte_budget_ok is False
assert len(doc.body.encode()) <= 80_000
assert "troncato" in doc.body
def test_task_doc_carries_phase_header(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
doc = generate_task_doc(session_dir=s, phase=4)
assert "fase 4" in doc.body.lower() or "fase 4" in doc.body
def test_task_doc_excludes_stale_decisions_post_rollback(tmp_path):
"""D15+D16: il task doc riflette lo stato effective, non quello stale.
Dopo rollback a F4, una sql_approved:7 stale non appare nel brief delle decisioni."""
from tht.decisions import append_decision
from tht.phase import current_phase
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
# simula: lavoro fino a F7, poi rollback a F4
append_decision(s, type="phase_approved", subject="phase:1")
append_decision(s, type="phase_approved", subject="phase:2")
append_decision(s, type="phase_approved", subject="phase:3")
append_decision(s, type="phase_approved", subject="phase:4")
append_decision(s, type="sql_approved", subject="phase:7", detail="SELECT 1")
append_decision(s, type="phase_reopened", subject="phase:4")
assert current_phase(s) == 4
doc = generate_task_doc(session_dir=s, phase=4)
# la decisione stale di fase 7 NON deve apparire nel brief
assert "sql_approved" not in doc.body
assert "phase:7" not in doc.body
def test_taskdoc_slices_to_promoted_tables(tmp_path):
s = tmp_path / "sess"
s.mkdir()
(s / "question.md").write_text("q")
(s / "schema_linking.json").write_text(
'{"question":"q","candidates":['
'{"kind":"table","name":"pazienti","decision":"promoted"},'
'{"kind":"table","name":"ricoveri","decision":"promoted"}],'
'"joins":[],"excluded":[],"open_questions":[]}'
)
doc = generate_task_doc(session_dir=s, phase=4, promoted_tables=["pazienti"])
assert "pazienti" in doc.body
assert "ricoveri" not in doc.body
-1
View File
@@ -1,6 +1,5 @@
"""Classificazione di column eligibility (principio trasversale Thoth).
Vedi docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md.
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
provengono solo da numerici, enum, temporali, booleani e testo breve.
"""
-102
View File
@@ -1,102 +0,0 @@
"""Per-step task document generator (spec D16, §4.9).
Emette un singolo documento compatto per fase/step, derivato dagli artefatti precedenti
e dalla vista effective delle decisioni, con byte budget enforced (target <20k token
per un modello 35B/<200k). MAI incorpora physical.yaml (~190k token, fatale).
D15+D16 complementari: il task doc e' generato dalla vista effective_decisions, quindi
post-rollback riflette automaticamente lo stato corretto (le decisioni stale di fasi
> current_phase sono escluse).
"""
from __future__ import annotations
import json
from dataclasses import dataclass
from pathlib import Path
from tht.phase import effective_decisions
from tht.session.models import SessionSnapshot
from tht.workflow import load_workflow
MAX_BODY_BYTES = 80_000 # ~20k token (target per task document di una fase)
_TRUNCATION_MARKER = "\n\n[...troncato per il budget di contesto D16...]"
@dataclass
class TaskDoc:
phase: int
body: str
byte_budget_ok: bool
def _slice_schema_linking(raw: str, promoted_tables: list[str] | None) -> str:
"""Riduce schema_linking.json alle sole tabelle promosse (D16 §4.9: 'solo le
tabelle/colonne promosse, non tutto lo schema'). Senza promoted_tables passa il
contenuto invariato (e' gia' la superficie decisionale, non lo schema fisico).
Su JSON malformato ritorna il raw (il bound piu' sotto lo tronca se enorme)."""
if not promoted_tables:
return raw
try:
data = json.loads(raw)
except json.JSONDecodeError:
return raw
allow = set(promoted_tables)
def table_of(cand: dict) -> str:
return str(cand.get("name", "")).split(".")[0]
if isinstance(data, dict) and isinstance(data.get("candidates"), list):
data["candidates"] = [c for c in data["candidates"] if table_of(c) in allow]
return json.dumps(data, ensure_ascii=False, indent=2)
def generate_task_doc(
session_dir: Path | str | SessionSnapshot,
phase: int,
promoted_tables: list[str] | None = None,
) -> TaskDoc:
"""Genera il documento di task per la fase `phase`.
Contenuto (compatti, mai artefatti integrali fatali):
- Domanda (question.md) se presente.
- Schema linking (schema_linking.json) solo da fase >= 4.
- Brief delle decisioni effective (esclude stale post-rollback, esclude ritirate).
- Header del task con il numero/nome della fase.
"""
snapshot = session_dir if isinstance(session_dir, SessionSnapshot) else None
session_dir = None if snapshot is not None else Path(session_dir)
parts: list[str] = []
question = snapshot.artifacts.get("question") if snapshot else (session_dir / "question.md").read_text() if (session_dir / "question.md").exists() else None
if question:
parts.append("## Domanda\n" + question)
linking = snapshot.artifacts.get("schema_linking") if snapshot else (session_dir / "schema_linking.json").read_text() if (session_dir / "schema_linking.json").exists() else None
if linking and phase >= 4:
sliced = _slice_schema_linking(linking, promoted_tables)
parts.append("## Schema linking (deciso)\n```json\n" + sliced + "\n```")
# Brief decisioni effective (D15-aware)
eff = effective_decisions(snapshot or session_dir)
if eff:
lines = [f"- {d.type} | {d.subject} | {d.detail}" for d in eff]
parts.append("## Decisioni effettive (effective)\n" + "\n".join(lines))
# Header fase
try:
wf = load_workflow()
name = wf.phase_name(phase)
header = f"## Task: fase {phase} ({name})"
except Exception: # noqa: BLE001 - task documents retain a phase-only fallback
header = f"## Task: fase {phase}"
parts.append(header)
body = "\n\n".join(parts)
# Enforcement del bound (D16): non solo segnalare -- troncare. Un task doc oltre
# budget collasserebbe un 35B; meglio un documento troncato e marcato.
encoded = body.encode()
budget = MAX_BODY_BYTES - len(_TRUNCATION_MARKER.encode())
if len(encoded) > MAX_BODY_BYTES:
body = encoded[:budget].decode("utf-8", "ignore") + _TRUNCATION_MARKER
return TaskDoc(phase=phase, body=body, byte_budget_ok=False)
return TaskDoc(phase=phase, body=body, byte_budget_ok=True)
-8
View File
@@ -1,8 +0,0 @@
import re
import unicodedata
def slugify(text: str) -> str:
"""Slug ASCII minuscolo; preserva gli underscore (nomi tabella)."""
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-")
+2 -2
View File
@@ -19,8 +19,8 @@
dalla sola parola inglese `name`.
**Nota: `skip_column` e `NAME_LIKE_TOKENS` non sono più usati dal flusso Thoth.**
La selezione delle colonne da indicizzare è ora governata dal *principio di eleggibilità
delle colonne* (spec: `docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md`).
La selezione delle colonne da indicizzare è ora governata dal principio di eleggibilità
implementato in `tht/mschema/eligibility.py`.
L'esclusione dei testi larghi avviene a monte tramite il flag `eligible` persistito in
`physical.yaml` (impostato da `tht schema introspect`), non tramite l'euristica sulla
lunghezza del vendorizzato. Le funzioni restano nel file per fedeltà alla sorgente upstream;