feat(harness): derive a 3-5 keyword session name (YAKE, no LLM)

New sessions get a concise Italian-keyword `name` instead of the truncated
question. `tht session new` (when no --name is given) derives it via a new
`_extract_name` helper using YAKE (pure-Python, unsupervised, Italian, no LLM),
dropping generic query verbs and keeping the top keywords in reading order;
falls back to `_summarize` if YAKE is unavailable. `create_session` core keeps
its `name=None` default — the policy lives at the CLI layer.

TDD: tests/test_session_name.py (unit + CliRunner integration). Full harness
suite 269 passed; ruff clean.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-30 15:54:54 +02:00
co-authored by Claude Opus 4.8
parent d27afd0896
commit c12bdcd257
4 changed files with 129 additions and 2 deletions
+1
View File
@@ -15,6 +15,7 @@ dependencies = [
"rich>=13.0", "rich>=13.0",
"requests>=2.31", "requests>=2.31",
"tqdm>=4.66", "tqdm>=4.66",
"yake>=0.4",
] ]
[project.scripts] [project.scripts]
+81
View File
@@ -0,0 +1,81 @@
"""Tests for the keyword-based session name derivation (YAKE, no LLM).
A new session's display `name` is a 3-5 word Italian-keyword summary of the question,
derived at the CLI layer (`tht session new`) with no LLM. `create_session` core keeps its
`name=None` default; the policy lives in `new_cmd`.
"""
import json
from typer.testing import CliRunner
from tht.config import DatabaseConfig
from tht.cli.session_cmd import session_app
from tht.session.store import _extract_name, _summarize, load_session
def _db():
return DatabaseConfig(
database="testdb", user="u", password="p", # noqa: S106
**{"schema": "public"},
)
class _FakeCfg:
def __init__(self, sessions):
self.database = _db()
self.paths = type("P", (), {"sessions": sessions})()
def test_extract_name_is_a_3_to_5_word_italian_summary():
q = "Elenca i pazienti anziani con esiti gravi nell'ultimo periodo."
name = _extract_name(q)
words = name.split()
assert 3 <= len(words) <= 5
# the generic query verb is dropped...
assert "elenca" not in [w.lower() for w in words]
# ...and the salient content survives
lower = name.lower()
assert "pazienti" in lower and "anziani" in lower
# deterministic
assert _extract_name(q) == name
def test_extract_name_falls_back_to_summarize_on_failure(monkeypatch):
import tht.session.store as store
def boom(*a, **k):
raise RuntimeError("yake down")
monkeypatch.setattr(store.yake, "KeywordExtractor", boom)
q = "Elenca i pazienti anziani con esiti gravi."
assert _extract_name(q) == _summarize(q)
def test_extract_name_empty_question_is_empty():
assert _extract_name("") == ""
assert _extract_name(" ") == ""
def test_new_cmd_autonames_when_no_name(tmp_path, monkeypatch):
from tht.cli import session_cmd
monkeypatch.setattr(session_cmd, "_load_config_or_exit", lambda _: _FakeCfg(tmp_path))
q = "Mostra i pazienti con interventi recenti e complicazioni rilevanti."
res = CliRunner().invoke(session_app, ["new", q, "--json"])
assert res.exit_code == 0, res.output
sid = json.loads(res.output)["id"]
name = load_session(sid, tmp_path).name
assert name == _extract_name(q)
assert name and 3 <= len(name.split()) <= 5
def test_new_cmd_explicit_name_wins(tmp_path, monkeypatch):
from tht.cli import session_cmd
monkeypatch.setattr(session_cmd, "_load_config_or_exit", lambda _: _FakeCfg(tmp_path))
res = CliRunner().invoke(
session_app, ["new", "Una domanda qualunque", "--name", "Mio nome", "--json"]
)
assert res.exit_code == 0, res.output
sid = json.loads(res.output)["id"]
assert load_session(sid, tmp_path).name == "Mio nome"
+3 -2
View File
@@ -78,11 +78,12 @@ def new_cmd(
config: Path = CONFIG_OPT, config: Path = CONFIG_OPT,
) -> None: ) -> None:
"""Crea una sessione e stampa il suo id (ultima riga dell'output).""" """Crea una sessione e stampa il suo id (ultima riga dell'output)."""
from tht.session.store import create_session from tht.session.store import _extract_name, create_session
cfg = _load_config_or_exit(config) cfg = _load_config_or_exit(config)
manifest = create_session(question, cfg.database, cfg.paths.sessions, manifest = create_session(question, cfg.database, cfg.paths.sessions,
provider=provider, model=model, thinking=thinking, name=name) provider=provider, model=model, thinking=thinking,
name=name or _extract_name(question))
if json_out: if json_out:
typer.echo(json.dumps({"id": manifest.id}, ensure_ascii=False)) typer.echo(json.dumps({"id": manifest.id}, ensure_ascii=False))
return return
+44
View File
@@ -3,6 +3,8 @@ import shutil
from datetime import UTC, datetime from datetime import UTC, datetime
from pathlib import Path from pathlib import Path
import yake
from tht.config import DatabaseConfig from tht.config import DatabaseConfig
from tht.session.models import SessionManifest from tht.session.models import SessionManifest
from tht.textutil import slugify from tht.textutil import slugify
@@ -10,6 +12,17 @@ from tht.textutil import slugify
MANIFEST = "session_manifest.yaml" MANIFEST = "session_manifest.yaml"
MAX_SLUG_CHARS = 40 MAX_SLUG_CHARS = 40
MAX_SUMMARY_CHARS = 120 MAX_SUMMARY_CHARS = 120
MAX_NAME_CHARS = 60
MAX_NAME_WORDS = 5
# Generic query verbs/words dropped from a derived session name (lowercased compare);
# they carry no content and would just crowd out the salient keywords.
_STOPNAME = {
"crea", "creare", "elenca", "elencare", "elencami", "mostra", "mostrami", "mostrare",
"dammi", "dai", "trova", "trovami", "cerca", "cercami", "voglio", "vorrei", "fammi",
"quanti", "quante", "quanto", "quanta", "conta", "numero", "lista", "elenco", "vedi",
"visualizza", "ottieni", "restituisci", "calcola", "esponi", "riporta", "fornisci",
}
class SessionError(Exception): class SessionError(Exception):
@@ -29,6 +42,37 @@ def _summarize(question: str) -> str:
return first[:MAX_SUMMARY_CHARS] return first[:MAX_SUMMARY_CHARS]
def _extract_name(question: str) -> str:
"""Nome-sintesi (3-5 parole-chiave) della domanda per la sidebar, senza LLM (YAKE).
Le parole-chiave restano nella lingua della domanda (è contenuto, non chrome UI).
Ripiega su `_summarize` se YAKE non è disponibile o non estrae nulla di utile."""
q = (question or "").strip()
if not q:
return ""
try:
extractor = yake.KeywordExtractor(lan="it", n=1, top=8, dedupLim=0.9)
ranked = [k for k, _ in extractor.extract_keywords(q)]
except Exception:
return _summarize(question)
seen: set[str] = set()
picked: list[str] = []
for k in ranked:
kl = k.lower()
if kl in _STOPNAME or kl in seen:
continue
seen.add(kl)
picked.append(k)
if len(picked) >= MAX_NAME_WORDS:
break
if not picked:
return _summarize(question)
ql = q.lower()
picked.sort(key=lambda k: ql.find(k.lower())) # natural reading order
name = " ".join(picked)
return (name[0].upper() + name[1:])[:MAX_NAME_CHARS]
def render_question_md(question: str, assumptions: list[str] | None = None) -> str: def render_question_md(question: str, assumptions: list[str] | None = None) -> str:
"""Rende question.md in modo deterministico: domanda + assunzioni opzionali. """Rende question.md in modo deterministico: domanda + assunzioni opzionali.