feat(harness): derive a 3-5 keyword session name (YAKE, no LLM)
New sessions get a concise Italian-keyword `name` instead of the truncated question. `tht session new` (when no --name is given) derives it via a new `_extract_name` helper using YAKE (pure-Python, unsupervised, Italian, no LLM), dropping generic query verbs and keeping the top keywords in reading order; falls back to `_summarize` if YAKE is unavailable. `create_session` core keeps its `name=None` default — the policy lives at the CLI layer. TDD: tests/test_session_name.py (unit + CliRunner integration). Full harness suite 269 passed; ruff clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -15,6 +15,7 @@ dependencies = [
|
||||
"rich>=13.0",
|
||||
"requests>=2.31",
|
||||
"tqdm>=4.66",
|
||||
"yake>=0.4",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Tests for the keyword-based session name derivation (YAKE, no LLM).
|
||||
|
||||
A new session's display `name` is a 3-5 word Italian-keyword summary of the question,
|
||||
derived at the CLI layer (`tht session new`) with no LLM. `create_session` core keeps its
|
||||
`name=None` default; the policy lives in `new_cmd`.
|
||||
"""
|
||||
import json
|
||||
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.config import DatabaseConfig
|
||||
from tht.cli.session_cmd import session_app
|
||||
from tht.session.store import _extract_name, _summarize, load_session
|
||||
|
||||
|
||||
def _db():
|
||||
return DatabaseConfig(
|
||||
database="testdb", user="u", password="p", # noqa: S106
|
||||
**{"schema": "public"},
|
||||
)
|
||||
|
||||
|
||||
class _FakeCfg:
|
||||
def __init__(self, sessions):
|
||||
self.database = _db()
|
||||
self.paths = type("P", (), {"sessions": sessions})()
|
||||
|
||||
|
||||
def test_extract_name_is_a_3_to_5_word_italian_summary():
|
||||
q = "Elenca i pazienti anziani con esiti gravi nell'ultimo periodo."
|
||||
name = _extract_name(q)
|
||||
words = name.split()
|
||||
assert 3 <= len(words) <= 5
|
||||
# the generic query verb is dropped...
|
||||
assert "elenca" not in [w.lower() for w in words]
|
||||
# ...and the salient content survives
|
||||
lower = name.lower()
|
||||
assert "pazienti" in lower and "anziani" in lower
|
||||
# deterministic
|
||||
assert _extract_name(q) == name
|
||||
|
||||
|
||||
def test_extract_name_falls_back_to_summarize_on_failure(monkeypatch):
|
||||
import tht.session.store as store
|
||||
|
||||
def boom(*a, **k):
|
||||
raise RuntimeError("yake down")
|
||||
|
||||
monkeypatch.setattr(store.yake, "KeywordExtractor", boom)
|
||||
q = "Elenca i pazienti anziani con esiti gravi."
|
||||
assert _extract_name(q) == _summarize(q)
|
||||
|
||||
|
||||
def test_extract_name_empty_question_is_empty():
|
||||
assert _extract_name("") == ""
|
||||
assert _extract_name(" ") == ""
|
||||
|
||||
|
||||
def test_new_cmd_autonames_when_no_name(tmp_path, monkeypatch):
|
||||
from tht.cli import session_cmd
|
||||
|
||||
monkeypatch.setattr(session_cmd, "_load_config_or_exit", lambda _: _FakeCfg(tmp_path))
|
||||
q = "Mostra i pazienti con interventi recenti e complicazioni rilevanti."
|
||||
res = CliRunner().invoke(session_app, ["new", q, "--json"])
|
||||
assert res.exit_code == 0, res.output
|
||||
sid = json.loads(res.output)["id"]
|
||||
name = load_session(sid, tmp_path).name
|
||||
assert name == _extract_name(q)
|
||||
assert name and 3 <= len(name.split()) <= 5
|
||||
|
||||
|
||||
def test_new_cmd_explicit_name_wins(tmp_path, monkeypatch):
|
||||
from tht.cli import session_cmd
|
||||
|
||||
monkeypatch.setattr(session_cmd, "_load_config_or_exit", lambda _: _FakeCfg(tmp_path))
|
||||
res = CliRunner().invoke(
|
||||
session_app, ["new", "Una domanda qualunque", "--name", "Mio nome", "--json"]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
sid = json.loads(res.output)["id"]
|
||||
assert load_session(sid, tmp_path).name == "Mio nome"
|
||||
@@ -78,11 +78,12 @@ def new_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Crea una sessione e stampa il suo id (ultima riga dell'output)."""
|
||||
from tht.session.store import create_session
|
||||
from tht.session.store import _extract_name, create_session
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
manifest = create_session(question, cfg.database, cfg.paths.sessions,
|
||||
provider=provider, model=model, thinking=thinking, name=name)
|
||||
provider=provider, model=model, thinking=thinking,
|
||||
name=name or _extract_name(question))
|
||||
if json_out:
|
||||
typer.echo(json.dumps({"id": manifest.id}, ensure_ascii=False))
|
||||
return
|
||||
|
||||
@@ -3,6 +3,8 @@ import shutil
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
|
||||
import yake
|
||||
|
||||
from tht.config import DatabaseConfig
|
||||
from tht.session.models import SessionManifest
|
||||
from tht.textutil import slugify
|
||||
@@ -10,6 +12,17 @@ from tht.textutil import slugify
|
||||
MANIFEST = "session_manifest.yaml"
|
||||
MAX_SLUG_CHARS = 40
|
||||
MAX_SUMMARY_CHARS = 120
|
||||
MAX_NAME_CHARS = 60
|
||||
MAX_NAME_WORDS = 5
|
||||
|
||||
# Generic query verbs/words dropped from a derived session name (lowercased compare);
|
||||
# they carry no content and would just crowd out the salient keywords.
|
||||
_STOPNAME = {
|
||||
"crea", "creare", "elenca", "elencare", "elencami", "mostra", "mostrami", "mostrare",
|
||||
"dammi", "dai", "trova", "trovami", "cerca", "cercami", "voglio", "vorrei", "fammi",
|
||||
"quanti", "quante", "quanto", "quanta", "conta", "numero", "lista", "elenco", "vedi",
|
||||
"visualizza", "ottieni", "restituisci", "calcola", "esponi", "riporta", "fornisci",
|
||||
}
|
||||
|
||||
|
||||
class SessionError(Exception):
|
||||
@@ -29,6 +42,37 @@ def _summarize(question: str) -> str:
|
||||
return first[:MAX_SUMMARY_CHARS]
|
||||
|
||||
|
||||
def _extract_name(question: str) -> str:
|
||||
"""Nome-sintesi (3-5 parole-chiave) della domanda per la sidebar, senza LLM (YAKE).
|
||||
|
||||
Le parole-chiave restano nella lingua della domanda (è contenuto, non chrome UI).
|
||||
Ripiega su `_summarize` se YAKE non è disponibile o non estrae nulla di utile."""
|
||||
q = (question or "").strip()
|
||||
if not q:
|
||||
return ""
|
||||
try:
|
||||
extractor = yake.KeywordExtractor(lan="it", n=1, top=8, dedupLim=0.9)
|
||||
ranked = [k for k, _ in extractor.extract_keywords(q)]
|
||||
except Exception:
|
||||
return _summarize(question)
|
||||
seen: set[str] = set()
|
||||
picked: list[str] = []
|
||||
for k in ranked:
|
||||
kl = k.lower()
|
||||
if kl in _STOPNAME or kl in seen:
|
||||
continue
|
||||
seen.add(kl)
|
||||
picked.append(k)
|
||||
if len(picked) >= MAX_NAME_WORDS:
|
||||
break
|
||||
if not picked:
|
||||
return _summarize(question)
|
||||
ql = q.lower()
|
||||
picked.sort(key=lambda k: ql.find(k.lower())) # natural reading order
|
||||
name = " ".join(picked)
|
||||
return (name[0].upper() + name[1:])[:MAX_NAME_CHARS]
|
||||
|
||||
|
||||
def render_question_md(question: str, assumptions: list[str] | None = None) -> str:
|
||||
"""Rende question.md in modo deterministico: domanda + assunzioni opzionali.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user