Files
ThothII/harness/tests/test_evidence_pi_restructurer.py
T
Codex 84084bba37
Publish documentation / publish (push) Successful in 30s
Fix Qwen session tool calls and expose thinking compatibility
2026-09-21 16:23:51 +02:00

248 lines
8.7 KiB
Python

import json
from pathlib import Path
from types import SimpleNamespace
import pytest
from tht.evidence import (
EvidencePreparationError,
PiEvidenceRestructurer,
RestructureRequest,
)
def _request():
return RestructureRequest.model_validate({
"source_file": "source/domain/patient.md",
"source_sha256": "sha256:" + "a" * 64,
"normalized_text": "I pazienti sotto i 18 anni sono pediatrici.\n",
})
def _candidate():
return {
"schema_version": 1,
"title": "Fascia pediatrica",
"kind": "domain",
"purposes": ["disambiguation"],
"applies_to": {"concepts": ["fascia pediatrica"]},
"language": "it",
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
"review_items": [],
"payload": {"rule": "La fascia pediatrica comprende i minori."},
}
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
from tht.evidence import authoring
monkeypatch.setenv("THT_DEFAULT_SESSION_MODEL", "zai/glm-5.2")
monkeypatch.setenv("PI_THINKING", "medium")
calls = []
def run(argv, **kwargs):
calls.append((argv, kwargs))
kwargs["stdout"].write(json.dumps({"candidates": [_candidate()]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
candidates = restructurer.restructure(_request())
assert candidates[0].title == "Fascia pediatrica"
argv, kwargs = calls[0]
assert argv[:8] == [
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
]
assert "--no-skills" in argv
assert argv[argv.index("--extension") + 1].endswith(
"/evidence-extensions/tht-evidence-json-mode.ts"
)
assert "--system-prompt" in argv
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
assert "--skill" not in argv
assert "--append-system-prompt" not in argv
assert "/skill:tht-evidence-authoring" not in argv
assert argv[argv.index("--provider") + 1] == "zai"
assert argv[argv.index("--model") + 1] == "glm-5.2"
assert argv[argv.index("--thinking") + 1] == "medium"
assert any(
"must not use kind enum or formula" in argument
for argument in argv
)
assert any(argument.startswith("@") for argument in argv)
assert kwargs["timeout"] == 12
assert kwargs["shell"] is False
assert kwargs["stderr"] is authoring.subprocess.DEVNULL
@pytest.mark.parametrize("result", [
SimpleNamespace(returncode=1, stdout="secret", stderr="secret"),
SimpleNamespace(returncode=0, stdout="not-json", stderr=""),
])
def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path, monkeypatch, result):
from tht.evidence import authoring
def run(*args, **kwargs):
kwargs["stdout"].write(result.stdout)
return result
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
with pytest.raises(EvidencePreparationError) as error:
restructurer.restructure(_request())
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
assert "secret" not in str(error.value)
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
skill = (
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
).read_text(encoding="utf-8")
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
for required_field in (
"schema_version",
"existing_id",
"supporting_excerpts",
"review_items",
"payload",
):
assert required_field in skill
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
extension = (
Path(__file__).parents[1] / ".pi" / "evidence-extensions" / "tht-evidence-json-mode.ts"
).read_text(encoding="utf-8")
assert 'pi.on("before_provider_request"' in extension
assert 'response_format: { type: "json_object" }' in extension
assert "temperature: 0" in extension
assert "registerTool" not in extension
def test_evidence_json_mode_is_not_auto_loaded_into_interactive_sessions():
from tht.evidence.authoring import authoring_skill_path
skill_path = authoring_skill_path()
restructurer = PiEvidenceRestructurer("pi", skill_path)
extension_path = restructurer._json_mode_extension
auto_extensions = skill_path.parents[2] / "extensions"
assert extension_path.is_file()
assert not extension_path.is_relative_to(auto_extensions)
assert not (auto_extensions / extension_path.name).exists()
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
tmp_path, monkeypatch, response,
):
from tht.evidence import authoring
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps(response))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"- **Pazienti** sotto i 18 anni sono pediatrici.",
)
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"Pazienti sotto i 18 anni sono pediatrici.",
)
assert [item.code for item in candidates[0].review_items] == [
"supporting_excerpt_reconciled",
]
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
tmp_path, monkeypatch,
):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": [
(
"pivot/denormalization (solo per le FACT): legge le righe correlate "
"nella tabella source_table, estrae source_column usando le join_keys."
)
],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
exact = (
"- **pivot/denormalization** (solo per le FACT):\n"
" - legge le righe correlate nella tabella `source_table`,\n"
" - estrae `source_column` usando le `join_keys`."
)
request = _request().model_copy(update={"normalized_text": exact + "\n"})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (exact,)