fix(evidence): stabilize real Pi authoring

This commit is contained in:
2026-08-25 15:21:59 +02:00
parent 610ae8c85a
commit 8ba87b68dc
7 changed files with 486 additions and 23 deletions
+35
View File
@@ -1,4 +1,5 @@
import hashlib
import threading
import unicodedata
import pytest
@@ -348,6 +349,40 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_root = tmp_path / "evidence" / "source" / "domain"
changed_text = source_text + "\nRegola revisionata.\n"
(source_root / "patient.md").write_text(changed_text, encoding="utf-8")
(source_root / "second.md").write_text(changed_text, encoding="utf-8")
barrier = threading.Barrier(2)
class ConcurrentRestructurer:
def __init__(self):
self.requests = []
def restructure(self, request):
self.requests.append(request)
barrier.wait(timeout=2)
return (_candidate(title=f"Unit {request.source_file}"),)
restructurer = ConcurrentRestructurer()
report = prepare_workspace_evidence(
tmp_path,
restructurer=restructurer,
git_status=lambda _: (),
max_workers=2,
)
assert report.model_calls == 2
assert sorted(request.source_file for request in restructurer.requests) == [
"source/domain/patient.md",
"source/domain/second.md",
]
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
+26
View File
@@ -14,6 +14,32 @@ def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
assert "validate" in result.output
def test_evidence_prepare_failure_identifies_the_source_file(monkeypatch, tmp_path):
from tht.cli import evidence_cmd
from tht.evidence import EvidencePreparationError
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
monkeypatch.setattr(
evidence_cmd,
"prepare_workspace_evidence",
lambda *args, **kwargs: (_ for _ in ()).throw(
EvidencePreparationError("pi_restructure_invalid", "source/domain/patient.md")
),
)
result = CliRunner().invoke(app, ["evidence", "prepare", str(tmp_path), "--json"])
assert result.exit_code == 1
assert result.stderr == ""
assert json.loads(result.stdout) == {
"code": "pi_restructure_invalid",
"operation": "evidence_prepare",
"schemaVersion": 1,
"sourceFile": "source/domain/patient.md",
"status": "failed",
}
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
from tht.cli import evidence_cmd
from tht.evidence.authoring import EvidenceResolutionReport
+158 -3
View File
@@ -1,4 +1,5 @@
import json
from pathlib import Path
from types import SimpleNamespace
import pytest
@@ -35,6 +36,9 @@ def _candidate():
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
from tht.evidence import authoring
monkeypatch.setenv("PI_PROVIDER", "zai")
monkeypatch.setenv("PI_MODEL", "glm-5.2")
monkeypatch.setenv("PI_THINKING", "medium")
calls = []
def run(argv, **kwargs):
@@ -43,7 +47,9 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
candidates = restructurer.restructure(_request())
@@ -52,7 +58,22 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
assert argv[:8] == [
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
]
assert "--skill" in argv
assert "--no-skills" in argv
assert argv[argv.index("--extension") + 1].endswith(
"/extensions/tht-evidence-json-mode.ts"
)
assert "--system-prompt" in argv
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
assert "--skill" not in argv
assert "--append-system-prompt" not in argv
assert "/skill:tht-evidence-authoring" not in argv
assert argv[argv.index("--provider") + 1] == "zai"
assert argv[argv.index("--model") + 1] == "glm-5.2"
assert argv[argv.index("--thinking") + 1] == "medium"
assert any(
"must not use kind enum or formula" in argument
for argument in argv
)
assert any(argument.startswith("@") for argument in argv)
assert kwargs["timeout"] == 12
assert kwargs["shell"] is False
@@ -71,10 +92,144 @@ def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path,
return result
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
with pytest.raises(EvidencePreparationError) as error:
restructurer.restructure(_request())
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
assert "secret" not in str(error.value)
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
skill = (
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
).read_text(encoding="utf-8")
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
for required_field in (
"schema_version",
"existing_id",
"supporting_excerpts",
"review_items",
"payload",
):
assert required_field in skill
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
extension = (
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
).read_text(encoding="utf-8")
assert 'pi.on("before_provider_request"' in extension
assert 'response_format: { type: "json_object" }' in extension
assert "temperature: 0" in extension
assert "registerTool" not in extension
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
tmp_path, monkeypatch, response,
):
from tht.evidence import authoring
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps(response))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"- **Pazienti** sotto i 18 anni sono pediatrici.",
)
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"Pazienti sotto i 18 anni sono pediatrici.",
)
assert [item.code for item in candidates[0].review_items] == [
"supporting_excerpt_reconciled",
]
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
tmp_path, monkeypatch,
):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": [
(
"pivot/denormalization (solo per le FACT): legge le righe correlate "
"nella tabella source_table, estrae source_column usando le join_keys."
)
],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
exact = (
"- **pivot/denormalization** (solo per le FACT):\n"
" - legge le righe correlate nella tabella `source_table`,\n"
" - estrae `source_column` usando le `join_keys`."
)
request = _request().model_copy(update={"normalized_text": exact + "\n"})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (exact,)