fix(evidence): stabilize real Pi authoring
This commit is contained in:
@@ -1,4 +1,5 @@
|
||||
import hashlib
|
||||
import threading
|
||||
import unicodedata
|
||||
|
||||
import pytest
|
||||
@@ -348,6 +349,40 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
|
||||
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||
|
||||
|
||||
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||
source_root = tmp_path / "evidence" / "source" / "domain"
|
||||
changed_text = source_text + "\nRegola revisionata.\n"
|
||||
(source_root / "patient.md").write_text(changed_text, encoding="utf-8")
|
||||
(source_root / "second.md").write_text(changed_text, encoding="utf-8")
|
||||
barrier = threading.Barrier(2)
|
||||
|
||||
class ConcurrentRestructurer:
|
||||
def __init__(self):
|
||||
self.requests = []
|
||||
|
||||
def restructure(self, request):
|
||||
self.requests.append(request)
|
||||
barrier.wait(timeout=2)
|
||||
return (_candidate(title=f"Unit {request.source_file}"),)
|
||||
|
||||
restructurer = ConcurrentRestructurer()
|
||||
|
||||
report = prepare_workspace_evidence(
|
||||
tmp_path,
|
||||
restructurer=restructurer,
|
||||
git_status=lambda _: (),
|
||||
max_workers=2,
|
||||
)
|
||||
|
||||
assert report.model_calls == 2
|
||||
assert sorted(request.source_file for request in restructurer.requests) == [
|
||||
"source/domain/patient.md",
|
||||
"source/domain/second.md",
|
||||
]
|
||||
|
||||
|
||||
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||
|
||||
@@ -14,6 +14,32 @@ def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
|
||||
assert "validate" in result.output
|
||||
|
||||
|
||||
def test_evidence_prepare_failure_identifies_the_source_file(monkeypatch, tmp_path):
|
||||
from tht.cli import evidence_cmd
|
||||
from tht.evidence import EvidencePreparationError
|
||||
|
||||
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
|
||||
monkeypatch.setattr(
|
||||
evidence_cmd,
|
||||
"prepare_workspace_evidence",
|
||||
lambda *args, **kwargs: (_ for _ in ()).throw(
|
||||
EvidencePreparationError("pi_restructure_invalid", "source/domain/patient.md")
|
||||
),
|
||||
)
|
||||
|
||||
result = CliRunner().invoke(app, ["evidence", "prepare", str(tmp_path), "--json"])
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.stderr == ""
|
||||
assert json.loads(result.stdout) == {
|
||||
"code": "pi_restructure_invalid",
|
||||
"operation": "evidence_prepare",
|
||||
"schemaVersion": 1,
|
||||
"sourceFile": "source/domain/patient.md",
|
||||
"status": "failed",
|
||||
}
|
||||
|
||||
|
||||
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
|
||||
from tht.cli import evidence_cmd
|
||||
from tht.evidence.authoring import EvidenceResolutionReport
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
@@ -35,6 +36,9 @@ def _candidate():
|
||||
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
monkeypatch.setenv("PI_PROVIDER", "zai")
|
||||
monkeypatch.setenv("PI_MODEL", "glm-5.2")
|
||||
monkeypatch.setenv("PI_THINKING", "medium")
|
||||
calls = []
|
||||
|
||||
def run(argv, **kwargs):
|
||||
@@ -43,7 +47,9 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
|
||||
|
||||
candidates = restructurer.restructure(_request())
|
||||
|
||||
@@ -52,7 +58,22 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
||||
assert argv[:8] == [
|
||||
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
|
||||
]
|
||||
assert "--skill" in argv
|
||||
assert "--no-skills" in argv
|
||||
assert argv[argv.index("--extension") + 1].endswith(
|
||||
"/extensions/tht-evidence-json-mode.ts"
|
||||
)
|
||||
assert "--system-prompt" in argv
|
||||
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
|
||||
assert "--skill" not in argv
|
||||
assert "--append-system-prompt" not in argv
|
||||
assert "/skill:tht-evidence-authoring" not in argv
|
||||
assert argv[argv.index("--provider") + 1] == "zai"
|
||||
assert argv[argv.index("--model") + 1] == "glm-5.2"
|
||||
assert argv[argv.index("--thinking") + 1] == "medium"
|
||||
assert any(
|
||||
"must not use kind enum or formula" in argument
|
||||
for argument in argv
|
||||
)
|
||||
assert any(argument.startswith("@") for argument in argv)
|
||||
assert kwargs["timeout"] == 12
|
||||
assert kwargs["shell"] is False
|
||||
@@ -71,10 +92,144 @@ def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path,
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
|
||||
|
||||
with pytest.raises(EvidencePreparationError) as error:
|
||||
restructurer.restructure(_request())
|
||||
|
||||
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
|
||||
assert "secret" not in str(error.value)
|
||||
|
||||
|
||||
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
|
||||
skill = (
|
||||
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
).read_text(encoding="utf-8")
|
||||
|
||||
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
|
||||
for required_field in (
|
||||
"schema_version",
|
||||
"existing_id",
|
||||
"supporting_excerpts",
|
||||
"review_items",
|
||||
"payload",
|
||||
):
|
||||
assert required_field in skill
|
||||
|
||||
|
||||
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
|
||||
extension = (
|
||||
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
|
||||
).read_text(encoding="utf-8")
|
||||
|
||||
assert 'pi.on("before_provider_request"' in extension
|
||||
assert 'response_format: { type: "json_object" }' in extension
|
||||
assert "temperature: 0" in extension
|
||||
assert "registerTool" not in extension
|
||||
|
||||
|
||||
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
|
||||
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
|
||||
tmp_path, monkeypatch, response,
|
||||
):
|
||||
from tht.evidence import authoring
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps(response))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
|
||||
|
||||
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
|
||||
|
||||
|
||||
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
request = _request().model_copy(update={
|
||||
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
|
||||
})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (
|
||||
"- **Pazienti** sotto i 18 anni sono pediatrici.",
|
||||
)
|
||||
|
||||
|
||||
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
request = _request().model_copy(update={
|
||||
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
|
||||
})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (
|
||||
"Pazienti sotto i 18 anni sono pediatrici.",
|
||||
)
|
||||
assert [item.code for item in candidates[0].review_items] == [
|
||||
"supporting_excerpt_reconciled",
|
||||
]
|
||||
|
||||
|
||||
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": [
|
||||
(
|
||||
"pivot/denormalization (solo per le FACT): legge le righe correlate "
|
||||
"nella tabella source_table, estrae source_column usando le join_keys."
|
||||
)
|
||||
],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
exact = (
|
||||
"- **pivot/denormalization** (solo per le FACT):\n"
|
||||
" - legge le righe correlate nella tabella `source_table`,\n"
|
||||
" - estrae `source_column` usando le `join_keys`."
|
||||
)
|
||||
request = _request().model_copy(update={"normalized_text": exact + "\n"})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (exact,)
|
||||
|
||||
Reference in New Issue
Block a user