feat(evidence): prepare curated evidence incrementally

This commit is contained in:
2026-08-24 20:08:27 +02:00
parent 6176410f42
commit f5c7cc6198
10 changed files with 899 additions and 7 deletions
+2
View File
@@ -10,6 +10,8 @@
"decision add",
"decision add-batch",
"decision add-join-set",
"evidence prepare",
"evidence validate",
"memory promote",
"memory save-one",
"memory search",
+1 -1
View File
@@ -32,7 +32,7 @@ def test_typer_tree_matches_the_approved_command_surface():
approved = _approved_surface()
expected = set(approved["maintained"]) | set(approved["enhanced"])
assert len(approved["maintained"]) == 55
assert len(approved["maintained"]) == 57
assert len(approved["enhanced"]) == 8
assert len(approved["erased"]) == 14
assert not (expected & set(approved["erased"]))
+159 -1
View File
@@ -1,4 +1,5 @@
import hashlib
import unicodedata
import pytest
from pydantic import ValidationError
@@ -6,14 +7,24 @@ from pydantic import ValidationError
from tht.evidence import (
CuratedEvidence,
EvidenceManifest,
EvidencePreparationError,
EvidenceRestructurer,
RestructureCandidate,
dump_curated_markdown,
dump_manifest,
load_curated_tree,
load_manifest,
prepare_workspace_evidence,
validate_workspace_evidence,
)
def _evidence(source_text: str, *, review_items=()):
normalized_source = unicodedata.normalize(
"NFC", source_text.replace("\r\n", "\n").replace("\r", "\n"),
)
if not normalized_source.endswith("\n"):
normalized_source += "\n"
return CuratedEvidence.model_validate({
"schema_version": 1,
"id": "evidence:fascia-pediatrica",
@@ -24,7 +35,7 @@ def _evidence(source_text: str, *, review_items=()):
"language": "it",
"provenance": {
"source_file": "source/domain/patient.md",
"source_sha256": "sha256:" + hashlib.sha256(source_text.encode()).hexdigest(),
"source_sha256": "sha256:" + hashlib.sha256(normalized_source.encode()).hexdigest(),
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
},
"review_items": list(review_items),
@@ -279,3 +290,150 @@ def test_workspace_validation_reports_unreadable_or_oversized_sources(tmp_path,
oversized = validate_workspace_evidence(tmp_path)
assert [finding.code for finding in oversized.findings] == ["source_oversized"]
def test_workspace_validation_rejects_a_source_symlink(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
outside = tmp_path / "outside.md"
outside.write_text(source_text, encoding="utf-8")
source_path.unlink()
source_path.symlink_to(outside)
report = validate_workspace_evidence(tmp_path)
assert [finding.code for finding in report.findings] == ["source_unsafe"]
class _Restructurer(EvidenceRestructurer):
def __init__(self, candidates):
self.candidates = candidates
self.requests = []
def restructure(self, request):
self.requests.append(request)
return tuple(self.candidates)
def _candidate(*, title="Fascia pediatrica", existing_id=None):
return RestructureCandidate.model_validate({
"schema_version": 1,
"existing_id": existing_id,
"title": title,
"kind": "domain",
"purposes": ["disambiguation"],
"applies_to": {"concepts": ["fascia pediatrica"]},
"language": "it",
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
"review_items": [],
"payload": {"rule": "La fascia pediatrica comprende i minori."},
})
def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
source_path.write_text(source_text + " La regola è revisionata.\n", encoding="utf-8")
restructurer = _Restructurer([_candidate(existing_id="evidence:fascia-pediatrica")])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 1
assert report.changed == ("source/domain/patient.md",)
assert report.created == ()
assert len(restructurer.requests) == 1
assert restructurer.requests[0].previous_units[0].id == "evidence:fascia-pediatrica"
assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
original = curated_path.read_text(encoding="utf-8")
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.unchanged == ("source/domain/patient.md",)
assert curated_path.read_text(encoding="utf-8") == original
assert restructurer.requests == []
def test_prepare_rejects_dirty_curated_state_before_model_call(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
restructurer = _Restructurer([])
with pytest.raises(EvidencePreparationError, match="authoring_worktree_dirty"):
prepare_workspace_evidence(
tmp_path,
restructurer=restructurer,
git_status=lambda _: (" M evidence/curated/domain/patient.md",),
)
assert restructurer.requests == []
def test_prepare_retains_units_from_a_removed_source_as_blocking_orphans(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
(tmp_path / "evidence" / "source" / "domain" / "patient.md").unlink()
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.orphaned == ("evidence:fascia-pediatrica",)
assert (tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md").is_file()
assert [finding.code for finding in report.findings] == ["orphaned_unit"]
def test_prepare_preserves_ids_without_orphans_for_a_unique_source_rename(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_root = tmp_path / "evidence" / "source" / "domain"
(source_root / "patient.md").rename(source_root / "patient-age.md")
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.orphaned == ()
assert report.unchanged == ("source/domain/patient-age.md",)
assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_rejects_an_unknown_existing_id_without_writing_the_batch(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
source_path.write_text(source_text + " Revisione.\n", encoding="utf-8")
original = (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8")
restructurer = _Restructurer([_candidate(existing_id="evidence:not-supplied")])
with pytest.raises(EvidencePreparationError, match="unknown_existing_id"):
prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8") == original
def test_prepare_marks_an_omitted_prior_unit_for_human_review(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
(tmp_path / "evidence" / "source" / "domain" / "patient.md").write_text(
"La classificazione pediatrica non è documentata.\n", encoding="utf-8",
)
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 1
assert [finding.code for finding in report.findings] == [
"supporting_excerpt_missing", "unresolved_review_item",
]
retained = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert retained.review_items[0].code == "source_no_longer_supports_unit"
+40
View File
@@ -0,0 +1,40 @@
import json
from typer.testing import CliRunner
from tht.cli import app
def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
result = CliRunner().invoke(app, ["evidence", "--help"])
assert result.exit_code == 0
assert "prepare" in result.output
assert "validate" in result.output
def test_evidence_validate_json_is_pristine_and_reports_review_required(monkeypatch, tmp_path):
from tht.cli import evidence_cmd
from tht.evidence import ValidationFinding, ValidationReport
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
monkeypatch.setattr(evidence_cmd, "validate_workspace_evidence", lambda root: ValidationReport((
ValidationFinding("error", "unresolved_review_item", "curated/domain/x.md", "Review needed."),
)))
result = CliRunner().invoke(app, ["evidence", "validate", str(tmp_path), "--json"])
assert result.exit_code == 3
assert result.stderr == ""
assert json.loads(result.stdout) == {
"findings": [{
"code": "unresolved_review_item",
"message": "Review needed.",
"path": "curated/domain/x.md",
"severity": "error",
}],
"operation": "evidence_validate",
"publishable": False,
"schemaVersion": 1,
"status": "review_required",
}
@@ -0,0 +1,80 @@
import json
from types import SimpleNamespace
import pytest
from tht.evidence import (
EvidencePreparationError,
PiEvidenceRestructurer,
RestructureRequest,
)
def _request():
return RestructureRequest.model_validate({
"source_file": "source/domain/patient.md",
"source_sha256": "sha256:" + "a" * 64,
"normalized_text": "I pazienti sotto i 18 anni sono pediatrici.\n",
})
def _candidate():
return {
"schema_version": 1,
"title": "Fascia pediatrica",
"kind": "domain",
"purposes": ["disambiguation"],
"applies_to": {"concepts": ["fascia pediatrica"]},
"language": "it",
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
"review_items": [],
"payload": {"rule": "La fascia pediatrica comprende i minori."},
}
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
from tht.evidence import authoring
calls = []
def run(argv, **kwargs):
calls.append((argv, kwargs))
kwargs["stdout"].write(json.dumps({"candidates": [_candidate()]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
candidates = restructurer.restructure(_request())
assert candidates[0].title == "Fascia pediatrica"
argv, kwargs = calls[0]
assert argv[:8] == [
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
]
assert "--skill" in argv
assert any(argument.startswith("@") for argument in argv)
assert kwargs["timeout"] == 12
assert kwargs["shell"] is False
assert kwargs["stderr"] is authoring.subprocess.DEVNULL
@pytest.mark.parametrize("result", [
SimpleNamespace(returncode=1, stdout="secret", stderr="secret"),
SimpleNamespace(returncode=0, stdout="not-json", stderr=""),
])
def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path, monkeypatch, result):
from tht.evidence import authoring
def run(*args, **kwargs):
kwargs["stdout"].write(result.stdout)
return result
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
with pytest.raises(EvidencePreparationError) as error:
restructurer.restructure(_request())
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
assert "secret" not in str(error.value)