feat(evidence): prepare curated evidence incrementally
This commit is contained in:
@@ -0,0 +1,15 @@
|
|||||||
|
# Evidence authoring response contract
|
||||||
|
|
||||||
|
You receive exactly one normalized Source Evidence request. Return one JSON object with
|
||||||
|
only a `candidates` array, without Markdown fences or explanatory text.
|
||||||
|
|
||||||
|
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
|
||||||
|
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
|
||||||
|
otherwise omit it. Preserve prior reviewed wording when it is still supported. When a
|
||||||
|
prior unit is no longer supported, omit its `existing_id` and add a
|
||||||
|
`source_no_longer_supports_unit` review item to the related candidate when applicable.
|
||||||
|
|
||||||
|
Every candidate must match schema version 1, use one of the eight declared Evidence
|
||||||
|
kinds, include one to five short exact excerpts copied from `normalized_text`, and add
|
||||||
|
review items for ambiguities. Do not assign a new canonical ID; the host does that
|
||||||
|
deterministically. Do not use tools or alter any repository state.
|
||||||
@@ -10,6 +10,8 @@
|
|||||||
"decision add",
|
"decision add",
|
||||||
"decision add-batch",
|
"decision add-batch",
|
||||||
"decision add-join-set",
|
"decision add-join-set",
|
||||||
|
"evidence prepare",
|
||||||
|
"evidence validate",
|
||||||
"memory promote",
|
"memory promote",
|
||||||
"memory save-one",
|
"memory save-one",
|
||||||
"memory search",
|
"memory search",
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ def test_typer_tree_matches_the_approved_command_surface():
|
|||||||
approved = _approved_surface()
|
approved = _approved_surface()
|
||||||
expected = set(approved["maintained"]) | set(approved["enhanced"])
|
expected = set(approved["maintained"]) | set(approved["enhanced"])
|
||||||
|
|
||||||
assert len(approved["maintained"]) == 55
|
assert len(approved["maintained"]) == 57
|
||||||
assert len(approved["enhanced"]) == 8
|
assert len(approved["enhanced"]) == 8
|
||||||
assert len(approved["erased"]) == 14
|
assert len(approved["erased"]) == 14
|
||||||
assert not (expected & set(approved["erased"]))
|
assert not (expected & set(approved["erased"]))
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import hashlib
|
import hashlib
|
||||||
|
import unicodedata
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from pydantic import ValidationError
|
from pydantic import ValidationError
|
||||||
@@ -6,14 +7,24 @@ from pydantic import ValidationError
|
|||||||
from tht.evidence import (
|
from tht.evidence import (
|
||||||
CuratedEvidence,
|
CuratedEvidence,
|
||||||
EvidenceManifest,
|
EvidenceManifest,
|
||||||
|
EvidencePreparationError,
|
||||||
|
EvidenceRestructurer,
|
||||||
|
RestructureCandidate,
|
||||||
dump_curated_markdown,
|
dump_curated_markdown,
|
||||||
dump_manifest,
|
dump_manifest,
|
||||||
|
load_curated_tree,
|
||||||
load_manifest,
|
load_manifest,
|
||||||
|
prepare_workspace_evidence,
|
||||||
validate_workspace_evidence,
|
validate_workspace_evidence,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _evidence(source_text: str, *, review_items=()):
|
def _evidence(source_text: str, *, review_items=()):
|
||||||
|
normalized_source = unicodedata.normalize(
|
||||||
|
"NFC", source_text.replace("\r\n", "\n").replace("\r", "\n"),
|
||||||
|
)
|
||||||
|
if not normalized_source.endswith("\n"):
|
||||||
|
normalized_source += "\n"
|
||||||
return CuratedEvidence.model_validate({
|
return CuratedEvidence.model_validate({
|
||||||
"schema_version": 1,
|
"schema_version": 1,
|
||||||
"id": "evidence:fascia-pediatrica",
|
"id": "evidence:fascia-pediatrica",
|
||||||
@@ -24,7 +35,7 @@ def _evidence(source_text: str, *, review_items=()):
|
|||||||
"language": "it",
|
"language": "it",
|
||||||
"provenance": {
|
"provenance": {
|
||||||
"source_file": "source/domain/patient.md",
|
"source_file": "source/domain/patient.md",
|
||||||
"source_sha256": "sha256:" + hashlib.sha256(source_text.encode()).hexdigest(),
|
"source_sha256": "sha256:" + hashlib.sha256(normalized_source.encode()).hexdigest(),
|
||||||
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
||||||
},
|
},
|
||||||
"review_items": list(review_items),
|
"review_items": list(review_items),
|
||||||
@@ -279,3 +290,150 @@ def test_workspace_validation_reports_unreadable_or_oversized_sources(tmp_path,
|
|||||||
oversized = validate_workspace_evidence(tmp_path)
|
oversized = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
assert [finding.code for finding in oversized.findings] == ["source_oversized"]
|
assert [finding.code for finding in oversized.findings] == ["source_oversized"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_rejects_a_source_symlink(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
|
||||||
|
outside = tmp_path / "outside.md"
|
||||||
|
outside.write_text(source_text, encoding="utf-8")
|
||||||
|
source_path.unlink()
|
||||||
|
source_path.symlink_to(outside)
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert [finding.code for finding in report.findings] == ["source_unsafe"]
|
||||||
|
|
||||||
|
|
||||||
|
class _Restructurer(EvidenceRestructurer):
|
||||||
|
def __init__(self, candidates):
|
||||||
|
self.candidates = candidates
|
||||||
|
self.requests = []
|
||||||
|
|
||||||
|
def restructure(self, request):
|
||||||
|
self.requests.append(request)
|
||||||
|
return tuple(self.candidates)
|
||||||
|
|
||||||
|
|
||||||
|
def _candidate(*, title="Fascia pediatrica", existing_id=None):
|
||||||
|
return RestructureCandidate.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"existing_id": existing_id,
|
||||||
|
"title": title,
|
||||||
|
"kind": "domain",
|
||||||
|
"purposes": ["disambiguation"],
|
||||||
|
"applies_to": {"concepts": ["fascia pediatrica"]},
|
||||||
|
"language": "it",
|
||||||
|
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
||||||
|
"review_items": [],
|
||||||
|
"payload": {"rule": "La fascia pediatrica comprende i minori."},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
|
||||||
|
source_path.write_text(source_text + " La regola è revisionata.\n", encoding="utf-8")
|
||||||
|
restructurer = _Restructurer([_candidate(existing_id="evidence:fascia-pediatrica")])
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert report.model_calls == 1
|
||||||
|
assert report.changed == ("source/domain/patient.md",)
|
||||||
|
assert report.created == ()
|
||||||
|
assert len(restructurer.requests) == 1
|
||||||
|
assert restructurer.requests[0].previous_units[0].id == "evidence:fascia-pediatrica"
|
||||||
|
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
|
||||||
|
original = curated_path.read_text(encoding="utf-8")
|
||||||
|
restructurer = _Restructurer([])
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert report.model_calls == 0
|
||||||
|
assert report.unchanged == ("source/domain/patient.md",)
|
||||||
|
assert curated_path.read_text(encoding="utf-8") == original
|
||||||
|
assert restructurer.requests == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_rejects_dirty_curated_state_before_model_call(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
restructurer = _Restructurer([])
|
||||||
|
|
||||||
|
with pytest.raises(EvidencePreparationError, match="authoring_worktree_dirty"):
|
||||||
|
prepare_workspace_evidence(
|
||||||
|
tmp_path,
|
||||||
|
restructurer=restructurer,
|
||||||
|
git_status=lambda _: (" M evidence/curated/domain/patient.md",),
|
||||||
|
)
|
||||||
|
|
||||||
|
assert restructurer.requests == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_retains_units_from_a_removed_source_as_blocking_orphans(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
(tmp_path / "evidence" / "source" / "domain" / "patient.md").unlink()
|
||||||
|
restructurer = _Restructurer([])
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert report.model_calls == 0
|
||||||
|
assert report.orphaned == ("evidence:fascia-pediatrica",)
|
||||||
|
assert (tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md").is_file()
|
||||||
|
assert [finding.code for finding in report.findings] == ["orphaned_unit"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_preserves_ids_without_orphans_for_a_unique_source_rename(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_root = tmp_path / "evidence" / "source" / "domain"
|
||||||
|
(source_root / "patient.md").rename(source_root / "patient-age.md")
|
||||||
|
restructurer = _Restructurer([])
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert report.model_calls == 0
|
||||||
|
assert report.orphaned == ()
|
||||||
|
assert report.unchanged == ("source/domain/patient-age.md",)
|
||||||
|
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_rejects_an_unknown_existing_id_without_writing_the_batch(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
|
||||||
|
source_path.write_text(source_text + " Revisione.\n", encoding="utf-8")
|
||||||
|
original = (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8")
|
||||||
|
restructurer = _Restructurer([_candidate(existing_id="evidence:not-supplied")])
|
||||||
|
|
||||||
|
with pytest.raises(EvidencePreparationError, match="unknown_existing_id"):
|
||||||
|
prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8") == original
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_marks_an_omitted_prior_unit_for_human_review(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
(tmp_path / "evidence" / "source" / "domain" / "patient.md").write_text(
|
||||||
|
"La classificazione pediatrica non è documentata.\n", encoding="utf-8",
|
||||||
|
)
|
||||||
|
restructurer = _Restructurer([])
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
|
||||||
|
|
||||||
|
assert report.model_calls == 1
|
||||||
|
assert [finding.code for finding in report.findings] == [
|
||||||
|
"supporting_excerpt_missing", "unresolved_review_item",
|
||||||
|
]
|
||||||
|
retained = load_curated_tree(tmp_path / "evidence" / "curated")[0]
|
||||||
|
assert retained.review_items[0].code == "source_no_longer_supports_unit"
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
import json
|
||||||
|
|
||||||
|
from typer.testing import CliRunner
|
||||||
|
|
||||||
|
from tht.cli import app
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
|
||||||
|
result = CliRunner().invoke(app, ["evidence", "--help"])
|
||||||
|
|
||||||
|
assert result.exit_code == 0
|
||||||
|
assert "prepare" in result.output
|
||||||
|
assert "validate" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_validate_json_is_pristine_and_reports_review_required(monkeypatch, tmp_path):
|
||||||
|
from tht.cli import evidence_cmd
|
||||||
|
from tht.evidence import ValidationFinding, ValidationReport
|
||||||
|
|
||||||
|
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
|
||||||
|
monkeypatch.setattr(evidence_cmd, "validate_workspace_evidence", lambda root: ValidationReport((
|
||||||
|
ValidationFinding("error", "unresolved_review_item", "curated/domain/x.md", "Review needed."),
|
||||||
|
)))
|
||||||
|
|
||||||
|
result = CliRunner().invoke(app, ["evidence", "validate", str(tmp_path), "--json"])
|
||||||
|
|
||||||
|
assert result.exit_code == 3
|
||||||
|
assert result.stderr == ""
|
||||||
|
assert json.loads(result.stdout) == {
|
||||||
|
"findings": [{
|
||||||
|
"code": "unresolved_review_item",
|
||||||
|
"message": "Review needed.",
|
||||||
|
"path": "curated/domain/x.md",
|
||||||
|
"severity": "error",
|
||||||
|
}],
|
||||||
|
"operation": "evidence_validate",
|
||||||
|
"publishable": False,
|
||||||
|
"schemaVersion": 1,
|
||||||
|
"status": "review_required",
|
||||||
|
}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
import json
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from tht.evidence import (
|
||||||
|
EvidencePreparationError,
|
||||||
|
PiEvidenceRestructurer,
|
||||||
|
RestructureRequest,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _request():
|
||||||
|
return RestructureRequest.model_validate({
|
||||||
|
"source_file": "source/domain/patient.md",
|
||||||
|
"source_sha256": "sha256:" + "a" * 64,
|
||||||
|
"normalized_text": "I pazienti sotto i 18 anni sono pediatrici.\n",
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def _candidate():
|
||||||
|
return {
|
||||||
|
"schema_version": 1,
|
||||||
|
"title": "Fascia pediatrica",
|
||||||
|
"kind": "domain",
|
||||||
|
"purposes": ["disambiguation"],
|
||||||
|
"applies_to": {"concepts": ["fascia pediatrica"]},
|
||||||
|
"language": "it",
|
||||||
|
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
||||||
|
"review_items": [],
|
||||||
|
"payload": {"rule": "La fascia pediatrica comprende i minori."},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
def run(argv, **kwargs):
|
||||||
|
calls.append((argv, kwargs))
|
||||||
|
kwargs["stdout"].write(json.dumps({"candidates": [_candidate()]}))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
|
||||||
|
|
||||||
|
candidates = restructurer.restructure(_request())
|
||||||
|
|
||||||
|
assert candidates[0].title == "Fascia pediatrica"
|
||||||
|
argv, kwargs = calls[0]
|
||||||
|
assert argv[:8] == [
|
||||||
|
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
|
||||||
|
]
|
||||||
|
assert "--skill" in argv
|
||||||
|
assert any(argument.startswith("@") for argument in argv)
|
||||||
|
assert kwargs["timeout"] == 12
|
||||||
|
assert kwargs["shell"] is False
|
||||||
|
assert kwargs["stderr"] is authoring.subprocess.DEVNULL
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("result", [
|
||||||
|
SimpleNamespace(returncode=1, stdout="secret", stderr="secret"),
|
||||||
|
SimpleNamespace(returncode=0, stdout="not-json", stderr=""),
|
||||||
|
])
|
||||||
|
def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path, monkeypatch, result):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
def run(*args, **kwargs):
|
||||||
|
kwargs["stdout"].write(result.stdout)
|
||||||
|
return result
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
|
||||||
|
|
||||||
|
with pytest.raises(EvidencePreparationError) as error:
|
||||||
|
restructurer.restructure(_request())
|
||||||
|
|
||||||
|
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
|
||||||
|
assert "secret" not in str(error.value)
|
||||||
@@ -42,6 +42,7 @@ from tht.cli.datamart_cmd import datamart_app
|
|||||||
from tht.cli.db_cmd import db_app
|
from tht.cli.db_cmd import db_app
|
||||||
from tht.cli.decision_cmd import decision_app
|
from tht.cli.decision_cmd import decision_app
|
||||||
from tht.cli.doctor_cmd import doctor
|
from tht.cli.doctor_cmd import doctor
|
||||||
|
from tht.cli.evidence_cmd import evidence_app
|
||||||
from tht.cli.memory_cmd import memory_app
|
from tht.cli.memory_cmd import memory_app
|
||||||
from tht.cli.ollama_cmd import ollama_app
|
from tht.cli.ollama_cmd import ollama_app
|
||||||
from tht.cli.phase_cmd import phase_app
|
from tht.cli.phase_cmd import phase_app
|
||||||
@@ -54,6 +55,7 @@ from tht.cli.vector_cmd import vector_app
|
|||||||
|
|
||||||
app.add_typer(phase_app, name="phase")
|
app.add_typer(phase_app, name="phase")
|
||||||
app.add_typer(preprocess_app, name="preprocess")
|
app.add_typer(preprocess_app, name="preprocess")
|
||||||
|
app.add_typer(evidence_app, name="evidence")
|
||||||
app.add_typer(config_app, name="config")
|
app.add_typer(config_app, name="config")
|
||||||
app.add_typer(schema_app, name="schema")
|
app.add_typer(schema_app, name="schema")
|
||||||
app.add_typer(session_app, name="session")
|
app.add_typer(session_app, name="session")
|
||||||
|
|||||||
@@ -1,5 +1,105 @@
|
|||||||
|
"""Curator-facing Evidence authoring commands, separate from runtime preprocessing."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import Annotated
|
||||||
|
|
||||||
|
import typer
|
||||||
|
|
||||||
|
from tht.evidence import (
|
||||||
|
EvidencePreparationError,
|
||||||
|
PiEvidenceRestructurer,
|
||||||
|
prepare_workspace_evidence,
|
||||||
|
validate_workspace_evidence,
|
||||||
|
)
|
||||||
|
|
||||||
|
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
|
||||||
|
|
||||||
|
|
||||||
def evidence_root(cfg) -> Path:
|
def _canonical_worktree(workspace_root: Path) -> Path:
|
||||||
return cfg.paths.artifacts / "evidence"
|
if workspace_root.is_symlink():
|
||||||
|
raise typer.BadParameter("workspace-root must not be a symlink")
|
||||||
|
root = workspace_root.resolve()
|
||||||
|
result = subprocess.run(
|
||||||
|
["git", "rev-parse", "--show-toplevel"],
|
||||||
|
cwd=root,
|
||||||
|
check=False,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
)
|
||||||
|
if result.returncode != 0 or Path(result.stdout.strip()).resolve() != root:
|
||||||
|
raise typer.BadParameter("workspace-root must be the root of a canonical Git worktree")
|
||||||
|
return root
|
||||||
|
|
||||||
|
|
||||||
|
def _emit(payload: dict[str, object], json_output: bool) -> None:
|
||||||
|
if json_output:
|
||||||
|
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
|
||||||
|
return
|
||||||
|
for key, value in payload.items():
|
||||||
|
typer.echo(f"{key}: {value}")
|
||||||
|
|
||||||
|
|
||||||
|
def _preparation_payload(report) -> dict[str, object]:
|
||||||
|
return {
|
||||||
|
"schemaVersion": 1,
|
||||||
|
"operation": "evidence_prepare",
|
||||||
|
"status": "prepared",
|
||||||
|
"changed": list(report.changed),
|
||||||
|
"unchanged": list(report.unchanged),
|
||||||
|
"created": list(report.created),
|
||||||
|
"orphaned": list(report.orphaned),
|
||||||
|
"findings": _findings_payload(report.findings),
|
||||||
|
"modelCalls": report.model_calls,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _findings_payload(findings) -> list[dict[str, object]]:
|
||||||
|
return [finding.__dict__ for finding in findings]
|
||||||
|
|
||||||
|
|
||||||
|
@evidence_app.command("prepare")
|
||||||
|
def prepare_cmd(
|
||||||
|
workspace_root: Path,
|
||||||
|
upgrade: Annotated[bool, typer.Option(help="Reprocess every source with the installed pipeline.")] = False,
|
||||||
|
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
|
||||||
|
) -> None:
|
||||||
|
"""Prepare changed Source Evidence without committing or publishing it."""
|
||||||
|
root = _canonical_worktree(workspace_root)
|
||||||
|
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||||
|
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||||
|
try:
|
||||||
|
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
|
||||||
|
except EvidencePreparationError as error:
|
||||||
|
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
|
||||||
|
raise typer.Exit(code=1) from error
|
||||||
|
_emit(_preparation_payload(report), json_output)
|
||||||
|
if report.findings:
|
||||||
|
raise typer.Exit(code=3)
|
||||||
|
|
||||||
|
|
||||||
|
@evidence_app.command("validate")
|
||||||
|
def validate_cmd(
|
||||||
|
workspace_root: Path,
|
||||||
|
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
|
||||||
|
) -> None:
|
||||||
|
"""Validate the authoring tree without writing, committing, or publishing it."""
|
||||||
|
root = _canonical_worktree(workspace_root)
|
||||||
|
report = validate_workspace_evidence(root)
|
||||||
|
payload = {
|
||||||
|
"schemaVersion": 1,
|
||||||
|
"operation": "evidence_validate",
|
||||||
|
"status": "valid" if report.publishable else "review_required",
|
||||||
|
"publishable": report.publishable,
|
||||||
|
"findings": _findings_payload(report.findings),
|
||||||
|
}
|
||||||
|
_emit(payload, json_output)
|
||||||
|
if report.publishable:
|
||||||
|
return
|
||||||
|
if all(finding.code in {"orphaned_unit", "unresolved_review_item"} for finding in report.findings):
|
||||||
|
raise typer.Exit(code=3)
|
||||||
|
raise typer.Exit(code=1)
|
||||||
|
|||||||
@@ -3,10 +3,17 @@
|
|||||||
from tht.evidence.acquisition import acquire, discover
|
from tht.evidence.acquisition import acquire, discover
|
||||||
from tht.evidence.authoring import (
|
from tht.evidence.authoring import (
|
||||||
EvidenceManifest,
|
EvidenceManifest,
|
||||||
|
EvidencePreparationError,
|
||||||
|
EvidencePreparationReport,
|
||||||
|
EvidenceRestructurer,
|
||||||
|
PiEvidenceRestructurer,
|
||||||
|
RestructureCandidate,
|
||||||
|
RestructureRequest,
|
||||||
ValidationFinding,
|
ValidationFinding,
|
||||||
ValidationReport,
|
ValidationReport,
|
||||||
dump_manifest,
|
dump_manifest,
|
||||||
load_manifest,
|
load_manifest,
|
||||||
|
prepare_workspace_evidence,
|
||||||
validate_workspace_evidence,
|
validate_workspace_evidence,
|
||||||
)
|
)
|
||||||
from tht.evidence.canonical import (
|
from tht.evidence.canonical import (
|
||||||
@@ -45,9 +52,15 @@ __all__ = [
|
|||||||
"CuratedEvidence",
|
"CuratedEvidence",
|
||||||
"EvidenceEmbedder",
|
"EvidenceEmbedder",
|
||||||
"EvidenceManifest",
|
"EvidenceManifest",
|
||||||
|
"EvidencePreparationError",
|
||||||
|
"EvidencePreparationReport",
|
||||||
|
"EvidenceRestructurer",
|
||||||
"EvidenceSource",
|
"EvidenceSource",
|
||||||
"EvidenceSourceError",
|
"EvidenceSourceError",
|
||||||
"EvidenceSourceErrorCategory",
|
"EvidenceSourceErrorCategory",
|
||||||
|
"PiEvidenceRestructurer",
|
||||||
|
"RestructureCandidate",
|
||||||
|
"RestructureRequest",
|
||||||
"SourceObject",
|
"SourceObject",
|
||||||
"ValidationFinding",
|
"ValidationFinding",
|
||||||
"ValidationReport",
|
"ValidationReport",
|
||||||
@@ -64,6 +77,7 @@ __all__ = [
|
|||||||
"load_manifest",
|
"load_manifest",
|
||||||
"normalize_aware_datetime",
|
"normalize_aware_datetime",
|
||||||
"parse_curated_markdown",
|
"parse_curated_markdown",
|
||||||
|
"prepare_workspace_evidence",
|
||||||
"project_session",
|
"project_session",
|
||||||
"resolve_citation",
|
"resolve_citation",
|
||||||
"validate_corpus_workspace",
|
"validate_corpus_workspace",
|
||||||
|
|||||||
@@ -3,17 +3,29 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import hashlib
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
import unicodedata
|
import unicodedata
|
||||||
|
from collections.abc import Callable
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Literal
|
from typing import Literal, Protocol
|
||||||
|
|
||||||
import yaml
|
import yaml
|
||||||
from pydantic import ValidationError, field_validator, model_validator
|
from pydantic import Field, ValidationError, field_validator, model_validator
|
||||||
|
|
||||||
from tht.evidence.canonical import (
|
from tht.evidence.canonical import (
|
||||||
CuratedEvidence,
|
CuratedEvidence,
|
||||||
|
EvidenceKind,
|
||||||
|
EvidencePurpose,
|
||||||
|
EvidenceScope,
|
||||||
|
ReviewItem,
|
||||||
StrictModel,
|
StrictModel,
|
||||||
|
dump_curated_markdown,
|
||||||
is_evidence_id,
|
is_evidence_id,
|
||||||
load_curated_tree,
|
load_curated_tree,
|
||||||
validate_source_file,
|
validate_source_file,
|
||||||
@@ -39,6 +51,128 @@ class ValidationReport:
|
|||||||
return not any(finding.severity == "error" for finding in self.findings)
|
return not any(finding.severity == "error" for finding in self.findings)
|
||||||
|
|
||||||
|
|
||||||
|
class EvidencePreparationError(RuntimeError):
|
||||||
|
"""A bounded authoring failure that is safe to report to a curator."""
|
||||||
|
|
||||||
|
def __init__(self, code: str, source_file: str | None = None) -> None:
|
||||||
|
self.code = code
|
||||||
|
self.source_file = source_file
|
||||||
|
message = code if source_file is None else f"{code}: {source_file}"
|
||||||
|
super().__init__(message)
|
||||||
|
|
||||||
|
|
||||||
|
class RestructureRequest(StrictModel):
|
||||||
|
"""The deterministic input passed to exactly one model call for one source."""
|
||||||
|
|
||||||
|
source_file: str
|
||||||
|
source_sha256: str
|
||||||
|
normalized_text: str
|
||||||
|
previous_units: tuple[CuratedEvidence, ...] = ()
|
||||||
|
|
||||||
|
@field_validator("source_file")
|
||||||
|
@classmethod
|
||||||
|
def _validate_source_file(cls, value: str) -> str:
|
||||||
|
return validate_source_file(value)
|
||||||
|
|
||||||
|
|
||||||
|
class RestructureCandidate(StrictModel):
|
||||||
|
"""A model proposal before the host allocates or reuses its stable ID."""
|
||||||
|
|
||||||
|
schema_version: Literal[1]
|
||||||
|
existing_id: str | None = None
|
||||||
|
title: str
|
||||||
|
kind: EvidenceKind
|
||||||
|
purposes: tuple[EvidencePurpose, ...]
|
||||||
|
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
||||||
|
language: str
|
||||||
|
supporting_excerpts: tuple[str, ...]
|
||||||
|
review_items: tuple[ReviewItem, ...] = ()
|
||||||
|
payload: dict[str, object]
|
||||||
|
|
||||||
|
@field_validator("existing_id")
|
||||||
|
@classmethod
|
||||||
|
def _validate_existing_id(cls, value: str | None) -> str | None:
|
||||||
|
if value is not None and not is_evidence_id(value):
|
||||||
|
raise ValueError("existing_id must use the evidence:<slug> form")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceRestructurer(Protocol):
|
||||||
|
"""Boundary for the ephemeral, read-only model restructuring call."""
|
||||||
|
|
||||||
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
||||||
|
|
||||||
|
|
||||||
|
class PiEvidenceRestructurer:
|
||||||
|
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
pi_executable: str,
|
||||||
|
skill_path: Path,
|
||||||
|
*,
|
||||||
|
timeout_seconds: int = 120,
|
||||||
|
) -> None:
|
||||||
|
self._pi_executable = pi_executable
|
||||||
|
self._skill_path = skill_path
|
||||||
|
self._timeout_seconds = timeout_seconds
|
||||||
|
|
||||||
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
||||||
|
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
||||||
|
request_path = Path(temporary) / "request.json"
|
||||||
|
request_path.write_text(
|
||||||
|
json.dumps(request.model_dump(mode="json"), ensure_ascii=False),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
argv = [
|
||||||
|
self._pi_executable,
|
||||||
|
"--mode", "text",
|
||||||
|
"--print",
|
||||||
|
"--no-session",
|
||||||
|
"--no-tools",
|
||||||
|
"--no-extensions",
|
||||||
|
"--no-context-files",
|
||||||
|
"--skill", str(self._skill_path),
|
||||||
|
f"@{request_path}",
|
||||||
|
"Return only the JSON object required by the Evidence authoring skill.",
|
||||||
|
]
|
||||||
|
try:
|
||||||
|
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
||||||
|
result = subprocess.run(
|
||||||
|
argv,
|
||||||
|
check=False,
|
||||||
|
stdout=response,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
timeout=self._timeout_seconds,
|
||||||
|
shell=False,
|
||||||
|
)
|
||||||
|
if response.tell() > 1024 * 1024:
|
||||||
|
raise EvidencePreparationError("pi_restructure_invalid", request.source_file)
|
||||||
|
response.seek(0)
|
||||||
|
stdout = response.read()
|
||||||
|
except (OSError, subprocess.TimeoutExpired) as error:
|
||||||
|
raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error
|
||||||
|
if result.returncode != 0:
|
||||||
|
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
||||||
|
try:
|
||||||
|
raw = json.loads(stdout)
|
||||||
|
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
|
||||||
|
raise ValueError("response shape")
|
||||||
|
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
|
||||||
|
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
||||||
|
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class EvidencePreparationReport:
|
||||||
|
changed: tuple[str, ...]
|
||||||
|
unchanged: tuple[str, ...]
|
||||||
|
created: tuple[str, ...]
|
||||||
|
orphaned: tuple[str, ...]
|
||||||
|
findings: tuple[ValidationFinding, ...]
|
||||||
|
model_calls: int
|
||||||
|
|
||||||
|
|
||||||
class ManifestSource(StrictModel):
|
class ManifestSource(StrictModel):
|
||||||
sha256: str
|
sha256: str
|
||||||
units: tuple[str, ...]
|
units: tuple[str, ...]
|
||||||
@@ -183,6 +317,10 @@ def _validate_manifest_source(
|
|||||||
findings: list[ValidationFinding],
|
findings: list[ValidationFinding],
|
||||||
) -> str | None:
|
) -> str | None:
|
||||||
path = evidence_root / source_path
|
path = evidence_root / source_path
|
||||||
|
if path.is_symlink():
|
||||||
|
findings.append(ValidationFinding("error", "source_unsafe", source_path,
|
||||||
|
"The manifest source must be a regular file below the Evidence tree."))
|
||||||
|
return None
|
||||||
if not path.is_file():
|
if not path.is_file():
|
||||||
findings.append(ValidationFinding("error", "source_missing", source_path,
|
findings.append(ValidationFinding("error", "source_missing", source_path,
|
||||||
"The manifest source does not exist."))
|
"The manifest source does not exist."))
|
||||||
@@ -192,7 +330,7 @@ def _validate_manifest_source(
|
|||||||
"The manifest source exceeds the authoring size limit."))
|
"The manifest source exceeds the authoring size limit."))
|
||||||
return None
|
return None
|
||||||
try:
|
try:
|
||||||
source = _normalize(path.read_text(encoding="utf-8"))
|
source = normalize_source_text(path.read_text(encoding="utf-8"))
|
||||||
except UnicodeDecodeError:
|
except UnicodeDecodeError:
|
||||||
findings.append(ValidationFinding("error", "source_unreadable", source_path,
|
findings.append(ValidationFinding("error", "source_unreadable", source_path,
|
||||||
"The manifest source is not valid UTF-8."))
|
"The manifest source is not valid UTF-8."))
|
||||||
@@ -207,6 +345,8 @@ def _validate_manifest_source(
|
|||||||
def _validate_unit(
|
def _validate_unit(
|
||||||
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
||||||
) -> list[ValidationFinding]:
|
) -> list[ValidationFinding]:
|
||||||
|
if evidence.id in manifest.orphans:
|
||||||
|
return []
|
||||||
path = evidence.provenance.source_file
|
path = evidence.provenance.source_file
|
||||||
findings: list[ValidationFinding] = []
|
findings: list[ValidationFinding] = []
|
||||||
manifest_source = manifest.sources.get(path)
|
manifest_source = manifest.sources.get(path)
|
||||||
@@ -254,3 +394,344 @@ def _validate_unit(
|
|||||||
|
|
||||||
def _normalize(text: str) -> str:
|
def _normalize(text: str) -> str:
|
||||||
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
|
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_source_text(text: str) -> str:
|
||||||
|
"""Normalize Source Evidence mechanically without changing its meaning."""
|
||||||
|
normalized = _normalize(text)
|
||||||
|
return normalized if normalized.endswith("\n") else normalized + "\n"
|
||||||
|
|
||||||
|
|
||||||
|
def prepare_workspace_evidence(
|
||||||
|
workspace_root: Path,
|
||||||
|
*,
|
||||||
|
restructurer: EvidenceRestructurer,
|
||||||
|
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||||
|
upgrade: bool = False,
|
||||||
|
) -> EvidencePreparationReport:
|
||||||
|
"""Prepare all changed Source Evidence without publishing or committing it.
|
||||||
|
|
||||||
|
Every deterministic operation happens locally. A changed source receives exactly
|
||||||
|
one call through ``restructurer``; all model results validate before the staged
|
||||||
|
authoring tree replaces the current one.
|
||||||
|
"""
|
||||||
|
workspace_root = workspace_root.resolve()
|
||||||
|
evidence_root = workspace_root / "evidence"
|
||||||
|
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||||
|
manifest_path = evidence_root / "manifest.yaml"
|
||||||
|
try:
|
||||||
|
manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade)
|
||||||
|
documents = load_curated_tree(evidence_root / "curated")
|
||||||
|
except (OSError, ValidationError, ValueError) as error:
|
||||||
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
||||||
|
|
||||||
|
source_texts = _load_source_texts(evidence_root)
|
||||||
|
documents_by_id = {document.id: document for document in documents}
|
||||||
|
if len(documents_by_id) != len(documents):
|
||||||
|
raise EvidencePreparationError("duplicate_evidence_id")
|
||||||
|
source_units = {
|
||||||
|
source_file: tuple(source.units)
|
||||||
|
for source_file, source in manifest.sources.items()
|
||||||
|
}
|
||||||
|
orphaned = set(manifest.orphans)
|
||||||
|
created: list[str] = []
|
||||||
|
changed: list[str] = []
|
||||||
|
unchanged: list[str] = []
|
||||||
|
model_calls = 0
|
||||||
|
|
||||||
|
renamed_sources = _preserve_uniquely_renamed_sources(
|
||||||
|
source_texts, manifest, documents_by_id, source_units,
|
||||||
|
)
|
||||||
|
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
||||||
|
|
||||||
|
reserved_ids = set(documents_by_id)
|
||||||
|
for source_file in sorted(source_texts):
|
||||||
|
source_text = source_texts[source_file]
|
||||||
|
source_hash = _source_hash(source_text)
|
||||||
|
previous = tuple(
|
||||||
|
documents_by_id[unit]
|
||||||
|
for unit in source_units.get(source_file, ())
|
||||||
|
if unit in documents_by_id
|
||||||
|
)
|
||||||
|
manifest_source = manifest.sources.get(source_file)
|
||||||
|
if not upgrade and (source_file in renamed_sources or (
|
||||||
|
manifest_source is not None and manifest_source.sha256 == source_hash
|
||||||
|
)):
|
||||||
|
unchanged.append(source_file)
|
||||||
|
continue
|
||||||
|
changed.append(source_file)
|
||||||
|
request = RestructureRequest(
|
||||||
|
source_file=source_file,
|
||||||
|
source_sha256=source_hash,
|
||||||
|
normalized_text=source_text,
|
||||||
|
previous_units=previous,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
candidates = restructurer.restructure(request)
|
||||||
|
except EvidencePreparationError:
|
||||||
|
raise
|
||||||
|
except Exception as error:
|
||||||
|
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||||
|
model_calls += 1
|
||||||
|
selected_existing_ids: set[str] = set()
|
||||||
|
generated_ids: list[str] = []
|
||||||
|
for candidate in candidates:
|
||||||
|
if candidate.existing_id is not None:
|
||||||
|
if candidate.existing_id not in {document.id for document in previous}:
|
||||||
|
raise EvidencePreparationError("unknown_existing_id", source_file)
|
||||||
|
evidence_id = candidate.existing_id
|
||||||
|
selected_existing_ids.add(evidence_id)
|
||||||
|
else:
|
||||||
|
evidence_id = _allocate_evidence_id(candidate.title, reserved_ids)
|
||||||
|
reserved_ids.add(evidence_id)
|
||||||
|
created.append(evidence_id)
|
||||||
|
try:
|
||||||
|
evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash)
|
||||||
|
except ValidationError as error:
|
||||||
|
raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error
|
||||||
|
if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts):
|
||||||
|
raise EvidencePreparationError("supporting_excerpt_missing", source_file)
|
||||||
|
documents_by_id[evidence.id] = evidence
|
||||||
|
generated_ids.append(evidence.id)
|
||||||
|
for prior in previous:
|
||||||
|
if prior.id not in selected_existing_ids:
|
||||||
|
documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash)
|
||||||
|
generated_ids.append(prior.id)
|
||||||
|
source_units[source_file] = tuple(sorted(set(generated_ids)))
|
||||||
|
|
||||||
|
next_manifest = EvidenceManifest(
|
||||||
|
schema_version=1,
|
||||||
|
pipeline_version="evidence-authoring-v1",
|
||||||
|
sources={
|
||||||
|
source_file: ManifestSource(
|
||||||
|
sha256=_source_hash(source_texts[source_file]),
|
||||||
|
units=source_units.get(source_file, ()),
|
||||||
|
)
|
||||||
|
for source_file in sorted(source_texts)
|
||||||
|
},
|
||||||
|
orphans=tuple(sorted(orphaned)),
|
||||||
|
)
|
||||||
|
if not changed and next_manifest == manifest:
|
||||||
|
return EvidencePreparationReport(
|
||||||
|
changed=(),
|
||||||
|
unchanged=tuple(unchanged),
|
||||||
|
created=(),
|
||||||
|
orphaned=tuple(sorted(orphaned)),
|
||||||
|
findings=validate_workspace_evidence(workspace_root).findings,
|
||||||
|
model_calls=0,
|
||||||
|
)
|
||||||
|
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
|
||||||
|
return EvidencePreparationReport(
|
||||||
|
changed=tuple(changed),
|
||||||
|
unchanged=tuple(unchanged),
|
||||||
|
created=tuple(sorted(created)),
|
||||||
|
orphaned=tuple(sorted(orphaned)),
|
||||||
|
findings=findings,
|
||||||
|
model_calls=model_calls,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_manifest() -> EvidenceManifest:
|
||||||
|
return EvidenceManifest(
|
||||||
|
schema_version=1,
|
||||||
|
pipeline_version="evidence-authoring-v1",
|
||||||
|
sources={},
|
||||||
|
orphans=(),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest:
|
||||||
|
if not path.is_file():
|
||||||
|
return _empty_manifest()
|
||||||
|
try:
|
||||||
|
return load_manifest(path)
|
||||||
|
except ValueError as original_error:
|
||||||
|
if not upgrade:
|
||||||
|
raise EvidencePreparationError("pipeline_upgrade_required") from original_error
|
||||||
|
try:
|
||||||
|
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
||||||
|
if not isinstance(raw, dict) or raw.get("schema_version") != 1:
|
||||||
|
raise ValueError("manifest shape")
|
||||||
|
raw["pipeline_version"] = "evidence-authoring-v1"
|
||||||
|
return EvidenceManifest.model_validate(raw)
|
||||||
|
except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error:
|
||||||
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
||||||
|
|
||||||
|
|
||||||
|
def _git_status(workspace_root: Path) -> tuple[str, ...]:
|
||||||
|
result = subprocess.run(
|
||||||
|
["git", "status", "--porcelain"],
|
||||||
|
cwd=workspace_root,
|
||||||
|
check=False,
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
)
|
||||||
|
if result.returncode != 0:
|
||||||
|
raise EvidencePreparationError("canonical_git_worktree_required")
|
||||||
|
return tuple(line for line in result.stdout.splitlines() if line)
|
||||||
|
|
||||||
|
|
||||||
|
def _reject_dirty_authoring_state(
|
||||||
|
workspace_root: Path,
|
||||||
|
git_status: Callable[[Path], tuple[str, ...]],
|
||||||
|
) -> None:
|
||||||
|
for entry in git_status(workspace_root):
|
||||||
|
path = entry[3:] if len(entry) > 3 else entry
|
||||||
|
if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"):
|
||||||
|
raise EvidencePreparationError("authoring_worktree_dirty")
|
||||||
|
|
||||||
|
|
||||||
|
def _load_source_texts(evidence_root: Path) -> dict[str, str]:
|
||||||
|
source_root = evidence_root / "source"
|
||||||
|
if not source_root.is_dir():
|
||||||
|
return {}
|
||||||
|
sources: dict[str, str] = {}
|
||||||
|
for path in sorted(source_root.rglob("*")):
|
||||||
|
if path.is_dir():
|
||||||
|
continue
|
||||||
|
relative = path.relative_to(evidence_root).as_posix()
|
||||||
|
try:
|
||||||
|
validate_source_file(relative)
|
||||||
|
if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
|
||||||
|
raise ValueError("unsafe source")
|
||||||
|
sources[relative] = normalize_source_text(path.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, UnicodeDecodeError, ValueError) as error:
|
||||||
|
raise EvidencePreparationError("source_invalid", relative) from error
|
||||||
|
return sources
|
||||||
|
|
||||||
|
|
||||||
|
def _source_hash(source: str) -> str:
|
||||||
|
return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def _preserve_removed_sources(
|
||||||
|
source_texts: dict[str, str],
|
||||||
|
manifest: EvidenceManifest,
|
||||||
|
documents_by_id: dict[str, CuratedEvidence],
|
||||||
|
source_units: dict[str, tuple[str, ...]],
|
||||||
|
orphaned: set[str],
|
||||||
|
) -> None:
|
||||||
|
for source_file in manifest.sources:
|
||||||
|
if source_file in source_texts:
|
||||||
|
continue
|
||||||
|
unit_ids = source_units.pop(source_file, None)
|
||||||
|
if unit_ids is None:
|
||||||
|
continue
|
||||||
|
for unit in unit_ids:
|
||||||
|
if unit in documents_by_id:
|
||||||
|
orphaned.add(unit)
|
||||||
|
|
||||||
|
|
||||||
|
def _preserve_uniquely_renamed_sources(
|
||||||
|
source_texts: dict[str, str],
|
||||||
|
manifest: EvidenceManifest,
|
||||||
|
documents_by_id: dict[str, CuratedEvidence],
|
||||||
|
source_units: dict[str, tuple[str, ...]],
|
||||||
|
) -> set[str]:
|
||||||
|
renamed_sources: set[str] = set()
|
||||||
|
unmatched_sources = [path for path in source_texts if path not in manifest.sources]
|
||||||
|
missing_sources = [path for path in manifest.sources if path not in source_texts]
|
||||||
|
for old_path in missing_sources:
|
||||||
|
matches = [
|
||||||
|
path for path in unmatched_sources
|
||||||
|
if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256
|
||||||
|
]
|
||||||
|
if len(matches) != 1:
|
||||||
|
continue
|
||||||
|
new_path = matches[0]
|
||||||
|
unit_ids = source_units.pop(old_path, ())
|
||||||
|
source_units[new_path] = unit_ids
|
||||||
|
for unit_id in unit_ids:
|
||||||
|
document = documents_by_id.get(unit_id)
|
||||||
|
if document is None:
|
||||||
|
continue
|
||||||
|
documents_by_id[unit_id] = document.model_copy(update={
|
||||||
|
"provenance": document.provenance.model_copy(update={"source_file": new_path}),
|
||||||
|
})
|
||||||
|
unmatched_sources.remove(new_path)
|
||||||
|
renamed_sources.add(new_path)
|
||||||
|
return renamed_sources
|
||||||
|
|
||||||
|
|
||||||
|
def _candidate_to_evidence(
|
||||||
|
candidate: RestructureCandidate,
|
||||||
|
evidence_id: str,
|
||||||
|
source_file: str,
|
||||||
|
source_hash: str,
|
||||||
|
) -> CuratedEvidence:
|
||||||
|
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
|
||||||
|
data["id"] = evidence_id
|
||||||
|
data["provenance"] = {
|
||||||
|
"source_file": source_file,
|
||||||
|
"source_sha256": source_hash,
|
||||||
|
"supporting_excerpts": candidate.supporting_excerpts,
|
||||||
|
}
|
||||||
|
return CuratedEvidence.model_validate(data)
|
||||||
|
|
||||||
|
|
||||||
|
def _unsupported_unit(
|
||||||
|
evidence: CuratedEvidence,
|
||||||
|
source_file: str,
|
||||||
|
source_hash: str,
|
||||||
|
) -> CuratedEvidence:
|
||||||
|
review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit")
|
||||||
|
review_items += (ReviewItem(
|
||||||
|
code="source_no_longer_supports_unit",
|
||||||
|
message="The current source no longer supports this Evidence unit.",
|
||||||
|
),)
|
||||||
|
return evidence.model_copy(update={
|
||||||
|
"provenance": evidence.provenance.model_copy(update={
|
||||||
|
"source_file": source_file,
|
||||||
|
"source_sha256": source_hash,
|
||||||
|
}),
|
||||||
|
"review_items": review_items,
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str:
|
||||||
|
ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii")
|
||||||
|
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit"
|
||||||
|
base = f"evidence:{slug}"
|
||||||
|
if base not in reserved_ids:
|
||||||
|
return base
|
||||||
|
suffix = 2
|
||||||
|
while f"{base}-{suffix}" in reserved_ids:
|
||||||
|
suffix += 1
|
||||||
|
return f"{base}-{suffix}"
|
||||||
|
|
||||||
|
|
||||||
|
def _stage_and_apply_authoring_tree(
|
||||||
|
workspace_root: Path,
|
||||||
|
documents_by_id: dict[str, CuratedEvidence],
|
||||||
|
manifest: EvidenceManifest,
|
||||||
|
) -> tuple[ValidationFinding, ...]:
|
||||||
|
evidence_root = workspace_root / "evidence"
|
||||||
|
with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary:
|
||||||
|
staged_workspace = Path(temporary) / "workspace"
|
||||||
|
staged_evidence = staged_workspace / "evidence"
|
||||||
|
if evidence_root.exists():
|
||||||
|
shutil.copytree(evidence_root, staged_evidence, symlinks=True)
|
||||||
|
else:
|
||||||
|
staged_evidence.mkdir(parents=True)
|
||||||
|
staged_curated = staged_evidence / "curated"
|
||||||
|
if staged_curated.exists():
|
||||||
|
shutil.rmtree(staged_curated)
|
||||||
|
staged_curated.mkdir()
|
||||||
|
for evidence in documents_by_id.values():
|
||||||
|
destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md"
|
||||||
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
destination.write_text(dump_curated_markdown(evidence), encoding="utf-8")
|
||||||
|
(staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8")
|
||||||
|
validation = validate_workspace_evidence(staged_workspace)
|
||||||
|
if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings):
|
||||||
|
raise EvidencePreparationError("staged_authoring_state_invalid")
|
||||||
|
backup = Path(temporary) / "previous-evidence"
|
||||||
|
if evidence_root.exists():
|
||||||
|
os.replace(evidence_root, backup)
|
||||||
|
try:
|
||||||
|
os.replace(staged_evidence, evidence_root)
|
||||||
|
except OSError:
|
||||||
|
if backup.exists():
|
||||||
|
os.replace(backup, evidence_root)
|
||||||
|
raise
|
||||||
|
return validation.findings
|
||||||
|
|||||||
Reference in New Issue
Block a user