feat(evidence): prepare curated evidence incrementally

This commit is contained in:
2026-08-24 20:08:27 +02:00
parent 6176410f42
commit f5c7cc6198
10 changed files with 899 additions and 7 deletions
@@ -0,0 +1,15 @@
# Evidence authoring response contract
You receive exactly one normalized Source Evidence request. Return one JSON object with
only a `candidates` array, without Markdown fences or explanatory text.
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
otherwise omit it. Preserve prior reviewed wording when it is still supported. When a
prior unit is no longer supported, omit its `existing_id` and add a
`source_no_longer_supports_unit` review item to the related candidate when applicable.
Every candidate must match schema version 1, use one of the eight declared Evidence
kinds, include one to five short exact excerpts copied from `normalized_text`, and add
review items for ambiguities. Do not assign a new canonical ID; the host does that
deterministically. Do not use tools or alter any repository state.
+2
View File
@@ -10,6 +10,8 @@
"decision add",
"decision add-batch",
"decision add-join-set",
"evidence prepare",
"evidence validate",
"memory promote",
"memory save-one",
"memory search",
+1 -1
View File
@@ -32,7 +32,7 @@ def test_typer_tree_matches_the_approved_command_surface():
approved = _approved_surface()
expected = set(approved["maintained"]) | set(approved["enhanced"])
assert len(approved["maintained"]) == 55
assert len(approved["maintained"]) == 57
assert len(approved["enhanced"]) == 8
assert len(approved["erased"]) == 14
assert not (expected & set(approved["erased"]))
+159 -1
View File
@@ -1,4 +1,5 @@
import hashlib
import unicodedata
import pytest
from pydantic import ValidationError
@@ -6,14 +7,24 @@ from pydantic import ValidationError
from tht.evidence import (
CuratedEvidence,
EvidenceManifest,
EvidencePreparationError,
EvidenceRestructurer,
RestructureCandidate,
dump_curated_markdown,
dump_manifest,
load_curated_tree,
load_manifest,
prepare_workspace_evidence,
validate_workspace_evidence,
)
def _evidence(source_text: str, *, review_items=()):
normalized_source = unicodedata.normalize(
"NFC", source_text.replace("\r\n", "\n").replace("\r", "\n"),
)
if not normalized_source.endswith("\n"):
normalized_source += "\n"
return CuratedEvidence.model_validate({
"schema_version": 1,
"id": "evidence:fascia-pediatrica",
@@ -24,7 +35,7 @@ def _evidence(source_text: str, *, review_items=()):
"language": "it",
"provenance": {
"source_file": "source/domain/patient.md",
"source_sha256": "sha256:" + hashlib.sha256(source_text.encode()).hexdigest(),
"source_sha256": "sha256:" + hashlib.sha256(normalized_source.encode()).hexdigest(),
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
},
"review_items": list(review_items),
@@ -279,3 +290,150 @@ def test_workspace_validation_reports_unreadable_or_oversized_sources(tmp_path,
oversized = validate_workspace_evidence(tmp_path)
assert [finding.code for finding in oversized.findings] == ["source_oversized"]
def test_workspace_validation_rejects_a_source_symlink(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
outside = tmp_path / "outside.md"
outside.write_text(source_text, encoding="utf-8")
source_path.unlink()
source_path.symlink_to(outside)
report = validate_workspace_evidence(tmp_path)
assert [finding.code for finding in report.findings] == ["source_unsafe"]
class _Restructurer(EvidenceRestructurer):
def __init__(self, candidates):
self.candidates = candidates
self.requests = []
def restructure(self, request):
self.requests.append(request)
return tuple(self.candidates)
def _candidate(*, title="Fascia pediatrica", existing_id=None):
return RestructureCandidate.model_validate({
"schema_version": 1,
"existing_id": existing_id,
"title": title,
"kind": "domain",
"purposes": ["disambiguation"],
"applies_to": {"concepts": ["fascia pediatrica"]},
"language": "it",
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
"review_items": [],
"payload": {"rule": "La fascia pediatrica comprende i minori."},
})
def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
source_path.write_text(source_text + " La regola è revisionata.\n", encoding="utf-8")
restructurer = _Restructurer([_candidate(existing_id="evidence:fascia-pediatrica")])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 1
assert report.changed == ("source/domain/patient.md",)
assert report.created == ()
assert len(restructurer.requests) == 1
assert restructurer.requests[0].previous_units[0].id == "evidence:fascia-pediatrica"
assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
original = curated_path.read_text(encoding="utf-8")
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.unchanged == ("source/domain/patient.md",)
assert curated_path.read_text(encoding="utf-8") == original
assert restructurer.requests == []
def test_prepare_rejects_dirty_curated_state_before_model_call(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
restructurer = _Restructurer([])
with pytest.raises(EvidencePreparationError, match="authoring_worktree_dirty"):
prepare_workspace_evidence(
tmp_path,
restructurer=restructurer,
git_status=lambda _: (" M evidence/curated/domain/patient.md",),
)
assert restructurer.requests == []
def test_prepare_retains_units_from_a_removed_source_as_blocking_orphans(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
(tmp_path / "evidence" / "source" / "domain" / "patient.md").unlink()
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.orphaned == ("evidence:fascia-pediatrica",)
assert (tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md").is_file()
assert [finding.code for finding in report.findings] == ["orphaned_unit"]
def test_prepare_preserves_ids_without_orphans_for_a_unique_source_rename(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_root = tmp_path / "evidence" / "source" / "domain"
(source_root / "patient.md").rename(source_root / "patient-age.md")
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 0
assert report.orphaned == ()
assert report.unchanged == ("source/domain/patient-age.md",)
assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_rejects_an_unknown_existing_id_without_writing_the_batch(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
source_path.write_text(source_text + " Revisione.\n", encoding="utf-8")
original = (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8")
restructurer = _Restructurer([_candidate(existing_id="evidence:not-supplied")])
with pytest.raises(EvidencePreparationError, match="unknown_existing_id"):
prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert (tmp_path / "evidence" / "manifest.yaml").read_text(encoding="utf-8") == original
def test_prepare_marks_an_omitted_prior_unit_for_human_review(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
(tmp_path / "evidence" / "source" / "domain" / "patient.md").write_text(
"La classificazione pediatrica non è documentata.\n", encoding="utf-8",
)
restructurer = _Restructurer([])
report = prepare_workspace_evidence(tmp_path, restructurer=restructurer, git_status=lambda _: ())
assert report.model_calls == 1
assert [finding.code for finding in report.findings] == [
"supporting_excerpt_missing", "unresolved_review_item",
]
retained = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert retained.review_items[0].code == "source_no_longer_supports_unit"
+40
View File
@@ -0,0 +1,40 @@
import json
from typer.testing import CliRunner
from tht.cli import app
def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
result = CliRunner().invoke(app, ["evidence", "--help"])
assert result.exit_code == 0
assert "prepare" in result.output
assert "validate" in result.output
def test_evidence_validate_json_is_pristine_and_reports_review_required(monkeypatch, tmp_path):
from tht.cli import evidence_cmd
from tht.evidence import ValidationFinding, ValidationReport
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
monkeypatch.setattr(evidence_cmd, "validate_workspace_evidence", lambda root: ValidationReport((
ValidationFinding("error", "unresolved_review_item", "curated/domain/x.md", "Review needed."),
)))
result = CliRunner().invoke(app, ["evidence", "validate", str(tmp_path), "--json"])
assert result.exit_code == 3
assert result.stderr == ""
assert json.loads(result.stdout) == {
"findings": [{
"code": "unresolved_review_item",
"message": "Review needed.",
"path": "curated/domain/x.md",
"severity": "error",
}],
"operation": "evidence_validate",
"publishable": False,
"schemaVersion": 1,
"status": "review_required",
}
@@ -0,0 +1,80 @@
import json
from types import SimpleNamespace
import pytest
from tht.evidence import (
EvidencePreparationError,
PiEvidenceRestructurer,
RestructureRequest,
)
def _request():
return RestructureRequest.model_validate({
"source_file": "source/domain/patient.md",
"source_sha256": "sha256:" + "a" * 64,
"normalized_text": "I pazienti sotto i 18 anni sono pediatrici.\n",
})
def _candidate():
return {
"schema_version": 1,
"title": "Fascia pediatrica",
"kind": "domain",
"purposes": ["disambiguation"],
"applies_to": {"concepts": ["fascia pediatrica"]},
"language": "it",
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
"review_items": [],
"payload": {"rule": "La fascia pediatrica comprende i minori."},
}
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
from tht.evidence import authoring
calls = []
def run(argv, **kwargs):
calls.append((argv, kwargs))
kwargs["stdout"].write(json.dumps({"candidates": [_candidate()]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
candidates = restructurer.restructure(_request())
assert candidates[0].title == "Fascia pediatrica"
argv, kwargs = calls[0]
assert argv[:8] == [
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
]
assert "--skill" in argv
assert any(argument.startswith("@") for argument in argv)
assert kwargs["timeout"] == 12
assert kwargs["shell"] is False
assert kwargs["stderr"] is authoring.subprocess.DEVNULL
@pytest.mark.parametrize("result", [
SimpleNamespace(returncode=1, stdout="secret", stderr="secret"),
SimpleNamespace(returncode=0, stdout="not-json", stderr=""),
])
def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path, monkeypatch, result):
from tht.evidence import authoring
def run(*args, **kwargs):
kwargs["stdout"].write(result.stdout)
return result
monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
with pytest.raises(EvidencePreparationError) as error:
restructurer.restructure(_request())
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
assert "secret" not in str(error.value)
+2
View File
@@ -42,6 +42,7 @@ from tht.cli.datamart_cmd import datamart_app
from tht.cli.db_cmd import db_app
from tht.cli.decision_cmd import decision_app
from tht.cli.doctor_cmd import doctor
from tht.cli.evidence_cmd import evidence_app
from tht.cli.memory_cmd import memory_app
from tht.cli.ollama_cmd import ollama_app
from tht.cli.phase_cmd import phase_app
@@ -54,6 +55,7 @@ from tht.cli.vector_cmd import vector_app
app.add_typer(phase_app, name="phase")
app.add_typer(preprocess_app, name="preprocess")
app.add_typer(evidence_app, name="evidence")
app.add_typer(config_app, name="config")
app.add_typer(schema_app, name="schema")
app.add_typer(session_app, name="session")
+102 -2
View File
@@ -1,5 +1,105 @@
"""Curator-facing Evidence authoring commands, separate from runtime preprocessing."""
from __future__ import annotations
import json
import os
import subprocess
from pathlib import Path
from typing import Annotated
import typer
from tht.evidence import (
EvidencePreparationError,
PiEvidenceRestructurer,
prepare_workspace_evidence,
validate_workspace_evidence,
)
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
def evidence_root(cfg) -> Path:
return cfg.paths.artifacts / "evidence"
def _canonical_worktree(workspace_root: Path) -> Path:
if workspace_root.is_symlink():
raise typer.BadParameter("workspace-root must not be a symlink")
root = workspace_root.resolve()
result = subprocess.run(
["git", "rev-parse", "--show-toplevel"],
cwd=root,
check=False,
capture_output=True,
text=True,
)
if result.returncode != 0 or Path(result.stdout.strip()).resolve() != root:
raise typer.BadParameter("workspace-root must be the root of a canonical Git worktree")
return root
def _emit(payload: dict[str, object], json_output: bool) -> None:
if json_output:
typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True))
return
for key, value in payload.items():
typer.echo(f"{key}: {value}")
def _preparation_payload(report) -> dict[str, object]:
return {
"schemaVersion": 1,
"operation": "evidence_prepare",
"status": "prepared",
"changed": list(report.changed),
"unchanged": list(report.unchanged),
"created": list(report.created),
"orphaned": list(report.orphaned),
"findings": _findings_payload(report.findings),
"modelCalls": report.model_calls,
}
def _findings_payload(findings) -> list[dict[str, object]]:
return [finding.__dict__ for finding in findings]
@evidence_app.command("prepare")
def prepare_cmd(
workspace_root: Path,
upgrade: Annotated[bool, typer.Option(help="Reprocess every source with the installed pipeline.")] = False,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Prepare changed Source Evidence without committing or publishing it."""
root = _canonical_worktree(workspace_root)
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
try:
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
except EvidencePreparationError as error:
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
raise typer.Exit(code=1) from error
_emit(_preparation_payload(report), json_output)
if report.findings:
raise typer.Exit(code=3)
@evidence_app.command("validate")
def validate_cmd(
workspace_root: Path,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Validate the authoring tree without writing, committing, or publishing it."""
root = _canonical_worktree(workspace_root)
report = validate_workspace_evidence(root)
payload = {
"schemaVersion": 1,
"operation": "evidence_validate",
"status": "valid" if report.publishable else "review_required",
"publishable": report.publishable,
"findings": _findings_payload(report.findings),
}
_emit(payload, json_output)
if report.publishable:
return
if all(finding.code in {"orphaned_unit", "unresolved_review_item"} for finding in report.findings):
raise typer.Exit(code=3)
raise typer.Exit(code=1)
+14
View File
@@ -3,10 +3,17 @@
from tht.evidence.acquisition import acquire, discover
from tht.evidence.authoring import (
EvidenceManifest,
EvidencePreparationError,
EvidencePreparationReport,
EvidenceRestructurer,
PiEvidenceRestructurer,
RestructureCandidate,
RestructureRequest,
ValidationFinding,
ValidationReport,
dump_manifest,
load_manifest,
prepare_workspace_evidence,
validate_workspace_evidence,
)
from tht.evidence.canonical import (
@@ -45,9 +52,15 @@ __all__ = [
"CuratedEvidence",
"EvidenceEmbedder",
"EvidenceManifest",
"EvidencePreparationError",
"EvidencePreparationReport",
"EvidenceRestructurer",
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"PiEvidenceRestructurer",
"RestructureCandidate",
"RestructureRequest",
"SourceObject",
"ValidationFinding",
"ValidationReport",
@@ -64,6 +77,7 @@ __all__ = [
"load_manifest",
"normalize_aware_datetime",
"parse_curated_markdown",
"prepare_workspace_evidence",
"project_session",
"resolve_citation",
"validate_corpus_workspace",
+484 -3
View File
@@ -3,17 +3,29 @@
from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import subprocess
import tempfile
import unicodedata
from collections.abc import Callable
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
from typing import Literal, Protocol
import yaml
from pydantic import ValidationError, field_validator, model_validator
from pydantic import Field, ValidationError, field_validator, model_validator
from tht.evidence.canonical import (
CuratedEvidence,
EvidenceKind,
EvidencePurpose,
EvidenceScope,
ReviewItem,
StrictModel,
dump_curated_markdown,
is_evidence_id,
load_curated_tree,
validate_source_file,
@@ -39,6 +51,128 @@ class ValidationReport:
return not any(finding.severity == "error" for finding in self.findings)
class EvidencePreparationError(RuntimeError):
"""A bounded authoring failure that is safe to report to a curator."""
def __init__(self, code: str, source_file: str | None = None) -> None:
self.code = code
self.source_file = source_file
message = code if source_file is None else f"{code}: {source_file}"
super().__init__(message)
class RestructureRequest(StrictModel):
"""The deterministic input passed to exactly one model call for one source."""
source_file: str
source_sha256: str
normalized_text: str
previous_units: tuple[CuratedEvidence, ...] = ()
@field_validator("source_file")
@classmethod
def _validate_source_file(cls, value: str) -> str:
return validate_source_file(value)
class RestructureCandidate(StrictModel):
"""A model proposal before the host allocates or reuses its stable ID."""
schema_version: Literal[1]
existing_id: str | None = None
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
supporting_excerpts: tuple[str, ...]
review_items: tuple[ReviewItem, ...] = ()
payload: dict[str, object]
@field_validator("existing_id")
@classmethod
def _validate_existing_id(cls, value: str | None) -> str | None:
if value is not None and not is_evidence_id(value):
raise ValueError("existing_id must use the evidence:<slug> form")
return value
class EvidenceRestructurer(Protocol):
"""Boundary for the ephemeral, read-only model restructuring call."""
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
def __init__(
self,
pi_executable: str,
skill_path: Path,
*,
timeout_seconds: int = 120,
) -> None:
self._pi_executable = pi_executable
self._skill_path = skill_path
self._timeout_seconds = timeout_seconds
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
request_path = Path(temporary) / "request.json"
request_path.write_text(
json.dumps(request.model_dump(mode="json"), ensure_ascii=False),
encoding="utf-8",
)
argv = [
self._pi_executable,
"--mode", "text",
"--print",
"--no-session",
"--no-tools",
"--no-extensions",
"--no-context-files",
"--skill", str(self._skill_path),
f"@{request_path}",
"Return only the JSON object required by the Evidence authoring skill.",
]
try:
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
result = subprocess.run(
argv,
check=False,
stdout=response,
stderr=subprocess.DEVNULL,
timeout=self._timeout_seconds,
shell=False,
)
if response.tell() > 1024 * 1024:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file)
response.seek(0)
stdout = response.read()
except (OSError, subprocess.TimeoutExpired) as error:
raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error
if result.returncode != 0:
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
try:
raw = json.loads(stdout)
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
raise ValueError("response shape")
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
@dataclass(frozen=True)
class EvidencePreparationReport:
changed: tuple[str, ...]
unchanged: tuple[str, ...]
created: tuple[str, ...]
orphaned: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
model_calls: int
class ManifestSource(StrictModel):
sha256: str
units: tuple[str, ...]
@@ -183,6 +317,10 @@ def _validate_manifest_source(
findings: list[ValidationFinding],
) -> str | None:
path = evidence_root / source_path
if path.is_symlink():
findings.append(ValidationFinding("error", "source_unsafe", source_path,
"The manifest source must be a regular file below the Evidence tree."))
return None
if not path.is_file():
findings.append(ValidationFinding("error", "source_missing", source_path,
"The manifest source does not exist."))
@@ -192,7 +330,7 @@ def _validate_manifest_source(
"The manifest source exceeds the authoring size limit."))
return None
try:
source = _normalize(path.read_text(encoding="utf-8"))
source = normalize_source_text(path.read_text(encoding="utf-8"))
except UnicodeDecodeError:
findings.append(ValidationFinding("error", "source_unreadable", source_path,
"The manifest source is not valid UTF-8."))
@@ -207,6 +345,8 @@ def _validate_manifest_source(
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
if evidence.id in manifest.orphans:
return []
path = evidence.provenance.source_file
findings: list[ValidationFinding] = []
manifest_source = manifest.sources.get(path)
@@ -254,3 +394,344 @@ def _validate_unit(
def _normalize(text: str) -> str:
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
def normalize_source_text(text: str) -> str:
"""Normalize Source Evidence mechanically without changing its meaning."""
normalized = _normalize(text)
return normalized if normalized.endswith("\n") else normalized + "\n"
def prepare_workspace_evidence(
workspace_root: Path,
*,
restructurer: EvidenceRestructurer,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
upgrade: bool = False,
) -> EvidencePreparationReport:
"""Prepare all changed Source Evidence without publishing or committing it.
Every deterministic operation happens locally. A changed source receives exactly
one call through ``restructurer``; all model results validate before the staged
authoring tree replaces the current one.
"""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
manifest_path = evidence_root / "manifest.yaml"
try:
manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade)
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
source_texts = _load_source_texts(evidence_root)
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
source_units = {
source_file: tuple(source.units)
for source_file, source in manifest.sources.items()
}
orphaned = set(manifest.orphans)
created: list[str] = []
changed: list[str] = []
unchanged: list[str] = []
model_calls = 0
renamed_sources = _preserve_uniquely_renamed_sources(
source_texts, manifest, documents_by_id, source_units,
)
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
reserved_ids = set(documents_by_id)
for source_file in sorted(source_texts):
source_text = source_texts[source_file]
source_hash = _source_hash(source_text)
previous = tuple(
documents_by_id[unit]
for unit in source_units.get(source_file, ())
if unit in documents_by_id
)
manifest_source = manifest.sources.get(source_file)
if not upgrade and (source_file in renamed_sources or (
manifest_source is not None and manifest_source.sha256 == source_hash
)):
unchanged.append(source_file)
continue
changed.append(source_file)
request = RestructureRequest(
source_file=source_file,
source_sha256=source_hash,
normalized_text=source_text,
previous_units=previous,
)
try:
candidates = restructurer.restructure(request)
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
model_calls += 1
selected_existing_ids: set[str] = set()
generated_ids: list[str] = []
for candidate in candidates:
if candidate.existing_id is not None:
if candidate.existing_id not in {document.id for document in previous}:
raise EvidencePreparationError("unknown_existing_id", source_file)
evidence_id = candidate.existing_id
selected_existing_ids.add(evidence_id)
else:
evidence_id = _allocate_evidence_id(candidate.title, reserved_ids)
reserved_ids.add(evidence_id)
created.append(evidence_id)
try:
evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash)
except ValidationError as error:
raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error
if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts):
raise EvidencePreparationError("supporting_excerpt_missing", source_file)
documents_by_id[evidence.id] = evidence
generated_ids.append(evidence.id)
for prior in previous:
if prior.id not in selected_existing_ids:
documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash)
generated_ids.append(prior.id)
source_units[source_file] = tuple(sorted(set(generated_ids)))
next_manifest = EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={
source_file: ManifestSource(
sha256=_source_hash(source_texts[source_file]),
units=source_units.get(source_file, ()),
)
for source_file in sorted(source_texts)
},
orphans=tuple(sorted(orphaned)),
)
if not changed and next_manifest == manifest:
return EvidencePreparationReport(
changed=(),
unchanged=tuple(unchanged),
created=(),
orphaned=tuple(sorted(orphaned)),
findings=validate_workspace_evidence(workspace_root).findings,
model_calls=0,
)
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
return EvidencePreparationReport(
changed=tuple(changed),
unchanged=tuple(unchanged),
created=tuple(sorted(created)),
orphaned=tuple(sorted(orphaned)),
findings=findings,
model_calls=model_calls,
)
def _empty_manifest() -> EvidenceManifest:
return EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={},
orphans=(),
)
def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest:
if not path.is_file():
return _empty_manifest()
try:
return load_manifest(path)
except ValueError as original_error:
if not upgrade:
raise EvidencePreparationError("pipeline_upgrade_required") from original_error
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
if not isinstance(raw, dict) or raw.get("schema_version") != 1:
raise ValueError("manifest shape")
raw["pipeline_version"] = "evidence-authoring-v1"
return EvidenceManifest.model_validate(raw)
except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
def _git_status(workspace_root: Path) -> tuple[str, ...]:
result = subprocess.run(
["git", "status", "--porcelain"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
return tuple(line for line in result.stdout.splitlines() if line)
def _reject_dirty_authoring_state(
workspace_root: Path,
git_status: Callable[[Path], tuple[str, ...]],
) -> None:
for entry in git_status(workspace_root):
path = entry[3:] if len(entry) > 3 else entry
if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"):
raise EvidencePreparationError("authoring_worktree_dirty")
def _load_source_texts(evidence_root: Path) -> dict[str, str]:
source_root = evidence_root / "source"
if not source_root.is_dir():
return {}
sources: dict[str, str] = {}
for path in sorted(source_root.rglob("*")):
if path.is_dir():
continue
relative = path.relative_to(evidence_root).as_posix()
try:
validate_source_file(relative)
if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
raise ValueError("unsafe source")
sources[relative] = normalize_source_text(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, ValueError) as error:
raise EvidencePreparationError("source_invalid", relative) from error
return sources
def _source_hash(source: str) -> str:
return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
def _preserve_removed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
orphaned: set[str],
) -> None:
for source_file in manifest.sources:
if source_file in source_texts:
continue
unit_ids = source_units.pop(source_file, None)
if unit_ids is None:
continue
for unit in unit_ids:
if unit in documents_by_id:
orphaned.add(unit)
def _preserve_uniquely_renamed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
) -> set[str]:
renamed_sources: set[str] = set()
unmatched_sources = [path for path in source_texts if path not in manifest.sources]
missing_sources = [path for path in manifest.sources if path not in source_texts]
for old_path in missing_sources:
matches = [
path for path in unmatched_sources
if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256
]
if len(matches) != 1:
continue
new_path = matches[0]
unit_ids = source_units.pop(old_path, ())
source_units[new_path] = unit_ids
for unit_id in unit_ids:
document = documents_by_id.get(unit_id)
if document is None:
continue
documents_by_id[unit_id] = document.model_copy(update={
"provenance": document.provenance.model_copy(update={"source_file": new_path}),
})
unmatched_sources.remove(new_path)
renamed_sources.add(new_path)
return renamed_sources
def _candidate_to_evidence(
candidate: RestructureCandidate,
evidence_id: str,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
"source_sha256": source_hash,
"supporting_excerpts": candidate.supporting_excerpts,
}
return CuratedEvidence.model_validate(data)
def _unsupported_unit(
evidence: CuratedEvidence,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit")
review_items += (ReviewItem(
code="source_no_longer_supports_unit",
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
}),
"review_items": review_items,
})
def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str:
ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii")
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit"
base = f"evidence:{slug}"
if base not in reserved_ids:
return base
suffix = 2
while f"{base}-{suffix}" in reserved_ids:
suffix += 1
return f"{base}-{suffix}"
def _stage_and_apply_authoring_tree(
workspace_root: Path,
documents_by_id: dict[str, CuratedEvidence],
manifest: EvidenceManifest,
) -> tuple[ValidationFinding, ...]:
evidence_root = workspace_root / "evidence"
with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary:
staged_workspace = Path(temporary) / "workspace"
staged_evidence = staged_workspace / "evidence"
if evidence_root.exists():
shutil.copytree(evidence_root, staged_evidence, symlinks=True)
else:
staged_evidence.mkdir(parents=True)
staged_curated = staged_evidence / "curated"
if staged_curated.exists():
shutil.rmtree(staged_curated)
staged_curated.mkdir()
for evidence in documents_by_id.values():
destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md"
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(dump_curated_markdown(evidence), encoding="utf-8")
(staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8")
validation = validate_workspace_evidence(staged_workspace)
if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings):
raise EvidencePreparationError("staged_authoring_state_invalid")
backup = Path(temporary) / "previous-evidence"
if evidence_root.exists():
os.replace(evidence_root, backup)
try:
os.replace(staged_evidence, evidence_root)
except OSError:
if backup.exists():
os.replace(backup, evidence_root)
raise
return validation.findings