feat(evidence): prepare curated evidence incrementally

This commit is contained in:
2026-08-24 20:08:27 +02:00
parent 6176410f42
commit f5c7cc6198
10 changed files with 899 additions and 7 deletions
+14
View File
@@ -3,10 +3,17 @@
from tht.evidence.acquisition import acquire, discover
from tht.evidence.authoring import (
EvidenceManifest,
EvidencePreparationError,
EvidencePreparationReport,
EvidenceRestructurer,
PiEvidenceRestructurer,
RestructureCandidate,
RestructureRequest,
ValidationFinding,
ValidationReport,
dump_manifest,
load_manifest,
prepare_workspace_evidence,
validate_workspace_evidence,
)
from tht.evidence.canonical import (
@@ -45,9 +52,15 @@ __all__ = [
"CuratedEvidence",
"EvidenceEmbedder",
"EvidenceManifest",
"EvidencePreparationError",
"EvidencePreparationReport",
"EvidenceRestructurer",
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"PiEvidenceRestructurer",
"RestructureCandidate",
"RestructureRequest",
"SourceObject",
"ValidationFinding",
"ValidationReport",
@@ -64,6 +77,7 @@ __all__ = [
"load_manifest",
"normalize_aware_datetime",
"parse_curated_markdown",
"prepare_workspace_evidence",
"project_session",
"resolve_citation",
"validate_corpus_workspace",
+484 -3
View File
@@ -3,17 +3,29 @@
from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import subprocess
import tempfile
import unicodedata
from collections.abc import Callable
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
from typing import Literal, Protocol
import yaml
from pydantic import ValidationError, field_validator, model_validator
from pydantic import Field, ValidationError, field_validator, model_validator
from tht.evidence.canonical import (
CuratedEvidence,
EvidenceKind,
EvidencePurpose,
EvidenceScope,
ReviewItem,
StrictModel,
dump_curated_markdown,
is_evidence_id,
load_curated_tree,
validate_source_file,
@@ -39,6 +51,128 @@ class ValidationReport:
return not any(finding.severity == "error" for finding in self.findings)
class EvidencePreparationError(RuntimeError):
"""A bounded authoring failure that is safe to report to a curator."""
def __init__(self, code: str, source_file: str | None = None) -> None:
self.code = code
self.source_file = source_file
message = code if source_file is None else f"{code}: {source_file}"
super().__init__(message)
class RestructureRequest(StrictModel):
"""The deterministic input passed to exactly one model call for one source."""
source_file: str
source_sha256: str
normalized_text: str
previous_units: tuple[CuratedEvidence, ...] = ()
@field_validator("source_file")
@classmethod
def _validate_source_file(cls, value: str) -> str:
return validate_source_file(value)
class RestructureCandidate(StrictModel):
"""A model proposal before the host allocates or reuses its stable ID."""
schema_version: Literal[1]
existing_id: str | None = None
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
supporting_excerpts: tuple[str, ...]
review_items: tuple[ReviewItem, ...] = ()
payload: dict[str, object]
@field_validator("existing_id")
@classmethod
def _validate_existing_id(cls, value: str | None) -> str | None:
if value is not None and not is_evidence_id(value):
raise ValueError("existing_id must use the evidence:<slug> form")
return value
class EvidenceRestructurer(Protocol):
"""Boundary for the ephemeral, read-only model restructuring call."""
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
def __init__(
self,
pi_executable: str,
skill_path: Path,
*,
timeout_seconds: int = 120,
) -> None:
self._pi_executable = pi_executable
self._skill_path = skill_path
self._timeout_seconds = timeout_seconds
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
request_path = Path(temporary) / "request.json"
request_path.write_text(
json.dumps(request.model_dump(mode="json"), ensure_ascii=False),
encoding="utf-8",
)
argv = [
self._pi_executable,
"--mode", "text",
"--print",
"--no-session",
"--no-tools",
"--no-extensions",
"--no-context-files",
"--skill", str(self._skill_path),
f"@{request_path}",
"Return only the JSON object required by the Evidence authoring skill.",
]
try:
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
result = subprocess.run(
argv,
check=False,
stdout=response,
stderr=subprocess.DEVNULL,
timeout=self._timeout_seconds,
shell=False,
)
if response.tell() > 1024 * 1024:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file)
response.seek(0)
stdout = response.read()
except (OSError, subprocess.TimeoutExpired) as error:
raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error
if result.returncode != 0:
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
try:
raw = json.loads(stdout)
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
raise ValueError("response shape")
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
@dataclass(frozen=True)
class EvidencePreparationReport:
changed: tuple[str, ...]
unchanged: tuple[str, ...]
created: tuple[str, ...]
orphaned: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
model_calls: int
class ManifestSource(StrictModel):
sha256: str
units: tuple[str, ...]
@@ -183,6 +317,10 @@ def _validate_manifest_source(
findings: list[ValidationFinding],
) -> str | None:
path = evidence_root / source_path
if path.is_symlink():
findings.append(ValidationFinding("error", "source_unsafe", source_path,
"The manifest source must be a regular file below the Evidence tree."))
return None
if not path.is_file():
findings.append(ValidationFinding("error", "source_missing", source_path,
"The manifest source does not exist."))
@@ -192,7 +330,7 @@ def _validate_manifest_source(
"The manifest source exceeds the authoring size limit."))
return None
try:
source = _normalize(path.read_text(encoding="utf-8"))
source = normalize_source_text(path.read_text(encoding="utf-8"))
except UnicodeDecodeError:
findings.append(ValidationFinding("error", "source_unreadable", source_path,
"The manifest source is not valid UTF-8."))
@@ -207,6 +345,8 @@ def _validate_manifest_source(
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
if evidence.id in manifest.orphans:
return []
path = evidence.provenance.source_file
findings: list[ValidationFinding] = []
manifest_source = manifest.sources.get(path)
@@ -254,3 +394,344 @@ def _validate_unit(
def _normalize(text: str) -> str:
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
def normalize_source_text(text: str) -> str:
"""Normalize Source Evidence mechanically without changing its meaning."""
normalized = _normalize(text)
return normalized if normalized.endswith("\n") else normalized + "\n"
def prepare_workspace_evidence(
workspace_root: Path,
*,
restructurer: EvidenceRestructurer,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
upgrade: bool = False,
) -> EvidencePreparationReport:
"""Prepare all changed Source Evidence without publishing or committing it.
Every deterministic operation happens locally. A changed source receives exactly
one call through ``restructurer``; all model results validate before the staged
authoring tree replaces the current one.
"""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
manifest_path = evidence_root / "manifest.yaml"
try:
manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade)
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
source_texts = _load_source_texts(evidence_root)
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
source_units = {
source_file: tuple(source.units)
for source_file, source in manifest.sources.items()
}
orphaned = set(manifest.orphans)
created: list[str] = []
changed: list[str] = []
unchanged: list[str] = []
model_calls = 0
renamed_sources = _preserve_uniquely_renamed_sources(
source_texts, manifest, documents_by_id, source_units,
)
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
reserved_ids = set(documents_by_id)
for source_file in sorted(source_texts):
source_text = source_texts[source_file]
source_hash = _source_hash(source_text)
previous = tuple(
documents_by_id[unit]
for unit in source_units.get(source_file, ())
if unit in documents_by_id
)
manifest_source = manifest.sources.get(source_file)
if not upgrade and (source_file in renamed_sources or (
manifest_source is not None and manifest_source.sha256 == source_hash
)):
unchanged.append(source_file)
continue
changed.append(source_file)
request = RestructureRequest(
source_file=source_file,
source_sha256=source_hash,
normalized_text=source_text,
previous_units=previous,
)
try:
candidates = restructurer.restructure(request)
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
model_calls += 1
selected_existing_ids: set[str] = set()
generated_ids: list[str] = []
for candidate in candidates:
if candidate.existing_id is not None:
if candidate.existing_id not in {document.id for document in previous}:
raise EvidencePreparationError("unknown_existing_id", source_file)
evidence_id = candidate.existing_id
selected_existing_ids.add(evidence_id)
else:
evidence_id = _allocate_evidence_id(candidate.title, reserved_ids)
reserved_ids.add(evidence_id)
created.append(evidence_id)
try:
evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash)
except ValidationError as error:
raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error
if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts):
raise EvidencePreparationError("supporting_excerpt_missing", source_file)
documents_by_id[evidence.id] = evidence
generated_ids.append(evidence.id)
for prior in previous:
if prior.id not in selected_existing_ids:
documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash)
generated_ids.append(prior.id)
source_units[source_file] = tuple(sorted(set(generated_ids)))
next_manifest = EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={
source_file: ManifestSource(
sha256=_source_hash(source_texts[source_file]),
units=source_units.get(source_file, ()),
)
for source_file in sorted(source_texts)
},
orphans=tuple(sorted(orphaned)),
)
if not changed and next_manifest == manifest:
return EvidencePreparationReport(
changed=(),
unchanged=tuple(unchanged),
created=(),
orphaned=tuple(sorted(orphaned)),
findings=validate_workspace_evidence(workspace_root).findings,
model_calls=0,
)
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
return EvidencePreparationReport(
changed=tuple(changed),
unchanged=tuple(unchanged),
created=tuple(sorted(created)),
orphaned=tuple(sorted(orphaned)),
findings=findings,
model_calls=model_calls,
)
def _empty_manifest() -> EvidenceManifest:
return EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={},
orphans=(),
)
def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest:
if not path.is_file():
return _empty_manifest()
try:
return load_manifest(path)
except ValueError as original_error:
if not upgrade:
raise EvidencePreparationError("pipeline_upgrade_required") from original_error
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
if not isinstance(raw, dict) or raw.get("schema_version") != 1:
raise ValueError("manifest shape")
raw["pipeline_version"] = "evidence-authoring-v1"
return EvidenceManifest.model_validate(raw)
except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
def _git_status(workspace_root: Path) -> tuple[str, ...]:
result = subprocess.run(
["git", "status", "--porcelain"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
return tuple(line for line in result.stdout.splitlines() if line)
def _reject_dirty_authoring_state(
workspace_root: Path,
git_status: Callable[[Path], tuple[str, ...]],
) -> None:
for entry in git_status(workspace_root):
path = entry[3:] if len(entry) > 3 else entry
if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"):
raise EvidencePreparationError("authoring_worktree_dirty")
def _load_source_texts(evidence_root: Path) -> dict[str, str]:
source_root = evidence_root / "source"
if not source_root.is_dir():
return {}
sources: dict[str, str] = {}
for path in sorted(source_root.rglob("*")):
if path.is_dir():
continue
relative = path.relative_to(evidence_root).as_posix()
try:
validate_source_file(relative)
if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
raise ValueError("unsafe source")
sources[relative] = normalize_source_text(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, ValueError) as error:
raise EvidencePreparationError("source_invalid", relative) from error
return sources
def _source_hash(source: str) -> str:
return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
def _preserve_removed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
orphaned: set[str],
) -> None:
for source_file in manifest.sources:
if source_file in source_texts:
continue
unit_ids = source_units.pop(source_file, None)
if unit_ids is None:
continue
for unit in unit_ids:
if unit in documents_by_id:
orphaned.add(unit)
def _preserve_uniquely_renamed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
) -> set[str]:
renamed_sources: set[str] = set()
unmatched_sources = [path for path in source_texts if path not in manifest.sources]
missing_sources = [path for path in manifest.sources if path not in source_texts]
for old_path in missing_sources:
matches = [
path for path in unmatched_sources
if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256
]
if len(matches) != 1:
continue
new_path = matches[0]
unit_ids = source_units.pop(old_path, ())
source_units[new_path] = unit_ids
for unit_id in unit_ids:
document = documents_by_id.get(unit_id)
if document is None:
continue
documents_by_id[unit_id] = document.model_copy(update={
"provenance": document.provenance.model_copy(update={"source_file": new_path}),
})
unmatched_sources.remove(new_path)
renamed_sources.add(new_path)
return renamed_sources
def _candidate_to_evidence(
candidate: RestructureCandidate,
evidence_id: str,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
"source_sha256": source_hash,
"supporting_excerpts": candidate.supporting_excerpts,
}
return CuratedEvidence.model_validate(data)
def _unsupported_unit(
evidence: CuratedEvidence,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit")
review_items += (ReviewItem(
code="source_no_longer_supports_unit",
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
}),
"review_items": review_items,
})
def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str:
ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii")
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit"
base = f"evidence:{slug}"
if base not in reserved_ids:
return base
suffix = 2
while f"{base}-{suffix}" in reserved_ids:
suffix += 1
return f"{base}-{suffix}"
def _stage_and_apply_authoring_tree(
workspace_root: Path,
documents_by_id: dict[str, CuratedEvidence],
manifest: EvidenceManifest,
) -> tuple[ValidationFinding, ...]:
evidence_root = workspace_root / "evidence"
with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary:
staged_workspace = Path(temporary) / "workspace"
staged_evidence = staged_workspace / "evidence"
if evidence_root.exists():
shutil.copytree(evidence_root, staged_evidence, symlinks=True)
else:
staged_evidence.mkdir(parents=True)
staged_curated = staged_evidence / "curated"
if staged_curated.exists():
shutil.rmtree(staged_curated)
staged_curated.mkdir()
for evidence in documents_by_id.values():
destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md"
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(dump_curated_markdown(evidence), encoding="utf-8")
(staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8")
validation = validate_workspace_evidence(staged_workspace)
if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings):
raise EvidencePreparationError("staged_authoring_state_invalid")
backup = Path(temporary) / "previous-evidence"
if evidence_root.exists():
os.replace(evidence_root, backup)
try:
os.replace(staged_evidence, evidence_root)
except OSError:
if backup.exists():
os.replace(backup, evidence_root)
raise
return validation.findings