1087 lines
41 KiB
Python
1087 lines
41 KiB
Python
"""Validation for the Git-reviewed Evidence authoring workspace."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import tempfile
|
|
import unicodedata
|
|
from collections.abc import Callable
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
from dataclasses import dataclass
|
|
from difflib import SequenceMatcher
|
|
from pathlib import Path
|
|
from typing import Literal, Protocol
|
|
|
|
import yaml
|
|
from pydantic import Field, ValidationError, field_validator, model_validator
|
|
|
|
from tht.evidence.canonical import (
|
|
CuratedEvidence,
|
|
EvidenceKind,
|
|
EvidencePurpose,
|
|
EvidenceScope,
|
|
ReviewItem,
|
|
StrictModel,
|
|
dump_curated_markdown,
|
|
is_evidence_id,
|
|
load_curated_tree,
|
|
validate_source_file,
|
|
)
|
|
|
|
MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ValidationFinding:
|
|
severity: Literal["error", "warning"]
|
|
code: str
|
|
path: str
|
|
message: str
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ValidationReport:
|
|
findings: tuple[ValidationFinding, ...]
|
|
|
|
@property
|
|
def publishable(self) -> bool:
|
|
return not any(finding.severity == "error" for finding in self.findings)
|
|
|
|
|
|
class EvidencePreparationError(RuntimeError):
|
|
"""A bounded authoring failure that is safe to report to a curator."""
|
|
|
|
def __init__(self, code: str, source_file: str | None = None) -> None:
|
|
self.code = code
|
|
self.source_file = source_file
|
|
message = code if source_file is None else f"{code}: {source_file}"
|
|
super().__init__(message)
|
|
|
|
|
|
class RestructureRequest(StrictModel):
|
|
"""The deterministic input passed to exactly one model call for one source."""
|
|
|
|
source_file: str
|
|
source_sha256: str
|
|
normalized_text: str
|
|
previous_units: tuple[CuratedEvidence, ...] = ()
|
|
|
|
@field_validator("source_file")
|
|
@classmethod
|
|
def _validate_source_file(cls, value: str) -> str:
|
|
return validate_source_file(value)
|
|
|
|
|
|
class RestructureCandidate(StrictModel):
|
|
"""A model proposal before the host allocates or reuses its stable ID."""
|
|
|
|
schema_version: Literal[1]
|
|
existing_id: str | None = None
|
|
title: str
|
|
kind: EvidenceKind
|
|
purposes: tuple[EvidencePurpose, ...]
|
|
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
|
language: str
|
|
supporting_excerpts: tuple[str, ...]
|
|
review_items: tuple[ReviewItem, ...] = ()
|
|
payload: dict[str, object]
|
|
|
|
@field_validator("existing_id")
|
|
@classmethod
|
|
def _validate_existing_id(cls, value: str | None) -> str | None:
|
|
if value is not None and not is_evidence_id(value):
|
|
raise ValueError("existing_id must use the evidence:<slug> form")
|
|
return value
|
|
|
|
|
|
class EvidenceRestructurer(Protocol):
|
|
"""Boundary for the ephemeral, read-only model restructuring call."""
|
|
|
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
|
|
|
|
|
def _excerpt_signature(value: str) -> str:
|
|
value = "\n".join(
|
|
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
|
|
for line in value.splitlines()
|
|
)
|
|
value = re.sub(r"[*_`]", "", value)
|
|
return " ".join(value.split())
|
|
|
|
|
|
def _request_specific_constraints(source_text: str) -> str:
|
|
qualified_column = re.search(
|
|
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
|
|
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
|
|
r"(?![A-Za-z0-9_])",
|
|
source_text,
|
|
)
|
|
if qualified_column is None:
|
|
return (
|
|
"This request contains no literal schema.table.column identifier, so you "
|
|
"must not use kind enum or formula. Use domain instead and add a review "
|
|
"item when a missing qualification prevents the more specific kind."
|
|
)
|
|
return "Apply the Evidence authoring response contract exactly."
|
|
|
|
|
|
def _restore_exact_source_excerpts(
|
|
candidate: RestructureCandidate,
|
|
source_text: str,
|
|
) -> RestructureCandidate:
|
|
source_lines = source_text.splitlines()
|
|
source_spans: list[str] = []
|
|
for start, first_line in enumerate(source_lines):
|
|
if not first_line:
|
|
continue
|
|
for width in range(1, 6):
|
|
selected = source_lines[start:start + width]
|
|
if len(selected) != width or not selected[-1]:
|
|
continue
|
|
span = "\n".join(selected)
|
|
if len(span) <= 1000:
|
|
source_spans.append(span)
|
|
restored: list[str] = []
|
|
reconciled = False
|
|
for excerpt in candidate.supporting_excerpts:
|
|
if excerpt in source_text:
|
|
restored.append(excerpt)
|
|
continue
|
|
signature = _excerpt_signature(excerpt)
|
|
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
|
|
if len(matches) == 1:
|
|
restored.append(matches[0])
|
|
continue
|
|
ranked = sorted(
|
|
(
|
|
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
|
|
for line in source_spans
|
|
),
|
|
reverse=True,
|
|
)
|
|
best_score = ranked[0][0] if ranked else 0.0
|
|
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
|
|
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
|
|
restored.append(ranked[0][1])
|
|
reconciled = True
|
|
else:
|
|
restored.append(excerpt)
|
|
review_items = candidate.review_items
|
|
if reconciled and not any(
|
|
item.code == "supporting_excerpt_reconciled" for item in review_items
|
|
):
|
|
review_items = (*review_items, ReviewItem(
|
|
code="supporting_excerpt_reconciled",
|
|
message=(
|
|
"A model excerpt was reconciled to a unique exact source line; "
|
|
"confirm that the restored quotation supports this unit."
|
|
),
|
|
field="supporting_excerpts",
|
|
))
|
|
return candidate.model_copy(update={
|
|
"supporting_excerpts": tuple(restored),
|
|
"review_items": review_items,
|
|
})
|
|
|
|
|
|
class PiEvidenceRestructurer:
|
|
"""Invoke Pi once, without tools or session state, for one changed source."""
|
|
|
|
def __init__(
|
|
self,
|
|
pi_executable: str,
|
|
skill_path: Path,
|
|
*,
|
|
timeout_seconds: int = 300,
|
|
) -> None:
|
|
self._pi_executable = pi_executable
|
|
self._skill_path = skill_path
|
|
self._json_mode_extension = (
|
|
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
|
|
)
|
|
self._timeout_seconds = timeout_seconds
|
|
|
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
|
try:
|
|
system_prompt = self._skill_path.read_text(encoding="utf-8")
|
|
except (OSError, UnicodeError) as error:
|
|
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
|
|
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
|
request_path = Path(temporary) / "request.json"
|
|
request_path.write_text(
|
|
json.dumps(request.model_dump(mode="json"), ensure_ascii=False),
|
|
encoding="utf-8",
|
|
)
|
|
argv = [
|
|
self._pi_executable,
|
|
"--mode", "text",
|
|
"--print",
|
|
"--no-session",
|
|
"--no-tools",
|
|
"--no-extensions",
|
|
"--no-context-files",
|
|
"--no-skills",
|
|
"--extension", str(self._json_mode_extension),
|
|
]
|
|
for option, environment_name in (
|
|
("--provider", "PI_PROVIDER"),
|
|
("--model", "PI_MODEL"),
|
|
("--thinking", "PI_THINKING"),
|
|
):
|
|
value = os.environ.get(environment_name)
|
|
if value:
|
|
argv.extend((option, value))
|
|
argv.extend((
|
|
"--system-prompt", system_prompt,
|
|
f"@{request_path}",
|
|
(
|
|
"Return only the JSON object required by the Evidence authoring skill. "
|
|
+ _request_specific_constraints(request.normalized_text)
|
|
),
|
|
))
|
|
try:
|
|
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
|
result = subprocess.run(
|
|
argv,
|
|
check=False,
|
|
stdout=response,
|
|
stderr=subprocess.DEVNULL,
|
|
timeout=self._timeout_seconds,
|
|
shell=False,
|
|
)
|
|
if response.tell() > 1024 * 1024:
|
|
raise EvidencePreparationError("pi_restructure_invalid", request.source_file)
|
|
response.seek(0)
|
|
stdout = response.read()
|
|
except (OSError, subprocess.TimeoutExpired) as error:
|
|
raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error
|
|
if result.returncode != 0:
|
|
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
|
try:
|
|
raw = json.loads(stdout)
|
|
if isinstance(raw, dict) and set(raw) == {"candidates"}:
|
|
candidates = raw["candidates"]
|
|
elif isinstance(raw, list):
|
|
candidates = raw
|
|
elif isinstance(raw, dict):
|
|
candidates = [raw]
|
|
else:
|
|
raise ValueError("response shape")
|
|
if not isinstance(candidates, list):
|
|
raise TypeError("response shape")
|
|
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
|
|
return tuple(
|
|
_restore_exact_source_excerpts(candidate, request.normalized_text)
|
|
for candidate in parsed
|
|
)
|
|
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
|
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class EvidencePreparationReport:
|
|
changed: tuple[str, ...]
|
|
unchanged: tuple[str, ...]
|
|
created: tuple[str, ...]
|
|
orphaned: tuple[str, ...]
|
|
findings: tuple[ValidationFinding, ...]
|
|
model_calls: int
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class EvidenceMigrationReport:
|
|
"""Result of a deterministic Curated Evidence presentation-format migration."""
|
|
|
|
migrated: tuple[str, ...]
|
|
unchanged: tuple[str, ...]
|
|
findings: tuple[ValidationFinding, ...]
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class EvidenceResolutionReport:
|
|
"""The result of one curator-directed Evidence resolution."""
|
|
|
|
action: Literal["retired", "relinked"]
|
|
evidence_id: str
|
|
source_file: str | None
|
|
findings: tuple[ValidationFinding, ...]
|
|
|
|
|
|
class ManifestSource(StrictModel):
|
|
sha256: str
|
|
units: tuple[str, ...]
|
|
|
|
@field_validator("sha256")
|
|
@classmethod
|
|
def _validate_sha256(cls, value: str) -> str:
|
|
if not value.startswith("sha256:") or len(value) != 71:
|
|
raise ValueError("sha256 must be a sha256 digest")
|
|
try:
|
|
int(value.removeprefix("sha256:"), 16)
|
|
except ValueError as error:
|
|
raise ValueError("sha256 must be a sha256 digest") from error
|
|
return value
|
|
|
|
@field_validator("units")
|
|
@classmethod
|
|
def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
if any(not is_evidence_id(unit) for unit in value):
|
|
raise ValueError("units must use stable evidence identifiers")
|
|
return value
|
|
|
|
|
|
class EvidenceManifest(StrictModel):
|
|
schema_version: Literal[1]
|
|
pipeline_version: Literal["evidence-authoring-v1"]
|
|
sources: dict[str, ManifestSource]
|
|
orphans: tuple[str, ...]
|
|
|
|
@field_validator("sources")
|
|
@classmethod
|
|
def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]:
|
|
for source_path in value:
|
|
validate_source_file(source_path)
|
|
return value
|
|
|
|
@field_validator("orphans")
|
|
@classmethod
|
|
def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
if any(not is_evidence_id(unit) for unit in value):
|
|
raise ValueError("orphans must use stable evidence identifiers")
|
|
return value
|
|
|
|
@model_validator(mode="after")
|
|
def _validate_unit_membership(self) -> EvidenceManifest:
|
|
seen: set[str] = set()
|
|
for source in self.sources.values():
|
|
duplicate = seen.intersection(source.units)
|
|
if duplicate:
|
|
raise ValueError("a unit may belong to only one source")
|
|
seen.update(source.units)
|
|
if len(set(self.orphans)) != len(self.orphans):
|
|
raise ValueError("orphans must be unique")
|
|
return self
|
|
|
|
|
|
def load_manifest(path: Path) -> EvidenceManifest:
|
|
"""Load the managed, versioned authoring manifest."""
|
|
try:
|
|
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
except (OSError, UnicodeDecodeError, yaml.YAMLError) as error:
|
|
raise ValueError("manifest cannot be read") from error
|
|
try:
|
|
return EvidenceManifest.model_validate(raw)
|
|
except ValidationError as error:
|
|
raise ValueError("manifest is invalid") from error
|
|
|
|
|
|
def dump_manifest(manifest: EvidenceManifest) -> str:
|
|
"""Serialize the manifest deterministically for Git review."""
|
|
return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True)
|
|
|
|
|
|
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
|
|
"""Validate the curated corpus without writing the workspace."""
|
|
evidence_root = workspace_root / "evidence"
|
|
findings: list[ValidationFinding] = []
|
|
manifest_path = evidence_root / "manifest.yaml"
|
|
if not manifest_path.is_file():
|
|
return ValidationReport((ValidationFinding(
|
|
severity="error",
|
|
code="manifest_missing",
|
|
path="manifest.yaml",
|
|
message="The managed Evidence manifest is missing.",
|
|
),))
|
|
try:
|
|
manifest = load_manifest(manifest_path)
|
|
except ValueError:
|
|
return ValidationReport((ValidationFinding(
|
|
severity="error",
|
|
code="manifest_invalid",
|
|
path="manifest.yaml",
|
|
message="The managed Evidence manifest is invalid.",
|
|
),))
|
|
for orphan in manifest.orphans:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="orphaned_unit",
|
|
path="manifest.yaml",
|
|
message=f"Orphaned Evidence {orphan} must be resolved before publication.",
|
|
))
|
|
try:
|
|
documents = load_curated_tree(evidence_root / "curated")
|
|
except (OSError, ValidationError, ValueError):
|
|
return ValidationReport(tuple(findings + [ValidationFinding(
|
|
severity="error",
|
|
code="curated_invalid",
|
|
path="curated",
|
|
message="A Curated Evidence document is invalid or cannot be read.",
|
|
)]))
|
|
source_texts = {
|
|
source_path: _validate_manifest_source(evidence_root, source_path, source, findings)
|
|
for source_path, source in manifest.sources.items()
|
|
}
|
|
seen_ids: set[str] = set()
|
|
for evidence in documents:
|
|
if evidence.id in seen_ids:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="duplicate_evidence_id",
|
|
path="curated",
|
|
message=f"Evidence id {evidence.id} appears more than once.",
|
|
))
|
|
seen_ids.add(evidence.id)
|
|
findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file)))
|
|
for source in manifest.sources.values():
|
|
for unit in source.units:
|
|
if unit not in seen_ids:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="manifest_unit_without_curated",
|
|
path="manifest.yaml",
|
|
message=f"Manifest Evidence {unit} has no Curated document.",
|
|
))
|
|
return ValidationReport(tuple(findings))
|
|
|
|
|
|
def _validate_manifest_source(
|
|
evidence_root: Path,
|
|
source_path: str,
|
|
manifest_source: ManifestSource,
|
|
findings: list[ValidationFinding],
|
|
) -> str | None:
|
|
path = evidence_root / source_path
|
|
if path.is_symlink():
|
|
findings.append(ValidationFinding("error", "source_unsafe", source_path,
|
|
"The manifest source must be a regular file below the Evidence tree."))
|
|
return None
|
|
if not path.is_file():
|
|
findings.append(ValidationFinding("error", "source_missing", source_path,
|
|
"The manifest source does not exist."))
|
|
return None
|
|
if path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
|
|
findings.append(ValidationFinding("error", "source_oversized", source_path,
|
|
"The manifest source exceeds the authoring size limit."))
|
|
return None
|
|
try:
|
|
source = normalize_source_text(path.read_text(encoding="utf-8"))
|
|
except UnicodeDecodeError:
|
|
findings.append(ValidationFinding("error", "source_unreadable", source_path,
|
|
"The manifest source is not valid UTF-8."))
|
|
return None
|
|
digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
|
|
if digest != manifest_source.sha256:
|
|
findings.append(ValidationFinding("error", "source_hash_mismatch", source_path,
|
|
"The manifest hash does not match the normalized source."))
|
|
return source
|
|
|
|
|
|
def _validate_unit(
|
|
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
|
) -> list[ValidationFinding]:
|
|
if evidence.id in manifest.orphans:
|
|
return []
|
|
path = evidence.provenance.source_file
|
|
findings: list[ValidationFinding] = []
|
|
manifest_source = manifest.sources.get(path)
|
|
if manifest_source is None:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="manifest_source_missing",
|
|
path="manifest.yaml",
|
|
message="The manifest does not contain the provenance source.",
|
|
))
|
|
else:
|
|
if manifest_source.sha256 != evidence.provenance.source_sha256:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="manifest_source_hash_mismatch",
|
|
path="manifest.yaml",
|
|
message="The manifest hash does not match the Evidence provenance.",
|
|
))
|
|
if evidence.id not in manifest_source.units:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="manifest_unit_missing",
|
|
path="manifest.yaml",
|
|
message="The manifest does not link the Evidence unit to its source.",
|
|
))
|
|
if source is None:
|
|
return findings
|
|
for excerpt in evidence.provenance.supporting_excerpts:
|
|
if _normalize(excerpt) not in source:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="supporting_excerpt_missing",
|
|
path=path,
|
|
message="A supporting excerpt is absent from the normalized source.",
|
|
))
|
|
for item in evidence.review_items:
|
|
findings.append(ValidationFinding(
|
|
severity="error",
|
|
code="unresolved_review_item",
|
|
path=path,
|
|
message=f"Review item {item.code} must be resolved before publication.",
|
|
))
|
|
return findings
|
|
|
|
|
|
def _normalize(text: str) -> str:
|
|
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
|
|
|
|
|
|
def normalize_source_text(text: str) -> str:
|
|
"""Normalize Source Evidence mechanically without changing its meaning."""
|
|
normalized = _normalize(text)
|
|
return normalized if normalized.endswith("\n") else normalized + "\n"
|
|
|
|
|
|
def prepare_workspace_evidence(
|
|
workspace_root: Path,
|
|
*,
|
|
restructurer: EvidenceRestructurer,
|
|
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
|
upgrade: bool = False,
|
|
max_workers: int = 1,
|
|
) -> EvidencePreparationReport:
|
|
"""Prepare all changed Source Evidence without publishing or committing it.
|
|
|
|
Every deterministic operation happens locally. A changed source receives exactly
|
|
one call through ``restructurer``; all model results validate before the staged
|
|
authoring tree replaces the current one.
|
|
"""
|
|
if not 1 <= max_workers <= 8:
|
|
raise EvidencePreparationError("authoring_workers_invalid")
|
|
workspace_root = workspace_root.resolve()
|
|
evidence_root = workspace_root / "evidence"
|
|
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
|
manifest_path = evidence_root / "manifest.yaml"
|
|
try:
|
|
manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade)
|
|
documents = load_curated_tree(evidence_root / "curated")
|
|
except (OSError, ValidationError, ValueError) as error:
|
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
|
|
|
source_texts = _load_source_texts(evidence_root)
|
|
documents_by_id = {document.id: document for document in documents}
|
|
if len(documents_by_id) != len(documents):
|
|
raise EvidencePreparationError("duplicate_evidence_id")
|
|
source_units = {
|
|
source_file: tuple(source.units)
|
|
for source_file, source in manifest.sources.items()
|
|
}
|
|
orphaned = set(manifest.orphans)
|
|
created: list[str] = []
|
|
changed: list[str] = []
|
|
unchanged: list[str] = []
|
|
model_calls = 0
|
|
|
|
renamed_sources = _preserve_uniquely_renamed_sources(
|
|
source_texts, manifest, documents_by_id, source_units,
|
|
)
|
|
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
|
|
|
requests: list[tuple[str, RestructureRequest]] = []
|
|
for source_file in sorted(source_texts):
|
|
source_text = source_texts[source_file]
|
|
source_hash = _source_hash(source_text)
|
|
previous = tuple(
|
|
documents_by_id[unit]
|
|
for unit in source_units.get(source_file, ())
|
|
if unit in documents_by_id
|
|
)
|
|
manifest_source = manifest.sources.get(source_file)
|
|
if not upgrade and (source_file in renamed_sources or (
|
|
manifest_source is not None and manifest_source.sha256 == source_hash
|
|
)):
|
|
unchanged.append(source_file)
|
|
continue
|
|
changed.append(source_file)
|
|
request = RestructureRequest(
|
|
source_file=source_file,
|
|
source_sha256=source_hash,
|
|
normalized_text=source_text,
|
|
previous_units=previous,
|
|
)
|
|
requests.append((source_file, request))
|
|
|
|
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
|
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
futures = {
|
|
source_file: executor.submit(restructurer.restructure, request)
|
|
for source_file, request in requests
|
|
}
|
|
for source_file, _request in requests:
|
|
try:
|
|
candidates_by_source[source_file] = futures[source_file].result()
|
|
except EvidencePreparationError:
|
|
raise
|
|
except Exception as error:
|
|
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
|
|
|
reserved_ids = set(documents_by_id)
|
|
for source_file, request in requests:
|
|
source_text = source_texts[source_file]
|
|
source_hash = request.source_sha256
|
|
previous = request.previous_units
|
|
candidates = candidates_by_source[source_file]
|
|
model_calls += 1
|
|
selected_existing_ids: set[str] = set()
|
|
generated_ids: list[str] = []
|
|
for candidate in candidates:
|
|
if candidate.existing_id is not None:
|
|
if candidate.existing_id not in {document.id for document in previous}:
|
|
raise EvidencePreparationError("unknown_existing_id", source_file)
|
|
evidence_id = candidate.existing_id
|
|
selected_existing_ids.add(evidence_id)
|
|
else:
|
|
evidence_id = _allocate_evidence_id(candidate.title, reserved_ids)
|
|
reserved_ids.add(evidence_id)
|
|
created.append(evidence_id)
|
|
try:
|
|
evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash)
|
|
except ValidationError as error:
|
|
raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error
|
|
if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts):
|
|
raise EvidencePreparationError("supporting_excerpt_missing", source_file)
|
|
documents_by_id[evidence.id] = evidence
|
|
generated_ids.append(evidence.id)
|
|
for prior in previous:
|
|
if prior.id not in selected_existing_ids:
|
|
documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash)
|
|
generated_ids.append(prior.id)
|
|
source_units[source_file] = tuple(sorted(set(generated_ids)))
|
|
|
|
next_manifest = EvidenceManifest(
|
|
schema_version=1,
|
|
pipeline_version="evidence-authoring-v1",
|
|
sources={
|
|
source_file: ManifestSource(
|
|
sha256=_source_hash(source_texts[source_file]),
|
|
units=source_units.get(source_file, ()),
|
|
)
|
|
for source_file in sorted(source_texts)
|
|
},
|
|
orphans=tuple(sorted(orphaned)),
|
|
)
|
|
if not changed and next_manifest == manifest:
|
|
return EvidencePreparationReport(
|
|
changed=(),
|
|
unchanged=tuple(unchanged),
|
|
created=(),
|
|
orphaned=tuple(sorted(orphaned)),
|
|
findings=validate_workspace_evidence(workspace_root).findings,
|
|
model_calls=0,
|
|
)
|
|
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
|
|
return EvidencePreparationReport(
|
|
changed=tuple(changed),
|
|
unchanged=tuple(unchanged),
|
|
created=tuple(sorted(created)),
|
|
orphaned=tuple(sorted(orphaned)),
|
|
findings=findings,
|
|
model_calls=model_calls,
|
|
)
|
|
|
|
|
|
def migrate_workspace_evidence(
|
|
workspace_root: Path,
|
|
*,
|
|
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
|
) -> EvidenceMigrationReport:
|
|
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
|
|
workspace_root = workspace_root.resolve()
|
|
evidence_root = workspace_root / "evidence"
|
|
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
|
try:
|
|
manifest = load_manifest(evidence_root / "manifest.yaml")
|
|
documents = load_curated_tree(evidence_root / "curated")
|
|
except (OSError, ValidationError, ValueError) as error:
|
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
|
|
|
documents_by_id = {document.id: document for document in documents}
|
|
if len(documents_by_id) != len(documents):
|
|
raise EvidencePreparationError("duplicate_evidence_id")
|
|
migrated = tuple(sorted(
|
|
document.id for document in documents if document.schema_version in {1, 2}
|
|
))
|
|
unchanged = tuple(sorted(
|
|
document.id for document in documents if document.schema_version == 3
|
|
))
|
|
if not migrated:
|
|
return EvidenceMigrationReport(
|
|
migrated=(),
|
|
unchanged=unchanged,
|
|
findings=validate_workspace_evidence(workspace_root).findings,
|
|
)
|
|
upgraded = {
|
|
evidence_id: document.model_copy(update={"schema_version": 3})
|
|
for evidence_id, document in documents_by_id.items()
|
|
}
|
|
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
|
return EvidenceMigrationReport(
|
|
migrated=migrated,
|
|
unchanged=unchanged,
|
|
findings=findings,
|
|
)
|
|
|
|
|
|
def resolve_workspace_evidence(
|
|
workspace_root: Path,
|
|
evidence_id: str,
|
|
*,
|
|
retire: bool = False,
|
|
source: str | Path | None = None,
|
|
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
|
) -> EvidenceResolutionReport:
|
|
"""Retire or relink one Evidence unit as one staged curator update.
|
|
|
|
The operation intentionally neither creates a commit nor publishes anything. Both
|
|
the curated tree and the managed manifest are replaced only after the staged tree
|
|
has been written and structurally validated.
|
|
"""
|
|
if retire == (source is not None):
|
|
raise EvidencePreparationError("resolve_mode_invalid")
|
|
if not is_evidence_id(evidence_id):
|
|
raise EvidencePreparationError("evidence_id_invalid")
|
|
|
|
workspace_root = workspace_root.resolve()
|
|
evidence_root = workspace_root / "evidence"
|
|
_reject_dirty_worktree(workspace_root, git_status or _git_status)
|
|
try:
|
|
manifest = load_manifest(evidence_root / "manifest.yaml")
|
|
documents = load_curated_tree(evidence_root / "curated")
|
|
except (OSError, ValidationError, ValueError) as error:
|
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
|
|
|
documents_by_id = {document.id: document for document in documents}
|
|
if len(documents_by_id) != len(documents):
|
|
raise EvidencePreparationError("duplicate_evidence_id")
|
|
if evidence_id not in documents_by_id:
|
|
raise EvidencePreparationError("evidence_not_found")
|
|
|
|
source_units = {path: tuple(entry.units) for path, entry in manifest.sources.items()}
|
|
orphaned = set(manifest.orphans)
|
|
source_file: str | None = None
|
|
action: Literal["retired", "relinked"]
|
|
|
|
if retire:
|
|
action = "retired"
|
|
del documents_by_id[evidence_id]
|
|
for path, units in tuple(source_units.items()):
|
|
source_units[path] = tuple(unit for unit in units if unit != evidence_id)
|
|
orphaned.discard(evidence_id)
|
|
else:
|
|
action = "relinked"
|
|
assert source is not None
|
|
source_file = _resolve_source_file(evidence_root, source)
|
|
source_text = _load_resolution_source(evidence_root, source_file)
|
|
document = documents_by_id[evidence_id]
|
|
review_items = document.review_items
|
|
if all(_normalize(excerpt) in source_text for excerpt in document.provenance.supporting_excerpts):
|
|
review_items = tuple(
|
|
item for item in review_items if item.code != "source_no_longer_supports_unit"
|
|
)
|
|
documents_by_id[evidence_id] = document.model_copy(update={
|
|
"provenance": document.provenance.model_copy(update={
|
|
"source_file": source_file,
|
|
"source_sha256": _source_hash(source_text),
|
|
}),
|
|
"review_items": review_items,
|
|
})
|
|
for path, units in tuple(source_units.items()):
|
|
source_units[path] = tuple(unit for unit in units if unit != evidence_id)
|
|
source_units[source_file] = tuple(sorted((*source_units.get(source_file, ()), evidence_id)))
|
|
orphaned.discard(evidence_id)
|
|
|
|
sources = {
|
|
path: ManifestSource(
|
|
sha256=(
|
|
_source_hash(_load_resolution_source(evidence_root, path))
|
|
if path == source_file
|
|
else manifest.sources[path].sha256
|
|
),
|
|
units=units,
|
|
)
|
|
for path, units in sorted(source_units.items())
|
|
}
|
|
next_manifest = EvidenceManifest(
|
|
schema_version=1,
|
|
pipeline_version=manifest.pipeline_version,
|
|
sources=sources,
|
|
orphans=tuple(sorted(orphaned)),
|
|
)
|
|
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
|
|
return EvidenceResolutionReport(
|
|
action=action,
|
|
evidence_id=evidence_id,
|
|
source_file=source_file,
|
|
findings=findings,
|
|
)
|
|
|
|
|
|
def _empty_manifest() -> EvidenceManifest:
|
|
return EvidenceManifest(
|
|
schema_version=1,
|
|
pipeline_version="evidence-authoring-v1",
|
|
sources={},
|
|
orphans=(),
|
|
)
|
|
|
|
|
|
def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest:
|
|
if not path.is_file():
|
|
return _empty_manifest()
|
|
try:
|
|
return load_manifest(path)
|
|
except ValueError as original_error:
|
|
if not upgrade:
|
|
raise EvidencePreparationError("pipeline_upgrade_required") from original_error
|
|
try:
|
|
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
if not isinstance(raw, dict) or raw.get("schema_version") != 1:
|
|
raise ValueError("manifest shape")
|
|
raw["pipeline_version"] = "evidence-authoring-v1"
|
|
return EvidenceManifest.model_validate(raw)
|
|
except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error:
|
|
raise EvidencePreparationError("authoring_state_invalid") from error
|
|
|
|
|
|
def _resolve_source_file(evidence_root: Path, source: str | Path) -> str:
|
|
raw_source = source.as_posix() if isinstance(source, Path) else source
|
|
try:
|
|
source_file = validate_source_file(raw_source)
|
|
except ValueError as error:
|
|
raise EvidencePreparationError("source_invalid", raw_source) from error
|
|
path = evidence_root / source_file
|
|
ancestor = evidence_root
|
|
for part in Path(source_file).parts:
|
|
ancestor /= part
|
|
if ancestor.is_symlink():
|
|
raise EvidencePreparationError("source_invalid", source_file)
|
|
if not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
|
|
raise EvidencePreparationError("source_invalid", source_file)
|
|
return source_file
|
|
|
|
|
|
def _load_resolution_source(evidence_root: Path, source_file: str) -> str:
|
|
try:
|
|
return normalize_source_text((evidence_root / source_file).read_text(encoding="utf-8"))
|
|
except (OSError, UnicodeDecodeError) as error:
|
|
raise EvidencePreparationError("source_invalid", source_file) from error
|
|
|
|
|
|
def _git_status(workspace_root: Path) -> tuple[str, ...]:
|
|
result = subprocess.run(
|
|
["git", "status", "--porcelain", "--untracked-files=all"],
|
|
cwd=workspace_root,
|
|
check=False,
|
|
capture_output=True,
|
|
text=True,
|
|
)
|
|
if result.returncode != 0:
|
|
raise EvidencePreparationError("canonical_git_worktree_required")
|
|
prefix_result = subprocess.run(
|
|
["git", "rev-parse", "--show-prefix"],
|
|
cwd=workspace_root,
|
|
check=False,
|
|
capture_output=True,
|
|
text=True,
|
|
)
|
|
if prefix_result.returncode != 0:
|
|
raise EvidencePreparationError("canonical_git_worktree_required")
|
|
prefix = prefix_result.stdout.strip()
|
|
entries: list[str] = []
|
|
for line in result.stdout.splitlines():
|
|
if not line:
|
|
continue
|
|
path = line[3:] if len(line) > 3 else line
|
|
if prefix and path.startswith(prefix):
|
|
line = line[:3] + path.removeprefix(prefix)
|
|
entries.append(line)
|
|
return tuple(entries)
|
|
|
|
|
|
def _reject_dirty_authoring_state(
|
|
workspace_root: Path,
|
|
git_status: Callable[[Path], tuple[str, ...]],
|
|
) -> None:
|
|
for entry in git_status(workspace_root):
|
|
path = entry[3:] if len(entry) > 3 else entry
|
|
if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"):
|
|
raise EvidencePreparationError("authoring_worktree_dirty")
|
|
|
|
|
|
def _reject_dirty_worktree(
|
|
workspace_root: Path,
|
|
git_status: Callable[[Path], tuple[str, ...]],
|
|
) -> None:
|
|
if git_status(workspace_root):
|
|
raise EvidencePreparationError("worktree_dirty")
|
|
|
|
|
|
def _load_source_texts(evidence_root: Path) -> dict[str, str]:
|
|
source_root = evidence_root / "source"
|
|
if not source_root.is_dir():
|
|
return {}
|
|
sources: dict[str, str] = {}
|
|
for path in sorted(source_root.rglob("*")):
|
|
if path.is_dir():
|
|
continue
|
|
relative = path.relative_to(evidence_root).as_posix()
|
|
try:
|
|
validate_source_file(relative)
|
|
if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
|
|
raise ValueError("unsafe source")
|
|
sources[relative] = normalize_source_text(path.read_text(encoding="utf-8"))
|
|
except (OSError, UnicodeDecodeError, ValueError) as error:
|
|
raise EvidencePreparationError("source_invalid", relative) from error
|
|
return sources
|
|
|
|
|
|
def _source_hash(source: str) -> str:
|
|
return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _preserve_removed_sources(
|
|
source_texts: dict[str, str],
|
|
manifest: EvidenceManifest,
|
|
documents_by_id: dict[str, CuratedEvidence],
|
|
source_units: dict[str, tuple[str, ...]],
|
|
orphaned: set[str],
|
|
) -> None:
|
|
for source_file in manifest.sources:
|
|
if source_file in source_texts:
|
|
continue
|
|
unit_ids = source_units.pop(source_file, None)
|
|
if unit_ids is None:
|
|
continue
|
|
for unit in unit_ids:
|
|
if unit in documents_by_id:
|
|
orphaned.add(unit)
|
|
|
|
|
|
def _preserve_uniquely_renamed_sources(
|
|
source_texts: dict[str, str],
|
|
manifest: EvidenceManifest,
|
|
documents_by_id: dict[str, CuratedEvidence],
|
|
source_units: dict[str, tuple[str, ...]],
|
|
) -> set[str]:
|
|
renamed_sources: set[str] = set()
|
|
unmatched_sources = [path for path in source_texts if path not in manifest.sources]
|
|
missing_sources = [path for path in manifest.sources if path not in source_texts]
|
|
for old_path in missing_sources:
|
|
matches = [
|
|
path for path in unmatched_sources
|
|
if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256
|
|
]
|
|
if len(matches) != 1:
|
|
continue
|
|
new_path = matches[0]
|
|
unit_ids = source_units.pop(old_path, ())
|
|
source_units[new_path] = unit_ids
|
|
for unit_id in unit_ids:
|
|
document = documents_by_id.get(unit_id)
|
|
if document is None:
|
|
continue
|
|
documents_by_id[unit_id] = document.model_copy(update={
|
|
"provenance": document.provenance.model_copy(update={"source_file": new_path}),
|
|
})
|
|
unmatched_sources.remove(new_path)
|
|
renamed_sources.add(new_path)
|
|
return renamed_sources
|
|
|
|
|
|
def _candidate_to_evidence(
|
|
candidate: RestructureCandidate,
|
|
evidence_id: str,
|
|
source_file: str,
|
|
source_hash: str,
|
|
) -> CuratedEvidence:
|
|
data = candidate.model_dump(
|
|
mode="json",
|
|
exclude={"schema_version", "existing_id", "supporting_excerpts"},
|
|
)
|
|
data["schema_version"] = 3
|
|
data["id"] = evidence_id
|
|
data["provenance"] = {
|
|
"source_file": source_file,
|
|
"source_sha256": source_hash,
|
|
"supporting_excerpts": candidate.supporting_excerpts,
|
|
}
|
|
return CuratedEvidence.model_validate(data)
|
|
|
|
|
|
def _unsupported_unit(
|
|
evidence: CuratedEvidence,
|
|
source_file: str,
|
|
source_hash: str,
|
|
) -> CuratedEvidence:
|
|
review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit")
|
|
review_items += (ReviewItem(
|
|
code="source_no_longer_supports_unit",
|
|
message="The current source no longer supports this Evidence unit.",
|
|
),)
|
|
return evidence.model_copy(update={
|
|
"schema_version": 3,
|
|
"provenance": evidence.provenance.model_copy(update={
|
|
"source_file": source_file,
|
|
"source_sha256": source_hash,
|
|
}),
|
|
"review_items": review_items,
|
|
})
|
|
|
|
|
|
def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str:
|
|
ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii")
|
|
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit"
|
|
base = f"evidence:{slug}"
|
|
if base not in reserved_ids:
|
|
return base
|
|
suffix = 2
|
|
while f"{base}-{suffix}" in reserved_ids:
|
|
suffix += 1
|
|
return f"{base}-{suffix}"
|
|
|
|
|
|
def _stage_and_apply_authoring_tree(
|
|
workspace_root: Path,
|
|
documents_by_id: dict[str, CuratedEvidence],
|
|
manifest: EvidenceManifest,
|
|
) -> tuple[ValidationFinding, ...]:
|
|
evidence_root = workspace_root / "evidence"
|
|
with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary:
|
|
staged_workspace = Path(temporary) / "workspace"
|
|
staged_evidence = staged_workspace / "evidence"
|
|
if evidence_root.exists():
|
|
shutil.copytree(evidence_root, staged_evidence, symlinks=True)
|
|
else:
|
|
staged_evidence.mkdir(parents=True)
|
|
staged_curated = staged_evidence / "curated"
|
|
if staged_curated.exists():
|
|
shutil.rmtree(staged_curated)
|
|
staged_curated.mkdir()
|
|
for evidence in documents_by_id.values():
|
|
destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md"
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
destination.write_text(dump_curated_markdown(evidence), encoding="utf-8")
|
|
(staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8")
|
|
validation = validate_workspace_evidence(staged_workspace)
|
|
if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings):
|
|
raise EvidencePreparationError("staged_authoring_state_invalid")
|
|
backup = Path(temporary) / "previous-evidence"
|
|
if evidence_root.exists():
|
|
os.replace(evidence_root, backup)
|
|
try:
|
|
os.replace(staged_evidence, evidence_root)
|
|
except OSError:
|
|
if backup.exists():
|
|
os.replace(backup, evidence_root)
|
|
raise
|
|
return validation.findings
|