Files
ThothII/harness/tht/evidence/authoring.py
T
Codex 84084bba37
Publish documentation / publish (push) Successful in 30s
Fix Qwen session tool calls and expose thinking compatibility
2026-09-21 16:23:51 +02:00

1122 lines
43 KiB
Python

"""Validation for the Git-reviewed Evidence authoring workspace."""
from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import subprocess
import tempfile
import unicodedata
from collections.abc import Callable
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass
from difflib import SequenceMatcher
from pathlib import Path
from typing import Literal, Protocol
import yaml
from pydantic import Field, ValidationError, field_validator, model_validator
from tht.evidence.canonical import (
CuratedEvidence,
EvidenceKind,
EvidencePurpose,
EvidenceScope,
ManualEvidenceProvenance,
ReviewItem,
StrictModel,
dump_curated_markdown,
is_evidence_id,
load_curated_tree,
validate_source_file,
)
MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024
@dataclass(frozen=True)
class ValidationFinding:
severity: Literal["error", "warning"]
code: str
path: str
message: str
@dataclass(frozen=True)
class ValidationReport:
findings: tuple[ValidationFinding, ...]
@property
def publishable(self) -> bool:
return not any(finding.severity == "error" for finding in self.findings)
class EvidencePreparationError(RuntimeError):
"""A bounded authoring failure that is safe to report to a curator."""
def __init__(self, code: str, source_file: str | None = None) -> None:
self.code = code
self.source_file = source_file
message = code if source_file is None else f"{code}: {source_file}"
super().__init__(message)
class RestructureRequest(StrictModel):
"""The deterministic input passed to exactly one model call for one source."""
source_file: str
source_sha256: str
normalized_text: str
previous_units: tuple[CuratedEvidence, ...] = ()
@field_validator("source_file")
@classmethod
def _validate_source_file(cls, value: str) -> str:
return validate_source_file(value)
class RestructureCandidate(StrictModel):
"""A model proposal before the host allocates or reuses its stable ID."""
schema_version: Literal[1]
existing_id: str | None = None
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
supporting_excerpts: tuple[str, ...]
review_items: tuple[ReviewItem, ...] = ()
payload: dict[str, object]
@field_validator("existing_id")
@classmethod
def _validate_existing_id(cls, value: str | None) -> str | None:
if value is not None and not is_evidence_id(value):
raise ValueError("existing_id must use the evidence:<slug> form")
return value
class EvidenceRestructurer(Protocol):
"""Boundary for the ephemeral, read-only model restructuring call."""
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
def _excerpt_signature(value: str) -> str:
value = "\n".join(
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
for line in value.splitlines()
)
value = re.sub(r"[*_`]", "", value)
return " ".join(value.split())
def _request_specific_constraints(source_text: str) -> str:
qualified_column = re.search(
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
r"(?![A-Za-z0-9_])",
source_text,
)
if qualified_column is None:
return (
"This request contains no literal schema.table.column identifier, so you "
"must not use kind enum or formula. Use domain instead and add a review "
"item when a missing qualification prevents the more specific kind."
)
return "Apply the Evidence authoring response contract exactly."
def _restore_exact_source_excerpts(
candidate: RestructureCandidate,
source_text: str,
) -> RestructureCandidate:
source_lines = source_text.splitlines()
source_spans: list[str] = []
for start, first_line in enumerate(source_lines):
if not first_line:
continue
for width in range(1, 6):
selected = source_lines[start:start + width]
if len(selected) != width or not selected[-1]:
continue
span = "\n".join(selected)
if len(span) <= 1000:
source_spans.append(span)
restored: list[str] = []
reconciled = False
for excerpt in candidate.supporting_excerpts:
if excerpt in source_text:
restored.append(excerpt)
continue
signature = _excerpt_signature(excerpt)
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
if len(matches) == 1:
restored.append(matches[0])
continue
ranked = sorted(
(
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
for line in source_spans
),
reverse=True,
)
best_score = ranked[0][0] if ranked else 0.0
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
restored.append(ranked[0][1])
reconciled = True
else:
restored.append(excerpt)
review_items = candidate.review_items
if reconciled and not any(
item.code == "supporting_excerpt_reconciled" for item in review_items
):
review_items = (*review_items, ReviewItem(
code="supporting_excerpt_reconciled",
message=(
"A model excerpt was reconciled to a unique exact source line; "
"confirm that the restored quotation supports this unit."
),
field="supporting_excerpts",
))
return candidate.model_copy(update={
"supporting_excerpts": tuple(restored),
"review_items": review_items,
})
def authoring_skill_path() -> Path:
"""The installed wheel and the deployment's Pi resources live in different roots."""
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
def __init__(
self,
pi_executable: str,
skill_path: Path,
*,
timeout_seconds: int = 300,
) -> None:
self._pi_executable = pi_executable
self._skill_path = skill_path
self._json_mode_extension = (
skill_path.parents[2] / "evidence-extensions" / "tht-evidence-json-mode.ts"
)
self._timeout_seconds = timeout_seconds
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
try:
system_prompt = self._skill_path.read_text(encoding="utf-8")
except (OSError, UnicodeError) as error:
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
request_path = Path(temporary) / "request.json"
request_path.write_text(
json.dumps(request.model_dump(mode="json"), ensure_ascii=False),
encoding="utf-8",
)
argv = [
self._pi_executable,
"--mode", "text",
"--print",
"--no-session",
"--no-tools",
"--no-extensions",
"--no-context-files",
"--no-skills",
"--extension", str(self._json_mode_extension),
]
canonical_model = os.environ.get("THT_DEFAULT_SESSION_MODEL")
if canonical_model:
provider, separator, model = canonical_model.partition("/")
if separator != "/" or not provider or not model:
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
argv.extend(("--provider", provider, "--model", model))
thinking = os.environ.get("PI_THINKING")
if thinking:
argv.extend(("--thinking", thinking))
argv.extend((
"--system-prompt", system_prompt,
f"@{request_path}",
(
"Return only the JSON object required by the Evidence authoring skill. "
+ _request_specific_constraints(request.normalized_text)
),
))
try:
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
result = subprocess.run(
argv,
check=False,
stdout=response,
stderr=subprocess.DEVNULL,
timeout=self._timeout_seconds,
shell=False,
)
if response.tell() > 1024 * 1024:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file)
response.seek(0)
stdout = response.read()
except (OSError, subprocess.TimeoutExpired) as error:
raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error
if result.returncode != 0:
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
try:
raw = json.loads(stdout)
if isinstance(raw, dict) and set(raw) == {"candidates"}:
candidates = raw["candidates"]
elif isinstance(raw, list):
candidates = raw
elif isinstance(raw, dict):
candidates = [raw]
else:
raise ValueError("response shape")
if not isinstance(candidates, list):
raise TypeError("response shape")
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
return tuple(
_restore_exact_source_excerpts(candidate, request.normalized_text)
for candidate in parsed
)
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
@dataclass(frozen=True)
class EvidencePreparationReport:
changed: tuple[str, ...]
unchanged: tuple[str, ...]
created: tuple[str, ...]
orphaned: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
model_calls: int
@dataclass(frozen=True)
class EvidenceMigrationReport:
"""Result of a deterministic Curated Evidence presentation-format migration."""
migrated: tuple[str, ...]
unchanged: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
@dataclass(frozen=True)
class EvidenceResolutionReport:
"""The result of one curator-directed Evidence resolution."""
action: Literal["retired", "relinked"]
evidence_id: str
source_file: str | None
findings: tuple[ValidationFinding, ...]
class ManifestSource(StrictModel):
sha256: str
units: tuple[str, ...]
@field_validator("sha256")
@classmethod
def _validate_sha256(cls, value: str) -> str:
if not value.startswith("sha256:") or len(value) != 71:
raise ValueError("sha256 must be a sha256 digest")
try:
int(value.removeprefix("sha256:"), 16)
except ValueError as error:
raise ValueError("sha256 must be a sha256 digest") from error
return value
@field_validator("units")
@classmethod
def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if any(not is_evidence_id(unit) for unit in value):
raise ValueError("units must use stable evidence identifiers")
return value
class EvidenceManifest(StrictModel):
schema_version: Literal[1]
pipeline_version: Literal["evidence-authoring-v1"]
sources: dict[str, ManifestSource]
orphans: tuple[str, ...]
@field_validator("sources")
@classmethod
def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]:
for source_path in value:
validate_source_file(source_path)
return value
@field_validator("orphans")
@classmethod
def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if any(not is_evidence_id(unit) for unit in value):
raise ValueError("orphans must use stable evidence identifiers")
return value
@model_validator(mode="after")
def _validate_unit_membership(self) -> EvidenceManifest:
seen: set[str] = set()
for source in self.sources.values():
duplicate = seen.intersection(source.units)
if duplicate:
raise ValueError("a unit may belong to only one source")
seen.update(source.units)
if len(set(self.orphans)) != len(self.orphans):
raise ValueError("orphans must be unique")
return self
def load_manifest(path: Path) -> EvidenceManifest:
"""Load the managed, versioned authoring manifest."""
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, yaml.YAMLError) as error:
raise ValueError("manifest cannot be read") from error
try:
return EvidenceManifest.model_validate(raw)
except ValidationError as error:
raise ValueError("manifest is invalid") from error
def dump_manifest(manifest: EvidenceManifest) -> str:
"""Serialize the manifest deterministically for Git review."""
return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True)
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
"""Validate the curated corpus without writing the workspace."""
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").is_file():
from .local_archive import LocalEvidenceArchive
try:
LocalEvidenceArchive(workspace_root).validate()
return ValidationReport(())
except (OSError, ValueError) as error:
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
"evidence/curated", str(error)),))
findings: list[ValidationFinding] = []
manifest_path = evidence_root / "manifest.yaml"
if not manifest_path.is_file():
return ValidationReport((ValidationFinding(
severity="error",
code="manifest_missing",
path="manifest.yaml",
message="The managed Evidence manifest is missing.",
),))
try:
manifest = load_manifest(manifest_path)
except ValueError:
return ValidationReport((ValidationFinding(
severity="error",
code="manifest_invalid",
path="manifest.yaml",
message="The managed Evidence manifest is invalid.",
),))
for orphan in manifest.orphans:
findings.append(ValidationFinding(
severity="error",
code="orphaned_unit",
path="manifest.yaml",
message=f"Orphaned Evidence {orphan} must be resolved before publication.",
))
try:
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError):
return ValidationReport(tuple(findings + [ValidationFinding(
severity="error",
code="curated_invalid",
path="curated",
message="A Curated Evidence document is invalid or cannot be read.",
)]))
source_texts = {
source_path: _validate_manifest_source(evidence_root, source_path, source, findings)
for source_path, source in manifest.sources.items()
}
seen_ids: set[str] = set()
for evidence in documents:
if evidence.id in seen_ids:
findings.append(ValidationFinding(
severity="error",
code="duplicate_evidence_id",
path="curated",
message=f"Evidence id {evidence.id} appears more than once.",
))
seen_ids.add(evidence.id)
findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file)))
for source in manifest.sources.values():
for unit in source.units:
if unit not in seen_ids:
findings.append(ValidationFinding(
severity="error",
code="manifest_unit_without_curated",
path="manifest.yaml",
message=f"Manifest Evidence {unit} has no Curated document.",
))
return ValidationReport(tuple(findings))
def _validate_manifest_source(
evidence_root: Path,
source_path: str,
manifest_source: ManifestSource,
findings: list[ValidationFinding],
) -> str | None:
path = evidence_root / source_path
if path.is_symlink():
findings.append(ValidationFinding("error", "source_unsafe", source_path,
"The manifest source must be a regular file below the Evidence tree."))
return None
if not path.is_file():
findings.append(ValidationFinding("error", "source_missing", source_path,
"The manifest source does not exist."))
return None
if path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
findings.append(ValidationFinding("error", "source_oversized", source_path,
"The manifest source exceeds the authoring size limit."))
return None
try:
source = normalize_source_text(path.read_text(encoding="utf-8"))
except UnicodeDecodeError:
findings.append(ValidationFinding("error", "source_unreadable", source_path,
"The manifest source is not valid UTF-8."))
return None
digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
if digest != manifest_source.sha256:
findings.append(ValidationFinding("error", "source_hash_mismatch", source_path,
"The manifest hash does not match the normalized source."))
return source
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
if isinstance(evidence.provenance, ManualEvidenceProvenance):
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
for item in evidence.review_items]
if evidence.id in manifest.orphans:
return []
path = evidence.provenance.source_file
findings: list[ValidationFinding] = []
manifest_source = manifest.sources.get(path)
if manifest_source is None:
findings.append(ValidationFinding(
severity="error",
code="manifest_source_missing",
path="manifest.yaml",
message="The manifest does not contain the provenance source.",
))
else:
if manifest_source.sha256 != evidence.provenance.source_sha256:
findings.append(ValidationFinding(
severity="error",
code="manifest_source_hash_mismatch",
path="manifest.yaml",
message="The manifest hash does not match the Evidence provenance.",
))
if evidence.id not in manifest_source.units:
findings.append(ValidationFinding(
severity="error",
code="manifest_unit_missing",
path="manifest.yaml",
message="The manifest does not link the Evidence unit to its source.",
))
if source is None:
return findings
for excerpt in evidence.provenance.supporting_excerpts:
if _normalize(excerpt) not in source:
findings.append(ValidationFinding(
severity="error",
code="supporting_excerpt_missing",
path=path,
message="A supporting excerpt is absent from the normalized source.",
))
for item in evidence.review_items:
findings.append(ValidationFinding(
severity="error",
code="unresolved_review_item",
path=path,
message=f"Review item {item.code} must be resolved before publication.",
))
return findings
def _normalize(text: str) -> str:
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
def normalize_source_text(text: str) -> str:
"""Normalize Source Evidence mechanically without changing its meaning."""
normalized = _normalize(text)
return normalized if normalized.endswith("\n") else normalized + "\n"
def prepare_workspace_evidence(
workspace_root: Path,
*,
restructurer: EvidenceRestructurer,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
upgrade: bool = False,
max_workers: int = 1,
) -> EvidencePreparationReport:
"""Prepare all changed Source Evidence without publishing or committing it.
Every deterministic operation happens locally. A changed source receives exactly
one call through ``restructurer``; all model results validate before the staged
authoring tree replaces the current one.
"""
if not 1 <= max_workers <= 8:
raise EvidencePreparationError("authoring_workers_invalid")
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
manifest_path = evidence_root / "manifest.yaml"
try:
manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade)
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
source_texts = _load_source_texts(evidence_root)
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
source_units = {
source_file: tuple(source.units)
for source_file, source in manifest.sources.items()
}
orphaned = set(manifest.orphans)
created: list[str] = []
changed: list[str] = []
unchanged: list[str] = []
model_calls = 0
renamed_sources = _preserve_uniquely_renamed_sources(
source_texts, manifest, documents_by_id, source_units,
)
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
requests: list[tuple[str, RestructureRequest]] = []
for source_file in sorted(source_texts):
source_text = source_texts[source_file]
source_hash = _source_hash(source_text)
previous = tuple(
documents_by_id[unit]
for unit in source_units.get(source_file, ())
if unit in documents_by_id
)
manifest_source = manifest.sources.get(source_file)
if not upgrade and (source_file in renamed_sources or (
manifest_source is not None and manifest_source.sha256 == source_hash
)):
unchanged.append(source_file)
continue
changed.append(source_file)
request = RestructureRequest(
source_file=source_file,
source_sha256=source_hash,
normalized_text=source_text,
previous_units=previous,
)
requests.append((source_file, request))
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
with ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = {
source_file: executor.submit(restructurer.restructure, request)
for source_file, request in requests
}
for source_file, _request in requests:
try:
candidates_by_source[source_file] = futures[source_file].result()
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
reserved_ids = set(documents_by_id)
for source_file, request in requests:
source_text = source_texts[source_file]
source_hash = request.source_sha256
previous = request.previous_units
candidates = candidates_by_source[source_file]
model_calls += 1
selected_existing_ids: set[str] = set()
generated_ids: list[str] = []
for candidate in candidates:
if candidate.existing_id is not None:
if candidate.existing_id not in {document.id for document in previous}:
raise EvidencePreparationError("unknown_existing_id", source_file)
evidence_id = candidate.existing_id
selected_existing_ids.add(evidence_id)
else:
evidence_id = _allocate_evidence_id(candidate.title, reserved_ids)
reserved_ids.add(evidence_id)
created.append(evidence_id)
try:
evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash)
except ValidationError as error:
raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error
if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts):
raise EvidencePreparationError("supporting_excerpt_missing", source_file)
documents_by_id[evidence.id] = evidence
generated_ids.append(evidence.id)
for prior in previous:
if prior.id not in selected_existing_ids:
documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash)
generated_ids.append(prior.id)
source_units[source_file] = tuple(sorted(set(generated_ids)))
next_manifest = EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={
source_file: ManifestSource(
sha256=_source_hash(source_texts[source_file]),
units=source_units.get(source_file, ()),
)
for source_file in sorted(source_texts)
},
orphans=tuple(sorted(orphaned)),
)
if not changed and next_manifest == manifest:
return EvidencePreparationReport(
changed=(),
unchanged=tuple(unchanged),
created=(),
orphaned=tuple(sorted(orphaned)),
findings=validate_workspace_evidence(workspace_root).findings,
model_calls=0,
)
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
return EvidencePreparationReport(
changed=tuple(changed),
unchanged=tuple(unchanged),
created=tuple(sorted(created)),
orphaned=tuple(sorted(orphaned)),
findings=findings,
model_calls=model_calls,
)
def migrate_workspace_evidence(
workspace_root: Path,
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 4})
for evidence_id, document in documents_by_id.items()
}
migrated_ids: list[str] = []
unchanged_ids: list[str] = []
curated_root = evidence_root / "curated"
for evidence_id, document in upgraded.items():
path = curated_root / document.kind / f"{evidence_id.removeprefix('evidence:')}.md"
try:
presentation_is_current = path.read_text(encoding="utf-8") == dump_curated_markdown(
document,
)
except (OSError, UnicodeDecodeError):
presentation_is_current = False
target = unchanged_ids if presentation_is_current else migrated_ids
target.append(evidence_id)
migrated = tuple(sorted(migrated_ids))
unchanged = tuple(sorted(unchanged_ids))
if not migrated:
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=migrated,
unchanged=unchanged,
findings=findings,
)
def resolve_workspace_evidence(
workspace_root: Path,
evidence_id: str,
*,
retire: bool = False,
source: str | Path | None = None,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceResolutionReport:
"""Retire or relink one Evidence unit as one staged curator update.
The operation intentionally neither creates a commit nor publishes anything. Both
the curated tree and the managed manifest are replaced only after the staged tree
has been written and structurally validated.
"""
if retire == (source is not None):
raise EvidencePreparationError("resolve_mode_invalid")
if not is_evidence_id(evidence_id):
raise EvidencePreparationError("evidence_id_invalid")
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local/state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
_reject_dirty_worktree(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
if evidence_id not in documents_by_id:
raise EvidencePreparationError("evidence_not_found")
source_units = {path: tuple(entry.units) for path, entry in manifest.sources.items()}
orphaned = set(manifest.orphans)
source_file: str | None = None
action: Literal["retired", "relinked"]
if retire:
action = "retired"
del documents_by_id[evidence_id]
for path, units in tuple(source_units.items()):
source_units[path] = tuple(unit for unit in units if unit != evidence_id)
orphaned.discard(evidence_id)
else:
action = "relinked"
assert source is not None
source_file = _resolve_source_file(evidence_root, source)
source_text = _load_resolution_source(evidence_root, source_file)
document = documents_by_id[evidence_id]
review_items = document.review_items
if all(_normalize(excerpt) in source_text for excerpt in document.provenance.supporting_excerpts):
review_items = tuple(
item for item in review_items if item.code != "source_no_longer_supports_unit"
)
documents_by_id[evidence_id] = document.model_copy(update={
"provenance": document.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": _source_hash(source_text),
}),
"review_items": review_items,
})
for path, units in tuple(source_units.items()):
source_units[path] = tuple(unit for unit in units if unit != evidence_id)
source_units[source_file] = tuple(sorted((*source_units.get(source_file, ()), evidence_id)))
orphaned.discard(evidence_id)
sources = {
path: ManifestSource(
sha256=(
_source_hash(_load_resolution_source(evidence_root, path))
if path == source_file
else manifest.sources[path].sha256
),
units=units,
)
for path, units in sorted(source_units.items())
}
next_manifest = EvidenceManifest(
schema_version=1,
pipeline_version=manifest.pipeline_version,
sources=sources,
orphans=tuple(sorted(orphaned)),
)
findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest)
return EvidenceResolutionReport(
action=action,
evidence_id=evidence_id,
source_file=source_file,
findings=findings,
)
def _empty_manifest() -> EvidenceManifest:
return EvidenceManifest(
schema_version=1,
pipeline_version="evidence-authoring-v1",
sources={},
orphans=(),
)
def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest:
if not path.is_file():
return _empty_manifest()
try:
return load_manifest(path)
except ValueError as original_error:
if not upgrade:
raise EvidencePreparationError("pipeline_upgrade_required") from original_error
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
if not isinstance(raw, dict) or raw.get("schema_version") != 1:
raise ValueError("manifest shape")
raw["pipeline_version"] = "evidence-authoring-v1"
return EvidenceManifest.model_validate(raw)
except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
def _resolve_source_file(evidence_root: Path, source: str | Path) -> str:
raw_source = source.as_posix() if isinstance(source, Path) else source
try:
source_file = validate_source_file(raw_source)
except ValueError as error:
raise EvidencePreparationError("source_invalid", raw_source) from error
path = evidence_root / source_file
ancestor = evidence_root
for part in Path(source_file).parts:
ancestor /= part
if ancestor.is_symlink():
raise EvidencePreparationError("source_invalid", source_file)
if not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
raise EvidencePreparationError("source_invalid", source_file)
return source_file
def _load_resolution_source(evidence_root: Path, source_file: str) -> str:
try:
return normalize_source_text((evidence_root / source_file).read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError) as error:
raise EvidencePreparationError("source_invalid", source_file) from error
def _git_status(workspace_root: Path) -> tuple[str, ...]:
result = subprocess.run(
["git", "status", "--porcelain", "--untracked-files=all"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
prefix_result = subprocess.run(
["git", "rev-parse", "--show-prefix"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if prefix_result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
prefix = prefix_result.stdout.strip()
entries: list[str] = []
for line in result.stdout.splitlines():
if not line:
continue
path = line[3:] if len(line) > 3 else line
if prefix and path.startswith(prefix):
line = line[:3] + path.removeprefix(prefix)
entries.append(line)
return tuple(entries)
def _reject_dirty_authoring_state(
workspace_root: Path,
git_status: Callable[[Path], tuple[str, ...]],
) -> None:
for entry in git_status(workspace_root):
path = entry[3:] if len(entry) > 3 else entry
if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"):
raise EvidencePreparationError("authoring_worktree_dirty")
def _reject_dirty_worktree(
workspace_root: Path,
git_status: Callable[[Path], tuple[str, ...]],
) -> None:
if git_status(workspace_root):
raise EvidencePreparationError("worktree_dirty")
def _load_source_texts(evidence_root: Path) -> dict[str, str]:
source_root = evidence_root / "source"
if not source_root.is_dir():
return {}
sources: dict[str, str] = {}
for path in sorted(source_root.rglob("*")):
if path.is_dir():
continue
relative = path.relative_to(evidence_root).as_posix()
try:
validate_source_file(relative)
if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
raise ValueError("unsafe source")
sources[relative] = normalize_source_text(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, ValueError) as error:
raise EvidencePreparationError("source_invalid", relative) from error
return sources
def _source_hash(source: str) -> str:
return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
def _preserve_removed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
orphaned: set[str],
) -> None:
for source_file in manifest.sources:
if source_file in source_texts:
continue
unit_ids = source_units.pop(source_file, None)
if unit_ids is None:
continue
for unit in unit_ids:
if unit in documents_by_id:
orphaned.add(unit)
def _preserve_uniquely_renamed_sources(
source_texts: dict[str, str],
manifest: EvidenceManifest,
documents_by_id: dict[str, CuratedEvidence],
source_units: dict[str, tuple[str, ...]],
) -> set[str]:
renamed_sources: set[str] = set()
unmatched_sources = [path for path in source_texts if path not in manifest.sources]
missing_sources = [path for path in manifest.sources if path not in source_texts]
for old_path in missing_sources:
matches = [
path for path in unmatched_sources
if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256
]
if len(matches) != 1:
continue
new_path = matches[0]
unit_ids = source_units.pop(old_path, ())
source_units[new_path] = unit_ids
for unit_id in unit_ids:
document = documents_by_id.get(unit_id)
if document is None:
continue
documents_by_id[unit_id] = document.model_copy(update={
"provenance": document.provenance.model_copy(update={"source_file": new_path}),
})
unmatched_sources.remove(new_path)
renamed_sources.add(new_path)
return renamed_sources
def _candidate_to_evidence(
candidate: RestructureCandidate,
evidence_id: str,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
data = candidate.model_dump(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 4
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
"source_sha256": source_hash,
"supporting_excerpts": candidate.supporting_excerpts,
}
return CuratedEvidence.model_validate(data)
def _unsupported_unit(
evidence: CuratedEvidence,
source_file: str,
source_hash: str,
) -> CuratedEvidence:
review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit")
review_items += (ReviewItem(
code="source_no_longer_supports_unit",
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 4,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
}),
"review_items": review_items,
})
def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str:
ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii")
slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit"
base = f"evidence:{slug}"
if base not in reserved_ids:
return base
suffix = 2
while f"{base}-{suffix}" in reserved_ids:
suffix += 1
return f"{base}-{suffix}"
def _stage_and_apply_authoring_tree(
workspace_root: Path,
documents_by_id: dict[str, CuratedEvidence],
manifest: EvidenceManifest,
) -> tuple[ValidationFinding, ...]:
evidence_root = workspace_root / "evidence"
with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary:
staged_workspace = Path(temporary) / "workspace"
staged_evidence = staged_workspace / "evidence"
if evidence_root.exists():
shutil.copytree(evidence_root, staged_evidence, symlinks=True)
else:
staged_evidence.mkdir(parents=True)
staged_curated = staged_evidence / "curated"
if staged_curated.exists():
shutil.rmtree(staged_curated)
staged_curated.mkdir()
for evidence in documents_by_id.values():
destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md"
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(dump_curated_markdown(evidence), encoding="utf-8")
(staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8")
validation = validate_workspace_evidence(staged_workspace)
if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings):
raise EvidencePreparationError("staged_authoring_state_invalid")
backup = Path(temporary) / "previous-evidence"
if evidence_root.exists():
os.replace(evidence_root, backup)
try:
os.replace(staged_evidence, evidence_root)
except OSError:
if backup.exists():
os.replace(backup, evidence_root)
raise
return validation.findings