"""Validation for the Git-reviewed Evidence authoring workspace.""" from __future__ import annotations import hashlib import json import os import re import shutil import subprocess import tempfile import unicodedata from collections.abc import Callable from concurrent.futures import ThreadPoolExecutor from dataclasses import dataclass from difflib import SequenceMatcher from pathlib import Path from typing import Literal, Protocol import yaml from pydantic import Field, ValidationError, field_validator, model_validator from tht.evidence.canonical import ( CuratedEvidence, EvidenceKind, EvidencePurpose, EvidenceScope, ReviewItem, StrictModel, dump_curated_markdown, is_evidence_id, load_curated_tree, validate_source_file, ) MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024 @dataclass(frozen=True) class ValidationFinding: severity: Literal["error", "warning"] code: str path: str message: str @dataclass(frozen=True) class ValidationReport: findings: tuple[ValidationFinding, ...] @property def publishable(self) -> bool: return not any(finding.severity == "error" for finding in self.findings) class EvidencePreparationError(RuntimeError): """A bounded authoring failure that is safe to report to a curator.""" def __init__(self, code: str, source_file: str | None = None) -> None: self.code = code self.source_file = source_file message = code if source_file is None else f"{code}: {source_file}" super().__init__(message) class RestructureRequest(StrictModel): """The deterministic input passed to exactly one model call for one source.""" source_file: str source_sha256: str normalized_text: str previous_units: tuple[CuratedEvidence, ...] = () @field_validator("source_file") @classmethod def _validate_source_file(cls, value: str) -> str: return validate_source_file(value) class RestructureCandidate(StrictModel): """A model proposal before the host allocates or reuses its stable ID.""" schema_version: Literal[1] existing_id: str | None = None title: str kind: EvidenceKind purposes: tuple[EvidencePurpose, ...] applies_to: EvidenceScope = Field(default_factory=EvidenceScope) language: str supporting_excerpts: tuple[str, ...] review_items: tuple[ReviewItem, ...] = () payload: dict[str, object] @field_validator("existing_id") @classmethod def _validate_existing_id(cls, value: str | None) -> str | None: if value is not None and not is_evidence_id(value): raise ValueError("existing_id must use the evidence: form") return value class EvidenceRestructurer(Protocol): """Boundary for the ephemeral, read-only model restructuring call.""" def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ... def _excerpt_signature(value: str) -> str: value = "\n".join( re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line) for line in value.splitlines() ) value = re.sub(r"[*_`]", "", value) return " ".join(value.split()) def _request_specific_constraints(source_text: str) -> str: qualified_column = re.search( r"(? RestructureCandidate: source_lines = source_text.splitlines() source_spans: list[str] = [] for start, first_line in enumerate(source_lines): if not first_line: continue for width in range(1, 6): selected = source_lines[start:start + width] if len(selected) != width or not selected[-1]: continue span = "\n".join(selected) if len(span) <= 1000: source_spans.append(span) restored: list[str] = [] reconciled = False for excerpt in candidate.supporting_excerpts: if excerpt in source_text: restored.append(excerpt) continue signature = _excerpt_signature(excerpt) matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature) if len(matches) == 1: restored.append(matches[0]) continue ranked = sorted( ( (SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line) for line in source_spans ), reverse=True, ) best_score = ranked[0][0] if ranked else 0.0 runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0 if best_score >= 0.88 and best_score - runner_up_score >= 0.08: restored.append(ranked[0][1]) reconciled = True else: restored.append(excerpt) review_items = candidate.review_items if reconciled and not any( item.code == "supporting_excerpt_reconciled" for item in review_items ): review_items = (*review_items, ReviewItem( code="supporting_excerpt_reconciled", message=( "A model excerpt was reconciled to a unique exact source line; " "confirm that the restored quotation supports this unit." ), field="supporting_excerpts", )) return candidate.model_copy(update={ "supporting_excerpts": tuple(restored), "review_items": review_items, }) class PiEvidenceRestructurer: """Invoke Pi once, without tools or session state, for one changed source.""" def __init__( self, pi_executable: str, skill_path: Path, *, timeout_seconds: int = 300, ) -> None: self._pi_executable = pi_executable self._skill_path = skill_path self._json_mode_extension = ( skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts" ) self._timeout_seconds = timeout_seconds def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: try: system_prompt = self._skill_path.read_text(encoding="utf-8") except (OSError, UnicodeError) as error: raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary: request_path = Path(temporary) / "request.json" request_path.write_text( json.dumps(request.model_dump(mode="json"), ensure_ascii=False), encoding="utf-8", ) argv = [ self._pi_executable, "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files", "--no-skills", "--extension", str(self._json_mode_extension), ] for option, environment_name in ( ("--provider", "PI_PROVIDER"), ("--model", "PI_MODEL"), ("--thinking", "PI_THINKING"), ): value = os.environ.get(environment_name) if value: argv.extend((option, value)) argv.extend(( "--system-prompt", system_prompt, f"@{request_path}", ( "Return only the JSON object required by the Evidence authoring skill. " + _request_specific_constraints(request.normalized_text) ), )) try: with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response: result = subprocess.run( argv, check=False, stdout=response, stderr=subprocess.DEVNULL, timeout=self._timeout_seconds, shell=False, ) if response.tell() > 1024 * 1024: raise EvidencePreparationError("pi_restructure_invalid", request.source_file) response.seek(0) stdout = response.read() except (OSError, subprocess.TimeoutExpired) as error: raise EvidencePreparationError("pi_restructure_failed", request.source_file) from error if result.returncode != 0: raise EvidencePreparationError("pi_restructure_failed", request.source_file) try: raw = json.loads(stdout) if isinstance(raw, dict) and set(raw) == {"candidates"}: candidates = raw["candidates"] elif isinstance(raw, list): candidates = raw elif isinstance(raw, dict): candidates = [raw] else: raise ValueError("response shape") if not isinstance(candidates, list): raise TypeError("response shape") parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates) return tuple( _restore_exact_source_excerpts(candidate, request.normalized_text) for candidate in parsed ) except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error: raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error @dataclass(frozen=True) class EvidencePreparationReport: changed: tuple[str, ...] unchanged: tuple[str, ...] created: tuple[str, ...] orphaned: tuple[str, ...] findings: tuple[ValidationFinding, ...] model_calls: int @dataclass(frozen=True) class EvidenceMigrationReport: """Result of a deterministic Curated Evidence presentation-format migration.""" migrated: tuple[str, ...] unchanged: tuple[str, ...] findings: tuple[ValidationFinding, ...] @dataclass(frozen=True) class EvidenceResolutionReport: """The result of one curator-directed Evidence resolution.""" action: Literal["retired", "relinked"] evidence_id: str source_file: str | None findings: tuple[ValidationFinding, ...] class ManifestSource(StrictModel): sha256: str units: tuple[str, ...] @field_validator("sha256") @classmethod def _validate_sha256(cls, value: str) -> str: if not value.startswith("sha256:") or len(value) != 71: raise ValueError("sha256 must be a sha256 digest") try: int(value.removeprefix("sha256:"), 16) except ValueError as error: raise ValueError("sha256 must be a sha256 digest") from error return value @field_validator("units") @classmethod def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]: if any(not is_evidence_id(unit) for unit in value): raise ValueError("units must use stable evidence identifiers") return value class EvidenceManifest(StrictModel): schema_version: Literal[1] pipeline_version: Literal["evidence-authoring-v1"] sources: dict[str, ManifestSource] orphans: tuple[str, ...] @field_validator("sources") @classmethod def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]: for source_path in value: validate_source_file(source_path) return value @field_validator("orphans") @classmethod def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]: if any(not is_evidence_id(unit) for unit in value): raise ValueError("orphans must use stable evidence identifiers") return value @model_validator(mode="after") def _validate_unit_membership(self) -> EvidenceManifest: seen: set[str] = set() for source in self.sources.values(): duplicate = seen.intersection(source.units) if duplicate: raise ValueError("a unit may belong to only one source") seen.update(source.units) if len(set(self.orphans)) != len(self.orphans): raise ValueError("orphans must be unique") return self def load_manifest(path: Path) -> EvidenceManifest: """Load the managed, versioned authoring manifest.""" try: raw = yaml.safe_load(path.read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError, yaml.YAMLError) as error: raise ValueError("manifest cannot be read") from error try: return EvidenceManifest.model_validate(raw) except ValidationError as error: raise ValueError("manifest is invalid") from error def dump_manifest(manifest: EvidenceManifest) -> str: """Serialize the manifest deterministically for Git review.""" return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True) def validate_workspace_evidence(workspace_root: Path) -> ValidationReport: """Validate the curated corpus without writing the workspace.""" evidence_root = workspace_root / "evidence" findings: list[ValidationFinding] = [] manifest_path = evidence_root / "manifest.yaml" if not manifest_path.is_file(): return ValidationReport((ValidationFinding( severity="error", code="manifest_missing", path="manifest.yaml", message="The managed Evidence manifest is missing.", ),)) try: manifest = load_manifest(manifest_path) except ValueError: return ValidationReport((ValidationFinding( severity="error", code="manifest_invalid", path="manifest.yaml", message="The managed Evidence manifest is invalid.", ),)) for orphan in manifest.orphans: findings.append(ValidationFinding( severity="error", code="orphaned_unit", path="manifest.yaml", message=f"Orphaned Evidence {orphan} must be resolved before publication.", )) try: documents = load_curated_tree(evidence_root / "curated") except (OSError, ValidationError, ValueError): return ValidationReport(tuple(findings + [ValidationFinding( severity="error", code="curated_invalid", path="curated", message="A Curated Evidence document is invalid or cannot be read.", )])) source_texts = { source_path: _validate_manifest_source(evidence_root, source_path, source, findings) for source_path, source in manifest.sources.items() } seen_ids: set[str] = set() for evidence in documents: if evidence.id in seen_ids: findings.append(ValidationFinding( severity="error", code="duplicate_evidence_id", path="curated", message=f"Evidence id {evidence.id} appears more than once.", )) seen_ids.add(evidence.id) findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file))) for source in manifest.sources.values(): for unit in source.units: if unit not in seen_ids: findings.append(ValidationFinding( severity="error", code="manifest_unit_without_curated", path="manifest.yaml", message=f"Manifest Evidence {unit} has no Curated document.", )) return ValidationReport(tuple(findings)) def _validate_manifest_source( evidence_root: Path, source_path: str, manifest_source: ManifestSource, findings: list[ValidationFinding], ) -> str | None: path = evidence_root / source_path if path.is_symlink(): findings.append(ValidationFinding("error", "source_unsafe", source_path, "The manifest source must be a regular file below the Evidence tree.")) return None if not path.is_file(): findings.append(ValidationFinding("error", "source_missing", source_path, "The manifest source does not exist.")) return None if path.stat().st_size > MAX_AUTHORING_FILE_BYTES: findings.append(ValidationFinding("error", "source_oversized", source_path, "The manifest source exceeds the authoring size limit.")) return None try: source = normalize_source_text(path.read_text(encoding="utf-8")) except UnicodeDecodeError: findings.append(ValidationFinding("error", "source_unreadable", source_path, "The manifest source is not valid UTF-8.")) return None digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest() if digest != manifest_source.sha256: findings.append(ValidationFinding("error", "source_hash_mismatch", source_path, "The manifest hash does not match the normalized source.")) return source def _validate_unit( manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None, ) -> list[ValidationFinding]: if evidence.id in manifest.orphans: return [] path = evidence.provenance.source_file findings: list[ValidationFinding] = [] manifest_source = manifest.sources.get(path) if manifest_source is None: findings.append(ValidationFinding( severity="error", code="manifest_source_missing", path="manifest.yaml", message="The manifest does not contain the provenance source.", )) else: if manifest_source.sha256 != evidence.provenance.source_sha256: findings.append(ValidationFinding( severity="error", code="manifest_source_hash_mismatch", path="manifest.yaml", message="The manifest hash does not match the Evidence provenance.", )) if evidence.id not in manifest_source.units: findings.append(ValidationFinding( severity="error", code="manifest_unit_missing", path="manifest.yaml", message="The manifest does not link the Evidence unit to its source.", )) if source is None: return findings for excerpt in evidence.provenance.supporting_excerpts: if _normalize(excerpt) not in source: findings.append(ValidationFinding( severity="error", code="supporting_excerpt_missing", path=path, message="A supporting excerpt is absent from the normalized source.", )) for item in evidence.review_items: findings.append(ValidationFinding( severity="error", code="unresolved_review_item", path=path, message=f"Review item {item.code} must be resolved before publication.", )) return findings def _normalize(text: str) -> str: return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n")) def normalize_source_text(text: str) -> str: """Normalize Source Evidence mechanically without changing its meaning.""" normalized = _normalize(text) return normalized if normalized.endswith("\n") else normalized + "\n" def prepare_workspace_evidence( workspace_root: Path, *, restructurer: EvidenceRestructurer, git_status: Callable[[Path], tuple[str, ...]] | None = None, upgrade: bool = False, max_workers: int = 1, ) -> EvidencePreparationReport: """Prepare all changed Source Evidence without publishing or committing it. Every deterministic operation happens locally. A changed source receives exactly one call through ``restructurer``; all model results validate before the staged authoring tree replaces the current one. """ if not 1 <= max_workers <= 8: raise EvidencePreparationError("authoring_workers_invalid") workspace_root = workspace_root.resolve() evidence_root = workspace_root / "evidence" _reject_dirty_authoring_state(workspace_root, git_status or _git_status) manifest_path = evidence_root / "manifest.yaml" try: manifest = _load_preparation_manifest(manifest_path, upgrade=upgrade) documents = load_curated_tree(evidence_root / "curated") except (OSError, ValidationError, ValueError) as error: raise EvidencePreparationError("authoring_state_invalid") from error source_texts = _load_source_texts(evidence_root) documents_by_id = {document.id: document for document in documents} if len(documents_by_id) != len(documents): raise EvidencePreparationError("duplicate_evidence_id") source_units = { source_file: tuple(source.units) for source_file, source in manifest.sources.items() } orphaned = set(manifest.orphans) created: list[str] = [] changed: list[str] = [] unchanged: list[str] = [] model_calls = 0 renamed_sources = _preserve_uniquely_renamed_sources( source_texts, manifest, documents_by_id, source_units, ) _preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned) requests: list[tuple[str, RestructureRequest]] = [] for source_file in sorted(source_texts): source_text = source_texts[source_file] source_hash = _source_hash(source_text) previous = tuple( documents_by_id[unit] for unit in source_units.get(source_file, ()) if unit in documents_by_id ) manifest_source = manifest.sources.get(source_file) if not upgrade and (source_file in renamed_sources or ( manifest_source is not None and manifest_source.sha256 == source_hash )): unchanged.append(source_file) continue changed.append(source_file) request = RestructureRequest( source_file=source_file, source_sha256=source_hash, normalized_text=source_text, previous_units=previous, ) requests.append((source_file, request)) candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {} with ThreadPoolExecutor(max_workers=max_workers) as executor: futures = { source_file: executor.submit(restructurer.restructure, request) for source_file, request in requests } for source_file, _request in requests: try: candidates_by_source[source_file] = futures[source_file].result() except EvidencePreparationError: raise except Exception as error: raise EvidencePreparationError("restructuring_failed", source_file) from error reserved_ids = set(documents_by_id) for source_file, request in requests: source_text = source_texts[source_file] source_hash = request.source_sha256 previous = request.previous_units candidates = candidates_by_source[source_file] model_calls += 1 selected_existing_ids: set[str] = set() generated_ids: list[str] = [] for candidate in candidates: if candidate.existing_id is not None: if candidate.existing_id not in {document.id for document in previous}: raise EvidencePreparationError("unknown_existing_id", source_file) evidence_id = candidate.existing_id selected_existing_ids.add(evidence_id) else: evidence_id = _allocate_evidence_id(candidate.title, reserved_ids) reserved_ids.add(evidence_id) created.append(evidence_id) try: evidence = _candidate_to_evidence(candidate, evidence_id, source_file, source_hash) except ValidationError as error: raise EvidencePreparationError("invalid_restructure_candidate", source_file) from error if any(excerpt not in source_text for excerpt in evidence.provenance.supporting_excerpts): raise EvidencePreparationError("supporting_excerpt_missing", source_file) documents_by_id[evidence.id] = evidence generated_ids.append(evidence.id) for prior in previous: if prior.id not in selected_existing_ids: documents_by_id[prior.id] = _unsupported_unit(prior, source_file, source_hash) generated_ids.append(prior.id) source_units[source_file] = tuple(sorted(set(generated_ids))) next_manifest = EvidenceManifest( schema_version=1, pipeline_version="evidence-authoring-v1", sources={ source_file: ManifestSource( sha256=_source_hash(source_texts[source_file]), units=source_units.get(source_file, ()), ) for source_file in sorted(source_texts) }, orphans=tuple(sorted(orphaned)), ) if not changed and next_manifest == manifest: return EvidencePreparationReport( changed=(), unchanged=tuple(unchanged), created=(), orphaned=tuple(sorted(orphaned)), findings=validate_workspace_evidence(workspace_root).findings, model_calls=0, ) findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest) return EvidencePreparationReport( changed=tuple(changed), unchanged=tuple(unchanged), created=tuple(sorted(created)), orphaned=tuple(sorted(orphaned)), findings=findings, model_calls=model_calls, ) def migrate_workspace_evidence( workspace_root: Path, *, git_status: Callable[[Path], tuple[str, ...]] | None = None, ) -> EvidenceMigrationReport: """Rewrite legacy Curated units as table-free v3 Markdown without changing semantics.""" workspace_root = workspace_root.resolve() evidence_root = workspace_root / "evidence" _reject_dirty_authoring_state(workspace_root, git_status or _git_status) try: manifest = load_manifest(evidence_root / "manifest.yaml") documents = load_curated_tree(evidence_root / "curated") except (OSError, ValidationError, ValueError) as error: raise EvidencePreparationError("authoring_state_invalid") from error documents_by_id = {document.id: document for document in documents} if len(documents_by_id) != len(documents): raise EvidencePreparationError("duplicate_evidence_id") migrated = tuple(sorted( document.id for document in documents if document.schema_version in {1, 2} )) unchanged = tuple(sorted( document.id for document in documents if document.schema_version == 3 )) if not migrated: return EvidenceMigrationReport( migrated=(), unchanged=unchanged, findings=validate_workspace_evidence(workspace_root).findings, ) upgraded = { evidence_id: document.model_copy(update={"schema_version": 3}) for evidence_id, document in documents_by_id.items() } findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest) return EvidenceMigrationReport( migrated=migrated, unchanged=unchanged, findings=findings, ) def resolve_workspace_evidence( workspace_root: Path, evidence_id: str, *, retire: bool = False, source: str | Path | None = None, git_status: Callable[[Path], tuple[str, ...]] | None = None, ) -> EvidenceResolutionReport: """Retire or relink one Evidence unit as one staged curator update. The operation intentionally neither creates a commit nor publishes anything. Both the curated tree and the managed manifest are replaced only after the staged tree has been written and structurally validated. """ if retire == (source is not None): raise EvidencePreparationError("resolve_mode_invalid") if not is_evidence_id(evidence_id): raise EvidencePreparationError("evidence_id_invalid") workspace_root = workspace_root.resolve() evidence_root = workspace_root / "evidence" _reject_dirty_worktree(workspace_root, git_status or _git_status) try: manifest = load_manifest(evidence_root / "manifest.yaml") documents = load_curated_tree(evidence_root / "curated") except (OSError, ValidationError, ValueError) as error: raise EvidencePreparationError("authoring_state_invalid") from error documents_by_id = {document.id: document for document in documents} if len(documents_by_id) != len(documents): raise EvidencePreparationError("duplicate_evidence_id") if evidence_id not in documents_by_id: raise EvidencePreparationError("evidence_not_found") source_units = {path: tuple(entry.units) for path, entry in manifest.sources.items()} orphaned = set(manifest.orphans) source_file: str | None = None action: Literal["retired", "relinked"] if retire: action = "retired" del documents_by_id[evidence_id] for path, units in tuple(source_units.items()): source_units[path] = tuple(unit for unit in units if unit != evidence_id) orphaned.discard(evidence_id) else: action = "relinked" assert source is not None source_file = _resolve_source_file(evidence_root, source) source_text = _load_resolution_source(evidence_root, source_file) document = documents_by_id[evidence_id] review_items = document.review_items if all(_normalize(excerpt) in source_text for excerpt in document.provenance.supporting_excerpts): review_items = tuple( item for item in review_items if item.code != "source_no_longer_supports_unit" ) documents_by_id[evidence_id] = document.model_copy(update={ "provenance": document.provenance.model_copy(update={ "source_file": source_file, "source_sha256": _source_hash(source_text), }), "review_items": review_items, }) for path, units in tuple(source_units.items()): source_units[path] = tuple(unit for unit in units if unit != evidence_id) source_units[source_file] = tuple(sorted((*source_units.get(source_file, ()), evidence_id))) orphaned.discard(evidence_id) sources = { path: ManifestSource( sha256=( _source_hash(_load_resolution_source(evidence_root, path)) if path == source_file else manifest.sources[path].sha256 ), units=units, ) for path, units in sorted(source_units.items()) } next_manifest = EvidenceManifest( schema_version=1, pipeline_version=manifest.pipeline_version, sources=sources, orphans=tuple(sorted(orphaned)), ) findings = _stage_and_apply_authoring_tree(workspace_root, documents_by_id, next_manifest) return EvidenceResolutionReport( action=action, evidence_id=evidence_id, source_file=source_file, findings=findings, ) def _empty_manifest() -> EvidenceManifest: return EvidenceManifest( schema_version=1, pipeline_version="evidence-authoring-v1", sources={}, orphans=(), ) def _load_preparation_manifest(path: Path, *, upgrade: bool) -> EvidenceManifest: if not path.is_file(): return _empty_manifest() try: return load_manifest(path) except ValueError as original_error: if not upgrade: raise EvidencePreparationError("pipeline_upgrade_required") from original_error try: raw = yaml.safe_load(path.read_text(encoding="utf-8")) if not isinstance(raw, dict) or raw.get("schema_version") != 1: raise ValueError("manifest shape") raw["pipeline_version"] = "evidence-authoring-v1" return EvidenceManifest.model_validate(raw) except (OSError, UnicodeDecodeError, ValidationError, ValueError, yaml.YAMLError) as error: raise EvidencePreparationError("authoring_state_invalid") from error def _resolve_source_file(evidence_root: Path, source: str | Path) -> str: raw_source = source.as_posix() if isinstance(source, Path) else source try: source_file = validate_source_file(raw_source) except ValueError as error: raise EvidencePreparationError("source_invalid", raw_source) from error path = evidence_root / source_file ancestor = evidence_root for part in Path(source_file).parts: ancestor /= part if ancestor.is_symlink(): raise EvidencePreparationError("source_invalid", source_file) if not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES: raise EvidencePreparationError("source_invalid", source_file) return source_file def _load_resolution_source(evidence_root: Path, source_file: str) -> str: try: return normalize_source_text((evidence_root / source_file).read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError) as error: raise EvidencePreparationError("source_invalid", source_file) from error def _git_status(workspace_root: Path) -> tuple[str, ...]: result = subprocess.run( ["git", "status", "--porcelain", "--untracked-files=all"], cwd=workspace_root, check=False, capture_output=True, text=True, ) if result.returncode != 0: raise EvidencePreparationError("canonical_git_worktree_required") prefix_result = subprocess.run( ["git", "rev-parse", "--show-prefix"], cwd=workspace_root, check=False, capture_output=True, text=True, ) if prefix_result.returncode != 0: raise EvidencePreparationError("canonical_git_worktree_required") prefix = prefix_result.stdout.strip() entries: list[str] = [] for line in result.stdout.splitlines(): if not line: continue path = line[3:] if len(line) > 3 else line if prefix and path.startswith(prefix): line = line[:3] + path.removeprefix(prefix) entries.append(line) return tuple(entries) def _reject_dirty_authoring_state( workspace_root: Path, git_status: Callable[[Path], tuple[str, ...]], ) -> None: for entry in git_status(workspace_root): path = entry[3:] if len(entry) > 3 else entry if path == "evidence/manifest.yaml" or path.startswith("evidence/curated/"): raise EvidencePreparationError("authoring_worktree_dirty") def _reject_dirty_worktree( workspace_root: Path, git_status: Callable[[Path], tuple[str, ...]], ) -> None: if git_status(workspace_root): raise EvidencePreparationError("worktree_dirty") def _load_source_texts(evidence_root: Path) -> dict[str, str]: source_root = evidence_root / "source" if not source_root.is_dir(): return {} sources: dict[str, str] = {} for path in sorted(source_root.rglob("*")): if path.is_dir(): continue relative = path.relative_to(evidence_root).as_posix() try: validate_source_file(relative) if path.is_symlink() or not path.is_file() or path.stat().st_size > MAX_AUTHORING_FILE_BYTES: raise ValueError("unsafe source") sources[relative] = normalize_source_text(path.read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError, ValueError) as error: raise EvidencePreparationError("source_invalid", relative) from error return sources def _source_hash(source: str) -> str: return "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest() def _preserve_removed_sources( source_texts: dict[str, str], manifest: EvidenceManifest, documents_by_id: dict[str, CuratedEvidence], source_units: dict[str, tuple[str, ...]], orphaned: set[str], ) -> None: for source_file in manifest.sources: if source_file in source_texts: continue unit_ids = source_units.pop(source_file, None) if unit_ids is None: continue for unit in unit_ids: if unit in documents_by_id: orphaned.add(unit) def _preserve_uniquely_renamed_sources( source_texts: dict[str, str], manifest: EvidenceManifest, documents_by_id: dict[str, CuratedEvidence], source_units: dict[str, tuple[str, ...]], ) -> set[str]: renamed_sources: set[str] = set() unmatched_sources = [path for path in source_texts if path not in manifest.sources] missing_sources = [path for path in manifest.sources if path not in source_texts] for old_path in missing_sources: matches = [ path for path in unmatched_sources if _source_hash(source_texts[path]) == manifest.sources[old_path].sha256 ] if len(matches) != 1: continue new_path = matches[0] unit_ids = source_units.pop(old_path, ()) source_units[new_path] = unit_ids for unit_id in unit_ids: document = documents_by_id.get(unit_id) if document is None: continue documents_by_id[unit_id] = document.model_copy(update={ "provenance": document.provenance.model_copy(update={"source_file": new_path}), }) unmatched_sources.remove(new_path) renamed_sources.add(new_path) return renamed_sources def _candidate_to_evidence( candidate: RestructureCandidate, evidence_id: str, source_file: str, source_hash: str, ) -> CuratedEvidence: data = candidate.model_dump( mode="json", exclude={"schema_version", "existing_id", "supporting_excerpts"}, ) data["schema_version"] = 3 data["id"] = evidence_id data["provenance"] = { "source_file": source_file, "source_sha256": source_hash, "supporting_excerpts": candidate.supporting_excerpts, } return CuratedEvidence.model_validate(data) def _unsupported_unit( evidence: CuratedEvidence, source_file: str, source_hash: str, ) -> CuratedEvidence: review_items = tuple(item for item in evidence.review_items if item.code != "source_no_longer_supports_unit") review_items += (ReviewItem( code="source_no_longer_supports_unit", message="The current source no longer supports this Evidence unit.", ),) return evidence.model_copy(update={ "schema_version": 3, "provenance": evidence.provenance.model_copy(update={ "source_file": source_file, "source_sha256": source_hash, }), "review_items": review_items, }) def _allocate_evidence_id(title: str, reserved_ids: set[str]) -> str: ascii_title = unicodedata.normalize("NFKD", title).encode("ascii", "ignore").decode("ascii") slug = re.sub(r"[^a-z0-9]+", "-", ascii_title.lower()).strip("-") or "unit" base = f"evidence:{slug}" if base not in reserved_ids: return base suffix = 2 while f"{base}-{suffix}" in reserved_ids: suffix += 1 return f"{base}-{suffix}" def _stage_and_apply_authoring_tree( workspace_root: Path, documents_by_id: dict[str, CuratedEvidence], manifest: EvidenceManifest, ) -> tuple[ValidationFinding, ...]: evidence_root = workspace_root / "evidence" with tempfile.TemporaryDirectory(prefix=".evidence-authoring-", dir=workspace_root) as temporary: staged_workspace = Path(temporary) / "workspace" staged_evidence = staged_workspace / "evidence" if evidence_root.exists(): shutil.copytree(evidence_root, staged_evidence, symlinks=True) else: staged_evidence.mkdir(parents=True) staged_curated = staged_evidence / "curated" if staged_curated.exists(): shutil.rmtree(staged_curated) staged_curated.mkdir() for evidence in documents_by_id.values(): destination = staged_curated / evidence.kind / f"{evidence.id.removeprefix('evidence:')}.md" destination.parent.mkdir(parents=True, exist_ok=True) destination.write_text(dump_curated_markdown(evidence), encoding="utf-8") (staged_evidence / "manifest.yaml").write_text(dump_manifest(manifest), encoding="utf-8") validation = validate_workspace_evidence(staged_workspace) if any(finding.code in {"manifest_invalid", "curated_invalid"} for finding in validation.findings): raise EvidencePreparationError("staged_authoring_state_invalid") backup = Path(temporary) / "previous-evidence" if evidence_root.exists(): os.replace(evidence_root, backup) try: os.replace(staged_evidence, evidence_root) except OSError: if backup.exists(): os.replace(backup, evidence_root) raise return validation.findings