feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s
Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
@@ -22,6 +22,7 @@ from tht.evidence.authoring import (
|
||||
)
|
||||
from tht.evidence.canonical import (
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
load_curated_tree,
|
||||
parse_curated_markdown,
|
||||
@@ -37,6 +38,7 @@ from tht.evidence.contracts import (
|
||||
validate_namespaced_value,
|
||||
validate_safe_metadata,
|
||||
)
|
||||
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive
|
||||
from tht.evidence.preprocessing import EvidenceEmbedder, build_preprocessing_pipeline
|
||||
from tht.evidence.search import (
|
||||
ActiveEvidenceSearcher,
|
||||
@@ -58,6 +60,7 @@ from tht.evidence.sources import build_sources
|
||||
__all__ = [
|
||||
"AcquiredDocument",
|
||||
"ActiveEvidenceSearcher",
|
||||
"ArchiveConflict",
|
||||
"CorpusWorkspaceMismatchError",
|
||||
"CuratedEvidence",
|
||||
"EvidenceEmbedder",
|
||||
@@ -75,6 +78,8 @@ __all__ = [
|
||||
"EvidenceSource",
|
||||
"EvidenceSourceError",
|
||||
"EvidenceSourceErrorCategory",
|
||||
"LocalEvidenceArchive",
|
||||
"ManualEvidenceProvenance",
|
||||
"PiEvidenceRestructurer",
|
||||
"RestructureCandidate",
|
||||
"RestructureRequest",
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Local Evidence browsing and explicit consolidation, independent of DWH access."""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from .canonical import parse_curated_markdown
|
||||
from .local_archive import LocalEvidenceArchive, _content
|
||||
|
||||
|
||||
class ConsolidationError(RuntimeError):
|
||||
def __init__(self, message, *, saved=False):
|
||||
super().__init__(message)
|
||||
self.saved = saved
|
||||
|
||||
|
||||
def consolidate_from_config(config: Path):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import EvidencePreparationError, migrate_workspace_evidence
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ConsolidationError("Local Evidence is not configured for this workspace")
|
||||
root = cfg.evidence.local_archive_root
|
||||
archive = LocalEvidenceArchive(root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
try:
|
||||
# Explicit first consolidation performs the one-time legacy conversion.
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(root)
|
||||
else:
|
||||
archive.initialize()
|
||||
except (ValueError, OSError, EvidencePreparationError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
result = None
|
||||
|
||||
def activate(snapshot):
|
||||
nonlocal result
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise ConsolidationError("Evidence indexing is blocked; check unit size and review items", saved=True)
|
||||
except ConsolidationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Evidence files were saved, but indexing failed. Retry consolidation.", saved=True) from error
|
||||
|
||||
try:
|
||||
archive.consolidate(actor=os.environ.get("THT_PRINCIPAL_SUBJECT") or "installation operator",
|
||||
activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
return result
|
||||
|
||||
|
||||
def browse(root: Path, query: dict):
|
||||
"""Read complete working units and their active status without opening an index."""
|
||||
archive = LocalEvidenceArchive(root)
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
active = archive._snapshot(state["active"]) if state["active"] else None
|
||||
active_units = archive._units(archive._files(active), allow_review=True) if active else {}
|
||||
files = archive._files(archive.evidence)
|
||||
items, errors = [], []
|
||||
seen = set()
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode(), path=Path(relative))
|
||||
if unit.id in seen:
|
||||
raise ValueError("Duplicate Evidence identifier")
|
||||
seen.add(unit.id)
|
||||
old = active_units.get(unit.id)
|
||||
status = "review_required" if unit.review_items else "legacy" if unit.schema_version != 4 \
|
||||
else "active" if old and _content(old[1]) == _content(unit) and old[1].provenance == unit.provenance else "modified" if old else "new"
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative, "status": status,
|
||||
"revision": _content(unit)})
|
||||
except (ValueError, UnicodeError) as error:
|
||||
errors.append({"file": relative, "message": str(error)[:1500]})
|
||||
for identity, (relative, unit) in active_units.items():
|
||||
if identity not in seen:
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative,
|
||||
"status": "invalid" if relative in files else "removed", "revision": _content(unit)})
|
||||
def matches(item):
|
||||
for field in ("kind", "status", "language"):
|
||||
if query.get(field) and item[field] != query[field]:
|
||||
return False
|
||||
if query.get("purpose") and query["purpose"] not in item["purposes"]:
|
||||
return False
|
||||
for key, field in (("concept", "concepts"), ("table", "tables"), ("column", "columns")):
|
||||
if query.get(key) and not any(query[key].casefold() in v.casefold() for v in item["applies_to"][field]):
|
||||
return False
|
||||
provenance = item["provenance"]
|
||||
if query.get("source") and query["source"].casefold() not in str(provenance).casefold():
|
||||
return False
|
||||
return not query.get("q") or query["q"].casefold() in str(item).casefold()
|
||||
selected = [item for item in items if matches(item)]
|
||||
selected.sort(key=lambda item: (str(item.get(query.get("sort", "title"), "")).casefold(), item["id"]),
|
||||
reverse=query.get("direction") == "desc")
|
||||
page, size = int(query.get("page", 1)), int(query.get("page_size", 25))
|
||||
if page < 1 or not 1 <= size <= 100:
|
||||
raise ValueError("Invalid Evidence page")
|
||||
result = {"items": selected[(page-1)*size:page*size], "total": len(selected), "page": page,
|
||||
"page_size": size, "errors": errors, "active_revision": state["active"],
|
||||
"pending_revision": state["pending"], "initialized": bool(state.get("baseline") or state["pending"] or state["active"])}
|
||||
if query.get("id"):
|
||||
result["item"] = next((item for item in items if item["id"] == query["id"]), None)
|
||||
from .imports import reviews
|
||||
result["source_reviews"] = reviews(archive)
|
||||
return result
|
||||
|
||||
|
||||
def source_action(config, *, action, source_id=None, revision=None, decision=None, actor="installation operator"):
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import PiEvidenceRestructurer, authoring_skill_path, migrate_workspace_evidence
|
||||
from .imports import acquisition_sources, decide, refresh
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ValueError("Local Evidence is not configured")
|
||||
archive = LocalEvidenceArchive(cfg.evidence.local_archive_root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(archive.root)
|
||||
else:
|
||||
(archive.evidence / "curated").mkdir(parents=True, exist_ok=True)
|
||||
archive.initialize()
|
||||
if action == "refresh":
|
||||
skill = authoring_skill_path()
|
||||
return refresh(archive, acquisition_sources(cfg), PiEvidenceRestructurer(
|
||||
os.environ.get("THT_PI_EXECUTABLE", "pi"), skill))
|
||||
if action != "decide":
|
||||
raise ValueError("Unknown source action")
|
||||
|
||||
def activate(snapshot):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise RuntimeError("Indexing did not succeed")
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Source decision saved, but indexing failed. Retry the same decision.", saved=True) from error
|
||||
|
||||
try:
|
||||
return decide(archive, source_id=source_id, revision=revision, decision=decision,
|
||||
actor=actor, activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
from .imports import reviews
|
||||
if any(r["id"] == source_id and r["status"] == "applying" for r in reviews(archive)):
|
||||
raise ConsolidationError(f"Source decision saved. {str(error)[:1200]}. Retry the same decision.", saved=True) from error
|
||||
raise
|
||||
@@ -25,6 +25,7 @@ from tht.evidence.canonical import (
|
||||
EvidenceKind,
|
||||
EvidencePurpose,
|
||||
EvidenceScope,
|
||||
ManualEvidenceProvenance,
|
||||
ReviewItem,
|
||||
StrictModel,
|
||||
dump_curated_markdown,
|
||||
@@ -189,6 +190,12 @@ def _restore_exact_source_excerpts(
|
||||
})
|
||||
|
||||
|
||||
def authoring_skill_path() -> Path:
|
||||
"""The installed wheel and the deployment's Pi resources live in different roots."""
|
||||
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
|
||||
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
|
||||
|
||||
class PiEvidenceRestructurer:
|
||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||
|
||||
@@ -389,6 +396,14 @@ def dump_manifest(manifest: EvidenceManifest) -> str:
|
||||
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
|
||||
"""Validate the curated corpus without writing the workspace."""
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").is_file():
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
try:
|
||||
LocalEvidenceArchive(workspace_root).validate()
|
||||
return ValidationReport(())
|
||||
except (OSError, ValueError) as error:
|
||||
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
|
||||
"evidence/curated", str(error)),))
|
||||
findings: list[ValidationFinding] = []
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
if not manifest_path.is_file():
|
||||
@@ -485,6 +500,9 @@ def _validate_manifest_source(
|
||||
def _validate_unit(
|
||||
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
||||
) -> list[ValidationFinding]:
|
||||
if isinstance(evidence.provenance, ManualEvidenceProvenance):
|
||||
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
|
||||
for item in evidence.review_items]
|
||||
if evidence.id in manifest.orphans:
|
||||
return []
|
||||
path = evidence.provenance.source_file
|
||||
@@ -560,6 +578,8 @@ def prepare_workspace_evidence(
|
||||
raise EvidencePreparationError("authoring_workers_invalid")
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
try:
|
||||
@@ -695,10 +715,9 @@ def migrate_workspace_evidence(
|
||||
*,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
) -> EvidenceMigrationReport:
|
||||
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
|
||||
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
documents = load_curated_tree(evidence_root / "curated")
|
||||
@@ -709,7 +728,7 @@ def migrate_workspace_evidence(
|
||||
if len(documents_by_id) != len(documents):
|
||||
raise EvidencePreparationError("duplicate_evidence_id")
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 3})
|
||||
evidence_id: document.model_copy(update={"schema_version": 4})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
migrated_ids: list[str] = []
|
||||
@@ -728,12 +747,16 @@ def migrate_workspace_evidence(
|
||||
migrated = tuple(sorted(migrated_ids))
|
||||
unchanged = tuple(sorted(unchanged_ids))
|
||||
if not migrated:
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=(),
|
||||
unchanged=unchanged,
|
||||
findings=validate_workspace_evidence(workspace_root).findings,
|
||||
)
|
||||
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=migrated,
|
||||
unchanged=unchanged,
|
||||
@@ -762,6 +785,8 @@ def resolve_workspace_evidence(
|
||||
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local/state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
|
||||
_reject_dirty_worktree(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
@@ -1017,7 +1042,7 @@ def _candidate_to_evidence(
|
||||
mode="json",
|
||||
exclude={"schema_version", "existing_id", "supporting_excerpts"},
|
||||
)
|
||||
data["schema_version"] = 3
|
||||
data["schema_version"] = 4
|
||||
data["id"] = evidence_id
|
||||
data["provenance"] = {
|
||||
"source_file": source_file,
|
||||
@@ -1038,7 +1063,7 @@ def _unsupported_unit(
|
||||
message="The current source no longer supports this Evidence unit.",
|
||||
),)
|
||||
return evidence.model_copy(update={
|
||||
"schema_version": 3,
|
||||
"schema_version": 4,
|
||||
"provenance": evidence.provenance.model_copy(update={
|
||||
"source_file": source_file,
|
||||
"source_sha256": source_hash,
|
||||
|
||||
@@ -98,6 +98,19 @@ class ReviewItem(StrictModel):
|
||||
field: str | None = None
|
||||
|
||||
|
||||
class ManualEvidenceProvenance(StrictModel):
|
||||
"""The curator supports the current content; an earlier document is only its origin."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
kind: Literal["manual"] = "manual"
|
||||
declared_by: str = Field(min_length=1)
|
||||
original: EvidenceProvenance | None = None
|
||||
|
||||
@property
|
||||
def source_file(self) -> str:
|
||||
return f"Manual declaration: {self.declared_by}"
|
||||
|
||||
|
||||
class FormulaPayload(StrictModel):
|
||||
concept: str
|
||||
columns: tuple[str, ...]
|
||||
@@ -229,19 +242,23 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
|
||||
|
||||
class CuratedEvidence(StrictModel):
|
||||
schema_version: Literal[1, 2, 3]
|
||||
schema_version: Literal[1, 2, 3, 4]
|
||||
id: str
|
||||
title: str
|
||||
kind: EvidenceKind
|
||||
purposes: tuple[EvidencePurpose, ...]
|
||||
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
||||
language: str
|
||||
provenance: EvidenceProvenance
|
||||
provenance: EvidenceProvenance | ManualEvidenceProvenance
|
||||
review_items: tuple[ReviewItem, ...] = ()
|
||||
payload: EvidencePayload
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _validate_kind_payload(self) -> CuratedEvidence:
|
||||
if isinstance(self.provenance, ManualEvidenceProvenance) and self.schema_version != 4:
|
||||
raise ValueError("manual declarations require Curated unit schema v4")
|
||||
if self.schema_version == 4 and (not self.purposes or not self.language.strip()):
|
||||
raise ValueError("editable Evidence requires a language and at least one purpose")
|
||||
if not is_evidence_id(self.id):
|
||||
raise ValueError("id must use the evidence:<slug> form")
|
||||
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
|
||||
@@ -1056,13 +1073,19 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
_, frontmatter, body = text.split("---\n", 2)
|
||||
except ValueError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
try:
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
except yaml.YAMLError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
try:
|
||||
data = dict(raw)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
||||
if data.get("schema_version") == 2:
|
||||
data = _parse_v2_body(data, body)
|
||||
elif data.get("schema_version") == 4:
|
||||
from tht.evidence.editable import parse_document, parse_metadata
|
||||
data = parse_document(parse_metadata(frontmatter), body)
|
||||
else:
|
||||
if body.strip():
|
||||
raise ValueError("curated evidence must not contain an ignored body")
|
||||
@@ -1081,6 +1104,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
|
||||
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
||||
"""Render one canonical Curated Evidence Markdown document."""
|
||||
if value.schema_version == 4:
|
||||
from tht.evidence.editable import render_document
|
||||
return render_document(value)
|
||||
if value.schema_version == 3:
|
||||
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
|
||||
if value.schema_version == 2:
|
||||
|
||||
@@ -8,7 +8,12 @@ from typing import Self
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
|
||||
|
||||
from tht.evidence.canonical import EVIDENCE_KINDS, EVIDENCE_PURPOSES
|
||||
from tht.evidence.canonical import (
|
||||
EVIDENCE_KINDS,
|
||||
EVIDENCE_PURPOSES,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
)
|
||||
from tht.evidence.contracts import (
|
||||
canonical_provenance_uri,
|
||||
normalize_aware_datetime,
|
||||
@@ -76,10 +81,10 @@ def _validate_evidence_metadata(metadata: Mapping[str, JsonValue]) -> None:
|
||||
if not isinstance(metadata["language"], str) or not metadata["language"]:
|
||||
raise ValueError("typed Evidence metadata must contain language")
|
||||
provenance = metadata["provenance"]
|
||||
if not isinstance(provenance, dict) or set(provenance) != {
|
||||
"source_file", "source_sha256", "supporting_excerpts",
|
||||
}:
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance")
|
||||
if not isinstance(provenance, dict):
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance") # noqa: TRY004
|
||||
model = ManualEvidenceProvenance if provenance.get("kind") == "manual" else EvidenceProvenance
|
||||
model.model_validate(provenance)
|
||||
|
||||
|
||||
class _CanonicalValue(BaseModel):
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
"""Curated unit v4: visible Markdown fields are the sole human-content authority."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
_PAYLOAD_TYPE_BY_KIND,
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
_v2_labels,
|
||||
)
|
||||
|
||||
_LIST_FIELDS = {"synonyms", "variants", "columns", "tables"}
|
||||
|
||||
|
||||
def parse_metadata(text: str) -> dict:
|
||||
class UniqueKeysLoader(yaml.SafeLoader):
|
||||
pass
|
||||
|
||||
def mapping(loader, node):
|
||||
pairs = loader.construct_pairs(node, deep=True)
|
||||
result = {}
|
||||
for key, value in pairs:
|
||||
if key in result:
|
||||
raise ValueError(f"Duplicate metadata key: {key}")
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping)
|
||||
try:
|
||||
return yaml.load(text, Loader=UniqueKeysLoader)
|
||||
except (yaml.YAMLError, TypeError) as error:
|
||||
raise ValueError("Evidence metadata is malformed") from error
|
||||
|
||||
|
||||
def _sections(body: str, headings: dict[str, str]) -> dict[str, str]:
|
||||
"""Recognize structural H2s outside code fences; all other Markdown is content."""
|
||||
sections: dict[str, list[str]] = {}
|
||||
field = None
|
||||
fence = None
|
||||
for line in body.splitlines():
|
||||
match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line)
|
||||
if match:
|
||||
marker = match[1]
|
||||
if fence is None:
|
||||
fence = marker
|
||||
elif marker[0] == fence[0] and len(marker) >= len(fence):
|
||||
fence = None
|
||||
heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None
|
||||
if heading:
|
||||
if heading in sections:
|
||||
raise ValueError(f"Duplicate section: {line[3:]}")
|
||||
field = heading
|
||||
sections[field] = []
|
||||
elif field is not None:
|
||||
sections[field].append(line)
|
||||
elif line.strip():
|
||||
raise ValueError("Content must follow a documented section heading")
|
||||
if fence:
|
||||
raise ValueError("Unclosed Markdown code fence")
|
||||
return {key: "\n".join(lines).strip() for key, lines in sections.items()}
|
||||
|
||||
|
||||
def _list(text: str) -> list[str]:
|
||||
if not text:
|
||||
return []
|
||||
values = []
|
||||
for line in text.splitlines():
|
||||
if not line.startswith("- ") or not line[2:].strip():
|
||||
raise ValueError("List entries must use '- value', one per line")
|
||||
value = line[2:]
|
||||
# Quoted strings preserve multiline and unusual values during migration.
|
||||
values.append(json.loads(value) if value.startswith('"') else value)
|
||||
return values
|
||||
|
||||
|
||||
def _values(text: str) -> dict[str, str]:
|
||||
values = {}
|
||||
key = None
|
||||
lines = []
|
||||
for line in text.splitlines():
|
||||
if line.startswith("### "):
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
label = line[4:]
|
||||
key = json.loads(label) if label.startswith('"') else label
|
||||
if key in values:
|
||||
raise ValueError("Duplicate enum value")
|
||||
lines = []
|
||||
elif key is None:
|
||||
if line.strip():
|
||||
raise ValueError("Enum values require '### value' headings")
|
||||
else:
|
||||
lines.append(line)
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
return values
|
||||
|
||||
|
||||
def parse_document(metadata: dict, body: str) -> dict:
|
||||
data = dict(metadata)
|
||||
if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data):
|
||||
raise ValueError("Title and payload must be edited only in the Markdown body")
|
||||
lines = body.strip().splitlines()
|
||||
if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip():
|
||||
raise ValueError("A title starting with '# ' is required")
|
||||
data["title"] = lines[0][2:].strip()
|
||||
kind = data.get("kind")
|
||||
if kind not in _PAYLOAD_TYPE_BY_KIND:
|
||||
raise ValueError("Unknown Evidence kind")
|
||||
labels = _v2_labels(str(data.get("language", "")))
|
||||
fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields
|
||||
sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields})
|
||||
required = {name for name, field in fields.items() if field.is_required()}
|
||||
if not required <= sections.keys():
|
||||
raise ValueError(
|
||||
"Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys()))
|
||||
)
|
||||
payload = {}
|
||||
for key, content in sections.items():
|
||||
if key in _LIST_FIELDS:
|
||||
payload[key] = _list(content)
|
||||
elif key == "values":
|
||||
payload[key] = _values(content)
|
||||
elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"):
|
||||
payload[key] = content[7:-4]
|
||||
else:
|
||||
payload[key] = content
|
||||
if fields[key].is_required() and not payload[key] and key != "values":
|
||||
raise ValueError(f"Section {labels[key]} must not be empty")
|
||||
data["payload"] = payload
|
||||
data.setdefault(
|
||||
"provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump()
|
||||
)
|
||||
return data
|
||||
|
||||
|
||||
def render_document(value: CuratedEvidence) -> str:
|
||||
metadata = value.model_dump(mode="json", exclude={"title", "payload"})
|
||||
labels = _v2_labels(value.language)
|
||||
parts = [
|
||||
f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}"
|
||||
]
|
||||
for key, content in value.payload.model_dump(mode="json").items():
|
||||
if key in _LIST_FIELDS:
|
||||
rendered = "\n".join(
|
||||
"- "
|
||||
+ (
|
||||
json.dumps(v, ensure_ascii=False)
|
||||
if "\n" in v or v.startswith('"') or v != v.strip()
|
||||
else v
|
||||
)
|
||||
for v in content
|
||||
)
|
||||
elif key == "values":
|
||||
rendered = "\n\n".join(
|
||||
f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items()
|
||||
)
|
||||
elif key == "sql":
|
||||
rendered = f"```sql\n{content}\n```"
|
||||
else:
|
||||
rendered = content
|
||||
parts.append(f"## {labels[key]}\n\n{rendered}")
|
||||
rendered = "\n\n".join(parts) + "\n"
|
||||
# Migration must fail explicitly rather than silently changing unrepresentable content.
|
||||
restored = CuratedEvidence.model_validate(
|
||||
parse_document(metadata, rendered.split("---\n", 2)[2])
|
||||
)
|
||||
if restored != value:
|
||||
raise ValueError(
|
||||
f"{value.id}: content cannot be represented losslessly in editable Markdown"
|
||||
)
|
||||
return rendered
|
||||
@@ -0,0 +1,235 @@
|
||||
"""Explicit acquisition and durable source comparisons; never implicit runtime refresh."""
|
||||
|
||||
import base64
|
||||
import json
|
||||
|
||||
from .authoring import (
|
||||
RestructureRequest,
|
||||
_allocate_evidence_id,
|
||||
_candidate_to_evidence,
|
||||
normalize_source_text,
|
||||
)
|
||||
from .canonical import (
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
)
|
||||
from .corpus.normalize import _decode
|
||||
from .local_archive import ArchiveConflict, LocalEvidenceArchive, _atomic, _digest
|
||||
|
||||
MAX_DOCUMENTS = 200
|
||||
MAX_TOTAL_BYTES = 100 * 1024 * 1024
|
||||
|
||||
|
||||
def _origin(unit):
|
||||
return unit.provenance.original if isinstance(unit.provenance, ManualEvidenceProvenance) else unit.provenance
|
||||
|
||||
|
||||
def _records(archive):
|
||||
path = archive.metadata / "sources.json"
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source metadata must not use symlinks")
|
||||
return json.loads(path.read_text()) if path.exists() else {}
|
||||
|
||||
|
||||
def _write_records(archive, records):
|
||||
_atomic(archive.metadata / "sources.json", json.dumps(records, ensure_ascii=False, sort_keys=True))
|
||||
|
||||
|
||||
def reviews(archive):
|
||||
return [{k: v for k, v in row.items() if k not in {"expected", "text", "writes", "approved"}}
|
||||
for row in _records(archive).values()]
|
||||
|
||||
|
||||
def acquisition_sources(cfg):
|
||||
"""Local drafts and original files, plus configured read-only remote connectors."""
|
||||
from .adapters import FilesystemEvidenceSource
|
||||
from .sources import build_sources
|
||||
|
||||
root = cfg.evidence.local_archive_root / "evidence"
|
||||
# Canonical acquired versions are immutable lineage, never new input documents.
|
||||
patterns = [str(p.relative_to(root)) for folder in ("incoming", "source")
|
||||
for p in sorted((root / folder).rglob("*.md"))
|
||||
if not p.is_relative_to(root / "source/acquired")]
|
||||
result = [FilesystemEvidenceSource(root, patterns=patterns)] if patterns else []
|
||||
# Filesystem descriptors select the installation's local authoring tree after E2.
|
||||
remote = cfg.evidence.model_copy(update={"source_root": None,
|
||||
"sources": [s for s in cfg.evidence.sources if s.type != "filesystem"]})
|
||||
result.extend(build_sources(remote, acquisition=True))
|
||||
return result
|
||||
|
||||
|
||||
def refresh(archive: LocalEvidenceArchive, sources, restructurer):
|
||||
"""Acquire everything successfully before recording proposals. Missing is never deletion."""
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation before refreshing sources")
|
||||
records = _records(archive)
|
||||
if any(r["status"] == "applying" for r in records.values()):
|
||||
raise ArchiveConflict("Retry the pending source decision before refreshing")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
documents, total = {}, 0
|
||||
for adapter in sources:
|
||||
for item in adapter.discover():
|
||||
document = adapter.acquire(item)
|
||||
total += len(document.content)
|
||||
if len(documents) >= MAX_DOCUMENTS or total > MAX_TOTAL_BYTES:
|
||||
raise ValueError("Source refresh exceeds the local acquisition limit")
|
||||
relative = item.metadata.get("relative_path") if item.uri.startswith("file:") else None
|
||||
identity = f"local:{relative}" if relative else item.uri
|
||||
key = _digest(identity.encode())
|
||||
if key in documents:
|
||||
raise ValueError("Duplicate acquisition identity")
|
||||
documents[key] = (document, relative)
|
||||
reserved = set(units) | set(state["deleted_ids"])
|
||||
for record in records.values():
|
||||
reserved.update(u["id"] for u in record.get("proposed", []))
|
||||
changed, unchanged = 0, 0
|
||||
acquired = {}
|
||||
for key, (document, relative) in documents.items():
|
||||
text = normalize_source_text(_decode(document))
|
||||
sha = "sha256:" + _digest(text.encode())
|
||||
old = records.get(key)
|
||||
stale = old and old["status"] == "review" and any(
|
||||
p not in files or _digest(files[p]) != h for p, h in old["expected"].items())
|
||||
if old and old["sha256"] == sha and not stale:
|
||||
old["availability"] = "available"
|
||||
unchanged += 1
|
||||
continue
|
||||
current = {i: pair for i, pair in units.items()
|
||||
if (old and i in old["unit_ids"]) or
|
||||
(_origin(pair[1]) and _origin(pair[1]).source_file == relative)}
|
||||
# Seed imported E2 document identity without asking the model to recurate unchanged text.
|
||||
if old is None and current and all(_origin(u).source_sha256 == sha for _, u in current.values()):
|
||||
records[key] = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "unit_ids": sorted(current), "status": "accepted",
|
||||
"availability": "available", "revision": sha[7:], "proposed": []}
|
||||
unchanged += 1
|
||||
continue
|
||||
source_file = f"source/acquired/{key}/{sha[7:]}.md"
|
||||
request = RestructureRequest(source_file=source_file, source_sha256=sha,
|
||||
normalized_text=text, previous_units=tuple(u for _, u in current.values()))
|
||||
proposed = []
|
||||
seen = set()
|
||||
suppressed = relative in state["suppressed_sources"] or any(
|
||||
p.startswith(f"source/acquired/{key}/") for p in state["suppressed_sources"]) or (old and (
|
||||
old.get("suppressed", False) or any(i in state["deleted_ids"] for i in old["unit_ids"])))
|
||||
for candidate in restructurer.restructure(request):
|
||||
identity = candidate.existing_id
|
||||
if identity in state["deleted_ids"]:
|
||||
continue
|
||||
if identity is not None and identity not in current:
|
||||
raise ValueError("Source proposal refers to an unrelated Evidence identity")
|
||||
if identity is None:
|
||||
if suppressed:
|
||||
continue # A model-created identifier cannot bypass a curated deletion.
|
||||
identity = _allocate_evidence_id(candidate.title, reserved)
|
||||
reserved.add(identity)
|
||||
if identity in seen:
|
||||
raise ValueError("Source proposal repeats an Evidence identity")
|
||||
seen.add(identity)
|
||||
unit = _candidate_to_evidence(candidate, identity, source_file, sha)
|
||||
dump_curated_markdown(unit) # Refuse an uneditable proposal before saving any review.
|
||||
if any(excerpt not in text for excerpt in unit.provenance.supporting_excerpts):
|
||||
raise ValueError("Source proposal contains an excerpt absent from the acquired document")
|
||||
proposed.append(unit.model_dump(mode="json"))
|
||||
expected = {p: _digest(files[p]) for p, _ in current.values()}
|
||||
row = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "source_file": source_file, "text": text,
|
||||
"status": "review", "availability": "available", "suppressed": bool(suppressed),
|
||||
"unit_ids": sorted(current), "expected": expected,
|
||||
"current": [u.model_dump(mode="json") for _, u in current.values()], "proposed": proposed,
|
||||
"removed_ids": sorted(set(current) - seen)}
|
||||
row["revision"] = _digest(json.dumps(row, sort_keys=True).encode())
|
||||
records[key] = row
|
||||
acquired[key] = {"source": document.source.model_dump(mode="json"),
|
||||
"media_type": document.media_type, "raw_base64": base64.b64encode(document.content).decode()}
|
||||
changed += 1
|
||||
for key, row in records.items():
|
||||
if key not in documents:
|
||||
row["availability"] = "missing"
|
||||
# No writes above: an access/model failure preserves every previous review and active unit.
|
||||
for key, value in acquired.items():
|
||||
directory = archive.metadata / "acquisitions" / key
|
||||
if any(p.is_symlink() for p in [directory, directory.parent]):
|
||||
raise ValueError("Acquisition metadata must not use symlinks")
|
||||
_atomic(directory / f"{records[key]['sha256'][7:]}.json", json.dumps(value, ensure_ascii=False))
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"changed": changed, "unchanged": unchanged,
|
||||
"review": sum(r["status"] == "review" for r in records.values())}}
|
||||
|
||||
|
||||
def decide(archive, *, source_id, revision, decision, actor, activate):
|
||||
if decision not in {"keep", "replace"} or not actor.strip():
|
||||
raise ValueError("Choose keep or replace and supply a curator")
|
||||
with archive.operation():
|
||||
records = _records(archive)
|
||||
row = records.get(source_id)
|
||||
if row is None or row["revision"] != revision:
|
||||
raise ArchiveConflict("Source comparison changed; refresh the page")
|
||||
if row["status"] == "applying":
|
||||
if row["decision"] != decision:
|
||||
raise ArchiveConflict("Retry the saved source decision before changing it")
|
||||
state = archive._state()
|
||||
state.update(import_writes=row["writes"], approved_imports=row["approved"])
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
else:
|
||||
if row["status"] != "review":
|
||||
raise ArchiveConflict("This source comparison was already decided")
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation first")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
if any(_digest(files[p]) != h if p in files else True for p, h in row["expected"].items()):
|
||||
raise ArchiveConflict("Curated files changed since source review; refresh the source comparison")
|
||||
selected = [CuratedEvidence.model_validate(v) for v in row["proposed"]] if decision == "replace" else [units[i][1] for i in row["unit_ids"]]
|
||||
writes, approved = {}, {}
|
||||
def write(path, content):
|
||||
current = archive.evidence / path
|
||||
writes[path] = {"before": _digest(current.read_bytes()) if current.exists() else None, "after": content}
|
||||
if decision == "replace":
|
||||
write(row["source_file"], row["text"])
|
||||
for identity in row["unit_ids"]:
|
||||
write(units[identity][0], None)
|
||||
for value in selected:
|
||||
if value.review_items:
|
||||
raise ValueError("The proposal needs review; correct the input draft and refresh, or keep local content")
|
||||
if decision == "keep":
|
||||
origin = _origin(value)
|
||||
if origin:
|
||||
old = archive._snapshot(state["active"] or state["baseline"])
|
||||
candidate = old / origin.source_file
|
||||
if not candidate.is_file():
|
||||
candidate = archive.evidence / origin.source_file
|
||||
text = normalize_source_text(candidate.read_text())
|
||||
if "sha256:" + _digest(text.encode()) != origin.source_sha256:
|
||||
raise ValueError("The original source version is unavailable")
|
||||
path = f"source/acquired/{source_id}/{origin.source_sha256[7:]}.md"
|
||||
write(path, text)
|
||||
origin = EvidenceProvenance(source_file=path, source_sha256=origin.source_sha256,
|
||||
supporting_excerpts=origin.supporting_excerpts)
|
||||
value = value.model_copy(update={"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)})
|
||||
if value.id in units and value.id not in row["unit_ids"]:
|
||||
raise ArchiveConflict("A proposed Evidence identity was created elsewhere")
|
||||
path = units[value.id][0] if value.id in units and units[value.id][1].kind == value.kind else f"curated/{value.kind}/{value.id[9:]}.md"
|
||||
if path in files and value.id not in units:
|
||||
raise ArchiveConflict("The proposed file path is occupied")
|
||||
write(path, dump_curated_markdown(value))
|
||||
approved[value.id] = _digest(dump_curated_markdown(value).encode())
|
||||
# Journal before applying files; retry never silently clobbers an external edit.
|
||||
row.update(status="applying", decision=decision, decided_by=actor,
|
||||
next_unit_ids=[u.id for u in selected], writes=writes, approved=approved)
|
||||
_write_records(archive, records)
|
||||
state.update(import_writes=writes, approved_imports=approved)
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
if result["status"] != "active":
|
||||
raise ValueError("Source decision was saved but not activated; retry")
|
||||
row.update(status="accepted" if decision == "replace" else "kept", unit_ids=row["next_unit_ids"])
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"units": result["units"]}}
|
||||
@@ -0,0 +1,472 @@
|
||||
"""Persistent curated files, immutable consolidation candidates and explicit activation.
|
||||
|
||||
The working tree is primary data. Core consumers use only active_snapshot(); an
|
||||
editor save or failed activation never switches that pointer. No Git/network/DWH I/O.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import fcntl
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
MAX_CURATED_FILE_BYTES,
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
parse_curated_markdown,
|
||||
)
|
||||
|
||||
|
||||
class ArchiveConflict(ValueError):
|
||||
"""The curator must reconcile a concurrent change before replacing it."""
|
||||
|
||||
|
||||
def _digest(value: bytes) -> str:
|
||||
return hashlib.sha256(value).hexdigest()
|
||||
|
||||
|
||||
def _content(unit: CuratedEvidence) -> str:
|
||||
return _digest(
|
||||
json.dumps(
|
||||
unit.model_dump(mode="json", exclude={"schema_version", "provenance"}),
|
||||
sort_keys=True,
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
)
|
||||
|
||||
|
||||
def _atomic(path: Path, data: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w") as handle:
|
||||
handle.write(data)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
os.replace(temporary, path)
|
||||
finally:
|
||||
if os.path.exists(temporary):
|
||||
os.unlink(temporary)
|
||||
|
||||
|
||||
class LocalEvidenceArchive:
|
||||
def __init__(self, workspace_root: Path):
|
||||
self.root = workspace_root.resolve()
|
||||
self.evidence = self.root / "evidence"
|
||||
self.metadata = self.evidence / ".local"
|
||||
if self.evidence.is_symlink() or self.metadata.is_symlink():
|
||||
raise ValueError("The local Evidence archive must use persistent regular directories")
|
||||
|
||||
@contextmanager
|
||||
def operation(self):
|
||||
if any(
|
||||
path.is_symlink()
|
||||
for path in (
|
||||
self.evidence,
|
||||
self.metadata,
|
||||
self.metadata / "snapshots",
|
||||
self.metadata / "state.yaml",
|
||||
)
|
||||
):
|
||||
raise ValueError("Evidence archive metadata must not use symlinks")
|
||||
self.root.mkdir(parents=True, exist_ok=True)
|
||||
lock = self.root / ".evidence-archive.lock"
|
||||
if lock.is_symlink():
|
||||
raise ValueError("Evidence lock must not be a symlink")
|
||||
with lock.open("a") as handle:
|
||||
fcntl.flock(handle, fcntl.LOCK_EX)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
fcntl.flock(handle, fcntl.LOCK_UN)
|
||||
|
||||
def _state(self):
|
||||
path = self.metadata / "state.yaml"
|
||||
if not path.exists():
|
||||
return {
|
||||
"schema_version": 1,
|
||||
"active": None,
|
||||
"pending": None,
|
||||
"baseline": None,
|
||||
"deleted_ids": [],
|
||||
"suppressed_sources": [],
|
||||
}
|
||||
value = yaml.safe_load(path.read_text())
|
||||
if not isinstance(value, dict) or value.get("schema_version") != 1:
|
||||
raise ValueError("Unsupported local Evidence archive state")
|
||||
return value
|
||||
|
||||
def _write_state(self, state):
|
||||
_atomic(self.metadata / "state.yaml", yaml.safe_dump(state, sort_keys=True))
|
||||
|
||||
def _snapshot(self, revision: str) -> Path:
|
||||
if (
|
||||
not isinstance(revision, str)
|
||||
or len(revision) != 64
|
||||
or any(c not in "0123456789abcdef" for c in revision)
|
||||
):
|
||||
raise ValueError("Invalid Evidence snapshot identity")
|
||||
path = self.metadata / "snapshots" / revision
|
||||
if not path.is_dir() or path.is_symlink():
|
||||
raise ValueError("Consolidated Evidence snapshot is missing")
|
||||
return path
|
||||
|
||||
def active_snapshot(self) -> Path | None:
|
||||
"""Only this immutable source is eligible for core consumption."""
|
||||
with self.operation():
|
||||
revision = self._state()["active"]
|
||||
return self._snapshot(revision) if revision else None
|
||||
|
||||
def validate(self):
|
||||
"""Check the working files without modifying them or changing active content."""
|
||||
with self.operation():
|
||||
units = self._units(self._files(self.evidence))
|
||||
state = self._state()
|
||||
revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous = (
|
||||
self._units(self._files(self._snapshot(revision)), allow_review=True)
|
||||
if revision
|
||||
else {}
|
||||
)
|
||||
for identity, (_, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
if isinstance(unit.provenance, EvidenceProvenance) and (
|
||||
old is None or _content(old[1]) == _content(unit)
|
||||
):
|
||||
self._validate_document_source(unit.provenance)
|
||||
return len(units)
|
||||
|
||||
def _files(self, root: Path):
|
||||
result = {}
|
||||
curated = root / "curated"
|
||||
if curated.is_symlink():
|
||||
raise ValueError("Curated directory must not be a symlink")
|
||||
if not curated.is_dir():
|
||||
raise ValueError(f"{curated}: curated archive is unavailable; absence is not deletion")
|
||||
for path in sorted(curated.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError(f"{path}: symlinks are not supported in the curated archive")
|
||||
if path.suffix != ".md" or path.name.upper().startswith("README"):
|
||||
continue
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: Evidence file exceeds the size limit")
|
||||
result[path.relative_to(root).as_posix()] = path.read_bytes()
|
||||
return result
|
||||
|
||||
def _units(self, files, *, allow_review=False):
|
||||
units = {}
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode("utf-8"), path=Path(relative))
|
||||
if unit.schema_version != 4:
|
||||
raise ValueError("Run evidence migrate before consolidating legacy units")
|
||||
if unit.id in units:
|
||||
raise ValueError(f"Duplicate Evidence identity {unit.id}")
|
||||
if unit.review_items and not allow_review:
|
||||
raise ValueError("Resolve review items before consolidation")
|
||||
units[unit.id] = (relative, unit)
|
||||
except ValueError as error:
|
||||
raise ValueError(f"{relative}: {error}") from error
|
||||
return units
|
||||
|
||||
def initialize(self):
|
||||
"""Capture migrated/refined content before edits, without activating review items."""
|
||||
with self.operation():
|
||||
state = self._state()
|
||||
if state.get("baseline") or state["active"] or state["pending"]:
|
||||
return
|
||||
files = self._files(self.evidence)
|
||||
self._units(files, allow_review=True)
|
||||
source = self.evidence / "source"
|
||||
if source.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if source.exists():
|
||||
for path in sorted(source.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if path.is_file():
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: source exceeds the size limit")
|
||||
files[path.relative_to(self.evidence).as_posix()] = path.read_bytes()
|
||||
revision = _digest(b"".join(k.encode() + b"\0" + v for k, v in sorted(files.items())))
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".baseline-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
state["baseline"] = revision
|
||||
self._write_state(state)
|
||||
|
||||
def consolidate(self, *, actor: str, activate: Callable[[Path], None] | None = None):
|
||||
"""Persist one valid candidate; optionally activate it through the existing index stage.
|
||||
|
||||
A missing callback deliberately reports pending_activation, never success.
|
||||
A retry of unchanged files reuses the same candidate after an index failure.
|
||||
"""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._consolidate(actor, activate)
|
||||
|
||||
def _consolidate(self, actor, activate):
|
||||
state = self._state()
|
||||
self._finish_import(state)
|
||||
self._finish_normalization(state)
|
||||
original_files = self._files(self.evidence)
|
||||
units = self._units(original_files)
|
||||
previous_revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous_root = self._snapshot(previous_revision) if previous_revision else None
|
||||
previous = (
|
||||
self._units(self._files(previous_root), allow_review=True) if previous_root else {}
|
||||
)
|
||||
deleted = set(state["deleted_ids"]) | (previous.keys() - units.keys())
|
||||
suppressed = set(state["suppressed_sources"])
|
||||
for identity in previous.keys() - units.keys():
|
||||
provenance = previous[identity][1].provenance
|
||||
source = (
|
||||
provenance.original
|
||||
if isinstance(provenance, ManualEvidenceProvenance)
|
||||
else provenance
|
||||
)
|
||||
if source:
|
||||
suppressed.add(source.source_file)
|
||||
files = {}
|
||||
for identity, (relative, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
changed = old is not None and _content(old[1]) != _content(unit)
|
||||
approved = state.get("approved_imports", {}).get(identity) == _digest(
|
||||
dump_curated_markdown(unit).encode()
|
||||
)
|
||||
if approved:
|
||||
pass # An explicit source decision authorized this exact content and provenance.
|
||||
elif changed or (
|
||||
isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
and (old is None or unit.provenance.declared_by == "local curator")
|
||||
):
|
||||
origin = old[1].provenance if old else unit.provenance
|
||||
origin = origin.original if isinstance(origin, ManualEvidenceProvenance) else origin
|
||||
unit = unit.model_copy(
|
||||
update={
|
||||
"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)
|
||||
}
|
||||
)
|
||||
elif old and unit.provenance != old[1].provenance:
|
||||
raise ArchiveConflict(
|
||||
f"{relative}: provenance is managed; edit the content instead"
|
||||
)
|
||||
if isinstance(unit.provenance, EvidenceProvenance):
|
||||
self._validate_document_source(unit.provenance)
|
||||
files[relative] = dump_curated_markdown(unit).encode()
|
||||
origin = (
|
||||
unit.provenance.original
|
||||
if isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
else unit.provenance
|
||||
)
|
||||
if origin:
|
||||
candidates = [previous_root / origin.source_file] if previous_root else []
|
||||
candidates.append(self.evidence / origin.source_file)
|
||||
from .authoring import normalize_source_text
|
||||
|
||||
for source in candidates:
|
||||
if source.is_file() and not source.is_symlink():
|
||||
if source.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{source}: source exceeds the size limit")
|
||||
raw = source.read_bytes()
|
||||
if (
|
||||
"sha256:" + _digest(normalize_source_text(raw.decode()).encode())
|
||||
== origin.source_sha256
|
||||
):
|
||||
if origin.source_file in files and files[origin.source_file] != raw:
|
||||
raise ArchiveConflict(
|
||||
"Different source revisions require explicit source resolution"
|
||||
)
|
||||
files[origin.source_file] = raw
|
||||
break
|
||||
else:
|
||||
raise ValueError(
|
||||
f"{origin.source_file}: the recorded original document is unavailable"
|
||||
)
|
||||
# Record current declarations and original document lineage distinctly.
|
||||
manifest = {
|
||||
"schema_version": 1,
|
||||
"units": {
|
||||
identity: {
|
||||
"file": relative,
|
||||
"content_hash": _content(parse_curated_markdown(files[relative].decode())),
|
||||
"provenance": parse_curated_markdown(
|
||||
files[relative].decode()
|
||||
).provenance.model_dump(mode="json"),
|
||||
}
|
||||
for identity, (relative, _) in sorted(units.items())
|
||||
},
|
||||
"deleted_ids": sorted(deleted),
|
||||
"suppressed_sources": sorted(suppressed),
|
||||
}
|
||||
files["local-manifest.yaml"] = yaml.safe_dump(
|
||||
manifest, allow_unicode=True, sort_keys=True
|
||||
).encode()
|
||||
revision = _digest(
|
||||
b"".join(path.encode() + b"\0" + data + b"\0" for path, data in sorted(files.items()))
|
||||
)
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".candidate-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
if self._files(self.evidence) != original_files:
|
||||
raise ArchiveConflict(
|
||||
"Evidence files changed during consolidation; retry with the current files"
|
||||
)
|
||||
# The candidate exists first. Pending state makes every subsequent interruption recoverable.
|
||||
state.update(
|
||||
pending=revision,
|
||||
deleted_ids=sorted(deleted),
|
||||
suppressed_sources=sorted(suppressed),
|
||||
normalization={relative: _digest(data) for relative, data in original_files.items()},
|
||||
)
|
||||
state.pop("approved_imports", None)
|
||||
self._write_state(state)
|
||||
self._finish_normalization(state)
|
||||
if activate is None:
|
||||
return {
|
||||
"status": "pending_activation",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
}
|
||||
activate(destination)
|
||||
state.update(active=revision, pending=None)
|
||||
self._write_state(state)
|
||||
return {
|
||||
"status": "active",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
"units": len(units),
|
||||
"deleted": len(previous.keys() - units.keys()),
|
||||
}
|
||||
|
||||
def _finish_import(self, state):
|
||||
"""Replay an explicit source decision, rejecting intervening external edits."""
|
||||
writes = state.get("import_writes")
|
||||
if writes is None:
|
||||
return
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if not relative.startswith(("curated/", "source/acquired/")) or ".." in Path(relative).parts:
|
||||
raise ValueError("Invalid import destination")
|
||||
if any(p.is_symlink() for p in [path, *path.parents] if p != self.root.parent):
|
||||
raise ValueError("Import destinations must not use symlinks")
|
||||
current = _digest(path.read_bytes()) if path.exists() else None
|
||||
after = _digest(change["after"].encode()) if change["after"] is not None else None
|
||||
if current not in (change["before"], after):
|
||||
raise ArchiveConflict("Evidence changed during a source decision; restore or review the file")
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if change["after"] is None:
|
||||
path.unlink(missing_ok=True)
|
||||
else:
|
||||
_atomic(path, change["after"])
|
||||
del state["import_writes"]
|
||||
self._write_state(state)
|
||||
|
||||
def _finish_normalization(self, state):
|
||||
"""Replay interrupted managed writes only where the user's bytes are unchanged."""
|
||||
if "normalization" not in state:
|
||||
return
|
||||
snapshot = self._snapshot(state["pending"])
|
||||
current = self._files(self.evidence)
|
||||
for relative, original_hash in state["normalization"].items():
|
||||
if relative in current and _digest(current[relative]) == original_hash:
|
||||
_atomic(self.evidence / relative, (snapshot / relative).read_text())
|
||||
_atomic(
|
||||
self.evidence / "local-manifest.yaml", (snapshot / "local-manifest.yaml").read_text()
|
||||
)
|
||||
del state["normalization"]
|
||||
self._write_state(state)
|
||||
|
||||
def _validate_document_source(self, provenance):
|
||||
from .authoring import _normalize, normalize_source_text
|
||||
|
||||
path = self.evidence / provenance.source_file
|
||||
if (
|
||||
path.is_symlink()
|
||||
or not path.is_file()
|
||||
or not path.resolve().is_relative_to(self.evidence.resolve())
|
||||
):
|
||||
raise ValueError(f"{provenance.source_file}: source document is unavailable")
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{provenance.source_file}: source exceeds the size limit")
|
||||
source = normalize_source_text(path.read_text())
|
||||
if "sha256:" + _digest(source.encode()) != provenance.source_sha256:
|
||||
raise ArchiveConflict(
|
||||
f"{provenance.source_file}: source changed; explicit source refresh is required"
|
||||
)
|
||||
if any(_normalize(excerpt) not in source for excerpt in provenance.supporting_excerpts):
|
||||
raise ValueError(f"{provenance.source_file}: supporting excerpt is missing")
|
||||
|
||||
def save(self, value: CuratedEvidence, *, expected_revision: str | None, actor: str):
|
||||
"""Explicit workflow correction with optimistic concurrency; activation is a separate boundary."""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._save(value, expected_revision=expected_revision, actor=actor)
|
||||
|
||||
def _save(self, value, *, expected_revision, actor):
|
||||
"""Save while the caller holds operation(), including workflow receipt recovery."""
|
||||
self._finish_normalization(self._state())
|
||||
units = self._units(self._files(self.evidence))
|
||||
existing = units.get(value.id)
|
||||
if (existing is None) != (expected_revision is None):
|
||||
raise ArchiveConflict("Evidence was created or removed since review")
|
||||
if existing and _content(existing[1]) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
relative = (
|
||||
existing[0]
|
||||
if existing
|
||||
else f"curated/{value.kind}/{value.id.removeprefix('evidence:')}.md"
|
||||
)
|
||||
if existing and existing[1].kind != value.kind:
|
||||
raise ValueError("An update cannot change the Evidence kind")
|
||||
if existing:
|
||||
value = value.model_copy(update={"provenance": existing[1].provenance})
|
||||
_atomic(self.evidence / relative, dump_curated_markdown(value))
|
||||
return self._consolidate(actor, None)
|
||||
|
||||
def get(self, identity: str):
|
||||
with self.operation():
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
return {"unit": unit, "revision": _content(unit), "path": str(self.evidence / relative)}
|
||||
|
||||
def remove(self, identity: str, *, expected_revision: str, actor: str):
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
self._finish_normalization(self._state())
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
if _content(unit) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
(self.evidence / relative).unlink()
|
||||
return self._consolidate(actor, None)
|
||||
@@ -12,10 +12,18 @@ if TYPE_CHECKING:
|
||||
from tht.config import EvidenceSourcesConfig
|
||||
|
||||
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None) -> list[EvidenceSource]:
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None, *, acquisition: bool = False) -> list[EvidenceSource]:
|
||||
"""Build configured Evidence adapters in the existing deterministic order."""
|
||||
if evidence is None:
|
||||
return []
|
||||
if evidence.local_archive_root is not None and not acquisition:
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive
|
||||
archive = LocalEvidenceArchive(evidence.local_archive_root)
|
||||
if (archive.metadata / "state.yaml").exists():
|
||||
snapshot = archive.active_snapshot()
|
||||
if snapshot is None:
|
||||
raise ValueError("Local Evidence has not been activated; run Evidence consolidation")
|
||||
return [FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))]
|
||||
sources: list[EvidenceSource] = []
|
||||
if evidence.source_root is not None:
|
||||
legacy_root = evidence.source_root / evidence.evidence_dir
|
||||
|
||||
Reference in New Issue
Block a user