feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s

Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation.

Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
Codex
2026-09-10 10:31:34 +02:00
parent 8fe526dd6e
commit 82e2c91f42
168 changed files with 11914 additions and 1772 deletions
+5
View File
@@ -22,6 +22,7 @@ from tht.evidence.authoring import (
)
from tht.evidence.canonical import (
CuratedEvidence,
ManualEvidenceProvenance,
dump_curated_markdown,
load_curated_tree,
parse_curated_markdown,
@@ -37,6 +38,7 @@ from tht.evidence.contracts import (
validate_namespaced_value,
validate_safe_metadata,
)
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive
from tht.evidence.preprocessing import EvidenceEmbedder, build_preprocessing_pipeline
from tht.evidence.search import (
ActiveEvidenceSearcher,
@@ -58,6 +60,7 @@ from tht.evidence.sources import build_sources
__all__ = [
"AcquiredDocument",
"ActiveEvidenceSearcher",
"ArchiveConflict",
"CorpusWorkspaceMismatchError",
"CuratedEvidence",
"EvidenceEmbedder",
@@ -75,6 +78,8 @@ __all__ = [
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"LocalEvidenceArchive",
"ManualEvidenceProvenance",
"PiEvidenceRestructurer",
"RestructureCandidate",
"RestructureRequest",
+152
View File
@@ -0,0 +1,152 @@
"""Local Evidence browsing and explicit consolidation, independent of DWH access."""
import os
from pathlib import Path
from .canonical import parse_curated_markdown
from .local_archive import LocalEvidenceArchive, _content
class ConsolidationError(RuntimeError):
def __init__(self, message, *, saved=False):
super().__init__(message)
self.saved = saved
def consolidate_from_config(config: Path):
from tht.cli.preprocess_cmd import run_from_config
from tht.config import load_config
from .authoring import EvidencePreparationError, migrate_workspace_evidence
cfg = load_config(config)
if not cfg.evidence or not cfg.evidence.local_archive_root:
raise ConsolidationError("Local Evidence is not configured for this workspace")
root = cfg.evidence.local_archive_root
archive = LocalEvidenceArchive(root)
if not (archive.metadata / "state.yaml").exists():
try:
# Explicit first consolidation performs the one-time legacy conversion.
if (archive.evidence / "manifest.yaml").is_file():
migrate_workspace_evidence(root)
else:
archive.initialize()
except (ValueError, OSError, EvidencePreparationError) as error:
raise ConsolidationError(str(error)) from error
result = None
def activate(snapshot):
nonlocal result
try:
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise ConsolidationError("Evidence indexing is blocked; check unit size and review items", saved=True)
except ConsolidationError:
raise
except Exception as error:
raise ConsolidationError("Evidence files were saved, but indexing failed. Retry consolidation.", saved=True) from error
try:
archive.consolidate(actor=os.environ.get("THT_PRINCIPAL_SUBJECT") or "installation operator",
activate=activate)
except (ValueError, OSError) as error:
raise ConsolidationError(str(error)) from error
return result
def browse(root: Path, query: dict):
"""Read complete working units and their active status without opening an index."""
archive = LocalEvidenceArchive(root)
with archive.operation():
state = archive._state()
active = archive._snapshot(state["active"]) if state["active"] else None
active_units = archive._units(archive._files(active), allow_review=True) if active else {}
files = archive._files(archive.evidence)
items, errors = [], []
seen = set()
for relative, data in files.items():
try:
unit = parse_curated_markdown(data.decode(), path=Path(relative))
if unit.id in seen:
raise ValueError("Duplicate Evidence identifier")
seen.add(unit.id)
old = active_units.get(unit.id)
status = "review_required" if unit.review_items else "legacy" if unit.schema_version != 4 \
else "active" if old and _content(old[1]) == _content(unit) and old[1].provenance == unit.provenance else "modified" if old else "new"
items.append({**unit.model_dump(mode="json"), "file": relative, "status": status,
"revision": _content(unit)})
except (ValueError, UnicodeError) as error:
errors.append({"file": relative, "message": str(error)[:1500]})
for identity, (relative, unit) in active_units.items():
if identity not in seen:
items.append({**unit.model_dump(mode="json"), "file": relative,
"status": "invalid" if relative in files else "removed", "revision": _content(unit)})
def matches(item):
for field in ("kind", "status", "language"):
if query.get(field) and item[field] != query[field]:
return False
if query.get("purpose") and query["purpose"] not in item["purposes"]:
return False
for key, field in (("concept", "concepts"), ("table", "tables"), ("column", "columns")):
if query.get(key) and not any(query[key].casefold() in v.casefold() for v in item["applies_to"][field]):
return False
provenance = item["provenance"]
if query.get("source") and query["source"].casefold() not in str(provenance).casefold():
return False
return not query.get("q") or query["q"].casefold() in str(item).casefold()
selected = [item for item in items if matches(item)]
selected.sort(key=lambda item: (str(item.get(query.get("sort", "title"), "")).casefold(), item["id"]),
reverse=query.get("direction") == "desc")
page, size = int(query.get("page", 1)), int(query.get("page_size", 25))
if page < 1 or not 1 <= size <= 100:
raise ValueError("Invalid Evidence page")
result = {"items": selected[(page-1)*size:page*size], "total": len(selected), "page": page,
"page_size": size, "errors": errors, "active_revision": state["active"],
"pending_revision": state["pending"], "initialized": bool(state.get("baseline") or state["pending"] or state["active"])}
if query.get("id"):
result["item"] = next((item for item in items if item["id"] == query["id"]), None)
from .imports import reviews
result["source_reviews"] = reviews(archive)
return result
def source_action(config, *, action, source_id=None, revision=None, decision=None, actor="installation operator"):
from tht.config import load_config
from .authoring import PiEvidenceRestructurer, authoring_skill_path, migrate_workspace_evidence
from .imports import acquisition_sources, decide, refresh
cfg = load_config(config)
if not cfg.evidence or not cfg.evidence.local_archive_root:
raise ValueError("Local Evidence is not configured")
archive = LocalEvidenceArchive(cfg.evidence.local_archive_root)
if not (archive.metadata / "state.yaml").exists():
if (archive.evidence / "manifest.yaml").is_file():
migrate_workspace_evidence(archive.root)
else:
(archive.evidence / "curated").mkdir(parents=True, exist_ok=True)
archive.initialize()
if action == "refresh":
skill = authoring_skill_path()
return refresh(archive, acquisition_sources(cfg), PiEvidenceRestructurer(
os.environ.get("THT_PI_EXECUTABLE", "pi"), skill))
if action != "decide":
raise ValueError("Unknown source action")
def activate(snapshot):
from tht.cli.preprocess_cmd import run_from_config
try:
result = run_from_config(config, local_snapshot=snapshot)
if result.status != "succeeded":
raise RuntimeError("Indexing did not succeed")
except Exception as error:
raise ConsolidationError("Source decision saved, but indexing failed. Retry the same decision.", saved=True) from error
try:
return decide(archive, source_id=source_id, revision=revision, decision=decision,
actor=actor, activate=activate)
except (ValueError, OSError) as error:
from .imports import reviews
if any(r["id"] == source_id and r["status"] == "applying" for r in reviews(archive)):
raise ConsolidationError(f"Source decision saved. {str(error)[:1200]}. Retry the same decision.", saved=True) from error
raise
+30 -5
View File
@@ -25,6 +25,7 @@ from tht.evidence.canonical import (
EvidenceKind,
EvidencePurpose,
EvidenceScope,
ManualEvidenceProvenance,
ReviewItem,
StrictModel,
dump_curated_markdown,
@@ -189,6 +190,12 @@ def _restore_exact_source_excerpts(
})
def authoring_skill_path() -> Path:
"""The installed wheel and the deployment's Pi resources live in different roots."""
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
@@ -389,6 +396,14 @@ def dump_manifest(manifest: EvidenceManifest) -> str:
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
"""Validate the curated corpus without writing the workspace."""
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").is_file():
from .local_archive import LocalEvidenceArchive
try:
LocalEvidenceArchive(workspace_root).validate()
return ValidationReport(())
except (OSError, ValueError) as error:
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
"evidence/curated", str(error)),))
findings: list[ValidationFinding] = []
manifest_path = evidence_root / "manifest.yaml"
if not manifest_path.is_file():
@@ -485,6 +500,9 @@ def _validate_manifest_source(
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
if isinstance(evidence.provenance, ManualEvidenceProvenance):
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
for item in evidence.review_items]
if evidence.id in manifest.orphans:
return []
path = evidence.provenance.source_file
@@ -560,6 +578,8 @@ def prepare_workspace_evidence(
raise EvidencePreparationError("authoring_workers_invalid")
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local" / "state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
manifest_path = evidence_root / "manifest.yaml"
try:
@@ -695,10 +715,9 @@ def migrate_workspace_evidence(
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
@@ -709,7 +728,7 @@ def migrate_workspace_evidence(
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 3})
evidence_id: document.model_copy(update={"schema_version": 4})
for evidence_id, document in documents_by_id.items()
}
migrated_ids: list[str] = []
@@ -728,12 +747,16 @@ def migrate_workspace_evidence(
migrated = tuple(sorted(migrated_ids))
unchanged = tuple(sorted(unchanged_ids))
if not migrated:
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
from .local_archive import LocalEvidenceArchive
LocalEvidenceArchive(workspace_root).initialize()
return EvidenceMigrationReport(
migrated=migrated,
unchanged=unchanged,
@@ -762,6 +785,8 @@ def resolve_workspace_evidence(
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
if (evidence_root / ".local/state.yaml").exists():
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
_reject_dirty_worktree(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
@@ -1017,7 +1042,7 @@ def _candidate_to_evidence(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 3
data["schema_version"] = 4
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
@@ -1038,7 +1063,7 @@ def _unsupported_unit(
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 3,
"schema_version": 4,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
+29 -3
View File
@@ -98,6 +98,19 @@ class ReviewItem(StrictModel):
field: str | None = None
class ManualEvidenceProvenance(StrictModel):
"""The curator supports the current content; an earlier document is only its origin."""
model_config = ConfigDict(extra="forbid", frozen=True)
kind: Literal["manual"] = "manual"
declared_by: str = Field(min_length=1)
original: EvidenceProvenance | None = None
@property
def source_file(self) -> str:
return f"Manual declaration: {self.declared_by}"
class FormulaPayload(StrictModel):
concept: str
columns: tuple[str, ...]
@@ -229,19 +242,23 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1, 2, 3]
schema_version: Literal[1, 2, 3, 4]
id: str
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
provenance: EvidenceProvenance
provenance: EvidenceProvenance | ManualEvidenceProvenance
review_items: tuple[ReviewItem, ...] = ()
payload: EvidencePayload
@model_validator(mode="after")
def _validate_kind_payload(self) -> CuratedEvidence:
if isinstance(self.provenance, ManualEvidenceProvenance) and self.schema_version != 4:
raise ValueError("manual declarations require Curated unit schema v4")
if self.schema_version == 4 and (not self.purposes or not self.language.strip()):
raise ValueError("editable Evidence requires a language and at least one purpose")
if not is_evidence_id(self.id):
raise ValueError("id must use the evidence:<slug> form")
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
@@ -1056,13 +1073,19 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
try:
raw = yaml.safe_load(frontmatter)
except yaml.YAMLError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
try:
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
elif data.get("schema_version") == 4:
from tht.evidence.editable import parse_document, parse_metadata
data = parse_document(parse_metadata(frontmatter), body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
@@ -1081,6 +1104,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render one canonical Curated Evidence Markdown document."""
if value.schema_version == 4:
from tht.evidence.editable import render_document
return render_document(value)
if value.schema_version == 3:
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
if value.schema_version == 2:
+10 -5
View File
@@ -8,7 +8,12 @@ from typing import Self
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
from tht.evidence.canonical import EVIDENCE_KINDS, EVIDENCE_PURPOSES
from tht.evidence.canonical import (
EVIDENCE_KINDS,
EVIDENCE_PURPOSES,
EvidenceProvenance,
ManualEvidenceProvenance,
)
from tht.evidence.contracts import (
canonical_provenance_uri,
normalize_aware_datetime,
@@ -76,10 +81,10 @@ def _validate_evidence_metadata(metadata: Mapping[str, JsonValue]) -> None:
if not isinstance(metadata["language"], str) or not metadata["language"]:
raise ValueError("typed Evidence metadata must contain language")
provenance = metadata["provenance"]
if not isinstance(provenance, dict) or set(provenance) != {
"source_file", "source_sha256", "supporting_excerpts",
}:
raise ValueError("typed Evidence metadata must contain canonical provenance")
if not isinstance(provenance, dict):
raise ValueError("typed Evidence metadata must contain canonical provenance") # noqa: TRY004
model = ManualEvidenceProvenance if provenance.get("kind") == "manual" else EvidenceProvenance
model.model_validate(provenance)
class _CanonicalValue(BaseModel):
+177
View File
@@ -0,0 +1,177 @@
"""Curated unit v4: visible Markdown fields are the sole human-content authority."""
from __future__ import annotations
import json
import re
import yaml
from .canonical import (
_PAYLOAD_TYPE_BY_KIND,
CuratedEvidence,
ManualEvidenceProvenance,
_v2_labels,
)
_LIST_FIELDS = {"synonyms", "variants", "columns", "tables"}
def parse_metadata(text: str) -> dict:
class UniqueKeysLoader(yaml.SafeLoader):
pass
def mapping(loader, node):
pairs = loader.construct_pairs(node, deep=True)
result = {}
for key, value in pairs:
if key in result:
raise ValueError(f"Duplicate metadata key: {key}")
result[key] = value
return result
UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping)
try:
return yaml.load(text, Loader=UniqueKeysLoader)
except (yaml.YAMLError, TypeError) as error:
raise ValueError("Evidence metadata is malformed") from error
def _sections(body: str, headings: dict[str, str]) -> dict[str, str]:
"""Recognize structural H2s outside code fences; all other Markdown is content."""
sections: dict[str, list[str]] = {}
field = None
fence = None
for line in body.splitlines():
match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line)
if match:
marker = match[1]
if fence is None:
fence = marker
elif marker[0] == fence[0] and len(marker) >= len(fence):
fence = None
heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None
if heading:
if heading in sections:
raise ValueError(f"Duplicate section: {line[3:]}")
field = heading
sections[field] = []
elif field is not None:
sections[field].append(line)
elif line.strip():
raise ValueError("Content must follow a documented section heading")
if fence:
raise ValueError("Unclosed Markdown code fence")
return {key: "\n".join(lines).strip() for key, lines in sections.items()}
def _list(text: str) -> list[str]:
if not text:
return []
values = []
for line in text.splitlines():
if not line.startswith("- ") or not line[2:].strip():
raise ValueError("List entries must use '- value', one per line")
value = line[2:]
# Quoted strings preserve multiline and unusual values during migration.
values.append(json.loads(value) if value.startswith('"') else value)
return values
def _values(text: str) -> dict[str, str]:
values = {}
key = None
lines = []
for line in text.splitlines():
if line.startswith("### "):
if key is not None:
values[key] = "\n".join(lines).strip()
label = line[4:]
key = json.loads(label) if label.startswith('"') else label
if key in values:
raise ValueError("Duplicate enum value")
lines = []
elif key is None:
if line.strip():
raise ValueError("Enum values require '### value' headings")
else:
lines.append(line)
if key is not None:
values[key] = "\n".join(lines).strip()
return values
def parse_document(metadata: dict, body: str) -> dict:
data = dict(metadata)
if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data):
raise ValueError("Title and payload must be edited only in the Markdown body")
lines = body.strip().splitlines()
if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip():
raise ValueError("A title starting with '# ' is required")
data["title"] = lines[0][2:].strip()
kind = data.get("kind")
if kind not in _PAYLOAD_TYPE_BY_KIND:
raise ValueError("Unknown Evidence kind")
labels = _v2_labels(str(data.get("language", "")))
fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields
sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields})
required = {name for name, field in fields.items() if field.is_required()}
if not required <= sections.keys():
raise ValueError(
"Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys()))
)
payload = {}
for key, content in sections.items():
if key in _LIST_FIELDS:
payload[key] = _list(content)
elif key == "values":
payload[key] = _values(content)
elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"):
payload[key] = content[7:-4]
else:
payload[key] = content
if fields[key].is_required() and not payload[key] and key != "values":
raise ValueError(f"Section {labels[key]} must not be empty")
data["payload"] = payload
data.setdefault(
"provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump()
)
return data
def render_document(value: CuratedEvidence) -> str:
metadata = value.model_dump(mode="json", exclude={"title", "payload"})
labels = _v2_labels(value.language)
parts = [
f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}"
]
for key, content in value.payload.model_dump(mode="json").items():
if key in _LIST_FIELDS:
rendered = "\n".join(
"- "
+ (
json.dumps(v, ensure_ascii=False)
if "\n" in v or v.startswith('"') or v != v.strip()
else v
)
for v in content
)
elif key == "values":
rendered = "\n\n".join(
f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items()
)
elif key == "sql":
rendered = f"```sql\n{content}\n```"
else:
rendered = content
parts.append(f"## {labels[key]}\n\n{rendered}")
rendered = "\n\n".join(parts) + "\n"
# Migration must fail explicitly rather than silently changing unrepresentable content.
restored = CuratedEvidence.model_validate(
parse_document(metadata, rendered.split("---\n", 2)[2])
)
if restored != value:
raise ValueError(
f"{value.id}: content cannot be represented losslessly in editable Markdown"
)
return rendered
+235
View File
@@ -0,0 +1,235 @@
"""Explicit acquisition and durable source comparisons; never implicit runtime refresh."""
import base64
import json
from .authoring import (
RestructureRequest,
_allocate_evidence_id,
_candidate_to_evidence,
normalize_source_text,
)
from .canonical import (
CuratedEvidence,
EvidenceProvenance,
ManualEvidenceProvenance,
dump_curated_markdown,
)
from .corpus.normalize import _decode
from .local_archive import ArchiveConflict, LocalEvidenceArchive, _atomic, _digest
MAX_DOCUMENTS = 200
MAX_TOTAL_BYTES = 100 * 1024 * 1024
def _origin(unit):
return unit.provenance.original if isinstance(unit.provenance, ManualEvidenceProvenance) else unit.provenance
def _records(archive):
path = archive.metadata / "sources.json"
if path.is_symlink():
raise ValueError("Source metadata must not use symlinks")
return json.loads(path.read_text()) if path.exists() else {}
def _write_records(archive, records):
_atomic(archive.metadata / "sources.json", json.dumps(records, ensure_ascii=False, sort_keys=True))
def reviews(archive):
return [{k: v for k, v in row.items() if k not in {"expected", "text", "writes", "approved"}}
for row in _records(archive).values()]
def acquisition_sources(cfg):
"""Local drafts and original files, plus configured read-only remote connectors."""
from .adapters import FilesystemEvidenceSource
from .sources import build_sources
root = cfg.evidence.local_archive_root / "evidence"
# Canonical acquired versions are immutable lineage, never new input documents.
patterns = [str(p.relative_to(root)) for folder in ("incoming", "source")
for p in sorted((root / folder).rglob("*.md"))
if not p.is_relative_to(root / "source/acquired")]
result = [FilesystemEvidenceSource(root, patterns=patterns)] if patterns else []
# Filesystem descriptors select the installation's local authoring tree after E2.
remote = cfg.evidence.model_copy(update={"source_root": None,
"sources": [s for s in cfg.evidence.sources if s.type != "filesystem"]})
result.extend(build_sources(remote, acquisition=True))
return result
def refresh(archive: LocalEvidenceArchive, sources, restructurer):
"""Acquire everything successfully before recording proposals. Missing is never deletion."""
with archive.operation():
state = archive._state()
if state.get("pending") or state.get("import_writes"):
raise ArchiveConflict("Complete the pending consolidation before refreshing sources")
records = _records(archive)
if any(r["status"] == "applying" for r in records.values()):
raise ArchiveConflict("Retry the pending source decision before refreshing")
files = archive._files(archive.evidence)
units = archive._units(files, allow_review=True)
documents, total = {}, 0
for adapter in sources:
for item in adapter.discover():
document = adapter.acquire(item)
total += len(document.content)
if len(documents) >= MAX_DOCUMENTS or total > MAX_TOTAL_BYTES:
raise ValueError("Source refresh exceeds the local acquisition limit")
relative = item.metadata.get("relative_path") if item.uri.startswith("file:") else None
identity = f"local:{relative}" if relative else item.uri
key = _digest(identity.encode())
if key in documents:
raise ValueError("Duplicate acquisition identity")
documents[key] = (document, relative)
reserved = set(units) | set(state["deleted_ids"])
for record in records.values():
reserved.update(u["id"] for u in record.get("proposed", []))
changed, unchanged = 0, 0
acquired = {}
for key, (document, relative) in documents.items():
text = normalize_source_text(_decode(document))
sha = "sha256:" + _digest(text.encode())
old = records.get(key)
stale = old and old["status"] == "review" and any(
p not in files or _digest(files[p]) != h for p, h in old["expected"].items())
if old and old["sha256"] == sha and not stale:
old["availability"] = "available"
unchanged += 1
continue
current = {i: pair for i, pair in units.items()
if (old and i in old["unit_ids"]) or
(_origin(pair[1]) and _origin(pair[1]).source_file == relative)}
# Seed imported E2 document identity without asking the model to recurate unchanged text.
if old is None and current and all(_origin(u).source_sha256 == sha for _, u in current.values()):
records[key] = {"id": key, "uri": document.source.uri, "legacy_file": relative,
"sha256": sha, "unit_ids": sorted(current), "status": "accepted",
"availability": "available", "revision": sha[7:], "proposed": []}
unchanged += 1
continue
source_file = f"source/acquired/{key}/{sha[7:]}.md"
request = RestructureRequest(source_file=source_file, source_sha256=sha,
normalized_text=text, previous_units=tuple(u for _, u in current.values()))
proposed = []
seen = set()
suppressed = relative in state["suppressed_sources"] or any(
p.startswith(f"source/acquired/{key}/") for p in state["suppressed_sources"]) or (old and (
old.get("suppressed", False) or any(i in state["deleted_ids"] for i in old["unit_ids"])))
for candidate in restructurer.restructure(request):
identity = candidate.existing_id
if identity in state["deleted_ids"]:
continue
if identity is not None and identity not in current:
raise ValueError("Source proposal refers to an unrelated Evidence identity")
if identity is None:
if suppressed:
continue # A model-created identifier cannot bypass a curated deletion.
identity = _allocate_evidence_id(candidate.title, reserved)
reserved.add(identity)
if identity in seen:
raise ValueError("Source proposal repeats an Evidence identity")
seen.add(identity)
unit = _candidate_to_evidence(candidate, identity, source_file, sha)
dump_curated_markdown(unit) # Refuse an uneditable proposal before saving any review.
if any(excerpt not in text for excerpt in unit.provenance.supporting_excerpts):
raise ValueError("Source proposal contains an excerpt absent from the acquired document")
proposed.append(unit.model_dump(mode="json"))
expected = {p: _digest(files[p]) for p, _ in current.values()}
row = {"id": key, "uri": document.source.uri, "legacy_file": relative,
"sha256": sha, "source_file": source_file, "text": text,
"status": "review", "availability": "available", "suppressed": bool(suppressed),
"unit_ids": sorted(current), "expected": expected,
"current": [u.model_dump(mode="json") for _, u in current.values()], "proposed": proposed,
"removed_ids": sorted(set(current) - seen)}
row["revision"] = _digest(json.dumps(row, sort_keys=True).encode())
records[key] = row
acquired[key] = {"source": document.source.model_dump(mode="json"),
"media_type": document.media_type, "raw_base64": base64.b64encode(document.content).decode()}
changed += 1
for key, row in records.items():
if key not in documents:
row["availability"] = "missing"
# No writes above: an access/model failure preserves every previous review and active unit.
for key, value in acquired.items():
directory = archive.metadata / "acquisitions" / key
if any(p.is_symlink() for p in [directory, directory.parent]):
raise ValueError("Acquisition metadata must not use symlinks")
_atomic(directory / f"{records[key]['sha256'][7:]}.json", json.dumps(value, ensure_ascii=False))
_write_records(archive, records)
return {"status": "succeeded", "counts": {"changed": changed, "unchanged": unchanged,
"review": sum(r["status"] == "review" for r in records.values())}}
def decide(archive, *, source_id, revision, decision, actor, activate):
if decision not in {"keep", "replace"} or not actor.strip():
raise ValueError("Choose keep or replace and supply a curator")
with archive.operation():
records = _records(archive)
row = records.get(source_id)
if row is None or row["revision"] != revision:
raise ArchiveConflict("Source comparison changed; refresh the page")
if row["status"] == "applying":
if row["decision"] != decision:
raise ArchiveConflict("Retry the saved source decision before changing it")
state = archive._state()
state.update(import_writes=row["writes"], approved_imports=row["approved"])
archive._write_state(state)
result = archive._consolidate(actor, activate)
else:
if row["status"] != "review":
raise ArchiveConflict("This source comparison was already decided")
state = archive._state()
if state.get("pending") or state.get("import_writes"):
raise ArchiveConflict("Complete the pending consolidation first")
files = archive._files(archive.evidence)
units = archive._units(files, allow_review=True)
if any(_digest(files[p]) != h if p in files else True for p, h in row["expected"].items()):
raise ArchiveConflict("Curated files changed since source review; refresh the source comparison")
selected = [CuratedEvidence.model_validate(v) for v in row["proposed"]] if decision == "replace" else [units[i][1] for i in row["unit_ids"]]
writes, approved = {}, {}
def write(path, content):
current = archive.evidence / path
writes[path] = {"before": _digest(current.read_bytes()) if current.exists() else None, "after": content}
if decision == "replace":
write(row["source_file"], row["text"])
for identity in row["unit_ids"]:
write(units[identity][0], None)
for value in selected:
if value.review_items:
raise ValueError("The proposal needs review; correct the input draft and refresh, or keep local content")
if decision == "keep":
origin = _origin(value)
if origin:
old = archive._snapshot(state["active"] or state["baseline"])
candidate = old / origin.source_file
if not candidate.is_file():
candidate = archive.evidence / origin.source_file
text = normalize_source_text(candidate.read_text())
if "sha256:" + _digest(text.encode()) != origin.source_sha256:
raise ValueError("The original source version is unavailable")
path = f"source/acquired/{source_id}/{origin.source_sha256[7:]}.md"
write(path, text)
origin = EvidenceProvenance(source_file=path, source_sha256=origin.source_sha256,
supporting_excerpts=origin.supporting_excerpts)
value = value.model_copy(update={"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)})
if value.id in units and value.id not in row["unit_ids"]:
raise ArchiveConflict("A proposed Evidence identity was created elsewhere")
path = units[value.id][0] if value.id in units and units[value.id][1].kind == value.kind else f"curated/{value.kind}/{value.id[9:]}.md"
if path in files and value.id not in units:
raise ArchiveConflict("The proposed file path is occupied")
write(path, dump_curated_markdown(value))
approved[value.id] = _digest(dump_curated_markdown(value).encode())
# Journal before applying files; retry never silently clobbers an external edit.
row.update(status="applying", decision=decision, decided_by=actor,
next_unit_ids=[u.id for u in selected], writes=writes, approved=approved)
_write_records(archive, records)
state.update(import_writes=writes, approved_imports=approved)
archive._write_state(state)
result = archive._consolidate(actor, activate)
if result["status"] != "active":
raise ValueError("Source decision was saved but not activated; retry")
row.update(status="accepted" if decision == "replace" else "kept", unit_ids=row["next_unit_ids"])
_write_records(archive, records)
return {"status": "succeeded", "counts": {"units": result["units"]}}
+472
View File
@@ -0,0 +1,472 @@
"""Persistent curated files, immutable consolidation candidates and explicit activation.
The working tree is primary data. Core consumers use only active_snapshot(); an
editor save or failed activation never switches that pointer. No Git/network/DWH I/O.
"""
from __future__ import annotations
import fcntl
import hashlib
import json
import os
import tempfile
from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
import yaml
from .canonical import (
MAX_CURATED_FILE_BYTES,
CuratedEvidence,
EvidenceProvenance,
ManualEvidenceProvenance,
dump_curated_markdown,
parse_curated_markdown,
)
class ArchiveConflict(ValueError):
"""The curator must reconcile a concurrent change before replacing it."""
def _digest(value: bytes) -> str:
return hashlib.sha256(value).hexdigest()
def _content(unit: CuratedEvidence) -> str:
return _digest(
json.dumps(
unit.model_dump(mode="json", exclude={"schema_version", "provenance"}),
sort_keys=True,
ensure_ascii=False,
).encode()
)
def _atomic(path: Path, data: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent)
try:
with os.fdopen(fd, "w") as handle:
handle.write(data)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary, path)
finally:
if os.path.exists(temporary):
os.unlink(temporary)
class LocalEvidenceArchive:
def __init__(self, workspace_root: Path):
self.root = workspace_root.resolve()
self.evidence = self.root / "evidence"
self.metadata = self.evidence / ".local"
if self.evidence.is_symlink() or self.metadata.is_symlink():
raise ValueError("The local Evidence archive must use persistent regular directories")
@contextmanager
def operation(self):
if any(
path.is_symlink()
for path in (
self.evidence,
self.metadata,
self.metadata / "snapshots",
self.metadata / "state.yaml",
)
):
raise ValueError("Evidence archive metadata must not use symlinks")
self.root.mkdir(parents=True, exist_ok=True)
lock = self.root / ".evidence-archive.lock"
if lock.is_symlink():
raise ValueError("Evidence lock must not be a symlink")
with lock.open("a") as handle:
fcntl.flock(handle, fcntl.LOCK_EX)
try:
yield
finally:
fcntl.flock(handle, fcntl.LOCK_UN)
def _state(self):
path = self.metadata / "state.yaml"
if not path.exists():
return {
"schema_version": 1,
"active": None,
"pending": None,
"baseline": None,
"deleted_ids": [],
"suppressed_sources": [],
}
value = yaml.safe_load(path.read_text())
if not isinstance(value, dict) or value.get("schema_version") != 1:
raise ValueError("Unsupported local Evidence archive state")
return value
def _write_state(self, state):
_atomic(self.metadata / "state.yaml", yaml.safe_dump(state, sort_keys=True))
def _snapshot(self, revision: str) -> Path:
if (
not isinstance(revision, str)
or len(revision) != 64
or any(c not in "0123456789abcdef" for c in revision)
):
raise ValueError("Invalid Evidence snapshot identity")
path = self.metadata / "snapshots" / revision
if not path.is_dir() or path.is_symlink():
raise ValueError("Consolidated Evidence snapshot is missing")
return path
def active_snapshot(self) -> Path | None:
"""Only this immutable source is eligible for core consumption."""
with self.operation():
revision = self._state()["active"]
return self._snapshot(revision) if revision else None
def validate(self):
"""Check the working files without modifying them or changing active content."""
with self.operation():
units = self._units(self._files(self.evidence))
state = self._state()
revision = state["pending"] or state["active"] or state.get("baseline")
previous = (
self._units(self._files(self._snapshot(revision)), allow_review=True)
if revision
else {}
)
for identity, (_, unit) in units.items():
old = previous.get(identity)
if isinstance(unit.provenance, EvidenceProvenance) and (
old is None or _content(old[1]) == _content(unit)
):
self._validate_document_source(unit.provenance)
return len(units)
def _files(self, root: Path):
result = {}
curated = root / "curated"
if curated.is_symlink():
raise ValueError("Curated directory must not be a symlink")
if not curated.is_dir():
raise ValueError(f"{curated}: curated archive is unavailable; absence is not deletion")
for path in sorted(curated.rglob("*")):
if path.is_symlink():
raise ValueError(f"{path}: symlinks are not supported in the curated archive")
if path.suffix != ".md" or path.name.upper().startswith("README"):
continue
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{path}: Evidence file exceeds the size limit")
result[path.relative_to(root).as_posix()] = path.read_bytes()
return result
def _units(self, files, *, allow_review=False):
units = {}
for relative, data in files.items():
try:
unit = parse_curated_markdown(data.decode("utf-8"), path=Path(relative))
if unit.schema_version != 4:
raise ValueError("Run evidence migrate before consolidating legacy units")
if unit.id in units:
raise ValueError(f"Duplicate Evidence identity {unit.id}")
if unit.review_items and not allow_review:
raise ValueError("Resolve review items before consolidation")
units[unit.id] = (relative, unit)
except ValueError as error:
raise ValueError(f"{relative}: {error}") from error
return units
def initialize(self):
"""Capture migrated/refined content before edits, without activating review items."""
with self.operation():
state = self._state()
if state.get("baseline") or state["active"] or state["pending"]:
return
files = self._files(self.evidence)
self._units(files, allow_review=True)
source = self.evidence / "source"
if source.is_symlink():
raise ValueError("Source symlinks are not supported")
if source.exists():
for path in sorted(source.rglob("*")):
if path.is_symlink():
raise ValueError("Source symlinks are not supported")
if path.is_file():
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{path}: source exceeds the size limit")
files[path.relative_to(self.evidence).as_posix()] = path.read_bytes()
revision = _digest(b"".join(k.encode() + b"\0" + v for k, v in sorted(files.items())))
snapshots = self.metadata / "snapshots"
snapshots.mkdir(parents=True, exist_ok=True)
destination = snapshots / revision
if not destination.exists():
with tempfile.TemporaryDirectory(prefix=".baseline-", dir=snapshots) as tmp:
candidate = Path(tmp) / "snapshot"
(candidate / "curated").mkdir(parents=True)
for relative, data in files.items():
path = candidate / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
os.replace(candidate, destination)
state["baseline"] = revision
self._write_state(state)
def consolidate(self, *, actor: str, activate: Callable[[Path], None] | None = None):
"""Persist one valid candidate; optionally activate it through the existing index stage.
A missing callback deliberately reports pending_activation, never success.
A retry of unchanged files reuses the same candidate after an index failure.
"""
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
return self._consolidate(actor, activate)
def _consolidate(self, actor, activate):
state = self._state()
self._finish_import(state)
self._finish_normalization(state)
original_files = self._files(self.evidence)
units = self._units(original_files)
previous_revision = state["pending"] or state["active"] or state.get("baseline")
previous_root = self._snapshot(previous_revision) if previous_revision else None
previous = (
self._units(self._files(previous_root), allow_review=True) if previous_root else {}
)
deleted = set(state["deleted_ids"]) | (previous.keys() - units.keys())
suppressed = set(state["suppressed_sources"])
for identity in previous.keys() - units.keys():
provenance = previous[identity][1].provenance
source = (
provenance.original
if isinstance(provenance, ManualEvidenceProvenance)
else provenance
)
if source:
suppressed.add(source.source_file)
files = {}
for identity, (relative, unit) in units.items():
old = previous.get(identity)
changed = old is not None and _content(old[1]) != _content(unit)
approved = state.get("approved_imports", {}).get(identity) == _digest(
dump_curated_markdown(unit).encode()
)
if approved:
pass # An explicit source decision authorized this exact content and provenance.
elif changed or (
isinstance(unit.provenance, ManualEvidenceProvenance)
and (old is None or unit.provenance.declared_by == "local curator")
):
origin = old[1].provenance if old else unit.provenance
origin = origin.original if isinstance(origin, ManualEvidenceProvenance) else origin
unit = unit.model_copy(
update={
"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)
}
)
elif old and unit.provenance != old[1].provenance:
raise ArchiveConflict(
f"{relative}: provenance is managed; edit the content instead"
)
if isinstance(unit.provenance, EvidenceProvenance):
self._validate_document_source(unit.provenance)
files[relative] = dump_curated_markdown(unit).encode()
origin = (
unit.provenance.original
if isinstance(unit.provenance, ManualEvidenceProvenance)
else unit.provenance
)
if origin:
candidates = [previous_root / origin.source_file] if previous_root else []
candidates.append(self.evidence / origin.source_file)
from .authoring import normalize_source_text
for source in candidates:
if source.is_file() and not source.is_symlink():
if source.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{source}: source exceeds the size limit")
raw = source.read_bytes()
if (
"sha256:" + _digest(normalize_source_text(raw.decode()).encode())
== origin.source_sha256
):
if origin.source_file in files and files[origin.source_file] != raw:
raise ArchiveConflict(
"Different source revisions require explicit source resolution"
)
files[origin.source_file] = raw
break
else:
raise ValueError(
f"{origin.source_file}: the recorded original document is unavailable"
)
# Record current declarations and original document lineage distinctly.
manifest = {
"schema_version": 1,
"units": {
identity: {
"file": relative,
"content_hash": _content(parse_curated_markdown(files[relative].decode())),
"provenance": parse_curated_markdown(
files[relative].decode()
).provenance.model_dump(mode="json"),
}
for identity, (relative, _) in sorted(units.items())
},
"deleted_ids": sorted(deleted),
"suppressed_sources": sorted(suppressed),
}
files["local-manifest.yaml"] = yaml.safe_dump(
manifest, allow_unicode=True, sort_keys=True
).encode()
revision = _digest(
b"".join(path.encode() + b"\0" + data + b"\0" for path, data in sorted(files.items()))
)
snapshots = self.metadata / "snapshots"
snapshots.mkdir(parents=True, exist_ok=True)
destination = snapshots / revision
if not destination.exists():
with tempfile.TemporaryDirectory(prefix=".candidate-", dir=snapshots) as tmp:
candidate = Path(tmp) / "snapshot"
(candidate / "curated").mkdir(parents=True)
for relative, data in files.items():
path = candidate / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
os.replace(candidate, destination)
if self._files(self.evidence) != original_files:
raise ArchiveConflict(
"Evidence files changed during consolidation; retry with the current files"
)
# The candidate exists first. Pending state makes every subsequent interruption recoverable.
state.update(
pending=revision,
deleted_ids=sorted(deleted),
suppressed_sources=sorted(suppressed),
normalization={relative: _digest(data) for relative, data in original_files.items()},
)
state.pop("approved_imports", None)
self._write_state(state)
self._finish_normalization(state)
if activate is None:
return {
"status": "pending_activation",
"revision": revision,
"snapshot": str(destination),
}
activate(destination)
state.update(active=revision, pending=None)
self._write_state(state)
return {
"status": "active",
"revision": revision,
"snapshot": str(destination),
"units": len(units),
"deleted": len(previous.keys() - units.keys()),
}
def _finish_import(self, state):
"""Replay an explicit source decision, rejecting intervening external edits."""
writes = state.get("import_writes")
if writes is None:
return
for relative, change in writes.items():
path = self.evidence / relative
if not relative.startswith(("curated/", "source/acquired/")) or ".." in Path(relative).parts:
raise ValueError("Invalid import destination")
if any(p.is_symlink() for p in [path, *path.parents] if p != self.root.parent):
raise ValueError("Import destinations must not use symlinks")
current = _digest(path.read_bytes()) if path.exists() else None
after = _digest(change["after"].encode()) if change["after"] is not None else None
if current not in (change["before"], after):
raise ArchiveConflict("Evidence changed during a source decision; restore or review the file")
for relative, change in writes.items():
path = self.evidence / relative
if change["after"] is None:
path.unlink(missing_ok=True)
else:
_atomic(path, change["after"])
del state["import_writes"]
self._write_state(state)
def _finish_normalization(self, state):
"""Replay interrupted managed writes only where the user's bytes are unchanged."""
if "normalization" not in state:
return
snapshot = self._snapshot(state["pending"])
current = self._files(self.evidence)
for relative, original_hash in state["normalization"].items():
if relative in current and _digest(current[relative]) == original_hash:
_atomic(self.evidence / relative, (snapshot / relative).read_text())
_atomic(
self.evidence / "local-manifest.yaml", (snapshot / "local-manifest.yaml").read_text()
)
del state["normalization"]
self._write_state(state)
def _validate_document_source(self, provenance):
from .authoring import _normalize, normalize_source_text
path = self.evidence / provenance.source_file
if (
path.is_symlink()
or not path.is_file()
or not path.resolve().is_relative_to(self.evidence.resolve())
):
raise ValueError(f"{provenance.source_file}: source document is unavailable")
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError(f"{provenance.source_file}: source exceeds the size limit")
source = normalize_source_text(path.read_text())
if "sha256:" + _digest(source.encode()) != provenance.source_sha256:
raise ArchiveConflict(
f"{provenance.source_file}: source changed; explicit source refresh is required"
)
if any(_normalize(excerpt) not in source for excerpt in provenance.supporting_excerpts):
raise ValueError(f"{provenance.source_file}: supporting excerpt is missing")
def save(self, value: CuratedEvidence, *, expected_revision: str | None, actor: str):
"""Explicit workflow correction with optimistic concurrency; activation is a separate boundary."""
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
return self._save(value, expected_revision=expected_revision, actor=actor)
def _save(self, value, *, expected_revision, actor):
"""Save while the caller holds operation(), including workflow receipt recovery."""
self._finish_normalization(self._state())
units = self._units(self._files(self.evidence))
existing = units.get(value.id)
if (existing is None) != (expected_revision is None):
raise ArchiveConflict("Evidence was created or removed since review")
if existing and _content(existing[1]) != expected_revision:
raise ArchiveConflict("Evidence changed since review")
relative = (
existing[0]
if existing
else f"curated/{value.kind}/{value.id.removeprefix('evidence:')}.md"
)
if existing and existing[1].kind != value.kind:
raise ValueError("An update cannot change the Evidence kind")
if existing:
value = value.model_copy(update={"provenance": existing[1].provenance})
_atomic(self.evidence / relative, dump_curated_markdown(value))
return self._consolidate(actor, None)
def get(self, identity: str):
with self.operation():
relative, unit = self._units(self._files(self.evidence))[identity]
return {"unit": unit, "revision": _content(unit), "path": str(self.evidence / relative)}
def remove(self, identity: str, *, expected_revision: str, actor: str):
if not actor.strip():
raise ValueError("A curator identity is required")
with self.operation():
self._finish_normalization(self._state())
relative, unit = self._units(self._files(self.evidence))[identity]
if _content(unit) != expected_revision:
raise ArchiveConflict("Evidence changed since review")
(self.evidence / relative).unlink()
return self._consolidate(actor, None)
+9 -1
View File
@@ -12,10 +12,18 @@ if TYPE_CHECKING:
from tht.config import EvidenceSourcesConfig
def build_sources(evidence: EvidenceSourcesConfig | None) -> list[EvidenceSource]:
def build_sources(evidence: EvidenceSourcesConfig | None, *, acquisition: bool = False) -> list[EvidenceSource]:
"""Build configured Evidence adapters in the existing deterministic order."""
if evidence is None:
return []
if evidence.local_archive_root is not None and not acquisition:
from tht.evidence.local_archive import LocalEvidenceArchive
archive = LocalEvidenceArchive(evidence.local_archive_root)
if (archive.metadata / "state.yaml").exists():
snapshot = archive.active_snapshot()
if snapshot is None:
raise ValueError("Local Evidence has not been activated; run Evidence consolidation")
return [FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))]
sources: list[EvidenceSource] = []
if evidence.source_root is not None:
legacy_root = evidence.source_root / evidence.evidence_dir