fix(evidence): harden canonical corpus contracts

This commit is contained in:
2026-07-12 03:10:30 +02:00
parent d702aad93d
commit 4424fd3d90
5 changed files with 481 additions and 71 deletions
@@ -41,3 +41,33 @@ did not exist. After implementation, the focused suite passed.
treat these value objects as immutable and construct replacements rather than mutate collections.
- The adapter and normalization tasks should preserve the credential-free boundary by passing only
these records beyond acquisition.
## Review hardening follow-up
All six binding review areas were addressed in a separate TDD pass:
- JSON metadata is recursively converted to immutable `FrozenDict`/tuple values while retaining
stable object/array JSON serialization. Manifest document and chunk collections are tuples.
- Secret-key matching now normalizes camelCase and punctuation. It rejects credential-specific
names (passwords, API keys, access/refresh tokens, client/private keys, session cookies and
authorization) recursively, while deliberate benign labels such as generic `token` and `secret`
remain valid.
- Canonical URIs require a scheme and reject userinfo or credential-bearing query parameters.
- Namespaced IDs, SHA-256 content hashes, timezone-aware UTC timestamps, embedding/vector
compatibility, unique IDs, chunk referential/provenance integrity, contiguous per-document
ordinals and pipeline-version consistency are validated. Nested Pydantic instances are always
revalidated so `model_copy(update=...)` cannot bypass a manifest boundary.
- Acquired arbitrary bytes have explicit base64 JSON encoding and validation, covered by a JSON
round-trip test.
- `EvidenceSourceError` classifies transient/retryable versus permanent failures and exposes only
recursively immutable, credential-screened JSON details.
Follow-up verification:
- Focused contract suite: 39 passed.
- Focused Ruff: passed.
- Harness excluding Docker-backed L0 and the network-dependent wheel packaging test: 470 passed,
5 deselected.
- Fresh unrestricted harness attempt: 479 passed, 5 deselected; the same environmental boundary
remains (47 Docker socket setup errors, four Docker parity failures, one isolated `uv build`
network failure).
+79 -11
View File
@@ -1,4 +1,4 @@
from datetime import UTC, datetime
from datetime import UTC, datetime, timedelta, timezone
import pytest
from pydantic import ValidationError
@@ -12,11 +12,11 @@ def document(source_uri: str = "https://host/a.md") -> CanonicalDocument:
source_id="source:a",
source_uri=source_uri,
source_fingerprint="etag:abc",
content_hash="sha256:def",
content_hash=f"sha256:{'d' * 64}",
title="A",
content="# A",
media_type="text/markdown",
pipeline_version="normalize-v1",
pipeline_version="evidence-v1",
)
@@ -26,9 +26,9 @@ def chunk() -> CanonicalChunk:
document_id="doc:abc",
ordinal=0,
content="# A",
content_hash="sha256:def",
content_hash=f"sha256:{'e' * 64}",
source_uri="https://host/a.md",
pipeline_version="chunk-v1",
pipeline_version="evidence-v1",
)
@@ -65,17 +65,19 @@ def test_canonical_records_are_frozen(model):
model.pipeline_version = "changed" # type: ignore[misc]
def test_manifest_collections_have_independent_defaults():
def test_manifest_collections_are_immutable_tuples_with_json_arrays():
first = CorpusManifest(
manifest_id="one", created_at=datetime.now(UTC), pipeline_version="v1"
manifest_id="manifest:one", created_at=datetime.now(UTC), pipeline_version="v1"
)
second = CorpusManifest(
manifest_id="two", created_at=datetime.now(UTC), pipeline_version="v1"
manifest_id="manifest:two", created_at=datetime.now(UTC), pipeline_version="v1"
)
first.documents.append(document())
assert second.documents == []
with pytest.raises(AttributeError):
first.documents.append(document())
assert first.documents == ()
assert second.documents == ()
assert '"documents":[]' in first.model_dump_json()
def test_manifest_validates_embedding_compatibility_fields():
@@ -104,3 +106,69 @@ def test_canonical_metadata_rejects_secrets_and_non_json_values():
pipeline_version="v1",
metadata={"bad": object()},
)
def test_manifest_rejects_duplicate_ids_and_source_ids():
first = document()
duplicate_source = first.model_copy(
update={"document_id": "doc:other", "source_uri": "https://host/b.md"}
)
with pytest.raises(ValidationError, match="source_id"):
CorpusManifest(pipeline_version="evidence-v1", documents=[first, duplicate_source])
with pytest.raises(ValidationError, match="chunk_id"):
CorpusManifest(
pipeline_version="evidence-v1", documents=[first], chunks=[chunk(), chunk()]
)
def test_manifest_rejects_orphan_noncontiguous_and_inconsistent_chunks():
with pytest.raises(ValidationError, match="unknown document"):
CorpusManifest(pipeline_version="evidence-v1", chunks=[chunk()])
second = chunk().model_copy(update={"chunk_id": "chunk:abc:2", "ordinal": 2})
with pytest.raises(ValidationError, match="contiguous"):
CorpusManifest(
pipeline_version="evidence-v1", documents=[document()], chunks=[chunk(), second]
)
wrong_uri = chunk().model_copy(update={"source_uri": "https://host/wrong.md"})
with pytest.raises(ValidationError, match="source_uri"):
CorpusManifest(
pipeline_version="evidence-v1", documents=[document()], chunks=[wrong_uri]
)
def test_manifest_rejects_inconsistent_pipeline_versions():
wrong = document().model_copy(update={"pipeline_version": "other-v1"})
with pytest.raises(ValidationError, match="pipeline_version"):
CorpusManifest(pipeline_version="evidence-v1", documents=[wrong])
def test_vector_generation_requires_embedding_compatibility():
with pytest.raises(ValidationError, match="vector_generation"):
CorpusManifest(pipeline_version="evidence-v1", vector_generation="generation:one")
@pytest.mark.parametrize(
("field", "value"),
[
("document_id", "not-namespaced"),
("content_hash", "sha256:not-hex"),
("source_uri", "https://user:pass@host/a"),
("source_uri", "https://host/a?refresh_token=secret"),
],
)
def test_canonical_document_rejects_malformed_or_sensitive_provenance(field, value):
with pytest.raises(ValidationError):
CanonicalDocument.model_validate({**document().model_dump(), field: value})
def test_manifest_datetimes_are_aware_and_normalized_to_utc():
with pytest.raises(ValidationError, match="timezone-aware"):
CorpusManifest(created_at=datetime(2026, 7, 12), pipeline_version="evidence-v1")
plus_two = datetime(2026, 7, 12, 12, tzinfo=timezone(timedelta(hours=2)))
manifest = CorpusManifest(created_at=plus_two, pipeline_version="evidence-v1")
assert manifest.created_at.tzinfo is UTC
assert manifest.created_at.hour == 10
+146 -11
View File
@@ -1,9 +1,15 @@
from datetime import UTC, datetime
from datetime import UTC, datetime, timedelta, timezone
import pytest
from pydantic import ValidationError
from tht.ports.evidence import AcquiredDocument, EvidenceSource, SourceObject
from tht.ports.evidence import (
AcquiredDocument,
EvidenceSource,
EvidenceSourceError,
EvidenceSourceErrorCategory,
SourceObject,
)
class StubSource:
@@ -11,7 +17,7 @@ class StubSource:
return iter(
[
SourceObject(
source_id="handbook",
source_id="source:handbook",
uri="https://host/handbook.md",
fingerprint="sha256:abc",
)
@@ -35,35 +41,164 @@ def test_runtime_checkable_source_protocol():
def test_source_objects_are_frozen_and_metadata_defaults_are_independent():
first = SourceObject(source_id="a", uri="file:///a", fingerprint="sha256:a")
second = SourceObject(source_id="b", uri="file:///b", fingerprint="sha256:b")
first = SourceObject(source_id="source:a", uri="file:///a", fingerprint="sha256:a")
second = SourceObject(source_id="source:b", uri="file:///b", fingerprint="sha256:b")
with pytest.raises(ValidationError):
first.uri = "file:///changed" # type: ignore[misc]
first.metadata["owner"] = "team-a"
with pytest.raises(TypeError):
first.metadata["owner"] = "team-a"
assert second.metadata == {}
@pytest.mark.parametrize("metadata", [{"api_key": "secret"}, {"auth": {"token": "secret"}}])
def test_source_metadata_rejects_credentials(metadata):
@pytest.mark.parametrize(
"key",
[
"password",
"PassWd",
"api_key",
"x-api-key",
"accessToken",
"refresh.token",
"client secret",
"privateKey",
"session_cookie",
"Authorization",
],
)
def test_source_metadata_rejects_credential_specific_keys(key):
with pytest.raises(ValidationError, match="credential-like"):
SourceObject(
source_id="a", uri="https://host/a", fingerprint="etag:abc", metadata=metadata
source_id="source:a",
uri="https://host/a",
fingerprint="etag:abc",
metadata={"nested": [{key: "secret"}]},
)
def test_source_metadata_allows_benign_generic_token_and_secret_labels():
source = SourceObject(
source_id="source:a",
uri="https://host/a",
fingerprint="etag:abc",
metadata={"token": "word count token", "secret": False},
)
assert source.metadata["token"] == "word count token"
def test_nested_metadata_is_recursively_immutable_and_serializes_as_json():
source = SourceObject(
source_id="source:a",
uri="https://host/a",
fingerprint="etag:abc",
metadata={"nested": {"items": [1, {"ok": True}]}},
)
with pytest.raises(TypeError):
source.metadata["nested"]["items"][1]["ok"] = False
assert '"items":[1,{"ok":true}]' in source.model_dump_json()
@pytest.mark.parametrize(
"uri",
[
"https://user:pass@host/a",
"https://host/a?api_key=secret",
"https://host/a?accessToken=secret",
],
)
def test_source_uri_rejects_embedded_credentials(uri):
with pytest.raises(ValidationError, match="credentials"):
SourceObject(source_id="source:a", uri=uri, fingerprint="etag:abc")
def test_source_metadata_must_be_json_safe():
with pytest.raises(ValidationError):
SourceObject(
source_id="a",
source_id="source:a",
uri="file:///a",
fingerprint="sha256:a",
metadata={"path": object()},
)
def test_source_identity_and_fingerprint_must_be_namespaced():
with pytest.raises(ValidationError, match="namespaced"):
SourceObject(source_id="plain", uri="file:///a", fingerprint="sha256:a")
with pytest.raises(ValidationError, match="namespaced"):
SourceObject(source_id="source:a", uri="file:///a", fingerprint="plain")
def test_acquired_document_does_not_accept_credentials_as_extra_fields():
item = SourceObject(source_id="a", uri="https://host/a", fingerprint="etag:abc")
item = SourceObject(source_id="source:a", uri="https://host/a", fingerprint="etag:abc")
with pytest.raises(ValidationError):
AcquiredDocument(source=item, content=b"a", api_key="secret")
def test_acquired_binary_content_has_explicit_json_round_trip():
item = SourceObject(source_id="source:a", uri="https://host/a", fingerprint="etag:abc")
acquired = AcquiredDocument(source=item, content=b"\x00\xffbinary\x80")
payload = acquired.model_dump_json()
restored = AcquiredDocument.model_validate_json(payload)
assert restored.content == acquired.content
assert "binary" not in payload
def test_datetimes_must_be_aware_and_are_normalized_to_utc():
with pytest.raises(ValidationError, match="timezone-aware"):
SourceObject(
source_id="source:a",
uri="https://host/a",
fingerprint="etag:abc",
modified_at=datetime(2026, 7, 12),
)
source = SourceObject(
source_id="source:a",
uri="https://host/a",
fingerprint="etag:abc",
modified_at=datetime(2026, 7, 12, 4, tzinfo=timezone(timedelta(hours=2))),
)
assert source.modified_at.tzinfo is UTC
assert source.modified_at.hour == 2
acquired = AcquiredDocument(
source=source,
content=b"a",
acquired_at=datetime(2026, 7, 12, 2, tzinfo=UTC) + timedelta(hours=0),
)
assert acquired.acquired_at.utcoffset() == timedelta(0)
def test_source_errors_are_typed_retryable_and_safe():
transient = EvidenceSourceError(
"remote source unavailable",
category=EvidenceSourceErrorCategory.TRANSIENT,
details={"status": 503},
)
permanent = EvidenceSourceError(
"unsupported media type",
category=EvidenceSourceErrorCategory.PERMANENT,
)
assert transient.retryable is True
assert permanent.retryable is False
assert transient.details["status"] == 503
with pytest.raises(TypeError):
transient.details["status"] = 200
with pytest.raises(ValueError, match="credential-like"):
EvidenceSourceError(
"bad",
category=EvidenceSourceErrorCategory.PERMANENT,
details={"apiKey": "must-not-leak"},
)
with pytest.raises(ValidationError):
EvidenceSourceError(
"bad",
category=EvidenceSourceErrorCategory.PERMANENT,
details={"not_json": object()},
)
+92 -21
View File
@@ -1,64 +1,135 @@
"""Immutable records emitted by the Evidence preprocessing pipeline."""
import re
from datetime import UTC, datetime
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
from tht.ports.evidence import _reject_credentials
from tht.ports.evidence import (
normalize_aware_datetime,
validate_canonical_uri,
validate_namespaced_value,
validate_safe_metadata,
)
_NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$")
_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$")
def _validate_namespaced_id(value: str) -> str:
if not _NAMESPACED_ID.fullmatch(value):
raise ValueError("identifier must be namespaced as '<kind>:<stable-value>'")
return value
def _validate_hash(value: str) -> str:
if not _SHA256.fullmatch(value):
raise ValueError("content hash must be 'sha256:' followed by 64 lowercase hex digits")
return value
class _CanonicalValue(BaseModel):
model_config = ConfigDict(frozen=True, extra="forbid")
model_config = ConfigDict(
frozen=True, extra="forbid", validate_default=True, revalidate_instances="always"
)
class _WithMetadata(_CanonicalValue):
metadata: dict[str, JsonValue] = Field(default_factory=dict)
@field_validator("metadata")
@classmethod
def metadata_has_no_credentials(cls, value: dict[str, JsonValue]) -> dict[str, JsonValue]:
_reject_credentials(value)
return value
_frozen_metadata = field_validator("metadata")(validate_safe_metadata)
class CanonicalDocument(_WithMetadata):
document_id: str = Field(min_length=1)
source_id: str = Field(min_length=1)
source_uri: str = Field(min_length=1)
document_id: str
source_id: str
source_uri: str
source_fingerprint: str = Field(min_length=1)
content_hash: str = Field(min_length=1)
content_hash: str
title: str = ""
content: str
media_type: str = "text/plain"
modified_at: datetime | None = None
pipeline_version: str = Field(min_length=1)
_document_id = field_validator("document_id")(_validate_namespaced_id)
_source_id = field_validator("source_id")(_validate_namespaced_id)
_source_uri = field_validator("source_uri")(validate_canonical_uri)
_source_fingerprint = field_validator("source_fingerprint")(validate_namespaced_value)
_content_hash = field_validator("content_hash")(_validate_hash)
_modified_at = field_validator("modified_at")(normalize_aware_datetime)
class CanonicalChunk(_WithMetadata):
chunk_id: str = Field(min_length=1)
document_id: str = Field(min_length=1)
chunk_id: str
document_id: str
ordinal: int = Field(ge=0)
content: str
content_hash: str = Field(min_length=1)
source_uri: str = Field(min_length=1)
content_hash: str
source_uri: str
pipeline_version: str = Field(min_length=1)
_chunk_id = field_validator("chunk_id")(_validate_namespaced_id)
_document_id = field_validator("document_id")(_validate_namespaced_id)
_content_hash = field_validator("content_hash")(_validate_hash)
_source_uri = field_validator("source_uri")(validate_canonical_uri)
class CorpusManifest(_WithMetadata):
"""Description of one publishable canonical/vector generation."""
"""Description of one internally consistent publishable generation."""
schema_version: int = Field(default=1, ge=1)
manifest_id: str | None = None
created_at: datetime = Field(default_factory=lambda: datetime.now(UTC))
pipeline_version: str = Field(default="1", min_length=1)
pipeline_version: str = Field(default="evidence-v1", min_length=1)
embedding_model: str | None = None
embedding_dimensions: int | None = Field(default=None, gt=0)
vector_generation: str | None = None
documents: list[CanonicalDocument] = Field(default_factory=list)
chunks: list[CanonicalChunk] = Field(default_factory=list)
documents: tuple[CanonicalDocument, ...] = Field(default_factory=tuple)
chunks: tuple[CanonicalChunk, ...] = Field(default_factory=tuple)
_manifest_id = field_validator("manifest_id")(
lambda value: _validate_namespaced_id(value) if value is not None else None
)
_vector_generation = field_validator("vector_generation")(
lambda value: _validate_namespaced_id(value) if value is not None else None
)
_created_at = field_validator("created_at")(normalize_aware_datetime)
@model_validator(mode="after")
def embedding_fields_are_complete(self) -> "CorpusManifest":
def validate_generation(self) -> "CorpusManifest":
if (self.embedding_model is None) != (self.embedding_dimensions is None):
raise ValueError("embedding_model and embedding_dimensions must be set together")
if self.vector_generation is not None and self.embedding_model is None:
raise ValueError("vector_generation requires embedding model and dimension compatibility")
document_ids = [document.document_id for document in self.documents]
source_ids = [document.source_id for document in self.documents]
chunk_ids = [chunk.chunk_id for chunk in self.chunks]
self._require_unique("document_id", document_ids)
self._require_unique("source_id", source_ids)
self._require_unique("chunk_id", chunk_ids)
documents = {document.document_id: document for document in self.documents}
ordinals: dict[str, list[int]] = {}
for document in self.documents:
if document.pipeline_version != self.pipeline_version:
raise ValueError("document pipeline_version must match manifest pipeline_version")
for chunk in self.chunks:
document = documents.get(chunk.document_id)
if document is None:
raise ValueError(f"chunk references unknown document: {chunk.document_id}")
if chunk.pipeline_version != self.pipeline_version:
raise ValueError("chunk pipeline_version must match manifest pipeline_version")
if chunk.source_uri != document.source_uri:
raise ValueError("chunk source_uri must match its document provenance")
ordinals.setdefault(chunk.document_id, []).append(chunk.ordinal)
for document_id, values in ordinals.items():
if sorted(values) != list(range(len(values))):
raise ValueError(f"chunk ordinals must be unique and contiguous for {document_id}")
return self
@staticmethod
def _require_unique(field: str, values: list[str]) -> None:
if len(values) != len(set(values)):
raise ValueError(f"{field} values must be unique")
+134 -28
View File
@@ -1,42 +1,125 @@
"""Port for discovering and acquiring Evidence source objects.
"""Credential-free port for discovering and acquiring Evidence objects."""
Source adapters own transport details and credentials. The values crossing this
boundary are deliberately credential-free so they can safely become provenance.
"""
import re
from collections.abc import Iterable, Mapping, Sequence
from datetime import UTC, datetime
from enum import Enum
from typing import Protocol, runtime_checkable
from urllib.parse import parse_qsl, urlsplit
from datetime import datetime
from typing import Iterable, Protocol, runtime_checkable
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator
from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter, field_validator
_SECRET_KEYS = {
"api_key",
class FrozenDict(dict):
"""A JSON-serializable dict whose mutation operations are disabled."""
def _immutable(self, *args, **kwargs):
raise TypeError("frozen JSON metadata cannot be mutated")
__delitem__ = _immutable
__ior__ = _immutable
__setitem__ = _immutable
clear = _immutable
pop = _immutable
popitem = _immutable
setdefault = _immutable
update = _immutable
_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
_SEPARATORS = re.compile(r"[^a-z0-9]+")
_NAMESPACED_VALUE = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$")
_CREDENTIAL_KEYS = {
"apikey",
"authorization",
"authtoken",
"bearertoken",
"clientsecret",
"credential",
"credentials",
"password",
"secret",
"token",
"passwd",
"privatekey",
"refreshtoken",
"sessioncookie",
"xapikey",
"accesstoken",
}
_JSON_METADATA = TypeAdapter(dict[str, JsonValue])
def _reject_credentials(value: JsonValue, path: str = "metadata") -> JsonValue:
if isinstance(value, dict):
def _normalize_key(key: str) -> str:
return _SEPARATORS.sub("", _CAMEL_BOUNDARY.sub("_", key).lower())
def _is_credential_key(key: str) -> bool:
return _normalize_key(key) in _CREDENTIAL_KEYS
def _reject_credentials(value, path: str = "metadata") -> None:
if isinstance(value, Mapping):
for key, child in value.items():
normalized = key.lower().replace("-", "_")
if normalized in _SECRET_KEYS or normalized.endswith(("_password", "_secret", "_token")):
if _is_credential_key(str(key)):
raise ValueError(f"credential-like metadata key is not allowed: {path}.{key}")
_reject_credentials(child, f"{path}.{key}")
elif isinstance(value, list):
elif isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
for index, child in enumerate(value):
_reject_credentials(child, f"{path}[{index}]")
def freeze_json(value):
"""Recursively freeze a Pydantic-validated JSON value without changing its JSON shape."""
if isinstance(value, Mapping):
return FrozenDict({str(key): freeze_json(child) for key, child in value.items()})
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
return tuple(freeze_json(child) for child in value)
return value
def validate_safe_metadata(value: dict[str, JsonValue]) -> FrozenDict:
_reject_credentials(value)
return freeze_json(value)
def validate_canonical_uri(value: str) -> str:
try:
parsed = urlsplit(value)
_ = parsed.port
except ValueError as error:
raise ValueError("invalid canonical URI") from error
if not parsed.scheme:
raise ValueError("canonical URI must include a scheme")
if parsed.username is not None or parsed.password is not None:
raise ValueError("canonical URI must not contain credentials in userinfo")
for key, _ in parse_qsl(parsed.query, keep_blank_values=True):
if _is_credential_key(key):
raise ValueError("canonical URI must not contain credentials in query parameters")
return value
def normalize_aware_datetime(value: datetime | None) -> datetime | None:
if value is None:
return None
if value.tzinfo is None or value.utcoffset() is None:
raise ValueError("datetime must be timezone-aware")
return value.astimezone(UTC)
def validate_namespaced_value(value: str) -> str:
if not _NAMESPACED_VALUE.fullmatch(value):
raise ValueError("value must be namespaced as '<kind>:<stable-value>'")
return value
class _EvidenceValue(BaseModel):
model_config = ConfigDict(frozen=True, extra="forbid")
model_config = ConfigDict(
frozen=True,
extra="forbid",
revalidate_instances="always",
validate_default=True,
ser_json_bytes="base64",
val_json_bytes="base64",
)
class SourceObject(_EvidenceValue):
@@ -46,25 +129,48 @@ class SourceObject(_EvidenceValue):
modified_at: datetime | None = None
metadata: dict[str, JsonValue] = Field(default_factory=dict)
@field_validator("metadata")
@classmethod
def metadata_has_no_credentials(cls, value: dict[str, JsonValue]) -> dict[str, JsonValue]:
_reject_credentials(value)
return value
_source_id = field_validator("source_id")(validate_namespaced_value)
_fingerprint = field_validator("fingerprint")(validate_namespaced_value)
_safe_uri = field_validator("uri")(validate_canonical_uri)
_aware_modified_at = field_validator("modified_at")(normalize_aware_datetime)
_frozen_metadata = field_validator("metadata")(validate_safe_metadata)
class AcquiredDocument(_EvidenceValue):
"""Transport result; bytes use explicit base64 encoding in JSON mode."""
source: SourceObject
content: bytes
media_type: str | None = None
acquired_at: datetime | None = None
metadata: dict[str, JsonValue] = Field(default_factory=dict)
@field_validator("metadata")
@classmethod
def metadata_has_no_credentials(cls, value: dict[str, JsonValue]) -> dict[str, JsonValue]:
_reject_credentials(value)
return value
_aware_acquired_at = field_validator("acquired_at")(normalize_aware_datetime)
_frozen_metadata = field_validator("metadata")(validate_safe_metadata)
class EvidenceSourceErrorCategory(str, Enum):
TRANSIENT = "transient"
PERMANENT = "permanent"
class EvidenceSourceError(Exception):
"""Classified source failure with credential-free structured diagnostics."""
def __init__(
self,
message: str,
*,
category: EvidenceSourceErrorCategory,
details: dict[str, JsonValue] | None = None,
) -> None:
super().__init__(message)
self.category = EvidenceSourceErrorCategory(category)
self.details = validate_safe_metadata(_JSON_METADATA.validate_python(details or {}))
@property
def retryable(self) -> bool:
return self.category is EvidenceSourceErrorCategory.TRANSIENT
@runtime_checkable