feat(evidence): add filesystem and HTTP sources
This commit is contained in:
@@ -0,0 +1,124 @@
|
||||
"""Contained, deterministic filesystem Evidence source."""
|
||||
|
||||
import hashlib
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
|
||||
from tht.ports.evidence import (
|
||||
AcquiredDocument,
|
||||
EvidenceSourceError,
|
||||
EvidenceSourceErrorCategory,
|
||||
SourceObject,
|
||||
)
|
||||
|
||||
|
||||
class FilesystemEvidenceSource:
|
||||
def __init__(
|
||||
self,
|
||||
root: Path | str,
|
||||
*,
|
||||
patterns: tuple[str, ...] | list[str] = ("**/*.md",),
|
||||
max_bytes: int = 10 * 1024 * 1024,
|
||||
) -> None:
|
||||
if max_bytes < 1:
|
||||
raise ValueError("max_bytes must be positive")
|
||||
if not patterns or any(not pattern for pattern in patterns):
|
||||
raise ValueError("at least one non-empty discovery pattern is required")
|
||||
try:
|
||||
self.root = Path(root).expanduser().resolve(strict=True)
|
||||
except OSError as error:
|
||||
raise ValueError("filesystem evidence root is unavailable") from error
|
||||
if not self.root.is_dir():
|
||||
raise ValueError("filesystem evidence root must be a directory")
|
||||
self.patterns = tuple(patterns)
|
||||
self.max_bytes = max_bytes
|
||||
|
||||
def _contained(self, path: Path) -> Path:
|
||||
try:
|
||||
resolved = path.resolve(strict=True)
|
||||
resolved.relative_to(self.root)
|
||||
except (OSError, ValueError) as error:
|
||||
raise EvidenceSourceError(
|
||||
"unsafe filesystem object",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "path_validation"},
|
||||
) from error
|
||||
if not resolved.is_file():
|
||||
raise EvidenceSourceError(
|
||||
"unsupported filesystem object",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "path_validation"},
|
||||
)
|
||||
return resolved
|
||||
|
||||
def _read(self, path: Path) -> bytes:
|
||||
try:
|
||||
if path.stat().st_size > self.max_bytes:
|
||||
raise EvidenceSourceError(
|
||||
"filesystem object exceeds configured limit",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "read", "limit_bytes": self.max_bytes},
|
||||
)
|
||||
with path.open("rb") as stream:
|
||||
content = stream.read(self.max_bytes + 1)
|
||||
except EvidenceSourceError:
|
||||
raise
|
||||
except OSError as error:
|
||||
raise EvidenceSourceError(
|
||||
"filesystem read failed",
|
||||
category=EvidenceSourceErrorCategory.TRANSIENT,
|
||||
details={"operation": "read"},
|
||||
) from error
|
||||
if len(content) > self.max_bytes:
|
||||
raise EvidenceSourceError(
|
||||
"filesystem object exceeds configured limit",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "read", "limit_bytes": self.max_bytes},
|
||||
)
|
||||
return content
|
||||
|
||||
def _item(self, path: Path, content: bytes) -> SourceObject:
|
||||
relative = path.relative_to(self.root).as_posix()
|
||||
digest = hashlib.sha256(content).hexdigest()
|
||||
stable_id = hashlib.sha256(relative.encode()).hexdigest()
|
||||
modified = datetime.fromtimestamp(path.stat().st_mtime, tz=UTC)
|
||||
return SourceObject(
|
||||
source_id=f"filesystem:{stable_id}",
|
||||
uri=path.as_uri(),
|
||||
fingerprint=f"sha256:{digest}",
|
||||
modified_at=modified,
|
||||
metadata={"relative_path": relative},
|
||||
)
|
||||
|
||||
def discover(self):
|
||||
candidates = {path for pattern in self.patterns for path in self.root.glob(pattern)}
|
||||
for candidate in sorted(candidates, key=lambda path: path.as_posix()):
|
||||
path = self._contained(candidate)
|
||||
content = self._read(path)
|
||||
yield self._item(path, content)
|
||||
|
||||
def acquire(self, item: SourceObject) -> AcquiredDocument:
|
||||
if not item.uri.startswith("file:"):
|
||||
raise EvidenceSourceError(
|
||||
"object does not belong to filesystem source",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "acquire"},
|
||||
)
|
||||
from urllib.parse import unquote, urlsplit
|
||||
|
||||
parsed = urlsplit(item.uri)
|
||||
path = self._contained(Path(unquote(parsed.path)))
|
||||
content = self._read(path)
|
||||
expected = self._item(path, content)
|
||||
if item.source_id != expected.source_id:
|
||||
raise EvidenceSourceError(
|
||||
"object does not belong to filesystem source",
|
||||
category=EvidenceSourceErrorCategory.PERMANENT,
|
||||
details={"operation": "acquire"},
|
||||
)
|
||||
return AcquiredDocument(
|
||||
source=expected,
|
||||
content=content,
|
||||
media_type="text/markdown" if path.suffix.lower() == ".md" else None,
|
||||
acquired_at=datetime.now(UTC),
|
||||
)
|
||||
Reference in New Issue
Block a user