125 lines
4.8 KiB
Python
125 lines
4.8 KiB
Python
"""Contained, deterministic filesystem Evidence source."""
|
|
|
|
import hashlib
|
|
from datetime import UTC, datetime
|
|
from pathlib import Path
|
|
|
|
from tht.ports.evidence import (
|
|
AcquiredDocument,
|
|
EvidenceSourceError,
|
|
EvidenceSourceErrorCategory,
|
|
SourceObject,
|
|
)
|
|
|
|
|
|
class FilesystemEvidenceSource:
|
|
def __init__(
|
|
self,
|
|
root: Path | str,
|
|
*,
|
|
patterns: tuple[str, ...] | list[str] = ("**/*.md",),
|
|
max_bytes: int = 10 * 1024 * 1024,
|
|
) -> None:
|
|
if max_bytes < 1:
|
|
raise ValueError("max_bytes must be positive")
|
|
if not patterns or any(not pattern for pattern in patterns):
|
|
raise ValueError("at least one non-empty discovery pattern is required")
|
|
try:
|
|
self.root = Path(root).expanduser().resolve(strict=True)
|
|
except OSError as error:
|
|
raise ValueError("filesystem evidence root is unavailable") from error
|
|
if not self.root.is_dir():
|
|
raise ValueError("filesystem evidence root must be a directory")
|
|
self.patterns = tuple(patterns)
|
|
self.max_bytes = max_bytes
|
|
|
|
def _contained(self, path: Path) -> Path:
|
|
try:
|
|
resolved = path.resolve(strict=True)
|
|
resolved.relative_to(self.root)
|
|
except (OSError, ValueError) as error:
|
|
raise EvidenceSourceError(
|
|
"unsafe filesystem object",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "path_validation"},
|
|
) from error
|
|
if not resolved.is_file():
|
|
raise EvidenceSourceError(
|
|
"unsupported filesystem object",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "path_validation"},
|
|
)
|
|
return resolved
|
|
|
|
def _read(self, path: Path) -> bytes:
|
|
try:
|
|
if path.stat().st_size > self.max_bytes:
|
|
raise EvidenceSourceError(
|
|
"filesystem object exceeds configured limit",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "read", "limit_bytes": self.max_bytes},
|
|
)
|
|
with path.open("rb") as stream:
|
|
content = stream.read(self.max_bytes + 1)
|
|
except EvidenceSourceError:
|
|
raise
|
|
except OSError as error:
|
|
raise EvidenceSourceError(
|
|
"filesystem read failed",
|
|
category=EvidenceSourceErrorCategory.TRANSIENT,
|
|
details={"operation": "read"},
|
|
) from error
|
|
if len(content) > self.max_bytes:
|
|
raise EvidenceSourceError(
|
|
"filesystem object exceeds configured limit",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "read", "limit_bytes": self.max_bytes},
|
|
)
|
|
return content
|
|
|
|
def _item(self, path: Path, content: bytes) -> SourceObject:
|
|
relative = path.relative_to(self.root).as_posix()
|
|
digest = hashlib.sha256(content).hexdigest()
|
|
stable_id = hashlib.sha256(relative.encode()).hexdigest()
|
|
modified = datetime.fromtimestamp(path.stat().st_mtime, tz=UTC)
|
|
return SourceObject(
|
|
source_id=f"filesystem:{stable_id}",
|
|
uri=path.as_uri(),
|
|
fingerprint=f"sha256:{digest}",
|
|
modified_at=modified,
|
|
metadata={"relative_path": relative},
|
|
)
|
|
|
|
def discover(self):
|
|
candidates = {path for pattern in self.patterns for path in self.root.glob(pattern)}
|
|
for candidate in sorted(candidates, key=lambda path: path.as_posix()):
|
|
path = self._contained(candidate)
|
|
content = self._read(path)
|
|
yield self._item(path, content)
|
|
|
|
def acquire(self, item: SourceObject) -> AcquiredDocument:
|
|
if not item.uri.startswith("file:"):
|
|
raise EvidenceSourceError(
|
|
"object does not belong to filesystem source",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "acquire"},
|
|
)
|
|
from urllib.parse import unquote, urlsplit
|
|
|
|
parsed = urlsplit(item.uri)
|
|
path = self._contained(Path(unquote(parsed.path)))
|
|
content = self._read(path)
|
|
expected = self._item(path, content)
|
|
if item.source_id != expected.source_id:
|
|
raise EvidenceSourceError(
|
|
"object does not belong to filesystem source",
|
|
category=EvidenceSourceErrorCategory.PERMANENT,
|
|
details={"operation": "acquire"},
|
|
)
|
|
return AcquiredDocument(
|
|
source=expected,
|
|
content=content,
|
|
media_type="text/markdown" if path.suffix.lower() == ".md" else None,
|
|
acquired_at=datetime.now(UTC),
|
|
)
|