refactor(evidence): remove legacy Python layout (#31)

This commit is contained in:
2026-08-24 02:36:30 +02:00
parent 44f1efa5ba
commit 840848df94
36 changed files with 246 additions and 321 deletions
+4 -4
View File
@@ -1,7 +1,7 @@
import pytest
from tht.adapters.evidence import FilesystemEvidenceSource, HttpManifestEvidenceSource
from tht.adapters.factory import build_evidence_sources
from tht.evidence.adapters import FilesystemEvidenceSource, HttpManifestEvidenceSource
from tht.evidence import build_sources
from tht.config import (
ConfigError,
PgvectorDirectConfig,
@@ -369,7 +369,7 @@ evidence:
assert "example.test" not in repr(cfg.evidence)
assert "example.test" not in cfg.evidence.model_dump_json()
assert cfg.evidence.sources[1].allow_private_hosts is False
sources = build_evidence_sources(cfg)
sources = build_sources(cfg.evidence)
assert isinstance(sources[0], FilesystemEvidenceSource)
assert isinstance(sources[1], HttpManifestEvidenceSource)
assert "example.test" not in repr(sources[1])
@@ -381,7 +381,7 @@ evidence:
source_root: {tmp_path}
evidence_dir: curated
""")
legacy_source = build_evidence_sources(load_config(legacy))[0]
legacy_source = build_sources(load_config(legacy).evidence)[0]
assert isinstance(legacy_source, FilesystemEvidenceSource)
assert legacy_source.root == (tmp_path / "curated").resolve()
+2 -2
View File
@@ -2,8 +2,8 @@ import hashlib
import pytest
from tht.corpus.chunk import ChunkPolicy, chunk
from tht.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.chunk import ChunkPolicy, chunk
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
def document(content: str) -> CanonicalDocument:
+1 -1
View File
@@ -4,7 +4,7 @@ from datetime import UTC, datetime, timedelta, timezone
import pytest
from pydantic import ValidationError
from tht.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
def document(source_uri: str = "https://host/a.md") -> CanonicalDocument:
+2 -2
View File
@@ -3,8 +3,8 @@ from datetime import UTC, datetime
import pytest
from tht.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
from tht.ports.evidence import AcquiredDocument, SourceObject
from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
from tht.evidence.contracts import AcquiredDocument, SourceObject
def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument:
+9 -9
View File
@@ -2,11 +2,11 @@ from datetime import UTC, datetime, timedelta
import pytest
from tht.corpus.chunk import ChunkPolicy
from tht.corpus.pipeline import CorpusPipeline, PipelineError, PipelineResult
from tht.corpus.store import CorpusStore
from tht.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.ports.evidence import AcquiredDocument, SourceObject
from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.pipeline import CorpusPipeline, PipelineError, PipelineResult
from tht.evidence.corpus.store import CorpusStore
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.contracts import AcquiredDocument, SourceObject
from tht.ports.vector import VectorCapabilities, VectorHealth
@@ -273,7 +273,7 @@ def test_gc_preserves_vector_dependencies_of_retained_manifests(tmp_path):
def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path):
from types import SimpleNamespace
from tht.search.evidence import active_searcher
from tht.evidence.search import active_searcher
class Delegate:
def search(self, embedding, top_n=10, kinds=None, metadata_filter=None):
@@ -287,7 +287,7 @@ def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path):
def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_path):
from types import SimpleNamespace
from tht.search.evidence import ActiveEvidenceSearcher
from tht.evidence.search import ActiveEvidenceSearcher
store = CorpusStore(tmp_path / "corpus")
generation = store.stage(
@@ -319,7 +319,7 @@ def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_
def test_active_evidence_query_holds_lock_against_publish(tmp_path):
import threading
from types import SimpleNamespace
from tht.search.evidence import ActiveEvidenceSearcher
from tht.evidence.search import ActiveEvidenceSearcher
first_pipeline = pipeline(tmp_path, Source([(item("one", "a"), "old")]), vectors=Vectors())
first_pipeline.run()
@@ -450,7 +450,7 @@ def test_gc_rejects_workspace_mismatch_without_deleting(tmp_path):
@pytest.mark.parametrize("kinds", [None, ["evidence", "memory"], ["memory"]])
def test_active_search_rejects_workspace_mismatch_before_delegate(tmp_path, kinds):
from tht.search.evidence import ActiveEvidenceSearcher, CorpusWorkspaceMismatchError
from tht.evidence.search import ActiveEvidenceSearcher, CorpusWorkspaceMismatchError
vectors = Vectors()
owner = pipeline(tmp_path, Source([(item("one", "a"), "stable")]), vectors=vectors)
+5 -5
View File
@@ -1,8 +1,8 @@
import pytest
import os
from tht.corpus.models import CorpusManifest
from tht.corpus.store import CorpusStore, UnsafeCorpusPath
from tht.evidence.corpus.models import CorpusManifest
from tht.evidence.corpus.store import CorpusStore, UnsafeCorpusPath
def test_publish_switches_active_atomically_and_resolves_materialized_files(tmp_path):
@@ -60,7 +60,7 @@ def test_publish_restores_previous_active_when_directory_fsync_fails_after_repla
def test_read_document_rejects_symlink_hardlink_and_hash_mismatch(tmp_path):
from tht.corpus.models import CanonicalDocument
from tht.evidence.corpus.models import CanonicalDocument
content = "trusted"
digest = "sha256:" + __import__("hashlib").sha256(content.encode()).hexdigest()
@@ -111,7 +111,7 @@ def test_published_inventory_excludes_staged_and_invalid_newer_directories(tmp_p
def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch):
from tht.corpus.models import CanonicalDocument
from tht.evidence.corpus.models import CanonicalDocument
import hashlib
content = "active bytes"
@@ -140,7 +140,7 @@ def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_
def test_materialized_snapshot_uses_identified_manifest_when_active_changes(tmp_path):
from tht.corpus.models import CanonicalDocument
from tht.evidence.corpus.models import CanonicalDocument
import hashlib
def doc(content, fingerprint):
+13 -27
View File
@@ -1,12 +1,13 @@
from datetime import UTC, datetime
import hashlib
import inspect
from pathlib import Path
from types import SimpleNamespace
import pytest
from tht.corpus.models import CanonicalDocument, CorpusManifest
from tht.corpus.store import CorpusStore
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.decisions import DecisionRecord
from tht.evidence import (
acquire,
@@ -18,15 +19,12 @@ from tht.evidence import (
project_session,
resolve_citation,
)
from tht.ports.evidence import (
from tht.evidence.contracts import (
AcquiredDocument,
EvidenceSourceError,
EvidenceSourceErrorCategory,
SourceObject,
)
from tht.search.evidence import active_searcher as legacy_active_searcher
from tht.search.evidence import resolve_evidence_file
from tht.session.artifacts import build_evidence_entries
from tht.session.models import Candidate, SchemaLinking
@@ -91,8 +89,6 @@ def test_acquisition_facade_preserves_classified_errors():
def test_source_factory_preserves_legacy_first_order_and_filesystem_configuration(tmp_path):
from tht.adapters.factory import build_evidence_sources
legacy_root = tmp_path / "legacy"
configured_root = tmp_path / "configured"
(legacy_root / "evidence").mkdir(parents=True)
@@ -108,16 +104,14 @@ def test_source_factory_preserves_legacy_first_order_and_filesystem_configuratio
)],
))
legacy = build_evidence_sources(cfg)
current = build_sources(cfg.evidence)
assert [type(source) for source in current] == [type(source) for source in legacy]
assert [source.root for source in current] == [
(legacy_root / "evidence").resolve(),
configured_root.resolve(),
]
assert current[1].patterns == legacy[1].patterns == ("*.md",)
assert current[1].max_bytes == legacy[1].max_bytes == 1024
assert current[1].patterns == ("*.md",)
assert current[1].max_bytes == 1024
def test_preprocessing_factory_forwards_only_evidence_pipeline_dependencies(monkeypatch):
@@ -127,7 +121,7 @@ def test_preprocessing_factory_forwards_only_evidence_pipeline_dependencies(monk
def __init__(self, **kwargs):
captured.update(kwargs)
monkeypatch.setattr("tht.corpus.pipeline.CorpusPipeline", FakePipeline)
monkeypatch.setattr("tht.evidence.preprocessing.CorpusPipeline", FakePipeline)
dependencies = {
"store": object(),
"sources": [object()],
@@ -176,18 +170,14 @@ class OrderedDelegate:
def test_search_facade_preserves_active_filtering_and_global_order(tmp_path):
cfg = _active_config(tmp_path)
legacy_delegate = OrderedDelegate()
facade_delegate = OrderedDelegate()
legacy = legacy_active_searcher(
cfg, legacy_delegate, workspace_id="workspace-a",
).search([1.0], top_n=2, kinds=["evidence", "memory"])
current = active_searcher(
cfg, facade_delegate, workspace_id="workspace-a",
).search([1.0], top_n=2, kinds=["evidence", "memory"])
assert [hit.id for hit in current] == [hit.id for hit in legacy] == ["higher", "lower"]
assert facade_delegate.calls == legacy_delegate.calls
assert [hit.id for hit in current] == ["higher", "lower"]
assert facade_delegate.calls == [([1.0], 2, ["memory"], None)]
def test_retrieval_entries_preserve_hit_order_and_existing_projection_shape():
@@ -225,14 +215,12 @@ def test_citation_facade_matches_active_corpus_resolution(tmp_path):
store = _canonical_store(tmp_path / "corpus", "evi-used")
materialized = tmp_path / "materialized"
legacy = resolve_evidence_file(
store, "evi-used", materialized_root=materialized,
)
current = resolve_citation(
store, "evi-used", materialized_root=materialized,
)
assert current == legacy
assert current.endswith(".md")
assert Path(current).read_text(encoding="utf-8") == "# evi-used\n"
assert resolve_citation(store, "missing", materialized_root=materialized) == ""
@@ -245,7 +233,7 @@ def test_session_projection_routes_corpus_citations_through_the_facade(tmp_path,
calls.append((store.root, evidence_id, materialized_root))
return f"/materialized/{evidence_id}.md"
monkeypatch.setattr("tht.evidence.resolve_citation", fake_resolve)
monkeypatch.setattr("tht.evidence.session.resolve_citation", fake_resolve)
linking = SchemaLinking(
question="q",
candidates=[Candidate(
@@ -257,7 +245,7 @@ def test_session_projection_routes_corpus_citations_through_the_facade(tmp_path,
)],
)
assert build_evidence_entries([], linking, evidence_root) == [{
assert project_session([], linking, evidence_root) == [{
"id": "evi-used",
"file": "/materialized/evi-used.md",
"esito": "usata",
@@ -299,10 +287,8 @@ def test_session_projection_facade_preserves_outcome_precedence_and_order(tmp_pa
)],
)
legacy = build_evidence_entries(decisions, linking, evidence_root)
current = project_session(decisions, linking, evidence_root)
assert current == legacy
assert [(row["id"], row["esito"], row["decision_seq"]) for row in current] == [
("used", "usata", 17),
("accepted", "accettata", 21),
+40
View File
@@ -0,0 +1,40 @@
from pathlib import Path
HARNESS_ROOT = Path(__file__).resolve().parents[1]
LEGACY_PATHS = (
"tht/ports/evidence.py",
"tht/adapters/evidence",
"tht/corpus",
"tht/search/evidence.py",
"tht/session/artifacts.py",
)
LEGACY_REFERENCES = (
"tht.ports.evidence",
"tht.adapters.evidence",
"tht.corpus",
"tht.search.evidence",
"tht.session.artifacts",
"build_evidence_sources",
"build_evidence_entries",
)
def test_replaced_evidence_layout_is_removed():
remaining = [path for path in LEGACY_PATHS if (HARNESS_ROOT / path).exists()]
assert remaining == []
def test_production_and_tests_use_only_the_evidence_module():
offenders = []
for root in (HARNESS_ROOT / "tht", HARNESS_ROOT / "tests"):
for path in root.rglob("*.py"):
if path == Path(__file__).resolve() or "__pycache__" in path.parts:
continue
contents = path.read_text(encoding="utf-8")
matched = [reference for reference in LEGACY_REFERENCES if reference in contents]
if matched:
offenders.append((path.relative_to(HARNESS_ROOT).as_posix(), matched))
assert offenders == []
+1 -1
View File
@@ -3,7 +3,7 @@ from datetime import UTC, datetime, timedelta, timezone
import pytest
from pydantic import ValidationError
from tht.ports.evidence import (
from tht.evidence.contracts import (
AcquiredDocument,
EvidenceSource,
EvidenceSourceError,
@@ -2,8 +2,8 @@ import os
import pytest
from tht.adapters.evidence import FilesystemEvidenceSource
from tht.ports.evidence import EvidenceSourceError
from tht.evidence.adapters import FilesystemEvidenceSource
from tht.evidence.contracts import EvidenceSourceError
def test_filesystem_discovery_is_stable_and_acquisition_is_bounded(tmp_path):
+2 -2
View File
@@ -4,8 +4,8 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
import pytest
from tht.adapters.evidence import HttpManifestEvidenceSource
from tht.ports.evidence import EvidenceSourceError
from tht.evidence.adapters import HttpManifestEvidenceSource
from tht.evidence.contracts import EvidenceSourceError
class Handler(BaseHTTPRequestHandler):
@@ -6,8 +6,8 @@ import yaml
from pydantic import SecretStr
from typer.testing import CliRunner
from tht.adapters.evidence import HttpManifestEvidenceSource
from tht.adapters.factory import build_evidence_sources
from tht.evidence.adapters import HttpManifestEvidenceSource
from tht.evidence import build_sources
from tht.cli import app
from tht.config import ConfigError, load_config
@@ -108,7 +108,7 @@ def test_signed_http_file_resolves_in_memory_and_preserves_provenance_order(tmp_
assert_no_canaries(repr(cfg))
assert_no_canaries(cfg.model_dump_json())
adapter = build_evidence_sources(cfg)[0]
adapter = build_sources(cfg.evidence)[0]
assert isinstance(adapter, HttpManifestEvidenceSource)
assert_no_canaries(repr(adapter))
+15 -15
View File
@@ -2,7 +2,7 @@ from datetime import UTC, datetime
import pytest
from tht.ports.evidence import EvidenceSourceError
from tht.evidence.contracts import EvidenceSourceError
class Body:
@@ -28,7 +28,7 @@ class Client:
def test_s3_canonical_uri_version_fingerprint_and_closed_body():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
source = S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client)
item = next(iter(source.discover()))
@@ -39,14 +39,14 @@ def test_s3_canonical_uri_version_fingerprint_and_closed_body():
def test_s3_etag_fallback_and_bounds():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
with pytest.raises(ValueError):
S3EvidenceSource(bucket="evidence", client=client, max_objects=0)
def test_s3_rejects_private_or_insecure_endpoint_without_explicit_opt_in():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
with pytest.raises(ValueError, match="trusted"):
S3EvidenceSource(bucket="evidence", endpoint_url="https://127.0.0.1:9000", client=Client())
with pytest.raises(ValueError, match="HTTPS"):
@@ -59,20 +59,20 @@ def test_s3_rejects_private_or_insecure_endpoint_without_explicit_opt_in():
@pytest.mark.parametrize("bucket", ["UPPER", "bad_bucket", "-start", "end-", "a..b"])
def test_s3_rejects_invalid_bucket_names(bucket):
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
with pytest.raises(ValueError, match="bucket"):
S3EvidenceSource(bucket=bucket, client=Client())
@pytest.mark.parametrize("bucket", ["127.0.0.1", "192.168.1.1"])
def test_s3_rejects_ip_shaped_bucket(bucket):
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
with pytest.raises(ValueError, match="bucket"):
S3EvidenceSource(bucket=bucket, client=Client())
def test_s3_rejects_endpoint_query_path_fragment_and_untrusted_custom_host():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
for endpoint in ("https://s3.example.test/path", "https://s3.example.test/?x=1",
"https://s3.example.test/#x"):
with pytest.raises(ValueError, match="root"):
@@ -83,7 +83,7 @@ def test_s3_rejects_endpoint_query_path_fragment_and_untrusted_custom_host():
def test_s3_rejects_out_of_prefix_key_and_missing_validator():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
client.list_objects_v2 = lambda **kwargs: {"Contents": [{"Key": "other/a.md", "ETag": '"x"'}]}
with pytest.raises(EvidenceSourceError):
@@ -94,7 +94,7 @@ def test_s3_rejects_out_of_prefix_key_and_missing_validator():
def test_s3_rejects_leading_slash_prefix_empty_and_control_keys():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
with pytest.raises(ValueError, match="prefix"):
S3EvidenceSource(bucket="evidence", prefix="/clinical", client=Client())
for key in ("", "clinical/a\x00.md", "clinical/a\x7f.md"):
@@ -106,7 +106,7 @@ def test_s3_rejects_leading_slash_prefix_empty_and_control_keys():
@pytest.mark.parametrize("prefix", ["/bad", "x" * 1025, "bad\x00prefix", "bad\x7fprefix"])
def test_s3_rejects_invalid_prefix_before_client_request(prefix):
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
with pytest.raises(ValueError, match="prefix"):
S3EvidenceSource(bucket="evidence", prefix=prefix, client=client)
@@ -114,7 +114,7 @@ def test_s3_rejects_invalid_prefix_before_client_request(prefix):
def test_s3_hard_page_limit_never_requests_page_max_plus_one():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
def listing(**kwargs):
client.list_calls += 1
@@ -128,7 +128,7 @@ def test_s3_hard_page_limit_never_requests_page_max_plus_one():
def test_s3_acquire_rejects_exact_etag_drift_and_closes_body():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
source = S3EvidenceSource(bucket="evidence", client=client)
item = next(iter(source.discover()))
@@ -140,7 +140,7 @@ def test_s3_acquire_rejects_exact_etag_drift_and_closes_body():
def test_s3_acquire_rejects_forged_reconstructed_item_before_get():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
source = S3EvidenceSource(bucket="evidence", client=client)
item = next(iter(source.discover()))
@@ -153,14 +153,14 @@ def test_s3_acquire_rejects_forged_reconstructed_item_before_get():
@pytest.mark.parametrize("host", ["127.0.0.1", "10.0.0.1", "169.254.1.1", "0.0.0.0",
"[::1]", "[fe80::1]", "[::]"])
def test_s3_literal_non_global_endpoint_requires_private_opt_in(host):
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
with pytest.raises(ValueError, match="private"):
S3EvidenceSource(bucket="evidence", endpoint_url=f"https://{host}:9000",
trusted_endpoint=True, client=Client())
def test_s3_size_limit_closes_body():
from tht.adapters.evidence.s3 import S3EvidenceSource
from tht.evidence.adapters.s3 import S3EvidenceSource
client = Client()
source = S3EvidenceSource(bucket="evidence", client=client, max_bytes=4)
item = next(iter(source.discover()))
@@ -6,10 +6,10 @@ from datetime import UTC, datetime
from tht.adapters.vector.qdrant import point_id
from tht.cli.vector_cmd import sync_canonical_records
from tht.corpus.chunk import ChunkPolicy
from tht.corpus.models import CanonicalChunk
from tht.corpus.pipeline import CorpusPipeline
from tht.corpus.store import CorpusStore
from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.models import CanonicalChunk
from tht.evidence.corpus.pipeline import CorpusPipeline
from tht.evidence.corpus.store import CorpusStore
from tht.memory import MemoryRecord, save_one_memory
from tht.mschema.models import (
Annotations,
@@ -6,8 +6,8 @@ import hashlib
import pytest
from tht.corpus.models import CanonicalDocument, CorpusManifest
from tht.corpus.store import CorpusStore
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.decisions import DecisionRecord, append_decision
from tht.evidence import project_session
from tht.phase import current_phase, effective_decisions