Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
400 lines
14 KiB
Python
400 lines
14 KiB
Python
import json
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
from typer.testing import CliRunner
|
|
|
|
from tht.cli import app
|
|
|
|
|
|
def _runtime_config(
|
|
tmp_path: Path, name: str = "workspace.yaml", evidence_schema_version: int | None = None,
|
|
) -> Path:
|
|
path = tmp_path / name
|
|
(tmp_path / "evidence").mkdir(exist_ok=True)
|
|
evidence_version = "" if evidence_schema_version is None else f"\n schema_version: {evidence_schema_version}"
|
|
path.write_text(
|
|
f"""
|
|
runtime_identity:
|
|
workspace_id: psd-clinical
|
|
workspace_revision: {'a' * 40}
|
|
dwh:
|
|
type: postgres_direct
|
|
connection: {{database: analytics, schema: mart, user: reader, password: secret}}
|
|
vectors:
|
|
type: qdrant
|
|
base_url: http://qdrant:6333
|
|
collection: psd-clinical
|
|
embeddings:
|
|
provider: ollama_internal
|
|
base_url: http://embedding:11434
|
|
model: qwen3-embedding:0.6b
|
|
dim: 1024
|
|
evidence:
|
|
{evidence_version}
|
|
sources:
|
|
- type: filesystem
|
|
root: {tmp_path / 'evidence'}
|
|
roots:
|
|
sessions: {tmp_path / 'sessions'}
|
|
artifacts: {tmp_path / 'artifacts'}
|
|
indexes: {tmp_path / 'indexes'}
|
|
"""
|
|
)
|
|
return path
|
|
|
|
|
|
def test_v2_invalid_materialized_corpus_stops_before_vector_store_construction(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
from tht.evidence import ValidationFinding, ValidationReport
|
|
|
|
config = _runtime_config(tmp_path, evidence_schema_version=2)
|
|
vector_store_constructed = False
|
|
monkeypatch.setattr(
|
|
"tht.evidence.validate_workspace_evidence",
|
|
lambda root: ValidationReport((ValidationFinding(
|
|
"error", "manifest_missing", "manifest.yaml", "missing",
|
|
),)),
|
|
)
|
|
|
|
def forbidden_vector_store(*args, **kwargs):
|
|
nonlocal vector_store_constructed
|
|
vector_store_constructed = True
|
|
raise AssertionError("invalid corpus must not reach vector upsert setup")
|
|
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", forbidden_vector_store)
|
|
|
|
with pytest.raises(RuntimeError, match="curated Evidence corpus is invalid"):
|
|
command.run_from_config(config)
|
|
|
|
assert vector_store_constructed is False
|
|
|
|
|
|
def test_v2_valid_materialized_corpus_is_validated_before_preprocessing(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
from tht.evidence import ValidationReport
|
|
|
|
config = _runtime_config(tmp_path, evidence_schema_version=2)
|
|
calls = []
|
|
|
|
class FakePipeline:
|
|
def run_as_job(self, **kwargs):
|
|
calls.append(("run", kwargs))
|
|
return "completed"
|
|
|
|
monkeypatch.setattr("tht.evidence.validate_workspace_evidence", lambda root: calls.append(("validate", root)) or ValidationReport(()))
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: calls.append(("vector", require_write)) or object())
|
|
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
|
monkeypatch.setattr("tht.evidence.build_sources", lambda evidence: [])
|
|
monkeypatch.setattr("tht.evidence.build_preprocessing_pipeline", lambda **kwargs: FakePipeline())
|
|
|
|
assert command.run_from_config(config) == "completed"
|
|
assert calls[:2] == [("validate", tmp_path), ("vector", True)]
|
|
|
|
|
|
def test_clear_removes_only_derived_paths_and_preserves_memory(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
monkeypatch.setenv("THT_PROFILE", "server")
|
|
config = _runtime_config(tmp_path)
|
|
for path in (
|
|
tmp_path / ".tht-dwh",
|
|
tmp_path / ".tht-jobs",
|
|
tmp_path / "corpus",
|
|
tmp_path / "indexes" / "lsh",
|
|
):
|
|
path.mkdir(parents=True, exist_ok=True)
|
|
(path / "derived").write_text("generated", encoding="utf-8")
|
|
physical = tmp_path / "artifacts" / "mschema" / "physical.yaml"
|
|
physical.parent.mkdir(parents=True)
|
|
physical.write_text("generated", encoding="utf-8")
|
|
memory = tmp_path / "memory" / "registry.jsonl"
|
|
memory.parent.mkdir()
|
|
memory.write_text("durable", encoding="utf-8")
|
|
|
|
class Store:
|
|
def clear_reference(self):
|
|
return True
|
|
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *_args, **_kwargs: Store())
|
|
counts = command.clear_from_config(config)
|
|
|
|
assert counts == {"referenceCollections": 1, "derivedPaths": 5}
|
|
assert memory.read_text(encoding="utf-8") == "durable"
|
|
assert not (tmp_path / ".tht-dwh").exists()
|
|
assert not (tmp_path / "indexes" / "lsh").exists()
|
|
|
|
|
|
def test_preprocess_evidence_json_is_pristine(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "succeeded",
|
|
"generation": "gen:abc",
|
|
"published": True,
|
|
"counts": {"changed": 0, "unchanged": 0, "removed": 0, "documents": 0, "chunks": 0},
|
|
"changed": [],
|
|
"unchanged": [],
|
|
"removed": [],
|
|
"manifest_id": "manifest-1",
|
|
"run_id": "a" * 32,
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 0, response.output
|
|
assert response.stderr == ""
|
|
assert json.loads(response.stdout) == {
|
|
"changed": [],
|
|
"code": "ok",
|
|
"counts": {"changed": 0, "chunks": 0, "documents": 0, "removed": 0, "unchanged": 0},
|
|
"generation": "gen:abc",
|
|
"manifest_id": "manifest-1",
|
|
"operation": "preprocess_evidence",
|
|
"published": True,
|
|
"removed": [],
|
|
"resumed_from": None,
|
|
"run_id": "a" * 32,
|
|
"schemaVersion": 1,
|
|
"status": "succeeded",
|
|
"unchanged": [],
|
|
"workspaceId": "psd-clinical",
|
|
"workspaceRevision": "a" * 40,
|
|
}
|
|
|
|
|
|
def test_preprocess_failure_is_structured_and_nonzero(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
monkeypatch.setattr(
|
|
command,
|
|
"run_from_config",
|
|
lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail")),
|
|
)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code != 0
|
|
payload = json.loads(response.stdout)
|
|
assert payload == {
|
|
"code": "preprocessing_failed",
|
|
"error": "preprocessing failed",
|
|
"operation": "preprocess_evidence",
|
|
"schemaVersion": 1,
|
|
"status": "failed",
|
|
"workspaceId": "psd-clinical",
|
|
"workspaceRevision": "a" * 40,
|
|
}
|
|
assert "secret detail" not in response.output
|
|
|
|
|
|
def test_preprocess_failed_job_report_is_sanitized_json_and_nonzero(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "failed",
|
|
"run_id": "a" * 32,
|
|
"published": False,
|
|
"generation": "gen:" + "b" * 32,
|
|
"changed": ["fs:one"],
|
|
"unchanged": [],
|
|
"removed": [],
|
|
"counts": {"changed": 1, "unchanged": 0, "removed": 0, "documents": 1, "chunks": 1},
|
|
"manifest_id": "manifest-1",
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 1
|
|
payload = json.loads(response.stdout)
|
|
assert payload["status"] == "failed"
|
|
assert payload["error"] == "preprocessing job failed"
|
|
assert payload["workspaceId"] == "psd-clinical"
|
|
assert "traceback" not in response.output.lower()
|
|
|
|
|
|
def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path):
|
|
from test_corpus_pipeline import Source, item, pipeline
|
|
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = pipeline(
|
|
tmp_path,
|
|
Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))]),
|
|
).run_as_job(
|
|
workspace_id="demo",
|
|
workspace_root=tmp_path,
|
|
config_fingerprint="sha256:" + "1" * 64,
|
|
input_fingerprint="sha256:" + "2" * 64,
|
|
)
|
|
assert result.status == "failed"
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
|
|
assert response.exit_code == 1
|
|
assert json.loads(response.stdout)["status"] == "failed"
|
|
assert "SENSITIVE EVIDENCE" not in response.output
|
|
assert "secret" not in response.output
|
|
|
|
|
|
def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
result = SimpleNamespace(
|
|
model_dump=lambda mode=None: {
|
|
"status": "succeeded",
|
|
"run_id": "a" * 32,
|
|
"generation": "gen:" + "b" * 64,
|
|
"published": True,
|
|
"changed": ["fs:item"] * 100,
|
|
"unchanged": ["fs:item"] * 100,
|
|
"removed": ["fs:item"] * 100,
|
|
"counts": {"changed": 1001, "unchanged": 902, "removed": 803},
|
|
"manifest_id": "manifest-1",
|
|
"resumed_from": None,
|
|
}
|
|
)
|
|
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
|
|
|
|
response = CliRunner().invoke(app, ["preprocess", "evidence", "-c", str(config)])
|
|
|
|
assert response.exit_code == 0, response.output
|
|
assert "changed=1001 unchanged=902 removed=803" in response.output
|
|
|
|
|
|
def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
called = False
|
|
|
|
def forbidden(*args, **kwargs):
|
|
nonlocal called
|
|
called = True
|
|
|
|
monkeypatch.setattr(command, "run_from_config", forbidden)
|
|
response = CliRunner().invoke(
|
|
app,
|
|
[
|
|
"preprocess",
|
|
"evidence",
|
|
"--resume",
|
|
"gen:" + "a" * 32,
|
|
"--json",
|
|
"-c",
|
|
str(config),
|
|
],
|
|
)
|
|
assert response.exit_code != 0
|
|
assert json.loads(response.output) == {
|
|
"code": "invalid_resume",
|
|
"error": "resume requires a preprocessing run id",
|
|
"operation": "preprocess_evidence",
|
|
"schemaVersion": 1,
|
|
"status": "failed",
|
|
}
|
|
assert called is False
|
|
|
|
|
|
def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path, name="3")
|
|
calls = {}
|
|
|
|
class FakePipeline:
|
|
def __init__(
|
|
self,
|
|
*,
|
|
store,
|
|
sources,
|
|
embedder,
|
|
vector_store,
|
|
embedding_id,
|
|
embedding_model,
|
|
embedding_dimensions,
|
|
chunk_policy,
|
|
pipeline_version,
|
|
retain_published_generations,
|
|
sparse_language,
|
|
candidate_evaluator,
|
|
):
|
|
calls["init"] = {
|
|
"embedding_id": embedding_id,
|
|
"embedding_model": embedding_model,
|
|
"embedding_dimensions": embedding_dimensions,
|
|
"pipeline_version": pipeline_version,
|
|
"sparse_language": sparse_language,
|
|
"candidate_evaluator": candidate_evaluator,
|
|
}
|
|
|
|
def run_as_job(self, **kwargs):
|
|
calls["run_as_job"] = kwargs
|
|
return SimpleNamespace(model_dump=lambda mode=None: {"status": "succeeded"})
|
|
|
|
monkeypatch.setattr("tht.evidence.build_sources", lambda cfg: [])
|
|
monkeypatch.setattr(
|
|
"tht.evidence.build_preprocessing_pipeline",
|
|
lambda **kwargs: FakePipeline(**kwargs),
|
|
)
|
|
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: object())
|
|
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
|
|
|
command.run_from_config(config)
|
|
|
|
assert calls["init"]["embedding_id"] == "ollama/qwen3-embedding:0.6b"
|
|
assert calls["init"]["sparse_language"] == "english"
|
|
assert calls["init"]["candidate_evaluator"] is None
|
|
assert calls["run_as_job"]["workspace_id"] == "psd-clinical"
|
|
assert calls["run_as_job"]["input_fingerprint"] != calls["run_as_job"]["config_fingerprint"]
|
|
|
|
|
|
@pytest.mark.parametrize("schema_version, source_type", [
|
|
(1, "filesystem"),
|
|
(1, "http"),
|
|
(1, "s3"),
|
|
(2, "http"),
|
|
(2, "s3"),
|
|
])
|
|
def test_candidate_evaluation_is_not_required_outside_v2_filesystem_corpora(
|
|
schema_version, source_type,
|
|
):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
cfg = SimpleNamespace(
|
|
evidence=SimpleNamespace(
|
|
schema_version=schema_version,
|
|
local_archive_root=None,
|
|
source_root=None,
|
|
sources=[SimpleNamespace(type=source_type)],
|
|
),
|
|
language="it",
|
|
)
|
|
|
|
assert command._candidate_evaluator(cfg, vector_store=object(), embedder=object()) is None
|
|
|
|
|
|
def test_preprocess_evidence_gc_json_is_pristine(monkeypatch, tmp_path):
|
|
import tht.cli.preprocess_cmd as command
|
|
|
|
config = _runtime_config(tmp_path)
|
|
monkeypatch.setattr(command, "gc_from_config", lambda *a, **k: {
|
|
"status": "succeeded",
|
|
"dry_run": True,
|
|
"evicted": [],
|
|
"failures": [],
|
|
})
|
|
response = CliRunner().invoke(
|
|
app,
|
|
["preprocess", "evidence", "gc", "--dry-run", "--json", "-c", str(config)],
|
|
)
|
|
assert response.exit_code == 0, response.output
|
|
assert json.loads(response.output)["dry_run"] is True
|