feat: pristine harness JSON interfaces and require-existing semantic mode (P2)

This commit is contained in:
2026-08-11 18:40:11 +02:00
parent 3cfc8c53e6
commit ca391ba59c
12 changed files with 1204 additions and 252 deletions
+5 -1
View File
@@ -8,7 +8,11 @@ def test_local_compose_uses_the_generic_external_endpoint_contract():
compose = yaml.safe_load((root / "compose.yaml").read_text())
local = yaml.safe_load((root / "deploy/compose.local.yaml").read_text())
assert set(compose["services"]) == {"core", "frontend", "qdrant", "embedding", "embedding-model-init"}
assert set(compose["services"]) == {
"core", "frontend", "qdrant", "embedding", "embedding-model-init", "workspace-maintenance",
}
# workspace-maintenance is profile-gated: it must not be part of the default local startup.
assert compose["services"]["workspace-maintenance"].get("profiles") == ["workspace-maintenance"]
assert local["services"]["core"]["environment"]["AUTH_MODE"] == "none"
assert local["services"]["core"]["ports"] == ["127.0.0.1:${THOTH_CORE_HTTP_PORT:-8787}:8787"]
assert local["services"]["frontend"]["ports"] == ["127.0.0.1:${THOTH_HTTP_PORT:-8080}:8080"]
+188 -43
View File
@@ -1,4 +1,5 @@
import json
from pathlib import Path
from types import SimpleNamespace
from typer.testing import CliRunner
@@ -6,68 +7,152 @@ from typer.testing import CliRunner
from tht.cli import app
def _runtime_config(tmp_path: Path, name: str = "workspace.yaml") -> Path:
path = tmp_path / name
(tmp_path / "evidence").mkdir(exist_ok=True)
path.write_text(
f"""
runtime_identity:
workspace_id: psd-clinical
workspace_revision: {'a' * 40}
dwh:
type: postgres_direct
connection: {{database: analytics, schema: mart, user: reader, password: secret}}
vectors:
type: qdrant
base_url: http://qdrant:6333
collection: psd-clinical
embeddings:
provider: ollama_internal
base_url: http://embedding:11434
model: qwen3-embedding:0.6b
dim: 1024
evidence:
sources:
- type: filesystem
root: {tmp_path / 'evidence'}
roots:
sessions: {tmp_path / 'sessions'}
artifacts: {tmp_path / 'artifacts'}
indexes: {tmp_path / 'indexes'}
"""
)
return path
def test_preprocess_evidence_json_is_pristine(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
result = SimpleNamespace(model_dump=lambda mode=None: {
"status": "succeeded", "generation": "gen:abc", "published": True
})
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(
app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")]
config = _runtime_config(tmp_path)
result = SimpleNamespace(
model_dump=lambda mode=None: {
"status": "succeeded",
"generation": "gen:abc",
"published": True,
"counts": {"changed": 0, "unchanged": 0, "removed": 0, "documents": 0, "chunks": 0},
"changed": [],
"unchanged": [],
"removed": [],
"manifest_id": "manifest-1",
"run_id": "a" * 32,
"resumed_from": None,
}
)
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
assert response.exit_code == 0, response.output
assert json.loads(response.output)["generation"] == "gen:abc"
assert response.stderr == ""
assert json.loads(response.stdout) == {
"changed": [],
"code": "ok",
"counts": {"changed": 0, "chunks": 0, "documents": 0, "removed": 0, "unchanged": 0},
"generation": "gen:abc",
"manifest_id": "manifest-1",
"operation": "preprocess_evidence",
"published": True,
"removed": [],
"resumed_from": None,
"run_id": "a" * 32,
"schemaVersion": 1,
"status": "succeeded",
"unchanged": [],
"workspaceId": "psd-clinical",
"workspaceRevision": "a" * 40,
}
def test_preprocess_failure_is_structured_and_nonzero(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
monkeypatch.setattr(command, "run_from_config", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail")))
response = CliRunner().invoke(
app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")]
config = _runtime_config(tmp_path)
monkeypatch.setattr(
command,
"run_from_config",
lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail")),
)
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
assert response.exit_code != 0
assert json.loads(response.output) == {"status": "failed", "error": "preprocessing failed"}
payload = json.loads(response.stdout)
assert payload == {
"code": "preprocessing_failed",
"error": "preprocessing failed",
"operation": "preprocess_evidence",
"schemaVersion": 1,
"status": "failed",
"workspaceId": "psd-clinical",
"workspaceRevision": "a" * 40,
}
assert "secret detail" not in response.output
def test_preprocess_failed_job_report_is_sanitized_json_and_nonzero(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
result = SimpleNamespace(model_dump=lambda mode=None: {
"status": "failed", "run_id": "a" * 32, "published": False,
"generation": "gen:" + "b" * 32, "changed": ["fs:one"],
})
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(
app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")]
config = _runtime_config(tmp_path)
result = SimpleNamespace(
model_dump=lambda mode=None: {
"status": "failed",
"run_id": "a" * 32,
"published": False,
"generation": "gen:" + "b" * 32,
"changed": ["fs:one"],
"unchanged": [],
"removed": [],
"counts": {"changed": 1, "unchanged": 0, "removed": 0, "documents": 1, "chunks": 1},
"manifest_id": "manifest-1",
"resumed_from": None,
}
)
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
assert response.exit_code == 1
payload = json.loads(response.output)
payload = json.loads(response.stdout)
assert payload["status"] == "failed"
assert payload["error"] == "preprocessing job failed"
assert payload["workspaceId"] == "psd-clinical"
assert "traceback" not in response.output.lower()
def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
from test_corpus_pipeline import Source, item, pipeline
import tht.cli.preprocess_cmd as command
config = _runtime_config(tmp_path)
result = pipeline(
tmp_path, Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))])
tmp_path,
Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))]),
).run_as_job(
workspace_id="demo", workspace_root=tmp_path,
workspace_id="demo",
workspace_root=tmp_path,
config_fingerprint="sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64,
)
assert result.status == "failed"
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(
app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")]
)
response = CliRunner().invoke(app, ["preprocess", "evidence", "--json", "-c", str(config)])
assert response.exit_code == 1
assert json.loads(response.output)["status"] == "failed"
assert json.loads(response.stdout)["status"] == "failed"
assert "SENSITIVE EVIDENCE" not in response.output
assert "secret" not in response.output
@@ -75,19 +160,24 @@ def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path
def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
result = SimpleNamespace(model_dump=lambda mode=None: {
"status": "succeeded", "run_id": "a" * 32,
"generation": "gen:" + "b" * 64, "published": True,
"changed": ["fs:item"] * 100,
"unchanged": ["fs:item"] * 100,
"removed": ["fs:item"] * 100,
"counts": {"changed": 1001, "unchanged": 902, "removed": 803},
})
config = _runtime_config(tmp_path)
result = SimpleNamespace(
model_dump=lambda mode=None: {
"status": "succeeded",
"run_id": "a" * 32,
"generation": "gen:" + "b" * 64,
"published": True,
"changed": ["fs:item"] * 100,
"unchanged": ["fs:item"] * 100,
"removed": ["fs:item"] * 100,
"counts": {"changed": 1001, "unchanged": 902, "removed": 803},
"manifest_id": "manifest-1",
"resumed_from": None,
}
)
monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result)
response = CliRunner().invoke(
app, ["preprocess", "evidence", "-c", str(tmp_path / "workspace.yaml")]
)
response = CliRunner().invoke(app, ["preprocess", "evidence", "-c", str(config)])
assert response.exit_code == 0, response.output
assert "changed=1001 unchanged=902 removed=803" in response.output
@@ -96,6 +186,7 @@ def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_p
def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
config = _runtime_config(tmp_path)
called = False
def forbidden(*args, **kwargs):
@@ -106,26 +197,80 @@ def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatc
response = CliRunner().invoke(
app,
[
"preprocess", "evidence", "--resume", "gen:" + "a" * 32,
"--json", "-c", str(tmp_path / "workspace.yaml"),
"preprocess",
"evidence",
"--resume",
"gen:" + "a" * 32,
"--json",
"-c",
str(config),
],
)
assert response.exit_code != 0
assert json.loads(response.output) == {
"status": "failed", "error": "resume requires a preprocessing run id"
"code": "invalid_resume",
"error": "resume requires a preprocessing run id",
"operation": "preprocess_evidence",
"schemaVersion": 1,
"status": "failed",
}
assert called is False
def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
config = _runtime_config(tmp_path, name="3")
calls = {}
class FakePipeline:
def __init__(
self,
*,
store,
sources,
embedder,
vector_store,
embedding_model,
embedding_dimensions,
chunk_policy,
pipeline_version,
retain_published_generations,
):
calls["init"] = {
"embedding_model": embedding_model,
"embedding_dimensions": embedding_dimensions,
"pipeline_version": pipeline_version,
}
def run_as_job(self, **kwargs):
calls["run_as_job"] = kwargs
return SimpleNamespace(model_dump=lambda mode=None: {"status": "succeeded"})
monkeypatch.setattr("tht.adapters.factory.build_evidence_sources", lambda cfg: [])
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: object())
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
monkeypatch.setattr("tht.corpus.pipeline.CorpusPipeline", FakePipeline)
command.run_from_config(config)
assert calls["run_as_job"]["workspace_id"] == "psd-clinical"
assert calls["run_as_job"]["input_fingerprint"] != calls["run_as_job"]["config_fingerprint"]
def test_preprocess_evidence_gc_json_is_pristine(monkeypatch, tmp_path):
import tht.cli.preprocess_cmd as command
config = _runtime_config(tmp_path)
monkeypatch.setattr(command, "gc_from_config", lambda *a, **k: {
"status": "succeeded", "dry_run": True, "evicted": [], "failures": [],
"status": "succeeded",
"dry_run": True,
"evicted": [],
"failures": [],
})
response = CliRunner().invoke(
app, ["preprocess", "evidence", "gc", "--dry-run", "--json", "-c",
str(tmp_path / "workspace.yaml")]
app,
["preprocess", "evidence", "gc", "--dry-run", "--json", "-c", str(config)],
)
assert response.exit_code == 0, response.output
assert json.loads(response.output)["dry_run"] is True
+74 -3
View File
@@ -1,5 +1,6 @@
from __future__ import annotations
import hashlib
import json
from datetime import UTC, datetime
from pathlib import Path
@@ -9,6 +10,7 @@ from typer.testing import CliRunner
from tht.cli import app
from tht.memory import MemoryRecord, save_registry
from tht.ports.vector import VectorStoreError
class _FakeEmbedder:
@@ -33,6 +35,10 @@ class _FakeVectorStore:
return 3
def _sha_file(path: Path) -> str:
return "sha256:" + hashlib.sha256(path.read_bytes()).hexdigest()
def _qdrant_runtime_config(tmp_path: Path) -> Path:
cfg = tmp_path / "workspace.yaml"
cfg.write_text(
@@ -76,9 +82,7 @@ tables:
type: bigint
"""
)
(tmp_path / "artifacts" / "mschema" / "annotations.yaml").write_text(
"tables: {}\n"
)
(tmp_path / "artifacts" / "mschema" / "annotations.yaml").write_text("tables: {}\n")
def _memory_record() -> MemoryRecord:
@@ -111,6 +115,73 @@ def test_vector_index_schema_accepts_qdrant_only_runtime_config(tmp_path, monkey
assert store.upserts
def test_vector_index_schema_json_is_pristine_and_reports_artifacts(tmp_path, monkeypatch):
cfg = _qdrant_runtime_config(tmp_path)
_write_schema_artifacts(tmp_path)
store = _FakeVectorStore()
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store)
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder())
response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)])
assert response.exit_code == 0, response.output
assert response.stderr == ""
assert json.loads(response.stdout) == {
"artifactIdentities": [
{
"digest": _sha_file(tmp_path / "artifacts" / "mschema" / "annotations.yaml"),
"kind": "schema_annotations",
},
{
"digest": _sha_file(tmp_path / "artifacts" / "mschema" / "physical.yaml"),
"kind": "physical_schema",
},
],
"code": "ok",
"collection": "psd-clinical",
"counts": {
"added": 2,
"columns": 1,
"deleted": 0,
"records": 2,
"tables": 1,
"unchanged": 0,
"updated": 0,
},
"operation": "index_schema",
"schemaVersion": 1,
"status": "succeeded",
"workspaceId": "psd-clinical",
"workspaceRevision": "a" * 40,
}
def test_vector_index_schema_json_failure_is_pristine(tmp_path, monkeypatch):
cfg = _qdrant_runtime_config(tmp_path)
_write_schema_artifacts(tmp_path)
def boom(cfg, require_write):
raise VectorStoreError("semantic_index_incompatible")
monkeypatch.setattr("tht.adapters.factory.build_vector_store", boom)
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder())
response = CliRunner().invoke(app, ["vector", "index-schema", "--json", "-c", str(cfg)])
assert response.exit_code == 1
assert response.stderr == ""
assert json.loads(response.stdout) == {
"code": "semantic_index_incompatible",
"error": "semantic index incompatible",
"operation": "index_schema",
"schemaVersion": 1,
"status": "failed",
"workspaceId": "psd-clinical",
"workspaceRevision": "a" * 40,
}
def test_memory_promote_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch):
cfg = _qdrant_runtime_config(tmp_path)
store = _FakeVectorStore()
+89 -1
View File
@@ -39,6 +39,7 @@ class FakeQdrantHttp:
self.malformed_query = False
self.malformed_scroll = False
self.scroll_pages: list[dict] | None = None
self.drop_collection_on_points = False
def request(self, method, url, *, json=None, timeout=None):
self.calls.append((method, url, json))
@@ -75,6 +76,9 @@ class FakeQdrantHttp:
return FakeResponse(200, {"status": "ok"})
if method == "PUT" and path == "/collections/workspace-semantic/points":
if self.drop_collection_on_points:
self.collection = None
return FakeResponse(404, {"status": "error"})
for point in json["points"]:
self.points[point["id"]] = point
return FakeResponse(200, {"result": {"status": "acknowledged"}})
@@ -161,13 +165,16 @@ def _write_record(record_id: str, kind: str, *, metadata=None):
)
def _store(fake: FakeQdrantHttp) -> QdrantVectorStore:
def _store(
fake: FakeQdrantHttp, *, collection_lifecycle: str = "self_heal"
) -> QdrantVectorStore:
return QdrantVectorStore(
base_url="http://qdrant:6333",
collection="workspace-semantic",
workspace_id="demo",
workspace_revision="a" * 40,
expected_dimension=1024,
collection_lifecycle=collection_lifecycle,
request=fake.request,
)
@@ -212,6 +219,87 @@ def test_upsert_refuses_collection_dimension_or_distance_mismatch_without_recrea
assert creates == []
def test_upsert_require_existing_refuses_missing_collection_without_creating():
fake = FakeQdrantHttp()
store = _store(fake, collection_lifecycle="require_existing")
with pytest.raises(VectorStoreError, match="semantic_index_incompatible"):
store.upsert("memory", [_write_record("memory:1", "memory")])
assert fake.collection is None
creates = [
call
for call in fake.calls
if call[0] == "PUT" and call[1].endswith("/collections/workspace-semantic")
]
assert creates == []
def test_upsert_require_existing_refuses_incompatible_collection_without_mutating():
fake = FakeQdrantHttp(dimension=384, distance="Dot")
fake.collection = {"vectors": {"size": 384, "distance": "Dot"}}
store = _store(fake, collection_lifecycle="require_existing")
with pytest.raises(VectorStoreError, match="semantic_index_incompatible"):
store.upsert("memory", [_write_record("memory:1", "memory")])
assert fake.payload_indexes == set()
mutating = [call for call in fake.calls if call[0] == "PUT"]
assert mutating == []
def test_upsert_require_existing_writes_to_existing_compatible_collection():
fake = FakeQdrantHttp()
fake.collection = {"vectors": {"size": 1024, "distance": "Cosine"}}
fake.payload_indexes = {
"content_hash",
"document_id",
"kind",
"record_key",
"record_kind",
"vector_generation",
"workspace_id",
"workspace_revision",
}
store = _store(fake, collection_lifecycle="require_existing")
assert store.upsert("memory", [_write_record("memory:1", "memory")]) == 1
create_or_index = [
call
for call in fake.calls
if call[0] == "PUT" and not call[1].endswith("/points?wait=true")
]
assert create_or_index == []
def test_upsert_require_existing_fails_if_collection_disappears_after_preflight():
fake = FakeQdrantHttp()
fake.collection = {"vectors": {"size": 1024, "distance": "Cosine"}}
fake.payload_indexes = {
"content_hash",
"document_id",
"kind",
"record_key",
"record_kind",
"vector_generation",
"workspace_id",
"workspace_revision",
}
fake.drop_collection_on_points = True
store = _store(fake, collection_lifecycle="require_existing")
with pytest.raises(VectorStoreError, match="HTTP 404"):
store.upsert("memory", [_write_record("memory:1", "memory")])
creates = [
call
for call in fake.calls
if call[0] == "PUT" and call[1].endswith("/collections/workspace-semantic")
]
assert creates == []
def test_health_fails_when_the_bound_collection_is_missing():
fake = FakeQdrantHttp()
@@ -113,6 +113,61 @@ def test_signed_http_file_resolves_in_memory_and_preserves_provenance_order(tmp_
assert_no_canaries(repr(adapter))
def test_qdrant_runtime_config_keeps_require_existing_and_http_policy_fields(tmp_path):
path = tmp_path / "runtime.yaml"
path.write_text(
yaml.safe_dump(
{
"runtime_identity": {
"workspace_id": "psd-clinical",
"workspace_revision": "a" * 40,
},
"dwh": {
"type": "postgres_direct",
"connection": {
"database": "analytics",
"schema": "public",
"user": "reader",
"password": "secret",
},
},
"vectors": {
"type": "qdrant",
"base_url": "http://qdrant:6333",
"collection": "psd-clinical",
"collection_lifecycle": "require_existing",
},
"embeddings": {
"provider": "ollama_internal",
"base_url": "http://embedding:11434",
"model": "qwen3-embedding:0.6b",
"dim": 1024,
},
"evidence": {
"sources": [
{
"type": "http",
"urls": ["https://evidence.example.test/guide.md"],
"allow_private_hosts": True,
"max_redirects": 2,
"max_cache_bytes": 1234,
}
]
},
}
)
)
cfg = load_config(path)
rendered = cfg.model_dump(mode="json")
assert cfg.vectors.collection_lifecycle == "require_existing"
assert rendered["vectors"]["collection_lifecycle"] == "require_existing"
assert rendered["evidence"]["sources"][0]["allow_private_hosts"] is True
assert rendered["evidence"]["sources"][0]["max_redirects"] == 2
assert rendered["evidence"]["sources"][0]["max_cache_bytes"] == 1234
def test_signed_http_file_requires_explicit_provenance_urls(tmp_path):
secret_file = tmp_path / "signed-urls.json"
secret_file.write_text(json.dumps([
+292 -45
View File
@@ -1,4 +1,6 @@
from datetime import datetime
import hashlib
import json
from datetime import UTC, datetime
import yaml
from typer.testing import CliRunner
@@ -15,10 +17,19 @@ from tht.mschema.models import (
)
from tht.mschema.render import to_mschema_text, to_schema_dict
RUNNER = CliRunner()
def _json_sha(value) -> str:
payload = json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
return "sha256:" + hashlib.sha256(payload.encode("utf-8")).hexdigest()
def _physical():
return PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1),
database="d",
schema="s",
introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={
"dim_patient": TablePhysical(
columns={"cod_paz": ColumnPhysical(type="bigint", pk=True)},
@@ -42,10 +53,16 @@ def _annotations_with_fks():
tables={
"fact_ablazione": TableAnnotation(
foreign_keys=[
ForeignKey(columns=["cod_paz"], ref_table="dim_patient",
ref_columns=["cod_paz"]),
ForeignKey(columns=["data_time_key"], ref_table="dim_time",
ref_columns=["day_key"]),
ForeignKey(
columns=["cod_paz"],
ref_table="dim_patient",
ref_columns=["cod_paz"],
),
ForeignKey(
columns=["data_time_key"],
ref_table="dim_time",
ref_columns=["day_key"],
),
],
)
}
@@ -61,8 +78,11 @@ def test_mschema_text_renders_annotation_fks():
def test_schema_dict_merges_annotation_fks():
d = to_schema_dict(_physical(), _annotations_with_fks())
fks = d["fact_ablazione"]["foreign_keys"]
assert {"columns": ["cod_paz"], "ref_table": "dim_patient",
"ref_columns": ["cod_paz"]} in fks
assert {
"columns": ["cod_paz"],
"ref_table": "dim_patient",
"ref_columns": ["cod_paz"],
} in fks
def test_find_orphans_flags_broken_annotation_fk():
@@ -70,10 +90,16 @@ def test_find_orphans_flags_broken_annotation_fk():
tables={
"fact_ablazione": TableAnnotation(
foreign_keys=[
ForeignKey(columns=["cod_paz"], ref_table="dim_sparita",
ref_columns=["x"]),
ForeignKey(columns=["colonna_sparita"], ref_table="dim_time",
ref_columns=["day_key"]),
ForeignKey(
columns=["cod_paz"],
ref_table="dim_sparita",
ref_columns=["x"],
),
ForeignKey(
columns=["colonna_sparita"],
ref_table="dim_time",
ref_columns=["day_key"],
),
],
)
}
@@ -91,22 +117,139 @@ def _write_workspace(tmp_path):
_physical().to_yaml(tmp_path / "artifacts" / "mschema" / "physical.yaml")
cfg = tmp_path / "workspace.yaml"
cfg.write_text(
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, sessions: {tmp_path/'s'}}}\n"
f"""
runtime_identity:
workspace_id: demo
workspace_revision: {'a' * 40}
database: {{database: d, schema: s, user: u, password: p, transport: direct}}
paths: {{artifacts: {tmp_path / 'artifacts'}, indexes: {tmp_path / 'i'}, sessions: {tmp_path / 's'}}}
"""
)
return cfg
def test_suggest_fks_prints_candidates(tmp_path):
cfg = _write_workspace(tmp_path)
res = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(cfg)])
res = RUNNER.invoke(app, ["schema", "suggest-fks", "-c", str(cfg)])
assert res.exit_code == 0, res.output
data = yaml.safe_load(res.output.rsplit("\n", 2)[0].split("FK candidate")[0])
fks = data["tables"]["fact_ablazione"]["foreign_keys"]
assert {"columns": ["cod_paz"], "ref_table": "dim_patient",
"ref_columns": ["cod_paz"]} in fks
assert {"columns": ["data_time_key"], "ref_table": "dim_time",
"ref_columns": ["day_key"]} in fks
assert {
"columns": ["cod_paz"],
"ref_table": "dim_patient",
"ref_columns": ["cod_paz"],
} in fks
assert {
"columns": ["data_time_key"],
"ref_table": "dim_time",
"ref_columns": ["day_key"],
} in fks
def test_suggest_fks_json_is_pristine_and_stable(tmp_path):
cfg = _write_workspace(tmp_path)
first = tmp_path / "second.sql"
first.write_text(
"SELECT f.esito FROM datawarehouse.fact_ablazione f "
"JOIN datawarehouse.dim_patient p ON f.cod_paz = p.cod_paz"
)
second = tmp_path / "first.sql"
second.write_text(
"SELECT dt.year FROM datawarehouse.fact_ablazione f "
"JOIN datawarehouse.dim_time dt ON f.data_time_key = dt.day_key"
)
response = RUNNER.invoke(
app,
[
"schema",
"suggest-fks",
"-c",
str(cfg),
"--from-sql",
str(first),
"--from-sql",
str(second),
"--json",
],
)
assert response.exit_code == 0, response.output
assert response.stderr == ""
payload = json.loads(response.stdout)
assert payload["schemaVersion"] == 1
assert payload["status"] == "succeeded"
assert payload["code"] == "ok"
assert payload["operation"] == "schema_suggest_fks"
assert payload["workspaceId"] == "demo"
assert payload["workspaceRevision"] == "a" * 40
assert payload["counts"] == {
"ambiguousColumns": 0,
"candidateTables": 1,
"candidates": 2,
"minedJoins": 2,
"sqlFiles": 2,
}
assert payload["candidateDocument"] == {
"annotations": {
"tables": {
"fact_ablazione": {
"foreign_keys": [
{
"columns": ["cod_paz"],
"ref_columns": ["cod_paz"],
"ref_table": "dim_patient",
},
{
"columns": ["data_time_key"],
"ref_columns": ["day_key"],
"ref_table": "dim_time",
},
]
}
}
},
"counts": {"candidateTables": 1, "candidates": 2},
"schemaVersion": 1,
}
assert payload["candidate_count"] == payload["counts"]["candidates"]
candidate_yaml = payload["candidate_yaml"]
assert payload["candidateDigest"] == "sha256:" + hashlib.sha256(candidate_yaml.encode("utf-8")).hexdigest()
assert yaml.safe_load(candidate_yaml) == payload["candidateDocument"]["annotations"]
rerun = RUNNER.invoke(
app,
[
"schema",
"suggest-fks",
"-c",
str(cfg),
"--from-sql",
str(second),
"--from-sql",
str(first),
"--json",
],
)
assert rerun.exit_code == 0, rerun.output
assert json.loads(rerun.stdout) == payload
def test_suggest_fks_json_rejects_invalid_assume_without_prose(tmp_path):
cfg = _write_workspace(tmp_path)
response = RUNNER.invoke(
app,
["schema", "suggest-fks", "-c", str(cfg), "--assume", "cod_x=nope", "--json"],
)
assert response.exit_code == 1
assert response.stderr == ""
payload = json.loads(response.stdout)
assert payload["schemaVersion"] == 1
assert payload["status"] == "failed"
assert payload["code"] == "invalid_argument"
assert payload["error"] == "invalid assume mapping"
def test_mine_join_pairs_from_approved_sql():
@@ -122,23 +265,22 @@ def test_mine_join_pairs_from_approved_sql():
"""
pairs = mine_join_pairs(sql, _physical())
assert pairs[("fact_ablazione", "data_time_key", "dim_time", "day_key")] == 1
# il join CTE-CTE (abl.year=b.year) non produce coppie
assert len(pairs) == 1
def test_mine_join_pairs_ignores_non_pk_pairs_and_bad_sql():
from tht.mschema.fkmine import mine_join_pairs
# esito=esito: nessun lato e' PK -> scartato
sql = ("SELECT * FROM fact_ablazione a JOIN fact_ablazione b "
"ON a.esito = b.esito")
sql = "SELECT * FROM fact_ablazione a JOIN fact_ablazione b ON a.esito = b.esito"
assert len(mine_join_pairs(sql, _physical())) == 0
assert len(mine_join_pairs("WITH broken (", _physical())) == 0
def test_suggest_fks_skips_generic_and_ambiguous_pks(tmp_path):
phys = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1),
database="d",
schema="s",
introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={
"dim_a": TablePhysical(columns={"id": ColumnPhysical(type="int", pk=True)}),
"dim_b": TablePhysical(columns={"id": ColumnPhysical(type="int", pk=True)}),
@@ -158,14 +300,14 @@ def test_suggest_fks_skips_generic_and_ambiguous_pks(tmp_path):
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, sessions: {tmp_path/'s'}}}\n"
)
res = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(cfg)])
res = RUNNER.invoke(app, ["schema", "suggest-fks", "-c", str(cfg)])
assert res.exit_code == 0, res.output
assert "nessuna FK da suggerire" in res.output # id generico, cod_x ambigua
assert "cod_x" in res.output # segnalata come ambigua saltata
assert "nessuna FK da suggerire" in res.output
assert "cod_x" in res.output
# --assume disambigua la PK multi-proprietario
res2 = CliRunner().invoke(
app, ["schema", "suggest-fks", "-c", str(cfg), "--assume", "cod_x=dim_c1"]
res2 = RUNNER.invoke(
app,
["schema", "suggest-fks", "-c", str(cfg), "--assume", "cod_x=dim_c1"],
)
assert res2.exit_code == 0, res2.output
yaml_text = "\n".join(
@@ -173,16 +315,19 @@ def test_suggest_fks_skips_generic_and_ambiguous_pks(tmp_path):
)
data = yaml.safe_load(yaml_text)
fact_fks = data["tables"]["fact_f"]["foreign_keys"]
assert {"columns": ["cod_x"], "ref_table": "dim_c1",
"ref_columns": ["cod_x"]} in fact_fks
# dim_c2.cod_x -> dim_c1 (estensione 1:1), ma NON dim_c1 -> se stessa
assert {
"columns": ["cod_x"],
"ref_table": "dim_c1",
"ref_columns": ["cod_x"],
} in fact_fks
assert "dim_c1" not in data["tables"] or all(
fk["ref_table"] != "dim_c1" for fk in data["tables"].get("dim_c1", {}).get("foreign_keys", [])
fk["ref_table"] != "dim_c1"
for fk in data["tables"].get("dim_c1", {}).get("foreign_keys", [])
)
# --assume con tabella inesistente -> errore chiaro
res3 = CliRunner().invoke(
app, ["schema", "suggest-fks", "-c", str(cfg), "--assume", "cod_x=nope"]
res3 = RUNNER.invoke(
app,
["schema", "suggest-fks", "-c", str(cfg), "--assume", "cod_x=nope"],
)
assert res3.exit_code == 1
assert "non valido" in res3.output
@@ -190,20 +335,122 @@ def test_suggest_fks_skips_generic_and_ambiguous_pks(tmp_path):
def test_suggest_fks_from_sql_mines_joins(tmp_path):
cfg = _write_workspace(tmp_path)
sqldir = tmp_path / "approved"
sqldir.mkdir()
(sqldir / "q1.sql").write_text(
sql_file = tmp_path / "approved.sql"
sql_file.write_text(
"SELECT f.esito FROM datawarehouse.fact_ablazione f "
"JOIN datawarehouse.dim_patient p ON f.cod_paz = p.cod_paz"
)
res = CliRunner().invoke(
app, ["schema", "suggest-fks", "-c", str(cfg), "--from-sql", str(sqldir)]
res = RUNNER.invoke(
app,
["schema", "suggest-fks", "-c", str(cfg), "--from-sql", str(sql_file)],
)
assert res.exit_code == 0, res.output
assert "Minati 1 equi-join da 1 file SQL" in res.output
assert "ref_table: dim_patient" in res.output
def test_schema_check_json_validates_staged_annotations_without_mutating_runtime(tmp_path):
cfg = _write_workspace(tmp_path)
runtime_annotations = tmp_path / "artifacts" / "mschema" / "annotations.yaml"
runtime_annotations.write_text("tables: {}\n")
reviewed = tmp_path / "reviewed.yaml"
_annotations_with_fks().to_yaml(reviewed)
response = RUNNER.invoke(
app,
[
"schema",
"check",
"-c",
str(cfg),
"--annotations",
str(reviewed),
"--reviewed-candidates",
"sha256:" + "b" * 64,
"--json",
],
)
assert response.exit_code == 0, response.output
assert response.stderr == ""
payload = json.loads(response.stdout)
assert payload["orphan_count"] == 0
assert payload["reviewed_candidates_digest"] == "sha256:" + "b" * 64
assert payload["annotations_digest"] == "sha256:" + hashlib.sha256(reviewed.read_bytes()).hexdigest()
assert payload == {
"annotationsDigest": _json_sha(
{
"annotations": {
"tables": {
"fact_ablazione": {
"foreign_keys": [
{
"columns": ["cod_paz"],
"ref_columns": ["cod_paz"],
"ref_table": "dim_patient",
},
{
"columns": ["data_time_key"],
"ref_columns": ["day_key"],
"ref_table": "dim_time",
},
]
}
}
},
"schemaVersion": 1,
}
),
"annotations_digest": "sha256:" + hashlib.sha256(reviewed.read_bytes()).hexdigest(),
"code": "ok",
"orphan_count": 0,
"reviewed_candidates_digest": "sha256:" + "b" * 64,
"counts": {"annotationTables": 1, "foreignKeys": 2, "orphans": 0},
"operation": "schema_check",
"orphans": [],
"reviewedCandidates": "sha256:" + "b" * 64,
"schemaVersion": 1,
"status": "succeeded",
"workspaceId": "demo",
"workspaceRevision": "a" * 40,
"zeroOrphans": True,
}
assert runtime_annotations.read_text() == "tables: {}\n"
def test_schema_check_json_reports_orphans_without_prose(tmp_path):
cfg = _write_workspace(tmp_path)
reviewed = tmp_path / "reviewed.yaml"
Annotations(
tables={
"fact_ablazione": TableAnnotation(
foreign_keys=[
ForeignKey(
columns=["cod_paz"],
ref_table="dim_missing",
ref_columns=["cod_paz"],
)
]
)
}
).to_yaml(reviewed)
response = RUNNER.invoke(
app,
["schema", "check", "-c", str(cfg), "--annotations", str(reviewed), "--json"],
)
assert response.exit_code == 3
assert response.stderr == ""
payload = json.loads(response.stdout)
assert payload["schemaVersion"] == 1
assert payload["status"] == "blocked"
assert payload["code"] == "annotation_invalid"
assert payload["zeroOrphans"] is False
assert payload["counts"]["orphans"] == 1
assert payload["orphans"] == ["fact_ablazione.fk(cod_paz)->dim_missing"]
def test_suggest_fks_write_merges_and_is_idempotent(tmp_path):
cfg = _write_workspace(tmp_path)
ann_path = tmp_path / "artifacts" / "mschema" / "annotations.yaml"
@@ -211,13 +458,13 @@ def test_suggest_fks_write_merges_and_is_idempotent(tmp_path):
tables={"fact_ablazione": TableAnnotation(description="Ablazioni")}
).to_yaml(ann_path)
res = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(cfg), "--write"])
res = RUNNER.invoke(app, ["schema", "suggest-fks", "-c", str(cfg), "--write"])
assert res.exit_code == 0, res.output
ann = Annotations.from_yaml(ann_path)
assert ann.tables["fact_ablazione"].description == "Ablazioni" # non distrutta
assert ann.tables["fact_ablazione"].description == "Ablazioni"
assert len(ann.tables["fact_ablazione"].foreign_keys) == 2
res2 = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(cfg), "--write"])
res2 = RUNNER.invoke(app, ["schema", "suggest-fks", "-c", str(cfg), "--write"])
assert "nessuna FK da suggerire" in res2.output
ann2 = Annotations.from_yaml(ann_path)
assert len(ann2.tables["fact_ablazione"].foreign_keys) == 2