fix: complete P2 Task 2 schema machine contracts

This commit is contained in:
2026-08-11 05:38:25 +02:00
parent b235132d89
commit 554de4ff2f
4 changed files with 356 additions and 30 deletions
+148
View File
@@ -281,3 +281,151 @@ def test_schema_check_json_reports_orphan_count_without_prose(tmp_path):
"status": "failed", "code": "annotation_orphans", "orphan_count": 1,
"orphans": ["gone"],
}
def test_suggest_fks_json_rejects_more_than_32_staged_sql_files(tmp_path):
import json
cfg = _write_workspace(tmp_path)
staged = tmp_path / "many"
staged.mkdir()
for i in range(33):
(staged / f"q{i:02d}.sql").write_text("SELECT 1")
response = CliRunner().invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(staged)]
)
assert response.exit_code == 1
assert json.loads(response.stdout) == {"status": "failed", "code": "staged_sql_too_many"}
def test_staged_sql_size_limits_are_inclusive(tmp_path):
import json
cfg = _write_workspace(tmp_path)
staged = tmp_path / "boundary"
staged.mkdir()
# Exactly one MiB per file and exactly 16 MiB in aggregate are accepted.
body = "-- padding\n" + ("x" * (1 << 20))
body = body[: 1 << 20]
for i in range(16):
(staged / f"q{i:02d}.sql").write_bytes(body.encode())
response = CliRunner().invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(staged)]
)
assert response.exit_code == 0, response.output
assert json.loads(response.stdout)["staged_sql_count"] == 16
def test_schema_check_json_success_and_reviewed_annotations_are_preserved(tmp_path):
import json
cfg = _write_workspace(tmp_path)
annotations_path = tmp_path / "artifacts" / "mschema" / "annotations.yaml"
reviewed = _annotations_with_fks()
reviewed.to_yaml(annotations_path)
before = annotations_path.read_text()
check = CliRunner().invoke(app, ["schema", "check", "--json", "-c", str(cfg)])
assert check.exit_code == 0, check.output
assert json.loads(check.stdout) == {
"status": "succeeded", "code": "ok", "orphan_count": 0, "orphans": []
}
suggest = CliRunner().invoke(app, ["schema", "suggest-fks", "--json", "-c", str(cfg)])
assert suggest.exit_code == 0, suggest.output
assert json.loads(suggest.stdout)["candidate_count"] == 0
assert annotations_path.read_text() == before
def test_schema_human_check_and_suggest_report_missing_physical_schema(tmp_path):
cfg = _write_workspace(tmp_path)
physical = tmp_path / "artifacts" / "mschema" / "physical.yaml"
physical.unlink()
expected = "physical.yaml non trovato. Esegui prima `tht schema introspect`."
check = CliRunner().invoke(app, ["schema", "check", "-c", str(cfg)])
assert check.exit_code == 1
assert expected in check.output
suggest = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(cfg)])
assert suggest.exit_code == 1
assert expected in suggest.output
def test_schema_human_check_reports_ignored_columns(tmp_path):
cfg = _write_workspace(tmp_path)
physical = _physical()
physical.tables["fact_ablazione"].columns["esito"].eligible = False
physical.tables["fact_ablazione"].columns["esito"].eligibility_reason = "test ignored"
physical.to_yaml(tmp_path / "artifacts" / "mschema" / "physical.yaml")
response = CliRunner().invoke(app, ["schema", "check", "-c", str(cfg)])
assert response.exit_code == 0, response.output
assert "Colonne ignorate (testo ampio, 1):" in response.output
assert "fact_ablazione.esito (test ignored)" in response.output
def test_staged_sql_file_size_over_one_mib_is_rejected(tmp_path):
import json
cfg = _write_workspace(tmp_path)
staged = tmp_path / "oversize"
staged.mkdir()
(staged / "too-large.sql").write_bytes(b"x" * ((1 << 20) + 1))
response = CliRunner().invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(staged)]
)
assert response.exit_code == 1
assert json.loads(response.stdout) == {"status": "failed", "code": "staged_sql_too_large"}
def test_staged_sql_aggregate_over_16_mib_is_rejected(tmp_path):
import json
cfg = _write_workspace(tmp_path)
staged = tmp_path / "aggregate"
staged.mkdir()
for i in range(16):
(staged / f"q{i:02d}.sql").write_bytes(b"x" * (1 << 20))
(staged / "over.sql").write_bytes(b"x")
response = CliRunner().invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(staged)]
)
assert response.exit_code == 1
assert json.loads(response.stdout) == {"status": "failed", "code": "staged_sql_too_large"}
def test_suggest_fks_candidate_order_is_stable_for_reordered_staged_inputs(tmp_path):
import json
cfg = _write_workspace(tmp_path)
first = tmp_path / "first"
second = tmp_path / "second"
first.mkdir()
second.mkdir()
(first / "join.sql").write_text(
"SELECT * FROM fact_ablazione f JOIN dim_patient p ON f.cod_paz = p.cod_paz"
)
(second / "join.sql").write_text(
"SELECT * FROM fact_ablazione f JOIN dim_time p ON f.data_time_key = p.day_key"
)
runner = CliRunner()
one = runner.invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(first), "--from-sql", str(second)]
)
two = runner.invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg), "--from-sql", str(second), "--from-sql", str(first)]
)
assert one.exit_code == two.exit_code == 0
assert json.loads(one.stdout) == json.loads(two.stdout)
def test_schema_human_check_missing_config_has_original_error(tmp_path):
response = CliRunner().invoke(app, ["schema", "check", "-c", str(tmp_path / "missing.yaml")])
assert response.exit_code == 1
assert response.output
assert "ERRORE:" in response.output
def test_schema_human_suggest_missing_config_has_original_error(tmp_path):
response = CliRunner().invoke(app, ["schema", "suggest-fks", "-c", str(tmp_path / "missing.yaml")])
assert response.exit_code == 1
assert response.output
assert "ERRORE:" in response.output