fix: harden Task 2 schema quality boundaries

This commit is contained in:
2026-08-11 06:09:07 +02:00
parent d5f752dd7b
commit 9e2c2b37b3
3 changed files with 183 additions and 64 deletions
@@ -1,10 +1,13 @@
# ruff: noqa: DTZ001
import warnings
from datetime import datetime
import pytest
import yaml
from typer.testing import CliRunner
from tht.cli import app
from tht.cli.schema_cmd import check_schema_data
from tht.mschema.merge import find_orphans
from tht.mschema.models import (
Annotations,
@@ -163,6 +166,10 @@ def test_suggest_fks_skips_generic_and_ambiguous_pks(tmp_path):
assert res.exit_code == 0, res.output
assert "nessuna FK da suggerire" in res.output # id generico, cod_x ambigua
assert "cod_x" in res.output # segnalata come ambigua saltata
exact = CliRunner().invoke(
app, ["schema", "suggest-fks", "--json", "-c", str(cfg)]
)
assert yaml.safe_load(exact.stdout)["ambiguous_columns"] == ["cod_x"]
# --assume disambigua la PK multi-proprietario
res2 = CliRunner().invoke(
@@ -205,6 +212,18 @@ def test_suggest_fks_from_sql_mines_joins(tmp_path):
assert "ref_table: dim_patient" in res.output
def test_suggest_fks_reports_staged_file_without_mined_joins(tmp_path):
cfg = _write_workspace(tmp_path)
staged = tmp_path / "approved"
staged.mkdir()
(staged / "no-joins.sql").write_text("SELECT 1")
response = CliRunner().invoke(
app, ["schema", "suggest-fks", "-c", str(cfg), "--from-sql", str(staged)]
)
assert response.exit_code == 0, response.output
assert "Minati 0 equi-join da 1 file SQL" in response.output
def test_suggest_fks_write_merges_and_is_idempotent(tmp_path):
cfg = _write_workspace(tmp_path)
ann_path = tmp_path / "artifacts" / "mschema" / "annotations.yaml"
@@ -435,3 +454,59 @@ def test_schema_human_suggest_missing_config_has_original_error(tmp_path):
assert response.exit_code == 1
assert response.output
assert "ERRORE:" in response.output
def test_human_config_keeps_legacy_warning_but_json_suppresses_it(tmp_path):
cfg = _write_workspace(tmp_path)
with pytest.warns(FutureWarning, match="legacy workspace resource keys"):
check_schema_data(cfg)
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
check_schema_data(cfg, suppress_legacy_warning=True)
assert not [item for item in caught if issubclass(item.category, FutureWarning)]
def test_staged_sql_enumeration_stops_at_count_sentinel(tmp_path, monkeypatch):
from tht.cli.schema_cmd import _MachineSchemaError, _staged_sql_files
staged = tmp_path / "staged"
staged.mkdir()
files = [staged / f"q{i:03d}.sql" for i in range(100)]
for path in files:
path.write_text("SELECT 1")
yielded = 0
def bounded_rglob(_self, _pattern):
nonlocal yielded
for path in files:
yielded += 1
yield path
monkeypatch.setattr(type(staged), "rglob", bounded_rglob)
with pytest.raises(_MachineSchemaError, match="staged_sql_too_many"):
_staged_sql_files([staged])
assert yielded == 33
def test_staged_sql_oversize_read_is_bounded(tmp_path, monkeypatch):
from tht.cli.schema_cmd import _MachineSchemaError, _read_staged_sql
staged = tmp_path / "oversize.sql"
staged.write_bytes(b"unused")
reads = []
class BoundedReader:
def __enter__(self):
return self
def __exit__(self, *_args):
return False
def read(self, size):
reads.append(size)
return b"x" * size
monkeypatch.setattr(type(staged), "open", lambda *_args, **_kwargs: BoundedReader())
with pytest.raises(_MachineSchemaError, match="staged_sql_too_large"):
_read_staged_sql([staged])
assert reads == [(1 << 20) + 1]