Chiusura fase di ristrutturazione e modularizzazione del workflow per favorire sviluppo modulare

This commit is contained in:
2026-08-24 13:32:20 +02:00
parent fa2298653b
commit 6062cb010e
96 changed files with 304 additions and 259 deletions
+3
View File
@@ -42,6 +42,9 @@ line-length = 100
[tool.ruff.lint.per-file-ignores] [tool.ruff.lint.per-file-ignores]
"tht/cli/__init__.py" = ["E402"] "tht/cli/__init__.py" = ["E402"]
[tool.ruff.lint.flake8-bugbear]
extend-immutable-calls = ["typer.Argument", "typer.Option"]
[tool.pytest.ini_options] [tool.pytest.ini_options]
testpaths = ["tests"] testpaths = ["tests"]
markers = [ markers = [
+2 -2
View File
@@ -6,6 +6,7 @@ is not assumed reliable' gains real teeth for the data layer.
""" """
import pytest import pytest
from sqlalchemy import create_engine, text from sqlalchemy import create_engine, text
from sqlalchemy.exc import SQLAlchemyError
from tht.db.connection import can_create_in_schema, ping, writable_tables from tht.db.connection import can_create_in_schema, ping, writable_tables
@@ -43,8 +44,7 @@ def test_read_only_role_cannot_insert(ro_url):
by our engine).""" by our engine)."""
engine = create_engine(ro_url) engine = create_engine(ro_url)
try: try:
with pytest.raises(Exception): with pytest.raises(SQLAlchemyError), engine.begin() as conn:
with engine.begin() as conn:
conn.execute(text('INSERT INTO dw.dim_pazienti VALUES (999, %s, %s)'), conn.execute(text('INSERT INTO dw.dim_pazienti VALUES (999, %s, %s)'),
("test", "test")) ("test", "test"))
finally: finally:
+1 -1
View File
@@ -40,7 +40,7 @@ def test_unique_values_for_lsh_returns_most_frequent(admin_engine):
schema = introspect(admin_engine, "testdb", "dw") schema = introspect(admin_engine, "testdb", "dw")
# Before classify_all, all text columns are eligible=True by default. Sampling # Before classify_all, all text columns are eligible=True by default. Sampling
# only touches text types regardless. # only touches text types regardless.
values, skipped, truncated = unique_values_for_lsh( values, _skipped, _truncated = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=100) admin_engine, schema, LshConfig(max_values_per_column=100)
) )
# dim_pazienti.citta: Milano, Bergamo, Brescia (3 distinct, all eligible text) # dim_pazienti.citta: Milano, Bergamo, Brescia (3 distinct, all eligible text)
@@ -38,7 +38,7 @@ def test_ablazione_returns_multiple_columns(l2_env):
schema_name = ws.database.db_schema schema_name = ws.database.db_schema
try: try:
lsh, minhashes, meta = load_index(index_dir, schema_name) lsh, minhashes, meta = load_index(index_dir, schema_name)
except Exception as e: except Exception as e: # noqa: BLE001 - any unusable external index skips this L2 probe
pytest.skip(f"LSH index not built yet (run tht preprocess dwh --steps lsh -c {WORKSPACE}): {e}") pytest.skip(f"LSH index not built yet (run tht preprocess dwh --steps lsh -c {WORKSPACE}): {e}")
hits = query_index(lsh, minhashes, "ablazione", meta, top_n=20) hits = query_index(lsh, minhashes, "ablazione", meta, top_n=20)
+1 -1
View File
@@ -1,7 +1,7 @@
import json import json
import uuid
from datetime import UTC, datetime from datetime import UTC, datetime
from types import SimpleNamespace from types import SimpleNamespace
import uuid
from typer.testing import CliRunner from typer.testing import CliRunner
@@ -1,7 +1,7 @@
import uuid
from datetime import UTC, datetime from datetime import UTC, datetime
from pathlib import Path from pathlib import Path
from types import SimpleNamespace from types import SimpleNamespace
import uuid
from tht.cli import session_cmd from tht.cli import session_cmd
from tht.decisions import DecisionInput from tht.decisions import DecisionInput
+4 -2
View File
@@ -1,6 +1,7 @@
from tht.decisions import DecisionType
import typing import typing
from tht.decisions import DecisionType
def test_column_decision_types_exist(): def test_column_decision_types_exist():
allowed = set(typing.get_args(DecisionType)) allowed = set(typing.get_args(DecisionType))
@@ -9,8 +10,9 @@ def test_column_decision_types_exist():
def test_f4_emits_column_types(): def test_f4_emits_column_types():
import yaml
from pathlib import Path from pathlib import Path
import yaml
wf = yaml.safe_load(Path("workflow.yaml").read_text()) wf = yaml.safe_load(Path("workflow.yaml").read_text())
f4 = next(p for p in wf["phases"] if p["id"] == "F4") f4 = next(p for p in wf["phases"] if p["id"] == "F4")
assert "column_promoted" in f4["emits"] assert "column_promoted" in f4["emits"]
+2 -2
View File
@@ -1,7 +1,5 @@
import pytest import pytest
from tht.evidence.adapters import FilesystemEvidenceSource, HttpManifestEvidenceSource
from tht.evidence import build_sources
from tht.config import ( from tht.config import (
ConfigError, ConfigError,
PgvectorDirectConfig, PgvectorDirectConfig,
@@ -12,6 +10,8 @@ from tht.config import (
load_config, load_config,
workspace_id_for_config, workspace_id_for_config,
) )
from tht.evidence import build_sources
from tht.evidence.adapters import FilesystemEvidenceSource, HttpManifestEvidenceSource
def test_direct_vector_passwords_load_from_file_references(monkeypatch, tmp_path): def test_direct_vector_passwords_load_from_file_references(monkeypatch, tmp_path):
+4 -1
View File
@@ -209,7 +209,10 @@ def test_model_copy_revalidates_records_and_manifests():
def test_manifest_datetimes_are_aware_and_normalized_to_utc(): def test_manifest_datetimes_are_aware_and_normalized_to_utc():
with pytest.raises(ValidationError, match="timezone-aware"): with pytest.raises(ValidationError, match="timezone-aware"):
CorpusManifest(created_at=datetime(2026, 7, 12), pipeline_version="evidence-v1") CorpusManifest(
created_at=datetime(2026, 7, 12), # noqa: DTZ001 - verifies rejection
pipeline_version="evidence-v1",
)
plus_two = datetime(2026, 7, 12, 12, tzinfo=timezone(timedelta(hours=2))) plus_two = datetime(2026, 7, 12, 12, tzinfo=timezone(timedelta(hours=2)))
manifest = CorpusManifest(created_at=plus_two, pipeline_version="evidence-v1") manifest = CorpusManifest(created_at=plus_two, pipeline_version="evidence-v1")
+1 -1
View File
@@ -3,8 +3,8 @@ from datetime import UTC, datetime
import pytest import pytest
from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
from tht.evidence.contracts import AcquiredDocument, SourceObject from tht.evidence.contracts import AcquiredDocument, SourceObject
from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument: def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument:
+20 -17
View File
@@ -2,11 +2,11 @@ from datetime import UTC, datetime, timedelta
import pytest import pytest
from tht.evidence.contracts import AcquiredDocument, SourceObject
from tht.evidence.corpus.chunk import ChunkPolicy from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.corpus.pipeline import CorpusPipeline, PipelineError, PipelineResult from tht.evidence.corpus.pipeline import CorpusPipeline, PipelineError, PipelineResult
from tht.evidence.corpus.store import CorpusStore from tht.evidence.corpus.store import CorpusStore
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.contracts import AcquiredDocument, SourceObject
from tht.ports.vector import VectorCapabilities, VectorHealth from tht.ports.vector import VectorCapabilities, VectorHealth
@@ -273,6 +273,7 @@ def test_gc_preserves_vector_dependencies_of_retained_manifests(tmp_path):
def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path): def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path):
from types import SimpleNamespace from types import SimpleNamespace
from tht.evidence.search import active_searcher from tht.evidence.search import active_searcher
class Delegate: class Delegate:
@@ -287,6 +288,7 @@ def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path):
def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_path): def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_path):
from types import SimpleNamespace from types import SimpleNamespace
from tht.evidence.search import ActiveEvidenceSearcher from tht.evidence.search import ActiveEvidenceSearcher
store = CorpusStore(tmp_path / "corpus") store = CorpusStore(tmp_path / "corpus")
@@ -319,6 +321,7 @@ def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_
def test_active_evidence_query_holds_lock_against_publish(tmp_path): def test_active_evidence_query_holds_lock_against_publish(tmp_path):
import threading import threading
from types import SimpleNamespace from types import SimpleNamespace
from tht.evidence.search import ActiveEvidenceSearcher from tht.evidence.search import ActiveEvidenceSearcher
first_pipeline = pipeline(tmp_path, Source([(item("one", "a"), "old")]), vectors=Vectors()) first_pipeline = pipeline(tmp_path, Source([(item("one", "a"), "old")]), vectors=Vectors())
@@ -519,9 +522,9 @@ def test_unchanged_job_reuses_active_generation_without_new_directory(tmp_path):
"hello", "hello",
)]) )])
candidate = pipeline(tmp_path, source) candidate = pipeline(tmp_path, source)
args = dict(workspace_id="demo", workspace_root=tmp_path, args = {"workspace_id": "demo", "workspace_root": tmp_path,
config_fingerprint="sha256:" + "1" * 64, "config_fingerprint": "sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64) "input_fingerprint": "sha256:" + "2" * 64}
first = candidate.run_as_job(**args) first = candidate.run_as_job(**args)
snapshot = first.manifest.metadata["source_snapshot"]["fs:one"] snapshot = first.manifest.metadata["source_snapshot"]["fs:one"]
assert snapshot == { assert snapshot == {
@@ -546,9 +549,9 @@ def test_job_source_snapshot_change_forces_publish_with_same_fingerprint(tmp_pat
metadata={"media_type": "text/markdown", "size": 5, "label": "original"}, metadata={"media_type": "text/markdown", "size": 5, "label": "original"},
) )
vectors = Vectors() vectors = Vectors()
args = dict(workspace_id="demo", workspace_root=tmp_path, args = {"workspace_id": "demo", "workspace_root": tmp_path,
config_fingerprint="sha256:" + "1" * 64, "config_fingerprint": "sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64) "input_fingerprint": "sha256:" + "2" * 64}
first = pipeline(tmp_path, Source([(original, "hello")]), vectors=vectors).run_as_job(**args) first = pipeline(tmp_path, Source([(original, "hello")]), vectors=vectors).run_as_job(**args)
updates = { updates = {
"uri": "file:///safe/renamed.md", "uri": "file:///safe/renamed.md",
@@ -567,9 +570,9 @@ def test_job_source_snapshot_change_forces_publish_with_same_fingerprint(tmp_pat
def test_job_binding_change_forces_publish(tmp_path, fingerprint_name): def test_job_binding_change_forces_publish(tmp_path, fingerprint_name):
vectors = Vectors() vectors = Vectors()
source_object = item("one", "a") source_object = item("one", "a")
args = dict(workspace_id="demo", workspace_root=tmp_path, args = {"workspace_id": "demo", "workspace_root": tmp_path,
config_fingerprint="sha256:" + "1" * 64, "config_fingerprint": "sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64) "input_fingerprint": "sha256:" + "2" * 64}
first = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors).run_as_job(**args) first = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors).run_as_job(**args)
args[fingerprint_name] = "sha256:" + "3" * 64 args[fingerprint_name] = "sha256:" + "3" * 64
source = Source([(source_object, "hello")]) source = Source([(source_object, "hello")])
@@ -587,9 +590,9 @@ def test_job_incomplete_active_contract_never_noops(tmp_path, damage):
vectors = Vectors() vectors = Vectors()
source_object = item("one", "a") source_object = item("one", "a")
args = dict(workspace_id="demo", workspace_root=tmp_path, args = {"workspace_id": "demo", "workspace_root": tmp_path,
config_fingerprint="sha256:" + "1" * 64, "config_fingerprint": "sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64) "input_fingerprint": "sha256:" + "2" * 64}
candidate = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors) candidate = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors)
first = candidate.run_as_job(**args) first = candidate.run_as_job(**args)
if damage == "legacy_metadata": if damage == "legacy_metadata":
@@ -628,9 +631,9 @@ def test_job_corrupt_canonical_document_or_chunk_never_noops(tmp_path, damage):
modified_at=datetime(2026, 1, 1, tzinfo=UTC), modified_at=datetime(2026, 1, 1, tzinfo=UTC),
metadata={"media_type": "text/markdown", "size": 11, "owner": "docs"}, metadata={"media_type": "text/markdown", "size": 11, "owner": "docs"},
) )
args = dict(workspace_id="demo", workspace_root=tmp_path, args = {"workspace_id": "demo", "workspace_root": tmp_path,
config_fingerprint="sha256:" + "1" * 64, "config_fingerprint": "sha256:" + "1" * 64,
input_fingerprint="sha256:" + "2" * 64) "input_fingerprint": "sha256:" + "2" * 64}
candidate = pipeline( candidate = pipeline(
tmp_path, Source([(source_object, "hello world")]), vectors=vectors, tmp_path, Source([(source_object, "hello world")]), vectors=vectors,
policy=ChunkPolicy(version="chunk-v1", max_chars=6), policy=ChunkPolicy(version="chunk-v1", max_chars=6),
+6 -3
View File
@@ -1,6 +1,7 @@
import pytest
import os import os
import pytest
from tht.evidence.corpus.models import CorpusManifest from tht.evidence.corpus.models import CorpusManifest
from tht.evidence.corpus.store import CorpusStore, UnsafeCorpusPath from tht.evidence.corpus.store import CorpusStore, UnsafeCorpusPath
@@ -111,9 +112,10 @@ def test_published_inventory_excludes_staged_and_invalid_newer_directories(tmp_p
def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch): def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch):
from tht.evidence.corpus.models import CanonicalDocument
import hashlib import hashlib
from tht.evidence.corpus.models import CanonicalDocument
content = "active bytes" content = "active bytes"
document = CanonicalDocument( document = CanonicalDocument(
document_id="doc:" + "c" * 64, source_id="fs:copy", source_uri="file:///copy", document_id="doc:" + "c" * 64, source_id="fs:copy", source_uri="file:///copy",
@@ -140,9 +142,10 @@ def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_
def test_materialized_snapshot_uses_identified_manifest_when_active_changes(tmp_path): def test_materialized_snapshot_uses_identified_manifest_when_active_changes(tmp_path):
from tht.evidence.corpus.models import CanonicalDocument
import hashlib import hashlib
from tht.evidence.corpus.models import CanonicalDocument
def doc(content, fingerprint): def doc(content, fingerprint):
return CanonicalDocument( return CanonicalDocument(
document_id="doc:" + hashlib.sha256(content.encode()).hexdigest(), document_id="doc:" + hashlib.sha256(content.encode()).hexdigest(),
+7 -7
View File
@@ -6,7 +6,7 @@ cte_plan_doc.json entry, and the last CteTestRecord for the CTE. --json output m
pristine (only valid JSON on stdout). pristine (only valid JSON on stdout).
""" """
import json import json
from datetime import datetime from datetime import UTC, datetime
from typer.testing import CliRunner from typer.testing import CliRunner
@@ -18,7 +18,7 @@ from tht.session.store import create_session
def _db(): def _db():
return DatabaseConfig(database="testdb", user="u", password="p", **{"schema": "public"}) # noqa: S106 return DatabaseConfig(database="testdb", user="u", password="p", schema="public")
def _patch_cfg(monkeypatch, tmp_path): def _patch_cfg(monkeypatch, tmp_path):
@@ -46,7 +46,7 @@ def test_info_happy_path_json(tmp_path, monkeypatch):
sid, sdir = _make_session(tmp_path, ["a", "b"]) sid, sdir = _make_session(tmp_path, ["a", "b"])
(sdir / "ctes" / "b.sql").write_text("WITH b AS (SELECT 1)") (sdir / "ctes" / "b.sql").write_text("WITH b AS (SELECT 1)")
append_cte_test(sdir, CteTestRecord( append_cte_test(sdir, CteTestRecord(
name="b", ts=datetime(2025, 1, 1, 12, 0), sql_hash="h1", status="ok", name="b", ts=datetime(2025, 1, 1, 12, 0, tzinfo=UTC), sql_hash="h1", status="ok",
columns=["x"], row_sample=1, execution_ms=5, preview_rows=[[1]], columns=["x"], row_sample=1, execution_ms=5, preview_rows=[[1]],
)) ))
_patch_cfg(monkeypatch, tmp_path) _patch_cfg(monkeypatch, tmp_path)
@@ -105,10 +105,10 @@ def test_info_last_test_is_most_recent_record(tmp_path, monkeypatch):
sid, sdir = _make_session(tmp_path, ["a"]) sid, sdir = _make_session(tmp_path, ["a"])
(sdir / "ctes" / "a.sql").write_text("WITH a AS (SELECT 1)") (sdir / "ctes" / "a.sql").write_text("WITH a AS (SELECT 1)")
append_cte_test(sdir, CteTestRecord( append_cte_test(sdir, CteTestRecord(
name="a", ts=datetime(2025, 1, 1), sql_hash="old", status="ok", name="a", ts=datetime(2025, 1, 1, tzinfo=UTC), sql_hash="old", status="ok",
)) ))
append_cte_test(sdir, CteTestRecord( append_cte_test(sdir, CteTestRecord(
name="a", ts=datetime(2025, 1, 2), sql_hash="new", status="ok", name="a", ts=datetime(2025, 1, 2, tzinfo=UTC), sql_hash="new", status="ok",
)) ))
_patch_cfg(monkeypatch, tmp_path) _patch_cfg(monkeypatch, tmp_path)
@@ -134,7 +134,7 @@ def test_info_missing_plan_exit_1(tmp_path, monkeypatch):
def test_info_name_not_in_plan_exit_1(tmp_path, monkeypatch): def test_info_name_not_in_plan_exit_1(tmp_path, monkeypatch):
sid, sdir = _make_session(tmp_path, ["a"]) sid, _sdir = _make_session(tmp_path, ["a"])
_patch_cfg(monkeypatch, tmp_path) _patch_cfg(monkeypatch, tmp_path)
res = CliRunner().invoke(cte_app, ["info", "zzz", "--session", sid, "--json"]) res = CliRunner().invoke(cte_app, ["info", "zzz", "--session", sid, "--json"])
assert res.exit_code == 1 assert res.exit_code == 1
@@ -142,7 +142,7 @@ def test_info_name_not_in_plan_exit_1(tmp_path, monkeypatch):
def test_info_missing_sql_file_exit_1(tmp_path, monkeypatch): def test_info_missing_sql_file_exit_1(tmp_path, monkeypatch):
sid, sdir = _make_session(tmp_path, ["a"]) sid, _sdir = _make_session(tmp_path, ["a"])
_patch_cfg(monkeypatch, tmp_path) _patch_cfg(monkeypatch, tmp_path)
res = CliRunner().invoke(cte_app, ["info", "a", "--session", sid, "--json"]) res = CliRunner().invoke(cte_app, ["info", "a", "--session", sid, "--json"])
assert res.exit_code == 1 assert res.exit_code == 1
+1 -1
View File
@@ -10,7 +10,7 @@ from tht.session.store import create_session
def _db(): def _db():
return DatabaseConfig(database="testdb", user="u", password="p", **{"schema": "public"}) # noqa: S106 return DatabaseConfig(database="testdb", user="u", password="p", schema="public")
def _patch_cfg(monkeypatch, tmp_path): def _patch_cfg(monkeypatch, tmp_path):
+1 -1
View File
@@ -17,7 +17,7 @@ from tht.session.store import create_session
def _db(): def _db():
return DatabaseConfig(database="testdb", user="u", password="p", **{"schema": "public"}) # noqa: S106 return DatabaseConfig(database="testdb", user="u", password="p", schema="public")
def _patch_cfg(monkeypatch, tmp_path): def _patch_cfg(monkeypatch, tmp_path):
+6 -6
View File
@@ -7,7 +7,7 @@ and the ledger I/O (load/append, tolerant of JSON-array and JSONL formats).
Pure logic, no DB. Pure logic, no DB.
""" """
import json import json
from datetime import date, datetime from datetime import UTC, date, datetime
from decimal import Decimal from decimal import Decimal
import pytest import pytest
@@ -82,10 +82,10 @@ def test_build_test_sql_single_cte():
# --- ledger I/O: load_cte_tests / append_cte_test --------------------------- # --- ledger I/O: load_cte_tests / append_cte_test ---------------------------
def _record(**kw) -> CteTestRecord: def _record(**kw) -> CteTestRecord:
base = dict( base = {
name="ablazione_q", ts=datetime(2025, 1, 1, 12, 0), sql_hash="abc123", "name": "ablazione_q", "ts": datetime(2025, 1, 1, 12, 0, tzinfo=UTC), "sql_hash": "abc123",
status="ok", columns=["x"], row_sample=5, execution_ms=42, "status": "ok", "columns": ["x"], "row_sample": 5, "execution_ms": 42,
) }
base.update(kw) base.update(kw)
return CteTestRecord(**base) return CteTestRecord(**base)
@@ -147,7 +147,7 @@ def test_jsonable_passes_through_native_types():
def test_jsonable_coerces_decimal_and_date_to_str(): def test_jsonable_coerces_decimal_and_date_to_str():
assert _jsonable(Decimal("12.34")) == "12.34" assert _jsonable(Decimal("12.34")) == "12.34"
assert _jsonable(date(2025, 1, 1)) == "2025-01-01" assert _jsonable(date(2025, 1, 1)) == "2025-01-01"
assert _jsonable(datetime(2025, 1, 1, 12, 0, 0)) == "2025-01-01 12:00:00" assert _jsonable(datetime(2025, 1, 1, 12, 0, 0)) == "2025-01-01 12:00:00" # noqa: DTZ001
def test_jsonable_truncates_long_strings(): def test_jsonable_truncates_long_strings():
+1 -1
View File
@@ -17,7 +17,7 @@ def _concurrent_append_worker(session, subject, start, ready, done):
try: try:
append_decision(session, type="concept_clarified", subject=subject) append_decision(session, type="concept_clarified", subject=subject)
done.put((subject, None)) done.put((subject, None))
except Exception as error: # pragma: no cover - surfaced through the parent assertion except Exception as error: # noqa: BLE001 # pragma: no cover - sent to parent
done.put((subject, repr(error))) done.put((subject, repr(error)))
+2 -1
View File
@@ -57,8 +57,9 @@ def test_empty_session_returns_empty_list(tmp_path):
def test_decision_type_literal_includes_retracted(): def test_decision_type_literal_includes_retracted():
"""decision_retracted e' un tipo valido (pydantic lo accetta).""" """decision_retracted e' un tipo valido (pydantic lo accetta)."""
from datetime import UTC, datetime
from tht.decisions import DecisionRecord from tht.decisions import DecisionRecord
from datetime import datetime, UTC
d = DecisionRecord( d = DecisionRecord(
seq=1, ts=datetime.now(UTC), type="decision_retracted", seq=1, ts=datetime.now(UTC), type="decision_retracted",
subject="phase:4", retracts=1, subject="phase:4", retracts=1,
-1
View File
@@ -5,7 +5,6 @@ from typer.testing import CliRunner
from tht.cli import app from tht.cli import app
runner = CliRunner() runner = CliRunner()
+1 -1
View File
@@ -1,12 +1,12 @@
import pytest import pytest
from sqlalchemy.exc import OperationalError from sqlalchemy.exc import OperationalError
from tht.adapters.dwh import PostgresDwhAdapter
from tht.config import DatabaseConfig, RestConfig from tht.config import DatabaseConfig, RestConfig
from tht.db.sampling import distinct_values_rest, sample_column_rest from tht.db.sampling import distinct_values_rest, sample_column_rest
from tht.execute import ExecutionError from tht.execute import ExecutionError
from tht.ports import DistinctValues, DwhAdapter from tht.ports import DistinctValues, DwhAdapter
from tht.rest.client import RestError from tht.rest.client import RestError
from tht.adapters.dwh import PostgresDwhAdapter
def postgres_factory(): def postgres_factory():
+2 -2
View File
@@ -5,10 +5,10 @@ import pytest
from tht.execute import ExecResult, PlanSummary from tht.execute import ExecResult, PlanSummary
from tht.mschema.models import PhysicalSchema from tht.mschema.models import PhysicalSchema
from tht.ports.dwh import ( from tht.ports.dwh import (
DistinctValues,
DwhAdapter, DwhAdapter,
DwhCapabilities, DwhCapabilities,
DwhHealth, DwhHealth,
DistinctValues,
UnsupportedCapability, UnsupportedCapability,
) )
@@ -55,10 +55,10 @@ def test_contract_types_are_public_and_capabilities_are_immutable():
def test_all_contract_types_are_exported_from_public_package(): def test_all_contract_types_are_exported_from_public_package():
from tht.ports import DistinctValues as PublicDistinctValues
from tht.ports import DwhAdapter as PublicDwhAdapter from tht.ports import DwhAdapter as PublicDwhAdapter
from tht.ports import DwhCapabilities as PublicDwhCapabilities from tht.ports import DwhCapabilities as PublicDwhCapabilities
from tht.ports import DwhHealth as PublicDwhHealth from tht.ports import DwhHealth as PublicDwhHealth
from tht.ports import DistinctValues as PublicDistinctValues
from tht.ports import UnsupportedCapability as PublicUnsupportedCapability from tht.ports import UnsupportedCapability as PublicUnsupportedCapability
result = PublicDistinctValues(values=["a"], truncated=True) result = PublicDistinctValues(values=["a"], truncated=True)
+22 -14
View File
@@ -6,13 +6,16 @@ from typer.testing import CliRunner
from tht.cli import app from tht.cli import app
from tht.config import load_config from tht.config import load_config
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline from tht.jobs.dwh_pipeline import (
from tht.jobs.dwh_pipeline import active_generation_dir, config_dwh_binding, fingerprint DwhPreprocessPipeline,
from tht.jobs.dwh_pipeline import resolve_dwh_snapshot active_generation_dir,
from tht.jobs.dwh_pipeline import lease_dwh_snapshot config_dwh_binding,
fingerprint,
lease_dwh_snapshot,
resolve_dwh_snapshot,
)
from tht.jobs.locking import _lock_name from tht.jobs.locking import _lock_name
FP = "sha256:" + hashlib.sha256(b"test").hexdigest() FP = "sha256:" + hashlib.sha256(b"test").hexdigest()
@@ -104,8 +107,7 @@ def test_unowned_reads_fail_closed_without_creating_any_files(tmp_path):
cfg = snapshot_config(tmp_path) cfg = snapshot_config(tmp_path)
with pytest.raises(Exception, match="not initialized"): with pytest.raises(Exception, match="not initialized"):
resolve_dwh_snapshot(cfg) resolve_dwh_snapshot(cfg)
with pytest.raises(Exception, match="not initialized"): with pytest.raises(Exception, match="not initialized"), lease_dwh_snapshot(cfg):
with lease_dwh_snapshot(cfg):
pass pass
assert not (tmp_path / ".tht-dwh").exists() assert not (tmp_path / ".tht-dwh").exists()
@@ -125,7 +127,7 @@ def test_writer_claim_allows_only_lock_and_empty_generations(tmp_path):
pipeline = DwhPreprocessPipeline( pipeline = DwhPreprocessPipeline(
workspace_id="demo", workspace_root=root.parent, workspace_id="demo", workspace_root=root.parent,
config_fingerprint=FP, input_fingerprint=FP, config_fingerprint=FP, input_fingerprint=FP,
introspect=lambda output: calls.append("called"), introspect=lambda output, calls=calls: calls.append("called"),
build_lsh=lambda physical, output: None, build_lsh=lambda physical, output: None,
) )
with pytest.raises(Exception, match="unbound"): with pytest.raises(Exception, match="unbound"):
@@ -188,6 +190,7 @@ def test_owner_publication_remains_on_locked_root_when_path_is_swapped(
monkeypatch, tmp_path, monkeypatch, tmp_path,
): ):
import pytest import pytest
import tht.jobs.dwh_pipeline as module import tht.jobs.dwh_pipeline as module
real_replace = module.os.replace real_replace = module.os.replace
@@ -211,7 +214,7 @@ def test_owner_publication_remains_on_locked_root_when_path_is_swapped(
introspect=lambda output: (_ for _ in ()).throw(AssertionError("callback called")), introspect=lambda output: (_ for _ in ()).throw(AssertionError("callback called")),
build_lsh=lambda physical, output: None, build_lsh=lambda physical, output: None,
) )
with pytest.raises(Exception): with pytest.raises(Exception, match="root"):
pipeline.run() pipeline.run()
assert swapped assert swapped
assert (moved / "OWNER.json").is_file() assert (moved / "OWNER.json").is_file()
@@ -394,7 +397,7 @@ def test_missing_active_with_generations_and_symlink_owner_marker_fail_closed(tm
def _capture_error(operation): def _capture_error(operation):
try: try:
return operation() return operation()
except Exception as error: except Exception as error: # noqa: BLE001 - helper returns the exact injected failure
return error return error
@@ -513,6 +516,7 @@ def test_unsafe_lsh_filename_is_rejected(tmp_path):
def test_active_fsync_failure_restores_previous_pointer(monkeypatch, tmp_path): def test_active_fsync_failure_restores_previous_pointer(monkeypatch, tmp_path):
import os import os
import tht.jobs.dwh_pipeline as module import tht.jobs.dwh_pipeline as module
def build(physical, output): def build(physical, output):
for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json"): for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json"):
@@ -585,7 +589,7 @@ def test_snapshot_root_swap_after_lease_never_reads_replacement(monkeypatch, tmp
try: try:
with lease_dwh_snapshot(snapshot_config(tmp_path)) as snapshot: with lease_dwh_snapshot(snapshot_config(tmp_path)) as snapshot:
assert snapshot.physical.read_text() == "trusted" assert snapshot.physical.read_text() == "trusted"
except Exception as error: except Exception as error: # noqa: BLE001 - either safe refusal path is acceptable
assert "ACTIVE" in str(error) or "root" in str(error) assert "ACTIVE" in str(error) or "root" in str(error)
assert swapped assert swapped
assert (replacement / "sentinel").read_text() == "replacement-secret" assert (replacement / "sentinel").read_text() == "replacement-secret"
@@ -623,7 +627,9 @@ def test_snapshot_copies_each_validated_artifact_once_without_reopen(monkeypatch
def test_reconcile_mismatch_closes_active_generation_fd(monkeypatch, tmp_path): def test_reconcile_mismatch_closes_active_generation_fd(monkeypatch, tmp_path):
import os import os
from types import SimpleNamespace from types import SimpleNamespace
import pytest import pytest
import tht.jobs.dwh_pipeline as module import tht.jobs.dwh_pipeline as module
pipeline = DwhPreprocessPipeline( pipeline = DwhPreprocessPipeline(
@@ -637,7 +643,7 @@ def test_reconcile_mismatch_closes_active_generation_fd(monkeypatch, tmp_path):
real_active = module._active_generation_fd real_active = module._active_generation_fd
def mismatched_active(root_fd, binding): def mismatched_active(root_fd, binding):
generation, generation_fd = real_active(root_fd, binding) _generation, generation_fd = real_active(root_fd, binding)
return "f" * 32, generation_fd return "f" * 32, generation_fd
monkeypatch.setattr(module, "_active_generation_fd", mismatched_active) monkeypatch.setattr(module, "_active_generation_fd", mismatched_active)
@@ -676,9 +682,10 @@ def test_pipeline_releases_materialized_snapshot_after_every_run(tmp_path):
def test_corrupt_resume_checkpoint_releases_materialized_snapshot(tmp_path): def test_corrupt_resume_checkpoint_releases_materialized_snapshot(tmp_path):
import tht.jobs.dwh_pipeline as module
import pytest import pytest
import tht.jobs.dwh_pipeline as module
pipeline = DwhPreprocessPipeline( pipeline = DwhPreprocessPipeline(
workspace_id="demo", workspace_root=tmp_path, workspace_id="demo", workspace_root=tmp_path,
config_fingerprint=FP, input_fingerprint=FP, config_fingerprint=FP, input_fingerprint=FP,
@@ -699,9 +706,10 @@ def test_corrupt_resume_checkpoint_releases_materialized_snapshot(tmp_path):
def test_job_spec_construction_failure_releases_materialized_snapshot(monkeypatch, tmp_path): def test_job_spec_construction_failure_releases_materialized_snapshot(monkeypatch, tmp_path):
import tht.jobs.dwh_pipeline as module
import pytest import pytest
import tht.jobs.dwh_pipeline as module
pipeline = DwhPreprocessPipeline( pipeline = DwhPreprocessPipeline(
workspace_id="demo", workspace_root=tmp_path, workspace_id="demo", workspace_root=tmp_path,
config_fingerprint=FP, input_fingerprint=FP, config_fingerprint=FP, input_fingerprint=FP,
@@ -1,13 +1,11 @@
from datetime import UTC, datetime
import hashlib import hashlib
import inspect import inspect
from datetime import UTC, datetime
from pathlib import Path from pathlib import Path
from types import SimpleNamespace from types import SimpleNamespace
import pytest import pytest
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.decisions import DecisionRecord from tht.decisions import DecisionRecord
from tht.evidence import ( from tht.evidence import (
acquire, acquire,
@@ -25,6 +23,8 @@ from tht.evidence.contracts import (
EvidenceSourceErrorCategory, EvidenceSourceErrorCategory,
SourceObject, SourceObject,
) )
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.session.models import Candidate, SchemaLinking from tht.session.models import Candidate, SchemaLinking
-1
View File
@@ -1,6 +1,5 @@
from pathlib import Path from pathlib import Path
HARNESS_ROOT = Path(__file__).resolve().parents[1] HARNESS_ROOT = Path(__file__).resolve().parents[1]
LEGACY_PATHS = ( LEGACY_PATHS = (
"tht/ports/evidence.py", "tht/ports/evidence.py",
+6 -2
View File
@@ -154,7 +154,7 @@ def test_datetimes_must_be_aware_and_are_normalized_to_utc():
source_id="source:a", source_id="source:a",
uri="https://host/a", uri="https://host/a",
fingerprint="etag:abc", fingerprint="etag:abc",
modified_at=datetime(2026, 7, 12), modified_at=datetime(2026, 7, 12), # noqa: DTZ001 - verifies rejection
) )
source = SourceObject( source = SourceObject(
@@ -236,4 +236,8 @@ def test_model_copy_revalidates_source_and_acquired_records():
with pytest.raises(ValidationError, match="namespaced"): with pytest.raises(ValidationError, match="namespaced"):
source.model_copy(update={"source_id": "invalid"}) source.model_copy(update={"source_id": "invalid"})
with pytest.raises(ValidationError, match="timezone-aware"): with pytest.raises(ValidationError, match="timezone-aware"):
acquired.model_copy(update={"acquired_at": datetime(2026, 7, 12)}) acquired.model_copy(
update={
"acquired_at": datetime(2026, 7, 12) # noqa: DTZ001 - verifies rejection
}
)
+1 -1
View File
@@ -46,7 +46,7 @@ def test_union_query_accepts_limit():
def test_with_cte_query_accepts_limit(): def test_with_cte_query_accepts_limit():
sql = "WITH cte AS (SELECT 1) SELECT * FROM cte" sql = "WITH cte AS (SELECT 1) SELECT * FROM cte"
out, injected = _inject_limit(sql, 10) _out, injected = _inject_limit(sql, 10)
assert injected is True assert injected is True
+1
View File
@@ -70,6 +70,7 @@ def test_retrieve_empty_on_missing_dir(tmp_path):
def test_concept_formula_decision_types_exist(): def test_concept_formula_decision_types_exist():
import typing import typing
from tht.decisions import DecisionType from tht.decisions import DecisionType
args = typing.get_args(DecisionType) args = typing.get_args(DecisionType)
assert "concept_formula_approved" in args assert "concept_formula_approved" in args
+6 -5
View File
@@ -1,6 +1,7 @@
import threading
import socket import socket
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from typing import ClassVar
import pytest import pytest
@@ -12,8 +13,8 @@ class Handler(BaseHTTPRequestHandler):
etag_requests = 0 etag_requests = 0
etag_body_responses = 0 etag_body_responses = 0
redirect_target = "/redirected-v1" redirect_target = "/redirected-v1"
redirect_request_validators = [] redirect_request_validators: ClassVar[list[str | None]] = []
final_request_validators = [] final_request_validators: ClassVar[list[tuple[str, str | None]]] = []
def do_GET(self): def do_GET(self):
if self.path.startswith("/etag"): if self.path.startswith("/etag"):
@@ -198,7 +199,6 @@ class FakeSocket:
class FakeResponse: class FakeResponse:
status_code = 200 status_code = 200
headers = {}
is_redirect = False is_redirect = False
def __init__(self, *, peer="127.0.0.1", stream_error=None, location=None): def __init__(self, *, peer="127.0.0.1", stream_error=None, location=None):
@@ -214,6 +214,7 @@ class FakeResponse:
self.is_redirect = False self.is_redirect = False
self.status_code = 200 self.status_code = 200
self.headers = {} self.headers = {}
self.headers = {}
def iter_content(self, chunk_size): def iter_content(self, chunk_size):
if self.stream_error: if self.stream_error:
@@ -265,7 +266,7 @@ def test_http_closes_response_when_streaming_fails():
source = HttpManifestEvidenceSource( source = HttpManifestEvidenceSource(
["https://example.test/doc"], allow_private_hosts=True ["https://example.test/doc"], allow_private_hosts=True
) )
response = FakeResponse(stream_error=socket.timeout("read timed out")) response = FakeResponse(stream_error=TimeoutError("read timed out"))
source._session = FakeSession(response) source._session = FakeSession(response)
with pytest.raises(EvidenceSourceError): with pytest.raises(EvidenceSourceError):
list(source.discover()) list(source.discover())
+4 -2
View File
@@ -37,8 +37,10 @@ def test_same_workspace_and_job_are_exclusive_across_processes(tmp_path):
def test_evidence_and_dwh_jobs_have_distinct_locks(tmp_path): def test_evidence_and_dwh_jobs_have_distinct_locks(tmp_path):
with WorkspaceJobLock(tmp_path, "demo", "evidence"): with (
with WorkspaceJobLock(tmp_path, "demo", "dwh"): WorkspaceJobLock(tmp_path, "demo", "evidence"),
WorkspaceJobLock(tmp_path, "demo", "dwh"),
):
pass pass
+2 -1
View File
@@ -1,11 +1,12 @@
import json import json
import os import os
import pytest import pytest
from pydantic import ValidationError from pydantic import ValidationError
import tht.jobs.runner as runner_module
from tht.jobs.models import JobSpec from tht.jobs.models import JobSpec
from tht.jobs.runner import CorruptCheckpointError, StageArtifacts, run_job from tht.jobs.runner import CorruptCheckpointError, StageArtifacts, run_job
import tht.jobs.runner as runner_module
def _spec(tmp_path, **updates): def _spec(tmp_path, **updates):
+2 -6
View File
@@ -3,7 +3,7 @@ import hashlib
import pytest import pytest
from tht.jobs.dwh_pipeline import DwhPreprocessPipeline from tht.jobs.dwh_pipeline import DwhPreprocessPipeline
from tht.jobs.runner import CorruptCheckpointError
FP = "sha256:" + hashlib.sha256(b"test").hexdigest() FP = "sha256:" + hashlib.sha256(b"test").hexdigest()
@@ -67,12 +67,8 @@ def test_resume_rejects_a_different_stage_selection(tmp_path):
introspect=lambda output: output.write_text("catalog"), introspect=lambda output: output.write_text("catalog"),
build_lsh=lambda physical, output: None, build_lsh=lambda physical, output: None,
) )
try: with pytest.raises(CorruptCheckpointError, match="incompatible"):
pipeline.run(("lsh",), resume_run_id=failed.run_id) pipeline.run(("lsh",), resume_run_id=failed.run_id)
except Exception as error:
assert "incompatible" in str(error)
else:
raise AssertionError("resume with different stages must fail")
def test_resume_rejects_tampered_succeeded_stage_artifact(tmp_path): def test_resume_rejects_tampered_succeeded_stage_artifact(tmp_path):
+1 -1
View File
@@ -1,5 +1,5 @@
from tht.session.store import create_session
from tht.session.models import SessionManifest from tht.session.models import SessionManifest
from tht.session.store import create_session
def test_manifest_persists_pi_fields(tmp_path): def test_manifest_persists_pi_fields(tmp_path):
+7 -7
View File
@@ -5,18 +5,18 @@ fonte delle memory. L'hit di search_similar deve bastare per ricostruire la deci
completa. Per questo memory_vector_records mette subject/detail/rationale nel metadata completa. Per questo memory_vector_records mette subject/detail/rationale nel metadata
del VectorRecord (pack_metadata li serializza nel jsonb via **record.metadata). del VectorRecord (pack_metadata li serializza nel jsonb via **record.metadata).
""" """
from datetime import datetime from datetime import UTC, datetime
from tht.memory import MemoryRecord, memory_vector_records from tht.memory import MemoryRecord, memory_vector_records
def _record(**kw) -> MemoryRecord: def _record(**kw) -> MemoryRecord:
base = dict( base = {
id="mem-x", ts=datetime(2025, 1, 1), session_id="s", decision_seq=1, "id": "mem-x", "ts": datetime(2025, 1, 1, tzinfo=UTC), "session_id": "s", "decision_seq": 1,
type="concept_clarified", subject="paziente attivo", detail="flag_attivo = TRUE", "type": "concept_clarified", "subject": "paziente attivo", "detail": "flag_attivo = TRUE",
rationale="perche' serve", question_context="dammi pazienti", "rationale": "perche' serve", "question_context": "dammi pazienti",
tables=[], concepts=["paziente attivo"], "tables": [], "concepts": ["paziente attivo"],
) }
base.update(kw) base.update(kw)
return MemoryRecord(**base) return MemoryRecord(**base)
+8 -8
View File
@@ -6,7 +6,7 @@ la rende un passo del workflow. Questi test fissano il contratto harness-side:
- i candidati rifiutati al gate (memory_promotion_declined, detail "seq:<n>") non - i candidati rifiutati al gate (memory_promotion_declined, detail "seq:<n>") non
vengono riproposti da reusable_promotions/preview_promotions. vengono riproposti da reusable_promotions/preview_promotions.
""" """
from datetime import datetime from datetime import UTC, datetime
from tht.decisions import DecisionRecord, append_decision from tht.decisions import DecisionRecord, append_decision
from tht.memory import ( from tht.memory import (
@@ -23,7 +23,7 @@ from tht.workflow import load_workflow
def test_promotion_decision_types_are_valid(): def test_promotion_decision_types_are_valid():
for t in ("memory_promoted", "memory_promotion_declined"): for t in ("memory_promoted", "memory_promotion_declined"):
d = DecisionRecord( d = DecisionRecord(
seq=1, ts=datetime(2026, 1, 1), type=t, subject="fact_x", detail="seq:5" seq=1, ts=datetime(2026, 1, 1, tzinfo=UTC), type=t, subject="fact_x", detail="seq:5"
) )
assert d.type == t assert d.type == t
@@ -36,14 +36,14 @@ def test_promotion_decision_types_min_phase_is_f8():
def _manifest() -> SessionManifest: def _manifest() -> SessionManifest:
return SessionManifest( return SessionManifest(
id="s1", created_at=datetime(2026, 1, 1), question="domanda originale", id="s1", created_at=datetime(2026, 1, 1, tzinfo=UTC), question="domanda originale",
database="db", schema="public", database="db", schema="public",
) )
def test_declined_promotion_seqs_parses_seq_detail(): def test_declined_promotion_seqs_parses_seq_detail():
d = DecisionRecord( d = DecisionRecord(
seq=9, ts=datetime(2026, 1, 1), type="memory_promotion_declined", seq=9, ts=datetime(2026, 1, 1, tzinfo=UTC), type="memory_promotion_declined",
subject="fact_x", detail="seq:5", subject="fact_x", detail="seq:5",
) )
assert declined_promotion_seqs([d]) == {5} assert declined_promotion_seqs([d]) == {5}
@@ -51,9 +51,9 @@ def test_declined_promotion_seqs_parses_seq_detail():
def test_declined_promotion_seqs_ignores_malformed_and_other_types(): def test_declined_promotion_seqs_ignores_malformed_and_other_types():
ds = [ ds = [
DecisionRecord(seq=1, ts=datetime(2026, 1, 1), DecisionRecord(seq=1, ts=datetime(2026, 1, 1, tzinfo=UTC),
type="memory_promotion_declined", subject="x", detail=""), type="memory_promotion_declined", subject="x", detail=""),
DecisionRecord(seq=2, ts=datetime(2026, 1, 1), DecisionRecord(seq=2, ts=datetime(2026, 1, 1, tzinfo=UTC),
type="table_promoted", subject="x", detail="seq:3"), type="table_promoted", subject="x", detail="seq:3"),
] ]
assert declined_promotion_seqs(ds) == set() assert declined_promotion_seqs(ds) == set()
@@ -100,11 +100,11 @@ def test_only_concept_clarified_is_proposed_or_promoted(tmp_path):
def test_legacy_table_records_are_not_published_as_memory_vectors(): def test_legacy_table_records_are_not_published_as_memory_vectors():
records = [ records = [
MemoryRecord( MemoryRecord(
id="mem-0001", ts=datetime(2026, 1, 1), session_id="s1", id="mem-0001", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1",
decision_seq=1, type="table_promoted", subject="fact_a", decision_seq=1, type="table_promoted", subject="fact_a",
), ),
MemoryRecord( MemoryRecord(
id="mem-0002", ts=datetime(2026, 1, 1), session_id="s1", id="mem-0002", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1",
decision_seq=2, type="concept_clarified", subject="paziente attivo", decision_seq=2, type="concept_clarified", subject="paziente attivo",
detail="flag_attivo = TRUE", detail="flag_attivo = TRUE",
), ),
-1
View File
@@ -7,7 +7,6 @@ from pathlib import Path
import pytest import pytest
THT_ROOT = Path(__file__).resolve().parents[1] / "tht" THT_ROOT = Path(__file__).resolve().parents[1] / "tht"
DOMAIN_PACKAGES = ("evidence", "memory") DOMAIN_PACKAGES = ("evidence", "memory")
+5 -5
View File
@@ -4,7 +4,7 @@ Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
comes only from numerics, enums, temporals, booleans, and short text. Annotation comes only from numerics, enums, temporals, booleans, and short text. Annotation
override wins over the physical classification. override wins over the physical classification.
""" """
from datetime import datetime from datetime import UTC, datetime
from tht.config import EligibilityConfig from tht.config import EligibilityConfig
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
@@ -70,16 +70,16 @@ def test_array_type_is_wide_text():
def _schema_with(**columns) -> PhysicalSchema: def _schema_with(**columns) -> PhysicalSchema:
return PhysicalSchema( return PhysicalSchema(
database="db", schema="dw", introspected_at=datetime(2025, 1, 1), database="db", schema="dw", introspected_at=datetime(2025, 1, 1, tzinfo=UTC),
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})}, tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
) )
def test_classify_all_marks_wide_text_and_clears_examples(): def test_classify_all_marks_wide_text_and_clears_examples():
schema = _schema_with( schema = _schema_with(
note=dict(type="text", examples=["a" * 500, "b" * 400]), note={"type": "text", "examples": ["a" * 500, "b" * 400]},
cod=dict(type="varchar(10)", examples=["X", "Y"]), cod={"type": "varchar(10)", "examples": ["X", "Y"]},
etl_last_update=dict(type="timestamp", examples=[]), etl_last_update={"type": "timestamp", "examples": []},
) )
classify_all(schema, CFG) classify_all(schema, CFG)
cols = schema.tables["t"].columns cols = schema.tables["t"].columns
+2 -2
View File
@@ -3,7 +3,7 @@
Catches port breaks in the render layer (markdown reviewer report, mschema-text Catches port breaks in the render layer (markdown reviewer report, mschema-text
ThothAI style, schema-dict for AV-SQL). Pure logic, no I/O. ThothAI style, schema-dict for AV-SQL). Pure logic, no I/O.
""" """
from datetime import datetime from datetime import UTC, datetime
from tht.mschema.models import ( from tht.mschema.models import (
Annotations, Annotations,
@@ -20,7 +20,7 @@ def _fake_schema() -> PhysicalSchema:
return PhysicalSchema( return PhysicalSchema(
database="testdb", database="testdb",
schema="dw", schema="dw",
introspected_at=datetime(2025, 1, 1, 0, 0, 0), introspected_at=datetime(2025, 1, 1, 0, 0, 0, tzinfo=UTC),
tables={ tables={
"dim_pazienti": TablePhysical( "dim_pazienti": TablePhysical(
comment="Anagrafica pazienti", comment="Anagrafica pazienti",
+1 -1
View File
@@ -4,9 +4,9 @@ from types import SimpleNamespace
from typer.testing import CliRunner from typer.testing import CliRunner
from tht.config import EmbeddingsConfig
from tht.cli import ollama_cmd from tht.cli import ollama_cmd
from tht.cli.ollama_cmd import ensure_ollama, ollama_app from tht.cli.ollama_cmd import ensure_ollama, ollama_app
from tht.config import EmbeddingsConfig
def _cfg(**kw): def _cfg(**kw):
+1 -1
View File
@@ -1,4 +1,4 @@
from tht.decisions import append_decision, DecisionRecord from tht.decisions import DecisionRecord, append_decision
from tht.phase import current_phase, effective_decisions from tht.phase import current_phase, effective_decisions
+1 -1
View File
@@ -81,7 +81,7 @@ def test_successful_reopen_records_ledger_and_runs_teardown(tmp_path, monkeypatc
calls = [] calls = []
class _Report: class _Report:
deleted_files = [] deleted_files = ()
def _spy(_repository, snapshot, phase): def _spy(_repository, snapshot, phase):
# By the time teardown runs, the ledger already holds the reopen decision. # By the time teardown runs, the ledger already holds the reopen decision.
@@ -9,7 +9,6 @@ from tht.pi_skill_projection import (
render_projection, render_projection,
) )
BASELINE_SHA256 = "626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219" BASELINE_SHA256 = "626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219"
+2 -1
View File
@@ -1,4 +1,5 @@
import json import json
from tht.cli.sql_cmd import promoted_columns_for from tht.cli.sql_cmd import promoted_columns_for
@@ -21,6 +22,6 @@ def test_promoted_columns_for(tmp_path):
})) }))
class Cfg: class Cfg:
class paths: # noqa: N801 class paths:
sessions = tmp_path sessions = tmp_path
assert promoted_columns_for(Cfg, sid) == {"dim_patient.cod_paz"} assert promoted_columns_for(Cfg, sid) == {"dim_patient.cod_paz"}
@@ -6,10 +6,10 @@ import yaml
from pydantic import SecretStr from pydantic import SecretStr
from typer.testing import CliRunner from typer.testing import CliRunner
from tht.evidence.adapters import HttpManifestEvidenceSource
from tht.evidence import build_sources
from tht.cli import app from tht.cli import app
from tht.config import ConfigError, load_config from tht.config import ConfigError, load_config
from tht.evidence import build_sources
from tht.evidence.adapters import HttpManifestEvidenceSource
SIGNED_CANARY = "SIGNED-CANARY-QUERY" SIGNED_CANARY = "SIGNED-CANARY-QUERY"
ACCESS_CANARY = "ACCESS-CANARY" ACCESS_CANARY = "ACCESS-CANARY"
+3 -1
View File
@@ -99,7 +99,9 @@ def test_s3_rejects_leading_slash_prefix_empty_and_control_keys():
S3EvidenceSource(bucket="evidence", prefix="/clinical", client=Client()) S3EvidenceSource(bucket="evidence", prefix="/clinical", client=Client())
for key in ("", "clinical/a\x00.md", "clinical/a\x7f.md"): for key in ("", "clinical/a\x00.md", "clinical/a\x7f.md"):
client = Client() client = Client()
client.list_objects_v2 = lambda **kwargs: {"Contents": [{"Key": key, "ETag": '"x"'}]} client.list_objects_v2 = lambda key=key, **kwargs: {
"Contents": [{"Key": key, "ETag": '"x"'}]
}
with pytest.raises(EvidenceSourceError): with pytest.raises(EvidenceSourceError):
list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover()) list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover())
+3 -3
View File
@@ -1,5 +1,5 @@
import json import json
from datetime import datetime from datetime import UTC, datetime
from typer.testing import CliRunner from typer.testing import CliRunner
@@ -9,7 +9,7 @@ from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
def _write_catalog(tmp_path): def _write_catalog(tmp_path):
phys = PhysicalSchema( phys = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1), database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={ tables={
"dim_patient": TablePhysical( "dim_patient": TablePhysical(
comment="Anagrafica", comment="Anagrafica",
@@ -48,7 +48,7 @@ def _write_family_catalog(tmp_path):
) )
phys = PhysicalSchema( phys = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1), database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={ tables={
"fact_sost_impianto_pmk": _t("Sost PMK"), "fact_sost_impianto_pmk": _t("Sost PMK"),
"fact_sost_impianto_crt_d": _t("Sost CRT-D"), "fact_sost_impianto_crt_d": _t("Sost CRT-D"),
@@ -1,16 +1,16 @@
from datetime import datetime from datetime import UTC, datetime
from typer.testing import CliRunner from typer.testing import CliRunner
from tht.cli import app from tht.cli import app
from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
from tht.config import ExamplesConfig
from tht.cli.schema_cmd import _add_examples from tht.cli.schema_cmd import _add_examples
from tht.config import ExamplesConfig
from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
def _write_catalog(tmp_path): def _write_catalog(tmp_path):
phys = PhysicalSchema( phys = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1), database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={ tables={
"dim_patient": TablePhysical( "dim_patient": TablePhysical(
comment="Anagrafica", comment="Anagrafica",
@@ -48,7 +48,7 @@ def test_introspect_fresh_root_initializes_through_writer_job(tmp_path, monkeypa
cfg = _write_config(tmp_path) cfg = _write_config(tmp_path)
physical = PhysicalSchema( physical = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1), database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={"dim_patient": TablePhysical(columns={"id": ColumnPhysical(type="bigint")})}, tables={"dim_patient": TablePhysical(columns={"id": ColumnPhysical(type="bigint")})},
) )
@@ -90,7 +90,7 @@ def test_render_without_catalog_guides_fallback(tmp_path):
def test_examples_skip_one_unreadable_column_and_continue(caplog): def test_examples_skip_one_unreadable_column_and_continue(caplog):
physical = PhysicalSchema( physical = PhysicalSchema(
database="d", schema="s", introspected_at=datetime(2026, 1, 1), database="d", schema="s", introspected_at=datetime(2026, 1, 1, tzinfo=UTC),
tables={"t": TablePhysical(columns={ tables={"t": TablePhysical(columns={
"bad": ColumnPhysical(type="text"), "good": ColumnPhysical(type="text") "bad": ColumnPhysical(type="text"), "good": ColumnPhysical(type="text")
})}, })},
+11 -6
View File
@@ -1,20 +1,25 @@
"""Tests for `tht session documents --json` and build_documents.""" """Tests for `tht session documents --json` and build_documents."""
import json import json
from datetime import datetime from datetime import UTC, datetime
from typer.testing import CliRunner from typer.testing import CliRunner
from tht.cli.session_cmd import session_app
from tht.config import DatabaseConfig from tht.config import DatabaseConfig
from tht.decisions import DecisionRecord from tht.decisions import DecisionRecord
from tht.session.models import SessionSnapshot from tht.session.models import SessionSnapshot
from tht.session.store import build_documents, build_snapshot_documents, create_session, new_session_manifest from tht.session.store import (
from tht.cli.session_cmd import session_app build_documents,
build_snapshot_documents,
create_session,
new_session_manifest,
)
def _db(): def _db():
return DatabaseConfig( return DatabaseConfig(
database="testdb", user="u", password="p", # noqa: S106 database="testdb", user="u", password="p",
**{"schema": "public"}, schema="public",
) )
@@ -53,7 +58,7 @@ def test_build_documents_includes_existing_artifacts_only(tmp_path):
def _decision(seq: int, type_: str, subject: str, detail: str = "", rationale: str = ""): def _decision(seq: int, type_: str, subject: str, detail: str = "", rationale: str = ""):
return DecisionRecord( return DecisionRecord(
seq=seq, seq=seq,
ts=datetime(2026, 1, 1), ts=datetime(2026, 1, 1, tzinfo=UTC),
type=type_, type=type_,
subject=subject, subject=subject,
detail=detail, detail=detail,
+2 -2
View File
@@ -14,8 +14,8 @@ def _make_db():
return DatabaseConfig( return DatabaseConfig(
database="testdb", database="testdb",
user="testuser", user="testuser",
password="testpass", # noqa: S106 password="testpass",
**{"schema": "public"}, schema="public",
) )
+2 -2
View File
@@ -17,8 +17,8 @@ from tht.session.store import (
def _db(): def _db():
return DatabaseConfig( return DatabaseConfig(
database="testdb", user="u", password="p", # noqa: S106 database="testdb", user="u", password="p",
**{"schema": "public"}, schema="public",
) )
+4 -4
View File
@@ -8,15 +8,15 @@ import json
from typer.testing import CliRunner from typer.testing import CliRunner
from tht.config import DatabaseConfig
from tht.cli.session_cmd import session_app from tht.cli.session_cmd import session_app
from tht.config import DatabaseConfig
from tht.session.store import _extract_name, _summarize, load_session from tht.session.store import _extract_name, _summarize, load_session
def _db(): def _db():
return DatabaseConfig( return DatabaseConfig(
database="testdb", user="u", password="p", # noqa: S106 database="testdb", user="u", password="p",
**{"schema": "public"}, schema="public",
) )
@@ -41,7 +41,7 @@ def test_extract_name_is_a_3_to_5_word_italian_summary():
def test_extract_name_falls_back_to_summarize_on_failure(monkeypatch): def test_extract_name_falls_back_to_summarize_on_failure(monkeypatch):
import tht.session.store as store from tht.session import store
def boom(*a, **k): def boom(*a, **k):
raise RuntimeError("yake down") raise RuntimeError("yake down")
+2 -3
View File
@@ -7,12 +7,11 @@ from tht.decisions import DecisionInput
from tht.session.filesystem_repository import FilesystemSessionRepository from tht.session.filesystem_repository import FilesystemSessionRepository
from tht.session.models import PrincipalContext, SessionManifest, local_principal from tht.session.models import PrincipalContext, SessionManifest, local_principal
from tht.session.repository import build_session_repository, resolve_principal from tht.session.repository import build_session_repository, resolve_principal
from tht.session.store import SessionError from tht.session.store import SessionError, create_session
from tht.session.store import create_session
def _db() -> DatabaseConfig: def _db() -> DatabaseConfig:
return DatabaseConfig(database="testdb", schema="public", user="u", password="p") # noqa: S106 return DatabaseConfig(database="testdb", schema="public", user="u", password="p")
def _config(tmp_path): def _config(tmp_path):
@@ -1,7 +1,7 @@
import uuid import uuid
from tht.decisions import DecisionInput from tht.decisions import DecisionInput
from tht.phase import current_phase, cte_plan, next_cte from tht.phase import cte_plan, current_phase, next_cte
from tht.session.filesystem_repository import FilesystemSessionRepository from tht.session.filesystem_repository import FilesystemSessionRepository
from tht.session.models import PrincipalContext, SessionManifest from tht.session.models import PrincipalContext, SessionManifest
from tht.session.store import persist_verified_finalization from tht.session.store import persist_verified_finalization
+1 -1
View File
@@ -10,7 +10,7 @@ from tht.session.store import create_session, set_schema_linking
def _db(): def _db():
return DatabaseConfig(database="testdb", user="u", password="p", **{"schema": "public"}) # noqa: S106 return DatabaseConfig(database="testdb", user="u", password="p", schema="public")
def test_writes_and_revalidates(tmp_path): def test_writes_and_revalidates(tmp_path):
+1 -1
View File
@@ -9,7 +9,7 @@ from tht.session.store import create_session
def _db(): def _db():
return DatabaseConfig(database="testdb", user="u", password="p", **{"schema": "public"}) # noqa: S106 return DatabaseConfig(database="testdb", user="u", password="p", schema="public")
def _patch_cfg(monkeypatch, tmp_path): def _patch_cfg(monkeypatch, tmp_path):
-1
View File
@@ -6,7 +6,6 @@ Pure-logic tests (no DB needed):
""" """
import json import json
from tht.execute.limit import inject_limit_offset from tht.execute.limit import inject_limit_offset
+2 -2
View File
@@ -17,12 +17,12 @@ from tht.sqlcheck import validate_sql
def _schema() -> PhysicalSchema: def _schema() -> PhysicalSchema:
"""A tiny known schema for the object-existence checks.""" """A tiny known schema for the object-existence checks."""
from datetime import datetime from datetime import UTC, datetime
return PhysicalSchema( return PhysicalSchema(
database="db", database="db",
schema="dw", schema="dw",
introspected_at=datetime(2025, 1, 1), introspected_at=datetime(2025, 1, 1, tzinfo=UTC),
tables={ tables={
"dim_pazienti": TablePhysical( "dim_pazienti": TablePhysical(
columns={ columns={
+2 -2
View File
@@ -8,8 +8,8 @@ from tht.session.store import create_session, set_schema_linking, sync_schema_li
def _db(): def _db():
return DatabaseConfig( return DatabaseConfig(
database="testdb", user="u", password="p", # noqa: S106 database="testdb", user="u", password="p",
**{"schema": "public"}, schema="public",
) )
+2 -1
View File
@@ -64,7 +64,8 @@ def test_empty_hits_returns_empty():
def test_value_grounded_decision_type_exists(): def test_value_grounded_decision_type_exists():
# D14a adds the value_grounded decision type so the gate can record the # D14a adds the value_grounded decision type so the gate can record the
# reviewer's choice of which column(s) anchor a cited value. # reviewer's choice of which column(s) anchor a cited value.
from tht.decisions import DecisionType
import typing import typing
from tht.decisions import DecisionType
args = typing.get_args(DecisionType) args = typing.get_args(DecisionType)
assert "value_grounded" in args assert "value_grounded" in args
@@ -1,15 +1,15 @@
"""Executable baseline for persisted workflow behavior touched by the refactor.""" """Executable baseline for persisted workflow behavior touched by the refactor."""
import hashlib
from dataclasses import asdict from dataclasses import asdict
from datetime import UTC, datetime from datetime import UTC, datetime
import hashlib
import pytest import pytest
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.decisions import DecisionRecord, append_decision from tht.decisions import DecisionRecord, append_decision
from tht.evidence import project_session from tht.evidence import project_session
from tht.evidence.corpus.models import CanonicalDocument, CorpusManifest
from tht.evidence.corpus.store import CorpusStore
from tht.phase import current_phase, effective_decisions from tht.phase import current_phase, effective_decisions
from tht.session.models import Candidate, SchemaLinking from tht.session.models import Candidate, SchemaLinking
from tht.workflow import load_workflow from tht.workflow import load_workflow
+1 -1
View File
@@ -1,8 +1,8 @@
"""Direct PostgreSQL implementation of the DWH port.""" """Direct PostgreSQL implementation of the DWH port."""
from tht.config import DatabaseConfig
from sqlalchemy.exc import OperationalError, SQLAlchemyError from sqlalchemy.exc import OperationalError, SQLAlchemyError
from tht.config import DatabaseConfig
from tht.db import execute, sampling from tht.db import execute, sampling
from tht.db.connection import can_create_in_schema, make_engine, ping, writable_tables from tht.db.connection import can_create_in_schema, make_engine, ping, writable_tables
from tht.db.introspect import introspect from tht.db.introspect import introspect
+1 -1
View File
@@ -1,8 +1,8 @@
"""Thoth/PostgREST implementation of the DWH port.""" """Thoth/PostgREST implementation of the DWH port."""
from tht.config import DatabaseIdentityConfig, RestConfig from tht.config import DatabaseIdentityConfig, RestConfig
from tht.db.introspect import introspect_rest
from tht.db import sampling from tht.db import sampling
from tht.db.introspect import introspect_rest
from tht.execute import ExecResult, ExecutionError, PlanSummary from tht.execute import ExecResult, ExecutionError, PlanSummary
from tht.mschema.models import PhysicalSchema from tht.mschema.models import PhysicalSchema
from tht.ports.dwh import DistinctValues, DwhCapabilities, DwhHealth from tht.ports.dwh import DistinctValues, DwhCapabilities, DwhHealth
+1
View File
@@ -1,6 +1,7 @@
from pathlib import Path from pathlib import Path
import typer import typer
from tht.adapters.factory import build_dwh from tht.adapters.factory import build_dwh
from tht.cli.config_cmd import CONFIG_OPT from tht.cli.config_cmd import CONFIG_OPT
from tht.config import ConfigError, load_config from tht.config import ConfigError, load_config
+1 -1
View File
@@ -16,7 +16,7 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
continue continue
try: try:
distinct = dwh.distinct_values(table_name, column_name, limit=limit) distinct = dwh.distinct_values(table_name, column_name, limit=limit)
except Exception as exc: except Exception as exc: # noqa: BLE001 - an unreadable DWH column is non-fatal
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}")) skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
continue continue
vals = [str(value) for value in distinct.values if value not in (None, "")] vals = [str(value) for value in distinct.values if value not in (None, "")]
+1 -1
View File
@@ -459,8 +459,8 @@ def solved_search_cmd(
from rich.table import Table from rich.table import Table
from tht.cli.vector_cmd import make_embedder, open_searcher from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.memory import search_solved_questions from tht.memory import search_solved_questions
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config) cfg = _load_config_or_exit(config)
+3 -4
View File
@@ -12,8 +12,8 @@ import json
import typer import typer
from tht.phase import ( from tht.phase import (
auto_advance_eligible,
advance_problems, advance_problems,
auto_advance_eligible,
current_phase, current_phase,
) )
from tht.workflow import load_workflow from tht.workflow import load_workflow
@@ -93,8 +93,7 @@ def advance_cmd(
if cur > wf.max_phase: if cur > wf.max_phase:
typer.secho("Sessione già alla fase terminale.", fg=typer.colors.YELLOW) typer.secho("Sessione già alla fase terminale.", fg=typer.colors.YELLOW)
raise typer.Exit(0) raise typer.Exit(0)
if auto: if auto and not auto_advance_eligible(snapshot):
if not auto_advance_eligible(snapshot):
problems = advance_problems(snapshot, cur) problems = advance_problems(snapshot, cur)
for p in problems: for p in problems:
typer.echo(p) typer.echo(p)
@@ -164,7 +163,7 @@ def _cfg():
ws = os.environ.get("THT_WORKSPACE") or os.environ.get("THT_CONFIG") ws = os.environ.get("THT_WORKSPACE") or os.environ.get("THT_CONFIG")
config_path = Path(ws) if ws else Path("config/tht.yaml") config_path = Path(ws) if ws else Path("config/tht.yaml")
try: try:
from tht.cli.schema_cmd import _load_config_or_exit # noqa: F401 (portato in Onda 1.4) from tht.cli.schema_cmd import _load_config_or_exit
return _load_config_or_exit(config_path) return _load_config_or_exit(config_path)
except ImportError: except ImportError:
+2 -2
View File
@@ -96,9 +96,9 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
from tht.adapters.factory import build_vector_store from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder from tht.cli.vector_cmd import make_embedder
from tht.evidence import build_preprocessing_pipeline, build_sources
from tht.evidence.corpus.chunk import ChunkPolicy from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.store import CorpusStore from tht.evidence.corpus.store import CorpusStore
from tht.evidence import build_preprocessing_pipeline, build_sources
cfg = _load_config_or_exit(config) cfg = _load_config_or_exit(config)
if cfg.embeddings is None: if cfg.embeddings is None:
@@ -130,9 +130,9 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
from tht.adapters.factory import build_vector_store from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder from tht.cli.vector_cmd import make_embedder
from tht.evidence import build_preprocessing_pipeline, build_sources
from tht.evidence.corpus.chunk import ChunkPolicy from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.store import CorpusStore from tht.evidence.corpus.store import CorpusStore
from tht.evidence import build_preprocessing_pipeline, build_sources
cfg = _load_config_or_exit(config) cfg = _load_config_or_exit(config)
if cfg.embeddings is None: if cfg.embeddings is None:
+5 -5
View File
@@ -303,7 +303,7 @@ def _suggest_fk_result(physical, annotations, *, sql_inputs: list[tuple[str, str
@schema_app.command("check") @schema_app.command("check")
def check_cmd( def check_cmd(
config: Path = CONFIG_OPT, config: Path = CONFIG_OPT,
annotations: Path | None = typer.Option(None, "--annotations"), # noqa: B008 annotations: Path | None = typer.Option(None, "--annotations"),
reviewed_candidates: str | None = typer.Option(None, "--reviewed-candidates"), reviewed_candidates: str | None = typer.Option(None, "--reviewed-candidates"),
json_output: bool = typer.Option(False, "--json"), json_output: bool = typer.Option(False, "--json"),
) -> None: ) -> None:
@@ -411,11 +411,11 @@ _GENERIC_PK_NAMES = {"id", "key", "code"}
@schema_app.command("suggest-fks") @schema_app.command("suggest-fks")
def suggest_fks_cmd( def suggest_fks_cmd(
config: Path = CONFIG_OPT, config: Path = CONFIG_OPT,
from_sql: list[Path] = typer.Option( # noqa: B008 from_sql: list[Path] = typer.Option(
None, "--from-sql", None, "--from-sql",
help="Directory o file .sql approvati da cui minare i join reali (ripetibile).", help="Directory o file .sql approvati da cui minare i join reali (ripetibile).",
), ),
assume: list[str] = typer.Option( # noqa: B008 assume: list[str] = typer.Option(
None, "--assume", None, "--assume",
help="Disambigua una PK con piu' proprietari: col=tabella_ref " help="Disambigua una PK con piu' proprietari: col=tabella_ref "
"(es. cod_paz=dim_patient). Ripetibile.", "(es. cod_paz=dim_patient). Ripetibile.",
@@ -548,10 +548,10 @@ def render_cmd(
format: str = typer.Option( format: str = typer.Option(
"markdown", "--format", "-f", help="Formato: markdown | mschema-text | schema-dict" "markdown", "--format", "-f", help="Formato: markdown | mschema-text | schema-dict"
), ),
tables: list[str] = typer.Option( # noqa: B008 tables: list[str] = typer.Option(
None, "--table", "-t", help="Limita alle tabelle indicate (ripetibile)." None, "--table", "-t", help="Limita alle tabelle indicate (ripetibile)."
), ),
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."), # noqa: B008 output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
) -> None: ) -> None:
"""Serializza mschema (physical + annotations) nel formato richiesto.""" """Serializza mschema (physical + annotations) nel formato richiesto."""
import json import json
+1 -1
View File
@@ -258,9 +258,9 @@ def pack_cmd(
build_retrieval_entries, build_retrieval_entries,
validate_corpus_workspace, validate_corpus_workspace,
) )
from tht.memory import SOLVED_KIND
from tht.ports.vector import VectorReadUnavailable, VectorStoreError from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.search import combined_search, schema_tables from tht.search import combined_search, schema_tables
from tht.memory import SOLVED_KIND
from tht.vectorstore.embeddings import EmbeddingsError from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config) cfg = _load_config_or_exit(config)
+3 -3
View File
@@ -515,12 +515,12 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
promoted_tables_for, promoted_tables_for,
) )
from tht.ctetest import CteError, CteTestRecord, _iter_json_objects from tht.ctetest import CteError, CteTestRecord, _iter_json_objects
from tht.evidence import project_session
from tht.execute import ExecutionError from tht.execute import ExecutionError
from tht.execute.warnings import plan_warnings, runtime_warnings, static_warnings from tht.execute.warnings import plan_warnings, runtime_warnings, static_warnings
from tht.report import extract_reviewer_notes, render_validation_report
from tht.evidence import project_session
from tht.phase import cte_plan as effective_cte_plan from tht.phase import cte_plan as effective_cte_plan
from tht.phase import effective_decisions from tht.phase import effective_decisions
from tht.report import extract_reviewer_notes, render_validation_report
from tht.session.models import SchemaLinking from tht.session.models import SchemaLinking
from tht.sqlcheck import validate_sql from tht.sqlcheck import validate_sql
@@ -656,7 +656,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).", "Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
fg=typer.colors.CYAN, fg=typer.colors.CYAN,
) )
except Exception as e: except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
typer.secho( typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). " f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
f"Recupera con `tht memory solved-index {session_id}`.", f"Recupera con `tht memory solved-index {session_id}`.",
+3 -1
View File
@@ -5,10 +5,12 @@ from sqlalchemy import Engine
from tht.execute import ( from tht.execute import (
ExecResult, ExecResult,
PlanSummary, PlanSummary,
explain as _explain,
require_positive_int, require_positive_int,
run_controlled, run_controlled,
) )
from tht.execute import (
explain as _explain,
)
DEFAULT_TIMEOUT_MS = 30_000 DEFAULT_TIMEOUT_MS = 30_000
+4 -2
View File
@@ -45,8 +45,10 @@ def fetch_chain_pem(host: str, port: int = 443, timeout: int = 30) -> list[str]:
ctx.check_hostname = False ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE ctx.verify_mode = ssl.CERT_NONE
try: try:
with socket.create_connection((host, port), timeout=timeout) as sock: with (
with ctx.wrap_socket(sock, server_hostname=host) as tls: socket.create_connection((host, port), timeout=timeout) as sock,
ctx.wrap_socket(sock, server_hostname=host) as tls,
):
certs = _unverified_chain(tls) certs = _unverified_chain(tls)
except (OSError, ssl.SSLError) as e: except (OSError, ssl.SSLError) as e:
raise CaFetchError( raise CaFetchError(
+3 -3
View File
@@ -92,7 +92,7 @@ def add_examples(engine: Engine, physical: PhysicalSchema, cfg: ExamplesConfig)
''') ''')
try: try:
rows = conn.execute(q, {"lim": cfg.max_per_column}).fetchall() rows = conn.execute(q, {"lim": cfg.max_per_column}).fetchall()
except Exception as e: # colonna non leggibile: si salta, non si interrompe except Exception as e: # noqa: BLE001 - skip any unreadable DWH column
logger.warning("Campionamento saltato per %s.%s: %s", table_name, column_name, e) logger.warning("Campionamento saltato per %s.%s: %s", table_name, column_name, e)
continue continue
column.examples = [str(r[0]) for r in rows] column.examples = [str(r[0]) for r in rows]
@@ -164,7 +164,7 @@ def unique_values_for_lsh(
''') ''')
try: try:
rows = conn.execute(q, {"lim": cfg.max_values_per_column}).fetchall() rows = conn.execute(q, {"lim": cfg.max_values_per_column}).fetchall()
except Exception as e: except Exception as e: # noqa: BLE001 - skip any unreadable DWH column
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}")) skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
continue continue
vals = [str(r[0]) for r in rows] vals = [str(r[0]) for r in rows]
@@ -205,7 +205,7 @@ def unique_values_for_lsh_rest(
rows = client.top_values( rows = client.top_values(
schema, table_name, column_name, cfg.max_values_per_column schema, table_name, column_name, cfg.max_values_per_column
) )
except Exception as e: except Exception as e: # noqa: BLE001 - skip any unreadable REST column
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}")) skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
continue continue
vals = [str(r["value"]) for r in rows if r["value"] not in (None, "")] vals = [str(r["value"]) for r in rows if r["value"] not in (None, "")]
+3 -4
View File
@@ -24,16 +24,15 @@ from tht.evidence.search import (
from tht.evidence.session import project_session from tht.evidence.session import project_session
from tht.evidence.sources import build_sources from tht.evidence.sources import build_sources
__all__ = [ __all__ = [
"AcquiredDocument", "AcquiredDocument",
"ActiveEvidenceSearcher",
"CorpusWorkspaceMismatchError",
"EvidenceEmbedder",
"EvidenceSource", "EvidenceSource",
"EvidenceSourceError", "EvidenceSourceError",
"EvidenceSourceErrorCategory", "EvidenceSourceErrorCategory",
"EvidenceEmbedder",
"SourceObject", "SourceObject",
"ActiveEvidenceSearcher",
"CorpusWorkspaceMismatchError",
"acquire", "acquire",
"active_searcher", "active_searcher",
"build_preprocessing_pipeline", "build_preprocessing_pipeline",
+4 -1
View File
@@ -7,7 +7,10 @@ from datetime import UTC, datetime
from urllib.parse import quote, urlsplit from urllib.parse import quote, urlsplit
from tht.evidence.contracts import ( from tht.evidence.contracts import (
AcquiredDocument, EvidenceSourceError, EvidenceSourceErrorCategory, SourceObject, AcquiredDocument,
EvidenceSourceError,
EvidenceSourceErrorCategory,
SourceObject,
) )
-1
View File
@@ -15,7 +15,6 @@ from tht.evidence.contracts import (
validate_safe_metadata, validate_safe_metadata,
) )
_NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$") _NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$")
_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$") _SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$")
+1 -2
View File
@@ -10,9 +10,8 @@ from pydantic import JsonValue, TypeAdapter, ValidationError
from yaml.events import AliasEvent from yaml.events import AliasEvent
from yaml.nodes import MappingNode from yaml.nodes import MappingNode
from tht.evidence.corpus.models import CanonicalDocument
from tht.evidence.contracts import AcquiredDocument, canonical_provenance_uri from tht.evidence.contracts import AcquiredDocument, canonical_provenance_uri
from tht.evidence.corpus.models import CanonicalDocument
MAX_DOCUMENT_BYTES = 10 * 1024 * 1024 MAX_DOCUMENT_BYTES = 10 * 1024 * 1024
_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE) _CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE)
+13 -11
View File
@@ -4,6 +4,7 @@ from __future__ import annotations
import hashlib import hashlib
import json import json
import logging
import re import re
import uuid import uuid
from collections.abc import Mapping, Sequence from collections.abc import Mapping, Sequence
@@ -11,17 +12,16 @@ from dataclasses import asdict, dataclass, field
from datetime import UTC from datetime import UTC
from pathlib import Path from pathlib import Path
import tht.evidence.acquisition as evidence_acquisition
from tht.evidence.contracts import EvidenceSource, SourceObject, canonical_provenance_uri
from tht.evidence.corpus.chunk import ChunkPolicy, chunk from tht.evidence.corpus.chunk import ChunkPolicy, chunk
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.corpus.normalize import normalize from tht.evidence.corpus.normalize import normalize
from tht.evidence.corpus.store import CorpusStore from tht.evidence.corpus.store import CorpusStore
import tht.evidence.acquisition as evidence_acquisition
from tht.evidence.contracts import EvidenceSource, SourceObject, canonical_provenance_uri
from tht.ports.vector import VectorStore, VectorWriteRecord
from tht.vectorstore.records import VectorRecord
from tht.jobs.models import JobSpec from tht.jobs.models import JobSpec
from tht.jobs.runner import JobContext, StageArtifacts, run_job, seal_stage_artifacts from tht.jobs.runner import JobContext, StageArtifacts, run_job, seal_stage_artifacts
from tht.ports.vector import VectorStore, VectorWriteRecord
from tht.vectorstore.records import VectorRecord
EVIDENCE_STAGE_IDS = ( EVIDENCE_STAGE_IDS = (
"discover", "discover",
@@ -33,6 +33,8 @@ EVIDENCE_STAGE_IDS = (
"retention_cleanup", "retention_cleanup",
) )
logger = logging.getLogger(__name__)
class PipelineError(RuntimeError): class PipelineError(RuntimeError):
"""Credential-free failure at the preprocessing boundary.""" """Credential-free failure at the preprocessing boundary."""
@@ -212,14 +214,14 @@ class CorpusPipeline:
if purge_vector: if purge_vector:
try: try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id) self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception: except Exception: # noqa: BLE001 - retention reports per-generation failures
failures.append({"generation": generation, "error": "vector cleanup failed"}) failures.append({"generation": generation, "error": "vector cleanup failed"})
continue continue
try: try:
if purge_filesystem: if purge_filesystem:
self.store.discard(generation) self.store.discard(generation)
evicted.append(generation) evicted.append(generation)
except Exception: except Exception: # noqa: BLE001 - retention reports per-generation failures
failures.append({"generation": generation, "error": "filesystem cleanup failed"}) failures.append({"generation": generation, "error": "filesystem cleanup failed"})
return {"status": "partial" if failures else "succeeded", "dry_run": dry_run, return {"status": "partial" if failures else "succeeded", "dry_run": dry_run,
"active_generation": self.store.active_generation(), "evicted": evicted, "active_generation": self.store.active_generation(), "evicted": evicted,
@@ -378,7 +380,7 @@ class CorpusPipeline:
try: try:
active_assets_valid = active_assets_are_valid(previous) active_assets_valid = active_assets_are_valid(previous)
except Exception: except Exception: # noqa: BLE001 - any corrupt active asset disables reuse
active_assets_valid = False active_assets_valid = False
reusable = ( reusable = (
active_assets_valid active_assets_valid
@@ -538,7 +540,7 @@ class CorpusPipeline:
try: try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id) self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception: except Exception:
pass logger.debug("Failed to clean the compensated vector generation", exc_info=True)
write(context, "compensated.json", {"generation": generation}) write(context, "compensated.json", {"generation": generation})
def rotate_compensated_generation(context: JobContext) -> None: def rotate_compensated_generation(context: JobContext) -> None:
@@ -786,12 +788,12 @@ class CorpusPipeline:
try: try:
self.store.discard(generation) self.store.discard(generation)
except Exception: except Exception:
pass logger.debug("Failed to discard the unpublished evidence generation", exc_info=True)
if vector_written: if vector_written:
try: try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id) self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception: except Exception:
pass logger.debug("Failed to delete the unpublished vector generation", exc_info=True)
@staticmethod @staticmethod
def _vector_record( def _vector_record(
+5 -6
View File
@@ -2,22 +2,21 @@
from __future__ import annotations from __future__ import annotations
import json
import fcntl import fcntl
import hashlib
import json
import os import os
import re import re
import stat
import shutil import shutil
import uuid import stat
import hashlib
import threading import threading
import uuid
from contextlib import contextmanager
from datetime import UTC, datetime from datetime import UTC, datetime
from pathlib import Path from pathlib import Path
from contextlib import contextmanager
from tht.evidence.corpus.models import CorpusManifest from tht.evidence.corpus.models import CorpusManifest
_GENERATION = re.compile(r"^gen:[0-9a-f]{32}$") _GENERATION = re.compile(r"^gen:[0-9a-f]{32}$")
+2 -2
View File
@@ -46,7 +46,7 @@ class ConceptFormula(BaseModel):
return f"---\n{fm}---\n{self.sql}\n" return f"---\n{fm}---\n{self.sql}\n"
@classmethod @classmethod
def parse(cls, text: str) -> "ConceptFormula": def parse(cls, text: str) -> ConceptFormula:
if not text.startswith("---\n"): if not text.startswith("---\n"):
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)") raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
try: try:
@@ -55,7 +55,7 @@ class ConceptFormula(BaseModel):
raise ValueError("frontmatter malformato") from e raise ValueError("frontmatter malformato") from e
meta = yaml.safe_load(fm) meta = yaml.safe_load(fm)
if not isinstance(meta, dict): if not isinstance(meta, dict):
raise ValueError("frontmatter non valido") raise TypeError("frontmatter non valido")
return cls.model_validate({**meta, "sql": body.strip("\n")}) return cls.model_validate({**meta, "sql": body.strip("\n")})
+1 -1
View File
@@ -2,10 +2,10 @@
from typing import Protocol from typing import Protocol
from tht.evidence.contracts import EvidenceSource
from tht.evidence.corpus.chunk import ChunkPolicy from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.pipeline import CorpusPipeline from tht.evidence.corpus.pipeline import CorpusPipeline
from tht.evidence.corpus.store import CorpusStore from tht.evidence.corpus.store import CorpusStore
from tht.evidence.contracts import EvidenceSource
from tht.ports.vector import VectorStore from tht.ports.vector import VectorStore
+3 -2
View File
@@ -9,6 +9,7 @@ import re
import stat import stat
from pathlib import Path from pathlib import Path
from types import TracebackType from types import TracebackType
from typing import Self
class JobAlreadyRunningError(RuntimeError): class JobAlreadyRunningError(RuntimeError):
@@ -34,7 +35,7 @@ class WorkspaceJobLock:
) )
self._fd: int | None = None self._fd: int | None = None
def acquire(self) -> "WorkspaceJobLock": def acquire(self) -> WorkspaceJobLock:
if self._fd is not None: if self._fd is not None:
raise RuntimeError("job lock is already held by this object") raise RuntimeError("job lock is already held by this object")
root_fd = os.open(self.path.parents[2], os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) root_fd = os.open(self.path.parents[2], os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
@@ -85,7 +86,7 @@ class WorkspaceJobLock:
finally: finally:
os.close(fd) os.close(fd)
def __enter__(self) -> "WorkspaceJobLock": def __enter__(self) -> Self:
return self.acquire() return self.acquire()
def __exit__( def __exit__(
+11 -5
View File
@@ -7,8 +7,14 @@ from datetime import UTC, datetime
from pathlib import Path from pathlib import Path
from typing import Literal, Self from typing import Literal, Self
from pydantic import BaseModel, ConfigDict, Field, field_serializer, field_validator, model_validator from pydantic import (
BaseModel,
ConfigDict,
Field,
field_serializer,
field_validator,
model_validator,
)
_JOB_KEY = re.compile(r"^[a-z][a-z0-9_-]{0,63}$") _JOB_KEY = re.compile(r"^[a-z][a-z0-9_-]{0,63}$")
_RUN_ID = re.compile(r"^[0-9a-f]{32}$") _RUN_ID = re.compile(r"^[0-9a-f]{32}$")
@@ -95,7 +101,7 @@ class JobSpec(_FrozenModel):
data.update(update) data.update(update)
return type(self).model_validate(data) return type(self).model_validate(data)
def with_resume(self, run_id: str) -> "JobSpec": def with_resume(self, run_id: str) -> JobSpec:
return self.model_copy(update={"resume_run_id": run_id}) return self.model_copy(update={"resume_run_id": run_id})
@@ -121,7 +127,7 @@ class StageRun(_FrozenModel):
) )
@model_validator(mode="after") @model_validator(mode="after")
def state_shape(self) -> "StageRun": def state_shape(self) -> StageRun:
if self.status == "pending" and any( if self.status == "pending" and any(
value is not None for value in ( value is not None for value in (
self.started_at, self.finished_at, self.error, self.effect_state, self.started_at, self.finished_at, self.error, self.effect_state,
@@ -179,7 +185,7 @@ class JobRun(_FrozenModel):
_resumed_from = field_validator("resumed_from")(_validate_run_id) _resumed_from = field_validator("resumed_from")(_validate_run_id)
@model_validator(mode="after") @model_validator(mode="after")
def ledger_shape(self) -> "JobRun": def ledger_shape(self) -> JobRun:
names = [stage.name for stage in self.stages] names = [stage.name for stage in self.stages]
if len(names) != len(set(names)): if len(names) != len(set(names)):
raise ValueError("stage identifiers must be unique") raise ValueError("stage identifiers must be unique")
+4 -4
View File
@@ -2,12 +2,12 @@
from __future__ import annotations from __future__ import annotations
import json
import hashlib import hashlib
import json
import os import os
import uuid
import stat
import shutil import shutil
import stat
import uuid
from collections.abc import Callable, Sequence from collections.abc import Callable, Sequence
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path from pathlib import Path
@@ -325,7 +325,7 @@ def run_job(
_persist(checkpoint_path, run) _persist(checkpoint_path, run)
try: try:
stage_result = stage_callable(context) stage_result = stage_callable(context)
except Exception: except Exception: # noqa: BLE001 - stage failures are persisted as terminal reports
failed = stage.model_copy( failed = stage.model_copy(
update={ update={
"status": "failed", "status": "failed",
+1 -1
View File
@@ -122,7 +122,7 @@ def index_solved_question_best_effort(
store=store_factory(), store=store_factory(),
embedder=embedder_factory(), embedder=embedder_factory(),
) )
except Exception as error: except Exception as error: # noqa: BLE001 - callers receive a best-effort outcome
return SolvedIndexOutcome(upserted=None, error=str(error)) return SolvedIndexOutcome(upserted=None, error=str(error))
return SolvedIndexOutcome(upserted=upserted) return SolvedIndexOutcome(upserted=upserted)
+2 -2
View File
@@ -115,8 +115,8 @@ def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None
lines = [ lines = [
f"# Schema {physical.db_schema} ({physical.database})", f"# Schema {physical.db_schema} ({physical.database})",
"", "",
f"Introspezione: {physical.introspected_at.isoformat()} — " (f"Introspezione: {physical.introspected_at.isoformat()} — "
f"{len(physical.tables)} tabelle", f"{len(physical.tables)} tabelle"),
] ]
for table_name, table in physical.tables.items(): for table_name, table in physical.tables.items():
lines += ["", f"## {table_name}", ""] lines += ["", f"## {table_name}", ""]
+1 -2
View File
@@ -1,9 +1,8 @@
"""Deterministic builder for the single Pi-facing Thoth session skill.""" """Deterministic builder for the single Pi-facing Thoth session skill."""
import argparse import argparse
from pathlib import Path
import sys import sys
from pathlib import Path
HARNESS_ROOT = Path(__file__).resolve().parents[1] HARNESS_ROOT = Path(__file__).resolve().parents[1]
SKILL_ROOT = HARNESS_ROOT / ".pi" / "skills" / "tht-sessione" SKILL_ROOT = HARNESS_ROOT / ".pi" / "skills" / "tht-sessione"
+4 -4
View File
@@ -1,18 +1,18 @@
"""Stable interfaces implemented by Thoth infrastructure adapters.""" """Stable interfaces implemented by Thoth infrastructure adapters."""
from tht.ports.dwh import ( from tht.ports.dwh import (
DistinctValues,
DwhAdapter, DwhAdapter,
DwhCapabilities, DwhCapabilities,
DwhHealth, DwhHealth,
DistinctValues,
UnsupportedCapability, UnsupportedCapability,
) )
from tht.ports.vector import ( from tht.ports.vector import (
VectorCapabilities, VectorCapabilities,
VectorHealth, VectorHealth,
VectorHit, VectorHit,
VectorRecord,
VectorReadUnavailable, VectorReadUnavailable,
VectorRecord,
VectorStore, VectorStore,
VectorStoreError, VectorStoreError,
VectorWriteRecord, VectorWriteRecord,
@@ -20,16 +20,16 @@ from tht.ports.vector import (
) )
__all__ = [ __all__ = [
"DistinctValues",
"DwhAdapter", "DwhAdapter",
"DwhCapabilities", "DwhCapabilities",
"DwhHealth", "DwhHealth",
"DistinctValues",
"UnsupportedCapability", "UnsupportedCapability",
"VectorCapabilities", "VectorCapabilities",
"VectorHealth", "VectorHealth",
"VectorHit", "VectorHit",
"VectorRecord",
"VectorReadUnavailable", "VectorReadUnavailable",
"VectorRecord",
"VectorStore", "VectorStore",
"VectorStoreError", "VectorStoreError",
"VectorWriteRecord", "VectorWriteRecord",
+1 -1
View File
@@ -40,7 +40,7 @@ class RestClient:
try: try:
body = resp.json() body = resp.json()
detail = body.get("message") or body.get("details") or resp.text detail = body.get("message") or body.get("details") or resp.text
except Exception: except (requests.exceptions.JSONDecodeError, AttributeError, TypeError):
detail = resp.text detail = resp.text
return f"DWH REST rpc {fn} → HTTP {resp.status_code}: {detail}" return f"DWH REST rpc {fn} → HTTP {resp.status_code}: {detail}"
+2 -2
View File
@@ -9,8 +9,8 @@ import re
import shutil import shutil
import tempfile import tempfile
import uuid import uuid
from collections.abc import Sequence
from pathlib import Path from pathlib import Path
from typing import Sequence
import portalocker import portalocker
import yaml import yaml
@@ -147,7 +147,7 @@ class FilesystemSessionRepository:
return {} return {}
data = json.loads(path.read_text()) data = json.loads(path.read_text())
if not isinstance(data, dict): if not isinstance(data, dict):
raise ValueError(f"Invalid preferences: {path}") raise TypeError(f"Invalid preferences: {path}")
return data return data
def set_preferences(self, preferences: dict) -> None: def set_preferences(self, preferences: dict) -> None:
+1 -1
View File
@@ -8,7 +8,7 @@ from typing import Literal, Self
import portalocker import portalocker
import yaml import yaml
from pydantic import BaseModel, Field, ConfigDict from pydantic import BaseModel, ConfigDict, Field
from tht.decisions import DecisionRecord from tht.decisions import DecisionRecord
+2 -2
View File
@@ -6,13 +6,13 @@ import hashlib
import json import json
import re import re
import uuid import uuid
from collections.abc import Iterator, Sequence
from contextlib import contextmanager from contextlib import contextmanager
from dataclasses import dataclass from dataclasses import dataclass
from datetime import UTC, datetime from datetime import UTC, datetime
from importlib.resources import files from importlib.resources import files
from importlib.resources.abc import Traversable from importlib.resources.abc import Traversable
from pathlib import Path from pathlib import Path
from typing import Iterator, Sequence
from sqlalchemy import Engine, create_engine, text from sqlalchemy import Engine, create_engine, text
from sqlalchemy.engine import URL, make_url from sqlalchemy.engine import URL, make_url
@@ -204,7 +204,7 @@ class PostgresSessionRepository:
self._runtime_role = runtime_role self._runtime_role = runtime_role
@classmethod @classmethod
def from_config(cls, config, principal: PrincipalContext) -> "PostgresSessionRepository": def from_config(cls, config, principal: PrincipalContext) -> PostgresSessionRepository:
query = {"sslmode": config.sslmode} query = {"sslmode": config.sslmode}
if config.sslrootcert is not None: if config.sslrootcert is not None:
query["sslrootcert"] = str(config.sslrootcert) query["sslrootcert"] = str(config.sslrootcert)
+2 -1
View File
@@ -3,7 +3,8 @@
from __future__ import annotations from __future__ import annotations
import os import os
from typing import Protocol, Sequence from collections.abc import Sequence
from typing import Protocol
from tht.decisions import DecisionInput, DecisionRecord from tht.decisions import DecisionInput, DecisionRecord
from tht.session.models import PrincipalContext, SessionManifest, SessionSnapshot from tht.session.models import PrincipalContext, SessionManifest, SessionSnapshot
+2 -2
View File
@@ -55,7 +55,7 @@ def _extract_name(question: str) -> str:
try: try:
extractor = yake.KeywordExtractor(lan="it", n=1, top=8, dedupLim=0.9) extractor = yake.KeywordExtractor(lan="it", n=1, top=8, dedupLim=0.9)
ranked = [k for k, _ in extractor.extract_keywords(q)] ranked = [k for k, _ in extractor.extract_keywords(q)]
except Exception: except Exception: # noqa: BLE001 - keyword extraction has a deterministic fallback
return _summarize(question) return _summarize(question)
seen: set[str] = set() seen: set[str] = set()
picked: list[str] = [] picked: list[str] = []
@@ -117,7 +117,7 @@ def create_session(
from tht.workflow import load_workflow from tht.workflow import load_workflow
schema_version = load_workflow().schema_version schema_version = load_workflow().schema_version
except Exception: except Exception: # noqa: BLE001 - legacy sessions may predate workflow metadata
schema_version = None schema_version = None
manifest = SessionManifest( manifest = SessionManifest(
id=session_id, created_at=now, question=question, id=session_id, created_at=now, question=question,
+1 -1
View File
@@ -87,7 +87,7 @@ def generate_task_doc(
wf = load_workflow() wf = load_workflow()
name = wf.phase_name(phase) name = wf.phase_name(phase)
header = f"## Task: fase {phase} ({name})" header = f"## Task: fase {phase} ({name})"
except Exception: except Exception: # noqa: BLE001 - task documents retain a phase-only fallback
header = f"## Task: fase {phase}" header = f"## Task: fase {phase}"
parts.append(header) parts.append(header)
+7 -6
View File
@@ -4,11 +4,12 @@
"""Core LSH (MinHash) per la ricerca di valori simili nei campi del database.""" """Core LSH (MinHash) per la ricerca di valori simili nei campi del database."""
import logging import logging
from typing import Dict, List, Tuple
from datasketch import MinHash, MinHashLSH from datasketch import MinHash, MinHashLSH
from tqdm import tqdm from tqdm import tqdm
logger = logging.getLogger(__name__)
def create_minhash(signature_size: int, string: str, n_gram: int) -> MinHash: def create_minhash(signature_size: int, string: str, n_gram: int) -> MinHash:
m = MinHash(num_perm=signature_size) m = MinHash(num_perm=signature_size)
@@ -33,7 +34,7 @@ NAME_LIKE_TOKENS: tuple[str, ...] = (
def skip_column( def skip_column(
column_name: str, column_name: str,
column_values: List[str], column_values: list[str],
max_total_chars: int = 50000, max_total_chars: int = 50000,
max_avg_length: int = 20, max_avg_length: int = 20,
name_tokens: tuple[str, ...] = NAME_LIKE_TOKENS, name_tokens: tuple[str, ...] = NAME_LIKE_TOKENS,
@@ -51,20 +52,20 @@ def jaccard_similarity(m1: MinHash, m2: MinHash) -> float:
def create_lsh_index( def create_lsh_index(
unique_values: Dict[str, Dict[str, List[str]]], unique_values: dict[str, dict[str, list[str]]],
signature_size: int, signature_size: int,
n_gram: int, n_gram: int,
threshold: float, threshold: float,
verbose: bool = True, verbose: bool = True,
) -> Tuple[MinHashLSH, Dict[str, Tuple[MinHash, str, str, str]]]: ) -> tuple[MinHashLSH, dict[str, tuple[MinHash, str, str, str]]]:
lsh = MinHashLSH(threshold=threshold, num_perm=signature_size) lsh = MinHashLSH(threshold=threshold, num_perm=signature_size)
minhashes: Dict[str, Tuple[MinHash, str, str, str]] = {} minhashes: dict[str, tuple[MinHash, str, str, str]] = {}
total = sum( total = sum(
len(column_values) len(column_values)
for table_values in unique_values.values() for table_values in unique_values.values()
for column_values in table_values.values() for column_values in table_values.values()
) )
logging.info("Total unique values: %s", total) logger.info("Total unique values: %s", total)
progress_bar = tqdm(total=total, desc="Creating LSH") if verbose else None progress_bar = tqdm(total=total, desc="Creating LSH") if verbose else None
for table_name, table_values in unique_values.items(): for table_name, table_values in unique_values.items():
+3 -2
View File
@@ -82,8 +82,9 @@ def _collect_decision_mins(phases: list[PhaseSpec]) -> dict[str, int]:
dtype = value dtype = value
else: else:
continue continue
if isinstance(dtype, str): if isinstance(dtype, str) and (
if dtype not in mins or phase_num < mins[dtype]: dtype not in mins or phase_num < mins[dtype]
):
mins[dtype] = phase_num mins[dtype] = phase_num
else: else:
scan(value, phase_num) scan(value, phase_num)