Chiusura fase di ristrutturazione e modularizzazione del workflow per favorire sviluppo modulare

This commit is contained in:
2026-08-24 13:32:20 +02:00
parent fa2298653b
commit 6062cb010e
96 changed files with 304 additions and 259 deletions
+1 -1
View File
@@ -1,8 +1,8 @@
"""Direct PostgreSQL implementation of the DWH port."""
from tht.config import DatabaseConfig
from sqlalchemy.exc import OperationalError, SQLAlchemyError
from tht.config import DatabaseConfig
from tht.db import execute, sampling
from tht.db.connection import can_create_in_schema, make_engine, ping, writable_tables
from tht.db.introspect import introspect
+1 -1
View File
@@ -1,8 +1,8 @@
"""Thoth/PostgREST implementation of the DWH port."""
from tht.config import DatabaseIdentityConfig, RestConfig
from tht.db.introspect import introspect_rest
from tht.db import sampling
from tht.db.introspect import introspect_rest
from tht.execute import ExecResult, ExecutionError, PlanSummary
from tht.mschema.models import PhysicalSchema
from tht.ports.dwh import DistinctValues, DwhCapabilities, DwhHealth
+1
View File
@@ -1,6 +1,7 @@
from pathlib import Path
import typer
from tht.adapters.factory import build_dwh
from tht.cli.config_cmd import CONFIG_OPT
from tht.config import ConfigError, load_config
+1 -1
View File
@@ -16,7 +16,7 @@ def _extract_lsh_values(dwh, physical, annotations, limit):
continue
try:
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
except Exception as exc:
except Exception as exc: # noqa: BLE001 - an unreadable DWH column is non-fatal
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
continue
vals = [str(value) for value in distinct.values if value not in (None, "")]
+1 -1
View File
@@ -459,8 +459,8 @@ def solved_search_cmd(
from rich.table import Table
from tht.cli.vector_cmd import make_embedder, open_searcher
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.memory import search_solved_questions
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config)
+7 -8
View File
@@ -12,8 +12,8 @@ import json
import typer
from tht.phase import (
auto_advance_eligible,
advance_problems,
auto_advance_eligible,
current_phase,
)
from tht.workflow import load_workflow
@@ -93,12 +93,11 @@ def advance_cmd(
if cur > wf.max_phase:
typer.secho("Sessione già alla fase terminale.", fg=typer.colors.YELLOW)
raise typer.Exit(0)
if auto:
if not auto_advance_eligible(snapshot):
problems = advance_problems(snapshot, cur)
for p in problems:
typer.echo(p)
raise typer.Exit(6) # needs human confirmation (gate contract)
if auto and not auto_advance_eligible(snapshot):
problems = advance_problems(snapshot, cur)
for p in problems:
typer.echo(p)
raise typer.Exit(6) # needs human confirmation (gate contract)
session_repository(cfg).append_decisions(
session, [{"type": "phase_approved", "subject": f"phase:{cur}"}]
)
@@ -164,7 +163,7 @@ def _cfg():
ws = os.environ.get("THT_WORKSPACE") or os.environ.get("THT_CONFIG")
config_path = Path(ws) if ws else Path("config/tht.yaml")
try:
from tht.cli.schema_cmd import _load_config_or_exit # noqa: F401 (portato in Onda 1.4)
from tht.cli.schema_cmd import _load_config_or_exit
return _load_config_or_exit(config_path)
except ImportError:
+2 -2
View File
@@ -96,9 +96,9 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder
from tht.evidence import build_preprocessing_pipeline, build_sources
from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.store import CorpusStore
from tht.evidence import build_preprocessing_pipeline, build_sources
cfg = _load_config_or_exit(config)
if cfg.embeddings is None:
@@ -130,9 +130,9 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
from tht.adapters.factory import build_vector_store
from tht.cli.schema_cmd import _load_config_or_exit
from tht.cli.vector_cmd import make_embedder
from tht.evidence import build_preprocessing_pipeline, build_sources
from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.store import CorpusStore
from tht.evidence import build_preprocessing_pipeline, build_sources
cfg = _load_config_or_exit(config)
if cfg.embeddings is None:
+5 -5
View File
@@ -303,7 +303,7 @@ def _suggest_fk_result(physical, annotations, *, sql_inputs: list[tuple[str, str
@schema_app.command("check")
def check_cmd(
config: Path = CONFIG_OPT,
annotations: Path | None = typer.Option(None, "--annotations"), # noqa: B008
annotations: Path | None = typer.Option(None, "--annotations"),
reviewed_candidates: str | None = typer.Option(None, "--reviewed-candidates"),
json_output: bool = typer.Option(False, "--json"),
) -> None:
@@ -411,11 +411,11 @@ _GENERIC_PK_NAMES = {"id", "key", "code"}
@schema_app.command("suggest-fks")
def suggest_fks_cmd(
config: Path = CONFIG_OPT,
from_sql: list[Path] = typer.Option( # noqa: B008
from_sql: list[Path] = typer.Option(
None, "--from-sql",
help="Directory o file .sql approvati da cui minare i join reali (ripetibile).",
),
assume: list[str] = typer.Option( # noqa: B008
assume: list[str] = typer.Option(
None, "--assume",
help="Disambigua una PK con piu' proprietari: col=tabella_ref "
"(es. cod_paz=dim_patient). Ripetibile.",
@@ -548,10 +548,10 @@ def render_cmd(
format: str = typer.Option(
"markdown", "--format", "-f", help="Formato: markdown | mschema-text | schema-dict"
),
tables: list[str] = typer.Option( # noqa: B008
tables: list[str] = typer.Option(
None, "--table", "-t", help="Limita alle tabelle indicate (ripetibile)."
),
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."), # noqa: B008
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
) -> None:
"""Serializza mschema (physical + annotations) nel formato richiesto."""
import json
+1 -1
View File
@@ -258,9 +258,9 @@ def pack_cmd(
build_retrieval_entries,
validate_corpus_workspace,
)
from tht.memory import SOLVED_KIND
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
from tht.search import combined_search, schema_tables
from tht.memory import SOLVED_KIND
from tht.vectorstore.embeddings import EmbeddingsError
cfg = _load_config_or_exit(config)
+3 -3
View File
@@ -515,12 +515,12 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
promoted_tables_for,
)
from tht.ctetest import CteError, CteTestRecord, _iter_json_objects
from tht.evidence import project_session
from tht.execute import ExecutionError
from tht.execute.warnings import plan_warnings, runtime_warnings, static_warnings
from tht.report import extract_reviewer_notes, render_validation_report
from tht.evidence import project_session
from tht.phase import cte_plan as effective_cte_plan
from tht.phase import effective_decisions
from tht.report import extract_reviewer_notes, render_validation_report
from tht.session.models import SchemaLinking
from tht.sqlcheck import validate_sql
@@ -656,7 +656,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
fg=typer.colors.CYAN,
)
except Exception as e:
except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
typer.secho(
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
f"Recupera con `tht memory solved-index {session_id}`.",
+3 -1
View File
@@ -5,10 +5,12 @@ from sqlalchemy import Engine
from tht.execute import (
ExecResult,
PlanSummary,
explain as _explain,
require_positive_int,
run_controlled,
)
from tht.execute import (
explain as _explain,
)
DEFAULT_TIMEOUT_MS = 30_000
+5 -3
View File
@@ -45,9 +45,11 @@ def fetch_chain_pem(host: str, port: int = 443, timeout: int = 30) -> list[str]:
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE
try:
with socket.create_connection((host, port), timeout=timeout) as sock:
with ctx.wrap_socket(sock, server_hostname=host) as tls:
certs = _unverified_chain(tls)
with (
socket.create_connection((host, port), timeout=timeout) as sock,
ctx.wrap_socket(sock, server_hostname=host) as tls,
):
certs = _unverified_chain(tls)
except (OSError, ssl.SSLError) as e:
raise CaFetchError(
f"Impossibile connettersi a {host}:{port} per recuperare i certificati: {e}"
+3 -3
View File
@@ -92,7 +92,7 @@ def add_examples(engine: Engine, physical: PhysicalSchema, cfg: ExamplesConfig)
''')
try:
rows = conn.execute(q, {"lim": cfg.max_per_column}).fetchall()
except Exception as e: # colonna non leggibile: si salta, non si interrompe
except Exception as e: # noqa: BLE001 - skip any unreadable DWH column
logger.warning("Campionamento saltato per %s.%s: %s", table_name, column_name, e)
continue
column.examples = [str(r[0]) for r in rows]
@@ -164,7 +164,7 @@ def unique_values_for_lsh(
''')
try:
rows = conn.execute(q, {"lim": cfg.max_values_per_column}).fetchall()
except Exception as e:
except Exception as e: # noqa: BLE001 - skip any unreadable DWH column
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
continue
vals = [str(r[0]) for r in rows]
@@ -205,7 +205,7 @@ def unique_values_for_lsh_rest(
rows = client.top_values(
schema, table_name, column_name, cfg.max_values_per_column
)
except Exception as e:
except Exception as e: # noqa: BLE001 - skip any unreadable REST column
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
continue
vals = [str(r["value"]) for r in rows if r["value"] not in (None, "")]
+3 -4
View File
@@ -24,16 +24,15 @@ from tht.evidence.search import (
from tht.evidence.session import project_session
from tht.evidence.sources import build_sources
__all__ = [
"AcquiredDocument",
"ActiveEvidenceSearcher",
"CorpusWorkspaceMismatchError",
"EvidenceEmbedder",
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"EvidenceEmbedder",
"SourceObject",
"ActiveEvidenceSearcher",
"CorpusWorkspaceMismatchError",
"acquire",
"active_searcher",
"build_preprocessing_pipeline",
+4 -1
View File
@@ -7,7 +7,10 @@ from datetime import UTC, datetime
from urllib.parse import quote, urlsplit
from tht.evidence.contracts import (
AcquiredDocument, EvidenceSourceError, EvidenceSourceErrorCategory, SourceObject,
AcquiredDocument,
EvidenceSourceError,
EvidenceSourceErrorCategory,
SourceObject,
)
-1
View File
@@ -15,7 +15,6 @@ from tht.evidence.contracts import (
validate_safe_metadata,
)
_NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$")
_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$")
+1 -2
View File
@@ -10,9 +10,8 @@ from pydantic import JsonValue, TypeAdapter, ValidationError
from yaml.events import AliasEvent
from yaml.nodes import MappingNode
from tht.evidence.corpus.models import CanonicalDocument
from tht.evidence.contracts import AcquiredDocument, canonical_provenance_uri
from tht.evidence.corpus.models import CanonicalDocument
MAX_DOCUMENT_BYTES = 10 * 1024 * 1024
_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE)
+13 -11
View File
@@ -4,6 +4,7 @@ from __future__ import annotations
import hashlib
import json
import logging
import re
import uuid
from collections.abc import Mapping, Sequence
@@ -11,17 +12,16 @@ from dataclasses import asdict, dataclass, field
from datetime import UTC
from pathlib import Path
import tht.evidence.acquisition as evidence_acquisition
from tht.evidence.contracts import EvidenceSource, SourceObject, canonical_provenance_uri
from tht.evidence.corpus.chunk import ChunkPolicy, chunk
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest
from tht.evidence.corpus.normalize import normalize
from tht.evidence.corpus.store import CorpusStore
import tht.evidence.acquisition as evidence_acquisition
from tht.evidence.contracts import EvidenceSource, SourceObject, canonical_provenance_uri
from tht.ports.vector import VectorStore, VectorWriteRecord
from tht.vectorstore.records import VectorRecord
from tht.jobs.models import JobSpec
from tht.jobs.runner import JobContext, StageArtifacts, run_job, seal_stage_artifacts
from tht.ports.vector import VectorStore, VectorWriteRecord
from tht.vectorstore.records import VectorRecord
EVIDENCE_STAGE_IDS = (
"discover",
@@ -33,6 +33,8 @@ EVIDENCE_STAGE_IDS = (
"retention_cleanup",
)
logger = logging.getLogger(__name__)
class PipelineError(RuntimeError):
"""Credential-free failure at the preprocessing boundary."""
@@ -212,14 +214,14 @@ class CorpusPipeline:
if purge_vector:
try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception:
except Exception: # noqa: BLE001 - retention reports per-generation failures
failures.append({"generation": generation, "error": "vector cleanup failed"})
continue
try:
if purge_filesystem:
self.store.discard(generation)
evicted.append(generation)
except Exception:
except Exception: # noqa: BLE001 - retention reports per-generation failures
failures.append({"generation": generation, "error": "filesystem cleanup failed"})
return {"status": "partial" if failures else "succeeded", "dry_run": dry_run,
"active_generation": self.store.active_generation(), "evicted": evicted,
@@ -378,7 +380,7 @@ class CorpusPipeline:
try:
active_assets_valid = active_assets_are_valid(previous)
except Exception:
except Exception: # noqa: BLE001 - any corrupt active asset disables reuse
active_assets_valid = False
reusable = (
active_assets_valid
@@ -538,7 +540,7 @@ class CorpusPipeline:
try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception:
pass
logger.debug("Failed to clean the compensated vector generation", exc_info=True)
write(context, "compensated.json", {"generation": generation})
def rotate_compensated_generation(context: JobContext) -> None:
@@ -786,12 +788,12 @@ class CorpusPipeline:
try:
self.store.discard(generation)
except Exception:
pass
logger.debug("Failed to discard the unpublished evidence generation", exc_info=True)
if vector_written:
try:
self.vector_store.delete_generation("evidence", generation, self.workspace_id)
except Exception:
pass
logger.debug("Failed to delete the unpublished vector generation", exc_info=True)
@staticmethod
def _vector_record(
+5 -6
View File
@@ -2,22 +2,21 @@
from __future__ import annotations
import json
import fcntl
import hashlib
import json
import os
import re
import stat
import shutil
import uuid
import hashlib
import stat
import threading
import uuid
from contextlib import contextmanager
from datetime import UTC, datetime
from pathlib import Path
from contextlib import contextmanager
from tht.evidence.corpus.models import CorpusManifest
_GENERATION = re.compile(r"^gen:[0-9a-f]{32}$")
+2 -2
View File
@@ -46,7 +46,7 @@ class ConceptFormula(BaseModel):
return f"---\n{fm}---\n{self.sql}\n"
@classmethod
def parse(cls, text: str) -> "ConceptFormula":
def parse(cls, text: str) -> ConceptFormula:
if not text.startswith("---\n"):
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
try:
@@ -55,7 +55,7 @@ class ConceptFormula(BaseModel):
raise ValueError("frontmatter malformato") from e
meta = yaml.safe_load(fm)
if not isinstance(meta, dict):
raise ValueError("frontmatter non valido")
raise TypeError("frontmatter non valido")
return cls.model_validate({**meta, "sql": body.strip("\n")})
+1 -1
View File
@@ -2,10 +2,10 @@
from typing import Protocol
from tht.evidence.contracts import EvidenceSource
from tht.evidence.corpus.chunk import ChunkPolicy
from tht.evidence.corpus.pipeline import CorpusPipeline
from tht.evidence.corpus.store import CorpusStore
from tht.evidence.contracts import EvidenceSource
from tht.ports.vector import VectorStore
+3 -2
View File
@@ -9,6 +9,7 @@ import re
import stat
from pathlib import Path
from types import TracebackType
from typing import Self
class JobAlreadyRunningError(RuntimeError):
@@ -34,7 +35,7 @@ class WorkspaceJobLock:
)
self._fd: int | None = None
def acquire(self) -> "WorkspaceJobLock":
def acquire(self) -> WorkspaceJobLock:
if self._fd is not None:
raise RuntimeError("job lock is already held by this object")
root_fd = os.open(self.path.parents[2], os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
@@ -85,7 +86,7 @@ class WorkspaceJobLock:
finally:
os.close(fd)
def __enter__(self) -> "WorkspaceJobLock":
def __enter__(self) -> Self:
return self.acquire()
def __exit__(
+11 -5
View File
@@ -7,8 +7,14 @@ from datetime import UTC, datetime
from pathlib import Path
from typing import Literal, Self
from pydantic import BaseModel, ConfigDict, Field, field_serializer, field_validator, model_validator
from pydantic import (
BaseModel,
ConfigDict,
Field,
field_serializer,
field_validator,
model_validator,
)
_JOB_KEY = re.compile(r"^[a-z][a-z0-9_-]{0,63}$")
_RUN_ID = re.compile(r"^[0-9a-f]{32}$")
@@ -95,7 +101,7 @@ class JobSpec(_FrozenModel):
data.update(update)
return type(self).model_validate(data)
def with_resume(self, run_id: str) -> "JobSpec":
def with_resume(self, run_id: str) -> JobSpec:
return self.model_copy(update={"resume_run_id": run_id})
@@ -121,7 +127,7 @@ class StageRun(_FrozenModel):
)
@model_validator(mode="after")
def state_shape(self) -> "StageRun":
def state_shape(self) -> StageRun:
if self.status == "pending" and any(
value is not None for value in (
self.started_at, self.finished_at, self.error, self.effect_state,
@@ -179,7 +185,7 @@ class JobRun(_FrozenModel):
_resumed_from = field_validator("resumed_from")(_validate_run_id)
@model_validator(mode="after")
def ledger_shape(self) -> "JobRun":
def ledger_shape(self) -> JobRun:
names = [stage.name for stage in self.stages]
if len(names) != len(set(names)):
raise ValueError("stage identifiers must be unique")
+4 -4
View File
@@ -2,12 +2,12 @@
from __future__ import annotations
import json
import hashlib
import json
import os
import uuid
import stat
import shutil
import stat
import uuid
from collections.abc import Callable, Sequence
from dataclasses import dataclass
from pathlib import Path
@@ -325,7 +325,7 @@ def run_job(
_persist(checkpoint_path, run)
try:
stage_result = stage_callable(context)
except Exception:
except Exception: # noqa: BLE001 - stage failures are persisted as terminal reports
failed = stage.model_copy(
update={
"status": "failed",
+1 -1
View File
@@ -122,7 +122,7 @@ def index_solved_question_best_effort(
store=store_factory(),
embedder=embedder_factory(),
)
except Exception as error:
except Exception as error: # noqa: BLE001 - callers receive a best-effort outcome
return SolvedIndexOutcome(upserted=None, error=str(error))
return SolvedIndexOutcome(upserted=upserted)
+2 -2
View File
@@ -115,8 +115,8 @@ def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None
lines = [
f"# Schema {physical.db_schema} ({physical.database})",
"",
f"Introspezione: {physical.introspected_at.isoformat()} — "
f"{len(physical.tables)} tabelle",
(f"Introspezione: {physical.introspected_at.isoformat()} — "
f"{len(physical.tables)} tabelle"),
]
for table_name, table in physical.tables.items():
lines += ["", f"## {table_name}", ""]
+1 -2
View File
@@ -1,9 +1,8 @@
"""Deterministic builder for the single Pi-facing Thoth session skill."""
import argparse
from pathlib import Path
import sys
from pathlib import Path
HARNESS_ROOT = Path(__file__).resolve().parents[1]
SKILL_ROOT = HARNESS_ROOT / ".pi" / "skills" / "tht-sessione"
+4 -4
View File
@@ -1,18 +1,18 @@
"""Stable interfaces implemented by Thoth infrastructure adapters."""
from tht.ports.dwh import (
DistinctValues,
DwhAdapter,
DwhCapabilities,
DwhHealth,
DistinctValues,
UnsupportedCapability,
)
from tht.ports.vector import (
VectorCapabilities,
VectorHealth,
VectorHit,
VectorRecord,
VectorReadUnavailable,
VectorRecord,
VectorStore,
VectorStoreError,
VectorWriteRecord,
@@ -20,16 +20,16 @@ from tht.ports.vector import (
)
__all__ = [
"DistinctValues",
"DwhAdapter",
"DwhCapabilities",
"DwhHealth",
"DistinctValues",
"UnsupportedCapability",
"VectorCapabilities",
"VectorHealth",
"VectorHit",
"VectorRecord",
"VectorReadUnavailable",
"VectorRecord",
"VectorStore",
"VectorStoreError",
"VectorWriteRecord",
+1 -1
View File
@@ -40,7 +40,7 @@ class RestClient:
try:
body = resp.json()
detail = body.get("message") or body.get("details") or resp.text
except Exception:
except (requests.exceptions.JSONDecodeError, AttributeError, TypeError):
detail = resp.text
return f"DWH REST rpc {fn} → HTTP {resp.status_code}: {detail}"
+2 -2
View File
@@ -9,8 +9,8 @@ import re
import shutil
import tempfile
import uuid
from collections.abc import Sequence
from pathlib import Path
from typing import Sequence
import portalocker
import yaml
@@ -147,7 +147,7 @@ class FilesystemSessionRepository:
return {}
data = json.loads(path.read_text())
if not isinstance(data, dict):
raise ValueError(f"Invalid preferences: {path}")
raise TypeError(f"Invalid preferences: {path}")
return data
def set_preferences(self, preferences: dict) -> None:
+1 -1
View File
@@ -8,7 +8,7 @@ from typing import Literal, Self
import portalocker
import yaml
from pydantic import BaseModel, Field, ConfigDict
from pydantic import BaseModel, ConfigDict, Field
from tht.decisions import DecisionRecord
+2 -2
View File
@@ -6,13 +6,13 @@ import hashlib
import json
import re
import uuid
from collections.abc import Iterator, Sequence
from contextlib import contextmanager
from dataclasses import dataclass
from datetime import UTC, datetime
from importlib.resources import files
from importlib.resources.abc import Traversable
from pathlib import Path
from typing import Iterator, Sequence
from sqlalchemy import Engine, create_engine, text
from sqlalchemy.engine import URL, make_url
@@ -204,7 +204,7 @@ class PostgresSessionRepository:
self._runtime_role = runtime_role
@classmethod
def from_config(cls, config, principal: PrincipalContext) -> "PostgresSessionRepository":
def from_config(cls, config, principal: PrincipalContext) -> PostgresSessionRepository:
query = {"sslmode": config.sslmode}
if config.sslrootcert is not None:
query["sslrootcert"] = str(config.sslrootcert)
+2 -1
View File
@@ -3,7 +3,8 @@
from __future__ import annotations
import os
from typing import Protocol, Sequence
from collections.abc import Sequence
from typing import Protocol
from tht.decisions import DecisionInput, DecisionRecord
from tht.session.models import PrincipalContext, SessionManifest, SessionSnapshot
+2 -2
View File
@@ -55,7 +55,7 @@ def _extract_name(question: str) -> str:
try:
extractor = yake.KeywordExtractor(lan="it", n=1, top=8, dedupLim=0.9)
ranked = [k for k, _ in extractor.extract_keywords(q)]
except Exception:
except Exception: # noqa: BLE001 - keyword extraction has a deterministic fallback
return _summarize(question)
seen: set[str] = set()
picked: list[str] = []
@@ -117,7 +117,7 @@ def create_session(
from tht.workflow import load_workflow
schema_version = load_workflow().schema_version
except Exception:
except Exception: # noqa: BLE001 - legacy sessions may predate workflow metadata
schema_version = None
manifest = SessionManifest(
id=session_id, created_at=now, question=question,
+1 -1
View File
@@ -87,7 +87,7 @@ def generate_task_doc(
wf = load_workflow()
name = wf.phase_name(phase)
header = f"## Task: fase {phase} ({name})"
except Exception:
except Exception: # noqa: BLE001 - task documents retain a phase-only fallback
header = f"## Task: fase {phase}"
parts.append(header)
+7 -6
View File
@@ -4,11 +4,12 @@
"""Core LSH (MinHash) per la ricerca di valori simili nei campi del database."""
import logging
from typing import Dict, List, Tuple
from datasketch import MinHash, MinHashLSH
from tqdm import tqdm
logger = logging.getLogger(__name__)
def create_minhash(signature_size: int, string: str, n_gram: int) -> MinHash:
m = MinHash(num_perm=signature_size)
@@ -33,7 +34,7 @@ NAME_LIKE_TOKENS: tuple[str, ...] = (
def skip_column(
column_name: str,
column_values: List[str],
column_values: list[str],
max_total_chars: int = 50000,
max_avg_length: int = 20,
name_tokens: tuple[str, ...] = NAME_LIKE_TOKENS,
@@ -51,20 +52,20 @@ def jaccard_similarity(m1: MinHash, m2: MinHash) -> float:
def create_lsh_index(
unique_values: Dict[str, Dict[str, List[str]]],
unique_values: dict[str, dict[str, list[str]]],
signature_size: int,
n_gram: int,
threshold: float,
verbose: bool = True,
) -> Tuple[MinHashLSH, Dict[str, Tuple[MinHash, str, str, str]]]:
) -> tuple[MinHashLSH, dict[str, tuple[MinHash, str, str, str]]]:
lsh = MinHashLSH(threshold=threshold, num_perm=signature_size)
minhashes: Dict[str, Tuple[MinHash, str, str, str]] = {}
minhashes: dict[str, tuple[MinHash, str, str, str]] = {}
total = sum(
len(column_values)
for table_values in unique_values.values()
for column_values in table_values.values()
)
logging.info("Total unique values: %s", total)
logger.info("Total unique values: %s", total)
progress_bar = tqdm(total=total, desc="Creating LSH") if verbose else None
for table_name, table_values in unique_values.items():
+4 -3
View File
@@ -82,9 +82,10 @@ def _collect_decision_mins(phases: list[PhaseSpec]) -> dict[str, int]:
dtype = value
else:
continue
if isinstance(dtype, str):
if dtype not in mins or phase_num < mins[dtype]:
mins[dtype] = phase_num
if isinstance(dtype, str) and (
dtype not in mins or phase_num < mins[dtype]
):
mins[dtype] = phase_num
else:
scan(value, phase_num)
elif isinstance(node, list):