fix: harden model catalog projections

This commit is contained in:
Codex
2026-09-02 19:25:01 +02:00
parent ce4c31a6fb
commit a6a5bf2036
38 changed files with 573 additions and 83 deletions
+57 -5
View File
@@ -214,7 +214,7 @@ embeddings: {provider: ollama_internal, base_url: http://embedding:11434, model:
assert cfg.vectors.writer.api_key == "writer"
def test_accepts_only_internal_ollama_embedding_contract(tmp_path):
def test_accepts_catalog_selected_internal_ollama_embedding_contract(tmp_path):
workspace = tmp_path / "workspace.yaml"
workspace.write_text(
"""
@@ -225,8 +225,9 @@ resources:
embeddings:
provider: ollama_internal
base_url: http://embedding:11434
model: qwen3-embedding:0.6b
dimensions: 1024
id: ollama/bge-m3
model: bge-m3
dimensions: 1536
"""
)
@@ -234,8 +235,59 @@ resources:
assert cfg.embeddings.provider == "ollama_internal"
assert cfg.embeddings.base_url == "http://embedding:11434"
assert cfg.embeddings.model == "qwen3-embedding:0.6b"
assert cfg.embeddings.dim == 1024
assert cfg.embeddings.id == "ollama/bge-m3"
assert cfg.embeddings.model == "bge-m3"
assert cfg.embeddings.dim == 1536
def test_installation_embedding_projection_completes_model_free_runtime_source(
tmp_path, monkeypatch,
):
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_ID", "ollama/bge-m3")
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_MODEL", "bge-m3")
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_DIMENSIONS", "1536")
workspace = tmp_path / "workspace.yaml"
workspace.write_text(
"""
dwh:
type: postgres_direct
connection: {database: analytics, schema: mart, user: reader, password: secret}
resources:
embeddings:
provider: ollama_internal
base_url: http://embedding:11434
"""
)
cfg = load_config(workspace)
assert cfg.embeddings.id == "ollama/bge-m3"
assert cfg.embeddings.model == "bge-m3"
assert cfg.embeddings.dim == 1536
def test_workspace_embedding_values_cannot_override_installation_projection(
tmp_path, monkeypatch,
):
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_ID", "ollama/bge-m3")
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_MODEL", "bge-m3")
monkeypatch.setenv("THT_INTERNAL_EMBEDDING_DIMENSIONS", "1536")
workspace = tmp_path / "workspace.yaml"
workspace.write_text(
"""
dwh:
type: postgres_direct
connection: {database: analytics, schema: mart, user: reader, password: secret}
resources:
embeddings:
provider: ollama_internal
base_url: http://embedding:11434
model: workspace-owned-model
"""
)
with pytest.raises(ConfigError, match="proprietà dell'installazione|diverge"):
load_config(workspace)
def test_accepts_internal_qdrant_resource_contract(tmp_path):
+21
View File
@@ -40,6 +40,7 @@ def test_manifest_contains_provenance_without_credentials():
manifest_id="manifest:abc",
created_at=datetime(2026, 7, 12, tzinfo=UTC),
pipeline_version="evidence-v1",
embedding_id="ollama/nomic-embed-text",
embedding_model="nomic-embed-text",
embedding_dimensions=768,
documents=[document()],
@@ -52,6 +53,7 @@ def test_manifest_contains_provenance_without_credentials():
assert "etag:abc" in payload
assert "evidence-v1" in payload
assert "nomic-embed-text" in payload
assert '"embedding_id":"ollama/nomic-embed-text"' in payload
assert "api_key" not in payload
@@ -93,6 +95,25 @@ def test_manifest_validates_embedding_compatibility_fields():
)
def test_schema_v1_manifest_without_canonical_embedding_id_remains_readable():
manifest = CorpusManifest.model_validate({
"schema_version": 1,
"embedding_model": "legacy-model",
"embedding_dimensions": 768,
})
assert manifest.embedding_id is None
def test_schema_v2_embedding_generation_requires_canonical_id():
with pytest.raises(ValidationError, match="embedding_id"):
CorpusManifest(
schema_version=2,
embedding_model="model-v2",
embedding_dimensions=768,
)
def test_canonical_metadata_rejects_secrets_and_non_json_values():
with pytest.raises(ValidationError, match="credential-like"):
CanonicalDocument.model_validate(
+21 -1
View File
@@ -105,11 +105,13 @@ def item(name, fingerprint):
)
def pipeline(tmp_path, source, *, embedder=None, vectors=None, model="model-a", policy=None,
def pipeline(tmp_path, source, *, embedder=None, vectors=None, model="model-a",
embedding_id=None, policy=None,
retain=3, candidate_evaluator=None):
return CorpusPipeline(
store=CorpusStore(tmp_path / "corpus"), sources=[source],
embedder=embedder or Embedder(), vector_store=vectors or Vectors(),
embedding_id=embedding_id,
embedding_model=model, embedding_dimensions=3,
chunk_policy=policy or ChunkPolicy(version="chunk-v1", max_chars=100),
pipeline_version="evidence-v1",
@@ -875,6 +877,24 @@ def test_model_or_chunk_policy_change_forces_full_rebuild(tmp_path):
assert source.acquire_calls == ["fs:one"]
def test_canonical_embedding_id_change_forces_full_rebuild_and_is_persisted(tmp_path):
one = item("one", "a")
first = pipeline(
tmp_path, Source([(one, "hello")]), model="same-upstream",
embedding_id="ollama/catalog-a",
).run()
assert first.manifest.embedding_id == "ollama/catalog-a"
source = Source([(one, "hello")])
changed = pipeline(
tmp_path, source, model="same-upstream", embedding_id="ollama/catalog-b",
).run()
assert changed.changed == ("fs:one",)
assert changed.manifest.embedding_id == "ollama/catalog-b"
assert source.acquire_calls == ["fs:one"]
def test_partial_vector_failure_never_changes_active_or_exposes_generation(tmp_path):
one = item("one", "a")
good = pipeline(tmp_path, Source([(one, "old")]))
+16 -2
View File
@@ -21,7 +21,8 @@ def _write_cfg(tmp_path, raw):
return cfg
def _cfg(tmp_path, *, transport="thoth_rest", base_url="http://dwh.example.invalid", collection="psd", model="qwen3-embedding:0.6b"):
def _cfg(tmp_path, *, transport="thoth_rest", base_url="http://dwh.example.invalid",
collection="psd", model="qwen3-embedding:0.6b", dimensions=1024):
return {
"schemaVersion": 1,
"workspace": {"schema_version": 3, "id": "psd", "name": "PSD", "language": "it"},
@@ -34,7 +35,10 @@ def _cfg(tmp_path, *, transport="thoth_rest", base_url="http://dwh.example.inval
"connection": {"host": "h", "port": 5432, "database": "warehouse", "schema": "dw", "user": "reader", "password": "secret"},
},
"vectors": {"type": "qdrant", "base_url": "http://qdrant:6333", "collection": collection, "collection_lifecycle": "self_heal"},
"embeddings": {"provider": "ollama_internal", "base_url": "http://embedding:11434", "model": model, "dimensions": 1024},
"embeddings": {
"provider": "ollama_internal", "base_url": "http://embedding:11434",
"id": f"ollama/{model}", "model": model, "dimensions": dimensions,
},
"roots": {"artifacts": str(tmp_path / "artifacts"), "indexes": str(tmp_path / "indexes")},
"paths": {"artifacts": str(tmp_path / "artifacts"), "indexes": str(tmp_path / "indexes"), "sessions": str(tmp_path / "sessions")},
}
@@ -52,6 +56,16 @@ def test_canonical_json_is_deterministic_and_key_ordered(cfg):
assert keys == ["schemaVersion", "dwh", "vector", "embedding", "roots"]
def test_catalog_embedding_identity_and_dimensions_drive_effective_config(tmp_path):
cfg = _write_cfg(tmp_path, _cfg(tmp_path, model="bge-m3", dimensions=1536))
document = __import__("json").loads(canonical_effective_config_json(cfg))
assert document["embedding"] == {
"id": "ollama/bge-m3", "model": "bge-m3", "dimensions": 1536,
}
assert document["vector"]["dimensions"] == 1536
def test_canonical_excludes_credentials_and_evidence(cfg):
doc = canonical_effective_config_json(cfg)
assert "secret" not in doc
@@ -149,6 +149,7 @@ def test_preprocessing_factory_forwards_only_evidence_pipeline_dependencies(monk
"sources": [object()],
"embedder": object(),
"vector_store": object(),
"embedding_id": "ollama/model",
"embedding_model": "model",
"embedding_dimensions": 3,
"chunk_policy": object(),
+2
View File
@@ -10,6 +10,8 @@ from tht.config import EmbeddingsConfig
def _cfg(**kw):
kw.setdefault("model", "nomic-embed-text-v2-moe")
kw.setdefault("dim", 768)
emb = EmbeddingsConfig(base_url="http://localhost:11434", **kw)
return SimpleNamespace(embeddings=emb)
+3
View File
@@ -284,6 +284,7 @@ def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_pat
sources,
embedder,
vector_store,
embedding_id,
embedding_model,
embedding_dimensions,
chunk_policy,
@@ -293,6 +294,7 @@ def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_pat
candidate_evaluator,
):
calls["init"] = {
"embedding_id": embedding_id,
"embedding_model": embedding_model,
"embedding_dimensions": embedding_dimensions,
"pipeline_version": pipeline_version,
@@ -314,6 +316,7 @@ def test_run_from_config_uses_runtime_identity_workspace_id(monkeypatch, tmp_pat
command.run_from_config(config)
assert calls["init"]["embedding_id"] == "ollama/qwen3-embedding:0.6b"
assert calls["init"]["sparse_language"] == "english"
assert calls["init"]["candidate_evaluator"] is None
assert calls["run_as_job"]["workspace_id"] == "psd-clinical"
+2
View File
@@ -187,6 +187,7 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
embedder=embedder,
vector_store=vector_store,
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
embedding_model=cfg.embeddings.model, embedding_dimensions=cfg.embeddings.dim,
chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=cfg.vector.max_chunk_chars),
pipeline_version="evidence-v1",
@@ -222,6 +223,7 @@ def gc_from_config(config: Path, *, dry_run: bool = False):
pipeline = build_preprocessing_pipeline(
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
embedder=make_embedder(cfg.embeddings), vector_store=build_vector_store(cfg, require_write=True),
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
embedding_model=cfg.embeddings.model, embedding_dimensions=cfg.embeddings.dim,
chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=cfg.vector.max_chunk_chars),
pipeline_version="evidence-v1",
+84 -13
View File
@@ -60,8 +60,12 @@ def canonical_effective_config_document(cfg) -> dict:
return {
"schemaVersion": 1,
"dwh": dwh,
"vector": {"collection": collection, "dimensions": 1024, "distance": "cosine"},
"embedding": {"model": model, "dimensions": int(embed_dim)},
"vector": {"collection": collection, "dimensions": int(embed_dim), "distance": "cosine"},
"embedding": {
"id": getattr(embeddings, "id", None) or f"ollama/{model}",
"model": model,
"dimensions": int(embed_dim),
},
"roots": {
"artifacts": str(getattr(cfg.paths, "artifacts", Path("artifacts"))),
"indexes": str(getattr(cfg.paths, "indexes", Path("indexes"))),
@@ -499,8 +503,9 @@ class EvidenceSourcesConfig(BaseModel):
class EmbeddingsConfig(BaseModel):
provider: str = "ollama_internal"
base_url: str
model: str = "nomic-embed-text-v2-moe"
dim: int = Field(default=768, alias="dimensions")
id: str | None = None
model: str
dim: int = Field(alias="dimensions")
batch_size: int = 16
timeout: int = 300
connect_timeout: int = 5
@@ -642,6 +647,7 @@ def load_config(path: Path) -> Config:
if not isinstance(raw, dict):
raise ConfigError(f"Configurazione non valida (atteso un mapping YAML): {path}")
expanded = _resolve_secret_files(_resolve_evidence_secret_files(_expand_env(raw)))
_apply_installation_embedding_projection(expanded, path)
_validate_internal_embedding_contract(expanded, path)
_validate_internal_vector_contract(expanded, path)
translated, used_legacy = translate_legacy_config(expanded)
@@ -710,6 +716,56 @@ def load_config(path: Path) -> Config:
return cfg
def _apply_installation_embedding_projection(raw: dict[str, Any], path: Path) -> None:
projected = {
"id": os.environ.get("THT_INTERNAL_EMBEDDING_ID"),
"model": os.environ.get("THT_INTERNAL_EMBEDDING_MODEL"),
"dimensions": os.environ.get("THT_INTERNAL_EMBEDDING_DIMENSIONS"),
}
if all(value is None for value in projected.values()):
return
if any(value is None for value in projected.values()):
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"la proiezione embedding dell'installazione è incompleta"
)
embedding_id = projected["id"]
model = projected["model"]
try:
dimensions = int(projected["dimensions"] or "")
except ValueError:
dimensions = 0
if embedding_id != f"ollama/{model}" or dimensions <= 0:
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"la proiezione embedding dell'installazione non è canonica"
)
sections: list[tuple[dict[str, Any], str]] = []
embeddings = raw.get("embeddings")
if isinstance(embeddings, dict):
sections.append((embeddings, "dim"))
resources = raw.get("resources")
if isinstance(resources, dict) and isinstance(resources.get("embeddings"), dict):
sections.append((resources["embeddings"], "dimensions"))
for section, dimension_key in sections:
declared = {
"id": section.get("id"),
"model": section.get("model"),
"dimensions": section.get(dimension_key),
}
expected = {"id": embedding_id, "model": model, "dimensions": dimensions}
for key, value in declared.items():
if value is not None and str(value) != str(expected[key]):
raise ConfigError(
f"Configurazione non valida in {path}:\n"
f"{key} è proprietà dell'installazione e diverge dalla proiezione attiva"
)
section["id"] = embedding_id
section["model"] = model
section[dimension_key] = dimensions
def _validate_internal_embedding_contract(raw: dict[str, Any], path: Path) -> None:
resources = raw.get("resources")
if not isinstance(resources, dict):
@@ -720,9 +776,10 @@ def _validate_internal_embedding_contract(raw: dict[str, Any], path: Path) -> No
provider = embeddings.get("provider")
model = embeddings.get("model")
embedding_id = embeddings.get("id")
dimensions = embeddings.get("dimensions")
base_url = embeddings.get("base_url")
allowed = {"provider", "base_url", "model", "dimensions"}
allowed = {"provider", "base_url", "id", "model", "dimensions"}
unexpected = sorted(set(embeddings) - allowed)
if unexpected:
raise ConfigError(
@@ -734,15 +791,24 @@ def _validate_internal_embedding_contract(raw: dict[str, Any], path: Path) -> No
f"Configurazione non valida in {path}:\n"
"resources.embeddings.provider deve essere 'ollama_internal'"
)
if model != "qwen3-embedding:0.6b":
if not isinstance(model, str) or not model:
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"resources.embeddings.model deve essere 'qwen3-embedding:0.6b'"
"resources.embeddings.model deve essere valorizzato"
)
if dimensions != 1024:
try:
parsed_dimensions = int(dimensions)
except (TypeError, ValueError):
parsed_dimensions = 0
if isinstance(dimensions, bool) or parsed_dimensions <= 0:
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"resources.embeddings.dimensions deve essere 1024"
"resources.embeddings.dimensions deve essere un intero positivo"
)
if embedding_id is not None and embedding_id != f"ollama/{model}":
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"resources.embeddings.id deve essere l'identità canonica ollama/<model>"
)
if not _is_allowed_internal_embedding_url(base_url):
raise ConfigError(
@@ -805,15 +871,20 @@ def _validate_active_embeddings_config(
f"Configurazione non valida in {path}:\n"
"embeddings.provider deve essere 'ollama_internal'"
)
if embeddings.model != "qwen3-embedding:0.6b":
if not embeddings.model:
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"embeddings.model deve essere 'qwen3-embedding:0.6b'"
"embeddings.model deve essere valorizzato"
)
if embeddings.dim != 1024:
if embeddings.dim <= 0:
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"embeddings.dim deve essere 1024"
"embeddings.dim deve essere un intero positivo"
)
if embeddings.id is not None and embeddings.id != f"ollama/{embeddings.model}":
raise ConfigError(
f"Configurazione non valida in {path}:\n"
"embeddings.id deve essere l'identità canonica ollama/<model>"
)
if not _is_allowed_internal_embedding_url(embeddings.base_url):
raise ConfigError(
+17
View File
@@ -19,6 +19,9 @@ from tht.evidence.contracts import (
_NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$")
_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$")
_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
_CANONICAL_MODEL_ID = re.compile(
r"^[a-z][a-z0-9._-]{0,63}/[A-Za-z0-9][A-Za-z0-9._:-]{0,255}$"
)
_EVIDENCE_METADATA_KEYS = frozenset({
"evidence_id", "evidence_kind", "purposes", "scope", "language", "provenance",
})
@@ -37,6 +40,12 @@ def _validate_hash(value: str) -> str:
return value
def _validate_canonical_model_id(value: str) -> str:
if not _CANONICAL_MODEL_ID.fullmatch(value):
raise ValueError("model identifier must use canonical provider/model form")
return value
def _require_content_hash(content: str, content_hash: str) -> None:
expected = f"sha256:{hashlib.sha256(content.encode('utf-8')).hexdigest()}"
if content_hash != expected:
@@ -146,6 +155,7 @@ class CorpusManifest(_WithMetadata):
manifest_id: str | None = None
created_at: datetime = Field(default_factory=lambda: datetime.now(UTC))
pipeline_version: str = Field(default="evidence-v1", min_length=1)
embedding_id: str | None = None
embedding_model: str | None = None
embedding_dimensions: int | None = Field(default=None, gt=0)
vector_generation: str | None = None
@@ -155,6 +165,9 @@ class CorpusManifest(_WithMetadata):
_manifest_id = field_validator("manifest_id")(
lambda value: _validate_namespaced_id(value) if value is not None else None
)
_embedding_id = field_validator("embedding_id")(
lambda value: _validate_canonical_model_id(value) if value is not None else None
)
_vector_generation = field_validator("vector_generation")(
lambda value: _validate_namespaced_id(value) if value is not None else None
)
@@ -164,6 +177,10 @@ class CorpusManifest(_WithMetadata):
def validate_generation(self) -> "CorpusManifest":
if (self.embedding_model is None) != (self.embedding_dimensions is None):
raise ValueError("embedding_model and embedding_dimensions must be set together")
if self.embedding_id is not None and self.embedding_model is None:
raise ValueError("embedding_id requires embedding model and dimensions")
if self.schema_version >= 2 and self.embedding_model is not None and self.embedding_id is None:
raise ValueError("schema version 2 embedding generations require embedding_id")
if self.vector_generation is not None and self.embedding_model is None:
raise ValueError("vector_generation requires embedding model and dimension compatibility")
+10 -1
View File
@@ -125,6 +125,7 @@ class CorpusPipeline:
def __init__(
self, *, store: CorpusStore, sources: list[EvidenceSource], embedder,
vector_store: VectorStore, embedding_model: str, embedding_dimensions: int,
embedding_id: str | None = None,
chunk_policy: ChunkPolicy, pipeline_version: str, retain_published_generations: int = 3,
workspace_id: str | None = None, sparse_language: str = "italian",
candidate_evaluator: Callable[[CorpusManifest], object] | None = None,
@@ -133,6 +134,7 @@ class CorpusPipeline:
self.sources = sources
self.embedder = embedder
self.vector_store = vector_store
self.embedding_id = embedding_id or f"ollama/{embedding_model}"
self.embedding_model = embedding_model
self.embedding_dimensions = embedding_dimensions
self.chunk_policy = chunk_policy
@@ -284,6 +286,7 @@ class CorpusPipeline:
source_by_id = {item.source_id: (source, item) for source, item in discovered}
compatibility = _fingerprint({
"pipeline": self.pipeline_version,
"embedding_id": self.embedding_id,
"model": self.embedding_model,
"dimensions": self.embedding_dimensions,
"chunk_policy": asdict(self.chunk_policy),
@@ -294,6 +297,7 @@ class CorpusPipeline:
"compatibility_fingerprint": compatibility,
"pipeline_version": self.pipeline_version,
"chunk_policy_version": self.chunk_policy.version,
"embedding_id": self.embedding_id,
"embedding_model": self.embedding_model,
"embedding_dimensions": self.embedding_dimensions,
}
@@ -492,7 +496,9 @@ class CorpusPipeline:
) for document in documents
}
manifest = CorpusManifest(
schema_version=2,
pipeline_version=self.pipeline_version,
embedding_id=self.embedding_id,
embedding_model=self.embedding_model,
embedding_dimensions=self.embedding_dimensions,
vector_generation=plan["generation"],
@@ -723,7 +729,8 @@ class CorpusPipeline:
prior_documents = {doc.source_id: doc for doc in previous.documents} if previous else {}
fingerprints = {item.source_id: item.fingerprint for _, item in discovered}
compatibility = _fingerprint({
"pipeline": self.pipeline_version, "model": self.embedding_model,
"pipeline": self.pipeline_version, "embedding_id": self.embedding_id,
"model": self.embedding_model,
"dimensions": self.embedding_dimensions, "chunk_policy": asdict(self.chunk_policy),
})
previous_compatibility = previous.metadata.get("compatibility_fingerprint") if previous else None
@@ -761,7 +768,9 @@ class CorpusPipeline:
for document in documents
}
manifest = CorpusManifest(
schema_version=2,
pipeline_version=self.pipeline_version,
embedding_id=self.embedding_id,
embedding_model=self.embedding_model,
embedding_dimensions=self.embedding_dimensions,
vector_generation=generation,
+2
View File
@@ -22,6 +22,7 @@ def build_preprocessing_pipeline(
vector_store: VectorStore,
embedding_model: str,
embedding_dimensions: int,
embedding_id: str | None = None,
chunk_policy: ChunkPolicy,
pipeline_version: str,
retain_published_generations: int = 3,
@@ -35,6 +36,7 @@ def build_preprocessing_pipeline(
sources=sources,
embedder=embedder,
vector_store=vector_store,
embedding_id=embedding_id,
embedding_model=embedding_model,
embedding_dimensions=embedding_dimensions,
chunk_policy=chunk_policy,
-2
View File
@@ -39,8 +39,6 @@ evidence:
embeddings:
base_url: ${THT_OLLAMA_URL} # http://host.docker.internal:11434
model: nomic-embed-text-v2-moe
dim: 768
batch_size: 32
# Vector: diretto (read+write). L'assenza di vector_rest/vector_write_rest fa sì che
-2
View File
@@ -25,8 +25,6 @@ paths:
embeddings:
base_url: ${THT_OLLAMA_URL}
model: nomic-embed-text-v2-moe
dim: 768
batch_size: 64
# LOADING diretto del pgvector (server-only). Su workstation la lettura passa da
-2
View File
@@ -39,8 +39,6 @@ evidence:
embeddings:
base_url: ${THT_OLLAMA_URL} # es. http://localhost:11434
model: nomic-embed-text-v2-moe
dim: 768
batch_size: 32
vectors: