test: add internal semantic smoke
This commit is contained in:
Executable
+298
@@ -0,0 +1,298 @@
|
||||
#!/usr/bin/env bash
|
||||
# Live smoke for the mandatory internal semantic stack: disposable project/volumes, no host
|
||||
# ports on semantic services, exact cleanup via Task 13 labels only, and offline persistence.
|
||||
set -euo pipefail
|
||||
|
||||
root="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
# shellcheck source=./unified-deployment-smoke.sh
|
||||
source "$root/scripts/unified-deployment-smoke.sh"
|
||||
|
||||
task13_wait_internal_embedding_model() {
|
||||
printf '== Wait for the internal embedding model ==\n'
|
||||
for _attempt in $(seq 1 30); do
|
||||
if task13_compose exec -T core /opt/venv/bin/python - <<'PY' >>"$TASK13_LOG" 2>&1
|
||||
import json
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
|
||||
with urllib.request.urlopen("http://embedding:11434/api/tags", timeout=10) as response:
|
||||
payload = json.load(response)
|
||||
models = [entry.get("name") for entry in payload.get("models", []) if isinstance(entry, dict)]
|
||||
if "qwen3-embedding:0.6b" not in models:
|
||||
raise SystemExit(1)
|
||||
request = urllib.request.Request(
|
||||
"http://embedding:11434/api/embed",
|
||||
data=json.dumps({"model": "qwen3-embedding:0.6b", "input": ["warm semantic smoke"]}).encode("utf-8"),
|
||||
headers={"content-type": "application/json"},
|
||||
method="POST",
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=180) as response:
|
||||
payload = json.load(response)
|
||||
except urllib.error.URLError:
|
||||
raise SystemExit(1)
|
||||
embeddings = payload.get("embeddings")
|
||||
raise SystemExit(0 if isinstance(embeddings, list) and len(embeddings) == 1 and len(embeddings[0]) == 1024 else 1)
|
||||
PY
|
||||
then
|
||||
return 0
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
task13_log_failure "internal embedding model readiness"
|
||||
}
|
||||
|
||||
task13_semantic_python_probe() {
|
||||
local mode="$1"
|
||||
task13_compose exec -T core /opt/venv/bin/python - "$mode" <<'PY'
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
|
||||
import requests
|
||||
|
||||
from tht.adapters.vector.qdrant import QdrantVectorStore
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.vectorstore.embeddings import OllamaEmbeddings
|
||||
from tht.vectorstore.records import VectorRecord
|
||||
|
||||
MODE = sys.argv[1]
|
||||
WORKSPACE_ID = "task13-smoke"
|
||||
WORKSPACE_REVISION = "b" * 40
|
||||
COLLECTION = "task13-smoke"
|
||||
QDRANT = "http://qdrant:6333"
|
||||
EMBEDDING = "http://embedding:11434"
|
||||
MODEL = "qwen3-embedding:0.6b"
|
||||
DIM = 1024
|
||||
|
||||
|
||||
def embedder() -> OllamaEmbeddings:
|
||||
return OllamaEmbeddings(
|
||||
SimpleNamespace(
|
||||
base_url=EMBEDDING,
|
||||
model=MODEL,
|
||||
dim=DIM,
|
||||
connect_timeout=2.0,
|
||||
timeout=180.0,
|
||||
batch_size=8,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def store() -> QdrantVectorStore:
|
||||
return QdrantVectorStore(
|
||||
base_url=QDRANT,
|
||||
collection=COLLECTION,
|
||||
workspace_id=WORKSPACE_ID,
|
||||
workspace_revision=WORKSPACE_REVISION,
|
||||
expected_dimension=DIM,
|
||||
)
|
||||
|
||||
|
||||
def content_hash(text: str) -> str:
|
||||
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def query_payload(vector: list[float], must: list[dict]) -> dict:
|
||||
response = requests.post(
|
||||
f"{QDRANT}/collections/{COLLECTION}/points/query",
|
||||
json={
|
||||
"vector": vector,
|
||||
"limit": 1,
|
||||
"with_payload": True,
|
||||
"filter": {"must": must},
|
||||
},
|
||||
timeout=(2.0, 15.0),
|
||||
)
|
||||
response.raise_for_status()
|
||||
payload = response.json()
|
||||
points = payload.get("result", {}).get("points")
|
||||
if not isinstance(points, list) or len(points) != 1:
|
||||
raise RuntimeError(f"expected exactly one semantic point, got {payload!r}")
|
||||
point = points[0]
|
||||
result = point.get("payload")
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError(f"missing payload in query result: {point!r}")
|
||||
return result
|
||||
|
||||
|
||||
records = {
|
||||
"schema": {
|
||||
"collection": "schema_records",
|
||||
"record": VectorRecord(
|
||||
id="schema_table:fact_task13",
|
||||
kind="schema_table",
|
||||
ref="fact_task13",
|
||||
title="fact_task13",
|
||||
content="Tabella fact_task13 con una riga dedicata allo smoke semantico interno.",
|
||||
metadata={"table_name": "fact_task13"},
|
||||
),
|
||||
"query": "fact task13 smoke table",
|
||||
"must": [
|
||||
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
|
||||
{"key": "kind", "match": {"value": "schema"}},
|
||||
{"key": "record_kind", "match": {"value": "schema_table"}},
|
||||
{"key": "record_key", "match": {"value": "schema_table:fact_task13"}},
|
||||
],
|
||||
},
|
||||
"evidence": {
|
||||
"collection": "evidence",
|
||||
"record": VectorRecord(
|
||||
id="evidence:task13-doc:0",
|
||||
kind="evidence",
|
||||
ref="task13-doc",
|
||||
title="Task 13 Evidence",
|
||||
content="Evidence dedicata allo smoke semantico interno con filtro esatto per generazione.",
|
||||
metadata={
|
||||
"document_id": "task13-doc",
|
||||
"vector_generation": "gen:11111111111111111111111111111111",
|
||||
"status": "published",
|
||||
"tier": "gold",
|
||||
"tables": ["fact_task13"],
|
||||
"concepts": ["semantic smoke"],
|
||||
},
|
||||
),
|
||||
"query": "semantic smoke evidence generation",
|
||||
"must": [
|
||||
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
|
||||
{"key": "kind", "match": {"value": "evidence"}},
|
||||
{"key": "document_id", "match": {"value": "task13-doc"}},
|
||||
{"key": "vector_generation", "match": {"value": "gen:11111111111111111111111111111111"}},
|
||||
],
|
||||
},
|
||||
"memory": {
|
||||
"collection": "memory",
|
||||
"record": VectorRecord(
|
||||
id="mem-9000",
|
||||
kind="memory",
|
||||
ref="mem-9000",
|
||||
title="Task 13 Memory",
|
||||
content="Memoria riusabile per lo smoke semantico interno persistente.",
|
||||
metadata={
|
||||
"session_id": "task13-session",
|
||||
"decision_seq": 9,
|
||||
"subject": "semantic smoke memory",
|
||||
"type": "concept_clarified",
|
||||
"concepts": ["semantic smoke memory"],
|
||||
},
|
||||
),
|
||||
"query": "semantic smoke memory reusable",
|
||||
"must": [
|
||||
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
|
||||
{"key": "kind", "match": {"value": "memory"}},
|
||||
{"key": "record_kind", "match": {"value": "memory"}},
|
||||
{"key": "record_key", "match": {"value": "mem-9000"}},
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
embedding_client = embedder()
|
||||
vector_store = store()
|
||||
|
||||
if MODE == "seed":
|
||||
for family in records.values():
|
||||
record = family["record"]
|
||||
vector_store.upsert(
|
||||
family["collection"],
|
||||
[
|
||||
VectorWriteRecord(
|
||||
record=record,
|
||||
embedding=embedding_client.embed_query(record.content),
|
||||
content_hash=content_hash(record.content),
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
collection_info = requests.get(
|
||||
f"{QDRANT}/collections/{COLLECTION}",
|
||||
timeout=(2.0, 15.0),
|
||||
)
|
||||
collection_info.raise_for_status()
|
||||
payload = collection_info.json()
|
||||
size = payload.get("result", {}).get("config", {}).get("params", {}).get("vectors", {}).get("size")
|
||||
distance = payload.get("result", {}).get("config", {}).get("params", {}).get("vectors", {}).get("distance")
|
||||
if size != DIM or distance != "Cosine":
|
||||
raise RuntimeError(f"unexpected Qdrant collection shape: size={size!r} distance={distance!r}")
|
||||
|
||||
verified: dict[str, dict[str, str]] = {}
|
||||
for family_name, family in records.items():
|
||||
payload = query_payload(
|
||||
embedding_client.embed_query(family["query"]),
|
||||
family["must"],
|
||||
)
|
||||
if payload.get("workspace_id") != WORKSPACE_ID:
|
||||
raise RuntimeError(f"{family_name} query leaked another workspace")
|
||||
if payload.get("record_key") != family["record"].id:
|
||||
raise RuntimeError(f"{family_name} query returned the wrong record key: {payload!r}")
|
||||
verified[family_name] = {
|
||||
"record_key": payload["record_key"],
|
||||
"kind": payload["kind"],
|
||||
}
|
||||
|
||||
tags = requests.get(f"{EMBEDDING}/api/tags", timeout=(2.0, 15.0))
|
||||
tags.raise_for_status()
|
||||
models = [entry.get("name") for entry in tags.json().get("models", []) if isinstance(entry, dict)]
|
||||
if MODEL not in models:
|
||||
raise RuntimeError(f"missing cached embedding model {MODEL}")
|
||||
|
||||
print(json.dumps({
|
||||
"mode": MODE,
|
||||
"collection": COLLECTION,
|
||||
"model": MODEL,
|
||||
"verified": verified,
|
||||
}, sort_keys=True))
|
||||
PY
|
||||
}
|
||||
|
||||
task13_semantic_seed_and_assert() {
|
||||
local output
|
||||
printf '== Ensure the semantic collection and seed schema/evidence/memory ==\n'
|
||||
output="$(task13_semantic_python_probe seed)"
|
||||
printf '%s\n' "$output" >>"$TASK13_LOG"
|
||||
grep -Fq '"schema"' <<<"$output" || task13_fail "schema semantic verification did not run"
|
||||
grep -Fq '"evidence"' <<<"$output" || task13_fail "evidence semantic verification did not run"
|
||||
grep -Fq '"memory"' <<<"$output" || task13_fail "memory semantic verification did not run"
|
||||
}
|
||||
|
||||
task13_semantic_verify_persistence() {
|
||||
local output
|
||||
printf '== Restart offline and prove semantic points plus model cache persist ==\n'
|
||||
task13_write_environment /fixtures/offline.git
|
||||
task13_compose_logged "offline semantic recreation" up --detach --force-recreate --wait --wait-timeout 120
|
||||
task13_wait_internal_embedding_model
|
||||
task13_registry_status >>"$TASK13_LOG" 2>&1 || true
|
||||
output="$(task13_semantic_python_probe verify)"
|
||||
printf '%s\n' "$output" >>"$TASK13_LOG"
|
||||
grep -Fq '"schema"' <<<"$output" || task13_fail "schema semantic persistence did not verify"
|
||||
grep -Fq '"evidence"' <<<"$output" || task13_fail "evidence semantic persistence did not verify"
|
||||
grep -Fq '"memory"' <<<"$output" || task13_fail "memory semantic persistence did not verify"
|
||||
}
|
||||
|
||||
task13_internal_semantic_smoke_main() {
|
||||
local qdrant_before embedding_before
|
||||
task13_initialize
|
||||
task13_require_tools
|
||||
task13_write_fixture_files
|
||||
task13_write_environment /fixtures/remote.git
|
||||
task13_seed_registry
|
||||
task13_start_stack
|
||||
task13_assert_project_ownership
|
||||
task13_assert_built_image_ownership
|
||||
task13_wait_internal_embedding_model
|
||||
qdrant_before="$(task13_service_mount_fingerprint qdrant)"
|
||||
embedding_before="$(task13_service_mount_fingerprint embedding)"
|
||||
task13_semantic_seed_and_assert
|
||||
task13_semantic_verify_persistence
|
||||
[[ "$(task13_service_mount_fingerprint qdrant)" == "$qdrant_before" ]] \
|
||||
|| task13_fail "offline recreation changed qdrant volume identity"
|
||||
[[ "$(task13_service_mount_fingerprint embedding)" == "$embedding_before" ]] \
|
||||
|| task13_fail "offline recreation changed embedding model cache volume identity"
|
||||
printf 'Task 13 internal semantic smoke passed.\n'
|
||||
}
|
||||
|
||||
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
|
||||
task13_supervise "$TASK13_SMOKE_TIMEOUT" "internal semantic smoke" task13_internal_semantic_smoke_main
|
||||
fi
|
||||
Reference in New Issue
Block a user