test: add internal semantic smoke

This commit is contained in:
2026-08-08 20:29:47 +02:00
parent 1b1213317b
commit 8240fa4472
4 changed files with 441 additions and 3 deletions
+298
View File
@@ -0,0 +1,298 @@
#!/usr/bin/env bash
# Live smoke for the mandatory internal semantic stack: disposable project/volumes, no host
# ports on semantic services, exact cleanup via Task 13 labels only, and offline persistence.
set -euo pipefail
root="$(cd "$(dirname "$0")/.." && pwd -P)"
# shellcheck source=./unified-deployment-smoke.sh
source "$root/scripts/unified-deployment-smoke.sh"
task13_wait_internal_embedding_model() {
printf '== Wait for the internal embedding model ==\n'
for _attempt in $(seq 1 30); do
if task13_compose exec -T core /opt/venv/bin/python - <<'PY' >>"$TASK13_LOG" 2>&1
import json
import urllib.request
import urllib.error
with urllib.request.urlopen("http://embedding:11434/api/tags", timeout=10) as response:
payload = json.load(response)
models = [entry.get("name") for entry in payload.get("models", []) if isinstance(entry, dict)]
if "qwen3-embedding:0.6b" not in models:
raise SystemExit(1)
request = urllib.request.Request(
"http://embedding:11434/api/embed",
data=json.dumps({"model": "qwen3-embedding:0.6b", "input": ["warm semantic smoke"]}).encode("utf-8"),
headers={"content-type": "application/json"},
method="POST",
)
try:
with urllib.request.urlopen(request, timeout=180) as response:
payload = json.load(response)
except urllib.error.URLError:
raise SystemExit(1)
embeddings = payload.get("embeddings")
raise SystemExit(0 if isinstance(embeddings, list) and len(embeddings) == 1 and len(embeddings[0]) == 1024 else 1)
PY
then
return 0
fi
sleep 1
done
task13_log_failure "internal embedding model readiness"
}
task13_semantic_python_probe() {
local mode="$1"
task13_compose exec -T core /opt/venv/bin/python - "$mode" <<'PY'
from __future__ import annotations
import hashlib
import json
import sys
from types import SimpleNamespace
import requests
from tht.adapters.vector.qdrant import QdrantVectorStore
from tht.ports.vector import VectorWriteRecord
from tht.vectorstore.embeddings import OllamaEmbeddings
from tht.vectorstore.records import VectorRecord
MODE = sys.argv[1]
WORKSPACE_ID = "task13-smoke"
WORKSPACE_REVISION = "b" * 40
COLLECTION = "task13-smoke"
QDRANT = "http://qdrant:6333"
EMBEDDING = "http://embedding:11434"
MODEL = "qwen3-embedding:0.6b"
DIM = 1024
def embedder() -> OllamaEmbeddings:
return OllamaEmbeddings(
SimpleNamespace(
base_url=EMBEDDING,
model=MODEL,
dim=DIM,
connect_timeout=2.0,
timeout=180.0,
batch_size=8,
)
)
def store() -> QdrantVectorStore:
return QdrantVectorStore(
base_url=QDRANT,
collection=COLLECTION,
workspace_id=WORKSPACE_ID,
workspace_revision=WORKSPACE_REVISION,
expected_dimension=DIM,
)
def content_hash(text: str) -> str:
return hashlib.sha256(text.encode("utf-8")).hexdigest()
def query_payload(vector: list[float], must: list[dict]) -> dict:
response = requests.post(
f"{QDRANT}/collections/{COLLECTION}/points/query",
json={
"vector": vector,
"limit": 1,
"with_payload": True,
"filter": {"must": must},
},
timeout=(2.0, 15.0),
)
response.raise_for_status()
payload = response.json()
points = payload.get("result", {}).get("points")
if not isinstance(points, list) or len(points) != 1:
raise RuntimeError(f"expected exactly one semantic point, got {payload!r}")
point = points[0]
result = point.get("payload")
if not isinstance(result, dict):
raise RuntimeError(f"missing payload in query result: {point!r}")
return result
records = {
"schema": {
"collection": "schema_records",
"record": VectorRecord(
id="schema_table:fact_task13",
kind="schema_table",
ref="fact_task13",
title="fact_task13",
content="Tabella fact_task13 con una riga dedicata allo smoke semantico interno.",
metadata={"table_name": "fact_task13"},
),
"query": "fact task13 smoke table",
"must": [
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
{"key": "kind", "match": {"value": "schema"}},
{"key": "record_kind", "match": {"value": "schema_table"}},
{"key": "record_key", "match": {"value": "schema_table:fact_task13"}},
],
},
"evidence": {
"collection": "evidence",
"record": VectorRecord(
id="evidence:task13-doc:0",
kind="evidence",
ref="task13-doc",
title="Task 13 Evidence",
content="Evidence dedicata allo smoke semantico interno con filtro esatto per generazione.",
metadata={
"document_id": "task13-doc",
"vector_generation": "gen:11111111111111111111111111111111",
"status": "published",
"tier": "gold",
"tables": ["fact_task13"],
"concepts": ["semantic smoke"],
},
),
"query": "semantic smoke evidence generation",
"must": [
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
{"key": "kind", "match": {"value": "evidence"}},
{"key": "document_id", "match": {"value": "task13-doc"}},
{"key": "vector_generation", "match": {"value": "gen:11111111111111111111111111111111"}},
],
},
"memory": {
"collection": "memory",
"record": VectorRecord(
id="mem-9000",
kind="memory",
ref="mem-9000",
title="Task 13 Memory",
content="Memoria riusabile per lo smoke semantico interno persistente.",
metadata={
"session_id": "task13-session",
"decision_seq": 9,
"subject": "semantic smoke memory",
"type": "concept_clarified",
"concepts": ["semantic smoke memory"],
},
),
"query": "semantic smoke memory reusable",
"must": [
{"key": "workspace_id", "match": {"value": WORKSPACE_ID}},
{"key": "kind", "match": {"value": "memory"}},
{"key": "record_kind", "match": {"value": "memory"}},
{"key": "record_key", "match": {"value": "mem-9000"}},
],
},
}
embedding_client = embedder()
vector_store = store()
if MODE == "seed":
for family in records.values():
record = family["record"]
vector_store.upsert(
family["collection"],
[
VectorWriteRecord(
record=record,
embedding=embedding_client.embed_query(record.content),
content_hash=content_hash(record.content),
)
],
)
collection_info = requests.get(
f"{QDRANT}/collections/{COLLECTION}",
timeout=(2.0, 15.0),
)
collection_info.raise_for_status()
payload = collection_info.json()
size = payload.get("result", {}).get("config", {}).get("params", {}).get("vectors", {}).get("size")
distance = payload.get("result", {}).get("config", {}).get("params", {}).get("vectors", {}).get("distance")
if size != DIM or distance != "Cosine":
raise RuntimeError(f"unexpected Qdrant collection shape: size={size!r} distance={distance!r}")
verified: dict[str, dict[str, str]] = {}
for family_name, family in records.items():
payload = query_payload(
embedding_client.embed_query(family["query"]),
family["must"],
)
if payload.get("workspace_id") != WORKSPACE_ID:
raise RuntimeError(f"{family_name} query leaked another workspace")
if payload.get("record_key") != family["record"].id:
raise RuntimeError(f"{family_name} query returned the wrong record key: {payload!r}")
verified[family_name] = {
"record_key": payload["record_key"],
"kind": payload["kind"],
}
tags = requests.get(f"{EMBEDDING}/api/tags", timeout=(2.0, 15.0))
tags.raise_for_status()
models = [entry.get("name") for entry in tags.json().get("models", []) if isinstance(entry, dict)]
if MODEL not in models:
raise RuntimeError(f"missing cached embedding model {MODEL}")
print(json.dumps({
"mode": MODE,
"collection": COLLECTION,
"model": MODEL,
"verified": verified,
}, sort_keys=True))
PY
}
task13_semantic_seed_and_assert() {
local output
printf '== Ensure the semantic collection and seed schema/evidence/memory ==\n'
output="$(task13_semantic_python_probe seed)"
printf '%s\n' "$output" >>"$TASK13_LOG"
grep -Fq '"schema"' <<<"$output" || task13_fail "schema semantic verification did not run"
grep -Fq '"evidence"' <<<"$output" || task13_fail "evidence semantic verification did not run"
grep -Fq '"memory"' <<<"$output" || task13_fail "memory semantic verification did not run"
}
task13_semantic_verify_persistence() {
local output
printf '== Restart offline and prove semantic points plus model cache persist ==\n'
task13_write_environment /fixtures/offline.git
task13_compose_logged "offline semantic recreation" up --detach --force-recreate --wait --wait-timeout 120
task13_wait_internal_embedding_model
task13_registry_status >>"$TASK13_LOG" 2>&1 || true
output="$(task13_semantic_python_probe verify)"
printf '%s\n' "$output" >>"$TASK13_LOG"
grep -Fq '"schema"' <<<"$output" || task13_fail "schema semantic persistence did not verify"
grep -Fq '"evidence"' <<<"$output" || task13_fail "evidence semantic persistence did not verify"
grep -Fq '"memory"' <<<"$output" || task13_fail "memory semantic persistence did not verify"
}
task13_internal_semantic_smoke_main() {
local qdrant_before embedding_before
task13_initialize
task13_require_tools
task13_write_fixture_files
task13_write_environment /fixtures/remote.git
task13_seed_registry
task13_start_stack
task13_assert_project_ownership
task13_assert_built_image_ownership
task13_wait_internal_embedding_model
qdrant_before="$(task13_service_mount_fingerprint qdrant)"
embedding_before="$(task13_service_mount_fingerprint embedding)"
task13_semantic_seed_and_assert
task13_semantic_verify_persistence
[[ "$(task13_service_mount_fingerprint qdrant)" == "$qdrant_before" ]] \
|| task13_fail "offline recreation changed qdrant volume identity"
[[ "$(task13_service_mount_fingerprint embedding)" == "$embedding_before" ]] \
|| task13_fail "offline recreation changed embedding model cache volume identity"
printf 'Task 13 internal semantic smoke passed.\n'
}
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
task13_supervise "$TASK13_SMOKE_TIMEOUT" "internal semantic smoke" task13_internal_semantic_smoke_main
fi
+62
View File
@@ -17,6 +17,25 @@ const workspace = parse(readFileSync(workspacePath, "utf8"));
const core = config.services?.core; const core = config.services?.core;
const frontend = config.services?.frontend; const frontend = config.services?.frontend;
if (!core || !frontend) throw new Error("fixture render must contain core and frontend"); if (!core || !frontend) throw new Error("fixture render must contain core and frontend");
const qdrant = config.services?.qdrant;
const embedding = config.services?.embedding;
const modelInit = config.services?.["embedding-model-init"];
if (!qdrant || !embedding || !modelInit) {
throw new Error("fixture render must contain the private semantic services");
}
for (const [name, service, expectedExpose] of [
["qdrant", qdrant, "6333"],
["embedding", embedding, "11434"],
] as const) {
if ((service.ports || []).length !== 0) throw new Error(`${name} must not publish host ports`);
if ((service.expose || []).join(",") !== expectedExpose) {
throw new Error(`${name} must expose only ${expectedExpose}`);
}
}
if ((modelInit.ports || []).length !== 0) {
throw new Error("embedding-model-init must not publish host ports");
}
const expected = { const expected = {
THT_WS_TASK13_SMOKE_DWH_TRANSPORT: "postgres_direct", THT_WS_TASK13_SMOKE_DWH_TRANSPORT: "postgres_direct",
@@ -34,6 +53,49 @@ for (const [name, value] of Object.entries(expected)) {
} }
} }
const semanticRuntime = {
THT_INTERNAL_QDRANT_URL: "http://qdrant:6333",
THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434",
THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b",
THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024",
};
for (const [name, value] of Object.entries(semanticRuntime)) {
if (core.environment?.[name] !== value) {
throw new Error(`core semantic runtime ${name} is ${JSON.stringify(core.environment?.[name])}, want ${JSON.stringify(value)}`);
}
if (Object.hasOwn(frontend.environment || {}, name)) {
throw new Error(`semantic runtime escaped to frontend: ${name}`);
}
}
for (const name of ["THT_VEC_REST_URL", "THT_VEC_WRITE_REST_URL", "THT_OLLAMA_URL"]) {
if (Object.hasOwn(core.environment || {}, name) && core.environment?.[name] !== "") {
throw new Error(`fixture render reintroduced external semantic binding ${name}`);
}
}
if (workspace.workspace?.id !== "task13-smoke") throw new Error("fixture workspace id changed");
if (workspace.semantic_index?.vector_store?.engine !== "qdrant") {
throw new Error("fixture workspace must use qdrant");
}
if (workspace.semantic_index?.vector_store?.collection !== workspace.workspace?.id) {
throw new Error("fixture workspace must dedicate one qdrant collection per workspace id");
}
if (workspace.semantic_index?.vector_store?.dimensions !== 1024) {
throw new Error("fixture workspace qdrant dimension changed");
}
if (workspace.semantic_index?.vector_store?.distance !== "cosine") {
throw new Error("fixture workspace qdrant distance changed");
}
if (workspace.semantic_index?.embedding?.provider !== "ollama_internal") {
throw new Error("fixture workspace must use internal ollama embeddings");
}
if (workspace.semantic_index?.embedding?.model !== "qwen3-embedding:0.6b") {
throw new Error("fixture workspace embedding model changed");
}
if (workspace.semantic_index?.embedding?.dimensions !== 1024) {
throw new Error("fixture workspace embedding dimension changed");
}
const bundle = config.secrets?.thothii_secrets; const bundle = config.secrets?.thothii_secrets;
const bundleSource = bundle?.file; const bundleSource = bundle?.file;
if (typeof bundleSource !== "string" || !statSync(bundleSource).isFile()) { if (typeof bundleSource !== "string" || !statSync(bundleSource).isFile()) {
+39 -2
View File
@@ -102,7 +102,13 @@ checker=(node --import "$tsx_loader" "$root/scripts/task13-runtime-fixture-check
} }
"${checker[@]}" "$rendered" "$workspace" "$profile" "${checker[@]}" "$rendered" "$workspace" "$profile"
for mutation in wrong-service wrong-value wrong-secret-mount; do for mutation in \
wrong-service \
wrong-value \
wrong-secret-mount \
wrong-qdrant-service \
wrong-embedding-service \
external-semantic-urls; do
mutated="$fixture/$mutation.json" mutated="$fixture/$mutation.json"
node - "$rendered" "$mutated" "$mutation" <<'NODE' node - "$rendered" "$mutated" "$mutation" <<'NODE'
const fs = require("fs"); const fs = require("fs");
@@ -115,8 +121,15 @@ if (mutation === "wrong-service") {
delete config.services.core.environment[name]; delete config.services.core.environment[name];
} else if (mutation === "wrong-value") { } else if (mutation === "wrong-value") {
config.services.core.environment.THT_WS_TASK13_SMOKE_DWH_HOST = "wrong.task13.invalid"; config.services.core.environment.THT_WS_TASK13_SMOKE_DWH_HOST = "wrong.task13.invalid";
} else { } else if (mutation === "wrong-secret-mount") {
config.secrets.thothii_secrets.file = source + ".missing"; config.secrets.thothii_secrets.file = source + ".missing";
} else if (mutation === "wrong-qdrant-service") {
config.services.core.environment.THT_INTERNAL_QDRANT_URL = "http://vector:6333";
} else if (mutation === "wrong-embedding-service") {
config.services.core.environment.THT_INTERNAL_EMBEDDING_URL = "http://ollama:11434";
} else if (mutation === "external-semantic-urls") {
config.services.core.environment.THT_INTERNAL_QDRANT_URL = "https://qdrant.example.test";
config.services.core.environment.THT_INTERNAL_EMBEDDING_URL = "https://embedding.example.test";
} }
fs.writeFileSync(destination, JSON.stringify(config)); fs.writeFileSync(destination, JSON.stringify(config));
NODE NODE
@@ -127,4 +140,28 @@ NODE
fi fi
done done
for mutation in collection-reuse dimension-change; do
mutated="$fixture/$mutation.yaml"
node - "$root/backend/package.json" "$workspace" "$mutated" "$mutation" <<'NODE'
const fs = require("fs");
const { createRequire } = require("module");
const requireFromBackend = createRequire(process.argv[2]);
const yaml = requireFromBackend("yaml");
const [source, destination, mutation] = process.argv.slice(3);
const workspace = yaml.parse(fs.readFileSync(source, "utf8"));
if (mutation === "collection-reuse") {
workspace.semantic_index.vector_store.collection = "shared-semantic";
} else if (mutation === "dimension-change") {
workspace.semantic_index.vector_store.dimensions = 1536;
workspace.semantic_index.embedding.dimensions = 1536;
}
fs.writeFileSync(destination, yaml.stringify(workspace));
NODE
if "${checker[@]}" "$rendered" "$mutated" "$profile" \
>"$fixture/$mutation.out" 2>"$fixture/$mutation.err"; then
echo "runtime fixture checker accepted workspace mutation: $mutation" >&2
exit 1
fi
done
echo "Task 13 $profile rendered runtime fixture contract passed." echo "Task 13 $profile rendered runtime fixture contract passed."
+42 -1
View File
@@ -304,6 +304,15 @@ services:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
labels: labels:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
qdrant:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding-model-init:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
networks: networks:
thothii: thothii:
labels: labels:
@@ -321,6 +330,12 @@ volumes:
sessions: sessions:
labels: labels:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
qdrant-data:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding-models:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
EOF EOF
chmod 0600 "$TASK13_OVERRIDE" chmod 0600 "$TASK13_OVERRIDE"
@@ -405,12 +420,28 @@ services:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
labels: labels:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
qdrant:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding-model-init:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
session-migrate: session-migrate:
image: $TASK13_CORE_IMAGE image: $TASK13_CORE_IMAGE
networks: networks:
thothii: thothii:
labels: labels:
io.thothii.task13.run: "$TASK13_RUN_ID" io.thothii.task13.run: "$TASK13_RUN_ID"
volumes:
qdrant-data:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
embedding-models:
labels:
io.thothii.task13.run: "$TASK13_RUN_ID"
EOF EOF
chmod 0600 "$TASK13_OVERRIDE" chmod 0600 "$TASK13_OVERRIDE"
@@ -744,6 +775,12 @@ task13_mount_fingerprint() {
| LC_ALL=C sort | LC_ALL=C sort
} }
task13_service_mount_fingerprint() {
local service="$1"
docker inspect --format '{{range .Mounts}}{{println .Destination "=" .Type ":" .Name}}{{end}}' \
"$(task13_compose ps -q "$service")" | LC_ALL=C sort
}
task13_prepare_persistence() { task13_prepare_persistence() {
task13_compose exec -T core sh -ceu ' task13_compose exec -T core sh -ceu '
printf %s settings-preserved > /data/settings/task13-settings printf %s settings-preserved > /data/settings/task13-settings
@@ -1368,6 +1405,9 @@ task13_self_test_public_timeout_contract() {
grep -Eq 'task13_supervise[[:space:]].*task13_smoke_main[[:space:]]+update' \ grep -Eq 'task13_supervise[[:space:]].*task13_smoke_main[[:space:]]+update' \
"$root/scripts/thothctl-update-smoke.sh" \ "$root/scripts/thothctl-update-smoke.sh" \
|| task13_fail "direct update smoke invocation lacks an internal supervisor" || task13_fail "direct update smoke invocation lacks an internal supervisor"
grep -Eq 'task13_supervise[[:space:]].*task13_internal_semantic_smoke_main' \
"$root/scripts/internal-semantic-smoke.sh" \
|| task13_fail "direct internal semantic smoke invocation lacks an internal supervisor"
} }
task13_self_test_windows_release_contract() { task13_self_test_windows_release_contract() {
@@ -1428,7 +1468,8 @@ task13_self_test_source_contract() {
registry_function='task13_start_''registry' registry_function='task13_start_''registry'
if rg -n 'docker[[:space:]]+(system[[:space:]]+)?prune' \ if rg -n 'docker[[:space:]]+(system[[:space:]]+)?prune' \
"$root/scripts/unified-deployment-smoke.sh" \ "$root/scripts/unified-deployment-smoke.sh" \
"$root/scripts/thothctl-update-smoke.sh" >/dev/null; then "$root/scripts/thothctl-update-smoke.sh" \
"$root/scripts/internal-semantic-smoke.sh" >/dev/null; then
task13_fail "Task 13 smoke scripts must never prune global Docker state" task13_fail "Task 13 smoke scripts must never prune global Docker state"
fi fi
! grep -Fq -- "$host_network" "$root/scripts/unified-deployment-smoke.sh" \ ! grep -Fq -- "$host_network" "$root/scripts/unified-deployment-smoke.sh" \