From 320d9ea74e3491f5d016506e16cfe1c80034e7f3 Mon Sep 17 00:00:00 2001 From: mptyl Date: Sat, 8 Aug 2026 18:38:42 +0200 Subject: [PATCH] feat: run qdrant and ollama inside thothii --- .../task-7-report.md | 54 +++++++ compose.yaml | 60 ++++++- deploy/compose.embedding-gpu.yaml | 7 + deploy/env/local.env.example | 6 +- deploy/env/server.env.example | 6 +- docker/core.Dockerfile | 4 +- docker/embedding-model-init.sh | 81 ++++++++++ scripts/run-stack.sh | 11 +- scripts/test-default-compose.sh | 74 +++++++-- scripts/test-internal-semantic-compose.sh | 148 ++++++++++++++++++ scripts/test-unified-compose.sh | 79 +++++++++- 11 files changed, 502 insertions(+), 28 deletions(-) create mode 100644 .superpowers/sdd/2026-08-08-internal-qdrant-ollama/task-7-report.md create mode 100644 deploy/compose.embedding-gpu.yaml create mode 100755 docker/embedding-model-init.sh create mode 100755 scripts/test-internal-semantic-compose.sh diff --git a/.superpowers/sdd/2026-08-08-internal-qdrant-ollama/task-7-report.md b/.superpowers/sdd/2026-08-08-internal-qdrant-ollama/task-7-report.md new file mode 100644 index 00000000..6d756d85 --- /dev/null +++ b/.superpowers/sdd/2026-08-08-internal-qdrant-ollama/task-7-report.md @@ -0,0 +1,54 @@ +# Task 7 report — mandatory Qdrant and Ollama Compose services + +Date: 2026-08-08 + +Status: completed + +Summary: + +- Added mandatory private `qdrant`, `embedding`, and `embedding-model-init` services to the base Compose stack. +- Pinned Qdrant `v1.18.2` and Ollama `0.32.0` by immutable multi-arch digest. +- Persisted Qdrant storage in `qdrant-data` and Ollama model cache in `embedding-models`. +- Wired `core` to fixed internal semantic endpoints: + - `THT_INTERNAL_QDRANT_URL=http://qdrant:6333` + - `THT_INTERNAL_EMBEDDING_URL=http://embedding:11434` + - `THT_INTERNAL_EMBEDDING_MODEL=qwen3-embedding:0.6b` + - `THT_INTERNAL_EMBEDDING_DIMENSIONS=1024` +- Removed external vector / embedding endpoint requirements from the local and server env examples. +- Added an idempotent Ollama model bootstrap script that: + - waits up to a bounded deadline for `/api/tags` + - skips `ollama pull` when the model is already cached + - pulls `qwen3-embedding:0.6b` only when needed + - verifies the model appears in `/api/tags` after pull +- Added optional GPU override file `deploy/compose.embedding-gpu.yaml`; base Compose remains CPU-only. +- Updated `scripts/run-stack.sh` so the GPU override is included only when `THOTH_ENABLE_EMBEDDING_GPU=1`. + +Verification: + +- RED confirmed before implementation: + - `./scripts/test-default-compose.sh` failed on missing required services. + - `./scripts/test-unified-compose.sh` failed on missing required services. + - `./scripts/test-internal-semantic-compose.sh` failed because the GPU override file did not exist. +- GREEN after implementation: + - `./scripts/test-default-compose.sh` + - `./scripts/test-unified-compose.sh` + - `./scripts/test-internal-semantic-compose.sh` + - `git diff --check` +- Additional shell verification: + - `scripts/run-stack.sh --wait` includes only base + local Compose files by default. + - `THOTH_ENABLE_EMBEDDING_GPU=1 scripts/run-stack.sh --wait` adds `deploy/compose.embedding-gpu.yaml`. + +Resolved image digests: + +- `qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c` +- `ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a` + +Self-review: + +- The first bootstrap-script draft depended on tools not guaranteed inside the Ollama image. This was corrected after image inspection; the final script uses only confirmed image tools (`bash`, `ollama`, `grep`) plus raw HTTP over `/dev/tcp`. +- The server overlay intentionally replaces most named core volumes with bind mounts, so the unified contract was tightened to require named semantic-cache volumes there while preserving the local/base named-volume checks. + +Concerns: + +- The model bootstrap waits for Ollama readiness and verifies cache state, but the first real cold-start will still take time to download `qwen3-embedding:0.6b`. +- The GPU override requests generic Docker GPU capability only; actual GPU availability remains host/runtime dependent and intentionally stays opt-in. diff --git a/compose.yaml b/compose.yaml index ff25b0cc..1caa666e 100644 --- a/compose.yaml +++ b/compose.yaml @@ -24,10 +24,11 @@ services: THT_SECRETS_FILE: /run/secrets/thothii.secrets THT_DB_NAME: ${THT_DB_NAME:-} THT_DWH_REST_URL: ${THT_DWH_REST_URL:-} - THT_VEC_REST_URL: ${THT_VEC_REST_URL:-} - THT_VEC_WRITE_REST_URL: ${THT_VEC_WRITE_REST_URL:-} - THT_OLLAMA_URL: ${THT_OLLAMA_URL:-} THT_LLM_URL: ${THT_LLM_URL:-} + THT_INTERNAL_QDRANT_URL: http://qdrant:6333 + THT_INTERNAL_EMBEDDING_URL: http://embedding:11434 + THT_INTERNAL_EMBEDDING_MODEL: qwen3-embedding:0.6b + THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024" MAX_PI_PROCESSES: ${MAX_PI_PROCESSES:-4} volumes: - settings:/data/settings @@ -46,6 +47,11 @@ services: timeout: 3s retries: 5 start_period: 30s + depends_on: + qdrant: + condition: service_healthy + embedding-model-init: + condition: service_completed_successfully networks: - thothii @@ -69,6 +75,52 @@ services: networks: - thothii + qdrant: + image: qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c + expose: + - "6333" + volumes: + - qdrant-data:/qdrant/storage + healthcheck: + test: + - CMD-SHELL + - > + /usr/bin/bash -lc "exec 3<>/dev/tcp/127.0.0.1/6333 && + printf 'GET /healthz HTTP/1.1\r\nHost: 127.0.0.1\r\nConnection: close\r\n\r\n' >&3 && + grep -q '200 OK' <&3" + interval: 15s + timeout: 3s + retries: 10 + start_period: 10s + networks: + - thothii + + embedding: + image: ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a + command: ["serve"] + expose: + - "11434" + volumes: + - embedding-models:/root/.ollama + networks: + - thothii + + embedding-model-init: + image: ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a + entrypoint: ["/usr/bin/bash", "/opt/thoth/embedding-model-init.sh"] + environment: + OLLAMA_BASE_URL: http://embedding:11434 + OLLAMA_MODEL: qwen3-embedding:0.6b + OLLAMA_WAIT_TIMEOUT_SEC: "180" + volumes: + - embedding-models:/root/.ollama + - ./docker/embedding-model-init.sh:/opt/thoth/embedding-model-init.sh:ro + depends_on: + embedding: + condition: service_started + networks: + - thothii + networks: thothii: @@ -77,6 +129,8 @@ volumes: pi-state: workspace-registry: sessions: + qdrant-data: + embedding-models: secrets: thothii_secrets: diff --git a/deploy/compose.embedding-gpu.yaml b/deploy/compose.embedding-gpu.yaml new file mode 100644 index 00000000..8b7731b8 --- /dev/null +++ b/deploy/compose.embedding-gpu.yaml @@ -0,0 +1,7 @@ +services: + embedding: + deploy: + resources: + reservations: + devices: + - capabilities: ["gpu"] diff --git a/deploy/env/local.env.example b/deploy/env/local.env.example index ddd341a6..712ceca2 100644 --- a/deploy/env/local.env.example +++ b/deploy/env/local.env.example @@ -13,7 +13,7 @@ THT_WORKSPACE_GIT_AUTHOR_EMAIL=thoth-workspace-registry@example.invalid THT_DB_NAME=warehouse THT_DWH_REST_URL=https://dwh.example.invalid -THT_VEC_REST_URL=https://vector.example.invalid -THT_VEC_WRITE_REST_URL=https://vector-write.example.invalid -THT_OLLAMA_URL=https://embeddings.example.invalid THT_LLM_URL=https://llm.example.invalid + +# Optional explicit GPU override for Linux hosts that expose a Docker-compatible GPU device. +# THOTH_ENABLE_EMBEDDING_GPU=1 diff --git a/deploy/env/server.env.example b/deploy/env/server.env.example index 2c5dbd77..5e162402 100644 --- a/deploy/env/server.env.example +++ b/deploy/env/server.env.example @@ -18,11 +18,11 @@ THT_WORKSPACE_GIT_AUTHOR_EMAIL=thoth-workspace-registry@example.invalid THT_DB_NAME=warehouse THT_DWH_REST_URL=https://dwh.example.invalid -THT_VEC_REST_URL=https://vector.example.invalid -THT_VEC_WRITE_REST_URL=https://vector-write.example.invalid -THT_OLLAMA_URL=https://embeddings.example.invalid THT_LLM_URL=https://llm.example.invalid +# Optional explicit GPU override for Linux hosts that expose a Docker-compatible GPU device. +# THOTH_ENABLE_EMBEDDING_GPU=1 + # Public server session storage. Values are endpoints, roles, or protected source-file paths. THT_SESSION_DB_HOST=sessions-db.example.invalid THT_SESSION_DB_PORT=5432 diff --git a/docker/core.Dockerfile b/docker/core.Dockerfile index 57ca8a9c..a891eb1d 100644 --- a/docker/core.Dockerfile +++ b/docker/core.Dockerfile @@ -91,10 +91,10 @@ ENV PATH="/opt/venv/bin:/usr/local/bin:$PATH" \ HOME=/home/thoth COPY scripts/verify-line-endings.sh /usr/local/bin/verify-line-endings -COPY docker/core-entrypoint.sh docker/session-migrate.sh docker/ensure-pi-trust.mjs /app/docker/ +COPY docker/core-entrypoint.sh docker/session-migrate.sh docker/ensure-pi-trust.mjs docker/embedding-model-init.sh /app/docker/ COPY docker/smoke/core-smoke.sh /app/docker/smoke/core-smoke.sh RUN /usr/local/bin/verify-line-endings /app/docker \ - && chmod +x /app/docker/core-entrypoint.sh /app/docker/session-migrate.sh /app/docker/smoke/core-smoke.sh + && chmod +x /app/docker/core-entrypoint.sh /app/docker/session-migrate.sh /app/docker/embedding-model-init.sh /app/docker/smoke/core-smoke.sh WORKDIR /app/backend USER thoth diff --git a/docker/embedding-model-init.sh b/docker/embedding-model-init.sh new file mode 100755 index 00000000..312794f1 --- /dev/null +++ b/docker/embedding-model-init.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +set -euo pipefail + +ollama_base_url=${OLLAMA_BASE_URL:-http://embedding:11434} +ollama_model=${OLLAMA_MODEL:-qwen3-embedding:0.6b} +wait_timeout_sec=${OLLAMA_WAIT_TIMEOUT_SEC:-180} + +case "$wait_timeout_sec" in + ''|*[!0-9]*) + echo "OLLAMA_WAIT_TIMEOUT_SEC must be an integer number of seconds" >&2 + exit 1 + ;; +esac + +case "$ollama_base_url" in + http://*) + host_and_path=${ollama_base_url#http://} + ;; + *) + echo "OLLAMA_BASE_URL must use http://" >&2 + exit 1 + ;; +esac + +host_port=${host_and_path%%/*} +ollama_host=${host_port%%:*} +ollama_port=${host_port##*:} +if [[ "$host_port" == "$ollama_host" ]]; then + ollama_port=80 +fi + +deadline=$((SECONDS + wait_timeout_sec)) +export OLLAMA_HOST="$ollama_base_url" + +fetch_tags() { + local response body + response=$( + exec 3<>"/dev/tcp/$ollama_host/$ollama_port" + printf 'GET /api/tags HTTP/1.1\r\nHost: %s\r\nConnection: close\r\n\r\n' "$ollama_host" >&3 + cat <&3 + ) || return 1 + [[ "$response" == *$' 200 '* || "$response" == HTTP/1.1$' 200'* || "$response" == HTTP/1.0$' 200'* ]] || return 1 + body=${response#*$'\r\n\r\n'} + if [[ "$body" == "$response" ]]; then + body=${response#*$'\n\n'} + fi + printf '%s' "$body" +} + +model_present() { + local compact_json + compact_json=$(printf '%s' "$1" | tr -d '[:space:]') + grep -Fq "\"name\":\"$ollama_model\"" <<<"$compact_json" +} + +wait_for_tags() { + local tags_json + while (( SECONDS <= deadline )); do + if tags_json=$(fetch_tags 2>/dev/null); then + printf '%s' "$tags_json" + return 0 + fi + sleep 1 + done + echo "timed out waiting for Ollama tags at $ollama_base_url/api/tags" >&2 + return 1 +} + +tags_json=$(wait_for_tags) +if model_present "$tags_json"; then + echo "embedding model already cached: $ollama_model" + exit 0 +fi + +ollama pull "$ollama_model" +tags_json=$(fetch_tags) +model_present "$tags_json" || { + echo "embedding model missing after pull: $ollama_model" >&2 + exit 1 +} +echo "embedding model ready: $ollama_model" diff --git a/scripts/run-stack.sh b/scripts/run-stack.sh index 3bf95077..bc879fa1 100755 --- a/scripts/run-stack.sh +++ b/scripts/run-stack.sh @@ -1,8 +1,8 @@ #!/usr/bin/env bash # run-stack.sh — avvia lo stack Compose locale di ThothII in primo piano. # -# Il core include Pi; DWH, vector DB, embedding e LLM sono endpoint esterni configurati -# in deploy/env/local.env. Non richiede un eseguibile Pi sull'host. +# Il core include Pi; DWH e LLM restano endpoint esterni configurati in deploy/env/local.env. +# Qdrant e Ollama embedding sono servizi Compose privati. Non richiede un eseguibile Pi sull'host. # # Preparazione: cp deploy/env/local.env.example deploy/env/local.env e compilare i valori. # Uso: ./scripts/run-stack.sh [argomenti aggiuntivi per docker compose up] @@ -16,5 +16,10 @@ LOCAL_ENV_FILE="${THT_LOCAL_ENV_FILE:-$ROOT/deploy/env/local.env}" exit 1 } +compose_files=(-f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml") +if [[ "${THOTH_ENABLE_EMBEDDING_GPU:-0}" == "1" ]]; then + compose_files+=(-f "$ROOT/deploy/compose.embedding-gpu.yaml") +fi + exec docker compose --env-file "$LOCAL_ENV_FILE" \ - -f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml" up --build "$@" + "${compose_files[@]}" up --build "$@" diff --git a/scripts/test-default-compose.sh b/scripts/test-default-compose.sh index aea682f8..a88b44ae 100755 --- a/scripts/test-default-compose.sh +++ b/scripts/test-default-compose.sh @@ -11,16 +11,70 @@ rendered=$(mktemp) trap 'rm -f "$rendered"' EXIT HUP INT TERM docker compose --env-file deploy/env/local.env.example \ - -f compose.yaml -f deploy/compose.local.yaml config >"$rendered" + -f compose.yaml -f deploy/compose.local.yaml config --format json >"$rendered" -grep -q '^ core:' "$rendered" -grep -q '^ frontend:' "$rendered" -grep -q 'host_ip: 127.0.0.1' "$rendered" -grep -q 'AUTH_MODE: none' "$rendered" -grep -q 'THT_WORKSPACE_INSTALLATION_ID: local' "$rendered" -if grep -Eqi 'omics_portal|chirone|localllm_default|/home/chirone' "$rendered"; then - echo "default Compose contains application-specific coupling" >&2 - exit 1 -fi +node - "$rendered" <<'NODE' +const fs = require("fs"); + +const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8")); +const services = Object.keys(config.services).sort(); +if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") { + throw new Error(`unexpected service set: ${services.join(",")}`); +} +if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) { + throw new Error("default Compose contains application-specific coupling"); +} +for (const volume of ["settings", "pi-state", "workspace-registry", "sessions", "qdrant-data", "embedding-models"]) { + if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`); +} +const core = config.services.core; +const frontend = config.services.frontend; +const qdrant = config.services.qdrant; +const embedding = config.services.embedding; +const modelInit = config.services["embedding-model-init"]; +if (!frontend.ports?.some((port) => port.host_ip === "127.0.0.1")) { + throw new Error("local frontend must publish a loopback port"); +} +for (const service of [qdrant, embedding, modelInit]) { + if ((service.ports || []).length !== 0) throw new Error("private semantic services must not publish host ports"); +} +if ((qdrant.expose || []).join(",") !== "6333") throw new Error("qdrant must expose only 6333"); +if ((embedding.expose || []).join(",") !== "11434") throw new Error("embedding must expose only 11434"); +if (qdrant.image !== "qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c") { + throw new Error("qdrant image must be pinned by version and digest"); +} +if (embedding.image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") { + throw new Error("embedding image must be pinned by version and digest"); +} +if (modelInit.image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") { + throw new Error("embedding-model-init image must be pinned by version and digest"); +} +const env = core.environment || {}; +for (const [key, value] of Object.entries({ + AUTH_MODE: "none", + THT_WORKSPACE_INSTALLATION_ID: "local", + THT_INTERNAL_QDRANT_URL: "http://qdrant:6333", + THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434", + THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b", + THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024", +})) { + if (env[key] !== value) throw new Error(`unexpected core ${key}: ${env[key]}`); +} +for (const forbidden of ["THT_VEC_REST_URL", "THT_VEC_WRITE_REST_URL", "THT_OLLAMA_URL"]) { + if (Object.hasOwn(env, forbidden) && env[forbidden] !== "") { + throw new Error(`core must not require external semantic binding ${forbidden}`); + } +} +const depends = core.depends_on || {}; +if (depends.qdrant?.condition !== "service_healthy") { + throw new Error("core must wait for qdrant health"); +} +if (depends["embedding-model-init"]?.condition !== "service_completed_successfully") { + throw new Error("core must wait for embedding-model-init success"); +} +if (JSON.stringify(embedding).includes('"devices"')) { + throw new Error("base embedding service must stay CPU-only"); +} +NODE echo "default Compose contract passed." diff --git a/scripts/test-internal-semantic-compose.sh b/scripts/test-internal-semantic-compose.sh new file mode 100755 index 00000000..66b67f04 --- /dev/null +++ b/scripts/test-internal-semantic-compose.sh @@ -0,0 +1,148 @@ +#!/usr/bin/env bash +set -euo pipefail + +cd "$(dirname "$0")/.." + +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +docker compose --env-file deploy/env/local.env.example \ + -f compose.yaml -f deploy/compose.local.yaml -f deploy/compose.embedding-gpu.yaml \ + config --format json >"$tmp/compose-gpu.json" + +node - "$tmp/compose-gpu.json" <<'NODE' +const fs = require("fs"); + +const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8")); +const devices = config.services.embedding?.deploy?.resources?.reservations?.devices; +if (!Array.isArray(devices) || devices.length !== 1) { + throw new Error("GPU override must add one embedding device reservation"); +} +const [device] = devices; +if (JSON.stringify(device.capabilities) !== JSON.stringify(["gpu"])) { + throw new Error("GPU override must request gpu capability only"); +} +NODE + +mock_bin="$tmp/mock-bin" +mkdir -p "$mock_bin" + +cat >"$mock_bin/ollama" <<'EOF' +#!/usr/bin/env bash +set -euo pipefail + +state_dir=${MOCK_STATE_DIR:?} +printf '%s\n' "$*" >>"$state_dir/ollama-calls" +if [[ "$1" != "pull" ]]; then + echo "unexpected ollama command: $*" >&2 + exit 1 +fi +cat >"$state_dir/tags.json" <"$tmp/mock-tags-server.py" <<'PY' +import http.server +import os +from pathlib import Path + +state_dir = Path(os.environ["MOCK_STATE_DIR"]) +port_file = Path(os.environ["MOCK_PORT_FILE"]) + + +class Handler(http.server.BaseHTTPRequestHandler): + def do_GET(self): + if self.path != "/api/tags": + self.send_response(404) + self.end_headers() + return + count_file = state_dir / "curl-count" + count = int(count_file.read_text() or "0") if count_file.exists() else 0 + count += 1 + count_file.write_text(str(count)) + fail_until = int((state_dir / "fail-until").read_text()) if (state_dir / "fail-until").exists() else 0 + if count <= fail_until: + self.send_response(503) + self.end_headers() + self.wfile.write(b'{"models":[]}') + return + payload = (state_dir / "tags.json").read_bytes() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, format, *args): + return + + +server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), Handler) +port_file.write_text(str(server.server_address[1])) +server.serve_forever() +PY + +start_server() { + local state_dir=$1 + local port_file="$state_dir/port" + MOCK_STATE_DIR="$state_dir" MOCK_PORT_FILE="$port_file" \ + python3 "$tmp/mock-tags-server.py" >/dev/null 2>&1 & + local server_pid=$! + for _ in $(seq 1 50); do + [[ -f "$port_file" ]] && break + sleep 0.1 + done + [[ -f "$port_file" ]] || { + echo "mock tags server did not start" >&2 + kill "$server_pid" >/dev/null 2>&1 || true + exit 1 + } + printf '%s %s\n' "$server_pid" "$(cat "$port_file")" +} + +run_cached() { + local state_dir="$tmp/cached" + mkdir -p "$state_dir" + cat >"$state_dir/tags.json" <<'JSON' +{"models":[{"name":"qwen3-embedding:0.6b"}]} +JSON + read -r server_pid port < <(start_server "$state_dir") + PATH="$mock_bin:$PATH" \ + MOCK_STATE_DIR="$state_dir" \ + OLLAMA_BASE_URL="http://127.0.0.1:$port" \ + OLLAMA_MODEL="qwen3-embedding:0.6b" \ + OLLAMA_WAIT_TIMEOUT_SEC=2 \ + ./docker/embedding-model-init.sh + kill "$server_pid" >/dev/null 2>&1 || true + wait "$server_pid" 2>/dev/null || true + if [[ -e "$state_dir/ollama-calls" ]]; then + echo "cached bootstrap must not call ollama pull" >&2 + exit 1 + fi +} + +run_pull() { + local state_dir="$tmp/pull" + mkdir -p "$state_dir" + cat >"$state_dir/tags.json" <<'JSON' +{"models":[]} +JSON + printf '2' >"$state_dir/fail-until" + read -r server_pid port < <(start_server "$state_dir") + PATH="$mock_bin:$PATH" \ + MOCK_STATE_DIR="$state_dir" \ + OLLAMA_BASE_URL="http://127.0.0.1:$port" \ + OLLAMA_MODEL="qwen3-embedding:0.6b" \ + OLLAMA_WAIT_TIMEOUT_SEC=5 \ + ./docker/embedding-model-init.sh + kill "$server_pid" >/dev/null 2>&1 || true + wait "$server_pid" 2>/dev/null || true + grep -qx 'pull qwen3-embedding:0.6b' "$state_dir/ollama-calls" +} + +run_cached +run_pull + +echo "internal semantic Compose/script contracts passed." diff --git a/scripts/test-unified-compose.sh b/scripts/test-unified-compose.sh index 4c78205f..79f4ff29 100755 --- a/scripts/test-unified-compose.sh +++ b/scripts/test-unified-compose.sh @@ -25,17 +25,65 @@ const fs = require("fs"); const [configPath, profile] = process.argv.slice(2); const config = JSON.parse(fs.readFileSync(configPath, "utf8")); const services = Object.keys(config.services).sort(); -if (services.join(",") !== "core,frontend") throw new Error("mandatory stack must be core,frontend"); +if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") { + throw new Error("mandatory stack must include core, frontend, qdrant, embedding, and embedding-model-init"); +} if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) { throw new Error("forbidden application coupling"); } if (!config.networks || !config.networks.thothii) throw new Error("base stack must define the thothii network"); -if (!Object.hasOwn(config.services.core.environment || {}, "THT_LLM_URL")) { +for (const volume of ["qdrant-data", "embedding-models"]) { + if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`); +} +if (profile === "local") { + for (const volume of ["settings", "pi-state", "workspace-registry", "sessions"]) { + if (!config.volumes || !config.volumes[volume]) throw new Error(`missing local required volume: ${volume}`); + } +} +const coreEnv = config.services.core.environment || {}; +if (!Object.hasOwn(coreEnv, "THT_LLM_URL")) { throw new Error("core must expose a generic THT_LLM_URL endpoint contract"); } +for (const [key, value] of Object.entries({ + THT_INTERNAL_QDRANT_URL: "http://qdrant:6333", + THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434", + THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b", + THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024", +})) { + if (coreEnv[key] !== value) throw new Error(`unexpected core ${key}: ${coreEnv[key]}`); +} +for (const forbidden of ["THT_VEC_REST_URL", "THT_VEC_WRITE_REST_URL", "THT_OLLAMA_URL"]) { + if (Object.hasOwn(coreEnv, forbidden) && coreEnv[forbidden] !== "") { + throw new Error(`core must not require external semantic binding ${forbidden}`); + } +} if (/docker\.sock|\/var\/run\/docker|docker[-_]?daemon/i.test(JSON.stringify(config.services))) { throw new Error("Compose must not mount the Docker socket or daemon"); } +if (config.services.qdrant.image !== "qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c") { + throw new Error("qdrant image must be pinned by version and digest"); +} +for (const serviceName of ["embedding", "embedding-model-init"]) { + if (config.services[serviceName].image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") { + throw new Error(`${serviceName} image must be pinned by version and digest`); + } +} +for (const serviceName of ["qdrant", "embedding", "embedding-model-init"]) { + if ((config.services[serviceName].ports || []).length !== 0) { + throw new Error(`${serviceName} must not publish a host port`); + } +} +if ((config.services.qdrant.expose || []).join(",") !== "6333") throw new Error("qdrant must expose only 6333"); +if ((config.services.embedding.expose || []).join(",") !== "11434") throw new Error("embedding must expose only 11434"); +if (config.services.core.depends_on?.qdrant?.condition !== "service_healthy") { + throw new Error("core must wait for qdrant health"); +} +if (config.services.core.depends_on?.["embedding-model-init"]?.condition !== "service_completed_successfully") { + throw new Error("core must wait for embedding-model-init success"); +} +if (JSON.stringify(config.services.embedding).includes('"devices"')) { + throw new Error("base embedding service must stay CPU-only"); +} const piAuthMounts = (config.services.core.volumes || []).filter( (mount) => mount.target === "/home/thoth/.pi/agent/auth.json", ); @@ -108,17 +156,40 @@ const fs = require("fs"); const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8")); const services = Object.keys(config.services).sort(); -if (services.join(",") !== "core,frontend") throw new Error("mandatory stack must be core,frontend"); +if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") { + throw new Error("mandatory stack must include core, frontend, qdrant, embedding, and embedding-model-init"); +} if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) { throw new Error("forbidden application coupling"); } if (!config.networks || !config.networks.thothii) throw new Error("base stack must define the thothii network"); -for (const volume of ["settings", "pi-state", "workspace-registry", "sessions"]) { +for (const volume of ["settings", "pi-state", "workspace-registry", "sessions", "qdrant-data", "embedding-models"]) { if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`); } if (!Object.hasOwn(config.services.core.environment || {}, "THT_LLM_URL")) { throw new Error("core must expose a generic THT_LLM_URL endpoint contract"); } +for (const [key, value] of Object.entries({ + THT_INTERNAL_QDRANT_URL: "http://qdrant:6333", + THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434", + THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b", + THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024", +})) { + if (config.services.core.environment?.[key] !== value) { + throw new Error(`unexpected core ${key}: ${config.services.core.environment?.[key]}`); + } +} +for (const serviceName of ["qdrant", "embedding", "embedding-model-init"]) { + if ((config.services[serviceName].ports || []).length !== 0) { + throw new Error(`${serviceName} must not publish a host port`); + } +} +if (config.services.core.depends_on?.qdrant?.condition !== "service_healthy") { + throw new Error("core must wait for qdrant health"); +} +if (config.services.core.depends_on?.["embedding-model-init"]?.condition !== "service_completed_successfully") { + throw new Error("core must wait for embedding-model-init success"); +} if (/docker\.sock|\/var\/run\/docker|docker[-_]?daemon/i.test(JSON.stringify(config.services))) { throw new Error("Compose must not mount the Docker socket or daemon"); }