feat: run qdrant and ollama inside thothii
This commit is contained in:
@@ -1,8 +1,8 @@
|
||||
#!/usr/bin/env bash
|
||||
# run-stack.sh — avvia lo stack Compose locale di ThothII in primo piano.
|
||||
#
|
||||
# Il core include Pi; DWH, vector DB, embedding e LLM sono endpoint esterni configurati
|
||||
# in deploy/env/local.env. Non richiede un eseguibile Pi sull'host.
|
||||
# Il core include Pi; DWH e LLM restano endpoint esterni configurati in deploy/env/local.env.
|
||||
# Qdrant e Ollama embedding sono servizi Compose privati. Non richiede un eseguibile Pi sull'host.
|
||||
#
|
||||
# Preparazione: cp deploy/env/local.env.example deploy/env/local.env e compilare i valori.
|
||||
# Uso: ./scripts/run-stack.sh [argomenti aggiuntivi per docker compose up]
|
||||
@@ -16,5 +16,10 @@ LOCAL_ENV_FILE="${THT_LOCAL_ENV_FILE:-$ROOT/deploy/env/local.env}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
compose_files=(-f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml")
|
||||
if [[ "${THOTH_ENABLE_EMBEDDING_GPU:-0}" == "1" ]]; then
|
||||
compose_files+=(-f "$ROOT/deploy/compose.embedding-gpu.yaml")
|
||||
fi
|
||||
|
||||
exec docker compose --env-file "$LOCAL_ENV_FILE" \
|
||||
-f "$ROOT/compose.yaml" -f "$ROOT/deploy/compose.local.yaml" up --build "$@"
|
||||
"${compose_files[@]}" up --build "$@"
|
||||
|
||||
@@ -11,16 +11,70 @@ rendered=$(mktemp)
|
||||
trap 'rm -f "$rendered"' EXIT HUP INT TERM
|
||||
|
||||
docker compose --env-file deploy/env/local.env.example \
|
||||
-f compose.yaml -f deploy/compose.local.yaml config >"$rendered"
|
||||
-f compose.yaml -f deploy/compose.local.yaml config --format json >"$rendered"
|
||||
|
||||
grep -q '^ core:' "$rendered"
|
||||
grep -q '^ frontend:' "$rendered"
|
||||
grep -q 'host_ip: 127.0.0.1' "$rendered"
|
||||
grep -q 'AUTH_MODE: none' "$rendered"
|
||||
grep -q 'THT_WORKSPACE_INSTALLATION_ID: local' "$rendered"
|
||||
if grep -Eqi 'omics_portal|chirone|localllm_default|/home/chirone' "$rendered"; then
|
||||
echo "default Compose contains application-specific coupling" >&2
|
||||
exit 1
|
||||
fi
|
||||
node - "$rendered" <<'NODE'
|
||||
const fs = require("fs");
|
||||
|
||||
const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8"));
|
||||
const services = Object.keys(config.services).sort();
|
||||
if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") {
|
||||
throw new Error(`unexpected service set: ${services.join(",")}`);
|
||||
}
|
||||
if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) {
|
||||
throw new Error("default Compose contains application-specific coupling");
|
||||
}
|
||||
for (const volume of ["settings", "pi-state", "workspace-registry", "sessions", "qdrant-data", "embedding-models"]) {
|
||||
if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`);
|
||||
}
|
||||
const core = config.services.core;
|
||||
const frontend = config.services.frontend;
|
||||
const qdrant = config.services.qdrant;
|
||||
const embedding = config.services.embedding;
|
||||
const modelInit = config.services["embedding-model-init"];
|
||||
if (!frontend.ports?.some((port) => port.host_ip === "127.0.0.1")) {
|
||||
throw new Error("local frontend must publish a loopback port");
|
||||
}
|
||||
for (const service of [qdrant, embedding, modelInit]) {
|
||||
if ((service.ports || []).length !== 0) throw new Error("private semantic services must not publish host ports");
|
||||
}
|
||||
if ((qdrant.expose || []).join(",") !== "6333") throw new Error("qdrant must expose only 6333");
|
||||
if ((embedding.expose || []).join(",") !== "11434") throw new Error("embedding must expose only 11434");
|
||||
if (qdrant.image !== "qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c") {
|
||||
throw new Error("qdrant image must be pinned by version and digest");
|
||||
}
|
||||
if (embedding.image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") {
|
||||
throw new Error("embedding image must be pinned by version and digest");
|
||||
}
|
||||
if (modelInit.image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") {
|
||||
throw new Error("embedding-model-init image must be pinned by version and digest");
|
||||
}
|
||||
const env = core.environment || {};
|
||||
for (const [key, value] of Object.entries({
|
||||
AUTH_MODE: "none",
|
||||
THT_WORKSPACE_INSTALLATION_ID: "local",
|
||||
THT_INTERNAL_QDRANT_URL: "http://qdrant:6333",
|
||||
THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434",
|
||||
THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b",
|
||||
THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024",
|
||||
})) {
|
||||
if (env[key] !== value) throw new Error(`unexpected core ${key}: ${env[key]}`);
|
||||
}
|
||||
for (const forbidden of ["THT_VEC_REST_URL", "THT_VEC_WRITE_REST_URL", "THT_OLLAMA_URL"]) {
|
||||
if (Object.hasOwn(env, forbidden) && env[forbidden] !== "") {
|
||||
throw new Error(`core must not require external semantic binding ${forbidden}`);
|
||||
}
|
||||
}
|
||||
const depends = core.depends_on || {};
|
||||
if (depends.qdrant?.condition !== "service_healthy") {
|
||||
throw new Error("core must wait for qdrant health");
|
||||
}
|
||||
if (depends["embedding-model-init"]?.condition !== "service_completed_successfully") {
|
||||
throw new Error("core must wait for embedding-model-init success");
|
||||
}
|
||||
if (JSON.stringify(embedding).includes('"devices"')) {
|
||||
throw new Error("base embedding service must stay CPU-only");
|
||||
}
|
||||
NODE
|
||||
|
||||
echo "default Compose contract passed."
|
||||
|
||||
Executable
+148
@@ -0,0 +1,148 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
tmp=$(mktemp -d)
|
||||
trap 'rm -rf "$tmp"' EXIT HUP INT TERM
|
||||
|
||||
docker compose --env-file deploy/env/local.env.example \
|
||||
-f compose.yaml -f deploy/compose.local.yaml -f deploy/compose.embedding-gpu.yaml \
|
||||
config --format json >"$tmp/compose-gpu.json"
|
||||
|
||||
node - "$tmp/compose-gpu.json" <<'NODE'
|
||||
const fs = require("fs");
|
||||
|
||||
const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8"));
|
||||
const devices = config.services.embedding?.deploy?.resources?.reservations?.devices;
|
||||
if (!Array.isArray(devices) || devices.length !== 1) {
|
||||
throw new Error("GPU override must add one embedding device reservation");
|
||||
}
|
||||
const [device] = devices;
|
||||
if (JSON.stringify(device.capabilities) !== JSON.stringify(["gpu"])) {
|
||||
throw new Error("GPU override must request gpu capability only");
|
||||
}
|
||||
NODE
|
||||
|
||||
mock_bin="$tmp/mock-bin"
|
||||
mkdir -p "$mock_bin"
|
||||
|
||||
cat >"$mock_bin/ollama" <<'EOF'
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
state_dir=${MOCK_STATE_DIR:?}
|
||||
printf '%s\n' "$*" >>"$state_dir/ollama-calls"
|
||||
if [[ "$1" != "pull" ]]; then
|
||||
echo "unexpected ollama command: $*" >&2
|
||||
exit 1
|
||||
fi
|
||||
cat >"$state_dir/tags.json" <<JSON
|
||||
{"models":[{"name":"${2}"}]}
|
||||
JSON
|
||||
EOF
|
||||
chmod +x "$mock_bin/ollama"
|
||||
|
||||
cat >"$tmp/mock-tags-server.py" <<'PY'
|
||||
import http.server
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
state_dir = Path(os.environ["MOCK_STATE_DIR"])
|
||||
port_file = Path(os.environ["MOCK_PORT_FILE"])
|
||||
|
||||
|
||||
class Handler(http.server.BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
if self.path != "/api/tags":
|
||||
self.send_response(404)
|
||||
self.end_headers()
|
||||
return
|
||||
count_file = state_dir / "curl-count"
|
||||
count = int(count_file.read_text() or "0") if count_file.exists() else 0
|
||||
count += 1
|
||||
count_file.write_text(str(count))
|
||||
fail_until = int((state_dir / "fail-until").read_text()) if (state_dir / "fail-until").exists() else 0
|
||||
if count <= fail_until:
|
||||
self.send_response(503)
|
||||
self.end_headers()
|
||||
self.wfile.write(b'{"models":[]}')
|
||||
return
|
||||
payload = (state_dir / "tags.json").read_bytes()
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(payload)))
|
||||
self.end_headers()
|
||||
self.wfile.write(payload)
|
||||
|
||||
def log_message(self, format, *args):
|
||||
return
|
||||
|
||||
|
||||
server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
||||
port_file.write_text(str(server.server_address[1]))
|
||||
server.serve_forever()
|
||||
PY
|
||||
|
||||
start_server() {
|
||||
local state_dir=$1
|
||||
local port_file="$state_dir/port"
|
||||
MOCK_STATE_DIR="$state_dir" MOCK_PORT_FILE="$port_file" \
|
||||
python3 "$tmp/mock-tags-server.py" >/dev/null 2>&1 &
|
||||
local server_pid=$!
|
||||
for _ in $(seq 1 50); do
|
||||
[[ -f "$port_file" ]] && break
|
||||
sleep 0.1
|
||||
done
|
||||
[[ -f "$port_file" ]] || {
|
||||
echo "mock tags server did not start" >&2
|
||||
kill "$server_pid" >/dev/null 2>&1 || true
|
||||
exit 1
|
||||
}
|
||||
printf '%s %s\n' "$server_pid" "$(cat "$port_file")"
|
||||
}
|
||||
|
||||
run_cached() {
|
||||
local state_dir="$tmp/cached"
|
||||
mkdir -p "$state_dir"
|
||||
cat >"$state_dir/tags.json" <<'JSON'
|
||||
{"models":[{"name":"qwen3-embedding:0.6b"}]}
|
||||
JSON
|
||||
read -r server_pid port < <(start_server "$state_dir")
|
||||
PATH="$mock_bin:$PATH" \
|
||||
MOCK_STATE_DIR="$state_dir" \
|
||||
OLLAMA_BASE_URL="http://127.0.0.1:$port" \
|
||||
OLLAMA_MODEL="qwen3-embedding:0.6b" \
|
||||
OLLAMA_WAIT_TIMEOUT_SEC=2 \
|
||||
./docker/embedding-model-init.sh
|
||||
kill "$server_pid" >/dev/null 2>&1 || true
|
||||
wait "$server_pid" 2>/dev/null || true
|
||||
if [[ -e "$state_dir/ollama-calls" ]]; then
|
||||
echo "cached bootstrap must not call ollama pull" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
run_pull() {
|
||||
local state_dir="$tmp/pull"
|
||||
mkdir -p "$state_dir"
|
||||
cat >"$state_dir/tags.json" <<'JSON'
|
||||
{"models":[]}
|
||||
JSON
|
||||
printf '2' >"$state_dir/fail-until"
|
||||
read -r server_pid port < <(start_server "$state_dir")
|
||||
PATH="$mock_bin:$PATH" \
|
||||
MOCK_STATE_DIR="$state_dir" \
|
||||
OLLAMA_BASE_URL="http://127.0.0.1:$port" \
|
||||
OLLAMA_MODEL="qwen3-embedding:0.6b" \
|
||||
OLLAMA_WAIT_TIMEOUT_SEC=5 \
|
||||
./docker/embedding-model-init.sh
|
||||
kill "$server_pid" >/dev/null 2>&1 || true
|
||||
wait "$server_pid" 2>/dev/null || true
|
||||
grep -qx 'pull qwen3-embedding:0.6b' "$state_dir/ollama-calls"
|
||||
}
|
||||
|
||||
run_cached
|
||||
run_pull
|
||||
|
||||
echo "internal semantic Compose/script contracts passed."
|
||||
@@ -25,17 +25,65 @@ const fs = require("fs");
|
||||
const [configPath, profile] = process.argv.slice(2);
|
||||
const config = JSON.parse(fs.readFileSync(configPath, "utf8"));
|
||||
const services = Object.keys(config.services).sort();
|
||||
if (services.join(",") !== "core,frontend") throw new Error("mandatory stack must be core,frontend");
|
||||
if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") {
|
||||
throw new Error("mandatory stack must include core, frontend, qdrant, embedding, and embedding-model-init");
|
||||
}
|
||||
if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) {
|
||||
throw new Error("forbidden application coupling");
|
||||
}
|
||||
if (!config.networks || !config.networks.thothii) throw new Error("base stack must define the thothii network");
|
||||
if (!Object.hasOwn(config.services.core.environment || {}, "THT_LLM_URL")) {
|
||||
for (const volume of ["qdrant-data", "embedding-models"]) {
|
||||
if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`);
|
||||
}
|
||||
if (profile === "local") {
|
||||
for (const volume of ["settings", "pi-state", "workspace-registry", "sessions"]) {
|
||||
if (!config.volumes || !config.volumes[volume]) throw new Error(`missing local required volume: ${volume}`);
|
||||
}
|
||||
}
|
||||
const coreEnv = config.services.core.environment || {};
|
||||
if (!Object.hasOwn(coreEnv, "THT_LLM_URL")) {
|
||||
throw new Error("core must expose a generic THT_LLM_URL endpoint contract");
|
||||
}
|
||||
for (const [key, value] of Object.entries({
|
||||
THT_INTERNAL_QDRANT_URL: "http://qdrant:6333",
|
||||
THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434",
|
||||
THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b",
|
||||
THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024",
|
||||
})) {
|
||||
if (coreEnv[key] !== value) throw new Error(`unexpected core ${key}: ${coreEnv[key]}`);
|
||||
}
|
||||
for (const forbidden of ["THT_VEC_REST_URL", "THT_VEC_WRITE_REST_URL", "THT_OLLAMA_URL"]) {
|
||||
if (Object.hasOwn(coreEnv, forbidden) && coreEnv[forbidden] !== "") {
|
||||
throw new Error(`core must not require external semantic binding ${forbidden}`);
|
||||
}
|
||||
}
|
||||
if (/docker\.sock|\/var\/run\/docker|docker[-_]?daemon/i.test(JSON.stringify(config.services))) {
|
||||
throw new Error("Compose must not mount the Docker socket or daemon");
|
||||
}
|
||||
if (config.services.qdrant.image !== "qdrant/qdrant:v1.18.2@sha256:75eab8c4ba42096724fdcfde8b4de0b5713d529dde32f285a1f86fdcb2c9e50c") {
|
||||
throw new Error("qdrant image must be pinned by version and digest");
|
||||
}
|
||||
for (const serviceName of ["embedding", "embedding-model-init"]) {
|
||||
if (config.services[serviceName].image !== "ollama/ollama:0.32.0@sha256:57f573b47f1f71ebb445789f279fe3e596a8beab182f7cf486db9205bad87c5a") {
|
||||
throw new Error(`${serviceName} image must be pinned by version and digest`);
|
||||
}
|
||||
}
|
||||
for (const serviceName of ["qdrant", "embedding", "embedding-model-init"]) {
|
||||
if ((config.services[serviceName].ports || []).length !== 0) {
|
||||
throw new Error(`${serviceName} must not publish a host port`);
|
||||
}
|
||||
}
|
||||
if ((config.services.qdrant.expose || []).join(",") !== "6333") throw new Error("qdrant must expose only 6333");
|
||||
if ((config.services.embedding.expose || []).join(",") !== "11434") throw new Error("embedding must expose only 11434");
|
||||
if (config.services.core.depends_on?.qdrant?.condition !== "service_healthy") {
|
||||
throw new Error("core must wait for qdrant health");
|
||||
}
|
||||
if (config.services.core.depends_on?.["embedding-model-init"]?.condition !== "service_completed_successfully") {
|
||||
throw new Error("core must wait for embedding-model-init success");
|
||||
}
|
||||
if (JSON.stringify(config.services.embedding).includes('"devices"')) {
|
||||
throw new Error("base embedding service must stay CPU-only");
|
||||
}
|
||||
const piAuthMounts = (config.services.core.volumes || []).filter(
|
||||
(mount) => mount.target === "/home/thoth/.pi/agent/auth.json",
|
||||
);
|
||||
@@ -108,17 +156,40 @@ const fs = require("fs");
|
||||
|
||||
const config = JSON.parse(fs.readFileSync(process.argv[2], "utf8"));
|
||||
const services = Object.keys(config.services).sort();
|
||||
if (services.join(",") !== "core,frontend") throw new Error("mandatory stack must be core,frontend");
|
||||
if (services.join(",") !== "core,embedding,embedding-model-init,frontend,qdrant") {
|
||||
throw new Error("mandatory stack must include core, frontend, qdrant, embedding, and embedding-model-init");
|
||||
}
|
||||
if (/omics_portal|chirone|localllm_default|\/home\/chirone/i.test(JSON.stringify(config))) {
|
||||
throw new Error("forbidden application coupling");
|
||||
}
|
||||
if (!config.networks || !config.networks.thothii) throw new Error("base stack must define the thothii network");
|
||||
for (const volume of ["settings", "pi-state", "workspace-registry", "sessions"]) {
|
||||
for (const volume of ["settings", "pi-state", "workspace-registry", "sessions", "qdrant-data", "embedding-models"]) {
|
||||
if (!config.volumes || !config.volumes[volume]) throw new Error(`missing required volume: ${volume}`);
|
||||
}
|
||||
if (!Object.hasOwn(config.services.core.environment || {}, "THT_LLM_URL")) {
|
||||
throw new Error("core must expose a generic THT_LLM_URL endpoint contract");
|
||||
}
|
||||
for (const [key, value] of Object.entries({
|
||||
THT_INTERNAL_QDRANT_URL: "http://qdrant:6333",
|
||||
THT_INTERNAL_EMBEDDING_URL: "http://embedding:11434",
|
||||
THT_INTERNAL_EMBEDDING_MODEL: "qwen3-embedding:0.6b",
|
||||
THT_INTERNAL_EMBEDDING_DIMENSIONS: "1024",
|
||||
})) {
|
||||
if (config.services.core.environment?.[key] !== value) {
|
||||
throw new Error(`unexpected core ${key}: ${config.services.core.environment?.[key]}`);
|
||||
}
|
||||
}
|
||||
for (const serviceName of ["qdrant", "embedding", "embedding-model-init"]) {
|
||||
if ((config.services[serviceName].ports || []).length !== 0) {
|
||||
throw new Error(`${serviceName} must not publish a host port`);
|
||||
}
|
||||
}
|
||||
if (config.services.core.depends_on?.qdrant?.condition !== "service_healthy") {
|
||||
throw new Error("core must wait for qdrant health");
|
||||
}
|
||||
if (config.services.core.depends_on?.["embedding-model-init"]?.condition !== "service_completed_successfully") {
|
||||
throw new Error("core must wait for embedding-model-init success");
|
||||
}
|
||||
if (/docker\.sock|\/var\/run\/docker|docker[-_]?daemon/i.test(JSON.stringify(config.services))) {
|
||||
throw new Error("Compose must not mount the Docker socket or daemon");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user