Add CPU embeddings for OpenClaw memory

This commit is contained in:
Mikei386
2026-09-21 08:31:45 +02:00
parent de3f8fd7aa
commit 302b08e051
13 changed files with 231 additions and 7 deletions
+59
View File
@@ -88,6 +88,65 @@ services:
cap_drop: [ALL]
security_opt: ["no-new-privileges:true"]
# Dedicated CPU-only embedding endpoint for OpenClaw memory search. It
# shares the WireGuard namespace so the API is reachable only through
# Athena's private VPN address, without consuming scarce GPU memory.
embedding:
build:
context: .
dockerfile: platform/docker/llama-cpp-cpu/Dockerfile
args:
LLAMA_CPP_COMMIT: ${LLAMA_CPP_COMMIT:-b29c606}
image: ${LLAMA_CPU_IMAGE:-mike-ai/llama.cpp-cpu:local}
container_name: mike-ai-embedding
restart: unless-stopped
network_mode: "service:wireguard-gateway"
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
command:
- --model
- "/models/${EMBEDDING_MODEL_FILE:?EMBEDDING_MODEL_FILE is required}"
- --alias
- embeddinggemma
- --embedding
- --pooling
- mean
- --ctx-size
- "4096"
- --batch-size
# OpenClaw's chunks plus embedding control tokens must fit in both the
# logical and physical batch. 256 rejected valid index chunks.
- "4096"
- --ubatch-size
- "1024"
- --parallel
- "4"
- --threads
- "${EMBEDDING_THREADS:-8}"
- --threads-batch
- "${EMBEDDING_THREADS:-8}"
- --n-gpu-layers
- "0"
- --host
- 0.0.0.0
- --port
- "8082"
- --no-ui
depends_on:
wireguard-gateway:
condition: service_healthy
cap_drop: [ALL]
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8082/health"]
interval: 10s
timeout: 5s
retries: 30
start_period: 10s
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast