Files
AI-Profile-Router/compose.yaml
T

388 lines
11 KiB
YAML

name: mike-ai
x-llama-common: &llama-common
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
restart: "no"
profiles: [inference]
gpus: all
ipc: host
read_only: true
tmpfs:
- /tmp:size=1g,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
environment:
NVIDIA_DRIVER_CAPABILITIES: compute,utility
dns: ["${AI_DNS:-1.1.1.1}"]
networks:
inference:
aliases: [llama-upstream]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"]
interval: 10s
timeout: 5s
retries: 60
start_period: 30s
services:
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast
labels:
com.mike-ai.llama-profile: fast
environment:
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-fast
- --ctx-size
- "${FAST_CONTEXT:-76800}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-medium:
<<: *llama-common
container_name: mike-ai-llama-medium
labels:
com.mike-ai.llama-profile: medium
environment:
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-medium
- --ctx-size
- "${MEDIUM_CONTEXT:-94208}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0
- --split-mode
- none
llama-long:
<<: *llama-common
container_name: mike-ai-llama-long
labels:
com.mike-ai.llama-profile: long
environment:
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-long
- --ctx-size
- "${LONG_CONTEXT:-131072}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --override-tensor
- blk.([0-9]|1[0-1]).ffn_.*=CPU
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- q4_0
- --spec-draft-type-v
- q4_0
llama-experimental:
<<: *llama-common
container_name: mike-ai-llama-experimental
labels:
com.mike-ai.llama-profile: experimental
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${EXPERIMENTAL_MODEL_FILE:?EXPERIMENTAL_MODEL_FILE is required}"
- --alias
- qwen-experimental
- --ctx-size
- "${EXPERIMENTAL_CONTEXT:-76800}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --device
- CUDA0
- --split-mode
- none
profile-controller:
build: ./platform/docker/profile-controller
image: mike-ai/profile-controller:local
container_name: mike-ai-profile-controller
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=16m"]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,long,experimental
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"]
interval: 10s
timeout: 3s
retries: 10
router:
build:
context: .
dockerfile: platform/docker/router/Dockerfile
image: mike-ai/profile-router:local
container_name: mike-ai-router
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=256m"]
volumes:
- ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro
- router-state:/var/lib/mike-ai-profile-router
- router-images:/data/images
environment:
ROUTER_HOST: 0.0.0.0
ROUTER_PORT: "8081"
ROUTER_AUTH_MODE: required
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
UPSTREAM_URL: http://llama-upstream:8080
PROFILE_CONTROL_URL: http://profile-controller:8090
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600"
IMAGE_DIR: /data/images
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "false"
ENABLE_TTS: "false"
ENABLE_STT: "false"
ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8081:8081"
networks: [frontend, control, inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
# The entrypoint fixes ownership of fresh named volumes and immediately
# drops to uid/gid 10002 via gosu before starting the router. Without this
# narrowly scoped capabilities a clean installation cannot initialize the
# volumes and then switch to its unprivileged runtime identity.
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"]
interval: 5s
timeout: 3s
retries: 24
start_period: 5s
depends_on:
profile-controller:
condition: service_healthy
open-webui:
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
container_name: mike-ai-open-webui
restart: unless-stopped
volumes:
- open-webui-data:/app/backend/data
environment:
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
OLLAMA_BASE_URL: ""
OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
# Seed native MCP connections on a fresh Open WebUI database. Secrets
# stay inside the tool containers, so these internal URLs need no keys.
TOOL_SERVER_CONNECTIONS: >-
[{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (lokal)","description":"Kompakte Websuche und Quellenvergleich"}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Home-Assistant-Werkzeuge mit serverseitigem Token"}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"ARR (lokal)","description":"Sonarr- und Radarr-Werkzeuge"}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (lokal, read-only)","description":"Begrenzte Unraid-Diagnose"}}]
DO_NOT_TRACK: "true"
SCARF_NO_ANALYTICS: "true"
ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080"
dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend, tools]
depends_on:
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
networks:
frontend:
internal: false
ipam:
config: [{subnet: 172.30.10.0/24}]
control:
internal: true
ipam:
config: [{subnet: 172.30.20.0/24}]
inference:
internal: true
ipam:
config: [{subnet: 172.30.30.0/24}]
tools:
external: true
name: mike-ai-tools
volumes:
open-webui-data:
router-state:
router-images: