Files
AI-Profile-Router/compose.yaml
T

521 lines
15 KiB
YAML

name: mike-ai
x-llama-common: &llama-common
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
restart: "no"
profiles: [inference]
gpus: all
ipc: host
read_only: true
tmpfs:
- /tmp:size=1g,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
environment:
NVIDIA_DRIVER_CAPABILITIES: compute,utility
dns: ["${AI_DNS:-1.1.1.1}"]
networks:
inference:
aliases: [llama-upstream]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"]
interval: 10s
timeout: 5s
retries: 60
start_period: 30s
services:
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast
labels:
com.mike-ai.llama-profile: fast
environment:
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-fast
- --ctx-size
- "${FAST_CONTEXT:-76800}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-medium:
<<: *llama-common
container_name: mike-ai-llama-medium
labels:
com.mike-ai.llama-profile: medium
environment:
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-medium
- --ctx-size
- "${MEDIUM_CONTEXT:-160000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-large:
<<: *llama-common
container_name: mike-ai-llama-large
labels:
com.mike-ai.llama-profile: large
environment:
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-large
- --ctx-size
- "${LARGE_CONTEXT:-192000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${LARGE_TENSOR_SPLIT:-86,14}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
llama-ultra:
<<: *llama-common
container_name: mike-ai-llama-ultra
labels:
com.mike-ai.llama-profile: ultra
environment:
NVIDIA_VISIBLE_DEVICES: ${ULTRA_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
- --alias
- qwen-ultra
- --ctx-size
- "${ULTRA_CONTEXT:-262144}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${ULTRA_TENSOR_SPLIT:-80,20}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-experimental:
<<: *llama-common
container_name: mike-ai-llama-experimental
labels:
com.mike-ai.llama-profile: experimental
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${EXPERIMENTAL_MODEL_FILE:?EXPERIMENTAL_MODEL_FILE is required}"
- --alias
- qwen-experimental
- --ctx-size
- "${EXPERIMENTAL_CONTEXT:-76800}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --device
- CUDA0
- --split-mode
- none
profile-controller:
build: ./platform/docker/profile-controller
image: mike-ai/profile-controller:local
container_name: mike-ai-profile-controller
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=16m"]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,large,ultra,experimental
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"]
interval: 10s
timeout: 3s
retries: 10
router:
build:
context: .
dockerfile: platform/docker/router/Dockerfile
image: mike-ai/profile-router:local
container_name: mike-ai-router
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=256m"]
volumes:
- ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro
- router-state:/var/lib/mike-ai-profile-router
- router-images:/data/images
environment:
ROUTER_HOST: 0.0.0.0
ROUTER_PORT: "8081"
ROUTER_AUTH_MODE: required
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
UPSTREAM_URL: http://llama-upstream:8080
PROFILE_CONTROL_URL: http://profile-controller:8090
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600"
IMAGE_DIR: /data/images
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "false"
ENABLE_TTS: "true"
TTS_WORKER_URL: http://piper:8085
TTS_MODEL: piper
TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "false"
ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8081:8081"
networks: [frontend, control, inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
# The entrypoint fixes ownership of fresh named volumes and immediately
# drops to uid/gid 10002 via gosu before starting the router. Without this
# narrowly scoped capabilities a clean installation cannot initialize the
# volumes and then switch to its unprivileged runtime identity.
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"]
interval: 5s
timeout: 3s
retries: 24
start_period: 5s
depends_on:
profile-controller:
condition: service_healthy
piper:
condition: service_healthy
piper:
build:
context: platform/docker/piper
args:
PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
image: mike-ai/piper:local
container_name: mike-ai-piper
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- piper-data:/data
environment:
PIPER_DATA_DIR: /data
PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
PIPER_VOICE_ALIAS: alloy
PIPER_HOST: 0.0.0.0
PIPER_PORT: "8085"
PIPER_MAX_TEXT_CHARS: "8000"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 30
start_period: 120s
open-webui:
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
container_name: mike-ai-open-webui
restart: unless-stopped
volumes:
- open-webui-data:/app/backend/data
# Upstream-supported static customization hooks. Keeping these files in
# the repository makes the global dark theme reproducible and update-safe.
- ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
environment:
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
DEFAULT_MODELS: mikeai-medium
OLLAMA_BASE_URL: ""
OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
AUDIO_TTS_ENGINE: openai
AUDIO_TTS_OPENAI_API_BASE_URL: http://router:8081/v1
AUDIO_TTS_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
AUDIO_TTS_MODEL: piper
AUDIO_TTS_VOICE: alloy
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
# Seed native MCP connections on a fresh Open WebUI database. Secrets
# stay inside the tool containers, so these internal URLs need no keys.
TOOL_SERVER_CONNECTIONS: >-
[{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (öffentlich, read-only)","description":"Für aktuelle öffentliche Internetdaten, Quellenprüfung, GitHub/Hugging Face und Produktsuche. Nicht für Home Assistant, Medienverwaltung oder NAS-Diagnose."}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Nur für Home-Assistant-Entitäten, Zustände, Historie, Automationen, Dashboards und HA-Diagnose. Nicht für Unraid, Sonarr/Radarr oder allgemeine Websuche."}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"Sonarr und Radarr (lokal)","description":"Nur für verwaltete Serien/Filme, fehlende Episoden, Queue und Suche über konfigurierte Indexer. Keine allgemeine Websuche; Schreibaktionen benötigen Vorschau und Freigabe."}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (Systemdiagnose)","description":"Nur für Unraid-Host, Array, Datenträger, Docker-Container, Shares, Netzwerk, UPS und Systemlogs. Nicht für Home Assistant oder Medieninhalte; Standardzugriff read-only."}}]
DO_NOT_TRACK: "true"
SCARF_NO_ANALYTICS: "true"
ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080"
dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend, tools]
depends_on:
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
networks:
frontend:
internal: false
ipam:
config: [{subnet: 172.30.10.0/24}]
control:
internal: true
ipam:
config: [{subnet: 172.30.20.0/24}]
inference:
internal: true
ipam:
config: [{subnet: 172.30.30.0/24}]
tools:
external: true
name: mike-ai-tools
volumes:
open-webui-data:
piper-data:
router-state:
router-images: