914 lines
30 KiB
YAML
914 lines
30 KiB
YAML
name: mike-ai
|
|
|
|
x-llama-common: &llama-common
|
|
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
|
|
restart: "no"
|
|
profiles: [inference]
|
|
gpus: all
|
|
ipc: host
|
|
read_only: true
|
|
tmpfs:
|
|
- /tmp:size=1g,mode=1777
|
|
volumes:
|
|
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
|
|
environment:
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
dns: ["${AI_DNS:-1.1.1.1}"]
|
|
networks:
|
|
inference:
|
|
aliases: [llama-upstream]
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
healthcheck:
|
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 60
|
|
start_period: 30s
|
|
|
|
services:
|
|
wireguard-gateway:
|
|
build: ./platform/docker/wireguard-gateway
|
|
image: mike-ai/wireguard-gateway:local
|
|
container_name: mike-ai-wireguard-gateway
|
|
restart: unless-stopped
|
|
cap_add: [NET_ADMIN]
|
|
devices:
|
|
- /dev/net/tun:/dev/net/tun
|
|
sysctls:
|
|
net.ipv4.ip_forward: "1"
|
|
net.ipv4.conf.all.src_valid_mark: "1"
|
|
net.ipv6.conf.all.forwarding: "1"
|
|
read_only: true
|
|
tmpfs:
|
|
- /run:size=16m,mode=0755
|
|
- /tmp:size=16m,mode=1777
|
|
volumes:
|
|
- "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro"
|
|
networks:
|
|
frontend:
|
|
ipv4_address: 172.30.10.254
|
|
tools:
|
|
ipv4_address: 172.30.40.254
|
|
tools-egress:
|
|
ipv4_address: 172.30.50.254
|
|
security_opt: ["no-new-privileges:true"]
|
|
healthcheck:
|
|
test: [CMD, /usr/local/sbin/mike-ai-wireguard-healthcheck]
|
|
interval: 10s
|
|
timeout: 3s
|
|
retries: 12
|
|
start_period: 10s
|
|
|
|
llama-fast:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-fast
|
|
labels:
|
|
com.mike-ai.llama-profile: fast
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
# CUDA0 remains the exclusive text-model device. The projector is kept
|
|
# on the secondary card so vision does not consume the 5080 context
|
|
# budget.
|
|
MTMD_BACKEND_DEVICE: CUDA1
|
|
command:
|
|
- --model
|
|
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
|
|
- --mmproj
|
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
|
- --mmproj-offload
|
|
- --mmproj-device
|
|
- CUDA1
|
|
- --alias
|
|
- qwen-fast
|
|
- --ctx-size
|
|
- "${FAST_CONTEXT:-76800}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --threads
|
|
- "${LLAMA_THREADS:-6}"
|
|
- --threads-batch
|
|
- "${LLAMA_THREADS_BATCH:-6}"
|
|
- --batch-size
|
|
- "64"
|
|
- --ubatch-size
|
|
- "32"
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --temperature
|
|
- "0.2"
|
|
- --top-p
|
|
- "0.8"
|
|
- --top-k
|
|
- "20"
|
|
- --device
|
|
- CUDA0
|
|
- --split-mode
|
|
- none
|
|
- --spec-type
|
|
- draft-mtp
|
|
- --spec-draft-n-max
|
|
- "2"
|
|
- --spec-draft-type-k
|
|
- f16
|
|
- --spec-draft-type-v
|
|
- f16
|
|
|
|
llama-medium:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-medium
|
|
labels:
|
|
com.mike-ai.llama-profile: medium
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
# Keep the language model split unchanged while placing the complete
|
|
# multimodal projector on the secondary RTX 3060.
|
|
MTMD_BACKEND_DEVICE: CUDA1
|
|
command:
|
|
- --model
|
|
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
|
|
- --mmproj
|
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
|
- --mmproj-offload
|
|
- --mmproj-device
|
|
- CUDA1
|
|
- --alias
|
|
- qwen-medium
|
|
- --ctx-size
|
|
- "${MEDIUM_CONTEXT:-160000}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --threads
|
|
- "${LLAMA_THREADS:-6}"
|
|
- --threads-batch
|
|
- "${LLAMA_THREADS_BATCH:-6}"
|
|
- --batch-size
|
|
- "64"
|
|
- --ubatch-size
|
|
- "32"
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
# Bound each individual thinking phase. Long agent jobs can still use
|
|
# many phases around tool calls, but one degenerate reasoning loop can
|
|
# no longer consume the complete response budget indefinitely.
|
|
- --reasoning-budget
|
|
- "8192"
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --temperature
|
|
# Qwen3.8's official thinking-mode sampler. The former 0.2 setting was
|
|
# overly deterministic and could lock reasoning into verbatim loops.
|
|
- "1.0"
|
|
- --top-p
|
|
- "0.95"
|
|
- --top-k
|
|
- "20"
|
|
- --device
|
|
- CUDA0,CUDA1
|
|
- --main-gpu
|
|
- "0"
|
|
- --split-mode
|
|
- layer
|
|
- --tensor-split
|
|
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
|
- --spec-type
|
|
- draft-mtp
|
|
- --spec-draft-n-max
|
|
- "3"
|
|
- --spec-draft-type-k
|
|
- f16
|
|
- --spec-draft-type-v
|
|
- f16
|
|
- --spec-draft-p-min
|
|
- "0.05"
|
|
|
|
llama-large:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-large
|
|
labels:
|
|
com.mike-ai.llama-profile: large
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
MTMD_BACKEND_DEVICE: CUDA1
|
|
command:
|
|
- --model
|
|
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
|
- --mmproj
|
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
|
- --mmproj-offload
|
|
- --mmproj-device
|
|
- CUDA1
|
|
- --alias
|
|
- qwen-large
|
|
- --ctx-size
|
|
- "${LARGE_CONTEXT:-192000}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --threads
|
|
- "${LLAMA_THREADS:-6}"
|
|
- --threads-batch
|
|
- "${LLAMA_THREADS_BATCH:-6}"
|
|
- --batch-size
|
|
- "64"
|
|
- --ubatch-size
|
|
- "32"
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --temperature
|
|
- "0.2"
|
|
- --top-p
|
|
- "0.8"
|
|
- --top-k
|
|
- "20"
|
|
- --device
|
|
- CUDA0,CUDA1
|
|
- --main-gpu
|
|
- "0"
|
|
- --split-mode
|
|
- layer
|
|
- --tensor-split
|
|
- "${LARGE_TENSOR_SPLIT:-86,14}"
|
|
- --spec-type
|
|
- draft-mtp
|
|
- --spec-draft-n-max
|
|
- "3"
|
|
- --spec-draft-type-k
|
|
- f16
|
|
- --spec-draft-type-v
|
|
- f16
|
|
|
|
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
|
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
|
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
|
llama-ultra:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-ultra
|
|
labels:
|
|
com.mike-ai.llama-profile: ultra
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${ULTRA_GPU_DEVICES:-0,1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
command:
|
|
- --model
|
|
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
|
|
- --alias
|
|
- qwen-ultra
|
|
- --ctx-size
|
|
- "${ULTRA_CONTEXT:-262144}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --threads
|
|
- "${LLAMA_THREADS:-6}"
|
|
- --threads-batch
|
|
- "${LLAMA_THREADS_BATCH:-6}"
|
|
- --batch-size
|
|
- "64"
|
|
- --ubatch-size
|
|
- "32"
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --temperature
|
|
- "0.2"
|
|
- --top-p
|
|
- "0.8"
|
|
- --top-k
|
|
- "20"
|
|
- --device
|
|
- CUDA0,CUDA1
|
|
- --main-gpu
|
|
- "0"
|
|
- --split-mode
|
|
- layer
|
|
- --tensor-split
|
|
- "${ULTRA_TENSOR_SPLIT:-80,20}"
|
|
- --spec-type
|
|
- draft-mtp
|
|
- --spec-draft-n-max
|
|
- "2"
|
|
- --spec-draft-type-k
|
|
- f16
|
|
- --spec-draft-type-v
|
|
- f16
|
|
|
|
# Deliberately less refusal-prone weight-level ablation. It remains behind
|
|
# the same authenticated router, tool permissions and confirmation guards as
|
|
# every other profile; "uncensored" never means unrestricted tool access.
|
|
llama-uncensored:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-uncensored
|
|
labels:
|
|
com.mike-ai.llama-profile: uncensored
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${UNCENSORED_GPU_DEVICES:-0,1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
MTMD_BACKEND_DEVICE: CUDA1
|
|
command:
|
|
- --model
|
|
- "/models/${UNCENSORED_MODEL_FILE:?UNCENSORED_MODEL_FILE is required}"
|
|
- --mmproj
|
|
- "/models/${UNCENSORED_PROJECTOR_FILE:?UNCENSORED_PROJECTOR_FILE is required}"
|
|
- --mmproj-offload
|
|
- --mmproj-device
|
|
- CUDA1
|
|
- --alias
|
|
- qwen-uncensored
|
|
- --ctx-size
|
|
- "${UNCENSORED_CONTEXT:-80000}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --threads
|
|
- "${LLAMA_THREADS:-6}"
|
|
- --threads-batch
|
|
- "${LLAMA_THREADS_BATCH:-6}"
|
|
- --batch-size
|
|
- "64"
|
|
- --ubatch-size
|
|
- "32"
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --temperature
|
|
- "0.2"
|
|
- --top-p
|
|
- "0.8"
|
|
- --top-k
|
|
- "20"
|
|
- --device
|
|
- CUDA0,CUDA1
|
|
- --main-gpu
|
|
- "0"
|
|
- --split-mode
|
|
- layer
|
|
- --tensor-split
|
|
- "${UNCENSORED_TENSOR_SPLIT:-90,10}"
|
|
- --spec-type
|
|
- draft-mtp
|
|
- --spec-draft-n-max
|
|
- "${UNCENSORED_MTP_MAX:-2}"
|
|
- --spec-draft-p-min
|
|
- "0.10"
|
|
- --spec-draft-type-k
|
|
- f16
|
|
- --spec-draft-type-v
|
|
- f16
|
|
|
|
llama-experimental:
|
|
<<: *llama-common
|
|
container_name: mike-ai-llama-experimental
|
|
labels:
|
|
com.mike-ai.llama-profile: experimental
|
|
environment:
|
|
SEARXNG_URL: http://searxng:8080
|
|
NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
command:
|
|
- --model
|
|
- "/models/${EXPERIMENTAL_MODEL_FILE:?EXPERIMENTAL_MODEL_FILE is required}"
|
|
- --alias
|
|
- qwen-experimental
|
|
- --ctx-size
|
|
- "${EXPERIMENTAL_CONTEXT:-76800}"
|
|
- --flash-attn
|
|
- "on"
|
|
- --cache-type-k
|
|
- q4_0
|
|
- --cache-type-v
|
|
- q4_0
|
|
- --parallel
|
|
- "1"
|
|
- --jinja
|
|
- --reasoning
|
|
- auto
|
|
- --reasoning-preserve
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --metrics
|
|
- --fit
|
|
- "off"
|
|
- --n-gpu-layers
|
|
- all
|
|
- --no-mmap
|
|
- --no-ui
|
|
- --device
|
|
- CUDA0
|
|
- --split-mode
|
|
- none
|
|
|
|
profile-controller:
|
|
build: ./platform/docker/profile-controller
|
|
image: mike-ai/profile-controller:local
|
|
container_name: mike-ai-profile-controller
|
|
restart: unless-stopped
|
|
read_only: true
|
|
tmpfs: ["/tmp:size=16m"]
|
|
volumes:
|
|
- /var/run/docker.sock:/var/run/docker.sock
|
|
environment:
|
|
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
|
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored,experimental
|
|
IMAGE_WORKER: flux
|
|
networks: [control]
|
|
security_opt: ["no-new-privileges:true"]
|
|
healthcheck:
|
|
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"]
|
|
interval: 10s
|
|
timeout: 3s
|
|
retries: 10
|
|
|
|
router:
|
|
build:
|
|
context: .
|
|
dockerfile: platform/docker/router/Dockerfile
|
|
image: mike-ai/profile-router:local
|
|
container_name: mike-ai-router
|
|
restart: unless-stopped
|
|
read_only: true
|
|
tmpfs: ["/tmp:size=256m"]
|
|
volumes:
|
|
- ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro
|
|
- router-state:/var/lib/mike-ai-profile-router
|
|
- router-images:/data/images
|
|
environment:
|
|
ROUTER_HOST: 0.0.0.0
|
|
ROUTER_PORT: "8081"
|
|
ROUTER_AUTH_MODE: required
|
|
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
|
ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json
|
|
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
|
|
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
|
|
UPSTREAM_URL: http://llama-upstream:8080
|
|
PROFILE_CONTROL_URL: http://profile-controller:8090
|
|
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
|
SWITCH_TIMEOUT: "600"
|
|
REQUEST_TIMEOUT: "600"
|
|
IMAGE_DIR: /data/images
|
|
IMAGE_WORKER_URL: http://flux-worker:8086
|
|
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
|
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
|
ENABLE_IMAGE_GENERATION: "true"
|
|
ENABLE_TTS: "true"
|
|
# Stable OpenAI compatibility names remain piper/alloy because an
|
|
# existing Open WebUI database persists those values. The gateway maps
|
|
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
|
|
# Piper if XTTS is unavailable, busy or returns an error.
|
|
TTS_WORKER_URL: http://tts-gateway:8085
|
|
TTS_MODEL: piper
|
|
TTS_VOICES: alloy
|
|
TTS_DEFAULT_VOICE: alloy
|
|
ENABLE_STT: "false"
|
|
networks: [frontend, control, inference]
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
# The entrypoint fixes ownership of fresh named volumes and immediately
|
|
# drops to uid/gid 10002 via gosu before starting the router. Without this
|
|
# narrowly scoped capabilities a clean installation cannot initialize the
|
|
# volumes and then switch to its unprivileged runtime identity.
|
|
cap_add: [CHOWN, SETUID, SETGID]
|
|
healthcheck:
|
|
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"]
|
|
interval: 5s
|
|
timeout: 3s
|
|
retries: 24
|
|
start_period: 5s
|
|
depends_on:
|
|
wireguard-gateway:
|
|
condition: service_healthy
|
|
profile-controller:
|
|
condition: service_healthy
|
|
piper:
|
|
condition: service_healthy
|
|
tts-gateway:
|
|
condition: service_healthy
|
|
|
|
flux-worker:
|
|
build:
|
|
context: platform/docker/flux-worker
|
|
args:
|
|
DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0}
|
|
TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1}
|
|
ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0}
|
|
HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0}
|
|
image: mike-ai/flux-worker:local
|
|
container_name: mike-ai-flux-worker
|
|
restart: "no"
|
|
profiles: [image]
|
|
labels:
|
|
com.mike-ai.image-worker: flux
|
|
gpus: all
|
|
read_only: true
|
|
tmpfs: ["/tmp:size=1g,mode=1777"]
|
|
volumes:
|
|
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
|
|
- router-images:/data/images
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
|
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
|
|
IMAGE_DIR: /data/images
|
|
networks: [inference]
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
healthcheck:
|
|
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8086/health', timeout=2)"]
|
|
interval: 5s
|
|
timeout: 3s
|
|
retries: 12
|
|
|
|
piper:
|
|
build:
|
|
context: platform/docker/piper
|
|
args:
|
|
PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
|
|
image: mike-ai/piper:local
|
|
container_name: mike-ai-piper
|
|
restart: unless-stopped
|
|
read_only: true
|
|
tmpfs:
|
|
- /tmp:size=256m,mode=1777
|
|
volumes:
|
|
- piper-data:/data
|
|
environment:
|
|
PIPER_DATA_DIR: /data
|
|
PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
|
|
PIPER_VOICE_ALIAS: alloy
|
|
PIPER_HOST: 0.0.0.0
|
|
PIPER_PORT: "8085"
|
|
PIPER_MAX_TEXT_CHARS: "8000"
|
|
networks: [frontend]
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
cap_add: [CHOWN, SETUID, SETGID]
|
|
healthcheck:
|
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 30
|
|
start_period: 120s
|
|
|
|
xtts:
|
|
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
|
|
container_name: mike-ai-xtts
|
|
restart: unless-stopped
|
|
deploy:
|
|
resources:
|
|
reservations:
|
|
devices:
|
|
- driver: nvidia
|
|
device_ids:
|
|
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
|
capabilities: [gpu]
|
|
read_only: true
|
|
shm_size: 1g
|
|
tmpfs:
|
|
- /tmp:size=1g,mode=1777
|
|
- /root/.cache:size=2g,mode=0700
|
|
volumes:
|
|
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
|
|
environment:
|
|
COQUI_TOS_AGREED: "1"
|
|
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
CUDA_VISIBLE_DEVICES: "0"
|
|
NUM_THREADS: "4"
|
|
networks: [frontend]
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
healthcheck:
|
|
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 36
|
|
start_period: 240s
|
|
|
|
tts-gateway:
|
|
build:
|
|
context: platform/docker/tts-gateway
|
|
image: mike-ai/tts-gateway:local
|
|
container_name: mike-ai-tts-gateway
|
|
restart: unless-stopped
|
|
read_only: true
|
|
tmpfs:
|
|
- /tmp:size=256m,mode=1777
|
|
environment:
|
|
TTS_GATEWAY_HOST: 0.0.0.0
|
|
TTS_GATEWAY_PORT: "8085"
|
|
XTTS_URL: http://xtts:80
|
|
PIPER_URL: http://piper:8085
|
|
TTS_VOICE_ALIAS: alloy
|
|
XTTS_SPEAKER: Annmarie Nele
|
|
TTS_DEFAULT_LANGUAGE: de
|
|
# Mixed-language clip stitching caused long pauses and unintelligible
|
|
# transitions. Keep full sentences in one stable German voice.
|
|
TTS_CODE_SWITCH_ENABLED: "false"
|
|
XTTS_QUEUE_TIMEOUT: "15"
|
|
XTTS_TIMEOUT: "120"
|
|
# Short sentence-sized requests avoid long generated silences and
|
|
# truncated weather/status summaries with Annmarie Nele.
|
|
XTTS_CHUNK_CHARS: "60"
|
|
PIPER_TIMEOUT: "120"
|
|
networks: [frontend]
|
|
depends_on:
|
|
piper:
|
|
condition: service_healthy
|
|
security_opt: ["no-new-privileges:true"]
|
|
cap_drop: [ALL]
|
|
healthcheck:
|
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
|
|
interval: 10s
|
|
timeout: 5s
|
|
retries: 12
|
|
start_period: 10s
|
|
|
|
open-webui:
|
|
build:
|
|
context: .
|
|
dockerfile: platform/openwebui/Dockerfile
|
|
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-agent-loop-v9}
|
|
container_name: mike-ai-open-webui
|
|
restart: unless-stopped
|
|
volumes:
|
|
- open-webui-data:/app/backend/data
|
|
# Upstream-supported static customization hooks. Keeping these files in
|
|
# the repository makes the global dark theme reproducible and update-safe.
|
|
- ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro
|
|
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
|
- ./platform/openwebui/theme/midnight-aurora.svg:/app/build/static/midnight-aurora.svg:ro
|
|
- ./platform/openwebui/theme/tool-status:/app/build/static/tool-status:ro
|
|
environment:
|
|
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
|
DEFAULT_MODELS: mikeai-medium
|
|
OLLAMA_BASE_URL: ""
|
|
OPENAI_API_BASE_URLS: http://router:8081/v1
|
|
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
|
# Open WebUI uses the router's OpenAI-compatible image endpoint. The
|
|
# router performs the exclusive RTX-5080 hot swap and restores the
|
|
# previously active Qwen profile after every image.
|
|
ENABLE_IMAGE_GENERATION: "true"
|
|
IMAGE_GENERATION_ENGINE: openai
|
|
# OpenWebUI v0.9.x exposes a fixed OpenAI image-model dropdown. The
|
|
# router accepts this compatibility alias and still executes local
|
|
# FLUX.2 Klein; no request is sent to OpenAI.
|
|
IMAGE_GENERATION_MODEL: gpt-image-1
|
|
# The local router can return embedded image data. Force that mode so
|
|
# OpenWebUI does not reject the router's private Docker/LAN URL through
|
|
# its correct SSRF protection.
|
|
IMAGE_URL_RESPONSE_MODELS_REGEX_PATTERN: "^$"
|
|
IMAGES_OPENAI_API_BASE_URL: http://router:8081/v1
|
|
IMAGES_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
|
IMAGE_SIZE: 1024x1024
|
|
IMAGE_STEPS: "4"
|
|
AUDIO_TTS_ENGINE: openai
|
|
AUDIO_TTS_OPENAI_API_BASE_URL: http://router:8081/v1
|
|
AUDIO_TTS_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
|
AUDIO_TTS_MODEL: piper
|
|
AUDIO_TTS_VOICE: alloy
|
|
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
|
|
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
|
# The derived image reserves the last round for a tool-free synthesis.
|
|
# Forty executions permit real multi-domain agent work. Exact-repeat,
|
|
# per-tool and total-execution limits in the derived image stop loops.
|
|
# Leave continuation headroom after the execution middleware budget:
|
|
# one additional model turn is required to synthesize the visible answer.
|
|
CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "48"
|
|
USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
# Seed native MCP connections on a fresh Open WebUI database. Secrets
|
|
# stay inside the tool containers, so these internal URLs need no keys.
|
|
TOOL_SERVER_CONNECTIONS: >-
|
|
[
|
|
{"url":"http://mike-ai-mcp-platform-context:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"athena-platform","name":"Athena Plattformwissen","description":"Zuerst aktivieren und verwenden, wenn an Athena/MikeAI, Modellen, Profilen, Router, OpenWebUI, MCPs, TTS/STT, Vision, Netzwerk oder Recovery gearbeitet wird. Liefert versionierte Dokumentation und einen begrenzten aktuellen Systemstand. Dokumentationspflege nur über Vorschau und ausdrückliche Freigabe; keine Container-, Shell-, Netzwerk-, Git- oder Secretrechte."}},
|
|
{"url":"http://tinysearch:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-general-local","name":"Allgemeines Web (TinySearch)","description":"Breite, portable Websuche und Seitenabruf für beliebige öffentliche Websites. In OpenWebUI ist native search_web/fetch_url standardmäßig aktiv; dieser Upstream-MCP ist die portable Alternative für Hermes, Pi und manuelle Nutzung. Kurze, gezielte Abfragen bevorzugen."}},
|
|
{"url":"http://mike-ai-mcp-github:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[],"function_name_filter_list":"search_repositories,get_file_contents,search_code"},"info":{"id":"github-local","name":"GitHub (offiziell, read-only)","description":"Für Repositorysuche, echte Datei-Inhalte und gezielte Code-Suche. Strikt read-only mit genau drei Werkzeugen; keine rekursiven Komplettbäume, allgemeine Webrecherche oder Änderungen."}},
|
|
{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Für Home-Assistant-Entitäten, Zustände, Historie, Automationen, Dashboards, HA-Diagnose und freigegebene YAML-Dateien. YAML-Lesen ist begrenzt; Änderungen benötigen serverseitige Vorschau, explizite Freigabe, Sicherung und Validierung. Nicht für Unraid, Sonarr/Radarr oder allgemeine Websuche."}},
|
|
{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"Sonarr und Radarr (lokal)","description":"Nur für verwaltete Serien/Filme, fehlende Episoden, Queue und Suche über konfigurierte Indexer. Keine allgemeine Websuche; Schreibaktionen benötigen Vorschau und Freigabe."}},
|
|
{"url":"http://mike-ai-mcp-navidrome:3000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"navidrome-local","name":"Navidrome (Musikbibliothek)","description":"Nur für die persönliche Navidrome-Musikbibliothek: Titel, Alben, Künstler, Playlists, Favoriten und Hörverlauf. Nicht für Sonarr/Radarr, allgemeine Websuche oder Audioausgabe auf dem KI-Host. Wegen des großen Werkzeugkatalogs nur bei Musikaufgaben aktivieren."}},
|
|
{"url":"http://mike-ai-mcp-deemix:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"deemix-local","name":"Deemix (bestehende Unraid-Instanz)","description":"Durchsucht Deezer über die vorhandene Deemix-Instanz auf Unraid und verwaltet deren Download-Queue. Keine zweite Deemix-Instanz. Schreibende Queue-Aktionen nur auf ausdrücklichen Benutzerauftrag; Status und Suche sind read-only."}}
|
|
]
|
|
DO_NOT_TRACK: "true"
|
|
SCARF_NO_ANALYTICS: "true"
|
|
dns: ["${AI_DNS:-1.1.1.1}"]
|
|
networks: [frontend, tools]
|
|
depends_on:
|
|
wireguard-gateway:
|
|
condition: service_healthy
|
|
router:
|
|
condition: service_healthy
|
|
security_opt: ["no-new-privileges:true"]
|
|
|
|
hermes:
|
|
image: ${HERMES_IMAGE:-nousresearch/hermes-agent@sha256:143bdb9086bb2db645346179f11091e621ef6b7f4f9e5049ae7454bfeb3a0495}
|
|
container_name: mike-ai-hermes
|
|
restart: unless-stopped
|
|
command: [/usr/local/bin/start-hermes-managed]
|
|
env_file:
|
|
- /data/hermes/.env
|
|
volumes:
|
|
- /data/hermes:/opt/data
|
|
- /data/hermes/workspace:/workspace
|
|
- ./platform/hermes/start-hermes-managed.sh:/usr/local/bin/start-hermes-managed:ro
|
|
- ./platform/hermes/patch-api-mcp-refresh.py:/usr/local/lib/mike-ai/patch-api-mcp-refresh.py:ro
|
|
environment:
|
|
HERMES_HOME: /opt/data
|
|
dns: ["${AI_DNS:-1.1.1.1}"]
|
|
networks: [frontend, tools, tools-egress]
|
|
depends_on:
|
|
wireguard-gateway:
|
|
condition: service_healthy
|
|
router:
|
|
condition: service_healthy
|
|
security_opt: ["no-new-privileges:true"]
|
|
healthcheck:
|
|
test: [CMD, curl, -fsS, "http://127.0.0.1:8642/health"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 20
|
|
start_period: 45s
|
|
|
|
# Optional, fully removable community chat surface. Chat execution goes
|
|
# through the existing Hermes gateway. Upstream's container entrypoint
|
|
# requires a writable Hermes home for its ownership/init checks; UI-only
|
|
# state still remains on a separate bind mount for easy removal.
|
|
hermes-webui:
|
|
image: ${HERMES_WEBUI_IMAGE:-mike-ai/hermes-webui:0.52.113-hermes-source-v1}
|
|
container_name: mike-ai-hermes-webui
|
|
restart: unless-stopped
|
|
profiles: [hermes-webui]
|
|
env_file:
|
|
- /data/hermes-webui/.env
|
|
volumes:
|
|
- /data/hermes:/home/hermeswebui/.hermes
|
|
- /data/hermes-webui/state:/state
|
|
- /data/hermes-webui/hermes-agent:/home/hermeswebui/.hermes/hermes-agent:ro
|
|
- /data/hermes/workspace:/workspace
|
|
environment:
|
|
HERMES_HOME: /home/hermeswebui/.hermes
|
|
HERMES_WEBUI_STATE_DIR: /state
|
|
HERMES_WEBUI_HOST: 0.0.0.0
|
|
HERMES_WEBUI_PORT: "8787"
|
|
HERMES_WEBUI_CHAT_BACKEND: gateway
|
|
HERMES_WEBUI_GATEWAY_BASE_URL: http://hermes:8642
|
|
HERMES_API_URL: http://hermes:8642
|
|
HERMES_WEBUI_AGENT_DIR: /home/hermeswebui/.hermes/hermes-agent
|
|
HERMES_WEBUI_GATEWAY_USE_RUNS_API: "true"
|
|
HERMES_SKIP_CHMOD: "1"
|
|
WANTED_UID: "10000"
|
|
WANTED_GID: "10000"
|
|
networks: [frontend]
|
|
depends_on:
|
|
hermes:
|
|
condition: service_healthy
|
|
security_opt: ["no-new-privileges:true"]
|
|
healthcheck:
|
|
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8787/health', timeout=3)"]
|
|
interval: 15s
|
|
timeout: 5s
|
|
retries: 20
|
|
start_period: 45s
|
|
|
|
# Persistent VPN listener for the optional WebUI. Sharing the existing
|
|
# WireGuard network namespace avoids recreating the remote-access gateway
|
|
# merely to add one listener.
|
|
hermes-webui-vpn-proxy:
|
|
image: mike-ai/wireguard-gateway:local
|
|
container_name: mike-ai-hermes-webui-vpn-proxy
|
|
restart: unless-stopped
|
|
profiles: [hermes-webui]
|
|
network_mode: "service:wireguard-gateway"
|
|
entrypoint: [socat]
|
|
command:
|
|
- TCP-LISTEN:8787,bind=192.168.1.212,reuseaddr,fork
|
|
- TCP:hermes-webui:8787
|
|
read_only: true
|
|
tmpfs:
|
|
- /tmp:size=4m,mode=1777
|
|
cap_drop: [ALL]
|
|
security_opt: ["no-new-privileges:true"]
|
|
depends_on:
|
|
wireguard-gateway:
|
|
condition: service_healthy
|
|
hermes-webui:
|
|
condition: service_healthy
|
|
|
|
networks:
|
|
frontend:
|
|
internal: false
|
|
ipam:
|
|
config: [{subnet: 172.30.10.0/24}]
|
|
control:
|
|
internal: true
|
|
ipam:
|
|
config: [{subnet: 172.30.20.0/24}]
|
|
inference:
|
|
internal: true
|
|
ipam:
|
|
config: [{subnet: 172.30.30.0/24}]
|
|
tools:
|
|
external: true
|
|
name: mike-ai-tools
|
|
tools-egress:
|
|
external: true
|
|
name: mike-ai-tools-egress
|
|
|
|
volumes:
|
|
open-webui-data:
|
|
piper-data:
|
|
router-state:
|
|
router-images:
|