name: mike-ai x-llama-common: &llama-common image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local} restart: "no" profiles: [inference] gpus: all ipc: host read_only: true tmpfs: - /tmp:size=1g,mode=1777 volumes: - "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro" environment: NVIDIA_DRIVER_CAPABILITIES: compute,utility dns: ["${AI_DNS:-1.1.1.1}"] networks: inference: aliases: [llama-upstream] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"] interval: 10s timeout: 5s retries: 60 start_period: 30s services: llama-fast: <<: *llama-common container_name: mike-ai-llama-fast labels: com.mike-ai.llama-profile: fast environment: NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: - --model - "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --no-mmproj-offload - --alias - qwen-fast - --ctx-size - "${FAST_CONTEXT:-76800}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "64" - --ubatch-size - "32" - --parallel - "1" - --jinja - --reasoning - auto - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0 - --split-mode - none - --spec-type - draft-mtp - --spec-draft-n-max - "2" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 llama-medium: <<: *llama-common container_name: mike-ai-llama-medium labels: com.mike-ai.llama-profile: medium environment: NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: - --model - "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --no-mmproj-offload - --alias - qwen-medium - --ctx-size - "${MEDIUM_CONTEXT:-94208}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "64" - --ubatch-size - "32" - --parallel - "1" - --jinja - --reasoning - auto - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0 - --split-mode - none llama-long: <<: *llama-common container_name: mike-ai-llama-long labels: com.mike-ai.llama-profile: long environment: NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: - --model - "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --no-mmproj-offload - --alias - qwen-long - --ctx-size - "${LONG_CONTEXT:-131072}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "64" - --ubatch-size - "32" - --parallel - "1" - --jinja - --reasoning - auto - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --override-tensor - blk.([0-9]|1[0-1]).ffn_.*=CPU - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0 - --split-mode - none - --spec-type - draft-mtp - --spec-draft-n-max - "2" - --spec-draft-type-k - q4_0 - --spec-draft-type-v - q4_0 llama-experimental: <<: *llama-common container_name: mike-ai-llama-experimental labels: com.mike-ai.llama-profile: experimental environment: SEARXNG_URL: http://searxng:8080 NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: - --model - "/models/${EXPERIMENTAL_MODEL_FILE:?EXPERIMENTAL_MODEL_FILE is required}" - --alias - qwen-experimental - --ctx-size - "${EXPERIMENTAL_CONTEXT:-76800}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --parallel - "1" - --jinja - --reasoning - auto - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --device - CUDA0 - --split-mode - none profile-controller: build: ./platform/docker/profile-controller image: mike-ai/profile-controller:local container_name: mike-ai-profile-controller restart: unless-stopped read_only: true tmpfs: ["/tmp:size=16m"] volumes: - /var/run/docker.sock:/var/run/docker.sock environment: CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" ALLOWED_PROFILES: fast,medium,long,experimental networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"] interval: 10s timeout: 3s retries: 10 router: build: context: . dockerfile: platform/docker/router/Dockerfile image: mike-ai/profile-router:local container_name: mike-ai-router restart: unless-stopped read_only: true tmpfs: ["/tmp:size=256m"] volumes: - ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro - router-state:/var/lib/mike-ai-profile-router - router-images:/data/images environment: ROUTER_HOST: 0.0.0.0 ROUTER_PORT: "8081" ROUTER_AUTH_MODE: required ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json ROUTER_MAX_CONCURRENT_REQUESTS: "16" UPSTREAM_URL: http://llama-upstream:8080 PROFILE_CONTROL_URL: http://profile-controller:8090 PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" SWITCH_TIMEOUT: "600" REQUEST_TIMEOUT: "600" IMAGE_DIR: /data/images CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" ENABLE_IMAGE_GENERATION: "false" ENABLE_TTS: "false" ENABLE_STT: "false" ports: - "${AI_BIND_ADDRESS:-127.0.0.1}:8081:8081" networks: [frontend, control, inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] # The entrypoint fixes ownership of fresh named volumes and immediately # drops to uid/gid 10002 via gosu before starting the router. Without this # narrowly scoped capabilities a clean installation cannot initialize the # volumes and then switch to its unprivileged runtime identity. cap_add: [CHOWN, SETUID, SETGID] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"] interval: 5s timeout: 3s retries: 24 start_period: 5s depends_on: profile-controller: condition: service_healthy open-webui: image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5} container_name: mike-ai-open-webui restart: unless-stopped volumes: - open-webui-data:/app/backend/data # Upstream-supported static customization hooks. Keeping these files in # the repository makes the global dark theme reproducible and update-safe. - ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro - ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro environment: WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}" OLLAMA_BASE_URL: "" OPENAI_API_BASE_URLS: http://router:8081/v1 OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false} ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} # Seed native MCP connections on a fresh Open WebUI database. Secrets # stay inside the tool containers, so these internal URLs need no keys. TOOL_SERVER_CONNECTIONS: >- [{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (lokal)","description":"Kompakte Websuche und Quellenvergleich"}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Home-Assistant-Werkzeuge mit serverseitigem Token"}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"ARR (lokal)","description":"Sonarr- und Radarr-Werkzeuge"}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (lokal, read-only)","description":"Begrenzte Unraid-Diagnose"}}] DO_NOT_TRACK: "true" SCARF_NO_ANALYTICS: "true" ports: - "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080" dns: ["${AI_DNS:-1.1.1.1}"] networks: [frontend, tools] depends_on: router: condition: service_healthy security_opt: ["no-new-privileges:true"] networks: frontend: internal: false ipam: config: [{subnet: 172.30.10.0/24}] control: internal: true ipam: config: [{subnet: 172.30.20.0/24}] inference: internal: true ipam: config: [{subnet: 172.30.30.0/24}] tools: external: true name: mike-ai-tools volumes: open-webui-data: router-state: router-images: