name: mike-ai # One project and one command. Fach-MCPs remain in their own source file, but # Compose loads them into this same stack instead of a second project. include: - path: platform/mcp/compose.yaml x-llama-common: &llama-common image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local} restart: "no" profiles: [inference] gpus: all ipc: host read_only: true tmpfs: - /tmp:size=1g,mode=1777 volumes: - "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro" environment: NVIDIA_DRIVER_CAPABILITIES: compute,utility dns: ["${AI_DNS:-1.1.1.1}"] networks: inference: aliases: [llama-upstream] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"] interval: 10s timeout: 5s retries: 60 start_period: 30s services: wireguard-gateway: build: ./platform/docker/wireguard-gateway image: mike-ai/wireguard-gateway:local container_name: mike-ai-wireguard-gateway restart: unless-stopped cap_add: [NET_ADMIN] devices: - /dev/net/tun:/dev/net/tun sysctls: net.ipv4.ip_forward: "1" net.ipv4.conf.all.src_valid_mark: "1" net.ipv6.conf.all.forwarding: "1" read_only: true tmpfs: - /run:size=16m,mode=0755 - /tmp:size=16m,mode=1777 volumes: - "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro" networks: frontend: ipv4_address: 172.30.10.254 tools: ipv4_address: 172.30.40.254 tools-egress: ipv4_address: 172.30.50.254 security_opt: ["no-new-privileges:true"] healthcheck: test: [CMD, /usr/local/sbin/mike-ai-wireguard-healthcheck] interval: 10s timeout: 3s retries: 12 start_period: 10s llama-fast: <<: *llama-common container_name: mike-ai-llama-fast labels: com.mike-ai.llama-profile: fast environment: NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility # CUDA0 remains the exclusive text-model device. The projector is kept # on the secondary card so vision does not consume the 5080 context # budget. MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --mmproj-offload - --mmproj-device - CUDA1 - --alias - qwen-fast - --ctx-size - "${FAST_CONTEXT:-76800}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 # Keep the cross-chat prefix cache explicit. On-disk slot restore stays # disabled until the current upstream restore regressions are fixed. - --cache-prompt - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "${FAST_BATCH_SIZE:-64}" - --ubatch-size - "${FAST_UBATCH_SIZE:-32}" - --parallel - "${FAST_PARALLEL_SLOTS:-1}" - --kv-unified - --jinja - --reasoning - auto - --reasoning-preserve - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0 - --split-mode - none - --spec-type - draft-mtp - --spec-draft-n-max - "2" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 llama-medium: <<: *llama-common container_name: mike-ai-llama-medium labels: com.mike-ai.llama-profile: medium environment: NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --mmproj-offload - --mmproj-device - CUDA1 - --alias - qwen-medium - --ctx-size - "${MEDIUM_CONTEXT:-160000}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --cache-prompt - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "${MEDIUM_BATCH_SIZE:-2048}" - --ubatch-size - "${MEDIUM_UBATCH_SIZE:-128}" - --parallel - "${MEDIUM_PARALLEL_SLOTS:-1}" - --kv-unified - --jinja - --reasoning - auto # No fixed --reasoning-budget here: the router supplies a real budget # per request from the client's reasoning_effort selection. - --reasoning-preserve - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature # Qwen3.8's official thinking-mode sampler. The former 0.2 setting was # overly deterministic and could lock reasoning into verbatim loops. - "1.0" - --top-p - "0.95" - --top-k - "20" - --device - CUDA0,CUDA1 - --main-gpu - "0" - --split-mode - layer - --tensor-split - "${MEDIUM_TENSOR_SPLIT:-85,15}" - --spec-type - draft-mtp - --spec-draft-n-max - "3" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 - --spec-draft-p-min - "0.05" llama-large: <<: *llama-common container_name: mike-ai-llama-large labels: com.mike-ai.llama-profile: large environment: NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - --mmproj-offload - --mmproj-device - CUDA1 - --alias - qwen-large - --ctx-size - "${LARGE_CONTEXT:-192000}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --cache-prompt - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "${LARGE_BATCH_SIZE:-2048}" - --ubatch-size - "${LARGE_UBATCH_SIZE:-128}" - --parallel - "${LARGE_PARALLEL_SLOTS:-1}" - --kv-unified - --jinja - --reasoning - auto - --reasoning-preserve - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0,CUDA1 - --main-gpu - "0" - --split-mode - layer - --tensor-split - "${LARGE_TENSOR_SPLIT:-86,14}" - --spec-type - draft-mtp - --spec-draft-n-max - "3" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 # Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20 # combination completed the 220K fill test on RTX 5080 + RTX 3060. # Deliberately no vision projector: Ultra prioritizes maximum usable context. llama-ultra: <<: *llama-common container_name: mike-ai-llama-ultra labels: com.mike-ai.llama-profile: ultra environment: NVIDIA_VISIBLE_DEVICES: ${ULTRA_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: - --model - "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}" - --alias - qwen-ultra - --ctx-size - "${ULTRA_CONTEXT:-262144}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --cache-prompt - --cache-reuse - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "${ULTRA_BATCH_SIZE:-2048}" - --ubatch-size - "${ULTRA_UBATCH_SIZE:-128}" - --parallel - "${ULTRA_PARALLEL_SLOTS:-1}" - --kv-unified - --jinja - --reasoning - auto - --reasoning-preserve - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0,CUDA1 - --main-gpu - "0" - --split-mode - layer - --tensor-split - "${ULTRA_TENSOR_SPLIT:-80,20}" - --spec-type - draft-mtp - --spec-draft-n-max - "2" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 # Deliberately less refusal-prone weight-level ablation. It remains behind # the same authenticated router, tool permissions and confirmation guards as # every other profile; "uncensored" never means unrestricted tool access. llama-uncensored: <<: *llama-common container_name: mike-ai-llama-uncensored labels: com.mike-ai.llama-profile: uncensored environment: NVIDIA_VISIBLE_DEVICES: ${UNCENSORED_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${UNCENSORED_MODEL_FILE:?UNCENSORED_MODEL_FILE is required}" - --mmproj - "/models/${UNCENSORED_PROJECTOR_FILE:?UNCENSORED_PROJECTOR_FILE is required}" - --mmproj-offload - --mmproj-device - CUDA1 - --alias - qwen-uncensored - --ctx-size - "${UNCENSORED_CONTEXT:-80000}" - --flash-attn - "on" - --cache-type-k - q4_0 - --cache-type-v - q4_0 - --cache-prompt - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch - "${LLAMA_THREADS_BATCH:-6}" - --batch-size - "${UNCENSORED_BATCH_SIZE:-2048}" - --ubatch-size - "${UNCENSORED_UBATCH_SIZE:-128}" - --parallel - "${UNCENSORED_PARALLEL_SLOTS:-1}" - --kv-unified - --jinja - --reasoning - auto - --reasoning-preserve - --host - 0.0.0.0 - --port - "8080" - --metrics - --fit - "off" - --n-gpu-layers - all - --no-mmap - --no-ui - --temperature - "0.2" - --top-p - "0.8" - --top-k - "20" - --device - CUDA0,CUDA1 - --main-gpu - "0" - --split-mode - layer - --tensor-split - "${UNCENSORED_TENSOR_SPLIT:-90,10}" - --spec-type - draft-mtp - --spec-draft-n-max - "${UNCENSORED_MTP_MAX:-2}" - --spec-draft-p-min - "0.10" - --spec-draft-type-k - f16 - --spec-draft-type-v - f16 profile-controller: build: ./platform/docker/profile-controller image: mike-ai/profile-controller:local container_name: mike-ai-profile-controller restart: unless-stopped read_only: true tmpfs: ["/tmp:size=16m"] volumes: - /var/run/docker.sock:/var/run/docker.sock environment: CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" ALLOWED_PROFILES: fast,medium,large,ultra,uncensored IMAGE_WORKER: image RESTORE_WORKER: restore TTS_WORKER: qwen3 MUSIC_WORKER: acestep SEPARATOR_WORKER: bs-roformer VOICE_WORKER: vevo2 VOICE_CHANGE_WORKER: xvc APPLIO_WORKER: applio networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"] interval: 10s timeout: 3s retries: 10 router: build: context: . dockerfile: platform/docker/router/Dockerfile image: mike-ai/profile-router:local container_name: mike-ai-router restart: unless-stopped read_only: true tmpfs: ["/tmp:size=256m"] volumes: - ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro - ./config/global-system-policy.txt:/etc/mike-ai/global-system-policy.txt:ro - router-state:/var/lib/mike-ai-profile-router - router-images:/data/images environment: ROUTER_HOST: 0.0.0.0 ROUTER_PORT: "8081" ROUTER_AUTH_MODE: required ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json ROUTER_MAX_CONCURRENT_REQUESTS: "16" UPSTREAM_URL: http://llama-upstream:8080 PROFILE_CONTROL_URL: http://profile-controller:8090 PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" SWITCH_TIMEOUT: "600" REQUEST_TIMEOUT: "600" # Last-resort guard for every OpenAI-compatible client. Without a # request limit llama.cpp uses n_predict=-1 and a reasoning loop can # consume the complete context before yielding visible output. MAX_GENERATION_TOKENS: "8192" DEFAULT_REASONING_EFFORT: "${DEFAULT_REASONING_EFFORT:-off}" GLOBAL_SYSTEM_POLICY_FILE: /etc/mike-ai/global-system-policy.txt IMAGE_DIR: /data/images IMAGE_WORKER_URL: http://image-worker:8086 IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" ENABLE_IMAGE_GENERATION: "true" ENABLE_TTS: "true" # The gateway keeps text normalization, output conversion and native # PCM streaming in one stable API in front of Qwen3-TTS. TTS_WORKER_URL: http://tts-gateway:8085 TTS_MODEL: qwen3-tts TTS_VOICES: alloy TTS_DEFAULT_VOICE: alloy ENABLE_STT: "true" ENABLE_MUSIC_MODE: "true" MUSIC_START_TIMEOUT: "600" VOICE_CHANGE_START_TIMEOUT: "600" APPLIO_START_TIMEOUT: "900" STT_WORKER_URL: http://whisper:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] # The entrypoint fixes ownership of fresh named volumes and immediately # drops to uid/gid 10002 via gosu before starting the router. Without this # narrowly scoped capabilities a clean installation cannot initialize the # volumes and then switch to its unprivileged runtime identity. cap_add: [CHOWN, SETUID, SETGID] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"] interval: 5s timeout: 3s retries: 24 start_period: 5s depends_on: wireguard-gateway: condition: service_healthy profile-controller: condition: service_healthy tts-gateway: condition: service_healthy whisper: condition: service_healthy image-worker: build: context: platform/docker/image-worker args: DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0} TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1} ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0} HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0} image: mike-ai/image-worker:local container_name: mike-ai-image-worker restart: "no" profiles: [image] labels: com.mike-ai.image-worker: image gpus: all read_only: true tmpfs: ["/tmp:size=1g,mode=1777"] volumes: - "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro" - "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro" - router-images:/data/images environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" FLUX_COMPONENT_DIR: /models/components FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors IMAGE_DIR: /data/images networks: [inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8086/health', timeout=2)"] interval: 5s timeout: 3s retries: 12 qwen3-tts: image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} container_name: mike-ai-qwen3-tts restart: unless-stopped labels: com.mike-ai.tts-worker: qwen3 deploy: resources: reservations: devices: - driver: nvidia device_ids: - ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} capabilities: [gpu] read_only: true shm_size: 1g tmpfs: - /tmp:size=1g,mode=1777 volumes: - "${QWEN3_TTS_CACHE_DIR:-/data/models/qwen3-tts-cache}:/root/.cache/huggingface" - "${QWEN3_TTS_VOICES_DIR:-/data/models/qwen3-tts-voices}:/data/voices" environment: NVIDIA_VISIBLE_DEVICES: ${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} NVIDIA_DRIVER_CAPABILITIES: compute,utility CUDA_VISIBLE_DEVICES: "0" HF_HOME: /root/.cache/huggingface NUMBA_CACHE_DIR: /tmp/numba QWEN3_TTS_MODEL_ID: Qwen/Qwen3-TTS-12Hz-1.7B-Base QWEN3_TTS_DEFAULT_VOICE: serena QWEN3_TTS_VOICES_DIR: /data/voices networks: [frontend] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8001/health', timeout=2)"] interval: 10s timeout: 5s retries: 60 start_period: 600s tts-gateway: build: context: platform/docker/tts-gateway image: mike-ai/tts-gateway:local container_name: mike-ai-tts-gateway restart: unless-stopped read_only: true tmpfs: - /tmp:size=256m,mode=1777 environment: TTS_GATEWAY_HOST: 0.0.0.0 TTS_GATEWAY_PORT: "8085" QWEN_TTS_URL: http://qwen3-tts:8001 QWEN_TTS_MODEL: tts-1 QWEN_TTS_VOICE: serena QWEN_TTS_LANGUAGE: German QWEN_TTS_TIMEOUT: "120" TTS_VOICE_ALIAS: alloy TTS_DEFAULT_LANGUAGE: de # Mixed-language clip stitching caused long pauses and unintelligible # transitions. Keep full sentences in one stable German voice. TTS_CODE_SWITCH_ENABLED: "false" networks: [frontend] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"] interval: 10s timeout: 5s retries: 12 start_period: 10s whisper: build: context: . dockerfile: platform/docker/whisper/Dockerfile args: WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1} image: mike-ai/whisper:local container_name: mike-ai-whisper restart: unless-stopped read_only: true tmpfs: - /tmp:size=2g,mode=1777 volumes: - whisper-data:/models environment: WHISPER_HOST: 0.0.0.0 WHISPER_PORT: "8084" WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli WHISPER_MODEL: /models/ggml-small.bin WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin} WHISPER_SERVER_PORT: "8085" WHISPER_THREADS: ${WHISPER_THREADS:-8} WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de} networks: [frontend, inference] security_opt: ["no-new-privileges:true"] cap_drop: [ALL] # The entrypoint supervises whisper-server after dropping it to uid 10004. cap_add: [CHOWN, SETUID, SETGID, KILL] healthcheck: test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"] interval: 10s timeout: 5s retries: 90 start_period: 20m llama-dashboard: build: ./platform/llama-dashboard image: mike-ai/llama-dashboard:local container_name: mike-ai-llama-dashboard restart: unless-stopped networks: [frontend] gpus: all read_only: true tmpfs: - /tmp:size=16m,mode=1777 volumes: - /proc:/host/proc:ro - /data:/host/data:ro - /data/models:/host/models:ro - /data/emergency-backups:/backups:ro - /data/llama-dashboard:/var/lib/llama-dashboard environment: DASHBOARD_HOST: 0.0.0.0 DASHBOARD_PORT: "8099" ROUTER_URL: http://router:8081 ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}" MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}" SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}" VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}" VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}" APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}" MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models DASHBOARD_BACKUP_DIR: /backups DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3 DASHBOARD_HISTORY_INTERVAL: "15" DASHBOARD_DETAIL_RETENTION_DAYS: "21" NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility depends_on: wireguard-gateway: condition: service_healthy router: condition: service_healthy security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=2)"] interval: 10s timeout: 3s retries: 12 start_period: 10s portainer: image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa} container_name: mike-ai-portainer restart: unless-stopped networks: [frontend] command: [--no-setup-token] volumes: - /var/run/docker.sock:/var/run/docker.sock - portainer-data:/data depends_on: wireguard-gateway: condition: service_healthy security_opt: ["no-new-privileges:true"] backup: image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013} container_name: mike-ai-backup restart: unless-stopped environment: BACKUP_CRON_EXPRESSION: "0 */5 * * *" BACKUP_FILENAME: "athena-%Y-%m-%dT%H-%M-%S.tar.gz" BACKUP_LATEST_SYMLINK: athena-latest.tar.gz BACKUP_RETENTION_DAYS: "14" BACKUP_PRUNING_PREFIX: athena- volumes: - /var/run/docker.sock:/var/run/docker.sock:ro - /data/docker-backups:/archive - /etc/mike-ai:/backup/etc-mike-ai:ro # Include every deployed specialist UI/worker source tree, not just the # core checkout. Images themselves remain reproducible and are rebuilt. - /opt/mike-ai:/backup/opt-mike-ai:ro - router-state:/backup/volumes/router-state:ro - router-images:/backup/volumes/router-images:ro - portainer-data:/backup/volumes/portainer-data:ro security_opt: ["no-new-privileges:true"] networks: frontend: internal: false ipam: config: [{subnet: 172.30.10.0/24}] control: internal: true ipam: config: [{subnet: 172.30.20.0/24}] inference: internal: true ipam: config: [{subnet: 172.30.30.0/24}] tools: external: true name: mike-ai-tools tools-egress: external: true name: mike-ai-tools-egress volumes: whisper-data: router-state: router-images: portainer-data: name: portainer_data external: true