Files
AI-Profile-Router/compose.yaml
T

819 lines
23 KiB
YAML

name: mike-ai
# One project and one command. Fach-MCPs remain in their own source file, but
# Compose loads them into this same stack instead of a second project.
include:
- path: platform/mcp/compose.yaml
x-llama-common: &llama-common
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
restart: "no"
profiles: [inference]
gpus: all
ipc: host
read_only: true
tmpfs:
- /tmp:size=1g,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
environment:
NVIDIA_DRIVER_CAPABILITIES: compute,utility
dns: ["${AI_DNS:-1.1.1.1}"]
networks:
inference:
aliases: [llama-upstream]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"]
interval: 10s
timeout: 5s
retries: 60
start_period: 30s
services:
wireguard-gateway:
build: ./platform/docker/wireguard-gateway
image: mike-ai/wireguard-gateway:local
container_name: mike-ai-wireguard-gateway
restart: unless-stopped
cap_add: [NET_ADMIN]
devices:
- /dev/net/tun:/dev/net/tun
sysctls:
net.ipv4.ip_forward: "1"
net.ipv4.conf.all.src_valid_mark: "1"
net.ipv6.conf.all.forwarding: "1"
read_only: true
tmpfs:
- /run:size=16m,mode=0755
- /tmp:size=16m,mode=1777
volumes:
- "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro"
# Namespace-sharing services cannot publish ports themselves. The owner
# must keep these bindings so a gateway recreation cannot hide their UIs.
ports:
- "8099:8099"
- "9443:9443"
networks:
frontend:
ipv4_address: 172.30.10.254
tools:
ipv4_address: 172.30.40.254
tools-egress:
ipv4_address: 172.30.50.254
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, /usr/local/sbin/mike-ai-wireguard-healthcheck]
interval: 10s
timeout: 3s
retries: 12
start_period: 10s
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast
labels:
com.mike-ai.llama-profile: fast
environment:
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
# CUDA0 remains the exclusive text-model device. The projector is kept
# on the secondary card so vision does not consume the 5080 context
# budget.
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-fast
- --ctx-size
- "${FAST_CONTEXT:-76800}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
# Keep the cross-chat prefix cache explicit. On-disk slot restore stays
# disabled until the current upstream restore regressions are fixed.
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${FAST_BATCH_SIZE:-64}"
- --ubatch-size
- "${FAST_UBATCH_SIZE:-32}"
- --parallel
- "${FAST_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-medium:
<<: *llama-common
container_name: mike-ai-llama-medium
labels:
com.mike-ai.llama-profile: medium
environment:
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-medium
- --ctx-size
- "${MEDIUM_CONTEXT:-160000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${MEDIUM_BATCH_SIZE:-2048}"
- --ubatch-size
- "${MEDIUM_UBATCH_SIZE:-128}"
- --parallel
- "${MEDIUM_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
# Bound each individual thinking phase. Long agent jobs can still use
# many phases around tool calls, but one degenerate reasoning loop can
# no longer consume the complete response budget indefinitely.
- --reasoning-budget
- "8192"
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
# Qwen3.8's official thinking-mode sampler. The former 0.2 setting was
# overly deterministic and could lock reasoning into verbatim loops.
- "1.0"
- --top-p
- "0.95"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${MEDIUM_TENSOR_SPLIT:-85,15}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
- --spec-draft-p-min
- "0.05"
llama-large:
<<: *llama-common
container_name: mike-ai-llama-large
labels:
com.mike-ai.llama-profile: large
environment:
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-large
- --ctx-size
- "${LARGE_CONTEXT:-192000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${LARGE_BATCH_SIZE:-2048}"
- --ubatch-size
- "${LARGE_UBATCH_SIZE:-128}"
- --parallel
- "${LARGE_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${LARGE_TENSOR_SPLIT:-86,14}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
llama-ultra:
<<: *llama-common
container_name: mike-ai-llama-ultra
labels:
com.mike-ai.llama-profile: ultra
environment:
NVIDIA_VISIBLE_DEVICES: ${ULTRA_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
- --alias
- qwen-ultra
- --ctx-size
- "${ULTRA_CONTEXT:-262144}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-reuse
- "${LLAMA_CACHE_REUSE:-256}"
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${ULTRA_BATCH_SIZE:-2048}"
- --ubatch-size
- "${ULTRA_UBATCH_SIZE:-128}"
- --parallel
- "${ULTRA_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${ULTRA_TENSOR_SPLIT:-80,20}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
# Deliberately less refusal-prone weight-level ablation. It remains behind
# the same authenticated router, tool permissions and confirmation guards as
# every other profile; "uncensored" never means unrestricted tool access.
llama-uncensored:
<<: *llama-common
container_name: mike-ai-llama-uncensored
labels:
com.mike-ai.llama-profile: uncensored
environment:
NVIDIA_VISIBLE_DEVICES: ${UNCENSORED_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${UNCENSORED_MODEL_FILE:?UNCENSORED_MODEL_FILE is required}"
- --mmproj
- "/models/${UNCENSORED_PROJECTOR_FILE:?UNCENSORED_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-uncensored
- --ctx-size
- "${UNCENSORED_CONTEXT:-80000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${UNCENSORED_BATCH_SIZE:-2048}"
- --ubatch-size
- "${UNCENSORED_UBATCH_SIZE:-128}"
- --parallel
- "${UNCENSORED_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${UNCENSORED_TENSOR_SPLIT:-90,10}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "${UNCENSORED_MTP_MAX:-2}"
- --spec-draft-p-min
- "0.10"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
profile-controller:
build: ./platform/docker/profile-controller
image: mike-ai/profile-controller:local
container_name: mike-ai-profile-controller
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=16m"]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored
IMAGE_WORKER: image
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8090/health', timeout=2)"]
interval: 10s
timeout: 3s
retries: 10
router:
build:
context: .
dockerfile: platform/docker/router/Dockerfile
image: mike-ai/profile-router:local
container_name: mike-ai-router
restart: unless-stopped
read_only: true
tmpfs: ["/tmp:size=256m"]
volumes:
- ./router/router_profiles.json:/etc/mike-ai/router-profiles.json:ro
- ./config/global-system-policy.txt:/etc/mike-ai/global-system-policy.txt:ro
- router-state:/var/lib/mike-ai-profile-router
- router-images:/data/images
environment:
ROUTER_HOST: 0.0.0.0
ROUTER_PORT: "8081"
ROUTER_AUTH_MODE: required
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ROUTER_PROFILES_FILE: /etc/mike-ai/router-profiles.json
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
UPSTREAM_URL: http://llama-upstream:8080
PROFILE_CONTROL_URL: http://profile-controller:8090
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600"
# Last-resort guard for every OpenAI-compatible client. Without a
# request limit llama.cpp uses n_predict=-1 and a reasoning loop can
# consume the complete context before yielding visible output.
MAX_GENERATION_TOKENS: "8192"
DEFAULT_REASONING_EFFORT: "${DEFAULT_REASONING_EFFORT:-off}"
GLOBAL_SYSTEM_POLICY_FILE: /etc/mike-ai/global-system-policy.txt
IMAGE_DIR: /data/images
IMAGE_WORKER_URL: http://image-worker:8086
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
IMAGE_MODEL_NAME: FLUX.2-klein-4B
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true"
# Stable OpenAI compatibility names remain piper/alloy because an
# existing Open WebUI database persists those values. The gateway maps
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
# Piper if XTTS is unavailable, busy or returns an error.
TTS_WORKER_URL: http://tts-gateway:8085
TTS_MODEL: piper
TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "false"
networks: [frontend, control, inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
# The entrypoint fixes ownership of fresh named volumes and immediately
# drops to uid/gid 10002 via gosu before starting the router. Without this
# narrowly scoped capabilities a clean installation cannot initialize the
# volumes and then switch to its unprivileged runtime identity.
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8081/health', timeout=2)"]
interval: 5s
timeout: 3s
retries: 24
start_period: 5s
depends_on:
wireguard-gateway:
condition: service_healthy
profile-controller:
condition: service_healthy
piper:
condition: service_healthy
tts-gateway:
condition: service_healthy
image-worker:
build:
context: platform/docker/image-worker
args:
DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0}
TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1}
ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0}
HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0}
image: mike-ai/image-worker:local
container_name: mike-ai-image-worker
restart: "no"
profiles: [image]
labels:
com.mike-ai.image-worker: image
gpus: all
read_only: true
tmpfs: ["/tmp:size=1g,mode=1777"]
volumes:
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
- router-images:/data/images
environment:
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
IMAGE_DIR: /data/images
networks: [inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8086/health', timeout=2)"]
interval: 5s
timeout: 3s
retries: 12
piper:
build:
context: platform/docker/piper
args:
PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
image: mike-ai/piper:local
container_name: mike-ai-piper
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- piper-data:/data
environment:
PIPER_DATA_DIR: /data
PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
PIPER_VOICE_ALIAS: alloy
PIPER_HOST: 0.0.0.0
PIPER_PORT: "8085"
PIPER_MAX_TEXT_CHARS: "8000"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 30
start_period: 120s
xtts:
image: ${XTTS_IMAGE:-ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90}
container_name: mike-ai-xtts
restart: unless-stopped
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids:
- ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
capabilities: [gpu]
read_only: true
shm_size: 1g
tmpfs:
- /tmp:size=1g,mode=1777
- /root/.cache:size=2g,mode=0700
volumes:
- "${XTTS_CACHE_DIR:-/data/models/xtts-v2-cache}:/root/.local/share/tts"
environment:
COQUI_TOS_AGREED: "1"
NVIDIA_VISIBLE_DEVICES: ${XTTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
CUDA_VISIBLE_DEVICES: "0"
NUM_THREADS: "4"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1/languages"]
interval: 10s
timeout: 5s
retries: 36
start_period: 240s
tts-gateway:
build:
context: platform/docker/tts-gateway
image: mike-ai/tts-gateway:local
container_name: mike-ai-tts-gateway
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
environment:
TTS_GATEWAY_HOST: 0.0.0.0
TTS_GATEWAY_PORT: "8085"
XTTS_URL: http://xtts:80
PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy
XTTS_SPEAKER: Annmarie Nele
TTS_DEFAULT_LANGUAGE: de
# Mixed-language clip stitching caused long pauses and unintelligible
# transitions. Keep full sentences in one stable German voice.
TTS_CODE_SWITCH_ENABLED: "false"
XTTS_QUEUE_TIMEOUT: "15"
XTTS_TIMEOUT: "120"
# Short sentence-sized requests avoid long generated silences and
# truncated weather/status summaries with Annmarie Nele.
XTTS_CHUNK_CHARS: "60"
PIPER_TIMEOUT: "120"
networks: [frontend]
depends_on:
piper:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 12
start_period: 10s
llama-dashboard:
build: ./platform/llama-dashboard
image: mike-ai/llama-dashboard:local
container_name: mike-ai-llama-dashboard
restart: unless-stopped
network_mode: "service:wireguard-gateway"
gpus: all
read_only: true
tmpfs:
- /tmp:size=16m,mode=1777
volumes:
- /proc:/host/proc:ro
- /data:/host/data:ro
- /data/models:/host/models:ro
- /data/llama-dashboard:/var/lib/llama-dashboard
environment:
DASHBOARD_HOST: 0.0.0.0
DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
DASHBOARD_HISTORY_INTERVAL: "15"
DASHBOARD_DETAIL_RETENTION_DAYS: "21"
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
depends_on:
wireguard-gateway:
condition: service_healthy
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=2)"]
interval: 10s
timeout: 3s
retries: 12
start_period: 10s
portainer:
image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa}
container_name: mike-ai-portainer
restart: unless-stopped
network_mode: "service:wireguard-gateway"
command: [--no-setup-token]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
- portainer-data:/data
depends_on:
wireguard-gateway:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
backup:
image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013}
container_name: mike-ai-backup
restart: unless-stopped
environment:
BACKUP_CRON_EXPRESSION: "0 */5 * * *"
BACKUP_FILENAME: "athena-%Y-%m-%dT%H-%M-%S.tar.gz"
BACKUP_LATEST_SYMLINK: athena-latest.tar.gz
BACKUP_RETENTION_DAYS: "14"
BACKUP_PRUNING_PREFIX: athena-
volumes:
- /var/run/docker.sock:/var/run/docker.sock:ro
- /data/docker-backups:/archive
- /etc/mike-ai:/backup/etc-mike-ai:ro
- /opt/mike-ai/stack:/backup/stack:ro
- piper-data:/backup/volumes/piper-data:ro
- router-state:/backup/volumes/router-state:ro
- router-images:/backup/volumes/router-images:ro
- portainer-data:/backup/volumes/portainer-data:ro
security_opt: ["no-new-privileges:true"]
networks:
frontend:
internal: false
ipam:
config: [{subnet: 172.30.10.0/24}]
control:
internal: true
ipam:
config: [{subnet: 172.30.20.0/24}]
inference:
internal: true
ipam:
config: [{subnet: 172.30.30.0/24}]
tools:
external: true
name: mike-ai-tools
tools-egress:
external: true
name: mike-ai-tools-egress
volumes:
piper-data:
router-state:
router-images:
portainer-data:
name: portainer_data