Define four-profile production matrix with Medium default

This commit is contained in:
Mikei386
2026-08-22 11:30:38 +02:00
parent 977f8f5c76
commit 41b9c17f0f
33 changed files with 237 additions and 160 deletions
+35 -20
View File
@@ -100,7 +100,7 @@ services:
labels:
com.mike-ai.llama-profile: medium
environment:
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
@@ -111,7 +111,7 @@ services:
- --alias
- qwen-medium
- --ctx-size
- "${MEDIUM_CONTEXT:-94208}"
- "${MEDIUM_CONTEXT:-160000}"
- --flash-attn
- "on"
- --cache-type-k
@@ -149,28 +149,40 @@ services:
- --top-k
- "20"
- --device
- CUDA0
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- none
- layer
- --tensor-split
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-long:
llama-large:
<<: *llama-common
container_name: mike-ai-llama-long
container_name: mike-ai-llama-large
labels:
com.mike-ai.llama-profile: long
com.mike-ai.llama-profile: large
environment:
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}"
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-long
- qwen-large
- --ctx-size
- "${LONG_CONTEXT:-131072}"
- "${LARGE_CONTEXT:-192000}"
- --flash-attn
- "on"
- --cache-type-k
@@ -199,8 +211,6 @@ services:
- "off"
- --n-gpu-layers
- all
- --override-tensor
- blk.([0-9]|1[0-1]).ffn_.*=CPU
- --no-mmap
- --no-ui
- --temperature
@@ -210,19 +220,23 @@ services:
- --top-k
- "20"
- --device
- CUDA0
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- none
- layer
- --tensor-split
- "${LARGE_TENSOR_SPLIT:-86,14}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- "3"
- --spec-draft-type-k
- q4_0
- f16
- --spec-draft-type-v
- q4_0
- f16
# Text-only long-context profile. This exact IQ4_XS-pure / 256K / 80:20
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
llama-ultra:
@@ -347,7 +361,7 @@ services:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,long,ultra,experimental
ALLOWED_PROFILES: fast,medium,large,ultra,experimental
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -456,6 +470,7 @@ services:
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
environment:
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
DEFAULT_MODELS: qwen-medium
OLLAMA_BASE_URL: ""
OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"