Define four-profile production matrix with Medium default
This commit is contained in:
+35
-20
@@ -100,7 +100,7 @@ services:
|
||||
labels:
|
||||
com.mike-ai.llama-profile: medium
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
|
||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
command:
|
||||
- --model
|
||||
@@ -111,7 +111,7 @@ services:
|
||||
- --alias
|
||||
- qwen-medium
|
||||
- --ctx-size
|
||||
- "${MEDIUM_CONTEXT:-94208}"
|
||||
- "${MEDIUM_CONTEXT:-160000}"
|
||||
- --flash-attn
|
||||
- "on"
|
||||
- --cache-type-k
|
||||
@@ -149,28 +149,40 @@ services:
|
||||
- --top-k
|
||||
- "20"
|
||||
- --device
|
||||
- CUDA0
|
||||
- CUDA0,CUDA1
|
||||
- --main-gpu
|
||||
- "0"
|
||||
- --split-mode
|
||||
- none
|
||||
- layer
|
||||
- --tensor-split
|
||||
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
||||
- --spec-type
|
||||
- draft-mtp
|
||||
- --spec-draft-n-max
|
||||
- "3"
|
||||
- --spec-draft-type-k
|
||||
- f16
|
||||
- --spec-draft-type-v
|
||||
- f16
|
||||
|
||||
llama-long:
|
||||
llama-large:
|
||||
<<: *llama-common
|
||||
container_name: mike-ai-llama-long
|
||||
container_name: mike-ai-llama-large
|
||||
labels:
|
||||
com.mike-ai.llama-profile: long
|
||||
com.mike-ai.llama-profile: large
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
|
||||
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
command:
|
||||
- --model
|
||||
- "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}"
|
||||
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
||||
- --mmproj
|
||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||
- --no-mmproj-offload
|
||||
- --alias
|
||||
- qwen-long
|
||||
- qwen-large
|
||||
- --ctx-size
|
||||
- "${LONG_CONTEXT:-131072}"
|
||||
- "${LARGE_CONTEXT:-192000}"
|
||||
- --flash-attn
|
||||
- "on"
|
||||
- --cache-type-k
|
||||
@@ -199,8 +211,6 @@ services:
|
||||
- "off"
|
||||
- --n-gpu-layers
|
||||
- all
|
||||
- --override-tensor
|
||||
- blk.([0-9]|1[0-1]).ffn_.*=CPU
|
||||
- --no-mmap
|
||||
- --no-ui
|
||||
- --temperature
|
||||
@@ -210,19 +220,23 @@ services:
|
||||
- --top-k
|
||||
- "20"
|
||||
- --device
|
||||
- CUDA0
|
||||
- CUDA0,CUDA1
|
||||
- --main-gpu
|
||||
- "0"
|
||||
- --split-mode
|
||||
- none
|
||||
- layer
|
||||
- --tensor-split
|
||||
- "${LARGE_TENSOR_SPLIT:-86,14}"
|
||||
- --spec-type
|
||||
- draft-mtp
|
||||
- --spec-draft-n-max
|
||||
- "2"
|
||||
- "3"
|
||||
- --spec-draft-type-k
|
||||
- q4_0
|
||||
- f16
|
||||
- --spec-draft-type-v
|
||||
- q4_0
|
||||
- f16
|
||||
|
||||
# Text-only long-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
||||
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
||||
llama-ultra:
|
||||
@@ -347,7 +361,7 @@ services:
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
environment:
|
||||
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
ALLOWED_PROFILES: fast,medium,long,ultra,experimental
|
||||
ALLOWED_PROFILES: fast,medium,large,ultra,experimental
|
||||
networks: [control]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
@@ -456,6 +470,7 @@ services:
|
||||
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
||||
environment:
|
||||
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
||||
DEFAULT_MODELS: qwen-medium
|
||||
OLLAMA_BASE_URL: ""
|
||||
OPENAI_API_BASE_URLS: http://router:8081/v1
|
||||
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
|
||||
Reference in New Issue
Block a user