Offload vision projector to RTX 3060

This commit is contained in:
Mikei386
2026-08-22 17:19:56 +02:00
parent 7513a0c912
commit 8267a85a96
6 changed files with 119 additions and 6 deletions
+12 -4
View File
@@ -33,14 +33,18 @@ services:
labels:
com.mike-ai.llama-profile: fast
environment:
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
# CUDA0 remains the exclusive text-model device. The projector is kept
# on the secondary card so vision does not consume the 5080 context
# budget.
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --mmproj-offload
- --alias
- qwen-fast
- --ctx-size
@@ -102,12 +106,15 @@ services:
environment:
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
# Keep the language model split unchanged while placing the complete
# multimodal projector on the secondary RTX 3060.
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --mmproj-offload
- --alias
- qwen-medium
- --ctx-size
@@ -173,12 +180,13 @@ services:
environment:
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --mmproj-offload
- --alias
- qwen-large
- --ctx-size