Enable CPU vision projector for Ultra profile
This commit is contained in:
+5
-3
@@ -394,9 +394,8 @@ services:
|
||||
- --spec-draft-type-v
|
||||
- f16
|
||||
|
||||
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
||||
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
||||
# Maximum-context profile. Keep the vision projector on CPU so images work
|
||||
# without consuming the tightly budgeted GPU memory of the 256K context.
|
||||
llama-ultra:
|
||||
<<: *llama-common
|
||||
container_name: mike-ai-llama-ultra
|
||||
@@ -408,6 +407,9 @@ services:
|
||||
command:
|
||||
- --model
|
||||
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
|
||||
- --mmproj
|
||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||
- --no-mmproj-offload
|
||||
- --alias
|
||||
- qwen-ultra
|
||||
- --ctx-size
|
||||
|
||||
Reference in New Issue
Block a user