Enable CPU vision projector for Ultra profile

This commit is contained in:
Mikei386
2026-09-24 05:09:10 +02:00
parent 33c04150e3
commit 6378b50086
11 changed files with 46 additions and 14 deletions
+5 -3
View File
@@ -394,9 +394,8 @@ services:
- --spec-draft-type-v
- f16
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
# Maximum-context profile. Keep the vision projector on CPU so images work
# without consuming the tightly budgeted GPU memory of the 256K context.
llama-ultra:
<<: *llama-common
container_name: mike-ai-llama-ultra
@@ -408,6 +407,9 @@ services:
command:
- --model
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --no-mmproj-offload
- --alias
- qwen-ultra
- --ctx-size