Add tested 256K Ultra profile

This commit is contained in:
Mikei386
2026-08-22 10:44:14 +02:00
parent a194a2941d
commit f52cf14069
12 changed files with 111 additions and 13 deletions
+72 -1
View File
@@ -222,6 +222,77 @@ services:
- --spec-draft-type-v
- q4_0
# Text-only long-context profile. This exact IQ4-MIX / 256K / 80:20
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
llama-ultra:
<<: *llama-common
container_name: mike-ai-llama-ultra
labels:
com.mike-ai.llama-profile: ultra
environment:
NVIDIA_VISIBLE_DEVICES: ${ULTRA_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
- --model
- "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}"
- --alias
- qwen-ultra
- --ctx-size
- "${ULTRA_CONTEXT:-262144}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "64"
- --ubatch-size
- "32"
- --parallel
- "1"
- --jinja
- --reasoning
- auto
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "0.2"
- --top-p
- "0.8"
- --top-k
- "20"
- --device
- CUDA0,CUDA1
- --main-gpu
- "0"
- --split-mode
- layer
- --tensor-split
- "${ULTRA_TENSOR_SPLIT:-80,20}"
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
llama-experimental:
<<: *llama-common
container_name: mike-ai-llama-experimental
@@ -276,7 +347,7 @@ services:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,long,experimental
ALLOWED_PROFILES: fast,medium,long,ultra,experimental
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck: