Add dedicated Hermes compression model
This commit is contained in:
@@ -65,6 +65,75 @@ services:
|
||||
retries: 12
|
||||
start_period: 10s
|
||||
|
||||
# Small, non-thinking side model used only for Hermes context compression.
|
||||
# It shares the WireGuard namespace, so Unraid can reach port 8099 while the
|
||||
# university LAN receives no published host port. The stable GPU UUID keeps
|
||||
# it on the RTX 3060 even if PCI enumeration changes again.
|
||||
llama-compression:
|
||||
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
|
||||
container_name: mike-ai-llama-compression
|
||||
restart: unless-stopped
|
||||
profiles: [compression]
|
||||
gpus: all
|
||||
network_mode: "service:wireguard-gateway"
|
||||
depends_on:
|
||||
wireguard-gateway:
|
||||
condition: service_healthy
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp:size=256m,mode=1777
|
||||
volumes:
|
||||
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
cap_drop: [ALL]
|
||||
command:
|
||||
- --model
|
||||
- "/models/${COMPRESSION_MODEL_FILE:-qwen3.5-2b-compression/Qwen_Qwen3.5-2B-Q4_K_M.gguf}"
|
||||
- --alias
|
||||
- qwen-compression
|
||||
- --ctx-size
|
||||
- "${COMPRESSION_CONTEXT:-65536}"
|
||||
- --flash-attn
|
||||
- "on"
|
||||
- --cache-type-k
|
||||
- q4_0
|
||||
- --cache-type-v
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --parallel
|
||||
- "1"
|
||||
- --jinja
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8099"
|
||||
- --metrics
|
||||
- --n-gpu-layers
|
||||
- all
|
||||
- --device
|
||||
- CUDA0
|
||||
- --split-mode
|
||||
- none
|
||||
- --no-mmap
|
||||
- --no-ui
|
||||
- --batch-size
|
||||
- "512"
|
||||
- --ubatch-size
|
||||
- "256"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
- "${LLAMA_THREADS_BATCH:-6}"
|
||||
healthcheck:
|
||||
test: [CMD, curl, -fsS, "http://127.0.0.1:8099/health"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 60
|
||||
start_period: 30s
|
||||
|
||||
llama-fast:
|
||||
<<: *llama-common
|
||||
container_name: mike-ai-llama-fast
|
||||
|
||||
Reference in New Issue
Block a user