Add dedicated Hermes compression model

This commit is contained in:
Mikei386
2026-08-26 00:45:32 +02:00
parent e872829b11
commit 43d9903d13
8 changed files with 120 additions and 3 deletions
+69
View File
@@ -65,6 +65,75 @@ services:
retries: 12
start_period: 10s
# Small, non-thinking side model used only for Hermes context compression.
# It shares the WireGuard namespace, so Unraid can reach port 8099 while the
# university LAN receives no published host port. The stable GPU UUID keeps
# it on the RTX 3060 even if PCI enumeration changes again.
llama-compression:
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
container_name: mike-ai-llama-compression
restart: unless-stopped
profiles: [compression]
gpus: all
network_mode: "service:wireguard-gateway"
depends_on:
wireguard-gateway:
condition: service_healthy
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
environment:
NVIDIA_VISIBLE_DEVICES: ${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
command:
- --model
- "/models/${COMPRESSION_MODEL_FILE:-qwen3.5-2b-compression/Qwen_Qwen3.5-2B-Q4_K_M.gguf}"
- --alias
- qwen-compression
- --ctx-size
- "${COMPRESSION_CONTEXT:-65536}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --parallel
- "1"
- --jinja
- --host
- 0.0.0.0
- --port
- "8099"
- --metrics
- --n-gpu-layers
- all
- --device
- CUDA0
- --split-mode
- none
- --no-mmap
- --no-ui
- --batch-size
- "512"
- --ubatch-size
- "256"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8099/health"]
interval: 10s
timeout: 5s
retries: 60
start_period: 30s
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast