Replace compression sidecar with Hermes micro-compaction

This commit is contained in:
Mikei386
2026-08-26 06:18:51 +02:00
parent 438bd34b72
commit 896c12193e
8 changed files with 23 additions and 140 deletions
-90
View File
@@ -65,96 +65,6 @@ services:
retries: 12
start_period: 10s
# Small, non-thinking side model used only for Hermes context compression.
# It shares the WireGuard namespace, so Unraid can reach port 8099 while the
# university LAN receives no published host port. The stable GPU UUID keeps
# it on the RTX 3060 even if PCI enumeration changes again.
llama-compression:
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
container_name: mike-ai-llama-compression
restart: unless-stopped
profiles: [compression]
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}"]
capabilities: [gpu]
network_mode: "service:wireguard-gateway"
depends_on:
wireguard-gateway:
condition: service_healthy
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
environment:
NVIDIA_VISIBLE_DEVICES: ${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
command:
- --model
- "/models/${COMPRESSION_MODEL_FILE:-qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf}"
- --alias
- qwen-compression
- --ctx-size
- "${COMPRESSION_CONTEXT:-65536}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --parallel
- "1"
- --jinja
- --host
- 0.0.0.0
- --port
- "8099"
- --metrics
# Official Qwen3.5 non-thinking sampler. The presence penalty is
# essential here: without it one technical YAML benchmark repeated the
# summary structure for more than 40K tokens.
- --temperature
- "1.0"
- --top-p
- "1.0"
- --top-k
- "20"
- --presence-penalty
- "2.0"
# Hermes' own summary ceiling is 10K. Keep a little completion margin,
# but never allow an accidental repetition loop to fill the 65K slot.
- --n-predict
- "12000"
- --n-gpu-layers
- all
- --device
- CUDA0
- --split-mode
- none
- --no-mmap
- --no-ui
- --batch-size
- "512"
- --ubatch-size
- "256"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8099/health"]
interval: 10s
timeout: 5s
retries: 60
start_period: 30s
llama-fast:
<<: *llama-common
container_name: mike-ai-llama-fast