Replace compression sidecar with Hermes micro-compaction
This commit is contained in:
@@ -72,11 +72,6 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74
|
||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
||||
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
COMPRESSION_MODEL_FILE=qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf
|
||||
COMPRESSION_MODEL_URL=https://huggingface.co/bartowski/Qwen_Qwen3.5-4B-GGUF/resolve/main/Qwen_Qwen3.5-4B-Q4_K_M.gguf
|
||||
COMPRESSION_MODEL_SHA256=13c16f426047e2de38cd075bdade4a7bcbc8c774384876f677740cda65f8a983
|
||||
COMPRESSION_CONTEXT=65536
|
||||
COMPRESSION_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||
# All standard profiles use the MTP tensor embedded in their GGUF. A separate
|
||||
# draft-model artifact is neither downloaded nor passed to llama-server.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user