Increase llama prompt cache to 24 GiB

This commit is contained in:
Mikei386
2026-08-30 22:33:37 +02:00
parent a41d6dc28c
commit 0fa9a34886
3 changed files with 10 additions and 8 deletions
+2 -1
View File
@@ -52,7 +52,8 @@ EXPERIMENTAL_GPU_DEVICES=0
LLAMA_THREADS=6 LLAMA_THREADS=6
LLAMA_THREADS_BATCH=6 LLAMA_THREADS_BATCH=6
FAST_PARALLEL_SLOTS=1 FAST_PARALLEL_SLOTS=1
MEDIUM_PARALLEL_SLOTS=2 LLAMA_CACHE_RAM_MIB=24576
MEDIUM_PARALLEL_SLOTS=1
LARGE_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1
+6 -6
View File
@@ -99,7 +99,7 @@ services:
# disabled until the current upstream restore regressions are fixed. # disabled until the current upstream restore regressions are fixed.
- --cache-prompt - --cache-prompt
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads - --threads
- "${LLAMA_THREADS:-6}" - "${LLAMA_THREADS:-6}"
- --threads-batch - --threads-batch
@@ -176,7 +176,7 @@ services:
- q4_0 - q4_0
- --cache-prompt - --cache-prompt
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads - --threads
- "${LLAMA_THREADS:-6}" - "${LLAMA_THREADS:-6}"
- --threads-batch - --threads-batch
@@ -264,7 +264,7 @@ services:
- q4_0 - q4_0
- --cache-prompt - --cache-prompt
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads - --threads
- "${LLAMA_THREADS:-6}" - "${LLAMA_THREADS:-6}"
- --threads-batch - --threads-batch
@@ -342,7 +342,7 @@ services:
- --cache-reuse - --cache-reuse
- "${LLAMA_CACHE_REUSE:-256}" - "${LLAMA_CACHE_REUSE:-256}"
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads - --threads
- "${LLAMA_THREADS:-6}" - "${LLAMA_THREADS:-6}"
- --threads-batch - --threads-batch
@@ -424,7 +424,7 @@ services:
- q4_0 - q4_0
- --cache-prompt - --cache-prompt
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --threads - --threads
- "${LLAMA_THREADS:-6}" - "${LLAMA_THREADS:-6}"
- --threads-batch - --threads-batch
@@ -501,7 +501,7 @@ services:
- --cache-reuse - --cache-reuse
- "${LLAMA_CACHE_REUSE:-256}" - "${LLAMA_CACHE_REUSE:-256}"
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-24576}"
- --parallel - --parallel
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}" - "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
- --kv-unified - --kv-unified
+2 -1
View File
@@ -100,7 +100,8 @@ EXPERIMENTAL_CONTEXT=76800
LLAMA_THREADS=6 LLAMA_THREADS=6
LLAMA_THREADS_BATCH=6 LLAMA_THREADS_BATCH=6
FAST_PARALLEL_SLOTS=1 FAST_PARALLEL_SLOTS=1
MEDIUM_PARALLEL_SLOTS=2 LLAMA_CACHE_RAM_MIB=24576
MEDIUM_PARALLEL_SLOTS=1
LARGE_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1