Increase llama prompt cache to 24 GiB
This commit is contained in:
+2
-1
@@ -52,7 +52,8 @@ EXPERIMENTAL_GPU_DEVICES=0
|
||||
LLAMA_THREADS=6
|
||||
LLAMA_THREADS_BATCH=6
|
||||
FAST_PARALLEL_SLOTS=1
|
||||
MEDIUM_PARALLEL_SLOTS=2
|
||||
LLAMA_CACHE_RAM_MIB=24576
|
||||
MEDIUM_PARALLEL_SLOTS=1
|
||||
LARGE_PARALLEL_SLOTS=1
|
||||
ULTRA_PARALLEL_SLOTS=1
|
||||
UNCENSORED_PARALLEL_SLOTS=1
|
||||
|
||||
+6
-6
@@ -99,7 +99,7 @@ services:
|
||||
# disabled until the current upstream restore regressions are fixed.
|
||||
- --cache-prompt
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
@@ -176,7 +176,7 @@ services:
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
@@ -264,7 +264,7 @@ services:
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
@@ -342,7 +342,7 @@ services:
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
@@ -424,7 +424,7 @@ services:
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --threads
|
||||
- "${LLAMA_THREADS:-6}"
|
||||
- --threads-batch
|
||||
@@ -501,7 +501,7 @@ services:
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||
- --parallel
|
||||
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
|
||||
@@ -100,7 +100,8 @@ EXPERIMENTAL_CONTEXT=76800
|
||||
LLAMA_THREADS=6
|
||||
LLAMA_THREADS_BATCH=6
|
||||
FAST_PARALLEL_SLOTS=1
|
||||
MEDIUM_PARALLEL_SLOTS=2
|
||||
LLAMA_CACHE_RAM_MIB=24576
|
||||
MEDIUM_PARALLEL_SLOTS=1
|
||||
LARGE_PARALLEL_SLOTS=1
|
||||
ULTRA_PARALLEL_SLOTS=1
|
||||
UNCENSORED_PARALLEL_SLOTS=1
|
||||
|
||||
Reference in New Issue
Block a user