Increase llama prompt cache to 24 GiB
This commit is contained in:
1 parent
a41d6dc28c
commit
0fa9a34886
3 files changed
+10
-8
No files matched your search
+2
-1
@@ -52,7 +52,8 @@ EXPERIMENTAL_GPU_DEVICES=0
|
|||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
FAST_PARALLEL_SLOTS=1
|
FAST_PARALLEL_SLOTS=1
|
||||||
MEDIUM_PARALLEL_SLOTS=2
|
LLAMA_CACHE_RAM_MIB=24576
|
||||||
|
MEDIUM_PARALLEL_SLOTS=1
|
||||||
LARGE_PARALLEL_SLOTS=1
|
LARGE_PARALLEL_SLOTS=1
|
||||||
ULTRA_PARALLEL_SLOTS=1
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
UNCENSORED_PARALLEL_SLOTS=1
|
UNCENSORED_PARALLEL_SLOTS=1
|
||||||
|
|||||||
+6
-6
@@ -99,7 +99,7 @@ services:
|
|||||||
# disabled until the current upstream restore regressions are fixed.
|
# disabled until the current upstream restore regressions are fixed.
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -176,7 +176,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -264,7 +264,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -342,7 +342,7 @@ services:
|
|||||||
- --cache-reuse
|
- --cache-reuse
|
||||||
- "${LLAMA_CACHE_REUSE:-256}"
|
- "${LLAMA_CACHE_REUSE:-256}"
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -424,7 +424,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -501,7 +501,7 @@ services:
|
|||||||
- --cache-reuse
|
- --cache-reuse
|
||||||
- "${LLAMA_CACHE_REUSE:-256}"
|
- "${LLAMA_CACHE_REUSE:-256}"
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
|
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
|
||||||
- --kv-unified
|
- --kv-unified
|
||||||
|
|||||||
@@ -100,7 +100,8 @@ EXPERIMENTAL_CONTEXT=76800
|
|||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
FAST_PARALLEL_SLOTS=1
|
FAST_PARALLEL_SLOTS=1
|
||||||
MEDIUM_PARALLEL_SLOTS=2
|
LLAMA_CACHE_RAM_MIB=24576
|
||||||
|
MEDIUM_PARALLEL_SLOTS=1
|
||||||
LARGE_PARALLEL_SLOTS=1
|
LARGE_PARALLEL_SLOTS=1
|
||||||
ULTRA_PARALLEL_SLOTS=1
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
UNCENSORED_PARALLEL_SLOTS=1
|
UNCENSORED_PARALLEL_SLOTS=1
|
||||||
|
|||||||
Reference in new issue
Block a user