Increase llama prompt cache to 32 GiB
This commit is contained in:
1 parent
4153e535d3
commit
fb0cb40bed
4 files changed
+8
-8
No files matched your search
+1
-1
@@ -49,7 +49,7 @@ UNCENSORED_MTP_MAX=2
|
|||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
FAST_PARALLEL_SLOTS=1
|
FAST_PARALLEL_SLOTS=1
|
||||||
LLAMA_CACHE_RAM_MIB=24576
|
LLAMA_CACHE_RAM_MIB=32768
|
||||||
MEDIUM_PARALLEL_SLOTS=1
|
MEDIUM_PARALLEL_SLOTS=1
|
||||||
LARGE_PARALLEL_SLOTS=1
|
LARGE_PARALLEL_SLOTS=1
|
||||||
ULTRA_PARALLEL_SLOTS=1
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
|
|||||||
+5
-5
@@ -104,7 +104,7 @@ services:
|
|||||||
# disabled until the current upstream restore regressions are fixed.
|
# disabled until the current upstream restore regressions are fixed.
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
- "${LLAMA_CACHE_RAM_MIB:-32768}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -179,7 +179,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
- "${LLAMA_CACHE_RAM_MIB:-32768}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -264,7 +264,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
- "${LLAMA_CACHE_RAM_MIB:-32768}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -342,7 +342,7 @@ services:
|
|||||||
- --cache-reuse
|
- --cache-reuse
|
||||||
- "${LLAMA_CACHE_REUSE:-256}"
|
- "${LLAMA_CACHE_REUSE:-256}"
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
- "${LLAMA_CACHE_RAM_MIB:-32768}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
@@ -424,7 +424,7 @@ services:
|
|||||||
- q4_0
|
- q4_0
|
||||||
- --cache-prompt
|
- --cache-prompt
|
||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-24576}"
|
- "${LLAMA_CACHE_RAM_MIB:-32768}"
|
||||||
- --threads
|
- --threads
|
||||||
- "${LLAMA_THREADS:-6}"
|
- "${LLAMA_THREADS:-6}"
|
||||||
- --threads-batch
|
- --threads-batch
|
||||||
|
|||||||
@@ -98,7 +98,7 @@ UNCENSORED_MTP_MAX=2
|
|||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
FAST_PARALLEL_SLOTS=1
|
FAST_PARALLEL_SLOTS=1
|
||||||
LLAMA_CACHE_RAM_MIB=24576
|
LLAMA_CACHE_RAM_MIB=32768
|
||||||
MEDIUM_PARALLEL_SLOTS=1
|
MEDIUM_PARALLEL_SLOTS=1
|
||||||
LARGE_PARALLEL_SLOTS=1
|
LARGE_PARALLEL_SLOTS=1
|
||||||
ULTRA_PARALLEL_SLOTS=1
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
|
|||||||
+1
-1
@@ -358,7 +358,7 @@ IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}}
|
|||||||
FLUX_MODEL_DIR=${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}
|
FLUX_MODEL_DIR=${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}
|
||||||
LLAMA_THREADS=${LLAMA_THREADS:-6}
|
LLAMA_THREADS=${LLAMA_THREADS:-6}
|
||||||
LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6}
|
LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6}
|
||||||
LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-24576}
|
LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768}
|
||||||
DEFAULT_REASONING_EFFORT=${DEFAULT_REASONING_EFFORT:-off}
|
DEFAULT_REASONING_EFFORT=${DEFAULT_REASONING_EFFORT:-off}
|
||||||
EOF
|
EOF
|
||||||
chmod 0600 $SECRETS_DIR/stack.env
|
chmod 0600 $SECRETS_DIR/stack.env
|
||||||
|
|||||||
Reference in new issue
Block a user