From fb0cb40bed8e6cecb1076b7877a6d62051744b86 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Thu, 3 Sep 2026 10:03:43 +0200 Subject: [PATCH] Increase llama prompt cache to 32 GiB --- .env.example | 2 +- compose.yaml | 10 +++++----- config/install.env.example | 2 +- install.sh | 2 +- 4 files changed, 8 insertions(+), 8 deletions(-) diff --git a/.env.example b/.env.example index e19ce19..486fcd2 100644 --- a/.env.example +++ b/.env.example @@ -49,7 +49,7 @@ UNCENSORED_MTP_MAX=2 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 FAST_PARALLEL_SLOTS=1 -LLAMA_CACHE_RAM_MIB=24576 +LLAMA_CACHE_RAM_MIB=32768 MEDIUM_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1 diff --git a/compose.yaml b/compose.yaml index 83443ea..0bd9a0f 100644 --- a/compose.yaml +++ b/compose.yaml @@ -104,7 +104,7 @@ services: # disabled until the current upstream restore regressions are fixed. - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-24576}" + - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -179,7 +179,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-24576}" + - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -264,7 +264,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-24576}" + - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -342,7 +342,7 @@ services: - --cache-reuse - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-24576}" + - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -424,7 +424,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-24576}" + - "${LLAMA_CACHE_RAM_MIB:-32768}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch diff --git a/config/install.env.example b/config/install.env.example index 1a9da66..58a6c91 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -98,7 +98,7 @@ UNCENSORED_MTP_MAX=2 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 FAST_PARALLEL_SLOTS=1 -LLAMA_CACHE_RAM_MIB=24576 +LLAMA_CACHE_RAM_MIB=32768 MEDIUM_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1 diff --git a/install.sh b/install.sh index 3dda732..53f6214 100755 --- a/install.sh +++ b/install.sh @@ -358,7 +358,7 @@ IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}} FLUX_MODEL_DIR=${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B} LLAMA_THREADS=${LLAMA_THREADS:-6} LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6} -LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-24576} +LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768} DEFAULT_REASONING_EFFORT=${DEFAULT_REASONING_EFFORT:-off} EOF chmod 0600 $SECRETS_DIR/stack.env