From 0fa9a3488607c62e4fb5be5a8e0aeaddf8a12988 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sun, 30 Aug 2026 22:32:40 +0200 Subject: [PATCH] Increase llama prompt cache to 24 GiB --- .env.example | 3 ++- compose.yaml | 12 ++++++------ config/install.env.example | 3 ++- 3 files changed, 10 insertions(+), 8 deletions(-) diff --git a/.env.example b/.env.example index f8418bc..d6c9b28 100644 --- a/.env.example +++ b/.env.example @@ -52,7 +52,8 @@ EXPERIMENTAL_GPU_DEVICES=0 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 FAST_PARALLEL_SLOTS=1 -MEDIUM_PARALLEL_SLOTS=2 +LLAMA_CACHE_RAM_MIB=24576 +MEDIUM_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1 diff --git a/compose.yaml b/compose.yaml index 55113c0..b633fb4 100644 --- a/compose.yaml +++ b/compose.yaml @@ -99,7 +99,7 @@ services: # disabled until the current upstream restore regressions are fixed. - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -176,7 +176,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -264,7 +264,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -342,7 +342,7 @@ services: - --cache-reuse - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -424,7 +424,7 @@ services: - q4_0 - --cache-prompt - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --threads - "${LLAMA_THREADS:-6}" - --threads-batch @@ -501,7 +501,7 @@ services: - --cache-reuse - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-8192}" + - "${LLAMA_CACHE_RAM_MIB:-24576}" - --parallel - "${EXPERIMENTAL_PARALLEL_SLOTS:-1}" - --kv-unified diff --git a/config/install.env.example b/config/install.env.example index e12c4e6..da3221f 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -100,7 +100,8 @@ EXPERIMENTAL_CONTEXT=76800 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 FAST_PARALLEL_SLOTS=1 -MEDIUM_PARALLEL_SLOTS=2 +LLAMA_CACHE_RAM_MIB=24576 +MEDIUM_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1