From c25e57af5719ed6ccd3d56010bd2898cbe80597f Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Tue, 25 Aug 2026 19:26:48 +0200 Subject: [PATCH] Limit prompt cache reuse to text-only profiles --- compose.yaml | 13 ++----------- docs/CURRENT_REFERENCE.md | 10 ++++++++-- docs/UNRAID_AUTOMATIC_UPDATE_WORKFLOW.md | 9 +++++++++ platform/profiles/profile-fast.conf | 2 +- platform/profiles/profile-large.conf | 2 +- platform/profiles/profile-medium.conf | 2 +- 6 files changed, 22 insertions(+), 16 deletions(-) diff --git a/compose.yaml b/compose.yaml index f8c07d3..8eaae2e 100644 --- a/compose.yaml +++ b/compose.yaml @@ -90,12 +90,9 @@ services: - q4_0 - --cache-type-v - q4_0 - # Keep the cross-chat prefix cache explicit. cache-reuse tolerates small - # changes after Hermes' stable system-prompt prefix without enabling the - # currently unreliable on-disk slot restore path. + # Keep the cross-chat prefix cache explicit. On-disk slot restore stays + # disabled until the current upstream restore regressions are fixed. - --cache-prompt - - --cache-reuse - - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-8192}" - --threads @@ -172,8 +169,6 @@ services: - --cache-type-v - q4_0 - --cache-prompt - - --cache-reuse - - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-8192}" - --threads @@ -261,8 +256,6 @@ services: - --cache-type-v - q4_0 - --cache-prompt - - --cache-reuse - - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-8192}" - --threads @@ -421,8 +414,6 @@ services: - --cache-type-v - q4_0 - --cache-prompt - - --cache-reuse - - "${LLAMA_CACHE_REUSE:-256}" - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-8192}" - --threads diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index 0c5e91b..6dd45fc 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -44,8 +44,14 @@ Zielplattform. - Flash Attention - KV-Cache Q4_0 für K und V - explizites Prompt-Caching mit 8.192 MiB profilinternem RAM-Cache -- `--cache-reuse 256` für die Wiederverwendung langer stabiler Präfixe trotz - kleiner späterer Abweichungen +- exakte Wiederverwendung bereits verarbeiteter Prompt-Präfixe über + `--cache-prompt`; Hermes trennt seinen System-Prompt zusätzlich in einen + stabilen, einen kontextabhängigen und einen flüchtigen Teil +- `--cache-reuse 256` nur in den text-only-Profilen Ultra und Experimental; + llama.cpp deaktiviert diese unscharfe Wiederverwendung ausdrücklich, sobald + ein Vision-Projektor geladen ist +- kein persistenter Slot-Cache auf Datenträger, bis die bekannten + llama.cpp-Restore-Regressions behoben sind - MTP Draft, maximal drei Tokens - MTP-Akzeptanzschwelle 0,05; im Referenzlauf 77,26 statt 73,88 Tok/s - sechs Threads und sechs Batch-Threads diff --git a/docs/UNRAID_AUTOMATIC_UPDATE_WORKFLOW.md b/docs/UNRAID_AUTOMATIC_UPDATE_WORKFLOW.md index 5986b98..6375b33 100644 --- a/docs/UNRAID_AUTOMATIC_UPDATE_WORKFLOW.md +++ b/docs/UNRAID_AUTOMATIC_UPDATE_WORKFLOW.md @@ -59,6 +59,15 @@ Für eine frische Anzeige muss Unraids eigener Statuslauf `dynamix.docker.manager/scripts/dockerupdate check` abgeschlossen sein. Das ist eine Aktualisierung der Anzeige und kein erneuter Container-Rebuild. +Auch nach diesem nativen Statuslauf kann Unraid einzelne Images weiterhin als +Update markieren, obwohl Container-Image-ID und lokale Tag-Image-ID identisch +sind. Das kommt insbesondere bei Registry-/Manifest- und Multiarch-Digest- +Vergleichen vor. In diesem Konfliktfall ist das Ergebnis von +`unraid_docker_update_verified_batch` nach dem Pull maßgeblich: identische +unveränderliche Image-IDs bedeuten `already-current`; ein weiterer Rebuild nur +zum Entfernen der GUI-Anzeige ist weder nötig noch erwünscht. Die GUI-Meldung +ist dann ausdrücklich als Fehlanzeige zu melden. + Ein wiederholter Lauf muss bei einem aktuellen Image folgendes melden: ```text diff --git a/platform/profiles/profile-fast.conf b/platform/profiles/profile-fast.conf index d639c94..6efde84 100644 --- a/platform/profiles/profile-fast.conf +++ b/platform/profiles/profile-fast.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-large.conf b/platform/profiles/profile-large.conf index fd02f9e..b19ea43 100644 --- a/platform/profiles/profile-large.conf +++ b/platform/profiles/profile-large.conf @@ -3,4 +3,4 @@ Description=Legacy native Qwen Large 192K profile (Docker is the production path [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-medium.conf b/platform/profiles/profile-medium.conf index 67a4176..b9f7a89 100644 --- a/platform/profiles/profile-medium.conf +++ b/platform/profiles/profile-medium.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16