Limit prompt cache reuse to text-only profiles
This commit is contained in:
+2
-11
@@ -90,12 +90,9 @@ services:
|
||||
- q4_0
|
||||
- --cache-type-v
|
||||
- q4_0
|
||||
# Keep the cross-chat prefix cache explicit. cache-reuse tolerates small
|
||||
# changes after Hermes' stable system-prompt prefix without enabling the
|
||||
# currently unreliable on-disk slot restore path.
|
||||
# Keep the cross-chat prefix cache explicit. On-disk slot restore stays
|
||||
# disabled until the current upstream restore regressions are fixed.
|
||||
- --cache-prompt
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- --threads
|
||||
@@ -172,8 +169,6 @@ services:
|
||||
- --cache-type-v
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- --threads
|
||||
@@ -261,8 +256,6 @@ services:
|
||||
- --cache-type-v
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- --threads
|
||||
@@ -421,8 +414,6 @@ services:
|
||||
- --cache-type-v
|
||||
- q4_0
|
||||
- --cache-prompt
|
||||
- --cache-reuse
|
||||
- "${LLAMA_CACHE_REUSE:-256}"
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- --threads
|
||||
|
||||
@@ -44,8 +44,14 @@ Zielplattform.
|
||||
- Flash Attention
|
||||
- KV-Cache Q4_0 für K und V
|
||||
- explizites Prompt-Caching mit 8.192 MiB profilinternem RAM-Cache
|
||||
- `--cache-reuse 256` für die Wiederverwendung langer stabiler Präfixe trotz
|
||||
kleiner späterer Abweichungen
|
||||
- exakte Wiederverwendung bereits verarbeiteter Prompt-Präfixe über
|
||||
`--cache-prompt`; Hermes trennt seinen System-Prompt zusätzlich in einen
|
||||
stabilen, einen kontextabhängigen und einen flüchtigen Teil
|
||||
- `--cache-reuse 256` nur in den text-only-Profilen Ultra und Experimental;
|
||||
llama.cpp deaktiviert diese unscharfe Wiederverwendung ausdrücklich, sobald
|
||||
ein Vision-Projektor geladen ist
|
||||
- kein persistenter Slot-Cache auf Datenträger, bis die bekannten
|
||||
llama.cpp-Restore-Regressions behoben sind
|
||||
- MTP Draft, maximal drei Tokens
|
||||
- MTP-Akzeptanzschwelle 0,05; im Referenzlauf 77,26 statt 73,88 Tok/s
|
||||
- sechs Threads und sechs Batch-Threads
|
||||
|
||||
@@ -59,6 +59,15 @@ Für eine frische Anzeige muss Unraids eigener Statuslauf
|
||||
`dynamix.docker.manager/scripts/dockerupdate check` abgeschlossen sein. Das ist
|
||||
eine Aktualisierung der Anzeige und kein erneuter Container-Rebuild.
|
||||
|
||||
Auch nach diesem nativen Statuslauf kann Unraid einzelne Images weiterhin als
|
||||
Update markieren, obwohl Container-Image-ID und lokale Tag-Image-ID identisch
|
||||
sind. Das kommt insbesondere bei Registry-/Manifest- und Multiarch-Digest-
|
||||
Vergleichen vor. In diesem Konfliktfall ist das Ergebnis von
|
||||
`unraid_docker_update_verified_batch` nach dem Pull maßgeblich: identische
|
||||
unveränderliche Image-IDs bedeuten `already-current`; ein weiterer Rebuild nur
|
||||
zum Entfernen der GUI-Anzeige ist weder nötig noch erwünscht. Die GUI-Meldung
|
||||
ist dann ausdrücklich als Fehlanzeige zu melden.
|
||||
|
||||
Ein wiederholter Lauf muss bei einem aktuellen Image folgendes melden:
|
||||
|
||||
```text
|
||||
|
||||
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
@@ -3,4 +3,4 @@ Description=Legacy native Qwen Large 192K profile (Docker is the production path
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
Reference in New Issue
Block a user