From 2415b2716066775765d4c4e49e3e8ec7ad1c67a0 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sat, 12 Sep 2026 20:41:05 +0200 Subject: [PATCH] Update llama.cpp to b10930 and migrate model loading option --- compose.yaml | 18 ++++--- docs/CURRENT_RUNTIME_NOTES.md | 25 ++++----- docs/LLAMA_B10930_UPDATE_20260912.md | 76 +++++++++++++++++++++++++++ platform/llama/LLAMA_CPP_COMMIT | 2 +- platform/llama/README.md | 21 ++++++-- platform/profiles/profile-fast.conf | 2 +- platform/profiles/profile-large.conf | 2 +- platform/profiles/profile-medium.conf | 2 +- platform/profiles/profile-ultra.conf | 2 +- 9 files changed, 122 insertions(+), 28 deletions(-) create mode 100644 docs/LLAMA_B10930_UPDATE_20260912.md diff --git a/compose.yaml b/compose.yaml index 5757bf4..3a99093 100644 --- a/compose.yaml +++ b/compose.yaml @@ -129,7 +129,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature - "0.2" @@ -206,7 +207,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature # Qwen3.8's official thinking-mode sampler. The former 0.2 setting was @@ -292,7 +294,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature - "1.0" @@ -371,7 +374,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature - "0.2" @@ -449,7 +453,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature - "0.2" @@ -531,7 +536,8 @@ services: - "off" - --n-gpu-layers - all - - --no-mmap + - --load-mode + - none - --no-ui - --temperature - "0.2" diff --git a/docs/CURRENT_RUNTIME_NOTES.md b/docs/CURRENT_RUNTIME_NOTES.md index 2bd1b2c..992506f 100644 --- a/docs/CURRENT_RUNTIME_NOTES.md +++ b/docs/CURRENT_RUNTIME_NOTES.md @@ -17,19 +17,20 @@ andere private Ziele werden dadurch nicht freigeschaltet. ## Produktive llama.cpp-Runtime -Alle Textprofile verwenden llama.cpp Build 10781, -Commit `c7bda030e7faee594dbe7550185e857351ad405d`. Der Stand enthält die ab -Build 10751 verfügbare Korrektur für eine zwischenzeitliche -MTP-/KV-Cache-Initialisierungsregression. Der vorherige produktive Stand war -Build 10718, Commit `41ef91f7c8046087cdfbb276b79bff311ecf1c6d`, und bleibt über das alte -lokale Image `mike-ai/llama.cpp:b10718-fallback` als unmittelbarer -Rückfallpunkt erhalten. +Aktualisiert am 12. September 2026: Alle fünf Textprofilcontainer verwenden +llama.cpp **Build 10930**, Commit +`56381e407c0ccfb3a6f71e668a27a901001d22ce`. -Build 10781 wurde nach dem Bau produktiv verifiziert: Alle fünf -Profilcontainer verwenden dasselbe neue Image, ausschließlich Medium läuft, -Qwen Medium ist mit 160.000 Tokens Kontext gesund, der MTP-Kontext wurde -erfolgreich initialisiert, der Vision-Projektor geladen und eine lokale -Textprobe korrekt beantwortet. +Der tatsächliche vorherige Live-Build war 10872 +(`b31b71f3a076bfc4278daad442203a9c51c6e676`), nicht der hier zuvor +angegebene Build 10781. Er bleibt unter +`mike-ai/llama.cpp:b10872-pre-b10930` als Rückfallimage erhalten. + +Das Update enthält den Fix für die Drafter-Position nach Bildeingaben +[#28715](https://github.com/ggml-org/llama.cpp/pull/28715). Modellgewichte, +Profilkontexte, GPU-Aufteilung, Vision, Slots und MTP-Parameter bleiben +unverändert. Details und Testgrenzen stehen im +[Updateprotokoll](LLAMA_B10930_UPDATE_20260912.md). ## Qwen Medium: Vision-Projektor wieder aktiviert diff --git a/docs/LLAMA_B10930_UPDATE_20260912.md b/docs/LLAMA_B10930_UPDATE_20260912.md new file mode 100644 index 0000000..45b05f8 --- /dev/null +++ b/docs/LLAMA_B10930_UPDATE_20260912.md @@ -0,0 +1,76 @@ +# llama.cpp b10930 auf Athena + +Stand: 12. September 2026 + +## Produktiver Stand + +- Build: **10930** +- Upstream-Commit: `56381e407c0ccfb3a6f71e668a27a901001d22ce` +- Image: `mike-ai/llama.cpp:local`, zusätzlich `mike-ai/llama.cpp:b10930` +- Image-ID: `sha256:c9d78a9375143877574f3d7c6e0dea94ecfebb264e8dd9f57ac4e03f1c96044e` +- CUDA im Container: 12.8.1; Zielarchitekturen: 86 und 120 +- Build aus dem vorhandenen Dockerfile mit zwei parallelen Compiler-Prozessen + +Alle fünf auf Athena vorhandenen Textprofilcontainer wurden aktualisiert: +Fast, Medium, Large, Ultra und Uncensored. Modellgewichte, GPU-Verteilung, +Kontextfenster, Slots, Vision- und MTP-Einstellungen wurden beibehalten. +Der zusätzliche Git-Matrixeintrag `beta1` ist auf Athena nicht installiert +und wurde daher nicht als Laufzeittest gewertet. + +Der tatsächliche vorherige Live-Build war **10872**, Commit +`b31b71f3a076bfc4278daad442203a9c51c6e676`. Die bisherige Dokumentation +mit Build 10781 war veraltet. + +## Enthaltener Fix und notwendige CLI-Migration + +[Upstream-PR #28715](https://github.com/ggml-org/llama.cpp/pull/28715), +integriert am 11. September, korrigiert die an den Drafter übergebene Position +nach Bildeingaben. Der Fix betrifft spekulative Decodierung einschließlich MTP. +Er ist kein Nachweis dafür, dass sämtliche MMProj-/Prompt-Cache-Probleme oder +Hermes-Slot-Verdrängungen behoben sind. + +Der neue Build entfernt die bisherige Option `--no-mmap`. Ihre gleichwertige +Ersatzform ist `--load-mode none`. Compose und die Referenzprofil-Dateien sind +entsprechend angepasst. Beim ersten Start wurde die alte Option abgewiesen; +der automatische Rückfall auf b10872 funktionierte. Nach der Migration lädt +b10930 Modell, MTP-Kontext und Vision-Projektor erfolgreich. + +## Verifikation + +- Alle fünf Profile über den normalen Router nacheinander aktiviert und mit + einer kurzen Rechenaufgabe geprüft: jeweils korrekte Antwort. +- Uncensored vor und nach dem Update: Rechnen, strukturierter Tool-Aufruf, + synthetisches Bild mit rotem Quadrat links und blauem Kreis rechts sowie + eine Folgefrage nach dem Bild bestanden; MTP-Zähler bestätigen Drafting. +- Medium nach dem Update: dieselben kurzen Text-, Tool- und Vision-Tests. +- Uncensored-Langkontext: **72.802 Eingabetokens**; Kennung vom Anfang nach + langem synthetischem Fülltext korrekt wiedergegeben, vor und nach dem Update. + +| Messung auf Uncensored | b10872 | b10930 | +|---|---:|---:| +| 300 Ausgabetokens, synthetische Zahlenfolge | 65,92 Token/s | 66,79 Token/s | +| Verarbeitung des langen Prompts | 978,29 Token/s | 976,05 Token/s | +| Gesamtdauer Langkontextanfrage | 74,104 s | 74,248 s | + +Dies sind einzelne Funktions- und Vergleichsläufe, kein statistisch +abgesicherter Leistungsbenchmark. Die Vorher-Messung lief während des +CPU-Builds; Cachezustand und Hintergrundlast können Messwerte beeinflussen. +Die Zahlen zeigen hier praktisch gleiches Verhalten, keinen belegten +allgemeinen Geschwindigkeitsgewinn. Vollständig gefüllte 160K-, 192K- und +262K-Kontexte sowie Langzeitstabilität wurden nicht geprüft. + +## Rückfall und Betriebszustand + +Das unveränderte vorherige Image bleibt erhalten als +`mike-ai/llama.cpp:b10872-pre-b10930`. +Originaldateien, Buildlog, Testskripte und Messergebnisse liegen auf Athena in +`/data/deploy-backups/20260912-llama-b10930/`. + +Ein Rückfall benötigt sowohl das alte Image als auch seine bisherige +Startoption `--no-mmap`; nur das Image umzuschalten reicht nicht. Die gesicherte +Compose-Datei dient als Referenz. Spätere Änderungen dürfen beim Rückfall +nicht durch blindes Überschreiben verloren gehen. + +Nach der Prüfung ist wieder genau das zuvor aktive Profil **Uncensored** +aktiv. Die übrigen Profile bleiben bedarfsgesteuert gestoppt. Host, Treiber, +SSH, LAN, Firewall und WireGuard wurden nicht verändert oder neu gestartet. diff --git a/platform/llama/LLAMA_CPP_COMMIT b/platform/llama/LLAMA_CPP_COMMIT index b802600..f8db295 100644 --- a/platform/llama/LLAMA_CPP_COMMIT +++ b/platform/llama/LLAMA_CPP_COMMIT @@ -1 +1 @@ -c7bda030e7faee594dbe7550185e857351ad405d +56381e407c0ccfb3a6f71e668a27a901001d22ce diff --git a/platform/llama/README.md b/platform/llama/README.md index 414d84a..4d4871d 100644 --- a/platform/llama/README.md +++ b/platform/llama/README.md @@ -21,11 +21,22 @@ nach Standardbenchmark, Tool-Calling-Test und Kontexttest übernommen. - Fast, Medium, Large und Uncensored: integrierte Vision; der jeweilige Projektor liegt vollständig auf der RTX 3060 -Die produktive Runtime ist auf llama.cpp Build 10781, Commit -`c7bda030e7faee594dbe7550185e857351ad405d`, festgeschrieben. Dieser Stand -enthält die ab Build 10751 verfügbare Korrektur für eine zwischenzeitliche -MTP-/KV-Cache-Initialisierungsregression. Der vorherige produktive Stand war -Build 10718, Commit `41ef91f7c8046087cdfbb276b79bff311ecf1c6d`. +Die produktive Runtime ist seit dem 12. September 2026 auf llama.cpp +**Build 10930**, Commit `56381e407c0ccfb3a6f71e668a27a901001d22ce`, +festgeschrieben. Enthalten ist der am 11. September integrierte Fix +[„speculation after an image“ (#28715)](https://github.com/ggml-org/llama.cpp/pull/28715). +Er korrigiert die Positionsübergabe an den Drafter nach Bildeingaben und ist +nicht mit einer generellen Behebung aller MMProj-/Prompt-Cache-Probleme +gleichzusetzen. + +Der unmittelbar vorher laufende Build war **10872**, Commit +`b31b71f3a076bfc4278daad442203a9c51c6e676`; die frühere Angabe 10781 +war im Live-System bereits überholt. Das unveränderte Rückfallimage heißt +`mike-ai/llama.cpp:b10872-pre-b10930`. + +Build, Tests und Wiederherstellung sind in +[docs/LLAMA_B10930_UPDATE_20260912.md](../../docs/LLAMA_B10930_UPDATE_20260912.md) +dokumentiert. Die Dateien selbst sind nicht Bestandteil des Repositories. Pfade und Hashes werden im lokalen Modellmanifest verwaltet. diff --git a/platform/profiles/profile-fast.conf b/platform/profiles/profile-fast.conf index 6efde84..211386e 100644 --- a/platform/profiles/profile-fast.conf +++ b/platform/profiles/profile-fast.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-large.conf b/platform/profiles/profile-large.conf index b19ea43..d5efc5a 100644 --- a/platform/profiles/profile-large.conf +++ b/platform/profiles/profile-large.conf @@ -3,4 +3,4 @@ Description=Legacy native Qwen Large 192K profile (Docker is the production path [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-medium.conf b/platform/profiles/profile-medium.conf index bb2ed29..b115402 100644 --- a/platform/profiles/profile-medium.conf +++ b/platform/profiles/profile-medium.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-ultra.conf b/platform/profiles/profile-ultra.conf index d1e9279..7c429cf 100644 --- a/platform/profiles/profile-ultra.conf +++ b/platform/profiles/profile-ultra.conf @@ -3,4 +3,4 @@ Description=Legacy native Qwen Ultra 256K profile (Docker is the production path [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16