From 8a45a0d8515770d50fd19e4c15bef0072e1fef30 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Thu, 20 Aug 2026 14:11:25 +0200 Subject: [PATCH] Optimize Qwen fast profile for vision and 76K context --- ...wen38-fast-profile-benchmark-2026-08-20.md | 46 +++++++++++++++++++ platform/profiles/profile-fast.conf | 4 +- 2 files changed, 48 insertions(+), 2 deletions(-) create mode 100644 docs/qwen38-fast-profile-benchmark-2026-08-20.md diff --git a/docs/qwen38-fast-profile-benchmark-2026-08-20.md b/docs/qwen38-fast-profile-benchmark-2026-08-20.md new file mode 100644 index 0000000..9b898d1 --- /dev/null +++ b/docs/qwen38-fast-profile-benchmark-2026-08-20.md @@ -0,0 +1,46 @@ +# Qwen3.8-27B Fast-Profil-Test vom 20. August 2026 + +Hardware: RTX 5080 mit 16 GB VRAM. Modell: `Qwen3.8-27B-IQ4-MIX.gguf`. + +## Ergebnis + +Gewinner ist das Profil mit 76.800 Tokens Kontext, MTP2, Haupt-KV-Cache in Q4_0, +MTP-Draft-KV in F16 und einem nicht auf die GPU ausgelagerten BF16-Vision-Projektor. + +| Profil | Kontext | Kurztests TG | 30K belegt | 66K belegt | Vision | Stabilität | +| --- | ---: | --- | ---: | ---: | --- | --- | +| bisher: Draft-Q4 | 73.728 | 85,6 / 93,6 / 98,3 t/s | nicht gemessen | nicht gemessen | nein | stabil | +| Draft-F16 | 73.728 | 92,7 / 100,5 / 103,4 t/s | nicht gemessen | nicht gemessen | nein | stabil | +| Gewinner | 76.800 | 94,3 / 103,3 / 114,7 t/s | 91,5 t/s | 72,0 t/s | ja, CPU-Projektor | stabil | +| BeeLlama KVarN4/3 | 98.304 | 82,9 / 98,2 / 90,2 t/s | nicht erreicht | nicht erreicht | nein | CUDA-OOM beim großen Prefill | +| BeeLlama KVarN4/3 | 90.112 | Kurztest nicht wiederholt | nicht erreicht | nicht erreicht | nein | CUDA-OOM beim großen Prefill | + +Die realistischen drei Kurztests waren deutsche Analyse, Python-Code und strukturiertes JSON +mit jeweils bis zu 1.200 Ausgabetokens. Große Kontexttests verwendeten ausschließlich +synthetischen Fülltext. + +## Wichtige Erkenntnisse + +- Der F16-Draft-Cache benötigt bei diesem einlagigen MTP weniger VRAM als der quantisierte + Q4-Draft-Cache. Bei 73.728 Tokens stieg der freie VRAM von ungefähr 64 auf 168 MiB. +- 81.920 Tokens waren mit MTP nicht startfähig. 76.800 Tokens starteten und bestanden einen + Prefill mit ungefähr 66.000 synthetischen Tokens. +- Nahe dem vollen Kontext sinkt die Ausgabe trotz unverändertem Profil unter 80 t/s. Bei + ungefähr 30.000 belegten Tokens wurden noch 91,5 t/s erreicht, bei etwa 66.000 Tokens + 72,0 t/s. Das ist der zunehmende Attention-Aufwand, kein CPU-Offload. +- Der 931-MB-BF16-Vision-Projektor bleibt mittels `--no-mmproj-offload` im System-RAM. + Dadurch bleibt der VRAM-Verbrauch des Sprachmodells praktisch unverändert. Das synthetische + Testbild wurde korrekt erkannt; Bild-Prefill etwa 2,5 Sekunden, Ausgabe etwa 92–94 t/s. +- KVarN war in kurzen Tests vielversprechend, stürzte aber bei großen Prefills reproduzierbar + im CUDA-Flash-Attention-Kernel mit OOM ab und ist daher nicht produktionsgeeignet. + +## Aktives Fast-Profil + +Siehe `platform/profiles/profile-fast.conf`. Wichtige Parameter: + +- `--ctx-size 76800` +- `--cache-type-k q4_0 --cache-type-v q4_0` +- `--spec-draft-type-k f16 --spec-draft-type-v f16` +- `--spec-draft-n-max 2` +- `--mmproj .../mmproj-BF16.gguf --no-mmproj-offload` +- vollständiger GPU-Offload der Modellgewichte auf `CUDA0` diff --git a/platform/profiles/profile-fast.conf b/platform/profiles/profile-fast.conf index f30ac6f..48b29ac 100644 --- a/platform/profiles/profile-fast.conf +++ b/platform/profiles/profile-fast.conf @@ -1,6 +1,6 @@ [Unit] -Description=Local AI llama.cpp - Qwen Fast 72K MTP2 +Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-72k-mtp2 --ctx-size 73728 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0 +ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16