From 6378b50086e1856894d59cebf4dc091fb080d0ff Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Thu, 24 Sep 2026 05:09:10 +0200 Subject: [PATCH] Enable CPU vision projector for Ultra profile --- ATHENA.md | 2 ++ compose.yaml | 8 ++++--- config/profile-matrix.json | 4 ++-- docs/CONTAINER_INVENTORY.md | 2 +- docs/LIVE_STATE.md | 5 ++++ docs/STANDARD_PROFILE_MATRIX.md | 4 ++-- docs/ULTRA_CPU_VISION_20260924.md | 23 +++++++++++++++++++ .../references/architecture-and-modes.md | 2 +- platform/llama/README.md | 6 ++--- platform/profiles/profile-ultra.conf | 2 +- router/router_profiles.json | 2 +- 11 files changed, 46 insertions(+), 14 deletions(-) create mode 100644 docs/ULTRA_CPU_VISION_20260924.md diff --git a/ATHENA.md b/ATHENA.md index 8625097..deec0e2 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -4,6 +4,8 @@ Stand: **20. September 2026**, auf Athena geprüft. Die fünf Textprofil-Images tragen **llama.cpp b29c606** (0.4.1). Der [geprüfte Live-Stand](docs/LIVE_STATE.md) beschreibt Profile, GPUs und die Abweichung zwischen dem bereitgestellten Stack und dem Git-Checkout. +Seit dem 24. September verarbeitet auch Ultra Bilder; sein Vision-Projektor +läuft auf der CPU. [Änderung und Test](docs/ULTRA_CPU_VISION_20260924.md). Der [Updatebericht vom 15. September](docs/UPDATE_AUDIT_20260915.md) enthält die aktuellen Build- und Testbelege. Der ältere [b10930-Bericht](docs/LLAMA_B10930_UPDATE_20260912.md) dokumentiert einen diff --git a/compose.yaml b/compose.yaml index 168739d..ef20659 100644 --- a/compose.yaml +++ b/compose.yaml @@ -394,9 +394,8 @@ services: - --spec-draft-type-v - f16 - # Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20 - # combination completed the 220K fill test on RTX 5080 + RTX 3060. - # Deliberately no vision projector: Ultra prioritizes maximum usable context. + # Maximum-context profile. Keep the vision projector on CPU so images work + # without consuming the tightly budgeted GPU memory of the 256K context. llama-ultra: <<: *llama-common container_name: mike-ai-llama-ultra @@ -408,6 +407,9 @@ services: command: - --model - "/models/${ULTRA_MODEL_FILE:?ULTRA_MODEL_FILE is required}" + - --mmproj + - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" + - --no-mmproj-offload - --alias - qwen-ultra - --ctx-size diff --git a/config/profile-matrix.json b/config/profile-matrix.json index 9d0af93..ad91b91 100644 --- a/config/profile-matrix.json +++ b/config/profile-matrix.json @@ -47,9 +47,9 @@ "model_env": "ULTRA_MODEL_FILE", "model_family": "Qwen3.8-27B IQ4 XS Pure", "gpu_split": "80:20", - "vision": false, + "vision": true, "mtp": 2, - "description": "Maximaler Textkontext; bewusst ohne Vision-Projektor." + "description": "Maximaler Kontext mit Vision-Projektor auf der CPU." }, { "id": "uncensored", diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index 252f8e5..6bfa9e7 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -22,7 +22,7 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-llama-fast` | Qwen3.8-27B `IQ4-MIX`, Qwen-MMProj BF16 | Schnelles Q4-Text-/Vision-Profil mit 76.800 Token Kontext. | | `mike-ai-llama-large` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Q4-Text-/Vision-Profil mit 192.000 Token Kontext und Verteilung auf beide GPUs. | | `mike-ai-llama-medium` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Standard-Q4-Text-/Vision-Profil mit gemeinsamem 160.000-Token-KV-Pool, aktuell zwei Slots und Verteilung auf beide GPUs. | -| `mike-ai-llama-ultra` | Qwen3.8-27B `IQ4_XS-pure`, ohne Vision-Projektor | Maximales Langkontextprofil mit 262.144 Token Kontext und Verteilung auf beide GPUs. | +| `mike-ai-llama-ultra` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 auf CPU | Text-/Vision-Profil mit 262.144 Token Kontext und Verteilung des Textmodells auf beide GPUs. | | `mike-ai-llama-uncensored` | Qwen3.8-27B Abliterated `Q4_K_M`, eigener MMProj F16 | Spezialprofil mit 80.000 Token Kontext und gelockerten Modellgrenzen. | | `mike-ai-ltx2-studio` | LTX-Video-Backend | GPU-Worker für lokale Videogenerierung; beim Abgleich gestoppt. | | `mike-ai-mcp-athena-operator` | kein Modell | Stellt Hermes begrenzte Werkzeuge zum Prüfen, Ändern, Testen, Sichern und Versionieren von Athena bereit. | diff --git a/docs/LIVE_STATE.md b/docs/LIVE_STATE.md index bc06156..a5bcb96 100644 --- a/docs/LIVE_STATE.md +++ b/docs/LIVE_STATE.md @@ -1,5 +1,10 @@ # Geprüfter Live-Stand auf Athena +Nachtrag vom 24. September 2026: Ultra verarbeitet nun Bilder mit einem +CPU-seitigen BF16-Vision-Projektor. Der Stand und der Funktionstest sind in +[Ultra-Vision mit CPU-Projektor](ULTRA_CPU_VISION_20260924.md) dokumentiert. +Die folgende Tabelle bildet weiterhin den historischen Stand vom 21. September ab. + Stand: **21. September 2026**. Quelle: lesender SSH-Abgleich von Docker, Compose-Dateien, Git-Inhalten, Images, Health-Endpunkten und Backup-Timern. Containerzustände sind Momentaufnahmen; der Controller darf Profile danach diff --git a/docs/STANDARD_PROFILE_MATRIX.md b/docs/STANDARD_PROFILE_MATRIX.md index 27f9630..4ccb8d4 100644 --- a/docs/STANDARD_PROFILE_MATRIX.md +++ b/docs/STANDARD_PROFILE_MATRIX.md @@ -9,7 +9,7 @@ Standardprofil: **medium** · globales Ausgabelimit: **8192 Token** | fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 | | medium | `qwen-medium` | 160,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 2 | | large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 2 | -| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 | +| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | ja | 2 | | uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 | ## Zweck @@ -17,5 +17,5 @@ Standardprofil: **medium** · globales Ausgabelimit: **8192 Token** - **fast**: Schnelles Profil für kurze Chats und zügige Werkzeugaufgaben. - **medium**: Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben. - **large**: Großes Profil für umfangreiche Dokumente und lange technische Arbeiten. -- **ultra**: Maximaler Textkontext; bewusst ohne Vision-Projektor. +- **ultra**: Maximaler Kontext mit Vision-Projektor auf der CPU. - **uncensored**: Weniger restriktives Spezialprofil; Werkzeugrechte bleiben unverändert. diff --git a/docs/ULTRA_CPU_VISION_20260924.md b/docs/ULTRA_CPU_VISION_20260924.md new file mode 100644 index 0000000..75ed554 --- /dev/null +++ b/docs/ULTRA_CPU_VISION_20260924.md @@ -0,0 +1,23 @@ +# Ultra-Vision mit CPU-Projektor + +Stand: 24. September 2026. Ultra verwendet weiterhin Qwen3.8-27B IQ4_XS Pure +mit 262.144 Token Kontext. Der BF16-Vision-Projektor wird mit `--mmproj` +geladen und durch `--no-mmproj-offload` auf der CPU gehalten. So kann Ultra +Bilder verarbeiten, ohne den knapp bemessenen GPU-Speicher des 256K-Profils +zusätzlich mit dem Projektor zu belegen. Router und Profilmatrix melden Ultra +als vision-fähig. + +Der frühere text-only-Modus war die Ursache dafür, dass der Router Bildanfragen +an `qwen-ultra` abwies. Ein Test über den Router nach dem Deployment lieferte +HTTP 200 für ein einzelnes synthetisches Farbbild (`Red`) und für fünf +synthetische Bilder (`5`). Der Ultra-Start lud den multimodalen Projektor. +Gemessen wurden 11,47 Sekunden bis zur Betriebsbereitschaft und 1,66 bzw. +0,9 Sekunden für die beiden kleinen Testanfragen. Große Screenshots und lange +Kontexte wurden damit nicht vermessen; deren Laufzeit kann deutlich höher +sein. Nach dem Test wurde Medium wieder aktiviert, Ultra ist gestoppt. + +Vor der Änderung wurden die drei produktiven Dateien auf Athena unter +`/var/tmp/ultra-vision-cpu-20260924/` gesichert. Ein Rückbau muss Compose, +Router-Profilregister und Profilmatrix gemeinsam auf den vorigen Stand setzen +und den Router neu laden. Das Verzeichnis liegt nur temporär auf dem Host; +die dauerhafte Versionierung erfolgt im Git-Repository. diff --git a/platform/hermes/skills/athena-operator/references/architecture-and-modes.md b/platform/hermes/skills/athena-operator/references/architecture-and-modes.md index 8095193..c1b98d6 100644 --- a/platform/hermes/skills/athena-operator/references/architecture-and-modes.md +++ b/platform/hermes/skills/athena-operator/references/architecture-and-modes.md @@ -32,7 +32,7 @@ gateway, UI, CPU-STT, backup and operator containers may remain active. - Fast: Qwen3.8-27B IQ4-MIX, 76,800 tokens. - Medium: Qwen3.8-27B IQ4_XS-pure, 160,000 tokens, vision. - Large: the same Q4 model, 192,000 tokens, vision. -- Ultra: the same Q4 model, 262,144 tokens, no vision projector. +- Ultra: the same Q4 model, 262,144 tokens, vision projector on CPU. - Uncensored: Abliterated Q4_K_M, 80,000 tokens, vision. Medium, Large, Ultra and Beta distribute their runtime across both GPUs. Do not diff --git a/platform/llama/README.md b/platform/llama/README.md index 811acd3..66c9076 100644 --- a/platform/llama/README.md +++ b/platform/llama/README.md @@ -16,10 +16,10 @@ nach Standardbenchmark, Tool-Calling-Test und Kontexttest übernommen. - Fast: Qwen3.8-27B IQ4-MIX mit MTP2 - Medium und Large: Qwen3.8-27B IQ4_XS Pure mit MTP3 -- Ultra: Qwen3.8-27B IQ4_XS Pure mit MTP2 und maximalem Textkontext +- Ultra: Qwen3.8-27B IQ4_XS Pure mit MTP2, maximalem Kontext und Vision-Projektor auf CPU - Uncensored: Blackfrost Qwen3.8-27B Abliterated Q4_K_M mit MTP2 -- Fast, Medium, Large und Uncensored: integrierte Vision; der jeweilige - Projektor liegt vollständig auf der RTX 3060 +- Alle fünf Profile: integrierte Vision. Bei Fast, Medium, Large und + Uncensored liegt der Projektor auf der RTX 3060; bei Ultra auf der CPU. Die produktive Runtime ist seit dem 12. September 2026 auf llama.cpp **Build 10930**, Commit `56381e407c0ccfb3a6f71e668a27a901001d22ce`, diff --git a/platform/profiles/profile-ultra.conf b/platform/profiles/profile-ultra.conf index 7c429cf..5c5fc94 100644 --- a/platform/profiles/profile-ultra.conf +++ b/platform/profiles/profile-ultra.conf @@ -3,4 +3,4 @@ Description=Legacy native Qwen Ultra 256K profile (Docker is the production path [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --load-mode none --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/router/router_profiles.json b/router/router_profiles.json index b6bf0d4..f90ce41 100644 --- a/router/router_profiles.json +++ b/router/router_profiles.json @@ -18,7 +18,7 @@ "ultra": { "context": 262144, "model_alias": "qwen-ultra", - "vision": false + "vision": true }, "uncensored": { "context": 80000,