From 977f8f5c76cba9d5e5cf45f6da79440e98a24269 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sat, 22 Aug 2026 11:01:13 +0200 Subject: [PATCH] Validate Pure quant for 256K Ultra profile --- .env.example | 2 +- README.md | 3 +- compose.yaml | 2 +- config/install.env.example | 6 +-- dev/run_qwen38_pure_256k.py | 39 +++++++++++++++++++ docs/CURRENT_REFERENCE.md | 1 + docs/OPERATIONS.md | 4 +- docs/qwen38-pure-256k-benchmark-2026-08-22.md | 32 +++++++++++++++ platform/models/manifest.example.yaml | 7 ++-- 9 files changed, 85 insertions(+), 11 deletions(-) create mode 100644 dev/run_qwen38_pure_256k.py create mode 100644 docs/qwen38-pure-256k-benchmark-2026-08-22.md diff --git a/.env.example b/.env.example index b0b35fd..2232080 100644 --- a/.env.example +++ b/.env.example @@ -12,7 +12,7 @@ AI_DNS=192.168.1.1 FAST_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf MEDIUM_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf -ULTRA_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf +ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf EXPERIMENTAL_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf diff --git a/README.md b/README.md index 8325eed..eabe479 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,8 @@ WireGuard-Isolation. - vier schaltbare Profilcontainer plus ein isolierter Experimentalcontainer; davon ist immer exakt ein Inferenzcontainer aktiv - `/fast`, `/medium`, `/long` und `/ultra` über den Profile Router -- `/ultra`: getestetes text-only 256K-Profil (IQ4-MIX, beide GPUs, 80:20); etwa 59 Token/s im Referenzlauf und erfolgreicher 220K-Prompt-Fülltest +- `/ultra`: getestetes text-only 256K-Profil (IQ4_XS Pure, beide GPUs, + 80:20); etwa 68 Token/s und erfolgreicher 220K-Prompt-Fülltest - Open WebUI als einzige normale Oberfläche - SearXNG/Web-MCP ohne externen API-Schlüssel - zentrale MCP-Werkzeugebene: getrennte Container für Web, HA, ARR, Unraid diff --git a/compose.yaml b/compose.yaml index 755a2eb..3d1a712 100644 --- a/compose.yaml +++ b/compose.yaml @@ -222,7 +222,7 @@ services: - --spec-draft-type-v - q4_0 - # Text-only long-context profile. This exact IQ4-MIX / 256K / 80:20 + # Text-only long-context profile. This exact IQ4_XS-pure / 256K / 80:20 # combination completed the 220K fill test on RTX 5080 + RTX 3060. # Deliberately no vision projector: Ultra prioritizes maximum usable context. llama-ultra: diff --git a/config/install.env.example b/config/install.env.example index b198214..4b428d5 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -37,9 +37,9 @@ MEDIUM_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf LONG_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf LONG_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199 -ULTRA_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf -ULTRA_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf -ULTRA_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199 +ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf +ULTRA_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf +ULTRA_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675 EXPERIMENTAL_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf EXPERIMENTAL_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199 diff --git a/dev/run_qwen38_pure_256k.py b/dev/run_qwen38_pure_256k.py new file mode 100644 index 0000000..aad4bfa --- /dev/null +++ b/dev/run_qwen38_pure_256k.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""Focused 256K validation for the Qwen3.8 IQ4_XS-pure artifact. + +Reuses the synthetic dual-GPU runner. It reads no chats or user data, stops +production inference for the isolated test and restores the Fast profile in +the runner's finally block. +""" + +from __future__ import annotations + +import importlib.util +import pathlib +import sys + + +BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py") +SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE) +if SPEC is None or SPEC.loader is None: + raise RuntimeError(f"Could not load {BASE}") + +runner = importlib.util.module_from_spec(SPEC) +sys.modules[SPEC.name] = runner +SPEC.loader.exec_module(runner) + +runner.OUT = pathlib.Path("/data/benchmarks/qwen38-pure-256k-20260822") +runner.CASES = [ + runner.Case( + "pure-262k-80-20-mtp2", + "iq4-pure", + 262144, + (80, 20), + mtp=True, + mtp_max=2, + quality=False, + long_fill=0.84, + ), +] + +raise SystemExit(runner.main()) diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index b69d513..ce0d786 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -58,6 +58,7 @@ Zielplattform. | Fast | `qwen-fast` | 76.800 | IQ4-MIX, MTP2, vollständig GPU, CPU-mmproj | | Medium | `qwen-medium` | 94.208 | IQ4_XS Pure, ohne MTP, CPU-mmproj | | Long | `qwen-long` | 131.072 | IQ4-MIX, MTP2, FFN 0–11 auf CPU, CPU-mmproj | +| Ultra | `qwen-ultra` | 262.144 | IQ4_XS Pure, MTP2, beide GPUs 80:20, text-only; 68,2 Tok/s und 220K-Fülltest bestanden | ## Router diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index 447f8e1..a2d0a7d 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -74,11 +74,11 @@ Community Store nachgeladen. | Fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit, integrierte Vision | | Medium | `qwen-medium` | 94.208 | mehr Kontext, reine IQ4_XS-Variante | | Long | `qwen-long` | 131.072 | lange Hermes-/MCP-Sitzungen | -| Ultra | `qwen-ultra` | 262.144 | maximaler Textkontext, IQ4-MIX auf RTX 5080 + RTX 3060 (80:20), ohne Vision-Projektor | +| Ultra | `qwen-ultra` | 262.144 | maximaler Textkontext, IQ4_XS Pure auf RTX 5080 + RTX 3060 (80:20), ohne Vision-Projektor | Manuell wird mit `llama-profile fast|medium|long|ultra` gewechselt. Über HTTP stehen `POST /fast`, `/medium`, `/long` und `/ultra` zur Verfügung. Ultra erreichte im -Referenzlauf etwa 59 Token/s; ein Prompt-Fülltest mit rund 220.000 Tokens war +Referenzlauf etwa 68 Token/s; ein Prompt-Fülltest mit rund 220.000 Tokens war erfolgreich. Für eine spätere Version ist `large` als Alias für `long` vorgesehen; bestehende Namen bleiben kompatibel. diff --git a/docs/qwen38-pure-256k-benchmark-2026-08-22.md b/docs/qwen38-pure-256k-benchmark-2026-08-22.md new file mode 100644 index 0000000..ca4a448 --- /dev/null +++ b/docs/qwen38-pure-256k-benchmark-2026-08-22.md @@ -0,0 +1,32 @@ +# Qwen3.8 Pure – 256K-Validierung vom 22. August 2026 + +## Aufbau + +- Modell: `jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF` +- Datei: `qwen3.8-27b-IQ4_XS-pure.gguf` +- SHA-256: `ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675` +- Kontext: 262.144 Tokens +- GPUs: RTX 5080 + RTX 3060, Layer-Split 80:20 +- KV-Cache: Q4_0 für K und V +- MTP: aktiviert, maximal zwei Draft-Tokens +- Vision-Projektor: aus + +Der Test verwendete ausschließlich synthetische Daten. Es wurden keine Chats, +MCP-Antworten oder Nutzerdaten gelesen. + +## Ergebnis + +| Messung | IQ4_XS Pure | IQ4-MIX Referenz | +|---|---:|---:| +| Laden erfolgreich | ja | ja | +| Kurzer Ausgabetest | 68,19 Tok/s | 59,15 Tok/s | +| 220.190-Token-Prefill | 368,64 Tok/s | 357,86 Tok/s | +| Ausgabe nach 220K Prompt | 26,76 Tok/s | 26,32 Tok/s | +| Dauer des 220K-Tests | 599,91 s | 617,95 s | +| Sentinel wiedergefunden | ja | ja | +| OOM/Absturz | nein | nein | + +Pure war im kurzen Ausgabetest rund 15,3 Prozent schneller. Beim fast vollen +Kontext war die Ausgabegeschwindigkeit beider Varianten nahezu gleich. Da Pure +den vollständigen Langtest bestanden hat, ist es das ausgewählte Ultra-Profil. + diff --git a/platform/models/manifest.example.yaml b/platform/models/manifest.example.yaml index d3396d1..b18446b 100644 --- a/platform/models/manifest.example.yaml +++ b/platform/models/manifest.example.yaml @@ -8,11 +8,12 @@ models: sha256: "54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36" size_bytes: 14111614400 qwen_medium: - role: primary-text-medium - source: "REPLACE_WITH_MODEL_REPOSITORY" + role: primary-text-medium-and-ultra + source: "jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF" file: qwen3.8-27b-IQ4_XS-pure.gguf target: /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf - sha256: "REPLACE_AFTER_VERIFICATION" + sha256: "ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675" + size_bytes: 14534384640 qwen_vision_projector: role: integrated-vision-projector-for-all-qwen-profiles source: "REPLACE_WITH_MODEL_REPOSITORY"