Enable benchmarked two-slot medium profile

This commit is contained in:
Mikei386
2026-08-30 20:54:22 +02:00
parent db3719af7a
commit a41d6dc28c
7 changed files with 60 additions and 21 deletions
+7 -1
View File
@@ -40,7 +40,7 @@ UNCENSORED_UBATCH_SIZE=128
EXPERIMENTAL_CONTEXT=76800 EXPERIMENTAL_CONTEXT=76800
FAST_GPU_DEVICES=0,1 FAST_GPU_DEVICES=0,1
MEDIUM_GPU_DEVICES=0,1 MEDIUM_GPU_DEVICES=0,1
MEDIUM_TENSOR_SPLIT=90,10 MEDIUM_TENSOR_SPLIT=85,15
LARGE_GPU_DEVICES=0,1 LARGE_GPU_DEVICES=0,1
LARGE_TENSOR_SPLIT=86,14 LARGE_TENSOR_SPLIT=86,14
ULTRA_GPU_DEVICES=0,1 ULTRA_GPU_DEVICES=0,1
@@ -51,3 +51,9 @@ UNCENSORED_MTP_MAX=2
EXPERIMENTAL_GPU_DEVICES=0 EXPERIMENTAL_GPU_DEVICES=0
LLAMA_THREADS=6 LLAMA_THREADS=6
LLAMA_THREADS_BATCH=6 LLAMA_THREADS_BATCH=6
FAST_PARALLEL_SLOTS=1
MEDIUM_PARALLEL_SLOTS=2
LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1
EXPERIMENTAL_PARALLEL_SLOTS=1
+13 -7
View File
@@ -109,7 +109,8 @@ services:
- --ubatch-size - --ubatch-size
- "${FAST_UBATCH_SIZE:-32}" - "${FAST_UBATCH_SIZE:-32}"
- --parallel - --parallel
- "1" - "${FAST_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
@@ -185,7 +186,8 @@ services:
- --ubatch-size - --ubatch-size
- "${MEDIUM_UBATCH_SIZE:-128}" - "${MEDIUM_UBATCH_SIZE:-128}"
- --parallel - --parallel
- "1" - "${MEDIUM_PARALLEL_SLOTS:-2}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
@@ -221,7 +223,7 @@ services:
- --split-mode - --split-mode
- layer - layer
- --tensor-split - --tensor-split
- "${MEDIUM_TENSOR_SPLIT:-90,10}" - "${MEDIUM_TENSOR_SPLIT:-85,15}"
- --spec-type - --spec-type
- draft-mtp - draft-mtp
- --spec-draft-n-max - --spec-draft-n-max
@@ -272,7 +274,8 @@ services:
- --ubatch-size - --ubatch-size
- "${LARGE_UBATCH_SIZE:-128}" - "${LARGE_UBATCH_SIZE:-128}"
- --parallel - --parallel
- "1" - "${LARGE_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
@@ -349,7 +352,8 @@ services:
- --ubatch-size - --ubatch-size
- "${ULTRA_UBATCH_SIZE:-128}" - "${ULTRA_UBATCH_SIZE:-128}"
- --parallel - --parallel
- "1" - "${ULTRA_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
@@ -430,7 +434,8 @@ services:
- --ubatch-size - --ubatch-size
- "${UNCENSORED_UBATCH_SIZE:-128}" - "${UNCENSORED_UBATCH_SIZE:-128}"
- --parallel - --parallel
- "1" - "${UNCENSORED_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
@@ -498,7 +503,8 @@ services:
- --cache-ram - --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-8192}" - "${LLAMA_CACHE_RAM_MIB:-8192}"
- --parallel - --parallel
- "1" - "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
+7 -1
View File
@@ -82,7 +82,7 @@ FAST_UBATCH_SIZE=32
MEDIUM_CONTEXT=160000 MEDIUM_CONTEXT=160000
MEDIUM_BATCH_SIZE=2048 MEDIUM_BATCH_SIZE=2048
MEDIUM_UBATCH_SIZE=128 MEDIUM_UBATCH_SIZE=128
MEDIUM_TENSOR_SPLIT=90,10 MEDIUM_TENSOR_SPLIT=85,15
LARGE_CONTEXT=192000 LARGE_CONTEXT=192000
LARGE_BATCH_SIZE=2048 LARGE_BATCH_SIZE=2048
LARGE_UBATCH_SIZE=128 LARGE_UBATCH_SIZE=128
@@ -99,6 +99,12 @@ UNCENSORED_MTP_MAX=2
EXPERIMENTAL_CONTEXT=76800 EXPERIMENTAL_CONTEXT=76800
LLAMA_THREADS=6 LLAMA_THREADS=6
LLAMA_THREADS_BATCH=6 LLAMA_THREADS_BATCH=6
FAST_PARALLEL_SLOTS=1
MEDIUM_PARALLEL_SLOTS=2
LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1
EXPERIMENTAL_PARALLEL_SLOTS=1
PIPER_TTS_VERSION=1.6.0 PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
+6 -1
View File
@@ -7,6 +7,7 @@
"id": "fast", "id": "fast",
"alias": "qwen-fast", "alias": "qwen-fast",
"context": 76800, "context": 76800,
"parallel_slots": 1,
"model_env": "FAST_MODEL_FILE", "model_env": "FAST_MODEL_FILE",
"model_family": "Qwen3.8-27B IQ4 Mix", "model_family": "Qwen3.8-27B IQ4 Mix",
"gpu_split": "5080 only", "gpu_split": "5080 only",
@@ -18,9 +19,10 @@
"id": "medium", "id": "medium",
"alias": "qwen-medium", "alias": "qwen-medium",
"context": 160000, "context": 160000,
"parallel_slots": 2,
"model_env": "MEDIUM_MODEL_FILE", "model_env": "MEDIUM_MODEL_FILE",
"model_family": "Qwen3.8-27B IQ4 XS Pure", "model_family": "Qwen3.8-27B IQ4 XS Pure",
"gpu_split": "90:10", "gpu_split": "85:15",
"vision": true, "vision": true,
"mtp": 3, "mtp": 3,
"description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben." "description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben."
@@ -29,6 +31,7 @@
"id": "large", "id": "large",
"alias": "qwen-large", "alias": "qwen-large",
"context": 192000, "context": 192000,
"parallel_slots": 1,
"model_env": "LARGE_MODEL_FILE", "model_env": "LARGE_MODEL_FILE",
"model_family": "Qwen3.8-27B IQ4 XS Pure", "model_family": "Qwen3.8-27B IQ4 XS Pure",
"gpu_split": "86:14", "gpu_split": "86:14",
@@ -40,6 +43,7 @@
"id": "ultra", "id": "ultra",
"alias": "qwen-ultra", "alias": "qwen-ultra",
"context": 262144, "context": 262144,
"parallel_slots": 1,
"model_env": "ULTRA_MODEL_FILE", "model_env": "ULTRA_MODEL_FILE",
"model_family": "Qwen3.8-27B IQ4 XS Pure", "model_family": "Qwen3.8-27B IQ4 XS Pure",
"gpu_split": "80:20", "gpu_split": "80:20",
@@ -51,6 +55,7 @@
"id": "uncensored", "id": "uncensored",
"alias": "qwen-uncensored", "alias": "qwen-uncensored",
"context": 80000, "context": 80000,
"parallel_slots": 1,
"model_env": "UNCENSORED_MODEL_FILE", "model_env": "UNCENSORED_MODEL_FILE",
"model_family": "Qwen3.8-27B Abliterated Q4_K_M", "model_family": "Qwen3.8-27B Abliterated Q4_K_M",
"gpu_split": "90:10", "gpu_split": "90:10",
+14 -1
View File
@@ -19,4 +19,17 @@ Full-context results with the selected settings:
| ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile | | ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile |
| uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed | | uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed |
The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. Moving Medium from a 90:10 to an 89:11 GPU split freed memory but reduced both prefill and output speed. The production setting therefore remains 2048 / 128 at 90:10. The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. The original one-slot benchmark selected 2048 / 128 at 90:10.
## Medium two-slot benchmark
Medium now uses two parallel slots with unified KV, so both chats dynamically share one total 160K-token pool. The model weights remain loaded only once. To fit the additional scheduler buffers, the production GPU split is 85:15 while batch / ubatch remains 2048 / 128.
Identical fresh 100,297-token prompt with a deterministic 256-token completion:
| Mode | Slots | GPU split | Prefill | Output | Total time |
|---|---:|---:|---:|---:|---:|
| Previous | 1 | 90:10 | 1,072.12 tok/s | 48.51 tok/s | 98.81 s |
| Selected | 2 | 85:15 | 993.20 tok/s | 47.65 tok/s | 106.33 s |
Single-request cost: **7.4% lower prefill**, **1.8% lower output**, and **7.6% longer total time** for this near-full prompt. Two simultaneous fresh 10K prompts completed in 20 seconds; per-slot output measured 43.32 and 27.95 tok/s. Two very large simultaneous prefills can temporarily throttle an already-generating slot, so the total 160K pool should not be treated as two independent 160K contexts.
+7 -7
View File
@@ -4,13 +4,13 @@ Diese Datei wird aus `config/profile-matrix.json` erzeugt. Änderungen gehören
Standardprofil: **medium** · globales Ausgabelimit: **8192 Token** Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP | | Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |
|---|---|---:|---|---|---|---:| |---|---|---:|---:|---|---|---|---:|
| fast | `qwen-fast` | 76,800 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 | | fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
| medium | `qwen-medium` | 160,000 | Qwen3.8-27B IQ4 XS Pure | 90:10 | ja | 3 | | medium | `qwen-medium` | 160,000 | 2 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 |
| large | `qwen-large` | 192,000 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 | | large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
| ultra | `qwen-ultra` | 262,144 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 | | ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
| uncensored | `qwen-uncensored` | 80,000 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 | | uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
## Zweck ## Zweck
+6 -3
View File
@@ -59,12 +59,12 @@ def render_docs(data: dict) -> str:
"", "",
f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**", f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**",
"", "",
"| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |", "| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |",
"|---|---|---:|---|---|---|---:|", "|---|---|---:|---:|---|---|---|---:|",
] ]
for item in data["profiles"]: for item in data["profiles"]:
lines.append( lines.append(
f"| {item['id']} | `{item['alias']}` | {item['context']:,} | " f"| {item['id']} | `{item['alias']}` | {item['context']:,} | {item.get('parallel_slots', 1)} | "
f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | " f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | "
f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |" f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |"
) )
@@ -92,6 +92,9 @@ def verify_compose(data: dict, compose: pathlib.Path) -> None:
env_prefix = item["id"].upper() env_prefix = item["id"].upper()
if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block: if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block:
raise SystemExit(f"Compose context drift for {item['id']}") raise SystemExit(f"Compose context drift for {item['id']}")
slots = int(item.get("parallel_slots", 1))
if f"${{{env_prefix}_PARALLEL_SLOTS:-{slots}}}" not in block:
raise SystemExit(f"Compose parallel-slot drift for {item['id']}")
def main() -> None: def main() -> None: