Enable benchmarked two-slot medium profile
This commit is contained in:
+7
-1
@@ -40,7 +40,7 @@ UNCENSORED_UBATCH_SIZE=128
|
||||
EXPERIMENTAL_CONTEXT=76800
|
||||
FAST_GPU_DEVICES=0,1
|
||||
MEDIUM_GPU_DEVICES=0,1
|
||||
MEDIUM_TENSOR_SPLIT=90,10
|
||||
MEDIUM_TENSOR_SPLIT=85,15
|
||||
LARGE_GPU_DEVICES=0,1
|
||||
LARGE_TENSOR_SPLIT=86,14
|
||||
ULTRA_GPU_DEVICES=0,1
|
||||
@@ -51,3 +51,9 @@ UNCENSORED_MTP_MAX=2
|
||||
EXPERIMENTAL_GPU_DEVICES=0
|
||||
LLAMA_THREADS=6
|
||||
LLAMA_THREADS_BATCH=6
|
||||
FAST_PARALLEL_SLOTS=1
|
||||
MEDIUM_PARALLEL_SLOTS=2
|
||||
LARGE_PARALLEL_SLOTS=1
|
||||
ULTRA_PARALLEL_SLOTS=1
|
||||
UNCENSORED_PARALLEL_SLOTS=1
|
||||
EXPERIMENTAL_PARALLEL_SLOTS=1
|
||||
|
||||
+13
-7
@@ -109,7 +109,8 @@ services:
|
||||
- --ubatch-size
|
||||
- "${FAST_UBATCH_SIZE:-32}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${FAST_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
@@ -185,7 +186,8 @@ services:
|
||||
- --ubatch-size
|
||||
- "${MEDIUM_UBATCH_SIZE:-128}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${MEDIUM_PARALLEL_SLOTS:-2}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
@@ -221,7 +223,7 @@ services:
|
||||
- --split-mode
|
||||
- layer
|
||||
- --tensor-split
|
||||
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
||||
- "${MEDIUM_TENSOR_SPLIT:-85,15}"
|
||||
- --spec-type
|
||||
- draft-mtp
|
||||
- --spec-draft-n-max
|
||||
@@ -272,7 +274,8 @@ services:
|
||||
- --ubatch-size
|
||||
- "${LARGE_UBATCH_SIZE:-128}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${LARGE_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
@@ -349,7 +352,8 @@ services:
|
||||
- --ubatch-size
|
||||
- "${ULTRA_UBATCH_SIZE:-128}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${ULTRA_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
@@ -430,7 +434,8 @@ services:
|
||||
- --ubatch-size
|
||||
- "${UNCENSORED_UBATCH_SIZE:-128}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${UNCENSORED_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
@@ -498,7 +503,8 @@ services:
|
||||
- --cache-ram
|
||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||
- --parallel
|
||||
- "1"
|
||||
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
|
||||
- --kv-unified
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
|
||||
@@ -82,7 +82,7 @@ FAST_UBATCH_SIZE=32
|
||||
MEDIUM_CONTEXT=160000
|
||||
MEDIUM_BATCH_SIZE=2048
|
||||
MEDIUM_UBATCH_SIZE=128
|
||||
MEDIUM_TENSOR_SPLIT=90,10
|
||||
MEDIUM_TENSOR_SPLIT=85,15
|
||||
LARGE_CONTEXT=192000
|
||||
LARGE_BATCH_SIZE=2048
|
||||
LARGE_UBATCH_SIZE=128
|
||||
@@ -99,6 +99,12 @@ UNCENSORED_MTP_MAX=2
|
||||
EXPERIMENTAL_CONTEXT=76800
|
||||
LLAMA_THREADS=6
|
||||
LLAMA_THREADS_BATCH=6
|
||||
FAST_PARALLEL_SLOTS=1
|
||||
MEDIUM_PARALLEL_SLOTS=2
|
||||
LARGE_PARALLEL_SLOTS=1
|
||||
ULTRA_PARALLEL_SLOTS=1
|
||||
UNCENSORED_PARALLEL_SLOTS=1
|
||||
EXPERIMENTAL_PARALLEL_SLOTS=1
|
||||
PIPER_TTS_VERSION=1.6.0
|
||||
PIPER_VOICE=de_DE-thorsten-high
|
||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
"id": "fast",
|
||||
"alias": "qwen-fast",
|
||||
"context": 76800,
|
||||
"parallel_slots": 1,
|
||||
"model_env": "FAST_MODEL_FILE",
|
||||
"model_family": "Qwen3.8-27B IQ4 Mix",
|
||||
"gpu_split": "5080 only",
|
||||
@@ -18,9 +19,10 @@
|
||||
"id": "medium",
|
||||
"alias": "qwen-medium",
|
||||
"context": 160000,
|
||||
"parallel_slots": 2,
|
||||
"model_env": "MEDIUM_MODEL_FILE",
|
||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||
"gpu_split": "90:10",
|
||||
"gpu_split": "85:15",
|
||||
"vision": true,
|
||||
"mtp": 3,
|
||||
"description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben."
|
||||
@@ -29,6 +31,7 @@
|
||||
"id": "large",
|
||||
"alias": "qwen-large",
|
||||
"context": 192000,
|
||||
"parallel_slots": 1,
|
||||
"model_env": "LARGE_MODEL_FILE",
|
||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||
"gpu_split": "86:14",
|
||||
@@ -40,6 +43,7 @@
|
||||
"id": "ultra",
|
||||
"alias": "qwen-ultra",
|
||||
"context": 262144,
|
||||
"parallel_slots": 1,
|
||||
"model_env": "ULTRA_MODEL_FILE",
|
||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||
"gpu_split": "80:20",
|
||||
@@ -51,6 +55,7 @@
|
||||
"id": "uncensored",
|
||||
"alias": "qwen-uncensored",
|
||||
"context": 80000,
|
||||
"parallel_slots": 1,
|
||||
"model_env": "UNCENSORED_MODEL_FILE",
|
||||
"model_family": "Qwen3.8-27B Abliterated Q4_K_M",
|
||||
"gpu_split": "90:10",
|
||||
|
||||
@@ -19,4 +19,17 @@ Full-context results with the selected settings:
|
||||
| ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile |
|
||||
| uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed |
|
||||
|
||||
The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. Moving Medium from a 90:10 to an 89:11 GPU split freed memory but reduced both prefill and output speed. The production setting therefore remains 2048 / 128 at 90:10.
|
||||
The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. The original one-slot benchmark selected 2048 / 128 at 90:10.
|
||||
|
||||
## Medium two-slot benchmark
|
||||
|
||||
Medium now uses two parallel slots with unified KV, so both chats dynamically share one total 160K-token pool. The model weights remain loaded only once. To fit the additional scheduler buffers, the production GPU split is 85:15 while batch / ubatch remains 2048 / 128.
|
||||
|
||||
Identical fresh 100,297-token prompt with a deterministic 256-token completion:
|
||||
|
||||
| Mode | Slots | GPU split | Prefill | Output | Total time |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| Previous | 1 | 90:10 | 1,072.12 tok/s | 48.51 tok/s | 98.81 s |
|
||||
| Selected | 2 | 85:15 | 993.20 tok/s | 47.65 tok/s | 106.33 s |
|
||||
|
||||
Single-request cost: **7.4% lower prefill**, **1.8% lower output**, and **7.6% longer total time** for this near-full prompt. Two simultaneous fresh 10K prompts completed in 20 seconds; per-slot output measured 43.32 and 27.95 tok/s. Two very large simultaneous prefills can temporarily throttle an already-generating slot, so the total 160K pool should not be treated as two independent 160K contexts.
|
||||
|
||||
@@ -4,13 +4,13 @@ Diese Datei wird aus `config/profile-matrix.json` erzeugt. Änderungen gehören
|
||||
|
||||
Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
|
||||
|
||||
| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |
|
||||
|---|---|---:|---|---|---|---:|
|
||||
| fast | `qwen-fast` | 76,800 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
|
||||
| medium | `qwen-medium` | 160,000 | Qwen3.8-27B IQ4 XS Pure | 90:10 | ja | 3 |
|
||||
| large | `qwen-large` | 192,000 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
|
||||
| ultra | `qwen-ultra` | 262,144 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
|
||||
| uncensored | `qwen-uncensored` | 80,000 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
|
||||
| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |
|
||||
|---|---|---:|---:|---|---|---|---:|
|
||||
| fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
|
||||
| medium | `qwen-medium` | 160,000 | 2 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 |
|
||||
| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
|
||||
| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
|
||||
| uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
|
||||
|
||||
## Zweck
|
||||
|
||||
|
||||
@@ -59,12 +59,12 @@ def render_docs(data: dict) -> str:
|
||||
"",
|
||||
f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**",
|
||||
"",
|
||||
"| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |",
|
||||
"|---|---|---:|---|---|---|---:|",
|
||||
"| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |",
|
||||
"|---|---|---:|---:|---|---|---|---:|",
|
||||
]
|
||||
for item in data["profiles"]:
|
||||
lines.append(
|
||||
f"| {item['id']} | `{item['alias']}` | {item['context']:,} | "
|
||||
f"| {item['id']} | `{item['alias']}` | {item['context']:,} | {item.get('parallel_slots', 1)} | "
|
||||
f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | "
|
||||
f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |"
|
||||
)
|
||||
@@ -92,6 +92,9 @@ def verify_compose(data: dict, compose: pathlib.Path) -> None:
|
||||
env_prefix = item["id"].upper()
|
||||
if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block:
|
||||
raise SystemExit(f"Compose context drift for {item['id']}")
|
||||
slots = int(item.get("parallel_slots", 1))
|
||||
if f"${{{env_prefix}_PARALLEL_SLOTS:-{slots}}}" not in block:
|
||||
raise SystemExit(f"Compose parallel-slot drift for {item['id']}")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
|
||||
Reference in New Issue
Block a user