Enable benchmarked two-slot medium profile
This commit is contained in:
+7
-1
@@ -40,7 +40,7 @@ UNCENSORED_UBATCH_SIZE=128
|
|||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
FAST_GPU_DEVICES=0,1
|
FAST_GPU_DEVICES=0,1
|
||||||
MEDIUM_GPU_DEVICES=0,1
|
MEDIUM_GPU_DEVICES=0,1
|
||||||
MEDIUM_TENSOR_SPLIT=90,10
|
MEDIUM_TENSOR_SPLIT=85,15
|
||||||
LARGE_GPU_DEVICES=0,1
|
LARGE_GPU_DEVICES=0,1
|
||||||
LARGE_TENSOR_SPLIT=86,14
|
LARGE_TENSOR_SPLIT=86,14
|
||||||
ULTRA_GPU_DEVICES=0,1
|
ULTRA_GPU_DEVICES=0,1
|
||||||
@@ -51,3 +51,9 @@ UNCENSORED_MTP_MAX=2
|
|||||||
EXPERIMENTAL_GPU_DEVICES=0
|
EXPERIMENTAL_GPU_DEVICES=0
|
||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
|
FAST_PARALLEL_SLOTS=1
|
||||||
|
MEDIUM_PARALLEL_SLOTS=2
|
||||||
|
LARGE_PARALLEL_SLOTS=1
|
||||||
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
|
UNCENSORED_PARALLEL_SLOTS=1
|
||||||
|
EXPERIMENTAL_PARALLEL_SLOTS=1
|
||||||
|
|||||||
+13
-7
@@ -109,7 +109,8 @@ services:
|
|||||||
- --ubatch-size
|
- --ubatch-size
|
||||||
- "${FAST_UBATCH_SIZE:-32}"
|
- "${FAST_UBATCH_SIZE:-32}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${FAST_PARALLEL_SLOTS:-1}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
@@ -185,7 +186,8 @@ services:
|
|||||||
- --ubatch-size
|
- --ubatch-size
|
||||||
- "${MEDIUM_UBATCH_SIZE:-128}"
|
- "${MEDIUM_UBATCH_SIZE:-128}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${MEDIUM_PARALLEL_SLOTS:-2}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
@@ -221,7 +223,7 @@ services:
|
|||||||
- --split-mode
|
- --split-mode
|
||||||
- layer
|
- layer
|
||||||
- --tensor-split
|
- --tensor-split
|
||||||
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
- "${MEDIUM_TENSOR_SPLIT:-85,15}"
|
||||||
- --spec-type
|
- --spec-type
|
||||||
- draft-mtp
|
- draft-mtp
|
||||||
- --spec-draft-n-max
|
- --spec-draft-n-max
|
||||||
@@ -272,7 +274,8 @@ services:
|
|||||||
- --ubatch-size
|
- --ubatch-size
|
||||||
- "${LARGE_UBATCH_SIZE:-128}"
|
- "${LARGE_UBATCH_SIZE:-128}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${LARGE_PARALLEL_SLOTS:-1}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
@@ -349,7 +352,8 @@ services:
|
|||||||
- --ubatch-size
|
- --ubatch-size
|
||||||
- "${ULTRA_UBATCH_SIZE:-128}"
|
- "${ULTRA_UBATCH_SIZE:-128}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${ULTRA_PARALLEL_SLOTS:-1}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
@@ -430,7 +434,8 @@ services:
|
|||||||
- --ubatch-size
|
- --ubatch-size
|
||||||
- "${UNCENSORED_UBATCH_SIZE:-128}"
|
- "${UNCENSORED_UBATCH_SIZE:-128}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${UNCENSORED_PARALLEL_SLOTS:-1}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
@@ -498,7 +503,8 @@ services:
|
|||||||
- --cache-ram
|
- --cache-ram
|
||||||
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
- "${LLAMA_CACHE_RAM_MIB:-8192}"
|
||||||
- --parallel
|
- --parallel
|
||||||
- "1"
|
- "${EXPERIMENTAL_PARALLEL_SLOTS:-1}"
|
||||||
|
- --kv-unified
|
||||||
- --jinja
|
- --jinja
|
||||||
- --reasoning
|
- --reasoning
|
||||||
- auto
|
- auto
|
||||||
|
|||||||
@@ -82,7 +82,7 @@ FAST_UBATCH_SIZE=32
|
|||||||
MEDIUM_CONTEXT=160000
|
MEDIUM_CONTEXT=160000
|
||||||
MEDIUM_BATCH_SIZE=2048
|
MEDIUM_BATCH_SIZE=2048
|
||||||
MEDIUM_UBATCH_SIZE=128
|
MEDIUM_UBATCH_SIZE=128
|
||||||
MEDIUM_TENSOR_SPLIT=90,10
|
MEDIUM_TENSOR_SPLIT=85,15
|
||||||
LARGE_CONTEXT=192000
|
LARGE_CONTEXT=192000
|
||||||
LARGE_BATCH_SIZE=2048
|
LARGE_BATCH_SIZE=2048
|
||||||
LARGE_UBATCH_SIZE=128
|
LARGE_UBATCH_SIZE=128
|
||||||
@@ -99,6 +99,12 @@ UNCENSORED_MTP_MAX=2
|
|||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
|
FAST_PARALLEL_SLOTS=1
|
||||||
|
MEDIUM_PARALLEL_SLOTS=2
|
||||||
|
LARGE_PARALLEL_SLOTS=1
|
||||||
|
ULTRA_PARALLEL_SLOTS=1
|
||||||
|
UNCENSORED_PARALLEL_SLOTS=1
|
||||||
|
EXPERIMENTAL_PARALLEL_SLOTS=1
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
"id": "fast",
|
"id": "fast",
|
||||||
"alias": "qwen-fast",
|
"alias": "qwen-fast",
|
||||||
"context": 76800,
|
"context": 76800,
|
||||||
|
"parallel_slots": 1,
|
||||||
"model_env": "FAST_MODEL_FILE",
|
"model_env": "FAST_MODEL_FILE",
|
||||||
"model_family": "Qwen3.8-27B IQ4 Mix",
|
"model_family": "Qwen3.8-27B IQ4 Mix",
|
||||||
"gpu_split": "5080 only",
|
"gpu_split": "5080 only",
|
||||||
@@ -18,9 +19,10 @@
|
|||||||
"id": "medium",
|
"id": "medium",
|
||||||
"alias": "qwen-medium",
|
"alias": "qwen-medium",
|
||||||
"context": 160000,
|
"context": 160000,
|
||||||
|
"parallel_slots": 2,
|
||||||
"model_env": "MEDIUM_MODEL_FILE",
|
"model_env": "MEDIUM_MODEL_FILE",
|
||||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||||
"gpu_split": "90:10",
|
"gpu_split": "85:15",
|
||||||
"vision": true,
|
"vision": true,
|
||||||
"mtp": 3,
|
"mtp": 3,
|
||||||
"description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben."
|
"description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben."
|
||||||
@@ -29,6 +31,7 @@
|
|||||||
"id": "large",
|
"id": "large",
|
||||||
"alias": "qwen-large",
|
"alias": "qwen-large",
|
||||||
"context": 192000,
|
"context": 192000,
|
||||||
|
"parallel_slots": 1,
|
||||||
"model_env": "LARGE_MODEL_FILE",
|
"model_env": "LARGE_MODEL_FILE",
|
||||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||||
"gpu_split": "86:14",
|
"gpu_split": "86:14",
|
||||||
@@ -40,6 +43,7 @@
|
|||||||
"id": "ultra",
|
"id": "ultra",
|
||||||
"alias": "qwen-ultra",
|
"alias": "qwen-ultra",
|
||||||
"context": 262144,
|
"context": 262144,
|
||||||
|
"parallel_slots": 1,
|
||||||
"model_env": "ULTRA_MODEL_FILE",
|
"model_env": "ULTRA_MODEL_FILE",
|
||||||
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
"model_family": "Qwen3.8-27B IQ4 XS Pure",
|
||||||
"gpu_split": "80:20",
|
"gpu_split": "80:20",
|
||||||
@@ -51,6 +55,7 @@
|
|||||||
"id": "uncensored",
|
"id": "uncensored",
|
||||||
"alias": "qwen-uncensored",
|
"alias": "qwen-uncensored",
|
||||||
"context": 80000,
|
"context": 80000,
|
||||||
|
"parallel_slots": 1,
|
||||||
"model_env": "UNCENSORED_MODEL_FILE",
|
"model_env": "UNCENSORED_MODEL_FILE",
|
||||||
"model_family": "Qwen3.8-27B Abliterated Q4_K_M",
|
"model_family": "Qwen3.8-27B Abliterated Q4_K_M",
|
||||||
"gpu_split": "90:10",
|
"gpu_split": "90:10",
|
||||||
|
|||||||
@@ -19,4 +19,17 @@ Full-context results with the selected settings:
|
|||||||
| ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile |
|
| ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile |
|
||||||
| uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed |
|
| uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed |
|
||||||
|
|
||||||
The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. Moving Medium from a 90:10 to an 89:11 GPU split freed memory but reduced both prefill and output speed. The production setting therefore remains 2048 / 128 at 90:10.
|
The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. The original one-slot benchmark selected 2048 / 128 at 90:10.
|
||||||
|
|
||||||
|
## Medium two-slot benchmark
|
||||||
|
|
||||||
|
Medium now uses two parallel slots with unified KV, so both chats dynamically share one total 160K-token pool. The model weights remain loaded only once. To fit the additional scheduler buffers, the production GPU split is 85:15 while batch / ubatch remains 2048 / 128.
|
||||||
|
|
||||||
|
Identical fresh 100,297-token prompt with a deterministic 256-token completion:
|
||||||
|
|
||||||
|
| Mode | Slots | GPU split | Prefill | Output | Total time |
|
||||||
|
|---|---:|---:|---:|---:|---:|
|
||||||
|
| Previous | 1 | 90:10 | 1,072.12 tok/s | 48.51 tok/s | 98.81 s |
|
||||||
|
| Selected | 2 | 85:15 | 993.20 tok/s | 47.65 tok/s | 106.33 s |
|
||||||
|
|
||||||
|
Single-request cost: **7.4% lower prefill**, **1.8% lower output**, and **7.6% longer total time** for this near-full prompt. Two simultaneous fresh 10K prompts completed in 20 seconds; per-slot output measured 43.32 and 27.95 tok/s. Two very large simultaneous prefills can temporarily throttle an already-generating slot, so the total 160K pool should not be treated as two independent 160K contexts.
|
||||||
|
|||||||
@@ -4,13 +4,13 @@ Diese Datei wird aus `config/profile-matrix.json` erzeugt. Änderungen gehören
|
|||||||
|
|
||||||
Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
|
Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
|
||||||
|
|
||||||
| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |
|
| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |
|
||||||
|---|---|---:|---|---|---|---:|
|
|---|---|---:|---:|---|---|---|---:|
|
||||||
| fast | `qwen-fast` | 76,800 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
|
| fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
|
||||||
| medium | `qwen-medium` | 160,000 | Qwen3.8-27B IQ4 XS Pure | 90:10 | ja | 3 |
|
| medium | `qwen-medium` | 160,000 | 2 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 |
|
||||||
| large | `qwen-large` | 192,000 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
|
| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
|
||||||
| ultra | `qwen-ultra` | 262,144 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
|
| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
|
||||||
| uncensored | `qwen-uncensored` | 80,000 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
|
| uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
|
||||||
|
|
||||||
## Zweck
|
## Zweck
|
||||||
|
|
||||||
|
|||||||
@@ -59,12 +59,12 @@ def render_docs(data: dict) -> str:
|
|||||||
"",
|
"",
|
||||||
f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**",
|
f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**",
|
||||||
"",
|
"",
|
||||||
"| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |",
|
"| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |",
|
||||||
"|---|---|---:|---|---|---|---:|",
|
"|---|---|---:|---:|---|---|---|---:|",
|
||||||
]
|
]
|
||||||
for item in data["profiles"]:
|
for item in data["profiles"]:
|
||||||
lines.append(
|
lines.append(
|
||||||
f"| {item['id']} | `{item['alias']}` | {item['context']:,} | "
|
f"| {item['id']} | `{item['alias']}` | {item['context']:,} | {item.get('parallel_slots', 1)} | "
|
||||||
f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | "
|
f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | "
|
||||||
f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |"
|
f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |"
|
||||||
)
|
)
|
||||||
@@ -92,6 +92,9 @@ def verify_compose(data: dict, compose: pathlib.Path) -> None:
|
|||||||
env_prefix = item["id"].upper()
|
env_prefix = item["id"].upper()
|
||||||
if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block:
|
if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block:
|
||||||
raise SystemExit(f"Compose context drift for {item['id']}")
|
raise SystemExit(f"Compose context drift for {item['id']}")
|
||||||
|
slots = int(item.get("parallel_slots", 1))
|
||||||
|
if f"${{{env_prefix}_PARALLEL_SLOTS:-{slots}}}" not in block:
|
||||||
|
raise SystemExit(f"Compose parallel-slot drift for {item['id']}")
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
def main() -> None:
|
||||||
|
|||||||
Reference in New Issue
Block a user