From 014583e7f696e304bfe7d394d0f936f6d25235d9 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sun, 30 Aug 2026 20:54:22 +0200 Subject: [PATCH] Enable benchmarked two-slot medium profile --- .env.example | 8 +++++++- compose.yaml | 20 +++++++++++++------- config/install.env.example | 8 +++++++- config/profile-matrix.json | 7 ++++++- docs/PREFILL_BATCH_BENCHMARK_20260830.md | 15 ++++++++++++++- docs/STANDARD_PROFILE_MATRIX.md | 14 +++++++------- platform/scripts/sync-profile-matrix.py | 9 ++++++--- 7 files changed, 60 insertions(+), 21 deletions(-) diff --git a/.env.example b/.env.example index 885cb93..f8418bc 100644 --- a/.env.example +++ b/.env.example @@ -40,7 +40,7 @@ UNCENSORED_UBATCH_SIZE=128 EXPERIMENTAL_CONTEXT=76800 FAST_GPU_DEVICES=0,1 MEDIUM_GPU_DEVICES=0,1 -MEDIUM_TENSOR_SPLIT=90,10 +MEDIUM_TENSOR_SPLIT=85,15 LARGE_GPU_DEVICES=0,1 LARGE_TENSOR_SPLIT=86,14 ULTRA_GPU_DEVICES=0,1 @@ -51,3 +51,9 @@ UNCENSORED_MTP_MAX=2 EXPERIMENTAL_GPU_DEVICES=0 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 +FAST_PARALLEL_SLOTS=1 +MEDIUM_PARALLEL_SLOTS=2 +LARGE_PARALLEL_SLOTS=1 +ULTRA_PARALLEL_SLOTS=1 +UNCENSORED_PARALLEL_SLOTS=1 +EXPERIMENTAL_PARALLEL_SLOTS=1 diff --git a/compose.yaml b/compose.yaml index 1758ae4..55113c0 100644 --- a/compose.yaml +++ b/compose.yaml @@ -109,7 +109,8 @@ services: - --ubatch-size - "${FAST_UBATCH_SIZE:-32}" - --parallel - - "1" + - "${FAST_PARALLEL_SLOTS:-1}" + - --kv-unified - --jinja - --reasoning - auto @@ -185,7 +186,8 @@ services: - --ubatch-size - "${MEDIUM_UBATCH_SIZE:-128}" - --parallel - - "1" + - "${MEDIUM_PARALLEL_SLOTS:-2}" + - --kv-unified - --jinja - --reasoning - auto @@ -221,7 +223,7 @@ services: - --split-mode - layer - --tensor-split - - "${MEDIUM_TENSOR_SPLIT:-90,10}" + - "${MEDIUM_TENSOR_SPLIT:-85,15}" - --spec-type - draft-mtp - --spec-draft-n-max @@ -272,7 +274,8 @@ services: - --ubatch-size - "${LARGE_UBATCH_SIZE:-128}" - --parallel - - "1" + - "${LARGE_PARALLEL_SLOTS:-1}" + - --kv-unified - --jinja - --reasoning - auto @@ -349,7 +352,8 @@ services: - --ubatch-size - "${ULTRA_UBATCH_SIZE:-128}" - --parallel - - "1" + - "${ULTRA_PARALLEL_SLOTS:-1}" + - --kv-unified - --jinja - --reasoning - auto @@ -430,7 +434,8 @@ services: - --ubatch-size - "${UNCENSORED_UBATCH_SIZE:-128}" - --parallel - - "1" + - "${UNCENSORED_PARALLEL_SLOTS:-1}" + - --kv-unified - --jinja - --reasoning - auto @@ -498,7 +503,8 @@ services: - --cache-ram - "${LLAMA_CACHE_RAM_MIB:-8192}" - --parallel - - "1" + - "${EXPERIMENTAL_PARALLEL_SLOTS:-1}" + - --kv-unified - --jinja - --reasoning - auto diff --git a/config/install.env.example b/config/install.env.example index 7a8936e..e12c4e6 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -82,7 +82,7 @@ FAST_UBATCH_SIZE=32 MEDIUM_CONTEXT=160000 MEDIUM_BATCH_SIZE=2048 MEDIUM_UBATCH_SIZE=128 -MEDIUM_TENSOR_SPLIT=90,10 +MEDIUM_TENSOR_SPLIT=85,15 LARGE_CONTEXT=192000 LARGE_BATCH_SIZE=2048 LARGE_UBATCH_SIZE=128 @@ -99,6 +99,12 @@ UNCENSORED_MTP_MAX=2 EXPERIMENTAL_CONTEXT=76800 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 +FAST_PARALLEL_SLOTS=1 +MEDIUM_PARALLEL_SLOTS=2 +LARGE_PARALLEL_SLOTS=1 +ULTRA_PARALLEL_SLOTS=1 +UNCENSORED_PARALLEL_SLOTS=1 +EXPERIMENTAL_PARALLEL_SLOTS=1 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 diff --git a/config/profile-matrix.json b/config/profile-matrix.json index 3d83a5b..98e89bb 100644 --- a/config/profile-matrix.json +++ b/config/profile-matrix.json @@ -7,6 +7,7 @@ "id": "fast", "alias": "qwen-fast", "context": 76800, + "parallel_slots": 1, "model_env": "FAST_MODEL_FILE", "model_family": "Qwen3.8-27B IQ4 Mix", "gpu_split": "5080 only", @@ -18,9 +19,10 @@ "id": "medium", "alias": "qwen-medium", "context": 160000, + "parallel_slots": 2, "model_env": "MEDIUM_MODEL_FILE", "model_family": "Qwen3.8-27B IQ4 XS Pure", - "gpu_split": "90:10", + "gpu_split": "85:15", "vision": true, "mtp": 3, "description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben." @@ -29,6 +31,7 @@ "id": "large", "alias": "qwen-large", "context": 192000, + "parallel_slots": 1, "model_env": "LARGE_MODEL_FILE", "model_family": "Qwen3.8-27B IQ4 XS Pure", "gpu_split": "86:14", @@ -40,6 +43,7 @@ "id": "ultra", "alias": "qwen-ultra", "context": 262144, + "parallel_slots": 1, "model_env": "ULTRA_MODEL_FILE", "model_family": "Qwen3.8-27B IQ4 XS Pure", "gpu_split": "80:20", @@ -51,6 +55,7 @@ "id": "uncensored", "alias": "qwen-uncensored", "context": 80000, + "parallel_slots": 1, "model_env": "UNCENSORED_MODEL_FILE", "model_family": "Qwen3.8-27B Abliterated Q4_K_M", "gpu_split": "90:10", diff --git a/docs/PREFILL_BATCH_BENCHMARK_20260830.md b/docs/PREFILL_BATCH_BENCHMARK_20260830.md index caa487a..cf83b7e 100644 --- a/docs/PREFILL_BATCH_BENCHMARK_20260830.md +++ b/docs/PREFILL_BATCH_BENCHMARK_20260830.md @@ -19,4 +19,17 @@ Full-context results with the selected settings: | ultra | 257,998 | 698.2 tok/s | 21.1 tok/s | no | not configured for this profile | | uncensored | 77,998 | 1,050.5 tok/s | 40.0 tok/s | no | passed | -The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. Moving Medium from a 90:10 to an 89:11 GPU split freed memory but reduced both prefill and output speed. The production setting therefore remains 2048 / 128 at 90:10. +The larger logical batches 3072 and 4096 did not improve Medium at ubatch 128. The original one-slot benchmark selected 2048 / 128 at 90:10. + +## Medium two-slot benchmark + +Medium now uses two parallel slots with unified KV, so both chats dynamically share one total 160K-token pool. The model weights remain loaded only once. To fit the additional scheduler buffers, the production GPU split is 85:15 while batch / ubatch remains 2048 / 128. + +Identical fresh 100,297-token prompt with a deterministic 256-token completion: + +| Mode | Slots | GPU split | Prefill | Output | Total time | +|---|---:|---:|---:|---:|---:| +| Previous | 1 | 90:10 | 1,072.12 tok/s | 48.51 tok/s | 98.81 s | +| Selected | 2 | 85:15 | 993.20 tok/s | 47.65 tok/s | 106.33 s | + +Single-request cost: **7.4% lower prefill**, **1.8% lower output**, and **7.6% longer total time** for this near-full prompt. Two simultaneous fresh 10K prompts completed in 20 seconds; per-slot output measured 43.32 and 27.95 tok/s. Two very large simultaneous prefills can temporarily throttle an already-generating slot, so the total 160K pool should not be treated as two independent 160K contexts. diff --git a/docs/STANDARD_PROFILE_MATRIX.md b/docs/STANDARD_PROFILE_MATRIX.md index e2932fb..5ffdfac 100644 --- a/docs/STANDARD_PROFILE_MATRIX.md +++ b/docs/STANDARD_PROFILE_MATRIX.md @@ -4,13 +4,13 @@ Diese Datei wird aus `config/profile-matrix.json` erzeugt. Änderungen gehören Standardprofil: **medium** · globales Ausgabelimit: **8192 Token** -| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP | -|---|---|---:|---|---|---|---:| -| fast | `qwen-fast` | 76,800 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 | -| medium | `qwen-medium` | 160,000 | Qwen3.8-27B IQ4 XS Pure | 90:10 | ja | 3 | -| large | `qwen-large` | 192,000 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 | -| ultra | `qwen-ultra` | 262,144 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 | -| uncensored | `qwen-uncensored` | 80,000 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 | +| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP | +|---|---|---:|---:|---|---|---|---:| +| fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 | +| medium | `qwen-medium` | 160,000 | 2 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 | +| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 | +| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 | +| uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 | ## Zweck diff --git a/platform/scripts/sync-profile-matrix.py b/platform/scripts/sync-profile-matrix.py index c93e31e..ba7ac74 100755 --- a/platform/scripts/sync-profile-matrix.py +++ b/platform/scripts/sync-profile-matrix.py @@ -59,12 +59,12 @@ def render_docs(data: dict) -> str: "", f"Standardprofil: **{data['default_profile']}** · globales Ausgabelimit: **{data['max_output_tokens']} Token**", "", - "| Profil | API-Alias | Kontext | Modell | GPU-Verteilung | Vision | MTP |", - "|---|---|---:|---|---|---|---:|", + "| Profil | API-Alias | Gesamtkontext | Slots | Modell | GPU-Verteilung | Vision | MTP |", + "|---|---|---:|---:|---|---|---|---:|", ] for item in data["profiles"]: lines.append( - f"| {item['id']} | `{item['alias']}` | {item['context']:,} | " + f"| {item['id']} | `{item['alias']}` | {item['context']:,} | {item.get('parallel_slots', 1)} | " f"{item.get('model_family', '')} | {item.get('gpu_split', '')} | " f"{'ja' if item.get('vision') else 'nein'} | {item.get('mtp', '')} |" ) @@ -92,6 +92,9 @@ def verify_compose(data: dict, compose: pathlib.Path) -> None: env_prefix = item["id"].upper() if f"${{{env_prefix}_CONTEXT:-{item['context']}}}" not in block: raise SystemExit(f"Compose context drift for {item['id']}") + slots = int(item.get("parallel_slots", 1)) + if f"${{{env_prefix}_PARALLEL_SLOTS:-{slots}}}" not in block: + raise SystemExit(f"Compose parallel-slot drift for {item['id']}") def main() -> None: