diff --git a/.env.example b/.env.example index cb8ba9e..424275d 100644 --- a/.env.example +++ b/.env.example @@ -21,7 +21,7 @@ MEDIUM_CONTEXT=160000 LARGE_CONTEXT=192000 ULTRA_CONTEXT=262144 EXPERIMENTAL_CONTEXT=76800 -FAST_GPU_DEVICES=0 +FAST_GPU_DEVICES=0,1 MEDIUM_GPU_DEVICES=0,1 MEDIUM_TENSOR_SPLIT=90,10 LARGE_GPU_DEVICES=0,1 diff --git a/README.md b/README.md index a7612f2..037f443 100644 --- a/README.md +++ b/README.md @@ -102,3 +102,8 @@ router/ OpenAI-kompatibler Profile Router Die Profilwerte wurden auf RTX 5080 und RTX 3060 vermessen und bilden die verbindliche Standardmatrix. Neue Varianten ersetzen sie erst nach demselben Vergleichstest und einer dokumentierten Entscheidung. + +Der integrierte Vision-Projektor der Profile Fast, Medium und Large läuft +gezielt auf der RTX 3060. Das hält den knappen VRAM der RTX 5080 für Modell und +Kontext frei und beschleunigte den dokumentierten synthetischen Vision-Test +gegenüber CPU-Vision um etwa den Faktor 8,5 bei der Gesamtzeit. diff --git a/compose.yaml b/compose.yaml index 94c0d73..396e201 100644 --- a/compose.yaml +++ b/compose.yaml @@ -33,14 +33,18 @@ services: labels: com.mike-ai.llama-profile: fast environment: - NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0} + NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility + # CUDA0 remains the exclusive text-model device. The projector is kept + # on the secondary card so vision does not consume the 5080 context + # budget. + MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - - --no-mmproj-offload + - --mmproj-offload - --alias - qwen-fast - --ctx-size @@ -102,12 +106,15 @@ services: environment: NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility + # Keep the language model split unchanged while placing the complete + # multimodal projector on the secondary RTX 3060. + MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - - --no-mmproj-offload + - --mmproj-offload - --alias - qwen-medium - --ctx-size @@ -173,12 +180,13 @@ services: environment: NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1} NVIDIA_DRIVER_CAPABILITIES: compute,utility + MTMD_BACKEND_DEVICE: CUDA1 command: - --model - "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}" - --mmproj - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - - --no-mmproj-offload + - --mmproj-offload - --alias - qwen-large - --ctx-size diff --git a/dev/benchmark_vision.py b/dev/benchmark_vision.py new file mode 100644 index 0000000..8467966 --- /dev/null +++ b/dev/benchmark_vision.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Run a private, deterministic llama.cpp vision latency smoke test.""" + +from __future__ import annotations + +import argparse +import base64 +import json +import struct +import time +import urllib.request +import zlib + + +def chunk(kind: bytes, payload: bytes) -> bytes: + return ( + struct.pack(">I", len(payload)) + + kind + + payload + + struct.pack(">I", zlib.crc32(kind + payload) & 0xFFFFFFFF) + ) + + +def synthetic_png(width: int = 1024, height: int = 768) -> bytes: + rows = [] + for y in range(height): + row = bytearray([0]) + for x in range(width): + row.extend(((x * 255) // width, (y * 255) // height, ((x // 64 + y // 64) % 2) * 210)) + rows.append(bytes(row)) + payload = zlib.compress(b"".join(rows), level=6) + return ( + b"\x89PNG\r\n\x1a\n" + + chunk(b"IHDR", struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0)) + + chunk(b"IDAT", payload) + + chunk(b"IEND", b"") + ) + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--url", required=True) + parser.add_argument("--model", default="qwen-medium") + args = parser.parse_args() + + image = base64.b64encode(synthetic_png()).decode("ascii") + body = { + "model": args.model, + "messages": [ + { + "role": "user", + "content": [ + { + "type": "image_url", + "image_url": {"url": f"data:image/png;base64,{image}"}, + }, + { + "type": "text", + "text": "Beschreibe dieses synthetische Testbild knapp auf Deutsch.", + }, + ], + } + ], + "temperature": 0, + "max_tokens": 80, + "cache_prompt": False, + } + request = urllib.request.Request( + args.url.rstrip("/") + "/v1/chat/completions", + data=json.dumps(body).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + started = time.perf_counter() + with urllib.request.urlopen(request, timeout=600) as response: + result = json.load(response) + elapsed = time.perf_counter() - started + print( + json.dumps( + { + "wall_seconds": round(elapsed, 3), + "usage": result.get("usage"), + "timings": result.get("timings"), + "answer": result["choices"][0]["message"].get("content", "")[:240], + }, + ensure_ascii=False, + indent=2, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 148461e..a4343d8 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -68,6 +68,14 @@ halten. Diese Matrix wurde auf RTX 5080 und RTX 3060 gemessen und ist bis zu einer bewussten Neubewertung der verbindliche Produktionsstandard. +Bei Fast, Medium und Large bleibt das Sprachmodell wie in der Tabelle +verteilt, während `MTMD_BACKEND_DEVICE=CUDA1` den vollständigen +Multimodal-Projektor auf die RTX 3060 legt. Ein reproduzierbarer Test mit einem +synthetischen Bild (1024×768, 835 Eingabetoken) senkte die Bild-/Promptzeit im +Medium-Profil von 23,17 auf 1,68 Sekunden und die gesamte Anfrage von 24,96 +auf 2,94 Sekunden. Der Projektor belegte dabei rund 1,1 GiB zusätzlichen VRAM +auf der 3060. Ultra bleibt für maximalen Kontext bewusst text-only. + ## Netzwerk - Docker-Netze liegen ausschließlich unter `172.30.0.0/16`. diff --git a/install.sh b/install.sh index bf7729c..258c3fa 100755 --- a/install.sh +++ b/install.sh @@ -218,7 +218,7 @@ MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000} LARGE_CONTEXT=${LARGE_CONTEXT:-192000} ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144} EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800} -FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0} +FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10} LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}