Offload vision projector to RTX 3060
This commit is contained in:
+1
-1
@@ -21,7 +21,7 @@ MEDIUM_CONTEXT=160000
|
|||||||
LARGE_CONTEXT=192000
|
LARGE_CONTEXT=192000
|
||||||
ULTRA_CONTEXT=262144
|
ULTRA_CONTEXT=262144
|
||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
FAST_GPU_DEVICES=0
|
FAST_GPU_DEVICES=0,1
|
||||||
MEDIUM_GPU_DEVICES=0,1
|
MEDIUM_GPU_DEVICES=0,1
|
||||||
MEDIUM_TENSOR_SPLIT=90,10
|
MEDIUM_TENSOR_SPLIT=90,10
|
||||||
LARGE_GPU_DEVICES=0,1
|
LARGE_GPU_DEVICES=0,1
|
||||||
|
|||||||
@@ -102,3 +102,8 @@ router/ OpenAI-kompatibler Profile Router
|
|||||||
Die Profilwerte wurden auf RTX 5080 und RTX 3060 vermessen und bilden die
|
Die Profilwerte wurden auf RTX 5080 und RTX 3060 vermessen und bilden die
|
||||||
verbindliche Standardmatrix. Neue Varianten ersetzen sie erst nach demselben
|
verbindliche Standardmatrix. Neue Varianten ersetzen sie erst nach demselben
|
||||||
Vergleichstest und einer dokumentierten Entscheidung.
|
Vergleichstest und einer dokumentierten Entscheidung.
|
||||||
|
|
||||||
|
Der integrierte Vision-Projektor der Profile Fast, Medium und Large läuft
|
||||||
|
gezielt auf der RTX 3060. Das hält den knappen VRAM der RTX 5080 für Modell und
|
||||||
|
Kontext frei und beschleunigte den dokumentierten synthetischen Vision-Test
|
||||||
|
gegenüber CPU-Vision um etwa den Faktor 8,5 bei der Gesamtzeit.
|
||||||
|
|||||||
+12
-4
@@ -33,14 +33,18 @@ services:
|
|||||||
labels:
|
labels:
|
||||||
com.mike-ai.llama-profile: fast
|
com.mike-ai.llama-profile: fast
|
||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
|
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0,1}
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
# CUDA0 remains the exclusive text-model device. The projector is kept
|
||||||
|
# on the secondary card so vision does not consume the 5080 context
|
||||||
|
# budget.
|
||||||
|
MTMD_BACKEND_DEVICE: CUDA1
|
||||||
command:
|
command:
|
||||||
- --model
|
- --model
|
||||||
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
|
- "/models/${FAST_MODEL_FILE:?FAST_MODEL_FILE is required}"
|
||||||
- --mmproj
|
- --mmproj
|
||||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||||
- --no-mmproj-offload
|
- --mmproj-offload
|
||||||
- --alias
|
- --alias
|
||||||
- qwen-fast
|
- qwen-fast
|
||||||
- --ctx-size
|
- --ctx-size
|
||||||
@@ -102,12 +106,15 @@ services:
|
|||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
# Keep the language model split unchanged while placing the complete
|
||||||
|
# multimodal projector on the secondary RTX 3060.
|
||||||
|
MTMD_BACKEND_DEVICE: CUDA1
|
||||||
command:
|
command:
|
||||||
- --model
|
- --model
|
||||||
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
|
- "/models/${MEDIUM_MODEL_FILE:?MEDIUM_MODEL_FILE is required}"
|
||||||
- --mmproj
|
- --mmproj
|
||||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||||
- --no-mmproj-offload
|
- --mmproj-offload
|
||||||
- --alias
|
- --alias
|
||||||
- qwen-medium
|
- qwen-medium
|
||||||
- --ctx-size
|
- --ctx-size
|
||||||
@@ -173,12 +180,13 @@ services:
|
|||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
MTMD_BACKEND_DEVICE: CUDA1
|
||||||
command:
|
command:
|
||||||
- --model
|
- --model
|
||||||
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
||||||
- --mmproj
|
- --mmproj
|
||||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||||
- --no-mmproj-offload
|
- --mmproj-offload
|
||||||
- --alias
|
- --alias
|
||||||
- qwen-large
|
- qwen-large
|
||||||
- --ctx-size
|
- --ctx-size
|
||||||
|
|||||||
@@ -0,0 +1,92 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Run a private, deterministic llama.cpp vision latency smoke test."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import base64
|
||||||
|
import json
|
||||||
|
import struct
|
||||||
|
import time
|
||||||
|
import urllib.request
|
||||||
|
import zlib
|
||||||
|
|
||||||
|
|
||||||
|
def chunk(kind: bytes, payload: bytes) -> bytes:
|
||||||
|
return (
|
||||||
|
struct.pack(">I", len(payload))
|
||||||
|
+ kind
|
||||||
|
+ payload
|
||||||
|
+ struct.pack(">I", zlib.crc32(kind + payload) & 0xFFFFFFFF)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def synthetic_png(width: int = 1024, height: int = 768) -> bytes:
|
||||||
|
rows = []
|
||||||
|
for y in range(height):
|
||||||
|
row = bytearray([0])
|
||||||
|
for x in range(width):
|
||||||
|
row.extend(((x * 255) // width, (y * 255) // height, ((x // 64 + y // 64) % 2) * 210))
|
||||||
|
rows.append(bytes(row))
|
||||||
|
payload = zlib.compress(b"".join(rows), level=6)
|
||||||
|
return (
|
||||||
|
b"\x89PNG\r\n\x1a\n"
|
||||||
|
+ chunk(b"IHDR", struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0))
|
||||||
|
+ chunk(b"IDAT", payload)
|
||||||
|
+ chunk(b"IEND", b"")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--url", required=True)
|
||||||
|
parser.add_argument("--model", default="qwen-medium")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
image = base64.b64encode(synthetic_png()).decode("ascii")
|
||||||
|
body = {
|
||||||
|
"model": args.model,
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "image_url",
|
||||||
|
"image_url": {"url": f"data:image/png;base64,{image}"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "text",
|
||||||
|
"text": "Beschreibe dieses synthetische Testbild knapp auf Deutsch.",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"temperature": 0,
|
||||||
|
"max_tokens": 80,
|
||||||
|
"cache_prompt": False,
|
||||||
|
}
|
||||||
|
request = urllib.request.Request(
|
||||||
|
args.url.rstrip("/") + "/v1/chat/completions",
|
||||||
|
data=json.dumps(body).encode("utf-8"),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
)
|
||||||
|
started = time.perf_counter()
|
||||||
|
with urllib.request.urlopen(request, timeout=600) as response:
|
||||||
|
result = json.load(response)
|
||||||
|
elapsed = time.perf_counter() - started
|
||||||
|
print(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"wall_seconds": round(elapsed, 3),
|
||||||
|
"usage": result.get("usage"),
|
||||||
|
"timings": result.get("timings"),
|
||||||
|
"answer": result["choices"][0]["message"].get("content", "")[:240],
|
||||||
|
},
|
||||||
|
ensure_ascii=False,
|
||||||
|
indent=2,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -68,6 +68,14 @@ halten.
|
|||||||
Diese Matrix wurde auf RTX 5080 und RTX 3060 gemessen und ist bis zu einer
|
Diese Matrix wurde auf RTX 5080 und RTX 3060 gemessen und ist bis zu einer
|
||||||
bewussten Neubewertung der verbindliche Produktionsstandard.
|
bewussten Neubewertung der verbindliche Produktionsstandard.
|
||||||
|
|
||||||
|
Bei Fast, Medium und Large bleibt das Sprachmodell wie in der Tabelle
|
||||||
|
verteilt, während `MTMD_BACKEND_DEVICE=CUDA1` den vollständigen
|
||||||
|
Multimodal-Projektor auf die RTX 3060 legt. Ein reproduzierbarer Test mit einem
|
||||||
|
synthetischen Bild (1024×768, 835 Eingabetoken) senkte die Bild-/Promptzeit im
|
||||||
|
Medium-Profil von 23,17 auf 1,68 Sekunden und die gesamte Anfrage von 24,96
|
||||||
|
auf 2,94 Sekunden. Der Projektor belegte dabei rund 1,1 GiB zusätzlichen VRAM
|
||||||
|
auf der 3060. Ultra bleibt für maximalen Kontext bewusst text-only.
|
||||||
|
|
||||||
## Netzwerk
|
## Netzwerk
|
||||||
|
|
||||||
- Docker-Netze liegen ausschließlich unter `172.30.0.0/16`.
|
- Docker-Netze liegen ausschließlich unter `172.30.0.0/16`.
|
||||||
|
|||||||
+1
-1
@@ -218,7 +218,7 @@ MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
|
|||||||
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
||||||
ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144}
|
ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144}
|
||||||
EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800}
|
EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800}
|
||||||
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10}
|
MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10}
|
||||||
LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
|
|||||||
Reference in New Issue
Block a user