- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality) - quality: standard=30 Steps (Default), high=50 Steps - Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920 - GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen wiederherstellen (exakt vorheriges Profil) - Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock) - Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT - Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness - /status: image.phase, image.worker, image.model_loaded, qwen.available, qwen.active_chats - GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis) - image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 + enable_model_cpu_offload) - deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download, Image-Dir, systemd-Unit mit Image-Umgebungsvariablen - dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps, Quality-Compare), 32 lokale Tests - README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080), Python-Pakete Benchmarks (RTX 5080, 16 GB, CPU-Offload): - 512x512 / 10 Steps: ~9.3 s - 1024x1024 / 30 Steps: ~31.3 s - 1024x1024 / 50 Steps: ~45.3 s - 1920x1088 / 50 Steps: ~91 s - Peak-VRAM: ~8.4-8.9 GB - Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
67 lines
2.0 KiB
Bash
67 lines
2.0 KiB
Bash
#!/bin/bash
|
|
# FLUX GPU-Resident-Benchmark mit GARANTIERTER Qwen-Recovery.
|
|
#
|
|
# Ablauf:
|
|
# 1. Aktives Qwen-Profil aus override.conf merken
|
|
# 2. llama.cpp stoppen, VRAM-Abgabe verifizieren
|
|
# 3. Benchmark ausführen (Python, GPU-resident, ohne CPU-Offload)
|
|
# 4. IMMER (trap EXIT): llama.cpp wieder starten, Readiness prüfen
|
|
#
|
|
# Usage: flux_gpu_benchmark.sh
|
|
set -u
|
|
|
|
SERVICE=mike-ai-llama-ui.service
|
|
PROFILE_DIR=/etc/systemd/system/mike-ai-llama-ui.service.d
|
|
PY=/opt/mike-ai/ai-profile-router/venv/bin/python
|
|
SCRIPT="$(cd "$(dirname "$0")" && pwd)/flux_gpu_benchmark.py"
|
|
ROUTER_STATUS=http://127.0.0.1:8081/status
|
|
|
|
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
|
# vergleichen (robuster als String-Matching).
|
|
PROFILE=""
|
|
for p in fast medium long; do
|
|
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
|
PROFILE=$p
|
|
break
|
|
fi
|
|
done
|
|
if [ -z "$PROFILE" ]; then
|
|
echo "FEHLER: aktives Profil nicht erkannt (override.conf passt zu keinem Profil-File)"
|
|
exit 1
|
|
fi
|
|
echo "=== Aktives Qwen-Profil: $PROFILE (wird nach dem Test wiederhergestellt) ==="
|
|
|
|
restore() {
|
|
echo
|
|
echo "=== RECOVERY: stelle Profil $PROFILE wieder her ==="
|
|
systemctl start "$SERVICE" 2>&1 || echo "systemctl start fehlgeschlagen"
|
|
for i in $(seq 1 150); do
|
|
if curl -sf "$ROUTER_STATUS" 2>/dev/null | python3 -c '
|
|
import json, sys
|
|
d = json.load(sys.stdin)
|
|
u = d["upstream"]
|
|
assert u["reachable"] and u["model"], d
|
|
print("ready:", u["model"], "ctx", u["ctx"])
|
|
' 2>/dev/null; then
|
|
echo "=== RECOVERY OK: llama.cpp ist wieder inference-ready ==="
|
|
return 0
|
|
fi
|
|
sleep 2
|
|
done
|
|
echo "=== RECOVERY FEHLGESCHLAGEN: bitte manuell prüfen (systemctl status $SERVICE) ==="
|
|
return 1
|
|
}
|
|
trap restore EXIT
|
|
|
|
echo "=== Stoppe $SERVICE ==="
|
|
systemctl stop "$SERVICE"
|
|
sleep 3
|
|
echo "=== VRAM nach Stop (MiB) ==="
|
|
nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits
|
|
|
|
echo "=== Starte Benchmark ==="
|
|
"$PY" "$SCRIPT"
|
|
RC=$?
|
|
echo "=== Benchmark-Exit-Code: $RC ==="
|
|
exit $RC
|