router: Bildgenerierung mit FLUX.2 [klein] 4B Base (GPU-Hotswap)
- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality) - quality: standard=30 Steps (Default), high=50 Steps - Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920 - GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen wiederherstellen (exakt vorheriges Profil) - Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock) - Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT - Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness - /status: image.phase, image.worker, image.model_loaded, qwen.available, qwen.active_chats - GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis) - image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 + enable_model_cpu_offload) - deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download, Image-Dir, systemd-Unit mit Image-Umgebungsvariablen - dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps, Quality-Compare), 32 lokale Tests - README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080), Python-Pakete Benchmarks (RTX 5080, 16 GB, CPU-Offload): - 512x512 / 10 Steps: ~9.3 s - 1024x1024 / 30 Steps: ~31.3 s - 1024x1024 / 50 Steps: ~45.3 s - 1920x1088 / 50 Steps: ~91 s - Peak-VRAM: ~8.4-8.9 GB - Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
This commit is contained in:
+234
-13
@@ -13,7 +13,7 @@ FAIL=0
|
||||
|
||||
cleanup() {
|
||||
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" 2>/dev/null || true
|
||||
rm -f /tmp/mock_pid2
|
||||
rm -f /tmp/mock_pid2 /tmp/mock_upstream_pid
|
||||
wait 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT
|
||||
@@ -21,22 +21,41 @@ trap cleanup EXIT
|
||||
ok() { echo " PASS: $1"; PASS=$((PASS+1)); }
|
||||
bad() { echo " FAIL: $1"; FAIL=$((FAIL+1)); }
|
||||
|
||||
# --- Mock-llama.cpp starten --------------------------------------------------
|
||||
# --- Mock-llama.cpp starten (über Fake-systemctl) ------------------------------
|
||||
echo "== Starte Mock-llama.cpp (Port $UP_PORT)"
|
||||
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream.log 2>&1 &
|
||||
MOCK_PID=$!
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
|
||||
FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
|
||||
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream.log \
|
||||
bash dev/fake-systemctl.sh start
|
||||
sleep 0.5
|
||||
MOCK_PID=$(cat /tmp/mock_upstream_pid 2>/dev/null || echo "")
|
||||
|
||||
# --- Router starten -----------------------------------------------------------
|
||||
echo "== Starte Router (Port $RT_PORT)"
|
||||
rm -rf /tmp/test-images
|
||||
ROUTER_HOST=127.0.0.1 ROUTER_PORT="$RT_PORT" \
|
||||
UPSTREAM_URL="http://127.0.0.1:$UP_PORT" \
|
||||
PROFILE_SCRIPT="$PWD/dev/fake-llama-profile.sh" \
|
||||
PROFILE_DIR="$FAKE_DIR" \
|
||||
SWITCH_TIMEOUT=30 \
|
||||
SYSTEMCTL_BIN="$PWD/dev/fake-systemctl.sh" \
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
|
||||
FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
|
||||
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream_fake.log \
|
||||
IMAGE_WORKER="$PWD/dev/mock_image_worker.py" \
|
||||
IMAGE_PYTHON=python3 \
|
||||
IMAGE_DIR=/tmp/test-images \
|
||||
IMAGE_WORKER_LOG=/tmp/test_worker.log \
|
||||
IMAGE_GEN_TIMEOUT=30 \
|
||||
MOCK_WORKER_LOG=/tmp/test_worker_requests.jsonl \
|
||||
python3 router/ai_profile_router.py >/tmp/router_test.log 2>&1 &
|
||||
ROUTER_PID=$!
|
||||
sleep 0.5
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
|
||||
# --- 1. /v1/models -------------------------------------------------------------
|
||||
echo "== Test 1: /v1/models"
|
||||
@@ -150,30 +169,232 @@ cat /tmp/err9b.json; echo
|
||||
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
kill "$MOCK_PID" 2>/dev/null; wait "$MOCK_PID" 2>/dev/null || true
|
||||
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
bash dev/fake-systemctl.sh stop
|
||||
sleep 0.5
|
||||
CODE=$(curl -s -o /tmp/err10.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[]}')
|
||||
cat /tmp/err10.json; echo
|
||||
[ "$CODE" = "502" ] && ok "502 bei downem Upstream (Profil bereits aktiv)" || bad "erwartet 502, bekam $CODE"
|
||||
|
||||
# Mock nach ~3 s neu starten (simuliert systemctl restart durch das Profil-Skript)
|
||||
rm -f /tmp/mock_pid2
|
||||
(
|
||||
sleep 3
|
||||
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream2.log 2>&1 &
|
||||
echo $! > /tmp/mock_pid2
|
||||
) &
|
||||
# Mock neu starten (simuliert systemctl restart durch das Profil-Skript)
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream2.log \
|
||||
bash dev/fake-systemctl.sh start
|
||||
RESP=$(curl -sf -X POST "$BASE/fast")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
# neuen Mock als MOCK_PID übernehmen, damit Cleanup ihn beendet
|
||||
[ -f /tmp/mock_pid2 ] && MOCK_PID=$(cat /tmp/mock_pid2)
|
||||
[ -f /tmp/mock_upstream_pid ] && MOCK_PID=$(cat /tmp/mock_upstream_pid)
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["profile"]=="fast" and d["model"]=="mock-model-73728", d
|
||||
' && ok "Recovery: /fast wartet auf Upstream, dann Erfolg" || bad "Recovery"
|
||||
|
||||
# --- 11. Bildgenerierung (Mock-Worker) -------------------------------------------------
|
||||
echo "== Test 11: POST /v1/images/generations (1024x1024)"
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"ein rotes Haus","size":"1024x1024"}')
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
IMG_NAME=$(echo "$RESP" | python3 -c 'import json,sys; print(json.load(sys.stdin)["data"][0]["url"].rsplit("/",1)[1])')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert len(d["data"])==1, d
|
||||
assert d["data"][0]["url"].startswith("http://"), d
|
||||
' && [ -f "/tmp/test-images/$IMG_NAME" ] \
|
||||
&& ok "Bild generiert und gespeichert ($IMG_NAME)" || bad "Bildgenerierung"
|
||||
|
||||
# --- 12. Bild-Download -----------------------------------------------------------------
|
||||
echo "== Test 12: GET /images/<datei>"
|
||||
CODE=$(curl -s -o /tmp/test_dl.png -w "%{http_code}" -D /tmp/hdr12.txt "$BASE/images/$IMG_NAME")
|
||||
CTYPE=$(grep -i content-type /tmp/hdr12.txt | tr -d "\r")
|
||||
[ "$CODE" = "200" ] && [ -s /tmp/test_dl.png ] && echo "$CTYPE" | grep -qi "image/png" \
|
||||
&& ok "PNG-Download (200, $CTYPE)" || bad "PNG-Download (Code $CODE, $CTYPE)"
|
||||
|
||||
# --- 13. Bild-Liste ---------------------------------------------------------------------
|
||||
echo "== Test 13: GET /images"
|
||||
RESP=$(curl -sf "$BASE/images")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c "
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
names=[i['name'] for i in d['images']]
|
||||
assert '$IMG_NAME' in names, names
|
||||
" && ok "Bild in Liste enthalten" || bad "Bild-Liste"
|
||||
|
||||
# --- 14. Validierung ---------------------------------------------------------------------
|
||||
echo "== Test 14: Validierung (Größe, Prompt, n)"
|
||||
CODE=$(curl -s -o /tmp/err14a.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","size":"500x500"}')
|
||||
cat /tmp/err14a.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültiger Größe" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err14b.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"size":"1024x1024"}')
|
||||
cat /tmp/err14b.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei fehlendem Prompt" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err14c.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","n":9}')
|
||||
cat /tmp/err14c.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei n=9 (max 4)" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
# --- 15. b64_json + n=2 + Seed -------------------------------------------------------------
|
||||
echo "== Test 15: response_format=b64_json, n=2, seed"
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"zwei Bilder","size":"1024x1024","n":2,"seed":42,"response_format":"b64_json"}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys,base64
|
||||
d=json.load(sys.stdin)
|
||||
assert len(d["data"])==2, d
|
||||
for item in d["data"]:
|
||||
assert item["url"] is None, item
|
||||
png=base64.b64decode(item["b64_json"])
|
||||
assert png[:4]==b"\x89PNG", "kein PNG"
|
||||
' && ok "2 Bilder als b64_json (gültige PNGs)" || bad "b64_json/n=2"
|
||||
|
||||
# --- 16. /status zeigt Bild-Zustand ---------------------------------------------------------
|
||||
echo "== Test 16: /status mit Bild-Section"
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
img=d["image"]
|
||||
assert img["phase"]=="idle", img
|
||||
assert img["worker"]=="stopped", img # Worker wird nach Job beendet
|
||||
assert img["model_loaded"] is False, img
|
||||
assert img["last_image"], img
|
||||
assert img["last_error"] is None, img
|
||||
q=d["qwen"]
|
||||
assert q["available"] is True, q
|
||||
assert q["active_chats"]==0, q
|
||||
' && ok "Status: phase=idle, worker=stopped, qwen verfügbar" || bad "Status Bild-Section"
|
||||
|
||||
# --- 17. Qwen nach Bildgenerierung erreichbar -------------------------------------------------
|
||||
echo "== Test 17: Qwen nach Bildgenerierung erreichbar"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
|
||||
' && ok "Chat funktioniert nach Bildgenerierung" || bad "Chat nach Bild"
|
||||
|
||||
# --- 18. quality=standard → 30 Steps -------------------------------------------------------------
|
||||
echo "== Test 18: quality=standard → 30 Steps"
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"standard test","size":"1024x1024","quality":"standard"}')
|
||||
sleep 0.3
|
||||
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
|
||||
[ "$STEPS" = "30" ] && ok "quality=standard → 30 Steps" || bad "erwartet 30 Steps, bekam $STEPS"
|
||||
|
||||
# --- 19. quality=high → 50 Steps -------------------------------------------------------------------
|
||||
echo "== Test 19: quality=high → 50 Steps"
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"high test","size":"1024x1024","quality":"high"}')
|
||||
sleep 0.3
|
||||
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
|
||||
[ "$STEPS" = "50" ] && ok "quality=high → 50 Steps" || bad "erwartet 50 Steps, bekam $STEPS"
|
||||
|
||||
# --- 20. ungültige Qualität → 400 ------------------------------------------------------------------
|
||||
echo "== Test 20: ungültige Qualität → 400"
|
||||
CODE=$(curl -s -o /tmp/err20.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","quality":"bogus"}')
|
||||
cat /tmp/err20.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültiger Qualität" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
# --- 21. Image-Fehler → Qwen wiederhergestellt ------------------------------------------------------
|
||||
echo "== Test 21: Image-Fehler → Qwen wiederhergestellt"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
CODE=$(curl -s -o /tmp/err21.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"FAIL","size":"1024x1024"}')
|
||||
cat /tmp/err21.json; echo
|
||||
sleep 0.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["current_profile"]=="fast", d
|
||||
assert d["upstream"]["reachable"] is True, d
|
||||
assert d["qwen"]["available"] is True, d
|
||||
' && ok "Qwen nach Image-Fehler wiederhergestellt (fast, erreichbar)" || bad "Qwen nicht wiederhergestellt"
|
||||
|
||||
# --- 22. Fast → Image → Fast ------------------------------------------------------------------------
|
||||
echo "== Test 22: Fast → Image → Fast"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"fast test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "fast" ] && ok "Fast → Image → Fast" || bad "Profil nach Image: $PROFILE (erwartet fast)"
|
||||
|
||||
# --- 23. Medium → Image → Medium --------------------------------------------------------------------
|
||||
echo "== Test 23: Medium → Image → Medium"
|
||||
curl -sf -X POST "$BASE/medium" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"medium test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
||||
|
||||
# --- 24. Long → Image → Long ------------------------------------------------------------------------
|
||||
echo "== Test 24: Long → Image → Long"
|
||||
curl -sf -X POST "$BASE/long" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"long test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
|
||||
|
||||
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
||||
echo "== Test 25: /status während Image-Job"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img25.json 2>&1 &
|
||||
IMG_PID=$!
|
||||
sleep 1.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
img=d["image"]
|
||||
assert img["phase"]!="idle", img
|
||||
assert d["qwen"]["available"] is False, d
|
||||
' && ok "Status während Image-Job: phase!=idle, qwen unavailable" || bad "Status während Image-Job"
|
||||
wait $IMG_PID
|
||||
sleep 0.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["qwen"]["available"] is True, d
|
||||
assert d["image"]["phase"]=="idle", d
|
||||
' && ok "Nach Image-Job: qwen verfügbar, phase=idle" || bad "Nach Image-Job"
|
||||
|
||||
# --- 26. paralleler Chat während Image-Job (wartet, kein 502) ----------------------------------------
|
||||
echo "== Test 26: paralleler Chat während Image-Job (wartet, kein 502)"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img26.json 2>&1 &
|
||||
IMG_PID=$!
|
||||
sleep 1.5
|
||||
START=$(date +%s)
|
||||
CODE=$(curl -s -o /tmp/chat26.json -w "%{http_code}" "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
END=$(date +%s)
|
||||
ELAPSED=$((END-START))
|
||||
cat /tmp/chat26.json; echo
|
||||
wait $IMG_PID
|
||||
[ "$CODE" = "200" ] && [ "$ELAPSED" -ge 2 ] \
|
||||
&& ok "Chat wartete ${ELAPSED}s (kein 502), dann 200" || bad "Chat: Code $CODE, ${ELAPSED}s"
|
||||
|
||||
# --- Ergebnis --------------------------------------------------------------------------------------------
|
||||
echo
|
||||
echo "== Ergebnis: $PASS bestanden, $FAIL fehlgeschlagen =="
|
||||
|
||||
Reference in New Issue
Block a user