Files
AI-Profile-Router/dev/test_local.sh
T
Mikei386 7c5bbe2ffb router: Bildgenerierung mit FLUX.2 [klein] 4B Base (GPU-Hotswap)
- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality)
- quality: standard=30 Steps (Default), high=50 Steps
- Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920
- GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen
  wiederherstellen (exakt vorheriges Profil)
- Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock)
- Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT
- Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness
- /status: image.phase, image.worker, image.model_loaded, qwen.available,
  qwen.active_chats
- GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis)
- image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 +
  enable_model_cpu_offload)
- deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download,
  Image-Dir, systemd-Unit mit Image-Umgebungsvariablen
- dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps,
  Quality-Compare), 32 lokale Tests
- README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080),
  Python-Pakete

Benchmarks (RTX 5080, 16 GB, CPU-Offload):
- 512x512 / 10 Steps: ~9.3 s
- 1024x1024 / 30 Steps: ~31.3 s
- 1024x1024 / 50 Steps: ~45.3 s
- 1920x1088 / 50 Steps: ~91 s
- Peak-VRAM: ~8.4-8.9 GB
- Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
2026-08-19 08:51:37 +02:00

402 lines
18 KiB
Bash
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/bin/bash
# Lokaler Test des AI Profile Router (Mock-llama.cpp + Fake-Profil-Skript).
# Startet Mock + Router, führt alle Testfälle aus, beendet beide.
set -uo pipefail
cd "$(dirname "$0")/.."
UP_PORT=18080
RT_PORT=18081
BASE="http://127.0.0.1:$RT_PORT"
FAKE_DIR="$PWD/dev/fake-profile-dir"
PASS=0
FAIL=0
cleanup() {
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" 2>/dev/null || true
rm -f /tmp/mock_pid2 /tmp/mock_upstream_pid
wait 2>/dev/null || true
}
trap cleanup EXIT
ok() { echo " PASS: $1"; PASS=$((PASS+1)); }
bad() { echo " FAIL: $1"; FAIL=$((FAIL+1)); }
# --- Mock-llama.cpp starten (über Fake-systemctl) ------------------------------
echo "== Starte Mock-llama.cpp (Port $UP_PORT)"
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
FAKE_SYSTEMD_PORT="$UP_PORT" \
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
FAKE_SYSTEMD_LOG=/tmp/mock_upstream.log \
bash dev/fake-systemctl.sh start
sleep 0.5
MOCK_PID=$(cat /tmp/mock_upstream_pid 2>/dev/null || echo "")
# --- Router starten -----------------------------------------------------------
echo "== Starte Router (Port $RT_PORT)"
rm -rf /tmp/test-images
ROUTER_HOST=127.0.0.1 ROUTER_PORT="$RT_PORT" \
UPSTREAM_URL="http://127.0.0.1:$UP_PORT" \
PROFILE_SCRIPT="$PWD/dev/fake-llama-profile.sh" \
PROFILE_DIR="$FAKE_DIR" \
SWITCH_TIMEOUT=30 \
SYSTEMCTL_BIN="$PWD/dev/fake-systemctl.sh" \
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
FAKE_SYSTEMD_PORT="$UP_PORT" \
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
FAKE_SYSTEMD_LOG=/tmp/mock_upstream_fake.log \
IMAGE_WORKER="$PWD/dev/mock_image_worker.py" \
IMAGE_PYTHON=python3 \
IMAGE_DIR=/tmp/test-images \
IMAGE_WORKER_LOG=/tmp/test_worker.log \
IMAGE_GEN_TIMEOUT=30 \
MOCK_WORKER_LOG=/tmp/test_worker_requests.jsonl \
python3 router/ai_profile_router.py >/tmp/router_test.log 2>&1 &
ROUTER_PID=$!
sleep 0.5
rm -f /tmp/test_worker_requests.jsonl
# --- 1. /v1/models -------------------------------------------------------------
echo "== Test 1: /v1/models"
RESP=$(curl -sf "$BASE/v1/models")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
ids={m["id"]:m for m in d["data"]}
assert set(ids)=={"qwen-fast","qwen-medium","qwen-long"}, ids
assert ids["qwen-fast"]["context_length"]==73728
assert ids["qwen-medium"]["context_length"]==94208
assert ids["qwen-long"]["context_length"]==131072
' && ok "drei virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
# --- 2. /status -----------------------------------------------------------------
echo "== Test 2: /status"
RESP=$(curl -sf "$BASE/status")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["current_profile"]=="fast", d
assert d["upstream"]["reachable"] is True, d
assert d["upstream"]["ctx"]==73728, d
' && ok "Status zeigt Profil fast + erreichbares Upstream" || bad "/status"
# --- 3. Normales Forwarding (non-streaming) -------------------------------------
echo "== Test 3: Forwarding non-streaming (qwen-fast)"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["model"].startswith("mock-model-"), d
assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
' && ok "Request wurde weitergeleitet, Modell ersetzt" || bad "Forwarding"
# --- 4. Streaming ----------------------------------------------------------------
echo "== Test 4: Streaming (SSE)"
RESP=$(curl -sfN "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","stream":true,"messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP"
echo "$RESP" | grep -q "data: " && echo "$RESP" | grep -q "\[DONE\]" \
&& ok "SSE-Stream mit [DONE] erhalten" || bad "Streaming"
# --- 5. Tool Calls -----------------------------------------------------------------
echo "== Test 5: Tool Calls"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Wetter?"}],
"tools":[{"type":"function","function":{"name":"get_weather","parameters":{}}}]}' )
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
tc=d["choices"][0]["message"]["tool_calls"]
assert tc[0]["function"]["name"]=="get_weather", d
assert d["choices"][0]["finish_reason"]=="tool_calls", d
' && ok "Tool Calls transparent weitergereicht" || bad "Tool Calls"
# --- 6. Profilwechsel fast -> medium ------------------------------------------------
echo "== Test 6: Profilwechsel fast -> medium"
RESP=$(curl -sf -X POST "$BASE/medium")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["profile"]=="medium" and d["context_length"]==94208, d
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
curl -sf "$BASE/status" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["current_profile"]=="medium", d
assert d["upstream"]["ctx"]==94208, d
' && ok "Status bestätigt medium (ctx 94208)" || bad "Status nach Wechsel"
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
echo "== Test 7: Profilwechsel medium -> fast"
RESP=$(curl -sf -X POST "$BASE/fast")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["profile"]=="fast" and d["context_length"]==73728, d
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
echo "== Test 8: Chat mit qwen-long triggert Wechsel auf long"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-long","messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["model"]=="mock-model-131072", d
' && ok "qwen-long hat Profil long aktiviert und weitergeleitet" || bad "virtuelles Modell"
# --- 9. Ungültiges Profil ------------------------------------------------------------------
echo "== Test 9: Ungültiges Profil"
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" -X POST "$BASE/huge")
cat /tmp/err9.json; echo
[ "$CODE" = "400" ] && ok "400 bei unbekanntem Profil (POST /huge)" || bad "erwartet 400, bekam $CODE"
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
cat /tmp/err9b.json; echo
[ "$CODE" = "400" ] && ok "400 bei unbekanntem virtuellen Modell (qwen-huge)" || bad "erwartet 400, bekam $CODE"
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
echo "== Test 10: Upstream down -> 502, danach Recovery"
# Profil auf fast setzen (aus Test 8 ist long aktiv)
curl -sf -X POST "$BASE/fast" >/dev/null
# Mock stoppen (simuliert Crash) – über Fake-systemctl
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
bash dev/fake-systemctl.sh stop
sleep 0.5
CODE=$(curl -s -o /tmp/err10.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[]}')
cat /tmp/err10.json; echo
[ "$CODE" = "502" ] && ok "502 bei downem Upstream (Profil bereits aktiv)" || bad "erwartet 502, bekam $CODE"
# Mock neu starten (simuliert systemctl restart durch das Profil-Skript)
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
FAKE_SYSTEMD_LOG=/tmp/mock_upstream2.log \
bash dev/fake-systemctl.sh start
RESP=$(curl -sf -X POST "$BASE/fast")
echo "$RESP" | python3 -m json.tool
# neuen Mock als MOCK_PID übernehmen, damit Cleanup ihn beendet
[ -f /tmp/mock_upstream_pid ] && MOCK_PID=$(cat /tmp/mock_upstream_pid)
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["profile"]=="fast" and d["model"]=="mock-model-73728", d
' && ok "Recovery: /fast wartet auf Upstream, dann Erfolg" || bad "Recovery"
# --- 11. Bildgenerierung (Mock-Worker) -------------------------------------------------
echo "== Test 11: POST /v1/images/generations (1024x1024)"
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"ein rotes Haus","size":"1024x1024"}')
echo "$RESP" | python3 -m json.tool
IMG_NAME=$(echo "$RESP" | python3 -c 'import json,sys; print(json.load(sys.stdin)["data"][0]["url"].rsplit("/",1)[1])')
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert len(d["data"])==1, d
assert d["data"][0]["url"].startswith("http://"), d
' && [ -f "/tmp/test-images/$IMG_NAME" ] \
&& ok "Bild generiert und gespeichert ($IMG_NAME)" || bad "Bildgenerierung"
# --- 12. Bild-Download -----------------------------------------------------------------
echo "== Test 12: GET /images/<datei>"
CODE=$(curl -s -o /tmp/test_dl.png -w "%{http_code}" -D /tmp/hdr12.txt "$BASE/images/$IMG_NAME")
CTYPE=$(grep -i content-type /tmp/hdr12.txt | tr -d "\r")
[ "$CODE" = "200" ] && [ -s /tmp/test_dl.png ] && echo "$CTYPE" | grep -qi "image/png" \
&& ok "PNG-Download (200, $CTYPE)" || bad "PNG-Download (Code $CODE, $CTYPE)"
# --- 13. Bild-Liste ---------------------------------------------------------------------
echo "== Test 13: GET /images"
RESP=$(curl -sf "$BASE/images")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c "
import json,sys
d=json.load(sys.stdin)
names=[i['name'] for i in d['images']]
assert '$IMG_NAME' in names, names
" && ok "Bild in Liste enthalten" || bad "Bild-Liste"
# --- 14. Validierung ---------------------------------------------------------------------
echo "== Test 14: Validierung (Größe, Prompt, n)"
CODE=$(curl -s -o /tmp/err14a.json -w "%{http_code}" "$BASE/v1/images/generations" \
-H "Content-Type: application/json" -d '{"prompt":"x","size":"500x500"}')
cat /tmp/err14a.json; echo
[ "$CODE" = "400" ] && ok "400 bei ungültiger Größe" || bad "erwartet 400, bekam $CODE"
CODE=$(curl -s -o /tmp/err14b.json -w "%{http_code}" "$BASE/v1/images/generations" \
-H "Content-Type: application/json" -d '{"size":"1024x1024"}')
cat /tmp/err14b.json; echo
[ "$CODE" = "400" ] && ok "400 bei fehlendem Prompt" || bad "erwartet 400, bekam $CODE"
CODE=$(curl -s -o /tmp/err14c.json -w "%{http_code}" "$BASE/v1/images/generations" \
-H "Content-Type: application/json" -d '{"prompt":"x","n":9}')
cat /tmp/err14c.json; echo
[ "$CODE" = "400" ] && ok "400 bei n=9 (max 4)" || bad "erwartet 400, bekam $CODE"
# --- 15. b64_json + n=2 + Seed -------------------------------------------------------------
echo "== Test 15: response_format=b64_json, n=2, seed"
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"zwei Bilder","size":"1024x1024","n":2,"seed":42,"response_format":"b64_json"}')
echo "$RESP" | python3 -c '
import json,sys,base64
d=json.load(sys.stdin)
assert len(d["data"])==2, d
for item in d["data"]:
assert item["url"] is None, item
png=base64.b64decode(item["b64_json"])
assert png[:4]==b"\x89PNG", "kein PNG"
' && ok "2 Bilder als b64_json (gültige PNGs)" || bad "b64_json/n=2"
# --- 16. /status zeigt Bild-Zustand ---------------------------------------------------------
echo "== Test 16: /status mit Bild-Section"
RESP=$(curl -sf "$BASE/status")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
img=d["image"]
assert img["phase"]=="idle", img
assert img["worker"]=="stopped", img # Worker wird nach Job beendet
assert img["model_loaded"] is False, img
assert img["last_image"], img
assert img["last_error"] is None, img
q=d["qwen"]
assert q["available"] is True, q
assert q["active_chats"]==0, q
' && ok "Status: phase=idle, worker=stopped, qwen verfügbar" || bad "Status Bild-Section"
# --- 17. Qwen nach Bildgenerierung erreichbar -------------------------------------------------
echo "== Test 17: Qwen nach Bildgenerierung erreichbar"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
' && ok "Chat funktioniert nach Bildgenerierung" || bad "Chat nach Bild"
# --- 18. quality=standard → 30 Steps -------------------------------------------------------------
echo "== Test 18: quality=standard → 30 Steps"
rm -f /tmp/test_worker_requests.jsonl
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"standard test","size":"1024x1024","quality":"standard"}')
sleep 0.3
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
[ "$STEPS" = "30" ] && ok "quality=standard → 30 Steps" || bad "erwartet 30 Steps, bekam $STEPS"
# --- 19. quality=high → 50 Steps -------------------------------------------------------------------
echo "== Test 19: quality=high → 50 Steps"
rm -f /tmp/test_worker_requests.jsonl
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"high test","size":"1024x1024","quality":"high"}')
sleep 0.3
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
[ "$STEPS" = "50" ] && ok "quality=high → 50 Steps" || bad "erwartet 50 Steps, bekam $STEPS"
# --- 20. ungültige Qualität → 400 ------------------------------------------------------------------
echo "== Test 20: ungültige Qualität → 400"
CODE=$(curl -s -o /tmp/err20.json -w "%{http_code}" "$BASE/v1/images/generations" \
-H "Content-Type: application/json" -d '{"prompt":"x","quality":"bogus"}')
cat /tmp/err20.json; echo
[ "$CODE" = "400" ] && ok "400 bei ungültiger Qualität" || bad "erwartet 400, bekam $CODE"
# --- 21. Image-Fehler → Qwen wiederhergestellt ------------------------------------------------------
echo "== Test 21: Image-Fehler → Qwen wiederhergestellt"
curl -sf -X POST "$BASE/fast" >/dev/null
CODE=$(curl -s -o /tmp/err21.json -w "%{http_code}" "$BASE/v1/images/generations" \
-H "Content-Type: application/json" -d '{"prompt":"FAIL","size":"1024x1024"}')
cat /tmp/err21.json; echo
sleep 0.5
RESP=$(curl -sf "$BASE/status")
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["current_profile"]=="fast", d
assert d["upstream"]["reachable"] is True, d
assert d["qwen"]["available"] is True, d
' && ok "Qwen nach Image-Fehler wiederhergestellt (fast, erreichbar)" || bad "Qwen nicht wiederhergestellt"
# --- 22. Fast → Image → Fast ------------------------------------------------------------------------
echo "== Test 22: Fast → Image → Fast"
curl -sf -X POST "$BASE/fast" >/dev/null
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"fast test","size":"1024x1024"}')
sleep 0.5
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
[ "$PROFILE" = "fast" ] && ok "Fast → Image → Fast" || bad "Profil nach Image: $PROFILE (erwartet fast)"
# --- 23. Medium → Image → Medium --------------------------------------------------------------------
echo "== Test 23: Medium → Image → Medium"
curl -sf -X POST "$BASE/medium" >/dev/null
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"medium test","size":"1024x1024"}')
sleep 0.5
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
# --- 24. Long → Image → Long ------------------------------------------------------------------------
echo "== Test 24: Long → Image → Long"
curl -sf -X POST "$BASE/long" >/dev/null
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"long test","size":"1024x1024"}')
sleep 0.5
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
# --- 25. /status während Image-Job -------------------------------------------------------------------
echo "== Test 25: /status während Image-Job"
curl -sf -X POST "$BASE/fast" >/dev/null
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img25.json 2>&1 &
IMG_PID=$!
sleep 1.5
RESP=$(curl -sf "$BASE/status")
echo "$RESP" | python3 -m json.tool
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
img=d["image"]
assert img["phase"]!="idle", img
assert d["qwen"]["available"] is False, d
' && ok "Status während Image-Job: phase!=idle, qwen unavailable" || bad "Status während Image-Job"
wait $IMG_PID
sleep 0.5
RESP=$(curl -sf "$BASE/status")
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d["qwen"]["available"] is True, d
assert d["image"]["phase"]=="idle", d
' && ok "Nach Image-Job: qwen verfügbar, phase=idle" || bad "Nach Image-Job"
# --- 26. paralleler Chat während Image-Job (wartet, kein 502) ----------------------------------------
echo "== Test 26: paralleler Chat während Image-Job (wartet, kein 502)"
curl -sf -X POST "$BASE/fast" >/dev/null
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img26.json 2>&1 &
IMG_PID=$!
sleep 1.5
START=$(date +%s)
CODE=$(curl -s -o /tmp/chat26.json -w "%{http_code}" "$BASE/v1/chat/completions" \
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
END=$(date +%s)
ELAPSED=$((END-START))
cat /tmp/chat26.json; echo
wait $IMG_PID
[ "$CODE" = "200" ] && [ "$ELAPSED" -ge 2 ] \
&& ok "Chat wartete ${ELAPSED}s (kein 502), dann 200" || bad "Chat: Code $CODE, ${ELAPSED}s"
# --- Ergebnis --------------------------------------------------------------------------------------------
echo
echo "== Ergebnis: $PASS bestanden, $FAIL fehlgeschlagen =="
[ "$FAIL" -eq 0 ]