router: Bildgenerierung mit FLUX.2 [klein] 4B Base (GPU-Hotswap)
- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality) - quality: standard=30 Steps (Default), high=50 Steps - Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920 - GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen wiederherstellen (exakt vorheriges Profil) - Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock) - Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT - Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness - /status: image.phase, image.worker, image.model_loaded, qwen.available, qwen.active_chats - GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis) - image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 + enable_model_cpu_offload) - deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download, Image-Dir, systemd-Unit mit Image-Umgebungsvariablen - dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps, Quality-Compare), 32 lokale Tests - README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080), Python-Pakete Benchmarks (RTX 5080, 16 GB, CPU-Offload): - 512x512 / 10 Steps: ~9.3 s - 1024x1024 / 30 Steps: ~31.3 s - 1024x1024 / 50 Steps: ~45.3 s - 1920x1088 / 50 Steps: ~91 s - Peak-VRAM: ~8.4-8.9 GB - Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
This commit is contained in:
1 parent
c5d92acd93
commit
7c5bbe2ffb
15 files changed
+1796
-68
No files matched your search
Executable
+49
@@ -0,0 +1,49 @@
|
||||
#!/bin/bash
|
||||
# Fake systemctl für lokale Tests: verwaltet den Mock-llama.cpp-Prozess.
|
||||
# Simuliert: systemctl stop|start|status <service>
|
||||
#
|
||||
# Umgebungsvariablen (vom Router geerbt):
|
||||
# FAKE_SYSTEMD_PIDFILE PID-Datei des Mocks (Default /tmp/mock_upstream_pid)
|
||||
# FAKE_SYSTEMD_PORT Mock-Port (Default 18080)
|
||||
# FAKE_SYSTEMD_PROFILE_DIR Profil-Dir für den Mock
|
||||
# FAKE_SYSTEMD_MOCK Mock-Skript (Default dev/mock_upstream.py)
|
||||
# FAKE_SYSTEMD_LOG Log-Datei (Default /tmp/mock_upstream_fake.log)
|
||||
|
||||
CMD="${1:-}"
|
||||
PIDFILE="${FAKE_SYSTEMD_PIDFILE:-/tmp/mock_upstream_pid}"
|
||||
PORT="${FAKE_SYSTEMD_PORT:-18080}"
|
||||
PROFILE_DIR="${FAKE_SYSTEMD_PROFILE_DIR:-}"
|
||||
MOCK="${FAKE_SYSTEMD_MOCK:-dev/mock_upstream.py}"
|
||||
LOG="${FAKE_SYSTEMD_LOG:-/tmp/mock_upstream_fake.log}"
|
||||
|
||||
is_running() {
|
||||
[ -f "$PIDFILE" ] && kill -0 "$(cat "$PIDFILE")" 2>/dev/null
|
||||
}
|
||||
|
||||
case "$CMD" in
|
||||
stop)
|
||||
if is_running; then
|
||||
kill "$(cat "$PIDFILE")" 2>/dev/null || true
|
||||
rm -f "$PIDFILE"
|
||||
for _ in $(seq 1 50); do
|
||||
is_running || break
|
||||
sleep 0.1
|
||||
done
|
||||
fi
|
||||
exit 0
|
||||
;;
|
||||
start)
|
||||
if ! is_running; then
|
||||
MOCK_PROFILE_DIR="$PROFILE_DIR" MOCK_PORT="$PORT" \
|
||||
python3 "$MOCK" >>"$LOG" 2>&1 &
|
||||
echo $! > "$PIDFILE"
|
||||
fi
|
||||
exit 0
|
||||
;;
|
||||
status)
|
||||
is_running && exit 0 || exit 3
|
||||
;;
|
||||
*)
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,66 @@
|
||||
#!/bin/bash
|
||||
# FLUX-Benchmark-Wrapper mit GARANTIERTER Qwen-Recovery.
|
||||
#
|
||||
# Usage: flux_benchmark_run.sh <python-script>
|
||||
#
|
||||
# Ablauf:
|
||||
# 1. Aktives Qwen-Profil aus override.conf merken (bytegenauer Vergleich)
|
||||
# 2. llama.cpp stoppen, VRAM-Abgabe verifizieren
|
||||
# 3. Benchmark-Skript ausführen
|
||||
# 4. IMMER (trap EXIT): llama.cpp wieder starten, Readiness prüfen
|
||||
set -u
|
||||
|
||||
SERVICE=mike-ai-llama-ui.service
|
||||
PROFILE_DIR=/etc/systemd/system/mike-ai-llama-ui.service.d
|
||||
PY=/opt/mike-ai/ai-profile-router/venv/bin/python
|
||||
SCRIPT="${1:?Usage: flux_benchmark_run.sh <python-script>}"
|
||||
ROUTER_STATUS=http://127.0.0.1:8081/status
|
||||
|
||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||
# vergleichen (robuster als String-Matching).
|
||||
PROFILE=""
|
||||
for p in fast medium long; do
|
||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||
PROFILE=$p
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -z "$PROFILE" ]; then
|
||||
echo "FEHLER: aktives Profil nicht erkannt (override.conf passt zu keinem Profil-File)"
|
||||
exit 1
|
||||
fi
|
||||
echo "=== Aktives Qwen-Profil: $PROFILE (wird nach dem Test wiederhergestellt) ==="
|
||||
|
||||
restore() {
|
||||
echo
|
||||
echo "=== RECOVERY: stelle Profil $PROFILE wieder her ==="
|
||||
systemctl start "$SERVICE" 2>&1 || echo "systemctl start fehlgeschlagen"
|
||||
for i in $(seq 1 150); do
|
||||
if curl -sf "$ROUTER_STATUS" 2>/dev/null | python3 -c '
|
||||
import json, sys
|
||||
d = json.load(sys.stdin)
|
||||
u = d["upstream"]
|
||||
assert u["reachable"] and u["model"], d
|
||||
print("ready:", u["model"], "ctx", u["ctx"])
|
||||
' 2>/dev/null; then
|
||||
echo "=== RECOVERY OK: llama.cpp ist wieder inference-ready ==="
|
||||
return 0
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
echo "=== RECOVERY FEHLGESCHLAGEN: bitte manuell prüfen (systemctl status $SERVICE) ==="
|
||||
return 1
|
||||
}
|
||||
trap restore EXIT
|
||||
|
||||
echo "=== Stoppe $SERVICE ==="
|
||||
systemctl stop "$SERVICE"
|
||||
sleep 3
|
||||
echo "=== VRAM nach Stop (MiB) ==="
|
||||
nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits
|
||||
|
||||
echo "=== Starte Benchmark: $SCRIPT ==="
|
||||
"$PY" "$SCRIPT"
|
||||
RC=$?
|
||||
echo "=== Benchmark-Exit-Code: $RC ==="
|
||||
exit $RC
|
||||
@@ -0,0 +1,95 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FLUX.2 [klein] 4B Base – GPU-Resident-Benchmark (OHNE CPU-Offload).
|
||||
|
||||
Ziel: Prüfen, ob das Modell vollständig auf der RTX 5080 (16 GB) läuft.
|
||||
|
||||
Messen:
|
||||
- from_pretrained-Zeit
|
||||
- .to("cuda")-Zeit
|
||||
- VRAM (nvidia-smi + torch.cuda.memory_allocated / max_memory_allocated)
|
||||
- Generierungszeit, Peak-VRAM pro Auflösung
|
||||
|
||||
Auflösungen: 512x512 (10 steps) → 1024x1024 (50 steps) → 1920x1088 (50 steps)
|
||||
Bei OOM wird abgebrochen (CUDA-Kontext danach nicht mehr verlässlich).
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||
|
||||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||||
PROMPT = ("A detailed photograph of a red cube on a white marble table, "
|
||||
"soft studio lighting, shallow depth of field")
|
||||
|
||||
# (breite, hoehe, steps, seed, name)
|
||||
CASES = [
|
||||
(512, 512, 10, 0, "512x512-10s"),
|
||||
(1024, 1024, 50, 0, "1024x1024-50s"),
|
||||
(1920, 1088, 50, 0, "1920x1088-50s"),
|
||||
]
|
||||
|
||||
|
||||
def nvidia_vram() -> int:
|
||||
out = subprocess.check_output(
|
||||
["nvidia-smi", "--query-gpu=memory.used",
|
||||
"--format=csv,noheader,nounits"]).decode().strip()
|
||||
return int(out.split()[0])
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import torch
|
||||
print(f"torch {torch.__version__} | cuda {torch.cuda.is_available()} "
|
||||
f"| {torch.cuda.get_device_name(0)}", flush=True)
|
||||
from diffusers import Flux2KleinPipeline
|
||||
|
||||
# --- Laden (zuerst auf CPU, dann vollständig auf GPU) ---
|
||||
t0 = time.monotonic()
|
||||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||||
t_load = time.monotonic() - t0
|
||||
print(f"[load] from_pretrained: {t_load:.1f} s", flush=True)
|
||||
|
||||
t1 = time.monotonic()
|
||||
pipe.to("cuda")
|
||||
torch.cuda.synchronize()
|
||||
t_to = time.monotonic() - t1
|
||||
print(f"[load] .to(cuda): {t_to:.1f} s", flush=True)
|
||||
print(f"[load] VRAM nvidia-smi: {nvidia_vram()} MiB | "
|
||||
f"torch allocated: {torch.cuda.memory_allocated() / 1e9:.2f} GB",
|
||||
flush=True)
|
||||
|
||||
# --- Generierung ---
|
||||
for width, height, steps, seed, name in CASES:
|
||||
out = f"/tmp/flux-bench-{name}.png"
|
||||
torch.cuda.synchronize()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
t = time.monotonic()
|
||||
try:
|
||||
img = pipe(
|
||||
prompt=PROMPT,
|
||||
height=height,
|
||||
width=width,
|
||||
guidance_scale=4.0,
|
||||
num_inference_steps=steps,
|
||||
generator=torch.Generator(device="cuda").manual_seed(seed),
|
||||
).images[0]
|
||||
dt = time.monotonic() - t
|
||||
img.save(out)
|
||||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||||
print(f"[gen] {name}: {dt:.1f} s | peak torch {peak:.2f} GB | "
|
||||
f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
dt = time.monotonic() - t
|
||||
print(f"[gen] {name}: FEHLER nach {dt:.1f} s: {e!r}", flush=True)
|
||||
if "out of memory" in str(e).lower():
|
||||
print("[gen] OOM – Abbruch, größere Auflösungen nicht getestet",
|
||||
flush=True)
|
||||
break
|
||||
print("DONE", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,66 @@
|
||||
#!/bin/bash
|
||||
# FLUX GPU-Resident-Benchmark mit GARANTIERTER Qwen-Recovery.
|
||||
#
|
||||
# Ablauf:
|
||||
# 1. Aktives Qwen-Profil aus override.conf merken
|
||||
# 2. llama.cpp stoppen, VRAM-Abgabe verifizieren
|
||||
# 3. Benchmark ausführen (Python, GPU-resident, ohne CPU-Offload)
|
||||
# 4. IMMER (trap EXIT): llama.cpp wieder starten, Readiness prüfen
|
||||
#
|
||||
# Usage: flux_gpu_benchmark.sh
|
||||
set -u
|
||||
|
||||
SERVICE=mike-ai-llama-ui.service
|
||||
PROFILE_DIR=/etc/systemd/system/mike-ai-llama-ui.service.d
|
||||
PY=/opt/mike-ai/ai-profile-router/venv/bin/python
|
||||
SCRIPT="$(cd "$(dirname "$0")" && pwd)/flux_gpu_benchmark.py"
|
||||
ROUTER_STATUS=http://127.0.0.1:8081/status
|
||||
|
||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||
# vergleichen (robuster als String-Matching).
|
||||
PROFILE=""
|
||||
for p in fast medium long; do
|
||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||
PROFILE=$p
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ -z "$PROFILE" ]; then
|
||||
echo "FEHLER: aktives Profil nicht erkannt (override.conf passt zu keinem Profil-File)"
|
||||
exit 1
|
||||
fi
|
||||
echo "=== Aktives Qwen-Profil: $PROFILE (wird nach dem Test wiederhergestellt) ==="
|
||||
|
||||
restore() {
|
||||
echo
|
||||
echo "=== RECOVERY: stelle Profil $PROFILE wieder her ==="
|
||||
systemctl start "$SERVICE" 2>&1 || echo "systemctl start fehlgeschlagen"
|
||||
for i in $(seq 1 150); do
|
||||
if curl -sf "$ROUTER_STATUS" 2>/dev/null | python3 -c '
|
||||
import json, sys
|
||||
d = json.load(sys.stdin)
|
||||
u = d["upstream"]
|
||||
assert u["reachable"] and u["model"], d
|
||||
print("ready:", u["model"], "ctx", u["ctx"])
|
||||
' 2>/dev/null; then
|
||||
echo "=== RECOVERY OK: llama.cpp ist wieder inference-ready ==="
|
||||
return 0
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
echo "=== RECOVERY FEHLGESCHLAGEN: bitte manuell prüfen (systemctl status $SERVICE) ==="
|
||||
return 1
|
||||
}
|
||||
trap restore EXIT
|
||||
|
||||
echo "=== Stoppe $SERVICE ==="
|
||||
systemctl stop "$SERVICE"
|
||||
sleep 3
|
||||
echo "=== VRAM nach Stop (MiB) ==="
|
||||
nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits
|
||||
|
||||
echo "=== Starte Benchmark ==="
|
||||
"$PY" "$SCRIPT"
|
||||
RC=$?
|
||||
echo "=== Benchmark-Exit-Code: $RC ==="
|
||||
exit $RC
|
||||
@@ -0,0 +1,83 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FLUX.2 [klein] 4B Base – Benchmark MIT enable_model_cpu_offload().
|
||||
|
||||
Offizieller Pfad der Modellkarte ("runs on consumer hardware, with as
|
||||
little as 13GB VRAM"). Keine Qualitätsreduktion – nur langsamer
|
||||
(Weights wandern pro Layer zwischen CPU und GPU).
|
||||
|
||||
Messen: Load-Zeit, Generierungszeit, Peak-VRAM pro Auflösung.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||
|
||||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||||
PROMPT = ("A detailed photograph of a red cube on a white marble table, "
|
||||
"soft studio lighting, shallow depth of field")
|
||||
|
||||
CASES = [
|
||||
(512, 512, 10, 0, "512x512-10s"),
|
||||
(1024, 1024, 50, 0, "1024x1024-50s"),
|
||||
(1920, 1088, 50, 0, "1920x1088-50s"),
|
||||
]
|
||||
|
||||
|
||||
def nvidia_vram() -> int:
|
||||
out = subprocess.check_output(
|
||||
["nvidia-smi", "--query-gpu=memory.used",
|
||||
"--format=csv,noheader,nounits"]).decode().strip()
|
||||
return int(out.split()[0])
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import torch
|
||||
print(f"torch {torch.__version__} | cuda {torch.cuda.is_available()} "
|
||||
f"| {torch.cuda.get_device_name(0)}", flush=True)
|
||||
from diffusers import Flux2KleinPipeline
|
||||
|
||||
t0 = time.monotonic()
|
||||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||||
t_load = time.monotonic() - t0
|
||||
print(f"[load] from_pretrained: {t_load:.1f} s", flush=True)
|
||||
|
||||
t1 = time.monotonic()
|
||||
pipe.enable_model_cpu_offload()
|
||||
t_off = time.monotonic() - t1
|
||||
print(f"[load] enable_model_cpu_offload: {t_off:.1f} s", flush=True)
|
||||
print(f"[load] VRAM nvidia-smi (idle): {nvidia_vram()} MiB", flush=True)
|
||||
|
||||
for width, height, steps, seed, name in CASES:
|
||||
out = f"/tmp/flux-bench-offload-{name}.png"
|
||||
torch.cuda.synchronize()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
t = time.monotonic()
|
||||
try:
|
||||
img = pipe(
|
||||
prompt=PROMPT,
|
||||
height=height,
|
||||
width=width,
|
||||
guidance_scale=4.0,
|
||||
num_inference_steps=steps,
|
||||
generator=torch.Generator(device="cuda").manual_seed(seed),
|
||||
).images[0]
|
||||
dt = time.monotonic() - t
|
||||
img.save(out)
|
||||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||||
print(f"[gen] {name}: {dt:.1f} s | peak torch {peak:.2f} GB | "
|
||||
f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
dt = time.monotonic() - t
|
||||
print(f"[gen] {name}: FEHLER nach {dt:.1f} s: {e!r}", flush=True)
|
||||
if "out of memory" in str(e).lower():
|
||||
print("[gen] OOM – Abbruch", flush=True)
|
||||
break
|
||||
print("DONE", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,92 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FLUX.2 [klein] 4B Base – Qualitätsvergleich 30 vs. 50 Steps.
|
||||
|
||||
Identischer Prompt, identischer Seed, identische Parameter – nur
|
||||
num_inference_steps variiert (30 vs. 50). CPU-Offload, Base-Modell,
|
||||
keine anderen Änderungen.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||
|
||||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||||
SEED = 42
|
||||
WIDTH = HEIGHT = 1024
|
||||
GUIDANCE = 4.0
|
||||
|
||||
PROMPT = (
|
||||
"Ultra-realistic cinematic photograph of a woman in her early thirties "
|
||||
"sitting at a small outdoor café table in a rainy European city at night. "
|
||||
"Natural detailed skin texture with pores and subtle imperfections, "
|
||||
"realistic eyes and individual strands of wet hair, both hands clearly "
|
||||
"visible holding a ceramic coffee cup with anatomically correct fingers. "
|
||||
"She wears a dark wool coat over a finely textured knitted sweater. "
|
||||
"Raindrops on the table and glass surfaces, wet pavement reflecting warm "
|
||||
"café lights and cool blue street lighting, realistic depth of field, "
|
||||
"pedestrians and bicycles in the detailed background, complex reflections "
|
||||
"in windows and puddles. On the café window behind her is a clearly "
|
||||
"readable handwritten sign saying exactly: 'CAFÉ LUMIÈRE – OPEN UNTIL "
|
||||
"MIDNIGHT'. A small newspaper lies on the table with the clearly readable "
|
||||
"headline 'BERLIN AFTER DARK'. Photorealistic professional full-frame "
|
||||
"camera photograph, natural color grading, physically plausible lighting, "
|
||||
"realistic materials, fine micro-detail, no plastic skin, no illustration, "
|
||||
"no CGI look."
|
||||
)
|
||||
|
||||
CASES = [
|
||||
(30, "/tmp/flux-quality-30.png"),
|
||||
(50, "/tmp/flux-quality-50.png"),
|
||||
]
|
||||
|
||||
|
||||
def nvidia_vram() -> int:
|
||||
out = subprocess.check_output(
|
||||
["nvidia-smi", "--query-gpu=memory.used",
|
||||
"--format=csv,noheader,nounits"]).decode().strip()
|
||||
return int(out.split()[0])
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import torch
|
||||
print(f"torch {torch.__version__} | {torch.cuda.get_device_name(0)}",
|
||||
flush=True)
|
||||
print(f"SEED={SEED} | {WIDTH}x{HEIGHT} | guidance={GUIDANCE}", flush=True)
|
||||
print(f"PROMPT={PROMPT!r}", flush=True)
|
||||
from diffusers import Flux2KleinPipeline
|
||||
|
||||
t0 = time.monotonic()
|
||||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
print(f"[load] ready in {time.monotonic() - t0:.1f} s", flush=True)
|
||||
|
||||
for steps, out in CASES:
|
||||
torch.cuda.synchronize()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
t = time.monotonic()
|
||||
try:
|
||||
img = pipe(
|
||||
prompt=PROMPT,
|
||||
height=HEIGHT,
|
||||
width=WIDTH,
|
||||
guidance_scale=GUIDANCE,
|
||||
num_inference_steps=steps,
|
||||
generator=torch.Generator(device="cuda").manual_seed(SEED),
|
||||
).images[0]
|
||||
dt = time.monotonic() - t
|
||||
img.save(out)
|
||||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||||
print(f"[gen] {steps} steps: {dt:.1f} s | peak {peak:.2f} GB | "
|
||||
f"nvidia-smi {nvidia_vram()} MiB | seed={SEED} | {out}",
|
||||
flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[gen] {steps} steps: FEHLER: {e!r}", flush=True)
|
||||
print("DONE", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,66 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FLUX.2 [klein] 4B Base – Steps-Vergleich (Qualität vs. Latenz).
|
||||
|
||||
1024x1024, fester Seed, cpu_offload. Vergleicht 20/30/40/50 Steps.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||
|
||||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||||
PROMPT = ("A detailed photograph of a red cube on a white marble table, "
|
||||
"soft studio lighting, shallow depth of field")
|
||||
SEED = 0
|
||||
WIDTH = HEIGHT = 1024
|
||||
STEPS_LIST = [20, 30, 40, 50]
|
||||
|
||||
|
||||
def nvidia_vram() -> int:
|
||||
out = subprocess.check_output(
|
||||
["nvidia-smi", "--query-gpu=memory.used",
|
||||
"--format=csv,noheader,nounits"]).decode().strip()
|
||||
return int(out.split()[0])
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import torch
|
||||
print(f"torch {torch.__version__} | {torch.cuda.get_device_name(0)}",
|
||||
flush=True)
|
||||
from diffusers import Flux2KleinPipeline
|
||||
|
||||
t0 = time.monotonic()
|
||||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
print(f"[load] ready in {time.monotonic() - t0:.1f} s", flush=True)
|
||||
|
||||
for steps in STEPS_LIST:
|
||||
out = f"/tmp/flux-steps-{steps}.png"
|
||||
torch.cuda.synchronize()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
t = time.monotonic()
|
||||
try:
|
||||
img = pipe(
|
||||
prompt=PROMPT,
|
||||
height=HEIGHT,
|
||||
width=WIDTH,
|
||||
guidance_scale=4.0,
|
||||
num_inference_steps=steps,
|
||||
generator=torch.Generator(device="cuda").manual_seed(SEED),
|
||||
).images[0]
|
||||
dt = time.monotonic() - t
|
||||
img.save(out)
|
||||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||||
print(f"[gen] {steps} steps: {dt:.1f} s | peak {peak:.2f} GB | "
|
||||
f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[gen] {steps} steps: FEHLER: {e!r}", flush=True)
|
||||
print("DONE", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,90 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Mock-Bild-Worker für lokale Tests (gleiche Protokoll wie image_worker.py).
|
||||
|
||||
Erzeugt ein minimales 1x1-PNG statt eines echten Bildes.
|
||||
|
||||
Optionen (Umgebungsvariablen):
|
||||
MOCK_WORKER_DELAY Sekunden, die pro generate geschlafen werden
|
||||
(Default 0.3). Für Tests von parallelen Requests.
|
||||
MOCK_WORKER_LOG Datei, in die die Requests geloggt werden (JSON-Zeilen).
|
||||
Für Tests, die die Steps/Qualität prüfen wollen.
|
||||
|
||||
Sonder-Prompts:
|
||||
"FAIL" -> Worker antwortet mit Fehler (simuliert OOM/Crash).
|
||||
"SLOW" -> Worker schläft 5 s (für Parallel-Tests).
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Minimales 1x1-PNG (1 Byte rot)
|
||||
PNG_1x1 = bytes.fromhex(
|
||||
"89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c4"
|
||||
"890000000d49444154789c626001000000ffff03000006000557bfabd40000"
|
||||
"000049454e44ae426082"
|
||||
)
|
||||
|
||||
DELAY = float(os.environ.get("MOCK_WORKER_DELAY", "0.3"))
|
||||
LOG_FILE = os.environ.get("MOCK_WORKER_LOG", "")
|
||||
|
||||
|
||||
def _emit(payload: dict) -> None:
|
||||
sys.stdout.write(json.dumps(payload) + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def _log_request(req: dict) -> None:
|
||||
if not LOG_FILE:
|
||||
return
|
||||
try:
|
||||
with open(LOG_FILE, "a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(req) + "\n")
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def main() -> None:
|
||||
loaded = False
|
||||
_emit({"status": "ready"})
|
||||
for line in sys.stdin:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
req = json.loads(line)
|
||||
except ValueError:
|
||||
_emit({"status": "error", "message": "ungültiges JSON"})
|
||||
continue
|
||||
cmd = req.get("cmd")
|
||||
if cmd == "generate":
|
||||
_log_request(req)
|
||||
prompt = req.get("prompt", "")
|
||||
if prompt == "SLOW":
|
||||
time.sleep(5.0) # langsame Generierung (Parallel-Tests)
|
||||
else:
|
||||
time.sleep(DELAY) # simulierte Generierung
|
||||
if prompt == "FAIL":
|
||||
_emit({"status": "error",
|
||||
"message": "simulierter Fehler (OOM)"})
|
||||
continue
|
||||
output = req["output"]
|
||||
os.makedirs(os.path.dirname(output) or ".", exist_ok=True)
|
||||
with open(output, "wb") as f:
|
||||
f.write(PNG_1x1)
|
||||
_emit({"status": "ok", "path": output, "seconds": DELAY,
|
||||
"load_seconds": 0.1})
|
||||
loaded = True
|
||||
elif cmd == "unload":
|
||||
loaded = False
|
||||
_emit({"status": "ok"})
|
||||
elif cmd == "status":
|
||||
_emit({"status": "ok", "model_loaded": loaded})
|
||||
else:
|
||||
_emit({"status": "error",
|
||||
"message": f"unbekanntes Kommando: {cmd}"})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+234
-13
@@ -13,7 +13,7 @@ FAIL=0
|
||||
|
||||
cleanup() {
|
||||
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" 2>/dev/null || true
|
||||
rm -f /tmp/mock_pid2
|
||||
rm -f /tmp/mock_pid2 /tmp/mock_upstream_pid
|
||||
wait 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT
|
||||
@@ -21,22 +21,41 @@ trap cleanup EXIT
|
||||
ok() { echo " PASS: $1"; PASS=$((PASS+1)); }
|
||||
bad() { echo " FAIL: $1"; FAIL=$((FAIL+1)); }
|
||||
|
||||
# --- Mock-llama.cpp starten --------------------------------------------------
|
||||
# --- Mock-llama.cpp starten (über Fake-systemctl) ------------------------------
|
||||
echo "== Starte Mock-llama.cpp (Port $UP_PORT)"
|
||||
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream.log 2>&1 &
|
||||
MOCK_PID=$!
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
|
||||
FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
|
||||
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream.log \
|
||||
bash dev/fake-systemctl.sh start
|
||||
sleep 0.5
|
||||
MOCK_PID=$(cat /tmp/mock_upstream_pid 2>/dev/null || echo "")
|
||||
|
||||
# --- Router starten -----------------------------------------------------------
|
||||
echo "== Starte Router (Port $RT_PORT)"
|
||||
rm -rf /tmp/test-images
|
||||
ROUTER_HOST=127.0.0.1 ROUTER_PORT="$RT_PORT" \
|
||||
UPSTREAM_URL="http://127.0.0.1:$UP_PORT" \
|
||||
PROFILE_SCRIPT="$PWD/dev/fake-llama-profile.sh" \
|
||||
PROFILE_DIR="$FAKE_DIR" \
|
||||
SWITCH_TIMEOUT=30 \
|
||||
SYSTEMCTL_BIN="$PWD/dev/fake-systemctl.sh" \
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid \
|
||||
FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" \
|
||||
FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream_fake.log \
|
||||
IMAGE_WORKER="$PWD/dev/mock_image_worker.py" \
|
||||
IMAGE_PYTHON=python3 \
|
||||
IMAGE_DIR=/tmp/test-images \
|
||||
IMAGE_WORKER_LOG=/tmp/test_worker.log \
|
||||
IMAGE_GEN_TIMEOUT=30 \
|
||||
MOCK_WORKER_LOG=/tmp/test_worker_requests.jsonl \
|
||||
python3 router/ai_profile_router.py >/tmp/router_test.log 2>&1 &
|
||||
ROUTER_PID=$!
|
||||
sleep 0.5
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
|
||||
# --- 1. /v1/models -------------------------------------------------------------
|
||||
echo "== Test 1: /v1/models"
|
||||
@@ -150,30 +169,232 @@ cat /tmp/err9b.json; echo
|
||||
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
kill "$MOCK_PID" 2>/dev/null; wait "$MOCK_PID" 2>/dev/null || true
|
||||
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
bash dev/fake-systemctl.sh stop
|
||||
sleep 0.5
|
||||
CODE=$(curl -s -o /tmp/err10.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[]}')
|
||||
cat /tmp/err10.json; echo
|
||||
[ "$CODE" = "502" ] && ok "502 bei downem Upstream (Profil bereits aktiv)" || bad "erwartet 502, bekam $CODE"
|
||||
|
||||
# Mock nach ~3 s neu starten (simuliert systemctl restart durch das Profil-Skript)
|
||||
rm -f /tmp/mock_pid2
|
||||
(
|
||||
sleep 3
|
||||
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream2.log 2>&1 &
|
||||
echo $! > /tmp/mock_pid2
|
||||
) &
|
||||
# Mock neu starten (simuliert systemctl restart durch das Profil-Skript)
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
FAKE_SYSTEMD_PROFILE_DIR="$FAKE_DIR" FAKE_SYSTEMD_MOCK="$PWD/dev/mock_upstream.py" \
|
||||
FAKE_SYSTEMD_LOG=/tmp/mock_upstream2.log \
|
||||
bash dev/fake-systemctl.sh start
|
||||
RESP=$(curl -sf -X POST "$BASE/fast")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
# neuen Mock als MOCK_PID übernehmen, damit Cleanup ihn beendet
|
||||
[ -f /tmp/mock_pid2 ] && MOCK_PID=$(cat /tmp/mock_pid2)
|
||||
[ -f /tmp/mock_upstream_pid ] && MOCK_PID=$(cat /tmp/mock_upstream_pid)
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["profile"]=="fast" and d["model"]=="mock-model-73728", d
|
||||
' && ok "Recovery: /fast wartet auf Upstream, dann Erfolg" || bad "Recovery"
|
||||
|
||||
# --- 11. Bildgenerierung (Mock-Worker) -------------------------------------------------
|
||||
echo "== Test 11: POST /v1/images/generations (1024x1024)"
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"ein rotes Haus","size":"1024x1024"}')
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
IMG_NAME=$(echo "$RESP" | python3 -c 'import json,sys; print(json.load(sys.stdin)["data"][0]["url"].rsplit("/",1)[1])')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert len(d["data"])==1, d
|
||||
assert d["data"][0]["url"].startswith("http://"), d
|
||||
' && [ -f "/tmp/test-images/$IMG_NAME" ] \
|
||||
&& ok "Bild generiert und gespeichert ($IMG_NAME)" || bad "Bildgenerierung"
|
||||
|
||||
# --- 12. Bild-Download -----------------------------------------------------------------
|
||||
echo "== Test 12: GET /images/<datei>"
|
||||
CODE=$(curl -s -o /tmp/test_dl.png -w "%{http_code}" -D /tmp/hdr12.txt "$BASE/images/$IMG_NAME")
|
||||
CTYPE=$(grep -i content-type /tmp/hdr12.txt | tr -d "\r")
|
||||
[ "$CODE" = "200" ] && [ -s /tmp/test_dl.png ] && echo "$CTYPE" | grep -qi "image/png" \
|
||||
&& ok "PNG-Download (200, $CTYPE)" || bad "PNG-Download (Code $CODE, $CTYPE)"
|
||||
|
||||
# --- 13. Bild-Liste ---------------------------------------------------------------------
|
||||
echo "== Test 13: GET /images"
|
||||
RESP=$(curl -sf "$BASE/images")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c "
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
names=[i['name'] for i in d['images']]
|
||||
assert '$IMG_NAME' in names, names
|
||||
" && ok "Bild in Liste enthalten" || bad "Bild-Liste"
|
||||
|
||||
# --- 14. Validierung ---------------------------------------------------------------------
|
||||
echo "== Test 14: Validierung (Größe, Prompt, n)"
|
||||
CODE=$(curl -s -o /tmp/err14a.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","size":"500x500"}')
|
||||
cat /tmp/err14a.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültiger Größe" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err14b.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"size":"1024x1024"}')
|
||||
cat /tmp/err14b.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei fehlendem Prompt" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err14c.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","n":9}')
|
||||
cat /tmp/err14c.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei n=9 (max 4)" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
# --- 15. b64_json + n=2 + Seed -------------------------------------------------------------
|
||||
echo "== Test 15: response_format=b64_json, n=2, seed"
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"zwei Bilder","size":"1024x1024","n":2,"seed":42,"response_format":"b64_json"}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys,base64
|
||||
d=json.load(sys.stdin)
|
||||
assert len(d["data"])==2, d
|
||||
for item in d["data"]:
|
||||
assert item["url"] is None, item
|
||||
png=base64.b64decode(item["b64_json"])
|
||||
assert png[:4]==b"\x89PNG", "kein PNG"
|
||||
' && ok "2 Bilder als b64_json (gültige PNGs)" || bad "b64_json/n=2"
|
||||
|
||||
# --- 16. /status zeigt Bild-Zustand ---------------------------------------------------------
|
||||
echo "== Test 16: /status mit Bild-Section"
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
img=d["image"]
|
||||
assert img["phase"]=="idle", img
|
||||
assert img["worker"]=="stopped", img # Worker wird nach Job beendet
|
||||
assert img["model_loaded"] is False, img
|
||||
assert img["last_image"], img
|
||||
assert img["last_error"] is None, img
|
||||
q=d["qwen"]
|
||||
assert q["available"] is True, q
|
||||
assert q["active_chats"]==0, q
|
||||
' && ok "Status: phase=idle, worker=stopped, qwen verfügbar" || bad "Status Bild-Section"
|
||||
|
||||
# --- 17. Qwen nach Bildgenerierung erreichbar -------------------------------------------------
|
||||
echo "== Test 17: Qwen nach Bildgenerierung erreichbar"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
|
||||
' && ok "Chat funktioniert nach Bildgenerierung" || bad "Chat nach Bild"
|
||||
|
||||
# --- 18. quality=standard → 30 Steps -------------------------------------------------------------
|
||||
echo "== Test 18: quality=standard → 30 Steps"
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"standard test","size":"1024x1024","quality":"standard"}')
|
||||
sleep 0.3
|
||||
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
|
||||
[ "$STEPS" = "30" ] && ok "quality=standard → 30 Steps" || bad "erwartet 30 Steps, bekam $STEPS"
|
||||
|
||||
# --- 19. quality=high → 50 Steps -------------------------------------------------------------------
|
||||
echo "== Test 19: quality=high → 50 Steps"
|
||||
rm -f /tmp/test_worker_requests.jsonl
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"high test","size":"1024x1024","quality":"high"}')
|
||||
sleep 0.3
|
||||
STEPS=$(tail -1 /tmp/test_worker_requests.jsonl 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin)["steps"])' 2>/dev/null || echo "?")
|
||||
[ "$STEPS" = "50" ] && ok "quality=high → 50 Steps" || bad "erwartet 50 Steps, bekam $STEPS"
|
||||
|
||||
# --- 20. ungültige Qualität → 400 ------------------------------------------------------------------
|
||||
echo "== Test 20: ungültige Qualität → 400"
|
||||
CODE=$(curl -s -o /tmp/err20.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"x","quality":"bogus"}')
|
||||
cat /tmp/err20.json; echo
|
||||
[ "$CODE" = "400" ] && ok "400 bei ungültiger Qualität" || bad "erwartet 400, bekam $CODE"
|
||||
|
||||
# --- 21. Image-Fehler → Qwen wiederhergestellt ------------------------------------------------------
|
||||
echo "== Test 21: Image-Fehler → Qwen wiederhergestellt"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
CODE=$(curl -s -o /tmp/err21.json -w "%{http_code}" "$BASE/v1/images/generations" \
|
||||
-H "Content-Type: application/json" -d '{"prompt":"FAIL","size":"1024x1024"}')
|
||||
cat /tmp/err21.json; echo
|
||||
sleep 0.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["current_profile"]=="fast", d
|
||||
assert d["upstream"]["reachable"] is True, d
|
||||
assert d["qwen"]["available"] is True, d
|
||||
' && ok "Qwen nach Image-Fehler wiederhergestellt (fast, erreichbar)" || bad "Qwen nicht wiederhergestellt"
|
||||
|
||||
# --- 22. Fast → Image → Fast ------------------------------------------------------------------------
|
||||
echo "== Test 22: Fast → Image → Fast"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"fast test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "fast" ] && ok "Fast → Image → Fast" || bad "Profil nach Image: $PROFILE (erwartet fast)"
|
||||
|
||||
# --- 23. Medium → Image → Medium --------------------------------------------------------------------
|
||||
echo "== Test 23: Medium → Image → Medium"
|
||||
curl -sf -X POST "$BASE/medium" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"medium test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
||||
|
||||
# --- 24. Long → Image → Long ------------------------------------------------------------------------
|
||||
echo "== Test 24: Long → Image → Long"
|
||||
curl -sf -X POST "$BASE/long" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"long test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
|
||||
|
||||
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
||||
echo "== Test 25: /status während Image-Job"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img25.json 2>&1 &
|
||||
IMG_PID=$!
|
||||
sleep 1.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
img=d["image"]
|
||||
assert img["phase"]!="idle", img
|
||||
assert d["qwen"]["available"] is False, d
|
||||
' && ok "Status während Image-Job: phase!=idle, qwen unavailable" || bad "Status während Image-Job"
|
||||
wait $IMG_PID
|
||||
sleep 0.5
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["qwen"]["available"] is True, d
|
||||
assert d["image"]["phase"]=="idle", d
|
||||
' && ok "Nach Image-Job: qwen verfügbar, phase=idle" || bad "Nach Image-Job"
|
||||
|
||||
# --- 26. paralleler Chat während Image-Job (wartet, kein 502) ----------------------------------------
|
||||
echo "== Test 26: paralleler Chat während Image-Job (wartet, kein 502)"
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"SLOW","size":"1024x1024"}' >/tmp/img26.json 2>&1 &
|
||||
IMG_PID=$!
|
||||
sleep 1.5
|
||||
START=$(date +%s)
|
||||
CODE=$(curl -s -o /tmp/chat26.json -w "%{http_code}" "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
END=$(date +%s)
|
||||
ELAPSED=$((END-START))
|
||||
cat /tmp/chat26.json; echo
|
||||
wait $IMG_PID
|
||||
[ "$CODE" = "200" ] && [ "$ELAPSED" -ge 2 ] \
|
||||
&& ok "Chat wartete ${ELAPSED}s (kein 502), dann 200" || bad "Chat: Code $CODE, ${ELAPSED}s"
|
||||
|
||||
# --- Ergebnis --------------------------------------------------------------------------------------------
|
||||
echo
|
||||
echo "== Ergebnis: $PASS bestanden, $FAIL fehlgeschlagen =="
|
||||
|
||||
Reference in new issue
Block a user