- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality) - quality: standard=30 Steps (Default), high=50 Steps - Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920 - GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen wiederherstellen (exakt vorheriges Profil) - Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock) - Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT - Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness - /status: image.phase, image.worker, image.model_loaded, qwen.available, qwen.active_chats - GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis) - image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 + enable_model_cpu_offload) - deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download, Image-Dir, systemd-Unit mit Image-Umgebungsvariablen - dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps, Quality-Compare), 32 lokale Tests - README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080), Python-Pakete Benchmarks (RTX 5080, 16 GB, CPU-Offload): - 512x512 / 10 Steps: ~9.3 s - 1024x1024 / 30 Steps: ~31.3 s - 1024x1024 / 50 Steps: ~45.3 s - 1920x1088 / 50 Steps: ~91 s - Peak-VRAM: ~8.4-8.9 GB - Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
96 lines
3.2 KiB
Python
96 lines
3.2 KiB
Python
#!/usr/bin/env python3
|
||
"""FLUX.2 [klein] 4B Base – GPU-Resident-Benchmark (OHNE CPU-Offload).
|
||
|
||
Ziel: Prüfen, ob das Modell vollständig auf der RTX 5080 (16 GB) läuft.
|
||
|
||
Messen:
|
||
- from_pretrained-Zeit
|
||
- .to("cuda")-Zeit
|
||
- VRAM (nvidia-smi + torch.cuda.memory_allocated / max_memory_allocated)
|
||
- Generierungszeit, Peak-VRAM pro Auflösung
|
||
|
||
Auflösungen: 512x512 (10 steps) → 1024x1024 (50 steps) → 1920x1088 (50 steps)
|
||
Bei OOM wird abgebrochen (CUDA-Kontext danach nicht mehr verlässlich).
|
||
"""
|
||
|
||
import os
|
||
import subprocess
|
||
import time
|
||
|
||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||
|
||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||
PROMPT = ("A detailed photograph of a red cube on a white marble table, "
|
||
"soft studio lighting, shallow depth of field")
|
||
|
||
# (breite, hoehe, steps, seed, name)
|
||
CASES = [
|
||
(512, 512, 10, 0, "512x512-10s"),
|
||
(1024, 1024, 50, 0, "1024x1024-50s"),
|
||
(1920, 1088, 50, 0, "1920x1088-50s"),
|
||
]
|
||
|
||
|
||
def nvidia_vram() -> int:
|
||
out = subprocess.check_output(
|
||
["nvidia-smi", "--query-gpu=memory.used",
|
||
"--format=csv,noheader,nounits"]).decode().strip()
|
||
return int(out.split()[0])
|
||
|
||
|
||
def main() -> None:
|
||
import torch
|
||
print(f"torch {torch.__version__} | cuda {torch.cuda.is_available()} "
|
||
f"| {torch.cuda.get_device_name(0)}", flush=True)
|
||
from diffusers import Flux2KleinPipeline
|
||
|
||
# --- Laden (zuerst auf CPU, dann vollständig auf GPU) ---
|
||
t0 = time.monotonic()
|
||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||
t_load = time.monotonic() - t0
|
||
print(f"[load] from_pretrained: {t_load:.1f} s", flush=True)
|
||
|
||
t1 = time.monotonic()
|
||
pipe.to("cuda")
|
||
torch.cuda.synchronize()
|
||
t_to = time.monotonic() - t1
|
||
print(f"[load] .to(cuda): {t_to:.1f} s", flush=True)
|
||
print(f"[load] VRAM nvidia-smi: {nvidia_vram()} MiB | "
|
||
f"torch allocated: {torch.cuda.memory_allocated() / 1e9:.2f} GB",
|
||
flush=True)
|
||
|
||
# --- Generierung ---
|
||
for width, height, steps, seed, name in CASES:
|
||
out = f"/tmp/flux-bench-{name}.png"
|
||
torch.cuda.synchronize()
|
||
torch.cuda.reset_peak_memory_stats()
|
||
t = time.monotonic()
|
||
try:
|
||
img = pipe(
|
||
prompt=PROMPT,
|
||
height=height,
|
||
width=width,
|
||
guidance_scale=4.0,
|
||
num_inference_steps=steps,
|
||
generator=torch.Generator(device="cuda").manual_seed(seed),
|
||
).images[0]
|
||
dt = time.monotonic() - t
|
||
img.save(out)
|
||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||
print(f"[gen] {name}: {dt:.1f} s | peak torch {peak:.2f} GB | "
|
||
f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True)
|
||
except Exception as e: # noqa: BLE001
|
||
dt = time.monotonic() - t
|
||
print(f"[gen] {name}: FEHLER nach {dt:.1f} s: {e!r}", flush=True)
|
||
if "out of memory" in str(e).lower():
|
||
print("[gen] OOM – Abbruch, größere Auflösungen nicht getestet",
|
||
flush=True)
|
||
break
|
||
print("DONE", flush=True)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|