router: Bildgenerierung mit FLUX.2 [klein] 4B Base (GPU-Hotswap)
- POST /v1/images/generations (OpenAI-kompatibel, prompt/size/n/seed/quality) - quality: standard=30 Steps (Default), high=50 Steps - Größen: 1024x1024, 1536x1024, 1024x1536, 1920x1088, 1088x1920 - GPU-Hotswap: Qwen stoppen -> FLUX laden -> Bild -> FLUX entladen -> Qwen wiederherstellen (exakt vorheriges Profil) - Zentrales GPU/Modell-Lock (Profilwechsel und Bild teilen sich das Lock) - Chat-Requests warten während Bild-Job (kein 502), Timeout CHAT_WAIT_TIMEOUT - Robuste Recovery: try/finally, Worker-Beendigung, VRAM-Check, Qwen-Readiness - /status: image.phase, image.worker, image.model_loaded, qwen.available, qwen.active_chats - GET /images, GET /images/<datei> (validiert, nur images/-Verzeichnis) - image_worker.py: FLUX-Worker (eigener Prozess, JSON-Protokoll, bf16 + enable_model_cpu_offload) - deploy: venv (torch/diffusers/transformers/accelerate), Modell-Download, Image-Dir, systemd-Unit mit Image-Umgebungsvariablen - dev: Mock-Worker, fake-systemctl, Benchmarks (GPU-Resident, Offload, Steps, Quality-Compare), 32 lokale Tests - README: Bildgenerierung, Hotswap, Recovery, Benchmarks (RTX 5080), Python-Pakete Benchmarks (RTX 5080, 16 GB, CPU-Offload): - 512x512 / 10 Steps: ~9.3 s - 1024x1024 / 30 Steps: ~31.3 s - 1024x1024 / 50 Steps: ~45.3 s - 1920x1088 / 50 Steps: ~91 s - Peak-VRAM: ~8.4-8.9 GB - Hotswap-Gesamtzeit: ~41-42 s (1024x1024 / 30 Steps)
This commit is contained in:
@@ -0,0 +1,92 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FLUX.2 [klein] 4B Base – Qualitätsvergleich 30 vs. 50 Steps.
|
||||
|
||||
Identischer Prompt, identischer Seed, identische Parameter – nur
|
||||
num_inference_steps variiert (30 vs. 50). CPU-Offload, Base-Modell,
|
||||
keine anderen Änderungen.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1"
|
||||
os.environ["TRANSFORMERS_VERBOSITY"] = "error"
|
||||
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
||||
|
||||
MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B"
|
||||
SEED = 42
|
||||
WIDTH = HEIGHT = 1024
|
||||
GUIDANCE = 4.0
|
||||
|
||||
PROMPT = (
|
||||
"Ultra-realistic cinematic photograph of a woman in her early thirties "
|
||||
"sitting at a small outdoor café table in a rainy European city at night. "
|
||||
"Natural detailed skin texture with pores and subtle imperfections, "
|
||||
"realistic eyes and individual strands of wet hair, both hands clearly "
|
||||
"visible holding a ceramic coffee cup with anatomically correct fingers. "
|
||||
"She wears a dark wool coat over a finely textured knitted sweater. "
|
||||
"Raindrops on the table and glass surfaces, wet pavement reflecting warm "
|
||||
"café lights and cool blue street lighting, realistic depth of field, "
|
||||
"pedestrians and bicycles in the detailed background, complex reflections "
|
||||
"in windows and puddles. On the café window behind her is a clearly "
|
||||
"readable handwritten sign saying exactly: 'CAFÉ LUMIÈRE – OPEN UNTIL "
|
||||
"MIDNIGHT'. A small newspaper lies on the table with the clearly readable "
|
||||
"headline 'BERLIN AFTER DARK'. Photorealistic professional full-frame "
|
||||
"camera photograph, natural color grading, physically plausible lighting, "
|
||||
"realistic materials, fine micro-detail, no plastic skin, no illustration, "
|
||||
"no CGI look."
|
||||
)
|
||||
|
||||
CASES = [
|
||||
(30, "/tmp/flux-quality-30.png"),
|
||||
(50, "/tmp/flux-quality-50.png"),
|
||||
]
|
||||
|
||||
|
||||
def nvidia_vram() -> int:
|
||||
out = subprocess.check_output(
|
||||
["nvidia-smi", "--query-gpu=memory.used",
|
||||
"--format=csv,noheader,nounits"]).decode().strip()
|
||||
return int(out.split()[0])
|
||||
|
||||
|
||||
def main() -> None:
|
||||
import torch
|
||||
print(f"torch {torch.__version__} | {torch.cuda.get_device_name(0)}",
|
||||
flush=True)
|
||||
print(f"SEED={SEED} | {WIDTH}x{HEIGHT} | guidance={GUIDANCE}", flush=True)
|
||||
print(f"PROMPT={PROMPT!r}", flush=True)
|
||||
from diffusers import Flux2KleinPipeline
|
||||
|
||||
t0 = time.monotonic()
|
||||
pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
print(f"[load] ready in {time.monotonic() - t0:.1f} s", flush=True)
|
||||
|
||||
for steps, out in CASES:
|
||||
torch.cuda.synchronize()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
t = time.monotonic()
|
||||
try:
|
||||
img = pipe(
|
||||
prompt=PROMPT,
|
||||
height=HEIGHT,
|
||||
width=WIDTH,
|
||||
guidance_scale=GUIDANCE,
|
||||
num_inference_steps=steps,
|
||||
generator=torch.Generator(device="cuda").manual_seed(SEED),
|
||||
).images[0]
|
||||
dt = time.monotonic() - t
|
||||
img.save(out)
|
||||
peak = torch.cuda.max_memory_allocated() / 1e9
|
||||
print(f"[gen] {steps} steps: {dt:.1f} s | peak {peak:.2f} GB | "
|
||||
f"nvidia-smi {nvidia_vram()} MiB | seed={SEED} | {out}",
|
||||
flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[gen] {steps} steps: FEHLER: {e!r}", flush=True)
|
||||
print("DONE", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user