#!/usr/bin/env python3 """FLUX.2 [klein] 4B Base – Steps-Vergleich (Qualität vs. Latenz). 1024x1024, fester Seed, cpu_offload. Vergleicht 20/30/40/50 Steps. """ import os import subprocess import time os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1" os.environ["TRANSFORMERS_VERBOSITY"] = "error" os.environ["TOKENIZERS_PARALLELISM"] = "false" MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B" PROMPT = ("A detailed photograph of a red cube on a white marble table, " "soft studio lighting, shallow depth of field") SEED = 0 WIDTH = HEIGHT = 1024 STEPS_LIST = [20, 30, 40, 50] def nvidia_vram() -> int: out = subprocess.check_output( ["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader,nounits"]).decode().strip() return int(out.split()[0]) def main() -> None: import torch print(f"torch {torch.__version__} | {torch.cuda.get_device_name(0)}", flush=True) from diffusers import Flux2KleinPipeline t0 = time.monotonic() pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16) pipe.enable_model_cpu_offload() print(f"[load] ready in {time.monotonic() - t0:.1f} s", flush=True) for steps in STEPS_LIST: out = f"/tmp/flux-steps-{steps}.png" torch.cuda.synchronize() torch.cuda.reset_peak_memory_stats() t = time.monotonic() try: img = pipe( prompt=PROMPT, height=HEIGHT, width=WIDTH, guidance_scale=4.0, num_inference_steps=steps, generator=torch.Generator(device="cuda").manual_seed(SEED), ).images[0] dt = time.monotonic() - t img.save(out) peak = torch.cuda.max_memory_allocated() / 1e9 print(f"[gen] {steps} steps: {dt:.1f} s | peak {peak:.2f} GB | " f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True) except Exception as e: # noqa: BLE001 print(f"[gen] {steps} steps: FEHLER: {e!r}", flush=True) print("DONE", flush=True) if __name__ == "__main__": main()