#!/usr/bin/env python3 """FLUX.2 [klein] 4B Base – Benchmark MIT enable_model_cpu_offload(). Offizieller Pfad der Modellkarte ("runs on consumer hardware, with as little as 13GB VRAM"). Keine Qualitätsreduktion – nur langsamer (Weights wandern pro Layer zwischen CPU und GPU). Messen: Load-Zeit, Generierungszeit, Peak-VRAM pro Auflösung. """ import os import subprocess import time os.environ["HF_HUB_DISABLE_PROGRESS_BARS"] = "1" os.environ["TRANSFORMERS_VERBOSITY"] = "error" os.environ["TOKENIZERS_PARALLELISM"] = "false" MODEL = "/opt/mike-ai/models/FLUX.2-klein-base-4B" PROMPT = ("A detailed photograph of a red cube on a white marble table, " "soft studio lighting, shallow depth of field") CASES = [ (512, 512, 10, 0, "512x512-10s"), (1024, 1024, 50, 0, "1024x1024-50s"), (1920, 1088, 50, 0, "1920x1088-50s"), ] def nvidia_vram() -> int: out = subprocess.check_output( ["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader,nounits"]).decode().strip() return int(out.split()[0]) def main() -> None: import torch print(f"torch {torch.__version__} | cuda {torch.cuda.is_available()} " f"| {torch.cuda.get_device_name(0)}", flush=True) from diffusers import Flux2KleinPipeline t0 = time.monotonic() pipe = Flux2KleinPipeline.from_pretrained(MODEL, torch_dtype=torch.bfloat16) t_load = time.monotonic() - t0 print(f"[load] from_pretrained: {t_load:.1f} s", flush=True) t1 = time.monotonic() pipe.enable_model_cpu_offload() t_off = time.monotonic() - t1 print(f"[load] enable_model_cpu_offload: {t_off:.1f} s", flush=True) print(f"[load] VRAM nvidia-smi (idle): {nvidia_vram()} MiB", flush=True) for width, height, steps, seed, name in CASES: out = f"/tmp/flux-bench-offload-{name}.png" torch.cuda.synchronize() torch.cuda.reset_peak_memory_stats() t = time.monotonic() try: img = pipe( prompt=PROMPT, height=height, width=width, guidance_scale=4.0, num_inference_steps=steps, generator=torch.Generator(device="cuda").manual_seed(seed), ).images[0] dt = time.monotonic() - t img.save(out) peak = torch.cuda.max_memory_allocated() / 1e9 print(f"[gen] {name}: {dt:.1f} s | peak torch {peak:.2f} GB | " f"nvidia-smi {nvidia_vram()} MiB | {out}", flush=True) except Exception as e: # noqa: BLE001 dt = time.monotonic() - t print(f"[gen] {name}: FEHLER nach {dt:.1f} s: {e!r}", flush=True) if "out of memory" in str(e).lower(): print("[gen] OOM – Abbruch", flush=True) break print("DONE", flush=True) if __name__ == "__main__": main()