pipe = DiffusionPipeline.from_pretrained( "black-forest-labs/FLUX.1-dev", torch_dtype=torch.bfloat16, ) pipe.enable_model_cpu_offload() # Instead of full compilation, we apply regional compilation # here to take advantage of `fullgraph=True` and also to reduce # compilation time. More details can be found here: # https://hf.co/docs/diffusers/main/en/optimization/fp16#regional-compilation pipe.transformer.compile_repeated_blocks(fullgraph=True) image = pipe(**pipe_kwargs).images[0]