FLUX.1-Kontext-Dev

Running on Zero

App Files Files Community

cbensimon HF Staff commited on Jul 15

Commit

3df4fd5

1 Parent(s): 6631fab

Compilation

Browse files

Files changed (3) hide show

app.py +3 -0
optimization.py +54 -0
zerogpu.py +62 -0

app.py CHANGED Viewed

@@ -8,9 +8,12 @@ from PIL import Image
 from diffusers import FluxKontextPipeline
 from diffusers.utils import load_image
 MAX_SEED = np.iinfo(np.int32).max
 pipe = FluxKontextPipeline.from_pretrained("black-forest-labs/FLUX.1-Kontext-dev", torch_dtype=torch.bfloat16).to("cuda")
 @spaces.GPU
 def infer(input_image, prompt, seed=42, randomize_seed=False, guidance_scale=2.5, steps=28, progress=gr.Progress(track_tqdm=True)):

 from diffusers import FluxKontextPipeline
 from diffusers.utils import load_image
+from optimization import optimize_pipeline_
 MAX_SEED = np.iinfo(np.int32).max
 pipe = FluxKontextPipeline.from_pretrained("black-forest-labs/FLUX.1-Kontext-dev", torch_dtype=torch.bfloat16).to("cuda")
+optimize_pipeline_(pipe)
 @spaces.GPU
 def infer(input_image, prompt, seed=42, randomize_seed=False, guidance_scale=2.5, steps=28, progress=gr.Progress(track_tqdm=True)):

optimization.py ADDED Viewed

	@@ -0,0 +1,54 @@

+"""
+"""
+import spaces
+import torch
+from diffusers.pipelines.flux.pipeline_flux import FluxPipeline
+from torchao.quantization import quantize_
+from torchao.quantization import Float8DynamicActivationFloat8WeightConfig
+from zerogpu import aoti_compile
+def optimize_pipeline_(pipeline: FluxPipeline):
+    @spaces.GPU(duration=1500)
+    def compile_transformer():
+        pipeline.transformer.fuse_qkv_projections()
+        quantize_(pipeline.transformer, Float8DynamicActivationFloat8WeightConfig())
+        def _example_tensor(*shape):
+            return torch.randn(*shape, device='cuda', dtype=torch.bfloat16)
+        is_timestep_distilled = not pipeline.transformer.config.guidance_embeds
+        seq_length = 256 if is_timestep_distilled else 512
+        transformer_kwargs = {
+            'hidden_states': _example_tensor(1, 4096, 64),
+            'timestep': torch.tensor([1.], device='cuda', dtype=torch.bfloat16),
+            'guidance': None if is_timestep_distilled else torch.tensor([1.], device='cuda', dtype=torch.bfloat16),
+            'pooled_projections': _example_tensor(1, 768),
+            'encoder_hidden_states': _example_tensor(1, seq_length, 4096),
+            'txt_ids': _example_tensor(seq_length, 3),
+            'img_ids': _example_tensor(4096, 3),
+            'joint_attention_kwargs': {},
+            'return_dict': False,
+        }
+        inductor_configs = {
+            'conv_1x1_as_mm': True,
+            'epilogue_fusion': False,
+            'coordinate_descent_tuning': True,
+            'coordinate_descent_check_all_directions': True,
+            'max_autotune': True,
+            'triton.cudagraphs': True,
+        }
+        exported = torch.export.export(pipeline.transformer, args=(), kwargs=transformer_kwargs)
+        return aoti_compile(exported, inductor_configs)
+    transformer_config = pipeline.transformer.config
+    pipeline.transformer = compile_transformer()
+    pipeline.transformer.config = transformer_config

zerogpu.py ADDED Viewed

	@@ -0,0 +1,62 @@

+"""
+"""
+from contextvars import ContextVar
+from io import BytesIO
+from typing import Any
+from typing import cast
+import torch
+from torch._inductor.package.package import package_aoti
+from torch.export.pt2_archive._package import AOTICompiledModel
+from torch.export.pt2_archive._package_weights import TensorProperties
+from torch.export.pt2_archive._package_weights import Weights
+INDUCTOR_CONFIGS_OVERRIDES = {
+    'aot_inductor.package_constants_in_so': False,
+    'aot_inductor.package_constants_on_disk': True,
+    'aot_inductor.package': True,
+}
+class ZeroGPUCompiledModel:
+    def __init__(self, archive_file: torch.types.FileLike, weights: Weights, cuda: bool = False):
+        self.archive_file = archive_file
+        self.weights = weights
+        if cuda:
+            self.weights_to_cuda_()
+        self.compiled_model: ContextVar[AOTICompiledModel | None] = ContextVar('compiled_model', default=None)
+    def weights_to_cuda_(self):
+        for name in self.weights:
+            tensor, properties = self.weights.get_weight(name)
+            self.weights[name] = (tensor.to('cuda'), properties)
+    def __call__(self, *args, **kwargs):
+        if (compiled_model := self.compiled_model.get()) is None:
+            constants_map = {name: value[0] for name, value in self.weights.items()}
+            compiled_model = cast(AOTICompiledModel, torch._inductor.aoti_load_package(self.archive_file))
+            compiled_model.load_constants(constants_map, check_full_update=True, user_managed=True)
+            self.compiled_model.set(compiled_model)
+        return compiled_model(*args, **kwargs)
+    def __reduce__(self):
+        weight_dict: dict[str, tuple[torch.Tensor, TensorProperties]] = {}
+        for name in self.weights:
+            tensor, properties = self.weights.get_weight(name)
+            tensor_ = torch.empty_like(tensor, device='cpu').pin_memory()
+            weight_dict[name] = (tensor_.copy_(tensor).detach().share_memory_(), properties)
+        return ZeroGPUCompiledModel, (self.archive_file, Weights(weight_dict), True)
+def aoti_compile(
+    exported_program: torch.export.ExportedProgram,
+    inductor_configs: dict[str, Any] | None = None,
+):
+    inductor_configs = (inductor_configs or {}) | INDUCTOR_CONFIGS_OVERRIDES
+    gm = exported_program.module()
+    assert exported_program.example_inputs is not None
+    args, kwargs = exported_program.example_inputs
+    artifacts = torch._inductor.aot_compile(gm, args, kwargs, options=inductor_configs)
+    archive_file = BytesIO()
+    files = [file for file in artifacts if isinstance(file, str)]
+    package_aoti(archive_file, files)
+    weights, = (artifact for artifact in artifacts if isinstance(artifact, Weights))
+    return ZeroGPUCompiledModel(archive_file, weights)