Spaces:

black-forest-labs
/

FLUX.1-Kontext-Dev

Running on Zero

App Files Files Community

Compiled transformer

#12

by cbensimon HF Staff - opened 10 days ago

base: refs/heads/main

←

from: refs/pr/12

Discussion Files changed

+164

-0

Files changed (3) hide show

app.py +8 -0
optimization.py +60 -0
optimization_utils.py +96 -0

app.py CHANGED Viewed

@@ -1,3 +1,8 @@
 import gradio as gr
 import numpy as np
 import spaces
@@ -8,9 +13,12 @@ from PIL import Image
 from diffusers import FluxKontextPipeline
 from diffusers.utils import load_image
 MAX_SEED = np.iinfo(np.int32).max
 pipe = FluxKontextPipeline.from_pretrained("black-forest-labs/FLUX.1-Kontext-dev", torch_dtype=torch.bfloat16).to("cuda")
 @spaces.GPU
 def infer(input_image, prompt, seed=42, randomize_seed=False, guidance_scale=2.5, steps=28, progress=gr.Progress(track_tqdm=True)):

+# PyTorch 2.8 (temporary hack)
+import os
+os.system('pip install --upgrade --pre --extra-index-url https://download.pytorch.org/whl/nightly/cu126 "torch<2.9" spaces')
+# Actual demo code
 import gradio as gr
 import numpy as np
 import spaces
 from diffusers import FluxKontextPipeline
 from diffusers.utils import load_image
+from optimization import optimize_pipeline_
 MAX_SEED = np.iinfo(np.int32).max
 pipe = FluxKontextPipeline.from_pretrained("black-forest-labs/FLUX.1-Kontext-dev", torch_dtype=torch.bfloat16).to("cuda")
+optimize_pipeline_(pipe, image=Image.new("RGB", (512, 512)), prompt='prompt')
 @spaces.GPU
 def infer(input_image, prompt, seed=42, randomize_seed=False, guidance_scale=2.5, steps=28, progress=gr.Progress(track_tqdm=True)):

optimization.py ADDED Viewed

	@@ -0,0 +1,60 @@

+"""
+"""
+from typing import Any
+from typing import Callable
+from typing import ParamSpec
+import spaces
+import torch
+from torch.utils._pytree import tree_map_only
+from optimization_utils import capture_component_call
+from optimization_utils import aoti_compile
+P = ParamSpec('P')
+TRANSFORMER_HIDDEN_DIM = torch.export.Dim('hidden', min=4096, max=8212)
+TRANSFORMER_DYNAMIC_SHAPES = {
+    'hidden_states': {1: TRANSFORMER_HIDDEN_DIM},
+    'img_ids': {0: TRANSFORMER_HIDDEN_DIM},
+}
+INDUCTOR_CONFIGS = {
+    'conv_1x1_as_mm': True,
+    'epilogue_fusion': False,
+    'coordinate_descent_tuning': True,
+    'coordinate_descent_check_all_directions': True,
+    'max_autotune': True,
+    'triton.cudagraphs': True,
+}
+def optimize_pipeline_(pipeline: Callable[P, Any], *args: P.args, **kwargs: P.kwargs):
+    @spaces.GPU(duration=1500)
+    def compile_transformer():
+        with capture_component_call(pipeline, 'transformer') as call:
+            pipeline(*args, **kwargs)
+        dynamic_shapes = tree_map_only((torch.Tensor, bool), lambda t: None, call.kwargs)
+        dynamic_shapes |= TRANSFORMER_DYNAMIC_SHAPES
+        pipeline.transformer.fuse_qkv_projections()
+        exported = torch.export.export(
+            mod=pipeline.transformer,
+            args=call.args,
+            kwargs=call.kwargs,
+            dynamic_shapes=dynamic_shapes,
+        )
+        return aoti_compile(exported, INDUCTOR_CONFIGS)
+    transformer_config = pipeline.transformer.config
+    pipeline.transformer = compile_transformer()
+    pipeline.transformer.config = transformer_config # pyright: ignore[reportAttributeAccessIssue]

optimization_utils.py ADDED Viewed

	@@ -0,0 +1,96 @@

+"""
+"""
+import contextlib
+from contextvars import ContextVar
+from io import BytesIO
+from typing import Any
+from typing import cast
+from unittest.mock import patch
+import torch
+from torch._inductor.package.package import package_aoti
+from torch.export.pt2_archive._package import AOTICompiledModel
+from torch.export.pt2_archive._package_weights import TensorProperties
+from torch.export.pt2_archive._package_weights import Weights
+INDUCTOR_CONFIGS_OVERRIDES = {
+    'aot_inductor.package_constants_in_so': False,
+    'aot_inductor.package_constants_on_disk': True,
+    'aot_inductor.package': True,
+}
+class ZeroGPUCompiledModel:
+    def __init__(self, archive_file: torch.types.FileLike, weights: Weights, cuda: bool = False):
+        self.archive_file = archive_file
+        self.weights = weights
+        if cuda:
+            self.weights_to_cuda_()
+        self.compiled_model: ContextVar[AOTICompiledModel | None] = ContextVar('compiled_model', default=None)
+    def weights_to_cuda_(self):
+        for name in self.weights:
+            tensor, properties = self.weights.get_weight(name)
+            self.weights[name] = (tensor.to('cuda'), properties)
+    def __call__(self, *args, **kwargs):
+        if (compiled_model := self.compiled_model.get()) is None:
+            constants_map = {name: value[0] for name, value in self.weights.items()}
+            compiled_model = cast(AOTICompiledModel, torch._inductor.aoti_load_package(self.archive_file))
+            compiled_model.load_constants(constants_map, check_full_update=True, user_managed=True)
+            self.compiled_model.set(compiled_model)
+        return compiled_model(*args, **kwargs)
+    def __reduce__(self):
+        weight_dict: dict[str, tuple[torch.Tensor, TensorProperties]] = {}
+        for name in self.weights:
+            tensor, properties = self.weights.get_weight(name)
+            tensor_ = torch.empty_like(tensor, device='cpu').pin_memory()
+            weight_dict[name] = (tensor_.copy_(tensor).detach().share_memory_(), properties)
+        return ZeroGPUCompiledModel, (self.archive_file, Weights(weight_dict), True)
+def aoti_compile(
+    exported_program: torch.export.ExportedProgram,
+    inductor_configs: dict[str, Any] | None = None,
+):
+    inductor_configs = (inductor_configs or {}) | INDUCTOR_CONFIGS_OVERRIDES
+    gm = cast(torch.fx.GraphModule, exported_program.module())
+    assert exported_program.example_inputs is not None
+    args, kwargs = exported_program.example_inputs
+    artifacts = torch._inductor.aot_compile(gm, args, kwargs, options=inductor_configs)
+    archive_file = BytesIO()
+    files: list[str | Weights] = [file for file in artifacts if isinstance(file, str)]
+    package_aoti(archive_file, files)
+    weights, = (artifact for artifact in artifacts if isinstance(artifact, Weights))
+    return ZeroGPUCompiledModel(archive_file, weights)
+@contextlib.contextmanager
+def capture_component_call(
+    pipeline: Any,
+    component_name: str,
+    component_method='forward',
+):
+    class CapturedCallException(Exception):
+        def __init__(self, *args, **kwargs):
+            super().__init__()
+            self.args = args
+            self.kwargs = kwargs
+    class CapturedCall:
+        def __init__(self):
+            self.args: tuple[Any, ...] = ()
+            self.kwargs: dict[str, Any] = {}
+    component = getattr(pipeline, component_name)
+    captured_call = CapturedCall()
+    def capture_call(*args, **kwargs):
+        raise CapturedCallException(*args, **kwargs)
+    with patch.object(component, component_method, new=capture_call):
+        try:
+            yield captured_call
+        except CapturedCallException as e:
+            captured_call.args = e.args
+            captured_call.kwargs = e.kwargs