Qwen-Image-Edit-Fast

Running on Zero

App Files Files Community

multimodalart HF Staff commited on 4 days ago

Commit

d434b56

verified ·

1 Parent(s): 4e85690

Add optimizations (#1)

Browse files

- Add optimizations (f999ccda0036b0e1c33263a084fda2742d7f0fee)
- Create optimization.py (28a5a5501c48df1538fbedce90011acbf627b6b0)
- Update requirements.txt (38bb795e28951db7465262866a358315b317c2f1)

Files changed (3) hide show

app.py +14 -1
optimization.py +70 -0
requirements.txt +3 -1

app.py CHANGED Viewed

@@ -8,9 +8,15 @@ import json
 from PIL import Image
 from diffusers import QwenImageEditPipeline, FlowMatchEulerDiscreteScheduler
 from huggingface_hub import InferenceClient
 import math
 # --- Prompt Enhancement using Hugging Face InferenceClient ---
 def polish_prompt_hf(original_prompt, system_prompt):
     """
@@ -159,7 +165,7 @@ scheduler_config = {
 scheduler = FlowMatchEulerDiscreteScheduler.from_config(scheduler_config)
 # Load the edit pipeline with Lightning scheduler
-pipe = QwenImageEditPipeline.from_pretrained(
     "Qwen/Qwen-Image-Edit",
     scheduler=scheduler,
     torch_dtype=dtype
@@ -177,6 +183,13 @@ except Exception as e:
     print(f"Warning: Could not load Lightning LoRA weights: {e}")
     print("Continuing with base model...")
 # --- UI Constants and Helpers ---
 MAX_SEED = np.iinfo(np.int32).max

 from PIL import Image
 from diffusers import QwenImageEditPipeline, FlowMatchEulerDiscreteScheduler
 from huggingface_hub import InferenceClient
 import math
+from optimization import optimize_pipeline_
+from qwenimage.pipeline_qwen_image_edit import QwenImageEditPipeline as QwenImageEditPipelineCustom
+from qwenimage.transformer_qwenimage import QwenImageTransformer2DModel
+from qwenimage.qwen_fa3_processor import QwenDoubleStreamAttnProcessorFA3
 # --- Prompt Enhancement using Hugging Face InferenceClient ---
 def polish_prompt_hf(original_prompt, system_prompt):
     """
 scheduler = FlowMatchEulerDiscreteScheduler.from_config(scheduler_config)
 # Load the edit pipeline with Lightning scheduler
+pipe = QwenImageEditPipelineCustom.from_pretrained(
     "Qwen/Qwen-Image-Edit",
     scheduler=scheduler,
     torch_dtype=dtype
     print(f"Warning: Could not load Lightning LoRA weights: {e}")
     print("Continuing with base model...")
+# Apply the same optimizations from the first version
+pipe.transformer.__class__ = QwenImageTransformer2DModel
+pipe.transformer.set_attn_processor(QwenDoubleStreamAttnProcessorFA3())
+# --- Ahead-of-time compilation ---
+optimize_pipeline_(pipe, image=Image.new("RGB", (1024, 1024)), prompt="prompt")
 # --- UI Constants and Helpers ---
 MAX_SEED = np.iinfo(np.int32).max

optimization.py ADDED Viewed

	@@ -0,0 +1,70 @@

+"""
+"""
+from typing import Any
+from typing import Callable
+from typing import ParamSpec
+from torchao.quantization import quantize_
+from torchao.quantization import Float8DynamicActivationFloat8WeightConfig
+import spaces
+import torch
+from torch.utils._pytree import tree_map
+P = ParamSpec('P')
+TRANSFORMER_IMAGE_SEQ_LENGTH_DIM = torch.export.Dim('image_seq_length')
+TRANSFORMER_TEXT_SEQ_LENGTH_DIM = torch.export.Dim('text_seq_length')
+TRANSFORMER_DYNAMIC_SHAPES = {
+    'hidden_states': {
+        1: TRANSFORMER_IMAGE_SEQ_LENGTH_DIM,
+    },
+    'encoder_hidden_states': {
+        1: TRANSFORMER_TEXT_SEQ_LENGTH_DIM,
+    },
+    'encoder_hidden_states_mask': {
+        1: TRANSFORMER_TEXT_SEQ_LENGTH_DIM,
+    },
+    'image_rotary_emb': ({
+        0: TRANSFORMER_IMAGE_SEQ_LENGTH_DIM,
+    }, {
+        0: TRANSFORMER_TEXT_SEQ_LENGTH_DIM,
+    }),
+}
+INDUCTOR_CONFIGS = {
+    'conv_1x1_as_mm': True,
+    'epilogue_fusion': False,
+    'coordinate_descent_tuning': True,
+    'coordinate_descent_check_all_directions': True,
+    'max_autotune': True,
+    'triton.cudagraphs': True,
+}
+def optimize_pipeline_(pipeline: Callable[P, Any], *args: P.args, **kwargs: P.kwargs):
+    @spaces.GPU(duration=1500)
+    def compile_transformer():
+        with spaces.aoti_capture(pipeline.transformer) as call:
+            pipeline(*args, **kwargs)
+        dynamic_shapes = tree_map(lambda t: None, call.kwargs)
+        dynamic_shapes |= TRANSFORMER_DYNAMIC_SHAPES
+        # quantize_(pipeline.transformer, Float8DynamicActivationFloat8WeightConfig())
+        exported = torch.export.export(
+            mod=pipeline.transformer,
+            args=call.args,
+            kwargs=call.kwargs,
+            dynamic_shapes=dynamic_shapes,
+        )
+        return spaces.aoti_compile(exported, INDUCTOR_CONFIGS)
+    spaces.aoti_apply(compile_transformer(), pipeline.transformer)

requirements.txt CHANGED Viewed

@@ -1,4 +1,6 @@
-git+https://github.com/huggingface/diffusers.git@7a2b78bf0f788d311cc96b61e660a8e13e3b1e63
 transformers
 accelerate
 safetensors

+git+https://github.com/huggingface/diffusers.git@qwenimage-lru-cache-bypass
+kernels
+torchao==0.11.0
 transformers
 accelerate
 safetensors