Avoid ZeroGPU startup OOM with precompiled Wan AOTI blocks

#3
by evalstate HF Staff - opened
Files changed (1) hide show
  1. app.py +27 -9
app.py CHANGED
@@ -17,8 +17,10 @@ import random
17
  import gc
18
  from gradio_client import Client, handle_file # Import for API call
19
 
20
- # Import the optimization function from the separate file
21
- from optimization import optimize_pipeline_
 
 
22
 
23
  # --- Constants and Model Loading ---
24
  MODEL_ID = "Wan-AI/Wan2.2-I2V-A14B-Diffusers"
@@ -67,13 +69,29 @@ for i in range(3):
67
  torch.cuda.synchronize()
68
  torch.cuda.empty_cache()
69
 
70
- optimize_pipeline_(pipe,
71
- image=Image.new('RGB', (MAX_DIMENSION, MIN_DIMENSION)),
72
- prompt='prompt',
73
- height=MIN_DIMENSION,
74
- width=MAX_DIMENSION,
75
- num_frames=MAX_FRAMES_MODEL,
76
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
  print("All models loaded and optimized. Gradio app is ready.")
78
 
79
 
@@ -333,4 +351,4 @@ with gr.Blocks() as app:
333
  )
334
 
335
  if __name__ == "__main__":
336
- app.launch(mcp_server=True, theme=gr.themes.Citrus(), css=css)
 
17
  import gc
18
  from gradio_client import Client, handle_file # Import for API call
19
 
20
+ # Import the quantization configs used by the precompiled ZeroGPU blocks
21
+ from torchao.quantization import quantize_
22
+ from torchao.quantization import Float8DynamicActivationFloat8WeightConfig
23
+ from torchao.quantization import Int8WeightOnlyConfig
24
 
25
  # --- Constants and Model Loading ---
26
  MODEL_ID = "Wan-AI/Wan2.2-I2V-A14B-Diffusers"
 
69
  torch.cuda.synchronize()
70
  torch.cuda.empty_cache()
71
 
72
+ pipe.load_lora_weights(
73
+ "Kijai/WanVideo_comfy",
74
+ weight_name="Lightx2v/lightx2v_I2V_14B_480p_cfg_step_distill_rank128_bf16.safetensors",
75
+ adapter_name="lightx2v",
 
 
76
  )
77
+ pipe.load_lora_weights(
78
+ "Kijai/WanVideo_comfy",
79
+ weight_name="Lightx2v/lightx2v_I2V_14B_480p_cfg_step_distill_rank128_bf16.safetensors",
80
+ adapter_name="lightx2v_2",
81
+ load_into_transformer_2=True,
82
+ )
83
+ pipe.set_adapters(["lightx2v", "lightx2v_2"], adapter_weights=[1.0, 1.0])
84
+ pipe.fuse_lora(adapter_names=["lightx2v"], lora_scale=3.0, components=["transformer"])
85
+ pipe.fuse_lora(adapter_names=["lightx2v_2"], lora_scale=1.0, components=["transformer_2"])
86
+ pipe.unload_lora_weights()
87
+
88
+ quantize_(pipe.text_encoder, Int8WeightOnlyConfig())
89
+ quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())
90
+ quantize_(pipe.transformer_2, Float8DynamicActivationFloat8WeightConfig())
91
+
92
+ spaces.aoti_blocks_load(pipe.transformer, "zerogpu-aoti/Wan2", variant="fp8da")
93
+ spaces.aoti_blocks_load(pipe.transformer_2, "zerogpu-aoti/Wan2", variant="fp8da")
94
+
95
  print("All models loaded and optimized. Gradio app is ready.")
96
 
97
 
 
351
  )
352
 
353
  if __name__ == "__main__":
354
+ app.launch(mcp_server=True, theme=gr.themes.Citrus(), css=css)