Duke86Syl

Paused

App Files Files Community

John6666 commited on Sep 25

Commit

534feb5

verified ·

1 Parent(s): 4aea187

Upload 4 files

Browse files

Files changed (4) hide show

README.md +1 -1
app.py +23 -2
optimization.py +42 -0
requirements.txt +5 -2

README.md CHANGED Viewed

@@ -4,7 +4,7 @@ emoji: 🖼
 colorFrom: purple
 colorTo: red
 sdk: gradio
-sdk_version: 5.23.2
 app_file: app.py
 license: apache-2.0
 pinned: false

 colorFrom: purple
 colorTo: red
 sdk: gradio
+sdk_version: 5.47.1
 app_file: app.py
 license: apache-2.0
 pinned: false

app.py CHANGED Viewed

@@ -4,6 +4,7 @@ import random
 import torch
 import spaces
 import re
 from diffusers import (
     DiffusionPipeline,
     AutoencoderTiny,
@@ -11,6 +12,15 @@ from diffusers import (
 from huggingface_hub import hf_hub_download
 #from feifeilib.feifeichat import feifeichat
 import config
 styles_name = [style["name"] for style in config.style_list]
 MAX_SEED = np.iinfo(np.int32).max
@@ -49,7 +59,19 @@ def feifeimodload():
 pipe = feifeimodload()
-@spaces.GPU()
 def infer(prompt="",  styles_Radio="(None)", feife_select = False, bigboobs_select = True, seed=42, randomize_seed=False, width=1024, height=1024, num_inference_steps=4, guidancescale=3.5, num_feifei=0.35, nsfw_select=False, nsfw_slider=1, progress=gr.Progress(track_tqdm=True)):
     Duke86Syl_lora_name=[]
@@ -232,7 +254,6 @@ with gr.Blocks(css=css) as demo:
                     step=0.05,
                     value=0.35,
                 )
         gr.Examples(
             examples = examples,

 import torch
 import spaces
 import re
+import os
 from diffusers import (
     DiffusionPipeline,
     AutoencoderTiny,
 from huggingface_hub import hf_hub_download
 #from feifeilib.feifeichat import feifeichat
+IS_ZERO_GPU = bool(os.getenv("SPACES_ZERO_GPU"))
+IS_GPU_MODE = True if IS_ZERO_GPU else (True if torch.cuda.is_available() else False)
+if IS_ZERO_GPU:
+    import subprocess
+    subprocess.run("rm -rf /data-nvme/zerogpu-offload/*", env={}, shell=True)
+    torch.set_float32_matmul_precision("high")
+    torch.backends.cuda.matmul.allow_tf32 = True
+IS_COMPILE = False
 import config
 styles_name = [style["name"] for style in config.style_list]
 MAX_SEED = np.iinfo(np.int32).max
 pipe = feifeimodload()
+if IS_ZERO_GPU:
+    os.environ["DIFFUSERS_ENABLE_HUB_KERNELS"] = "yes"
+    pipe.transformer.set_attention_backend("flash_hub")
+    if IS_COMPILE:
+        from optimization import optimize_pipeline_
+        optimize_pipeline_(pipe, "prompt")
+def get_duration(prompt, styles_Radio, feife_select, bigboobs_select, seed, randomize_seed, width, height, num_inference_steps, guidancescale, num_feifei, nsfw_select, nsfw_slider, progress):
+    def_duration = 15.
+    def_steps = 4.
+    return int(def_duration * float(num_inference_steps) / def_steps)
+@spaces.GPU(duration=get_duration)
 def infer(prompt="",  styles_Radio="(None)", feife_select = False, bigboobs_select = True, seed=42, randomize_seed=False, width=1024, height=1024, num_inference_steps=4, guidancescale=3.5, num_feifei=0.35, nsfw_select=False, nsfw_slider=1, progress=gr.Progress(track_tqdm=True)):
     Duke86Syl_lora_name=[]
                     step=0.05,
                     value=0.35,
                 )
         gr.Examples(
             examples = examples,

optimization.py ADDED Viewed

	@@ -0,0 +1,42 @@

+"""
+"""
+from typing import Any
+from typing import Callable
+from typing import ParamSpec
+import spaces
+import torch
+P = ParamSpec('P')
+INDUCTOR_CONFIGS = {
+    'conv_1x1_as_mm': True,
+    'epilogue_fusion': False,
+    'coordinate_descent_tuning': True,
+    'coordinate_descent_check_all_directions': True,
+    'max_autotune': True,
+    'triton.cudagraphs': True,
+}
+def optimize_pipeline_(pipeline: Callable[P, Any], *args: P.args, **kwargs: P.kwargs):
+    @spaces.GPU(duration=1500)
+    def compile_transformer():
+        with spaces.aoti_capture(pipeline.transformer) as call:
+            pipeline(*args, **kwargs)
+        exported = torch.export.export(
+            mod=pipeline.transformer,
+            args=call.args,
+            kwargs=call.kwargs,
+        )
+        return spaces.aoti_compile(exported, INDUCTOR_CONFIGS)
+    pipeline.transformer.fuse_qkv_projections()
+    pipeline.transformer.set_attention_backend("flash_hub")
+    spaces.aoti_apply(compile_transformer(), pipeline.transformer)

requirements.txt CHANGED Viewed

@@ -3,7 +3,7 @@ gradio
 mistralai
 requests
 accelerate
-git+https://github.com/huggingface/diffusers.git
 invisible_watermark
 torch
 xformers
@@ -13,4 +13,7 @@ peft
 psutil
 gradio_client
 spaces
-openai

 mistralai
 requests
 accelerate
+git+https://github.com/huggingface/diffusers@fa-hub
 invisible_watermark
 torch
 xformers
 psutil
 gradio_client
 spaces
+openai
+kernels
+huggingface_hub[hf_xet]
+pydantic==2.10.6