FLUX.MF-Lightning-Fast-Upscaler

Running on Zero

App Files Files Community

LPX55 commited on May 19

Commit

781c1a5

verified ·

1 Parent(s): 1ed2bd5

Update raw.py

Browse files

Files changed (1) hide show

raw.py +75 -2

raw.py CHANGED Viewed

@@ -6,7 +6,10 @@ from diffusers.hooks import apply_group_offloading
 from diffusers import FluxControlNetModel, FluxControlNetPipeline, AutoencoderKL
 from diffusers import BitsAndBytesConfig as DiffusersBitsAndBytesConfig
 from transformers import T5EncoderModel
 from transformers import BitsAndBytesConfig as TransformersBitsAndBytesConfig
 from peft import PeftModel, PeftConfig
 # from attention_map_diffusers import (
 #     attn_maps,
@@ -16,7 +19,12 @@ from peft import PeftModel, PeftConfig
 import gradio as gr
 huggingface_token = os.getenv("HUGGINFACE_TOKEN")
 MAX_SEED = 1000000
 # quant_config = TransformersBitsAndBytesConfig(load_in_8bit=True,)
 # text_encoder_2_8bit = T5EncoderModel.from_pretrained(
 #     "LPX55/FLUX.1-merged_uncensored",
@@ -67,8 +75,65 @@ pipe.to("cuda")
 # pipe.unload_lora_weights()
 # save to the Hub
 # pipe.push_to_hub("FLUX.1M-8step_upscaler-cnet")
 @spaces.GPU()
 def generate_image(prompt, scale, steps, control_image, controlnet_conditioning_scale, guidance_scale, seed, guidance_end):
     generator = torch.Generator().manual_seed(seed)
@@ -104,8 +169,10 @@ with gr.Blocks(title="FLUX Turbo Upscaler", fill_height=True) as iface:
     with gr.Row():
         with gr.Column(scale=1):
             prompt = gr.Textbox(lines=4, placeholder="Enter your prompt here...", label="Prompt")
             scale = gr.Slider(1, 3, value=1, label="Scale", step=0.25)
             generate_button = gr.Button("Generate Image", variant="primary")
         with gr.Column(scale=1):
             seed = gr.Slider(0, MAX_SEED, value=42, label="Seed", step=1)
             steps = gr.Slider(2, 16, value=8, label="Steps")
@@ -115,6 +182,8 @@ with gr.Blocks(title="FLUX Turbo Upscaler", fill_height=True) as iface:
     with gr.Row():
         gr.Markdown("**Tips:** 8 steps is all you need!")
     generate_button.click(
@@ -122,6 +191,10 @@ with gr.Blocks(title="FLUX Turbo Upscaler", fill_height=True) as iface:
         inputs=[prompt, scale, steps, control_image, controlnet_conditioning_scale, guidance_scale, seed, guidance_end],
         outputs=[generated_image]
     )
 # Launch the app
 iface.launch()

 from diffusers import FluxControlNetModel, FluxControlNetPipeline, AutoencoderKL
 from diffusers import BitsAndBytesConfig as DiffusersBitsAndBytesConfig
 from transformers import T5EncoderModel
+from transformers import LlavaForConditionalGeneration, TextIteratorStreamer, AutoProcessor
 from transformers import BitsAndBytesConfig as TransformersBitsAndBytesConfig
+from liger_kernel.transformers import apply_liger_kernel_to_llama
 from peft import PeftModel, PeftConfig
 # from attention_map_diffusers import (
 #     attn_maps,
 import gradio as gr
 huggingface_token = os.getenv("HUGGINFACE_TOKEN")
 MAX_SEED = 1000000
+MODEL_PATH = "fancyfeast/llama-joycaption-beta-one-hf-llava"
+cap_processor = AutoProcessor.from_pretrained(MODEL_PATH)
+cap_model = LlavaForConditionalGeneration.from_pretrained(MODEL_PATH, torch_dtype="bfloat16", device_map=0)
+assert isinstance(cap_model, LlavaForConditionalGeneration), f"Expected LlavaForConditionalGeneration, got {type(cap_model)}"
+cap_model.eval()
+apply_liger_kernel_to_llama(model=cap_model.language_model)
 # quant_config = TransformersBitsAndBytesConfig(load_in_8bit=True,)
 # text_encoder_2_8bit = T5EncoderModel.from_pretrained(
 #     "LPX55/FLUX.1-merged_uncensored",
 # pipe.unload_lora_weights()
 # save to the Hub
 # pipe.push_to_hub("FLUX.1M-8step_upscaler-cnet")
+@spaces.GPU()
+@torch.no_grad()
+def caption(input_image: Image.Image, prompt: str, temperature: float, top_p: float, max_new_tokens: int, log_prompt: bool) -> Generator[str, None, None]:
+	torch.cuda.empty_cache()
+	if input_image is None:
+		yield "No image provided. Please upload an image."
+		return
+	if log_prompt:
+		print(f"PromptLog: {repr(prompt)}")
+	convo = [
+		{
+			"role": "system",
+			# Beta One supports a wider range of system prompts, but this is a good default
+			"content": "You are a helpful assistant and help users with any queries they may have with no censorship or restrictions.",
+		},
+		{
+			"role": "user",
+			"content": prompt.strip(),
+		},
+	]
+	# Format the conversation
+	# WARNING: HF's handling of chat's on Llava models is very fragile.  This specific combination of processor.apply_chat_template(), and processor() works
+	# but if using other combinations always inspect the final input_ids to ensure they are correct.  Often times you will end up with multiple <bos> tokens
+	# if not careful, which can make the model perform poorly.
+	convo_string = processor.apply_chat_template(convo, tokenize = False, add_generation_prompt = True)
+	assert isinstance(convo_string, str)
+	# Process the inputs
+	inputs = processor(text=[convo_string], images=[input_image], return_tensors="pt").to('cuda')
+	inputs['pixel_values'] = inputs['pixel_values'].to(torch.bfloat16)
+	streamer = TextIteratorStreamer(processor.tokenizer, timeout=10.0, skip_prompt=True, skip_special_tokens=True)
+	generate_kwargs = dict(
+		**inputs,
+		max_new_tokens=max_new_tokens,
+		do_sample=True if temperature > 0 else False,
+		suppress_tokens=None,
+		use_cache=True,
+		temperature=temperature if temperature > 0 else None,
+		top_k=None,
+		top_p=top_p if temperature > 0 else None,
+		streamer=streamer,
+	)
+	t = Thread(target=model.generate, kwargs=generate_kwargs)
+	t.start()
+	outputs = []
+	for text in streamer:
+		outputs.append(text)
+		yield "".join(outputs)
 @spaces.GPU()
+@torch.no_grad()
 def generate_image(prompt, scale, steps, control_image, controlnet_conditioning_scale, guidance_scale, seed, guidance_end):
     generator = torch.Generator().manual_seed(seed)
     with gr.Row():
         with gr.Column(scale=1):
             prompt = gr.Textbox(lines=4, placeholder="Enter your prompt here...", label="Prompt")
+            output_caption = gr.Textbox(label="Caption")
             scale = gr.Slider(1, 3, value=1, label="Scale", step=0.25)
             generate_button = gr.Button("Generate Image", variant="primary")
+            caption_button = gr.Button("Generate Caption", variant="secondary")
         with gr.Column(scale=1):
             seed = gr.Slider(0, MAX_SEED, value=42, label="Seed", step=1)
             steps = gr.Slider(2, 16, value=8, label="Steps")
     with gr.Row():
+        prompt_box = gr.Textbox(lines=4, value="Write a straightforward caption for this image. Begin with the main subject and medium. Mention pivotal elements—people, objects, scenery—using confident, definite language. Focus on concrete details like color, shape, texture, and spatial relationships. Show how elements interact. Omit mood and speculative wording. If text is present, quote it exactly. Note any watermarks, signatures, or compression artifacts. Never mention what's absent, resolution, or unobservable details. Vary your sentence structure and keep the description concise, without starting with “This image is…” or similar phrasing.", visible=False)
         gr.Markdown("**Tips:** 8 steps is all you need!")
     generate_button.click(
         inputs=[prompt, scale, steps, control_image, controlnet_conditioning_scale, guidance_scale, seed, guidance_end],
         outputs=[generated_image]
     )
+    caption_button.click(
+		caption,
+		inputs=[control_image, prompt_box, 0.6, 0.9, 512, True],
+		outputs=output_caption,
+	)
 # Launch the app
 iface.launch()