import os import sys # --- Patch diffusers before import --- # Find diffusers path without importing it pipeline_file = None for path in sys.path: candidate = os.path.join(path, "diffusers", "pipelines", "qwenimage", "pipeline_qwenimage_layered.py") if os.path.exists(candidate): pipeline_file = candidate break if pipeline_file: with open(pipeline_file, "r", encoding="utf-8") as f: content = f.read() # Remove the line that removes the first frame original_line = "latents = latents[:, :, 1:] # remove the first frame as it is the orgin input" patched_line = "# latents = latents[:, :, 1:] # remove the first frame as it is the orgin input (PATCHED)" if original_line in content: content = content.replace(original_line, patched_line) with open(pipeline_file, "w", encoding="utf-8") as f: f.write(content) print(f"Patched: {pipeline_file}") else: print(f"Already patched or line not found: {pipeline_file}") else: print("diffusers pipeline file not found") import uuid import numpy as np import random import tempfile import spaces import zipfile from PIL import Image from diffusers import QwenImageLayeredPipeline import torch from pptx import Presentation import gradio as gr LOG_DIR = "/tmp/local" MAX_SEED = np.iinfo(np.int32).max # --- Environment Variables for LoRA --- LORA_REPO = os.environ.get("LORA_REPO", "") LORA_PATH_IN_REPO = os.environ.get("LORA_PATH_IN_REPO", "") LORA_WEIGHT_DEFAULT = float(os.environ.get("LORA_WEIGHT_DEFAULT", "1.0")) # --- Environment Variables for Prompts --- DEFAULT_PROMPT = os.environ.get("DEFAULT_PROMPT", "") DEFAULT_NEG_PROMPT = os.environ.get("DEFAULT_NEG_PROMPT", " ") DEFAULT_LAYERS = int(os.environ.get("DEFAULT_LAYERS", "4")) DEFAULT_STEPS = int(os.environ.get("DEFAULT_STEPS", "20")) # Debug: Print environment variables print(f"=== Environment Variables ===") print(f"LORA_REPO: {LORA_REPO}") print(f"LORA_PATH_IN_REPO: {LORA_PATH_IN_REPO}") print(f"LORA_WEIGHT_DEFAULT: {LORA_WEIGHT_DEFAULT}") print(f"DEFAULT_PROMPT: {DEFAULT_PROMPT}") print(f"DEFAULT_NEG_PROMPT: {DEFAULT_NEG_PROMPT}") print(f"DEFAULT_LAYERS: {DEFAULT_LAYERS}") print(f"DEFAULT_STEPS: {DEFAULT_STEPS}") from huggingface_hub import login login(token=os.environ.get('HF_TOKEN')) dtype = torch.bfloat16 device = "cuda" if torch.cuda.is_available() else "cpu" pipeline = QwenImageLayeredPipeline.from_pretrained("Qwen/Qwen-Image-Layered", torch_dtype=dtype).to(device) # --- Load LoRA if configured --- LORA_ENABLED = bool(LORA_REPO and LORA_PATH_IN_REPO) if LORA_ENABLED: print(f"Loading LoRA from {LORA_REPO}/{LORA_PATH_IN_REPO}") pipeline.load_lora_weights(LORA_REPO, weight_name=LORA_PATH_IN_REPO, adapter_name="lora") print("LoRA loaded successfully") # pipeline.set_progress_bar_config(disable=None) def ensure_dirname(path: str): if path and not os.path.exists(path): os.makedirs(path, exist_ok=True) def random_str(length=8): return uuid.uuid4().hex[:length] def imagelist_to_pptx(img_files): with Image.open(img_files[0]) as img: img_width_px, img_height_px = img.size def px_to_emu(px, dpi=96): inch = px / dpi emu = inch * 914400 return int(emu) prs = Presentation() prs.slide_width = px_to_emu(img_width_px) prs.slide_height = px_to_emu(img_height_px) slide = prs.slides.add_slide(prs.slide_layouts[6]) left = top = 0 for img_path in img_files: slide.shapes.add_picture(img_path, left, top, width=px_to_emu(img_width_px), height=px_to_emu(img_height_px)) with tempfile.NamedTemporaryFile(suffix=".pptx", delete=False) as tmp: prs.save(tmp.name) return tmp.name def export_gallery(images): # images: list of image file paths images = [e[0] for e in images] pptx_path = imagelist_to_pptx(images) return pptx_path def export_gallery_zip(images): # images: list of tuples (file_path, caption) images = [e[0] for e in images] with tempfile.NamedTemporaryFile(suffix=".zip", delete=False) as tmp: with zipfile.ZipFile(tmp.name, 'w', zipfile.ZIP_DEFLATED) as zipf: for i, img_path in enumerate(images): # Get the file extension from original file ext = os.path.splitext(img_path)[1] or '.png' # Add each image to the zip with a numbered filename zipf.write(img_path, f"layer_{i+1}{ext}") return tmp.name @spaces.GPU(duration=180) def infer(input_image, seed=777, randomize_seed=False, prompt=None, neg_prompt=" ", true_guidance_scale=4.0, num_inference_steps=20, layer=4, cfg_norm=True, use_en_prompt=True, lora_weight=1.0, progress=gr.Progress(track_tqdm=True)): if randomize_seed: seed = random.randint(0, MAX_SEED) # Set LoRA weight if enabled if LORA_ENABLED: pipeline.set_adapters(["lora"], adapter_weights=[lora_weight]) print(f"LoRA weight set to: {lora_weight}") if isinstance(input_image, list): input_image = input_image[0] if isinstance(input_image, str): pil_image = Image.open(input_image).convert("RGB").convert("RGBA") elif isinstance(input_image, Image.Image): pil_image = input_image.convert("RGB").convert("RGBA") elif isinstance(input_image, np.ndarray): pil_image = Image.fromarray(input_image).convert("RGB").convert("RGBA") else: raise ValueError("Unsupported input_image type: %s" % type(input_image)) inputs = { "image": pil_image, "generator": torch.Generator(device='cuda').manual_seed(seed), "true_cfg_scale": true_guidance_scale, "prompt": prompt, "negative_prompt": neg_prompt, "num_inference_steps": num_inference_steps, "num_images_per_prompt": 1, "layers": layer, "resolution": 640, # Using different bucket (640, 1024) to determine the resolution. For this version, 640 is recommended "cfg_normalize": cfg_norm, # Whether enable cfg normalization. "use_en_prompt": use_en_prompt, } print(inputs) with torch.inference_mode(): output = pipeline(**inputs) output_images = output.images[0] output = [] temp_files = [] for i, image in enumerate(output_images): output.append(image) # Save to temp file for export tmp = tempfile.NamedTemporaryFile(suffix=".png", delete=False) image.save(tmp.name) temp_files.append(tmp.name) # Generate PPTX pptx_path = imagelist_to_pptx(temp_files) # Generate ZIP with tempfile.NamedTemporaryFile(suffix=".zip", delete=False) as tmp: with zipfile.ZipFile(tmp.name, 'w', zipfile.ZIP_DEFLATED) as zipf: for i, img_path in enumerate(temp_files): zipf.write(img_path, f"layer_{i+1}.png") zip_path = tmp.name return output, pptx_path, zip_path ensure_dirname(LOG_DIR) examples = [ "assets/test_images/1.png", "assets/test_images/2.png", "assets/test_images/3.png", "assets/test_images/4.png", "assets/test_images/5.png", "assets/test_images/6.png", "assets/test_images/7.png", "assets/test_images/8.png", "assets/test_images/9.png", "assets/test_images/10.png", "assets/test_images/11.png", "assets/test_images/12.png", "assets/test_images/13.png", ] css = """ #col-container { margin: 0 auto; max-width: 900px; } #logo-title { text-align: center; } """ with gr.Blocks(css=css) as demo: with gr.Column(elem_id="col-container"): gr.HTML("""

Qwen-Image-Layered with LoRA

Image Layer Decomposition

Author: X @tori29umai

""") with gr.Row(): with gr.Column(scale=1): input_image = gr.Image(label="Input Image", image_mode="RGBA") with gr.Accordion("Advanced Settings", open=False): prompt = gr.Textbox( label="Prompt (Optional)", placeholder="Please enter the prompt to descibe the image. (Optional)", value=DEFAULT_PROMPT, lines=2, ) neg_prompt = gr.Textbox( label="Negative Prompt (Optional)", placeholder="Please enter the negative prompt", value=DEFAULT_NEG_PROMPT, lines=2, ) seed = gr.Slider( label="Seed", minimum=0, maximum=MAX_SEED, step=1, value=0, ) randomize_seed = gr.Checkbox(label="Randomize seed", value=True) true_guidance_scale = gr.Slider( label="True guidance scale", minimum=1.0, maximum=10.0, step=0.1, value=4.0 ) num_inference_steps = gr.Slider( label="Number of inference steps", minimum=1, maximum=50, step=1, value=DEFAULT_STEPS, ) layer = gr.Slider( label="Layers", minimum=2, maximum=10, step=1, value=DEFAULT_LAYERS, ) cfg_norm = gr.Checkbox(label="Whether enable CFG normalization", value=True) use_en_prompt = gr.Checkbox(label="Automatic caption language if no prompt provided, True for EN, False for ZH", value=True) lora_weight = gr.Slider( label="LoRA Weight", minimum=0.0, maximum=2.0, step=0.1, value=LORA_WEIGHT_DEFAULT, visible=LORA_ENABLED, ) run_button = gr.Button("Decompose!", variant="primary") with gr.Column(scale=2): gallery = gr.Gallery(label="Layers", columns=4, rows=1, format="png") with gr.Row(): export_file = gr.File(label="Download PPTX") export_zip_file = gr.File(label="Download ZIP") gr.Examples(examples=examples, inputs=[input_image], outputs=[gallery, export_file, export_zip_file], fn=infer, examples_per_page=14, cache_examples=False, run_on_click=False, ) run_button.click( fn=infer, inputs=[ input_image, seed, randomize_seed, prompt, neg_prompt, true_guidance_scale, num_inference_steps, layer, cfg_norm, use_en_prompt, lora_weight, ], outputs=[gallery, export_file, export_zip_file], ) if __name__ == "__main__": demo.queue().launch()