try: try: import spaces except Exception: class spaces: @staticmethod def GPU(duration=120): def decorator(fn): return fn return decorator HAS_SPACES = True except ImportError: HAS_SPACES = False class spaces: @staticmethod def GPU(duration=180): def decorator(fn): return fn return decorator """ Static-Sound — Music-Driven Image-to-Video Uses Wan2.2 S2V (Sound-to-Video) via diffusers Audio drives the video generation from a reference image. """ import os, gc, uuid from pathlib import Path import torch import gradio as gr import numpy as np from PIL import Image from huggingface_hub import login, hf_hub_download, snapshot_download device = "cuda" if __import__("torch").cuda.is_available() else "cpu" print(f"[device] Using: {device}") if token := os.environ.get("HF_TOKEN"): login(token=token) DATA_ROOT = Path("/data") if Path("/data").exists() else Path("/tmp/sound") CACHE_DIR = DATA_ROOT / "hf_cache" OUTPUT_DIR = DATA_ROOT / "outputs" for d in [CACHE_DIR, OUTPUT_DIR]: d.mkdir(parents=True, exist_ok=True) os.environ["HF_HOME"] = str(CACHE_DIR) os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True" # Model IDs — Wan2.2 S2V (Sound-to-Video) S2V_MODEL = "Wan-AI/Wan2.2-TI2V-5B-Diffusers" LORA_REPO = "Comfy-Org/Wan_2.2_ComfyUI_Repackaged" LORA_FILE = "split_files/loras/wan2.2_t2v_lightx2v_4steps_lora_v1.1_high_noise.safetensors" _pipe = None def _load_pipe(): global _pipe if _pipe is not None: return _pipe from diffusers import WanImageToVideoPipeline from diffusers.models.transformers.transformer_wan import WanTransformer3DModel print("[load] Loading Wan2.2 S2V pipeline...") _pipe = WanImageToVideoPipeline.from_pretrained( S2V_MODEL, torch_dtype=torch.bfloat16, cache_dir=str(CACHE_DIR), ) print("[load] Pipeline ready ✅") return _pipe def _extract_audio_features(audio_path: str) -> dict: """Extract rhythm/beat features from audio to guide generation.""" import librosa y, sr = librosa.load(audio_path, sr=22050, mono=True) tempo, beats = librosa.beat.beat_track(y=y, sr=sr) duration = librosa.get_duration(y=y, sr=sr) rms = float(np.mean(librosa.feature.rms(y=y))) return { "tempo": float(tempo), "duration": duration, "energy": rms, "beats": len(beats), } @spaces.GPU(duration=180) def generate_sound_video( image: Image.Image, audio_file: str, prompt: str, neg_prompt: str, duration_sec: float, steps: int, guidance: float, seed: int, randomize_seed: bool, ): """ Generate a music-driven video from an image and audio file using Wan2.2 S2V. Args: image: Reference image to animate. audio_file: Audio file path (mp3/wav) to drive the video. prompt: Text description of desired motion. neg_prompt: Negative prompt. duration_sec: Video duration in seconds. steps: Inference steps. guidance: Guidance scale. seed: Random seed. randomize_seed: Whether to randomize seed. Returns: Path to generated MP4 video. """ if image is None: raise gr.Error("Please upload a reference image.") if audio_file is None: raise gr.Error("Please upload an audio file.") if randomize_seed: import random seed = random.randint(0, 2**31) # Extract audio features for prompt enhancement audio_info = _extract_audio_features(audio_file) enhanced_prompt = ( f"{prompt}. Tempo: {audio_info['tempo']:.0f} BPM, " f"energetic motion synchronized to music rhythm." ) pipe = _load_pipe() pipe.to(device) # Resize image w, h = image.size scale = min(832/w, 480/h) nw = max(16, int(w*scale)//16*16) nh = max(16, int(h*scale)//16*16) image = image.resize((nw, nh), Image.LANCZOS).convert("RGB") fps = 16 num_frames = max(8, min(400, int(duration_sec * fps))) generator = torch.Generator(device).manual_seed(int(seed)) output = pipe( image=image, prompt=enhanced_prompt, negative_prompt=neg_prompt or None, num_frames=num_frames, num_inference_steps=int(steps), guidance_scale=float(guidance), generator=generator, ) frames = output.frames[0] # Save video import imageio video_path = str(OUTPUT_DIR / f"{uuid.uuid4().hex}_video.mp4") writer = imageio.get_writer(video_path, fps=fps, codec="libx264", quality=8) for frame in frames: writer.append_data(np.array(frame)) writer.close() # Merge audio with video using moviepy try: from moviepy.editor import VideoFileClip, AudioFileClip video_clip = VideoFileClip(video_path) audio_clip = AudioFileClip(audio_file).subclip(0, min(video_clip.duration, audio_info["duration"])) final = video_clip.set_audio(audio_clip) out_path = str(OUTPUT_DIR / f"{uuid.uuid4().hex}_final.mp4") final.write_videofile(out_path, codec="libx264", audio_codec="aac", verbose=False, logger=None) video_clip.close(); audio_clip.close(); final.close() return out_path, int(seed), f"Tempo: {audio_info['tempo']:.0f} BPM | Duration: {audio_info['duration']:.1f}s | Energy: {audio_info['energy']:.4f}" except Exception as e: print(f"Audio merge failed: {e}") return video_path, int(seed), f"Audio merge failed — video only. Tempo: {audio_info['tempo']:.0f} BPM" # ── UI ──────────────────────────────────────────────────────────────────────── CSS = "footer{display:none!important}" HEADER = """
Music-Driven Image-to-Video · Wan 2.2 S2V · ZeroGPU