# SeedVR2-3B image restoration, Upsampler v3 recipe. # # Ported from ByteDance-Seed/SeedVR2-3B. What changed and why: # # 1. Image only. The upstream Space serves video and images from one entry # point, carrying the sequence-parallel plumbing, frame cutting and video # writing along with it. This Space backs an image tool, so that is all gone. # 2. ONE @spaces.GPU entry. Upstream decorates configure_runner, # generation_step AND generation_loop, so a single request booked three GPU # allocations of 100s each. ZeroGPU checks the requested duration against the # visitor's remaining quota, and an unauthenticated visitor has 120 seconds a # day in total, so the upstream shape cannot serve an anonymous user at all. # Everything now runs inside one call with a measured dynamic duration. # 3. No apex. Upstream installs a prebuilt `apex-0.1-cp310-...whl` and selects # `fusedrms` / `fusedln` norms in configs_3b/main.yaml. ZeroGPU runs Python # 3.12, where that wheel does not install, so every norm layer then failed. # The config now selects the `rms` / `layer` paths that the same source file # already implements in pure PyTorch, with identical parameter names and # shapes so the checkpoint loads unchanged. # 4. No hard flash-attn dependency. See models/dit_v2/attention.py. # # The model restores at a fixed ~3.7MP working resolution regardless of input # size (it was trained at high res and NaResize scales the input to meet it), so # GPU cost per request is essentially constant and no tiling is involved. import gc import os import mimetypes from pathlib import Path import gradio as gr import spaces import torch import torch.nn.functional as F from einops import rearrange from omegaconf import OmegaConf from PIL import Image from huggingface_hub import hf_hub_download from torchvision.transforms import Compose, Lambda, Normalize import torchvision.transforms as T from upsampler_theme import UPSAMPLER_CSS, UPSAMPLER_THEME, footer_html, header_html from data.image.transforms.divisible_crop import DivisibleCrop from data.image.transforms.na_resize import NaResize from data.video.transforms.rearrange import Rearrange from common.config import load_config from common.distributed import init_torch from common.seed import set_seed from projects.video_diffusion_sr.infer import VideoDiffusionInfer try: from projects.video_diffusion_sr.color_fix import wavelet_reconstruction USE_COLOR_FIX = True except ImportError: USE_COLOR_FIX = False print("color fix unavailable; output will not be wavelet-reconstructed") # Weights come from the official ByteDance repo rather than a re-upload. It is # the canonical source for these files and is Apache-2.0, so there is no mirror # in the chain that could change under us. WEIGHTS_REPO = "ByteDance-Seed/SeedVR2-3B" CKPT_DIR = Path("./ckpts") CKPT_DIR.mkdir(exist_ok=True) def _fetch(filename: str, target: Path) -> str: if target.exists(): return str(target) path = hf_hub_download(repo_id=WEIGHTS_REPO, filename=filename) target.symlink_to(path) return str(target) DIT_CKPT = _fetch("seedvr2_ema_3b.pth", CKPT_DIR / "seedvr2_ema_3b.pth") VAE_CKPT = _fetch("ema_vae.pth", CKPT_DIR / "ema_vae.pth") # The text branch is conditioned by two fixed embeddings shipped with the # weights, so this Space runs no text encoder at all. POS_EMB = _fetch("pos_emb.pt", Path("./pos_emb.pt")) NEG_EMB = _fetch("neg_emb.pt", Path("./neg_emb.pt")) # The resolution the model restores at, as an area. Upstream hardcodes # 2560*1440 for images with `downsample_only=False`, meaning small inputs are # scaled UP to it and large inputs down, because the model was only trained at # high resolution. Keeping that exact value keeps output quality identical to # the reference implementation. WORK_AREA = 2560 * 1440 # Single-process "distributed" context. The model code routes every device # placement through common.distributed, which reads these. os.environ.setdefault("MASTER_ADDR", "127.0.0.1") os.environ.setdefault("MASTER_PORT", "12355") os.environ.setdefault("RANK", "0") os.environ.setdefault("WORLD_SIZE", "1") os.environ.setdefault("LOCAL_RANK", "0") _runner = None def _ensure_runner(): """Build the runner once, inside a GPU context. Deliberately NOT done at module scope, even though ZeroGPU prefers that for placement: `init_torch` ends in `dist.init_process_group(backend="nccl")` and `torch.cuda.set_device`, which need a real device rather than the CUDA emulation that applies outside `@spaces.GPU`. Memoized because `init_process_group` raises if called twice, and because a warm worker should not reload 3B parameters per request. """ global _runner if _runner is not None: return _runner if not torch.distributed.is_initialized(): init_torch(cudnn_benchmark=False) runner = VideoDiffusionInfer(load_config(os.path.join("./configs_3b", "main.yaml"))) OmegaConf.set_readonly(runner.config, False) runner.configure_dit_model(device="cuda", checkpoint=DIT_CKPT) runner.configure_vae_model() if hasattr(runner.vae, "set_memory_limit"): runner.vae.set_memory_limit(**runner.config.vae.memory_limit) _runner = runner return _runner def _transform(): return Compose( [ NaResize(resolution=WORK_AREA**0.5, mode="area", downsample_only=False), Lambda(lambda x: torch.clamp(x, 0.0, 1.0)), DivisibleCrop((16, 16)), Normalize(0.5, 0.5), Rearrange("t c h w -> c t h w"), ] ) def _duration(image, steps=1, seed=666, progress=None) -> int: """Every request restores at the same ~3.7MP working resolution, so cost tracks the step count and little else. Measured on a cold worker, then given headroom; kept as small as honesty allows, because the request is checked against the visitor's remaining quota and a smaller one also ranks higher in the ZeroGPU queue.""" return int(min(120, 35 + 12 * int(steps))) @spaces.GPU(duration=_duration) @torch.no_grad() def upscale_image(image, steps=1, seed=666, progress=gr.Progress(track_tqdm=True)): if image is None: raise gr.Error("Upload an image first.") runner = _ensure_runner() runner.config.diffusion.cfg.scale = 1.0 runner.config.diffusion.cfg.rescale = 0.0 runner.config.diffusion.timesteps.sampling.steps = int(steps) runner.configure_diffusion() set_seed(int(seed) % (2**32), same_across_ranks=True) img = Image.open(image).convert("RGB") if isinstance(image, str) else image.convert("RGB") tensor = T.ToTensor()(img).unsqueeze(0) # (t=1, c, h, w) cond = _transform()(tensor.to("cuda")) original = cond latents = runner.vae_encode([cond]) text_embeds = { "texts_pos": [torch.load(POS_EMB).to("cuda")], "texts_neg": [torch.load(NEG_EMB).to("cuda")], } noise = [torch.randn_like(latent) for latent in latents] aug_noise = [torch.randn_like(latent) for latent in latents] def _add_noise(x, aug): t = torch.tensor([1000.0], device="cuda") * 0.1 shape = torch.tensor(x.shape[1:], device="cuda")[None] return runner.schedule.forward(x, aug, runner.timestep_transform(t, shape)) conditions = [ runner.get_condition(n, task="sr", latent_blur=_add_noise(latent, a)) for n, a, latent in zip(noise, aug_noise, latents) ] with torch.autocast("cuda", torch.bfloat16, enabled=True): videos = runner.inference( noises=noise, conditions=conditions, dit_offload=False, **text_embeds ) sample = videos[0] sample = ( rearrange(sample[:, None], "c t h w -> t c h w") if sample.ndim == 3 else rearrange(sample, "c t h w -> t c h w") ) reference = ( rearrange(original[:, None], "c t h w -> t c h w") if original.ndim == 3 else rearrange(original, "c t h w -> t c h w") ) if USE_COLOR_FIX: sample = wavelet_reconstruction(sample.to("cpu"), reference[: sample.size(0)].to("cpu")) else: sample = sample.to("cpu") sample = rearrange(sample, "t c h w -> t h w c") sample = sample.clip(-1, 1).mul_(0.5).add_(0.5).mul_(255).round().to(torch.uint8).numpy() del latents, conditions, videos gc.collect() torch.cuda.empty_cache() return Image.fromarray(sample[0]) with gr.Blocks(css=UPSAMPLER_CSS, theme=UPSAMPLER_THEME) as demo: gr.HTML( header_html( "SeedVR2 3B Image Upscaler", "One-step diffusion restoration that rebuilds real detail in blurry, " "compressed, and low-resolution photos.", ) ) with gr.Row(): with gr.Column(): image_in = gr.Image(label="Image", type="filepath") steps = gr.Slider(1, 4, value=1, step=1, label="Steps") seed = gr.Number(label="Seed", value=666, precision=0) run = gr.Button("Upscale Image", variant="primary") with gr.Column(): image_out = gr.Image(label="Result", type="pil") # No leading slash: gradio prefixes it, and "/upscale_image" here would # publish the endpoint as "//upscale_image". run.click( upscale_image, inputs=[image_in, steps, seed], outputs=[image_out], api_name="upscale_image", ) gr.HTML( footer_html( "SeedVR2-3B is ByteDance's one-step diffusion model for image and video " "restoration. It rebuilds genuine texture in photos that are blurry, " "heavily compressed, or simply too small, restoring at high resolution " "rather than smoothing detail away the way a conventional upscaler does.", "https://upsampler.com/free-image-upscaler-no-signup", "free image upscaler", ) ) demo.launch(ssr_mode=False, show_error=True)