luca115 commited on
Commit
867bc57
·
verified ·
1 Parent(s): 6598976

Upload folder using huggingface_hub

Browse files
Files changed (3) hide show
  1. app.py +15 -9
  2. configs_3b/main.yaml +7 -8
  3. upsampler_theme.py +0 -54
app.py CHANGED
@@ -25,12 +25,11 @@
25
 
26
  # Must precede the torch import: the allocator reads this at initialization.
27
  #
28
- # Without it the VAE decode dies with
29
- # "NVML_SUCCESS == r INTERNAL ASSERT FAILED at CUDACachingAllocator.cpp" once
30
- # denoising finishes and the decoder asks for its large contiguous buffers.
31
- # Same failure, same fix as wan-2-2-5b-video: ZeroGPU hands out a partitioned
32
- # GPU, and expandable segments keep the allocator off the NVML query path that
33
- # breaks there.
34
  import os
35
 
36
  os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
@@ -96,9 +95,16 @@ NEG_EMB = _fetch("neg_emb.pt", Path("./neg_emb.pt"))
96
  # scaled UP to it and large inputs down, because the model was only trained at
97
  # high resolution.
98
  #
99
- # Overridable so the value can be tuned against the ZeroGPU VAE-decode ceiling
100
- # without a code change: the decoder's peak allocation scales with this, and
101
- # that allocation is what trips the caching allocator's NVML assert.
 
 
 
 
 
 
 
102
  WORK_AREA = int(os.environ.get("SEEDVR2_WORK_AREA", 1920 * 1080))
103
 
104
  # Single-process "distributed" context. The model code routes every device
 
25
 
26
  # Must precede the torch import: the allocator reads this at initialization.
27
  #
28
+ # Fragmentation insurance, NOT the fix for the NVML assert this Space used to
29
+ # die on. That was measured: the ZeroGPU worker printed
30
+ # 'expandable_segments:True' on the runs that still crashed, so the setting was
31
+ # applied the whole time and made no difference. What actually decides it is
32
+ # WORK_AREA below. Kept because it costs nothing and reduces fragmentation.
 
33
  import os
34
 
35
  os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
 
95
  # scaled UP to it and large inputs down, because the model was only trained at
96
  # high resolution.
97
  #
98
+ # THIS is what decides whether the VAE decode survives on ZeroGPU. The
99
+ # decoder's peak contiguous allocation scales with it, and that allocation is
100
+ # what trips "NVML_SUCCESS == r INTERNAL ASSERT FAILED" in the caching
101
+ # allocator. Measured: 2560x1440 (the upstream value) fails, 1920x1080 passes
102
+ # in 14.7s. Left overridable so the ceiling can be probed without a code
103
+ # change, but do not raise the default without re-testing a real run.
104
+ #
105
+ # The cost is honest to state: the model was trained at high resolution, so
106
+ # restoring at 2MP rather than 3.7MP gives up some of its headroom on very
107
+ # large outputs. It still upscales ~4.9x from a 360x240 input.
108
  WORK_AREA = int(os.environ.get("SEEDVR2_WORK_AREA", 1920 * 1080))
109
 
110
  # Single-process "distributed" context. The model code routes every device
configs_3b/main.yaml CHANGED
@@ -53,14 +53,13 @@ vae:
53
  memory_limit:
54
  # CHANGED FROM UPSTREAM (Upsampler): 0.5 -> 0.15 GiB.
55
  #
56
- # This is the VAE's own decode-splitting threshold: a conv whose input would
57
- # exceed it is split and run in pieces. At the upstream 0.5 the decoder still
58
- # asks for buffers large enough to send ZeroGPU's caching allocator down the
59
- # NVML path that dies with "NVML_SUCCESS == r INTERNAL ASSERT FAILED", which
60
- # is the same class of failure wan-2-2-5b-video solved by keeping the decode
61
- # tiled and sliced. Smaller splits mean more convs but no single large
62
- # contiguous allocation. Output is unaffected: the splitting carries the
63
- # overlap cache between pieces, so it is exact rather than approximate.
64
  conv_max_mem: 0.15
65
  norm_max_mem: 0.15
66
  checkpoint: ./ckpts/ema_vae.pth
 
53
  memory_limit:
54
  # CHANGED FROM UPSTREAM (Upsampler): 0.5 -> 0.15 GiB.
55
  #
56
+ # The VAE's own decode-splitting threshold: a conv whose input would exceed
57
+ # it is split and run in pieces. Lowering this did NOT by itself clear the
58
+ # NVML assert this Space used to die on -- it moved the failure from the
59
+ # conv to the torch.cat that reassembles the splits. WORK_AREA in app.py is
60
+ # what actually decides that. Kept at 0.15 because smaller allocations are
61
+ # still the right default here and the splitting is exact: it carries its
62
+ # overlap cache between pieces, so output is unchanged.
 
63
  conv_max_mem: 0.15
64
  norm_max_mem: 0.15
65
  checkpoint: ./ckpts/ema_vae.pth
upsampler_theme.py DELETED
@@ -1,54 +0,0 @@
1
- """Shared Upsampler look-and-feel for every Upsampler/* HF Space.
2
-
3
- `create_and_push.py` uploads this file alongside each Space's app.py, so every
4
- Space imports the exact same theme, CSS, header, and footer. Keep this the
5
- single source of truth (v3 recipe): Soft indigo/purple theme, gradient primary
6
- button, Gradio's own footer hidden, 1000px max width, minimal header/footer.
7
- """
8
-
9
- import gradio as gr
10
-
11
- UPSAMPLER_THEME = gr.themes.Soft(
12
- primary_hue=gr.themes.colors.indigo,
13
- secondary_hue=gr.themes.colors.purple,
14
- neutral_hue=gr.themes.colors.slate,
15
- font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"],
16
- ).set(
17
- button_primary_background_fill="linear-gradient(90deg, #6366f1 0%, #a855f7 100%)",
18
- button_primary_background_fill_hover="linear-gradient(90deg, #4f46e5 0%, #9333ea 100%)",
19
- button_primary_text_color="#ffffff",
20
- button_primary_border_color="*primary_500",
21
- )
22
-
23
- # Hide Gradio's built-in footer and keep the app narrow and centered.
24
- UPSAMPLER_CSS = """
25
- footer { display: none !important; }
26
- .gradio-container { max-width: 1000px !important; margin: 0 auto !important; }
27
- #usp-header h1 { font-size: 1.7rem; font-weight: 700; margin: 0 0 .25rem; }
28
- #usp-header p { opacity: .6; margin: 0; }
29
- #usp-footer { opacity: .5; font-size: .85rem; margin-top: 1.25rem; }
30
- #usp-footer a { text-decoration: none; }
31
- """
32
-
33
-
34
- def header_html(title: str, subtitle: str) -> str:
35
- return f"""<div id="usp-header">
36
- <h1>{title}</h1>
37
- <p>{subtitle}</p>
38
- </div>"""
39
-
40
-
41
- _LINK_STYLE = "color:#8b7cf6;font-weight:600;text-decoration:none"
42
-
43
-
44
- def footer_html(description: str, tool_url: str, tool_anchor: str) -> str:
45
- """SEO footer (v4 recipe): a short paragraph describing what the model
46
- does (unique per Space, keyword-bearing) plus the Upsampler attribution
47
- with a deep link to the matching /free-* tool on upsampler.com. No model
48
- credit/license line (the README frontmatter carries the license). Spaces
49
- target model-name queries; the site pages keep the intent queries
50
- ("free X no signup"), so the two never compete."""
51
- return f"""<div id="usp-footer">
52
- <p style="margin:0 0 10px">{description}</p>
53
- <p style="margin:0">Maintained by <a href="https://upsampler.com" target="_blank" rel="noopener" style="{_LINK_STYLE}">Upsampler</a>. Check out the <a href="{tool_url}" target="_blank" rel="noopener" style="{_LINK_STYLE}">{tool_anchor}</a>, no sign-up required.</p>
54
- </div>"""