Joffrey Thomas commited on
Commit
aacd93e
·
1 Parent(s): 00ea211

Refactor app.py: Optimize model loading for ZeroGPU and adjust CSS injection for Gradio compatibility

Browse files
Files changed (1) hide show
  1. app.py +18 -7
app.py CHANGED
@@ -111,11 +111,20 @@ except Exception as _e:
111
 
112
  model_id = "mistralai/Magistral-Small-2509"
113
  tokenizer = AutoTokenizer.from_pretrained(model_id, tokenizer_type="mistral", use_fast=False)
114
- model = Mistral3ForConditionalGeneration.from_pretrained(
 
 
 
 
 
 
115
  model_id,
116
  torch_dtype=torch.bfloat16,
117
  low_cpu_mem_usage=True,
118
- ).eval()
 
 
 
119
 
120
 
121
  # SYSTEM_PROMPT_TEXT = (
@@ -584,6 +593,7 @@ def _read_text(path: str) -> str:
584
  return ""
585
 
586
  APP_CSS = _read_text('static/style.css') + "\n#lat_box, #lng_box { display:none; }\n" + """
 
587
  #next_btn { position: absolute; left: -9999px; }
588
  #lobby_group, #game_group{max-width:1024px;margin:24px auto;padding:16px;}
589
  #start_btn{height:48px;font-weight:700}
@@ -859,6 +869,11 @@ APP_BOOT_JS = """
859
  """.replace("__GMAPS_KEY__", GOOGLE_MAPS_API_KEY or '')
860
 
861
  with gr.Blocks(title="LLM GeoGuessr") as demo:
 
 
 
 
 
862
  user_profile = gr.State()
863
 
864
  with gr.Row():
@@ -1344,8 +1359,4 @@ with gr.Blocks(title="LLM GeoGuessr") as demo:
1344
 
1345
 
1346
  if __name__ == "__main__":
1347
- demo.queue().launch(
1348
- server_name="0.0.0.0",
1349
- server_port=7860,
1350
- css=APP_CSS,
1351
- )
 
111
 
112
  model_id = "mistralai/Magistral-Small-2509"
113
  tokenizer = AutoTokenizer.from_pretrained(model_id, tokenizer_type="mistral", use_fast=False)
114
+ # On ZeroGPU, CUDA is emulated at module load and becomes a real GPU only inside
115
+ # @spaces.GPU functions. The docs explicitly require model placement to happen at
116
+ # the root module level (`.to("cuda")`); lazy moves inside @spaces.GPU are much
117
+ # slower because tensor packing happens at startup.
118
+ # https://huggingface.co/docs/hub/spaces-zerogpu#model-loading
119
+ model = (
120
+ Mistral3ForConditionalGeneration.from_pretrained(
121
  model_id,
122
  torch_dtype=torch.bfloat16,
123
  low_cpu_mem_usage=True,
124
+ )
125
+ .to("cuda")
126
+ .eval()
127
+ )
128
 
129
 
130
  # SYSTEM_PROMPT_TEXT = (
 
593
  return ""
594
 
595
  APP_CSS = _read_text('static/style.css') + "\n#lat_box, #lng_box { display:none; }\n" + """
596
+ #app-styles { display: none !important; }
597
  #next_btn { position: absolute; left: -9999px; }
598
  #lobby_group, #game_group{max-width:1024px;margin:24px auto;padding:16px;}
599
  #start_btn{height:48px;font-weight:700}
 
869
  """.replace("__GMAPS_KEY__", GOOGLE_MAPS_API_KEY or '')
870
 
871
  with gr.Blocks(title="LLM GeoGuessr") as demo:
872
+ # Gradio 6 dropped `css=` on `Blocks(...)`, and `launch(css=...)` isn't
873
+ # always honoured under SSR. Injecting a <style> tag as a child of the app
874
+ # is the only place that's guaranteed to render in every version.
875
+ gr.HTML(f"<style>{APP_CSS}</style>", elem_id="app-styles")
876
+
877
  user_profile = gr.State()
878
 
879
  with gr.Row():
 
1359
 
1360
 
1361
  if __name__ == "__main__":
1362
+ demo.queue().launch(server_name="0.0.0.0", server_port=7860)