Spaces:
Running on Zero
Running on Zero
Download app.py from hugging-apps/deskforge-gemma4-e4b: direct link, hf CLI and curl.
- Browser
- Download file 9.14 kB
-
https://huggingface.co/spaces/hugging-apps/deskforge-gemma4-e4b/resolve/main/app.py
- Command line
-
hf download hf://spaces/hugging-apps/deskforge-gemma4-e4b/app.py
-
curl -L -o app.py https://huggingface.co/spaces/hugging-apps/deskforge-gemma4-e4b/resolve/main/app.py
9.14 kB
| import spaces # MUST come before torch / transformers (ZeroGPU) | |
| import re | |
| import time | |
| import gradio as gr | |
| import torch | |
| from PIL import Image, ImageDraw | |
| from transformers import AutoModelForImageTextToText, AutoProcessor | |
| MODEL = "docling-project/DeskForge-Gemma4-E4B" | |
| MAX_PIXELS = 2_097_152 # training-time cap on screenshot area | |
| SYSTEM_PROMPT = """You are a computer-use agent operating a desktop graphical interface. At each step you see the user's task, a screenshot of the current screen, and the actions you have already taken. Reply with the next action as pyautogui code and nothing else -- no explanation, no code fence, no commentary. | |
| Coordinates are fractions of the screen, not pixels: x runs from 0.0 at the left edge to 1.0 at the right edge, y from 0.0 at the top to 1.0 at the bottom. Write both with four decimals. | |
| These are the only actions available: | |
| pyautogui.click(x=0.0000, y=0.0000) | |
| pyautogui.doubleClick(x=0.0000, y=0.0000) | |
| pyautogui.rightClick(x=0.0000, y=0.0000) | |
| pyautogui.middleClick(x=0.0000, y=0.0000) | |
| computer.tripleClick(x=0.0000, y=0.0000) | |
| pyautogui.moveTo(x=0.0000, y=0.0000) | |
| pyautogui.dragTo(x=0.0000, y=0.0000, button='left') | |
| pyautogui.scroll(-4) | |
| pyautogui.hscroll(4) | |
| pyautogui.write(message='text to type') | |
| pyautogui.press('enter') | |
| pyautogui.hotkey(['ctrl', 'c']) | |
| computer.wait() | |
| computer.terminate(status='success') | |
| To scroll at a particular place, move there first and then scroll. When the task is finished, or cannot be finished, end with computer.terminate.""" | |
| def fit(image, max_pixels=MAX_PIXELS): | |
| """Downscale to at most max_pixels (bilinear, sides rounded down), as in training.""" | |
| w, h = image.size | |
| if w * h <= max_pixels: | |
| return image | |
| ratio = w / h | |
| height = (max_pixels / ratio) ** 0.5 | |
| return image.resize( | |
| (max(1, int(height * ratio)), max(1, int(height))), Image.BILINEAR | |
| ) | |
| def user_text(instruction): | |
| return f"Task: {instruction}\n\nActions already taken:\n(none -- this is the first step)\n\nNext action:" | |
| def chat(instruction): | |
| return [ | |
| {"role": "system", "content": SYSTEM_PROMPT}, | |
| { | |
| "role": "user", | |
| "content": [{"type": "image"}, {"type": "text", "text": user_text(instruction)}], | |
| }, | |
| ] | |
| POINT_RE = re.compile(r"x=([\d.]+), y=([\d.]+)") | |
| def to_pixels(action, screenshot): | |
| """Fractional screen coords -> pixel coords on the original screenshot.""" | |
| m = POINT_RE.search(action) | |
| if not m: | |
| return None | |
| x, y = map(float, m.groups()) | |
| return round(x * screenshot.width), round(y * screenshot.height) | |
| def draw_marker(image, point): | |
| """Draw a crosshair + circle at the predicted click point. | |
| A white underlay keeps the marker visible on both light and dark UIs. | |
| """ | |
| out = image.copy() | |
| d = ImageDraw.Draw(out) | |
| x, y = point | |
| r = max(10, min(out.width, out.height) // 60) | |
| # white underlay, then orange marker on top | |
| d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 255, 255), width=6) | |
| d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 64, 0), width=3) | |
| for (x0, y0, x1, y1) in [ | |
| (x - r * 1.8, y, x + r * 1.8, y), # horizontal | |
| (x, y - r * 1.8, x, y + r * 1.8), # vertical | |
| ]: | |
| d.line([x0, y0, x1, y1], fill=(255, 255, 255), width=9) | |
| d.line([x0, y0, x1, y1], fill=(255, 64, 0), width=4) | |
| return out | |
| processor = AutoProcessor.from_pretrained(MODEL) | |
| model = ( | |
| AutoModelForImageTextToText.from_pretrained(MODEL, dtype=torch.bfloat16) | |
| .eval() | |
| .to("cuda") | |
| ) | |
| def predict(instruction: str, screenshot, max_new_tokens: int = 128): | |
| """Predict the next computer-use action for a desktop screenshot. | |
| Args: | |
| instruction: the task to perform on the screen, e.g. "Open the File menu". | |
| screenshot: a PIL image of the desktop to act on. | |
| max_new_tokens: generation length cap for the emitted action. | |
| Returns: | |
| Tuple of (annotated screenshot with the predicted click point, | |
| predicted action code, fractional coordinates, inference seconds). | |
| """ | |
| if screenshot is None: | |
| raise gr.Error("Please provide a desktop screenshot first.") | |
| if not instruction or not instruction.strip(): | |
| raise gr.Error("Please enter an instruction, e.g. 'Open the File menu'.") | |
| if screenshot.mode != "RGB": | |
| screenshot = screenshot.convert("RGB") | |
| prompt = processor.apply_chat_template( | |
| chat(instruction.strip()), tokenize=False, add_generation_prompt=True, | |
| enable_thinking=False, | |
| ) | |
| inputs = processor( | |
| text=[prompt], images=[fit(screenshot)], return_tensors="pt" | |
| ).to(model.device) | |
| t0 = time.perf_counter() | |
| with torch.inference_mode(): | |
| output = model.generate(**inputs, max_new_tokens=int(max_new_tokens), do_sample=False) | |
| elapsed = time.perf_counter() - t0 | |
| action = processor.decode( | |
| output[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True | |
| ).strip() | |
| point = to_pixels(action, screenshot) | |
| annotated = draw_marker(screenshot, point) if point else screenshot.copy() | |
| m = POINT_RE.search(action) | |
| coords = f"x={m.group(1)}, y={m.group(2)}" if m else "— (no point action)" | |
| return annotated, action, coords, f"{elapsed:.1f}s" | |
| CSS = """ | |
| #col-container { max-width: 1200px; margin: 0 auto; } | |
| .dark .gradio-container { color: var(--body-text-color); } | |
| """ | |
| with gr.Blocks() as demo: | |
| with gr.Column(elem_id="col-container"): | |
| gr.Markdown( | |
| """ | |
| # 🖱️ DeskForge-Gemma4-E4B · GUI Grounding | |
| **DeskForge-Gemma4-E4B** ([model card](https://huggingface.co/docling-project/DeskForge-Gemma4-E4B)) | |
| is [Gemma 4 E4B](https://huggingface.co/google/gemma-4-E4B-it) fine-tuned on 200K grounding | |
| examples from [DeskForge-1M](https://huggingface.co/datasets/docling-project/DeskForge-1M). | |
| Given a desktop screenshot and a task, it replies with the next action as | |
| `pyautogui` code, with coordinates as **fractions of the screen** | |
| (x: 0.0 left → 1.0 right, y: 0.0 top → 1.0 bottom). The predicted click | |
| point is drawn on the screenshot. | |
| Paper: [DeskForge: Dense Supervision from Desktop Environments for Computer-Use Agents](https://arxiv.org/abs/2610.02320) · | |
| [Project page](https://saidgurbuz.github.io/deskforge/) · [GitHub](https://github.com/Saidgurbuz/deskforge) | |
| """ | |
| ) | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| screenshot_in = gr.Image( | |
| type="pil", label="Desktop screenshot", height=420, | |
| sources=["upload", "clipboard"], | |
| ) | |
| instruction = gr.Textbox( | |
| label="Task / instruction", | |
| placeholder="e.g. Open the File menu", | |
| lines=2, | |
| ) | |
| run_btn = gr.Button("Ground it", variant="primary") | |
| with gr.Accordion("Advanced", open=False): | |
| max_new_tokens = gr.Slider( | |
| 16, 256, value=128, step=8, label="Max new tokens", | |
| ) | |
| with gr.Column(scale=1): | |
| screenshot_out = gr.Image( | |
| type="pil", label="Annotated screenshot (predicted click point)", | |
| height=420, | |
| ) | |
| action_out = gr.Textbox(label="Predicted action (pyautogui)", lines=2) | |
| coords_out = gr.Textbox(label="Fractional coordinates", lines=1) | |
| time_out = gr.Textbox(label="Inference time", lines=1) | |
| gr.Examples( | |
| examples=[ | |
| ["Select the Export as QIF... option from the File menu.", "examples/homebank_file_menu.png"], | |
| ["Add the factorial function to the current expression.", "examples/qalculate_factorial.png"], | |
| ["Select Financial Mode in the calculator's mode dropdown.", "examples/calculator_financial_mode.png"], | |
| ["Open the file encoding dropdown to view available encoding options.", "examples/mousepad_encoding_dropdown.png"], | |
| ["Open the Image Viewer menu.", "examples/image_viewer_menu.png"], | |
| ["Open the Applications menu to view application categories.", "examples/applications_menu.png"], | |
| ["Enable the sidebar in the file manager.", "examples/nautilus_show_sidebar.png"], | |
| ["Select the Sort by Size option in the Transmission View menu.", "examples/transmission_sort_by_size.png"], | |
| ], | |
| inputs=[instruction, screenshot_in], | |
| fn=predict, | |
| outputs=[screenshot_out, action_out, coords_out, time_out], | |
| cache_examples=True, | |
| cache_mode="lazy", | |
| ) | |
| run_btn.click( | |
| predict, | |
| inputs=[instruction, screenshot_in, max_new_tokens], | |
| outputs=[screenshot_out, action_out, coords_out, time_out], | |
| api_name="predict", | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch(mcp_server=True, theme=gr.themes.Citrus(), css=CSS) |