import spaces # MUST come before torch / transformers (ZeroGPU) import re import time import gradio as gr import torch from PIL import Image, ImageDraw from transformers import AutoModelForImageTextToText, AutoProcessor MODEL = "docling-project/DeskForge-Gemma4-E4B" MAX_PIXELS = 2_097_152 # training-time cap on screenshot area SYSTEM_PROMPT = """You are a computer-use agent operating a desktop graphical interface. At each step you see the user's task, a screenshot of the current screen, and the actions you have already taken. Reply with the next action as pyautogui code and nothing else -- no explanation, no code fence, no commentary. Coordinates are fractions of the screen, not pixels: x runs from 0.0 at the left edge to 1.0 at the right edge, y from 0.0 at the top to 1.0 at the bottom. Write both with four decimals. These are the only actions available: pyautogui.click(x=0.0000, y=0.0000) pyautogui.doubleClick(x=0.0000, y=0.0000) pyautogui.rightClick(x=0.0000, y=0.0000) pyautogui.middleClick(x=0.0000, y=0.0000) computer.tripleClick(x=0.0000, y=0.0000) pyautogui.moveTo(x=0.0000, y=0.0000) pyautogui.dragTo(x=0.0000, y=0.0000, button='left') pyautogui.scroll(-4) pyautogui.hscroll(4) pyautogui.write(message='text to type') pyautogui.press('enter') pyautogui.hotkey(['ctrl', 'c']) computer.wait() computer.terminate(status='success') To scroll at a particular place, move there first and then scroll. When the task is finished, or cannot be finished, end with computer.terminate.""" def fit(image, max_pixels=MAX_PIXELS): """Downscale to at most max_pixels (bilinear, sides rounded down), as in training.""" w, h = image.size if w * h <= max_pixels: return image ratio = w / h height = (max_pixels / ratio) ** 0.5 return image.resize( (max(1, int(height * ratio)), max(1, int(height))), Image.BILINEAR ) def user_text(instruction): return f"Task: {instruction}\n\nActions already taken:\n(none -- this is the first step)\n\nNext action:" def chat(instruction): return [ {"role": "system", "content": SYSTEM_PROMPT}, { "role": "user", "content": [{"type": "image"}, {"type": "text", "text": user_text(instruction)}], }, ] POINT_RE = re.compile(r"x=([\d.]+), y=([\d.]+)") def to_pixels(action, screenshot): """Fractional screen coords -> pixel coords on the original screenshot.""" m = POINT_RE.search(action) if not m: return None x, y = map(float, m.groups()) return round(x * screenshot.width), round(y * screenshot.height) def draw_marker(image, point): """Draw a crosshair + circle at the predicted click point. A white underlay keeps the marker visible on both light and dark UIs. """ out = image.copy() d = ImageDraw.Draw(out) x, y = point r = max(10, min(out.width, out.height) // 60) # white underlay, then orange marker on top d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 255, 255), width=6) d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 64, 0), width=3) for (x0, y0, x1, y1) in [ (x - r * 1.8, y, x + r * 1.8, y), # horizontal (x, y - r * 1.8, x, y + r * 1.8), # vertical ]: d.line([x0, y0, x1, y1], fill=(255, 255, 255), width=9) d.line([x0, y0, x1, y1], fill=(255, 64, 0), width=4) return out processor = AutoProcessor.from_pretrained(MODEL) model = ( AutoModelForImageTextToText.from_pretrained(MODEL, dtype=torch.bfloat16) .eval() .to("cuda") ) @spaces.GPU(duration=120) def predict(instruction: str, screenshot, max_new_tokens: int = 128): """Predict the next computer-use action for a desktop screenshot. Args: instruction: the task to perform on the screen, e.g. "Open the File menu". screenshot: a PIL image of the desktop to act on. max_new_tokens: generation length cap for the emitted action. Returns: Tuple of (annotated screenshot with the predicted click point, predicted action code, fractional coordinates, inference seconds). """ if screenshot is None: raise gr.Error("Please provide a desktop screenshot first.") if not instruction or not instruction.strip(): raise gr.Error("Please enter an instruction, e.g. 'Open the File menu'.") if screenshot.mode != "RGB": screenshot = screenshot.convert("RGB") prompt = processor.apply_chat_template( chat(instruction.strip()), tokenize=False, add_generation_prompt=True, enable_thinking=False, ) inputs = processor( text=[prompt], images=[fit(screenshot)], return_tensors="pt" ).to(model.device) t0 = time.perf_counter() with torch.inference_mode(): output = model.generate(**inputs, max_new_tokens=int(max_new_tokens), do_sample=False) elapsed = time.perf_counter() - t0 action = processor.decode( output[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True ).strip() point = to_pixels(action, screenshot) annotated = draw_marker(screenshot, point) if point else screenshot.copy() m = POINT_RE.search(action) coords = f"x={m.group(1)}, y={m.group(2)}" if m else "— (no point action)" return annotated, action, coords, f"{elapsed:.1f}s" CSS = """ #col-container { max-width: 1200px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } """ with gr.Blocks() as demo: with gr.Column(elem_id="col-container"): gr.Markdown( """ # 🖱️ DeskForge-Gemma4-E4B · GUI Grounding **DeskForge-Gemma4-E4B** ([model card](https://huggingface.co/docling-project/DeskForge-Gemma4-E4B)) is [Gemma 4 E4B](https://huggingface.co/google/gemma-4-E4B-it) fine-tuned on 200K grounding examples from [DeskForge-1M](https://huggingface.co/datasets/docling-project/DeskForge-1M). Given a desktop screenshot and a task, it replies with the next action as `pyautogui` code, with coordinates as **fractions of the screen** (x: 0.0 left → 1.0 right, y: 0.0 top → 1.0 bottom). The predicted click point is drawn on the screenshot. Paper: [DeskForge: Dense Supervision from Desktop Environments for Computer-Use Agents](https://arxiv.org/abs/2610.02320) · [Project page](https://saidgurbuz.github.io/deskforge/) · [GitHub](https://github.com/Saidgurbuz/deskforge) """ ) with gr.Row(): with gr.Column(scale=1): screenshot_in = gr.Image( type="pil", label="Desktop screenshot", height=420, sources=["upload", "clipboard"], ) instruction = gr.Textbox( label="Task / instruction", placeholder="e.g. Open the File menu", lines=2, ) run_btn = gr.Button("Ground it", variant="primary") with gr.Accordion("Advanced", open=False): max_new_tokens = gr.Slider( 16, 256, value=128, step=8, label="Max new tokens", ) with gr.Column(scale=1): screenshot_out = gr.Image( type="pil", label="Annotated screenshot (predicted click point)", height=420, ) action_out = gr.Textbox(label="Predicted action (pyautogui)", lines=2) coords_out = gr.Textbox(label="Fractional coordinates", lines=1) time_out = gr.Textbox(label="Inference time", lines=1) gr.Examples( examples=[ ["Select the Export as QIF... option from the File menu.", "examples/homebank_file_menu.png"], ["Add the factorial function to the current expression.", "examples/qalculate_factorial.png"], ["Select Financial Mode in the calculator's mode dropdown.", "examples/calculator_financial_mode.png"], ["Open the file encoding dropdown to view available encoding options.", "examples/mousepad_encoding_dropdown.png"], ["Open the Image Viewer menu.", "examples/image_viewer_menu.png"], ["Open the Applications menu to view application categories.", "examples/applications_menu.png"], ["Enable the sidebar in the file manager.", "examples/nautilus_show_sidebar.png"], ["Select the Sort by Size option in the Transmission View menu.", "examples/transmission_sort_by_size.png"], ], inputs=[instruction, screenshot_in], fn=predict, outputs=[screenshot_out, action_out, coords_out, time_out], cache_examples=True, cache_mode="lazy", ) run_btn.click( predict, inputs=[instruction, screenshot_in, max_new_tokens], outputs=[screenshot_out, action_out, coords_out, time_out], api_name="predict", ) if __name__ == "__main__": demo.launch(mcp_server=True, theme=gr.themes.Citrus(), css=CSS)