multimodalart's picture
multimodalart HF Staff
Upload app.py with huggingface_hub
fe63442 verified
Raw History Blame Contribute Delete
9.14 kB
import spaces # MUST come before torch / transformers (ZeroGPU)
import re
import time
import gradio as gr
import torch
from PIL import Image, ImageDraw
from transformers import AutoModelForImageTextToText, AutoProcessor
MODEL = "docling-project/DeskForge-Gemma4-E4B"
MAX_PIXELS = 2_097_152 # training-time cap on screenshot area
SYSTEM_PROMPT = """You are a computer-use agent operating a desktop graphical interface. At each step you see the user's task, a screenshot of the current screen, and the actions you have already taken. Reply with the next action as pyautogui code and nothing else -- no explanation, no code fence, no commentary.
Coordinates are fractions of the screen, not pixels: x runs from 0.0 at the left edge to 1.0 at the right edge, y from 0.0 at the top to 1.0 at the bottom. Write both with four decimals.
These are the only actions available:
pyautogui.click(x=0.0000, y=0.0000)
pyautogui.doubleClick(x=0.0000, y=0.0000)
pyautogui.rightClick(x=0.0000, y=0.0000)
pyautogui.middleClick(x=0.0000, y=0.0000)
computer.tripleClick(x=0.0000, y=0.0000)
pyautogui.moveTo(x=0.0000, y=0.0000)
pyautogui.dragTo(x=0.0000, y=0.0000, button='left')
pyautogui.scroll(-4)
pyautogui.hscroll(4)
pyautogui.write(message='text to type')
pyautogui.press('enter')
pyautogui.hotkey(['ctrl', 'c'])
computer.wait()
computer.terminate(status='success')
To scroll at a particular place, move there first and then scroll. When the task is finished, or cannot be finished, end with computer.terminate."""
def fit(image, max_pixels=MAX_PIXELS):
"""Downscale to at most max_pixels (bilinear, sides rounded down), as in training."""
w, h = image.size
if w * h <= max_pixels:
return image
ratio = w / h
height = (max_pixels / ratio) ** 0.5
return image.resize(
(max(1, int(height * ratio)), max(1, int(height))), Image.BILINEAR
)
def user_text(instruction):
return f"Task: {instruction}\n\nActions already taken:\n(none -- this is the first step)\n\nNext action:"
def chat(instruction):
return [
{"role": "system", "content": SYSTEM_PROMPT},
{
"role": "user",
"content": [{"type": "image"}, {"type": "text", "text": user_text(instruction)}],
},
]
POINT_RE = re.compile(r"x=([\d.]+), y=([\d.]+)")
def to_pixels(action, screenshot):
"""Fractional screen coords -> pixel coords on the original screenshot."""
m = POINT_RE.search(action)
if not m:
return None
x, y = map(float, m.groups())
return round(x * screenshot.width), round(y * screenshot.height)
def draw_marker(image, point):
"""Draw a crosshair + circle at the predicted click point.
A white underlay keeps the marker visible on both light and dark UIs.
"""
out = image.copy()
d = ImageDraw.Draw(out)
x, y = point
r = max(10, min(out.width, out.height) // 60)
# white underlay, then orange marker on top
d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 255, 255), width=6)
d.ellipse([x - r, y - r, x + r, y + r], outline=(255, 64, 0), width=3)
for (x0, y0, x1, y1) in [
(x - r * 1.8, y, x + r * 1.8, y), # horizontal
(x, y - r * 1.8, x, y + r * 1.8), # vertical
]:
d.line([x0, y0, x1, y1], fill=(255, 255, 255), width=9)
d.line([x0, y0, x1, y1], fill=(255, 64, 0), width=4)
return out
processor = AutoProcessor.from_pretrained(MODEL)
model = (
AutoModelForImageTextToText.from_pretrained(MODEL, dtype=torch.bfloat16)
.eval()
.to("cuda")
)
@spaces.GPU(duration=30)
def predict(instruction: str, screenshot, max_new_tokens: int = 128):
"""Predict the next computer-use action for a desktop screenshot.
Args:
instruction: the task to perform on the screen, e.g. "Open the File menu".
screenshot: a PIL image of the desktop to act on.
max_new_tokens: generation length cap for the emitted action.
Returns:
Tuple of (annotated screenshot with the predicted click point,
predicted action code, fractional coordinates, inference seconds).
"""
if screenshot is None:
raise gr.Error("Please provide a desktop screenshot first.")
if not instruction or not instruction.strip():
raise gr.Error("Please enter an instruction, e.g. 'Open the File menu'.")
if screenshot.mode != "RGB":
screenshot = screenshot.convert("RGB")
prompt = processor.apply_chat_template(
chat(instruction.strip()), tokenize=False, add_generation_prompt=True,
enable_thinking=False,
)
inputs = processor(
text=[prompt], images=[fit(screenshot)], return_tensors="pt"
).to(model.device)
t0 = time.perf_counter()
with torch.inference_mode():
output = model.generate(**inputs, max_new_tokens=int(max_new_tokens), do_sample=False)
elapsed = time.perf_counter() - t0
action = processor.decode(
output[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True
).strip()
point = to_pixels(action, screenshot)
annotated = draw_marker(screenshot, point) if point else screenshot.copy()
m = POINT_RE.search(action)
coords = f"x={m.group(1)}, y={m.group(2)}" if m else "— (no point action)"
return annotated, action, coords, f"{elapsed:.1f}s"
CSS = """
#col-container { max-width: 1200px; margin: 0 auto; }
.dark .gradio-container { color: var(--body-text-color); }
"""
with gr.Blocks() as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(
"""
# 🖱️ DeskForge-Gemma4-E4B · GUI Grounding
**DeskForge-Gemma4-E4B** ([model card](https://huggingface.co/docling-project/DeskForge-Gemma4-E4B))
is [Gemma 4 E4B](https://huggingface.co/google/gemma-4-E4B-it) fine-tuned on 200K grounding
examples from [DeskForge-1M](https://huggingface.co/datasets/docling-project/DeskForge-1M).
Given a desktop screenshot and a task, it replies with the next action as
`pyautogui` code, with coordinates as **fractions of the screen**
(x: 0.0 left → 1.0 right, y: 0.0 top → 1.0 bottom). The predicted click
point is drawn on the screenshot.
Paper: [DeskForge: Dense Supervision from Desktop Environments for Computer-Use Agents](https://arxiv.org/abs/2610.02320) ·
[Project page](https://saidgurbuz.github.io/deskforge/) · [GitHub](https://github.com/Saidgurbuz/deskforge)
"""
)
with gr.Row():
with gr.Column(scale=1):
screenshot_in = gr.Image(
type="pil", label="Desktop screenshot", height=420,
sources=["upload", "clipboard"],
)
instruction = gr.Textbox(
label="Task / instruction",
placeholder="e.g. Open the File menu",
lines=2,
)
run_btn = gr.Button("Ground it", variant="primary")
with gr.Accordion("Advanced", open=False):
max_new_tokens = gr.Slider(
16, 256, value=128, step=8, label="Max new tokens",
)
with gr.Column(scale=1):
screenshot_out = gr.Image(
type="pil", label="Annotated screenshot (predicted click point)",
height=420,
)
action_out = gr.Textbox(label="Predicted action (pyautogui)", lines=2)
coords_out = gr.Textbox(label="Fractional coordinates", lines=1)
time_out = gr.Textbox(label="Inference time", lines=1)
gr.Examples(
examples=[
["Select the Export as QIF... option from the File menu.", "examples/homebank_file_menu.png"],
["Add the factorial function to the current expression.", "examples/qalculate_factorial.png"],
["Select Financial Mode in the calculator's mode dropdown.", "examples/calculator_financial_mode.png"],
["Open the file encoding dropdown to view available encoding options.", "examples/mousepad_encoding_dropdown.png"],
["Open the Image Viewer menu.", "examples/image_viewer_menu.png"],
["Open the Applications menu to view application categories.", "examples/applications_menu.png"],
["Enable the sidebar in the file manager.", "examples/nautilus_show_sidebar.png"],
["Select the Sort by Size option in the Transmission View menu.", "examples/transmission_sort_by_size.png"],
],
inputs=[instruction, screenshot_in],
fn=predict,
outputs=[screenshot_out, action_out, coords_out, time_out],
cache_examples=True,
cache_mode="lazy",
)
run_btn.click(
predict,
inputs=[instruction, screenshot_in, max_new_tokens],
outputs=[screenshot_out, action_out, coords_out, time_out],
api_name="predict",
)
if __name__ == "__main__":
demo.launch(mcp_server=True, theme=gr.themes.Citrus(), css=CSS)