""" Gradio UI wrapping the 6-node FastAPI agent flow. Lets Ting test each node visually, or run the full flow. """ import gradio as gr import base64, io, json, requests from PIL import Image # Start FastAPI in background thread import threading, uvicorn, sys, os sys.path.insert(0, os.path.dirname(__file__)) from app import app as fastapi_app def start_api(): uvicorn.run(fastapi_app, host="127.0.0.1", port=7861, log_level="error") thread = threading.Thread(target=start_api, daemon=True) thread.start() BASE = "http://127.0.0.1:7861" def run_full_flow(image, task, safety_mode, dry_run): if image is None: return "❌ Please upload a screenshot first.", "{}" # Encode image buf = io.BytesIO() image.save(buf, format="PNG") buf.seek(0) b64 = base64.b64encode(buf.getvalue()).decode() results = [] log = [] # NODE 1 — Capture (simulate with uploaded image) log.append("▶ NODE 1: Capture...") cap_result = {"node": "capture", "status": "ok", "image_b64": b64, "width": image.width, "height": image.height} results.append(cap_result) log.append(f" ✅ {image.width}×{image.height} image received") # NODE 2 — Detect log.append("▶ NODE 2: Detecting elements...") try: r = requests.post(f"{BASE}/node/detect", json={"image_b64": b64}, timeout=20) det = r.json() results.append(det) log.append(f" ✅ {det.get('elements_found', 0)} elements found") except Exception as e: det = {"elements": []} log.append(f" ⚠️ Detect error: {e}") # NODE 3 — Decide log.append("▶ NODE 3: LLM deciding action (Groq)...") try: r = requests.post(f"{BASE}/node/decide", json={ "task": task, "elements": det.get("elements", []), "safety_mode": safety_mode }, timeout=30) dec = r.json() results.append(dec) log.append(f" ✅ Action: {dec.get('action')} at ({dec.get('target_x')}, {dec.get('target_y')})") log.append(f" 💬 {dec.get('reasoning', '')}") if dec.get("blocked"): log.append(f" 🚨 BLOCKED — HITL required") except Exception as e: dec = {"action": "wait", "blocked": False} log.append(f" ⚠️ Decide error: {e}") # NODE 4 — Execute log.append(f"▶ NODE 4: Executing ({'DRY RUN' if dry_run else 'REAL'})...") try: r = requests.post(f"{BASE}/node/execute", json={ "action": dec.get("action", "wait"), "target_x": dec.get("target_x"), "target_y": dec.get("target_y"), "text": dec.get("text"), "dry_run": dry_run, "blocked": dec.get("blocked", False) }, timeout=15) exe = r.json() results.append(exe) log.append(f" ✅ {exe.get('message') or exe.get('status')}") except Exception as e: exe = {"status": "error"} log.append(f" ⚠️ Execute error: {e}") # NODE 5 — Verify (self-compare since no before/after in demo) log.append("▶ NODE 5: Verifying...") results.append({"node": "verify", "status": "ok", "verdict": "DEMO MODE — no real before/after", "diff_percent": 0}) log.append(" ✅ Verify complete (demo mode)") # NODE 6 — Report log.append("▶ NODE 6: Generating report...") try: r = requests.post(f"{BASE}/node/report", json={ "task": task, "results": results }, timeout=15) rep = r.json() log.append(f" ✅ {rep.get('overall')}") except Exception as e: rep = {"overall": "ERROR"} log.append(f" ⚠️ Report error: {e}") return "\n".join(log), json.dumps(rep, indent=2) # Draw detected elements on image def detect_only(image): if image is None: return None, "Upload an image first" buf = io.BytesIO() image.save(buf, format="PNG") b64 = base64.b64encode(buf.getvalue()).decode() try: r = requests.post(f"{BASE}/node/detect", json={"image_b64": b64}, timeout=20) det = r.json() # Draw boxes out = image.copy() from PIL import ImageDraw draw = ImageDraw.Draw(out) for el in det.get("elements", []): x, y = el["x"], el["y"] draw.ellipse([x-12, y-12, x+12, y+12], outline="red", width=3) draw.text((x+14, y-8), el.get("type","?"), fill="red") return out, f"Found {det.get('elements_found', 0)} elements" except Exception as e: return image, f"Error: {e}" with gr.Blocks(title="Vision GUI Agent — n8n Flow", theme=gr.themes.Soft()) as demo: gr.Markdown(""" # 🤖 AI Vision GUI Agent — n8n-Compatible Flow **6 nodes. Each runs in sequence like an n8n workflow.** `📸 Capture → 🔍 Detect → 🧠 Decide → ⚡ Execute → ✅ Verify → 📊 Report` """) with gr.Tab("🚀 Full Flow"): with gr.Row(): img_in = gr.Image(type="pil", label="Upload Screenshot") with gr.Column(): task_box = gr.Textbox(label="Task", placeholder="e.g. Click the submit button") safety_chk = gr.Checkbox(label="🛡️ Safety Mode (HITL for medical)", value=True) dry_chk = gr.Checkbox(label="🔄 Dry Run (simulate only)", value=True) run_btn = gr.Button("▶ Run All 6 Nodes", variant="primary") log_out = gr.Textbox(label="📋 Flow Log", lines=14) report_out = gr.Code(label="📊 Final Report (JSON)", language="json") run_btn.click(run_full_flow, [img_in, task_box, safety_chk, dry_chk], [log_out, report_out]) with gr.Tab("🔍 Detect Only"): with gr.Row(): detect_img_in = gr.Image(type="pil", label="Upload Screenshot") detect_img_out = gr.Image(label="Detected Elements (red dots)") detect_status = gr.Textbox(label="Status") gr.Button("🔍 Detect Elements").click(detect_only, [detect_img_in], [detect_img_out, detect_status]) with gr.Tab("📖 n8n Import Guide"): gr.Markdown(""" ## Import this flow into n8n 1. **Download** `vision_agent_n8n_workflow.json` from the Files tab 2. Open your **n8n instance** → Workflows → **Import from file** 3. Replace `YOUR-HF-SPACE.hf.space` with this Space's URL in each node 4. **Set GROQ_KEY** in this Space's Settings → Variables 5. Activate the workflow → POST to the webhook trigger URL: ```json { "task": "Click the blue Submit button", "screenshot_b64": "" } ``` ## Node Map | n8n Node | Endpoint | What it does | |----------|----------|-------------| | 📸 Capture | `/node/capture` | Accept screenshot | | 🔍 Detect | `/node/detect` | Find UI elements | | 🧠 Decide | `/node/decide` | Groq picks action | | 🛡️ Safety Gate | n8n IF node | Block medical risk | | ⚡ Execute | `/node/execute` | Run or simulate | | ✅ Verify | `/node/verify` | Check success | | 📊 Report | `/node/report` | Final summary | """) demo.launch(server_name="0.0.0.0", server_port=7860)