""" AI Vision GUI Agent — n8n-Compatible FastAPI Each endpoint = one n8n node. Output feeds into next node. """ from fastapi import FastAPI, UploadFile, File, HTTPException from fastapi.responses import JSONResponse from pydantic import BaseModel from PIL import Image, ImageDraw, ImageFilter import base64, io, json, re, time, requests, os from typing import Optional, List app = FastAPI(title="Vision GUI Agent — n8n Flow", version="1.0.0") GROQ_KEY = os.environ.get("GROQ_KEY", "") GROQ_URL = "https://api.groq.com/openai/v1/chat/completions" # ───────────────────────────────────────────── # NODE 1: /capture — Accepts screenshot upload # ───────────────────────────────────────────── @app.post("/node/capture") async def node_capture(file: UploadFile = File(...)): """ NODE 1 — Screen Capture Input Accepts a screenshot, returns base64 + metadata. n8n: HTTP Request node → POST this endpoint with image. """ data = await file.read() img = Image.open(io.BytesIO(data)).convert("RGB") w, h = img.size buf = io.BytesIO() img.save(buf, format="PNG") b64 = base64.b64encode(buf.getvalue()).decode() return { "node": "capture", "status": "ok", "width": w, "height": h, "image_b64": b64, "timestamp": time.time() } # ───────────────────────────────────────────── # NODE 2: /detect — Find UI elements by color/region # ───────────────────────────────────────────── class DetectInput(BaseModel): image_b64: str target_colors: Optional[List[str]] = ["#FF90E8", "#000000", "#FFFFFF", "#0066FF"] @app.post("/node/detect") async def node_detect(body: DetectInput): """ NODE 2 — Element Detection Finds clickable regions by color clustering. n8n: HTTP Request node → receives capture output, sends here. """ img_bytes = base64.b64decode(body.image_b64) img = Image.open(io.BytesIO(img_bytes)).convert("RGB") w, h = img.size # Detect high-contrast regions as potential UI elements gray = img.convert("L") elements = [] # Sample grid for color-distinct regions grid = 40 for y in range(0, h - grid, grid): for x in range(0, w - grid, grid): region = img.crop((x, y, x + grid, y + grid)) pixels = list(region.getdata()) avg_r = sum(p[0] for p in pixels) // len(pixels) avg_g = sum(p[1] for p in pixels) // len(pixels) avg_b = sum(p[2] for p in pixels) // len(pixels) brightness = (avg_r + avg_g + avg_b) / 3 # Flag as likely button if colorful and not background gray is_colorful = max(avg_r, avg_g, avg_b) - min(avg_r, avg_g, avg_b) > 40 is_not_white = brightness < 230 is_not_black = brightness > 20 if is_colorful and is_not_white and is_not_black: elements.append({ "x": x + grid // 2, "y": y + grid // 2, "color": f"rgb({avg_r},{avg_g},{avg_b})", "brightness": round(brightness, 1), "type": "button_candidate" }) # Deduplicate nearby points filtered = [] for e in elements: if not any(abs(e["x"] - f["x"]) < 60 and abs(e["y"] - f["y"]) < 60 for f in filtered): filtered.append(e) return { "node": "detect", "status": "ok", "elements_found": len(filtered), "elements": filtered[:20], # top 20 candidates "image_b64": body.image_b64 # pass-through for next node } # ───────────────────────────────────────────── # NODE 3: /decide — LLM picks the action # ───────────────────────────────────────────── class DecideInput(BaseModel): task: str elements: List[dict] image_b64: Optional[str] = None safety_mode: bool = True # HITL for medical apps @app.post("/node/decide") async def node_decide(body: DecideInput): """ NODE 3 — LLM Decision (Groq) Given task + detected elements → decide what to click/type. n8n: HTTP Request node → receives detect output + task string. """ if not GROQ_KEY: # Fallback: pick first element if body.elements: el = body.elements[0] return { "node": "decide", "action": "click", "target_x": el["x"], "target_y": el["y"], "reasoning": "No LLM key — defaulting to first element", "safe_to_execute": not body.safety_mode } raise HTTPException(400, "No elements and no GROQ_KEY") el_summary = json.dumps(body.elements[:10], indent=2) prompt = f"""You are a GUI navigation agent. Task: {body.task} Detected UI elements (x, y coords + color): {el_summary} Respond with JSON only: {{ "action": "click" | "type" | "scroll" | "wait", "target_x": , "target_y": , "text": "", "reasoning": "", "confidence": <0.0-1.0>, "safe_to_execute": }}""" resp = requests.post(GROQ_URL, headers={ "Authorization": f"Bearer {GROQ_KEY}", "Content-Type": "application/json" }, json={ "model": "llama-3.3-70b-versatile", "messages": [{"role": "user", "content": prompt}], "temperature": 0.1, "max_tokens": 300 }, timeout=15) raw = resp.json()["choices"][0]["message"]["content"] # Extract JSON from response match = re.search(r'\{.*\}', raw, re.DOTALL) decision = json.loads(match.group()) if match else {"action": "wait", "reasoning": raw} decision["node"] = "decide" # Safety override for medical tasks if body.safety_mode and not decision.get("safe_to_execute", True): decision["blocked"] = True decision["reason"] = "HITL required — sensitive action flagged" return decision # ───────────────────────────────────────────── # NODE 4: /execute — Simulate or real execution # ───────────────────────────────────────────── class ExecuteInput(BaseModel): action: str target_x: Optional[int] = None target_y: Optional[int] = None text: Optional[str] = None dry_run: bool = True # default: simulate only blocked: Optional[bool] = False @app.post("/node/execute") async def node_execute(body: ExecuteInput): """ NODE 4 — Action Executor dry_run=True → simulate (safe). dry_run=False → real pyautogui. n8n: HTTP Request node → receives decide output. """ if body.blocked: return {"node": "execute", "status": "blocked", "reason": "HITL required"} if body.dry_run: return { "node": "execute", "status": "simulated", "action": body.action, "target": {"x": body.target_x, "y": body.target_y}, "text": body.text, "message": f"DRY RUN: Would {body.action} at ({body.target_x}, {body.target_y})" } # Real execution (only when dry_run=False and running locally with display) try: import pyautogui pyautogui.FAILSAFE = True if body.action == "click" and body.target_x and body.target_y: pyautogui.click(body.target_x, body.target_y) elif body.action == "type" and body.text: pyautogui.typewrite(body.text, interval=0.05) elif body.action == "scroll": pyautogui.scroll(3) return {"node": "execute", "status": "executed", "action": body.action} except Exception as e: return {"node": "execute", "status": "error", "error": str(e)} # ───────────────────────────────────────────── # NODE 5: /verify — Compare before/after screenshots # ───────────────────────────────────────────── class VerifyInput(BaseModel): before_b64: str after_b64: str threshold: float = 0.02 # 2% pixel change = success @app.post("/node/verify") async def node_verify(body: VerifyInput): """ NODE 5 — Verify Action Succeeded Compares before/after screenshots. Reports pixel diff %. n8n: HTTP Request node — chain after execute. """ before = Image.open(io.BytesIO(base64.b64decode(body.before_b64))).convert("RGB") after = Image.open(io.BytesIO(base64.b64decode(body.after_b64))).convert("RGB") if before.size != after.size: after = after.resize(before.size) b_px = list(before.getdata()) a_px = list(after.getdata()) total = len(b_px) changed = sum(1 for b, a in zip(b_px, a_px) if abs(b[0]-a[0]) + abs(b[1]-a[1]) + abs(b[2]-a[2]) > 30) diff_pct = changed / total success = diff_pct >= body.threshold return { "node": "verify", "status": "ok", "diff_percent": round(diff_pct * 100, 2), "changed_pixels": changed, "total_pixels": total, "success": success, "verdict": "CHANGED ✅" if success else "NO CHANGE ⚠️ — may need retry" } # ───────────────────────────────────────────── # NODE 6: /report — Final output / notification # ───────────────────────────────────────────── class ReportInput(BaseModel): task: str results: List[dict] notify_webhook: Optional[str] = None @app.post("/node/report") async def node_report(body: ReportInput): """ NODE 6 — Summary Report + Optional Webhook Notify n8n: Final node — sends summary back to n8n or Slack/email. """ success_nodes = [r for r in body.results if r.get("status") in ["ok","executed","simulated"]] failed_nodes = [r for r in body.results if r.get("status") in ["error","blocked"]] report = { "node": "report", "task": body.task, "total_nodes": len(body.results), "succeeded": len(success_nodes), "failed": len(failed_nodes), "overall": "SUCCESS ✅" if not failed_nodes else "PARTIAL ⚠️", "summary": [{"node": r.get("node"), "status": r.get("status")} for r in body.results], "timestamp": time.strftime("%Y-%m-%d %H:%M:%S UTC", time.gmtime()) } # Optional: fire webhook (Slack, n8n, etc.) if body.notify_webhook: try: requests.post(body.notify_webhook, json=report, timeout=5) report["webhook_fired"] = True except: report["webhook_fired"] = False return report # ───────────────────────────────────────────── # HEALTH CHECK # ───────────────────────────────────────────── @app.get("/health") def health(): return {"status": "healthy", "nodes": ["capture","detect","decide","execute","verify","report"]} @app.get("/") def root(): return { "name": "AI Vision GUI Agent — n8n Flow", "docs": "/docs", "flow": "capture → detect → decide → execute → verify → report" }