#!/usr/bin/env python3 """Render Miril-DroneVLM-2B-2 JSON over a video with safe pointing and OpenCV tracking.""" from __future__ import annotations import argparse import json import shutil import subprocess import tempfile from dataclasses import dataclass from pathlib import Path from typing import Any import cv2 import numpy as np from PIL import Image, ImageDraw, ImageFont from inference import generate, load_model, parse_bare_json from router_contract import drawable_point GREEN = (42, 255, 128) WHITE = (232, 238, 235) PANEL = (7, 10, 11) @dataclass class Response: prompt: str label: str text: str payload: dict[str, Any] | None errors: list[str] def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser() parser.add_argument("--input", required=True) parser.add_argument("--output", required=True) parser.add_argument("--prompt", action="append", required=True) parser.add_argument("--model-id", default="MirilAI/Miril-DroneVLM-2B-2") parser.add_argument("--processor-id") parser.add_argument("--interval-seconds", type=float, default=5.0) parser.add_argument( "--crop", choices=("native", "center-square"), default="center-square" ) parser.add_argument("--crop-size", type=int, default=1000) parser.add_argument("--max-seconds", type=float, default=0.0) parser.add_argument("--max-new-tokens", type=int, default=384) parser.add_argument("--load-4bit", action="store_true") return parser.parse_args() def process_frame(frame: np.ndarray, crop: str, crop_size: int) -> Image.Image: image = Image.fromarray(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)) if crop == "native": return image width, height = image.size side = min(width, height) left = (width - side) // 2 top = (height - side) // 2 image = image.crop((left, top, left + side, top + side)) return image.resize((crop_size, crop_size), Image.Resampling.LANCZOS) def response_label(payload: dict[str, Any] | None, index: int) -> str: if not payload: return f"Prompt {index}" response_type = payload.get("type") if response_type == "caption": return "Scene" if response_type == "answer": return "Answer" if response_type == "location": return f"Location/{payload.get('intent', 'unknown')}" if response_type == "pointing": return f"Pointing/{payload.get('action', 'unknown')}" return f"Prompt {index}" def response_text(raw: str, payload: dict[str, Any] | None, errors: list[str]) -> str: if errors or payload is None: return f"REJECTED: {'; '.join(errors) or 'invalid JSON'}" response_type = payload["type"] if response_type == "caption": return payload["caption"] if response_type == "answer": return payload["answer"] point = drawable_point(payload) if point is None: return f"{payload['status']} / no point: {payload['caption']}" x, y, mode = point mode_text = "precise target" if mode == "precise_target" else "broad direction only" return f"{payload['status']} / {mode_text} / yx=[{y:.0f}, {x:.0f}]: {payload['caption']}" def infer_responses( model: Any, processor: Any, image: Image.Image, prompts: list[str], max_new_tokens: int, ) -> list[Response]: responses: list[Response] = [] for index, prompt in enumerate(prompts, start=1): raw = generate( model, processor, image, prompt, max_new_tokens=max_new_tokens, ) payload, errors = parse_bare_json(raw) responses.append( Response( prompt=prompt, label=response_label(payload, index), text=response_text(raw, payload, errors), payload=payload, errors=errors, ) ) return responses def find_visual_payload( responses: list[Response], ) -> tuple[dict[str, Any] | None, dict[str, Any] | None]: precise = None coarse = None for response in responses: point = drawable_point(response.payload) if point is None: continue if point[2] == "precise_target" and precise is None: precise = response.payload elif point[2] == "coarse_direction" and coarse is None: coarse = response.payload return precise, coarse def font(size: int, bold: bool = False) -> ImageFont.ImageFont: names = ( "/usr/share/fonts/truetype/dejavu/DejaVuSansMono-Bold.ttf" if bold else "/usr/share/fonts/truetype/dejavu/DejaVuSansMono.ttf", "/Library/Fonts/Arial Bold.ttf" if bold else "/Library/Fonts/Arial.ttf", ) for name in names: try: return ImageFont.truetype(name, size) except Exception: pass return ImageFont.load_default() def wrap( draw: ImageDraw.ImageDraw, text: str, face: ImageFont.ImageFont, width: int ) -> list[str]: words = text.split() if not words: return [""] lines: list[str] = [] current = words[0] for word in words[1:]: candidate = f"{current} {word}" if draw.textbbox((0, 0), candidate, font=face)[2] <= width: current = candidate else: lines.append(current) current = word lines.append(current) return lines def visible_text(text: str, reveal_index: int) -> str: if reveal_index >= 12: return text words = text.split() keep = max(1, (len(words) * (reveal_index + 1) + 11) // 12) return " ".join(words[:keep]) def layout_lines( draw: ImageDraw.ImageDraw, responses: list[Response], panel_width: int, height: int, ) -> tuple[ImageFont.ImageFont, ImageFont.ImageFont, int]: available_width = panel_width - 36 for size in range(19, 8, -1): regular = font(size) bold = font(size, bold=True) line_height = size + 6 line_count = 0 for response in responses: line_count += len( wrap( draw, f"{response.label} query: {response.prompt}", regular, available_width, ) ) line_count += len( wrap( draw, f"{response.label}: {response.text}", regular, available_width ) ) line_count += 1 if line_count * line_height <= height - 70: return regular, bold, line_height return font(9), font(9, bold=True), 15 def draw_marker( draw: ImageDraw.ImageDraw, payload: dict[str, Any], width: int, height: int, *, tracked_center: tuple[float, float] | None, ) -> None: point = drawable_point(payload) if point is None: return x, y, mode = point px = int(round(x / 1000.0 * width)) py = int(round(y / 1000.0 * height)) if mode == "precise_target" and tracked_center is not None: px, py = map(lambda value: int(round(value)), tracked_center) if mode == "coarse_direction": radius = max(40, min(width, height) // 11) draw.ellipse( (px - radius, py - radius, px + radius, py + radius), fill=(0, 230, 120, 34), outline=(0, 230, 120, 225), width=3, ) return half = 25 radius = 31 draw.ellipse( (px - radius, py - radius, px + radius, py + radius), fill=(0, 230, 120, 40), outline=(0, 230, 120, 230), width=2, ) draw.rectangle( (px - half, py - half, px + half, py + half), outline=(0, 230, 120), width=2 ) draw.line((px - 10, py, px + 10, py), fill=(0, 230, 120), width=2) draw.line((px, py - 10, px, py + 10), fill=(0, 230, 120), width=2) def render( image: Image.Image, responses: list[Response], *, tracked_center: tuple[float, float] | None, precise_payload: dict[str, Any] | None, coarse_payload: dict[str, Any] | None, reveal_index: int, ) -> Image.Image: width, height = image.size panel_width = width // 2 canvas = Image.new("RGB", (width + panel_width, height), PANEL) layer = Image.new("RGBA", image.size, (0, 0, 0, 0)) marker_draw = ImageDraw.Draw(layer) if precise_payload is not None and tracked_center is not None: draw_marker( marker_draw, precise_payload, width, height, tracked_center=tracked_center ) elif coarse_payload is not None and reveal_index < 12: draw_marker(marker_draw, coarse_payload, width, height, tracked_center=None) canvas.paste( Image.alpha_composite(image.convert("RGBA"), layer).convert("RGB"), (0, 0) ) draw = ImageDraw.Draw(canvas) regular, bold, line_height = layout_lines(draw, responses, panel_width, height) max_width = panel_width - 36 y = 18 overflow = False for response in responses: blocks = ( (f"{response.label} query:", response.prompt), (f"{response.label}:", visible_text(response.text, reveal_index)), ) for prefix, body in blocks: first = True for line in wrap(draw, f"{prefix} {body}", regular, max_width): if y + line_height > height - 48: draw.text((width + 18, y), "...", fill=WHITE, font=regular) overflow = True break if first and line.startswith(prefix): draw.text((width + 18, y), prefix, fill=GREEN, font=bold) prefix_width = draw.textbbox((0, 0), prefix, font=bold)[2] draw.text( (width + 22 + prefix_width, y), line[len(prefix) :].lstrip(), fill=WHITE, font=regular, ) else: draw.text((width + 18, y), line, fill=WHITE, font=regular) first = False y += line_height if overflow: break if overflow: break y += line_height brand = "Miril-DroneVLM-2B-2 / miril.ai" small = font(max(10, min(15, height // 55))) brand_width = draw.textbbox((0, 0), brand, font=small)[2] draw.text( (width + panel_width - brand_width - 18, height - 30), brand, fill=WHITE, font=small, ) return canvas def initial_center( payload: dict[str, Any] | None, width: int, height: int ) -> tuple[float, float] | None: point = drawable_point(payload) if point is None or point[2] != "precise_target": return None center = (point[0] / 1000.0 * width, point[1] / 1000.0 * height) return center if inside_margin(center, width, height) else None def inside_margin( center: tuple[float, float], width: int, height: int, margin: int = 50 ) -> bool: return margin < center[0] < width - margin and margin < center[1] < height - margin def track_center( previous: Image.Image, current: Image.Image, center: tuple[float, float], ) -> tuple[float, float] | None: width, height = current.size if not inside_margin(center, width, height): return None half = 50 left = int(round(center[0])) - half top = int(round(center[1])) - half right = left + 100 bottom = top + 100 if left < 0 or top < 0 or right > width or bottom > height: return None points = np.asarray( [ [x, y] for y in np.linspace(top + 12, bottom - 12, 4) for x in np.linspace(left + 12, right - 12, 4) ], dtype=np.float32, ).reshape(-1, 1, 2) previous_gray = cv2.cvtColor(np.asarray(previous), cv2.COLOR_RGB2GRAY) current_gray = cv2.cvtColor(np.asarray(current), cv2.COLOR_RGB2GRAY) moved, status, _ = cv2.calcOpticalFlowPyrLK( previous_gray, current_gray, points, None ) if moved is None or status is None: return None mask = status.reshape(-1) == 1 if int(mask.sum()) < 4: return None delta = np.median(moved[mask].reshape(-1, 2) - points[mask].reshape(-1, 2), axis=0) updated = (float(center[0] + delta[0]), float(center[1] + delta[1])) return updated if inside_margin(updated, width, height) else None def mux_h264(silent_path: Path, source_path: Path, output_path: Path) -> None: ffmpeg = shutil.which("ffmpeg") if ffmpeg is None: shutil.move(silent_path, output_path) return subprocess.run( [ ffmpeg, "-y", "-i", str(silent_path), "-i", str(source_path), "-map", "0:v:0", "-map", "1:a?", "-c:v", "libx264", "-crf", "20", "-preset", "medium", "-c:a", "aac", "-shortest", str(output_path), ], check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, ) def main() -> int: args = parse_args() prompts = [prompt.strip() for prompt in args.prompt if prompt.strip()][:3] if not prompts: raise ValueError("at least one non-empty --prompt is required") if args.interval_seconds < 1 or args.interval_seconds > 10: raise ValueError("--interval-seconds must be from 1 through 10") source_path = Path(args.input) output_path = Path(args.output) output_path.parent.mkdir(parents=True, exist_ok=True) model, processor = load_model( args.model_id, args.processor_id or args.model_id, args.load_4bit, ) capture = cv2.VideoCapture(str(source_path)) if not capture.isOpened(): raise ValueError(f"could not open {source_path}") fps = float(capture.get(cv2.CAP_PROP_FPS) or 24.0) max_frames = int(args.max_seconds * fps) if args.max_seconds > 0 else None interval_frames = max(1, int(round(args.interval_seconds * fps))) temporary_dir = Path(tempfile.mkdtemp(prefix="miril_drone_overlay_")) silent_path = temporary_dir / "silent.mp4" writer: Any = None previous: Image.Image | None = None responses: list[Response] = [] precise_payload = None coarse_payload = None center = None segment_index = 0 frame_index = 0 sidecar_samples: list[dict[str, Any]] = [] while max_frames is None or frame_index < max_frames: ok, frame = capture.read() if not ok: break image = process_frame(frame, args.crop, args.crop_size) if frame_index % interval_frames == 0 or not responses: responses = infer_responses( model, processor, image, prompts, args.max_new_tokens, ) precise_payload, coarse_payload = find_visual_payload(responses) center = initial_center(precise_payload, image.width, image.height) segment_index = 0 sidecar_samples.append( { "timestamp_s": round(frame_index / fps, 3), "responses": [ { "prompt": response.prompt, "payload": response.payload, "validation_errors": response.errors, } for response in responses ], } ) elif center is not None and previous is not None: center = track_center(previous, image, center) rendered = render( image, responses, tracked_center=center, precise_payload=precise_payload, coarse_payload=coarse_payload, reveal_index=segment_index, ) if writer is None: writer = cv2.VideoWriter( str(silent_path), cv2.VideoWriter_fourcc(*"mp4v"), fps, rendered.size, ) writer.write(cv2.cvtColor(np.asarray(rendered), cv2.COLOR_RGB2BGR)) previous = image segment_index += 1 frame_index += 1 capture.release() if writer is not None: writer.release() if frame_index == 0: raise ValueError("no frames were rendered") mux_h264(silent_path, source_path, output_path) sidecar = { "model_id": args.model_id, "input": str(source_path), "output": str(output_path), "prompts": prompts, "interval_seconds": args.interval_seconds, "source_fps": fps, "rendered_frames": frame_index, "tracking_box_px": 100, "edge_hide_margin_px": 50, "samples": sidecar_samples, } output_path.with_suffix(output_path.suffix + ".json").write_text( json.dumps(sidecar, indent=2, ensure_ascii=False) + "\n", encoding="utf-8", ) shutil.rmtree(temporary_dir, ignore_errors=True) return 0 if __name__ == "__main__": raise SystemExit(main())