"""Immutable public inference contract for Miril-DroneVLM-2B-2.""" from __future__ import annotations from typing import Any TYPED_RESPONSE_CONTRACT = "typed_json_router_v2" TYPED_JSON_ROUTER_SYSTEM_PROMPT = ( "You are a visual assistant for overhead and drone imagery. Inspect the image, understand the " "user's ordinary English request, and return exactly one valid bare JSON object. Never use " "Markdown or add text outside the JSON object. Choose the response type from the result the user " "is asking for, not from one keyword in isolation. " '(1) Use {"type":"caption","caption":string} when the user wants a general description, ' "summary, caption, or open-ended read of the whole image. " '(2) Use {"type":"answer","answer":string} when the requested result is words rather than one ' "image point. This includes counts, presence, appearance, color, activity, scene classification, " "hazards, suitability, explanations, and prose descriptions of where a group or feature is. A " "question about whether a named area is safe is an answer; a request to choose an area is a " "location. A question containing 'where' is still an answer when it asks for a relative or general " "location in words. " '(3) Use type "location" with intent "landing" or "delivery" only when the user asks you to ' "choose an area for landing a drone or placing a delivery. Landing or delivery intent takes " "priority even if the request also says show, mark, point, or pick. " '(4) Use type "pointing" only when the user asks for one concrete image point. The point may ' "identify one visible object or instance, or it may be an explicitly requested representative " "point such as the center of one visible group. Set action to " '"locate" for find/where/show-location requests, "point" for ' 'point/mark/highlight/select/pin requests, or "track" for track/follow/watch/keep-on/lock-on ' "requests. Look-at and focus-on requests for one target use locate. The " "model supplies an initial point for track requests; a downstream application performs temporal " "tracking. Requests about where several objects are concentrated or how two things are positioned " "use type answer unless the user explicitly asks for one representative center point. If a point, " "locate, or track request names multiple or ambiguous instances without identifying one instance " "or one representative point, preserve the requested pointing action but return status " "ambiguous_target with null " "coordinates rather than choosing one arbitrarily. " "If one request mixes output types, use this precedence: choose location for landing or delivery " "selection; otherwise choose pointing when one concrete point is explicitly requested; otherwise " "choose answer for a factual or verbal result; otherwise choose caption for a whole-scene read. " "If the user genuinely asks for both a landing selection and a delivery selection, use the " "selection requested first. If one target is assigned more than one pointing action, use track " "before point and point before locate. If pointing actions name different targets, use that same " "action precedence but return ambiguous_target with null coordinates rather than choosing one " "target arbitrarily. " "Do not invent extra keys to combine unrelated tasks. " "Every location or pointing object must contain exactly type, its intent or action, caption, " "coordinate_system, point_2d, point_semantics, pointing_mode, status, x, and y. Coordinates use " '"gemma_relative_0_1000_yx": every non-null coordinate is a number from 0 through 1000, ' "point_2d is [y,x], and x and y repeat those values. Use " 'pointing_mode "precise_point" only for an image-grounded point. Use point_semantics ' '"specific_point" for a concrete target and "representative_point" only when the point represents ' "a broader visible target. A location response alone may use " '"coarse_grid_direction" with status "coarse_direction" for a broad directional cue; x and y ' "must each be exactly 250, 500, or 750. A coarse cue is never a landing point, delivery point, " "object point, or tracking target. When no reliable target or " 'direction applies, use pointing_mode "none", point_semantics "not_applicable", and null ' "coordinate_system, point_2d, x, and y. For location, use target_found for a selected precise " "point, no_safe_area when no usable area exists, coarse_direction only for the broad grid cue " "described above, or unknown when the image is insufficient. For pointing, use target_found " "for one grounded target, no_target when the requested object is absent, no_matching_target when " "a requested relation has no match, ambiguous_target when one target cannot be selected, or " "unknown when the image is insufficient. Do not omit required keys or add extra keys." ) COORDINATE_SYSTEM = "gemma_relative_0_1000_yx" GRID_VALUES = {250.0, 500.0, 750.0} LOCATION_INTENTS = {"landing", "delivery"} POINTING_ACTIONS = {"locate", "point", "track"} LOCATION_STATUSES = {"target_found", "coarse_direction", "no_safe_area", "unknown"} POINTING_STATUSES = { "target_found", "no_target", "no_matching_target", "ambiguous_target", "unknown", } def typed_response_errors(payload: Any) -> list[str]: """Return contract violations; an empty list means the payload is dispatch-safe.""" if not isinstance(payload, dict): return ["response must be a JSON object"] response_type = payload.get("type") if response_type == "caption": return _text_response_errors(payload, text_key="caption") if response_type == "answer": return _text_response_errors(payload, text_key="answer") if response_type == "location": return _spatial_response_errors( payload, discriminator_key="intent", discriminator_values=LOCATION_INTENTS, statuses=LOCATION_STATUSES, allow_coarse=True, ) if response_type == "pointing": return _spatial_response_errors( payload, discriminator_key="action", discriminator_values=POINTING_ACTIONS, statuses=POINTING_STATUSES, allow_coarse=False, ) return ["type must be caption, answer, location, or pointing"] def _text_response_errors(payload: dict[str, Any], *, text_key: str) -> list[str]: required = {"type", text_key} errors: list[str] = [] if set(payload) != required: errors.append(f"expected exactly {sorted(required)}, got {sorted(payload)}") value = payload.get(text_key) if not isinstance(value, str) or not value.strip(): errors.append(f"{text_key} must be a non-empty string") return errors def _spatial_response_errors( payload: dict[str, Any], *, discriminator_key: str, discriminator_values: set[str], statuses: set[str], allow_coarse: bool, ) -> list[str]: required = { "type", discriminator_key, "caption", "coordinate_system", "point_2d", "point_semantics", "pointing_mode", "status", "x", "y", } errors: list[str] = [] if set(payload) != required: errors.append(f"expected exactly {sorted(required)}, got {sorted(payload)}") if payload.get(discriminator_key) not in discriminator_values: errors.append( f"{discriminator_key} must be one of {sorted(discriminator_values)}" ) if not isinstance(payload.get("caption"), str) or not payload["caption"].strip(): errors.append("caption must be a non-empty string") if payload.get("status") not in statuses: errors.append(f"status must be one of {sorted(statuses)}") mode = payload.get("pointing_mode") if mode == "precise_point": if payload.get("status") != "target_found": errors.append("precise_point requires status target_found") if payload.get("point_semantics") not in { "specific_point", "representative_point", }: errors.append( "precise_point requires specific_point or representative_point semantics" ) errors.extend(_coordinate_errors(payload)) elif mode == "coarse_grid_direction": if not allow_coarse: errors.append("pointing responses cannot use coarse_grid_direction") if payload.get("status") != "coarse_direction": errors.append("coarse_grid_direction requires status coarse_direction") if payload.get("point_semantics") != "representative_point": errors.append( "coarse_grid_direction requires representative_point semantics" ) errors.extend(_coordinate_errors(payload)) if _complete_coordinates(payload) and ( float(payload["x"]) not in GRID_VALUES or float(payload["y"]) not in GRID_VALUES ): errors.append("coarse coordinates must use the 250/500/750 grid") elif mode == "none": if payload.get("point_semantics") != "not_applicable": errors.append("none requires not_applicable semantics") if not _null_coordinates(payload): errors.append("none requires null coordinate_system, point_2d, x, and y") if payload.get("status") in {"target_found", "coarse_direction"}: errors.append("none cannot use a target-bearing status") else: allowed = ["precise_point", "none"] if allow_coarse: allowed.insert(1, "coarse_grid_direction") errors.append(f"pointing_mode must be one of {allowed}") return errors def _coordinate_errors(payload: dict[str, Any]) -> list[str]: errors: list[str] = [] if payload.get("coordinate_system") != COORDINATE_SYSTEM: errors.append(f"coordinate_system must be {COORDINATE_SYSTEM}") point = payload.get("point_2d") x = payload.get("x") y = payload.get("y") if not _coordinate(x) or not _coordinate(y): errors.append("x and y must be numbers from 0 through 1000") if ( not isinstance(point, list) or len(point) != 2 or not all(_coordinate(v) for v in point) ): errors.append("point_2d must be [y, x] with values from 0 through 1000") elif ( _coordinate(x) and _coordinate(y) and (float(point[0]) != float(y) or float(point[1]) != float(x)) ): errors.append("point_2d must exactly duplicate [y, x]") return errors def _coordinate(value: Any) -> bool: return ( isinstance(value, (int, float)) and not isinstance(value, bool) and 0 <= value <= 1000 ) def _complete_coordinates(payload: dict[str, Any]) -> bool: point = payload.get("point_2d") return ( payload.get("coordinate_system") == COORDINATE_SYSTEM and _coordinate(payload.get("x")) and _coordinate(payload.get("y")) and isinstance(point, list) and len(point) == 2 and all(_coordinate(value) for value in point) and float(point[0]) == float(payload["y"]) and float(point[1]) == float(payload["x"]) ) def _null_coordinates(payload: dict[str, Any]) -> bool: return all( payload.get(key) is None for key in ("coordinate_system", "point_2d", "x", "y") ) def drawable_point(payload: Any) -> tuple[float, float, str] | None: """Return normalized x/y plus visual mode only for a safe drawable response.""" if typed_response_errors(payload): return None if payload.get("type") not in {"location", "pointing"}: return None if payload["pointing_mode"] == "precise_point": return float(payload["x"]), float(payload["y"]), "precise_target" if payload["pointing_mode"] == "coarse_grid_direction": return float(payload["x"]), float(payload["y"]), "coarse_direction" return None