#!/usr/bin/env python3 # SPDX-License-Identifier: Apache-2.0 """The card's demo media (``media/``) from outputs of this port on the p150. Development tool of the workspace (CPU only; research venv: numpy, OpenCV, PIL), run after the device jobs that produce the TT outputs: PY=/home/ubuntu/experiments/tt-models/tools/research-venv/bin/python $PY code/scripts/make_demo.py --tt-public logs/public_tt --tt-sample logs/meteor/shipped_sample_tt.json \\ [--compare logs/public_tt/compare] - ``--tt-public``: ``code/scripts/run_public_frames.py``'s output (TT outputs of the public frames in the CPU-golden layout). Renders, with the research renderer ``research/meteor/public_data/scripts/render_media.py`` (privacy blur, dataset attribution in every footer), the card and the camera-head images of PandaSet 019 frame 40, PandaSet 090 frame 40 and nuScenes scene-0103 key-frame 9 (non-commercial, labelled ``_NC``), the PandaSet 019 animation, and a bird's-eye TT-vs-CPU comparison of PandaSet 019 frame 40 (no camera pixels). - ``--tt-sample``: the TT ``/predict`` body of the shipped synthetic sample (``tests/test_e2e_device.py`` keeps it); renders the eight synthetic cameras with the TT 2D boxes and the TT and CPU-reference bird's-eye views side by side. - ``media/ATTRIBUTION.md``: every file, its source frames, licence and changes. """ from __future__ import annotations import argparse import base64 import io import json import math import os import sys from pathlib import Path import numpy as np BUNDLE = Path(__file__).resolve().parents[2] ROOT = Path(os.environ.get("TT_MODELS_ROOT", "/home/ubuntu/experiments/tt-models")) PUBLIC = ROOT / "research" / "meteor" / "public_data" sys.path.insert(0, str(PUBLIC / "scripts")) sys.path.insert(0, str(ROOT / "research" / "meteor" / "scripts")) MEDIA = BUNDLE / "media" PKG = BUNDLE / "code" / "tt_meteor" TT_LABEL = "Tenstorrent p150 output: meteor-p150 port (Blackhole, ETH dispatch 12x10, 1 CQ, one metal trace)" CPU_LABEL = "fp32 CPU reference (the same network, ONNX Runtime parity)" CARDS = [("ps019", 20), ("ps090", 20), ("ns0103", 9)] def _rm(): import render_media as rm # research renderer (privacy blur, footers) return rm def tag_of(gid: str, i: int, F) -> str: return f"meteor_{gid}_f{i:04d}" + ("_NC" if F.dataset == "nuScenes" else "") def save_jpeg(path: Path, rgb: np.ndarray, q: int = 88) -> None: from PIL import Image Image.fromarray(rgb).save(path, quality=q, optimize=True) def save_png(path: Path, rgb: np.ndarray) -> None: from PIL import Image Image.fromarray(rgb).save(path, optimize=True) def public_renders(tt_dir: Path, entries: list) -> None: rm = _rm() for gid, i in CARDS: F = rm.Frame(gid, i, str(tt_dir / gid)) tag = tag_of(gid, i, F) lic = "CC BY 4.0 + PandaSet Dataset Terms" if F.dataset == "PandaSet" else "**CC BY-NC-SA 4.0, non-commercial**" rgb, nb = rm.card(F, TT_LABEL) save_jpeg(MEDIA / f"{tag}_card_tt.jpg", rgb) entries.append((f"{tag}_card_tt.jpg", F.dataset, F.desc, lic, "TT outputs: 8-slot camera mosaic with 2D boxes; " "BEV lanes / risk / 3D boxes / futures / plan vs the logged path; front view with projected " "3D boxes and the planned path", rgb.shape)) hd = rm.heads(F, TT_LABEL) save_jpeg(MEDIA / f"{tag}_heads_tt.jpg", hd) entries.append((f"{tag}_heads_tt.jpg", F.dataset, F.desc, lic, "TT outputs: per-camera seg2d overlay and " "depth (linear bins), 8 slots", hd.shape)) print("rendered", tag, flush=True) frames = sorted(int(p.stem[1:]) for p in (tt_dir / "ps019").glob("f[0-9][0-9][0-9][0-9].json")) p = MEDIA / "meteor_ps019_seq_tt.gif" Fs, dur = rm.animation("ps019", frames, str(p), TT_LABEL, golden_dir=str(tt_dir / "ps019")) from PIL import Image with Image.open(p) as im: shape = (im.height, im.width, 3) entries.append((p.name, "PandaSet", f"{Fs[0].desc} ... {Fs[-1].desc} (every 4th frame, {len(frames)} frames, " f"{dur} ms each: real time)", "CC BY 4.0 + PandaSet Dataset Terms", "TT outputs: the card as an " "animation (128-colour GIF)", shape)) print("rendered", p.name, flush=True) def _text(draw, xy, s, size=12, fill=(235, 235, 235), bold=False, anchor="la"): _rm().text(draw, xy, s, size, fill, bold, anchor) def bev_compare(tt_dir: Path, compare: dict, entries: list) -> None: """PandaSet 019 frame 40: CPU BEV | TT BEV | where the two lane maps differ (no camera pixels).""" import cv2 from PIL import Image, ImageDraw rm = _rm() gid, i = "ps019", 20 Fc = rm.Frame(gid, i, None) Ft = rm.Frame(gid, i, str(tt_dir / gid)) w, h = 300, 450 a, b = rm.bev_panel(Fc, w, h), rm.bev_panel(Ft, w, h) view_f, view_r, yh = 50.0, 25.0, 25.0 r0, r1 = int((80 - view_f) / 0.2), int((80 + view_r) / 0.2) c0, c1 = int((50 - yh) / 0.2), int((50 + yh) / 0.2) lc, lt = Fc.out["lane"][0][r0:r1, c0:c1], Ft.out["lane"][0][r0:r1, c0:c1] diff = np.full(lc.shape + (3,), 30, np.uint8) diff[lc > 0] = (70, 70, 75) diff[lc != lt] = (255, 60, 60) diff = cv2.resize(diff, (w, h), interpolation=cv2.INTER_NEAREST) H = 44 + 16 + h + 92 canvas = np.full((H, 3 * w + 4 * 8, 3), rm.BG, np.uint8) for k, img in enumerate((a, b, diff)): rm.paste(canvas, img, 8 + k * (w + 8), 60) im = Image.fromarray(canvas) d = ImageDraw.Draw(im) _text(d, (8, 6), f"METEOR v1.0 on {Fc.desc}: Tenstorrent p150 vs the fp32 CPU reference (bird's-eye view)", 14, bold=True) _text(d, (8, 26), "Pixel-free BEV panels: lane classes, risk heat, 3D boxes, futures, the 3 ego paths (selected: green); " "white circles: logged path", 10, fill=rm.TEXT2) for k, lab in enumerate(("fp32 CPU reference", "Tenstorrent p150 (this port)", "lane cells that differ (red)")): _text(d, (8 + k * (w + 8), 44), lab, 12, bold=True) r = compare.get(f"{gid}/f{i:04d}", {}) lines = [] if r: lines.append(f"Agreement on this frame: lane {r['lane_agree']:.4f}, seg2d {r['seg2d_agree']:.4f}, depth " f"{r['depth_agree']:.4f} (argmax maps); hm / reg / stationary / risk PCC {r['hm_pcc']:.5f} / " f"{r['reg_pcc']:.5f} / {r['stationary_pcc']:.5f} / {r['risk_pcc']:.5f};") lines.append(f"3D boxes matched {r['det3d_matched']:.3f} (same class, <= 1 m); selected ego path mean deviation " f"{r['ego_path_dev_m']:.3f} m, same mode: {r['ego_mode_equal']}; traffic light equal: {r['tl_equal']}.") lines += ["Contains data from PandaSet (Scale AI and Hesai), https://pandaset.org, CC BY 4.0 + PandaSet Dataset Terms " "(outputs only; no camera pixels).", "Scale AI and Hesai do not endorse this work."] for k, line in enumerate(lines): _text(d, (8, 60 + h + 8 + 15 * k), line, 10, fill=rm.TEXT1 if k < 2 else rm.TEXT2) out = np.asarray(im) path = MEDIA / "meteor_ps019_f0020_bev_tt_vs_cpu.png" save_png(path, out) entries.append((path.name, "PandaSet", Fc.desc, "CC BY 4.0 + PandaSet Dataset Terms", "BEV of the fp32 CPU " "reference and of the TT output side by side, and the lane cells that differ (no camera pixels)", out.shape)) print("rendered", path.name, flush=True) # ------------------------------------------------------------------------------------------- shipped sample def _decode_lane(body: dict) -> np.ndarray: from PIL import Image lane = body["lane"] a = np.asarray(Image.open(io.BytesIO(base64.b64decode(lane["data"])))) return a.reshape(lane["shape"]) def body_bev(body: dict, w: int, h: int, view_f=50.0, view_r=25.0, yh=25.0) -> np.ndarray: import cv2 rm = _rm() lane = _decode_lane(body) r0, r1 = int((80 - view_f) / 0.2), int((80 + view_r) / 0.2) c0, c1 = int((50 - yh) / 0.2), int((50 + yh) / 0.2) img = cv2.resize(np.asarray(rm.LANE_RGB, np.uint8)[lane[r0:r1, c0:c1]], (w, h), interpolation=cv2.INTER_NEAREST) s = h / (view_f + view_r) def px(x, y): return (yh - y) * s, (view_f - x) * s for dd in range(-int(view_r // 10) * 10, int(view_f) + 1, 10): yy = int(round((view_f - dd) * s)) cv2.line(img, (0, yy), (w, yy), (55, 55, 60), 1) if dd: cv2.putText(img, f"{dd}m", (3, yy - 3), cv2.FONT_HERSHEY_SIMPLEX, 0.32, (150, 150, 150), 1, cv2.LINE_AA) for det in body["detections"]: (x, y), (L, W), yaw = det["center"], det["size"], det["yaw"] col = rm.STAT_RGB if det.get("stationary") else (rm.VEH_RGB if det["label_id"] == 0 else rm.VRU_RGB) cors = rm.box_corners(x, y, 0, L, W, 0, yaw)[:4] p = np.array([px(cx, cy) for cx, cy, _ in cors]).round().astype(np.int32) cv2.polylines(img, [p], True, col, 2, cv2.LINE_AA) cx, cy = px(x, y) fx, fy = px(x + 0.5 * L * math.cos(yaw), y + 0.5 * L * math.sin(yaw)) cv2.line(img, (int(cx), int(cy)), (int(fx), int(fy)), col, 2, cv2.LINE_AA) if det.get("future") and not det.get("stationary") and det["label_id"] == 0: q = np.array([[cx, cy]] + [px(a, b) for a, b in det["future"]]).round().astype(np.int32) cv2.polylines(img, [q], False, col, 1, cv2.LINE_AA) ex, ey = px(0, 0) sel = body["plan"]["mode"] for k, path in enumerate(body["plan"]["paths"]): q = np.array([[ex, ey]] + [px(a, b) for a, b in path]).round().astype(np.int32) cv2.polylines(img, [q], False, rm.PLAN_RGB if k == sel else (0, 150, 60), 2 if k == sel else 1, cv2.LINE_AA) if k == sel: for pt in q[1:]: cv2.circle(img, tuple(int(v) for v in pt), 3, rm.PLAN_RGB, -1, cv2.LINE_AA) tri = np.array([[ex, ey - 9], [ex - 6, ey + 7], [ex + 6, ey + 7]]).round().astype(np.int32) cv2.fillPoly(img, [tri], (255, 255, 255)) return img def sample_render(tt_body: dict, entries: list) -> None: import cv2 from PIL import Image, ImageDraw rm = _rm() ref = json.loads((PKG / "samples" / "synthetic_8cam.reference.json").read_text()) spec = json.loads((PKG / "samples" / "synthetic_8cam.json").read_text()) tw, th, gap, x0, top, lab = 233, 131, 6, 4, 44, 14 bw, bh = 300, 450 y_bot = top + 2 * (lab + th) + 12 H = y_bot + 18 + bh + 70 canvas = np.full((H, 960, 3), rm.BG, np.uint8) for k, cam in enumerate(rm.TILE_ORDER): img = np.asarray(Image.open(PKG / "samples" / spec["images"][cam]).convert("RGB")).copy() for det in tt_body["detections_2d"].get(cam, []): a, b, c, e = (int(round(v)) for v in det["box_xyxy"]) cv2.rectangle(img, (a, b), (c, e), rm.DET10_RGB[det["label_id"] % 10], 2) img = cv2.resize(img, (tw, th), interpolation=cv2.INTER_AREA) rm.paste(canvas, img, x0 + (k % 4) * (tw + gap), top + (k // 4) * (lab + th) + lab) rm.paste(canvas, body_bev(ref, bw, bh), x0, y_bot + 18) rm.paste(canvas, body_bev(tt_body, bw, bh), x0 + bw + 8, y_bot + 18) im = Image.fromarray(canvas) d = ImageDraw.Draw(im) _text(d, (8, 6), "METEOR v1.0 on the shipped synthetic sample (samples/synthetic_8cam): Tenstorrent p150 vs the fp32 CPU " "reference", 14, bold=True) _text(d, (8, 26), "Synthetic test frame generated by this repository (Apache-2.0): a ray-cast street; the object " "pixels were optimised against the CPU reference", 10, fill=rm.WARN) for k, cam in enumerate(rm.TILE_ORDER): n2d = len(tt_body["detections_2d"].get(cam, [])) _text(d, (x0 + (k % 4) * (tw + gap), top + (k // 4) * (lab + th) + 1), f"{cam[4:]} (TT 2D boxes: {n2d})", 10, fill=rm.TEXT2) _text(d, (x0, y_bot), "BEV: fp32 CPU reference (stored)", 12, bold=True) _text(d, (x0 + bw + 8, y_bot), "BEV: Tenstorrent p150", 12, bold=True) ix = x0 + 2 * (bw + 8) + 4 lines = [f"3D boxes: TT {tt_body['num_detections']}, reference {ref['num_detections']}", f"plan mode: TT {tt_body['plan']['mode']} (p={tt_body['plan']['mode_probs'][tt_body['plan']['mode']]:.2f}), " f"reference {ref['plan']['mode']}", "path end (m): TT ({:.2f}, {:.2f}), reference ({:.2f}, {:.2f})".format(*tt_body["trajectory"][-1], *ref["trajectory"][-1]), f"traffic light: TT {tt_body['traffic_light']['state']}, reference {ref['traffic_light']['state']}", f"2D boxes: TT {sum(len(v) for v in tt_body['detections_2d'].values())}, reference " f"{sum(len(v) for v in ref['detections_2d'].values())}"] lane_t, lane_r = _decode_lane(tt_body), _decode_lane(ref) lines.append(f"lane map agreement: {float((lane_t == lane_r).mean()):.4f}") by_label: dict = {} for det in tt_body["detections"]: by_label.setdefault(det["label"], []).append(round(det["score"], 2)) for k, v in sorted(by_label.items()): lines.append(f"TT {k} scores: " + ", ".join(f"{x:.2f}" for x in v)) for k, line in enumerate(lines): _text(d, (ix, y_bot + 22 + 17 * k), line, 11) ly = y_bot + 22 + 17 * len(lines) + 10 for name, colr in (("vehicle", rm.VEH_RGB), ("VRU", rm.VRU_RGB), ("stationary", rm.STAT_RGB), ("plan", rm.PLAN_RGB)): d.rectangle([ix, ly + 3, ix + 8, ly + 11], outline=colr, width=2) _text(d, (ix + 12, ly), name, 10, fill=rm.TEXT2) ly += 15 _text(d, (8, H - 18), "Data generated by this repository (code/scripts/make_synthetic_sample.py), Apache-2.0. " "No third-party pixels.", 10, fill=rm.TEXT2) out = np.asarray(im) path = MEDIA / "meteor_synthetic_8cam_tt_vs_cpu.png" save_png(path, out) entries.append((path.name, "synthetic (this repository)", "code/tt_meteor/samples/synthetic_8cam", "Apache-2.0", "the shipped sample's eight cameras with the TT 2D boxes; BEV of the stored CPU reference and of " "the TT output", out.shape)) print("rendered", path.name, flush=True) def write_attribution(entries: list) -> None: rm = _rm() C = rm.C lines = ["# media/: sources and licences", "", "Every model output drawn here was computed by **this port on one Tenstorrent Blackhole p150** (ETH " "dispatch, 12x10 grid, 1 command queue, the `frame` metal trace), unless a panel says \"CPU reference\" " "(the port's fp32 CPU reference of the same network, which matches ONNX Runtime on " "`meteor_v157c3Z.onnx`). Renderer: `code/scripts/make_demo.py` (camera mosaics, heads and animation through " "`research/meteor/public_data/scripts/render_media.py` of the development workspace).", "", "METEOR was trained on TIER IV recordings only (no paper, no public training split). PandaSet and nuScenes " "are out of its training domain (another camera rig, country and ISP), so the detections show the domain " "gap, not the port. No METEOR demo-scene frame appears in any image.", "", "| file | data | frames | content | licence | size (px) |", "|---|---|---|---|---|---|"] for f, ds, frames, lic, content, shape in entries: lines.append(f"| `{f}` | {ds} | {frames} | {content} | {lic} | {shape[1]}x{shape[0]} |") lines += ["", "Changes to the dataset images: resized to the 768x432 METEOR inputs (and a centre crop for the " "virtual FRONT_NARROW camera), downscaled for display (<= 960 px wide), heads of annotated pedestrians " "(<= 30 m) and plate areas of annotated vehicles (<= 25 m) blurred, model outputs drawn. The BACK_NARROW " "tile is black because that camera is absent (the model gets a zero image plus the BACK_WIDE pose, its " "trained 7-camera configuration).", "", "Attribution texts (also printed in each image footer):", "", "- PandaSet: " + C.PS_ATTRIBUTION, "- nuScenes (files marked `_NC`, **non-commercial**): " + C.NS_ATTRIBUTION, "- Synthetic sample: data generated by this repository (`code/scripts/make_synthetic_sample.py`), " "Apache-2.0.", "", "Rules (research/DATASETS.md section 2): the nuScenes renders are **non-commercial (CC BY-NC-SA 4.0, " "ShareAlike)** and keep that label next to the image wherever they are shown; the PandaSet renders are " "CC BY 4.0 + the PandaSet Dataset Terms (do not use them to identify people; no use of the licensors' " "names or logos beyond the attribution). No raw dataset file is in this directory."] (MEDIA / "ATTRIBUTION.md").write_text("\n".join(lines) + "\n") def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--tt-public", type=Path, required=True) ap.add_argument("--tt-sample", type=Path, help="TT /predict body of the shipped sample (omit: no sample render)") ap.add_argument("--compare", type=Path, help="compare_tt.py JSON files (.json) of the TT run vs the goldens") a = ap.parse_args() MEDIA.mkdir(exist_ok=True) compare = {} if a.compare: for p in sorted(a.compare.glob("*.json")): for r in json.loads(p.read_text()).get("rows", []): compare[f"{p.stem}/f{r['frame']:04d}"] = r entries: list = [] public_renders(a.tt_public, entries) bev_compare(a.tt_public, compare, entries) if a.tt_sample: sample_render(json.loads(a.tt_sample.read_text()), entries) write_attribution(entries) if __name__ == "__main__": main()