#!/usr/bin/env python3 # SPDX-License-Identifier: Apache-2.0 """Goldens of the METEOR port from the bundle's fp32 CPU reference (``tt_meteor.reference``), cross-checked against the research ONNX Runtime goldens of the same frames. Run in the research venv (torch, onnx; no device):: cd bundles/meteor-p150 && OMP_NUM_THREADS=4 PYTHONPATH=$PWD/code \\ /home/ubuntu/experiments/tt-models/tools/research-venv/bin/python code/scripts/make_goldens.py [--frames ...] Writes (large files never go into the bundle): - ``research/meteor/goldens//taps.npz``: the reference's taps (``ttaw.golden`` format, JSON ``__meta__``): every tap for the shipped sample, the gate taps for the other frames; float tensors above 1 M elements as float16, the rest float32, the 19 outputs in their ONNX dtypes, the feed (``input.*``) and the lift geometry tables. - ``research/meteor/goldens/report.json``: the cross-check of every frame against its ORT golden (PCC, agreement). - ``tests/goldens/pandaset_019_f40_reference.json``: the decoded reference result of the PandaSet sample (small; the host / e2e tests and the card compare against it); - ``samples/pandaset_019_f40.reference.json``: the ``/predict`` body of the reference on the PandaSet sample request (``tests/test_e2e_device.py`` compares the served body with it). Both under ``staging_samples_pandaset/`` (the PandaSet sample does not ship until the user approves it; once it is swapped into ``code/tt_meteor/`` they are written there). The shipped synthetic sample's reference is written by ``code/scripts/make_synthetic_sample.py``. - ``--variants``: ``research/meteor/goldens//taps.npz`` (gate taps) of the LOAD-knob variants of the graph (``VARIANTS``: the D12 ``input_norm=imagenet`` option with ``depth_mean_bins=linear``, on the shipped sample with CAM_FRONT_NARROW absent: zero image + the donor's calibration, so the ``present`` mask matters; plus its ``_unzeroed`` counter-example, the mask not applied) for ``tests/test_variants_device.py``. The imagenet reference is itself checked against ONNX Runtime (``test_reference_cpu.py::test_imagenet_variant``). """ from __future__ import annotations import argparse import hashlib import json import sys import time from pathlib import Path import numpy as np BUNDLE = Path(__file__).resolve().parents[2] sys.path.insert(0, str(BUNDLE / "code")) from tt_meteor import __version__ # noqa: E402 from tt_meteor.api import load_sample # noqa: E402 from tt_meteor.host.calib import lift_geometry # noqa: E402 from tt_meteor.host.postprocess import PostConfig # noqa: E402 from tt_meteor.host.preprocess import MeteorFrame, load_scene_frame, scene_dir_of # noqa: E402 from tt_meteor.host.result import build_output # noqa: E402 from tt_meteor.reference import config as C # noqa: E402 from tt_meteor.reference.model import OUTPUT_NAMES # noqa: E402 from tt_meteor.reference.pipeline import MeteorReference # noqa: E402 from tt_meteor.ttaw.golden import TapRegistry, save_goldens # noqa: E402 from tt_meteor.ttaw.metrics import pcc # noqa: E402 ROOT = Path("/home/ubuntu/experiments/tt-models") R = ROOT / "research" / "meteor" OUT = R / "goldens" _PKG = BUNDLE / "code" / "tt_meteor" _ROOT = _PKG if (_PKG / "samples" / "pandaset_019_f40.json").is_file() else BUNDLE / "staging_samples_pandaset" SHIP = _ROOT / "samples" / "pandaset_019_f40" GATE_TAPS = ["img.fpn", "seg2d.logits", "depth.logits", "ctx.painted", "lift.weight", "lift.bev", "bev.raw", "bev.fused", "lane.pre", "det.feat", "det.hm_pre", "det.reg_pre", "traj.agent_delta", "ego.*", "bev.fused_mean", "refiner.e2e_ln", "refiner.e2e_res", "lift.valid", "lift.grid"] + list(OUTPUT_NAMES) EXACT_TAPS = {"lift.grid"} # kept float32 whatever their size (bit-exact geometry) # frame id -> (scene dir or None, frame index, ORT golden, all taps?, licence note) FRAMES = { "pandaset_019_f40": (SHIP, 0, R / "public_data/golden/ps019/full_f0020.npz", True, "PandaSet CC BY 4.0 + Terms (the shipped sample)"), "pandaset_090_f40": (R / "public_data/inputs/ps090", 20, R / "public_data/golden/ps090/full_f0020.npz", False, "PandaSet CC BY 4.0 + Terms"), "nuscenes_0103_kf09": (R / "public_data/inputs/ns0103", 9, R / "public_data/golden/ns0103/full_f0009.npz", False, "nuScenes CC BY-NC-SA 4.0: internal validation only, never shipped"), "meteor_valday_f040": (ROOT / "assets/meteor/hf_meteor-demo-scenes/valday", 40, R / "golden/valday_f040_ort.npz", False, "METEOR demo scenes: research / demonstration use only, never shipped"), } # variant id -> (base frame, MeteorConfig overrides, absent cameras, taps, present flags of the absent cameras) # ``_unzeroed``: the same zero images but every camera flagged present, i.e. what the graph computes when the imagenet # present mask is NOT applied (absent camera = the normalised zero image -mean / std): the counter-example that shows # the device's absent-camera outputs come from the zeroed input (``tests/test_variants_device.py``). _IMAGENET_LINEAR = {"input_norm": "imagenet", "depth_mean_bins": "linear"} VARIANTS = { "pandaset_019_f40_imagenet_linear": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",), GATE_TAPS, False), "pandaset_019_f40_imagenet_linear_unzeroed": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",), list(OUTPUT_NAMES), True), } def sha(a) -> str: return hashlib.sha256(np.ascontiguousarray(a).tobytes()).hexdigest() def store(taps: dict) -> dict: out = {} for k, v in taps.items(): v = np.asarray(v) if k in OUTPUT_NAMES or v.dtype in (np.bool_, np.uint8, np.float16) or v.dtype.kind in "iu": out[k] = v elif k in EXACT_TAPS: out[k] = v.astype(np.float32) elif v.size > (1 << 20): out[k] = v.astype(np.float16) else: out[k] = v.astype(np.float32) return out def compare(out: dict, gold) -> dict: rows = {} for k in OUTPUT_NAMES: a, b = np.asarray(out[k]), np.asarray(gold[k]) if b.dtype == np.uint8: rows[k] = {"agreement": float((a == b).mean())} else: rows[k] = {"pcc": pcc(a.astype(np.float64), b.astype(np.float64)), "max_abs": float(np.abs(a.astype(np.float64) - b.astype(np.float64)).max())} return rows def summary(res, frame: MeteorFrame, out: dict) -> dict: return {"ego": np.asarray(out["ego"], np.float64).reshape(-1).tolist(), "tl": np.asarray(out["tl"], np.float64).reshape(-1).tolist(), "boxes3d": [b.to_dict() for b in sorted(res.boxes3d, key=lambda z: -z.score)], "boxes2d": {cam: [b.to_dict() for b in bs] for cam, bs in zip(C.CAMERAS, res.boxes2d)}, "unknown": res.unknown, "plan_mode": int(res.plan["mode"]), "stationary_healthy": res.stationary_healthy, "argmax_sha256": {k: sha(out[k]) for k in ("lane", "seg2d", "depth")}, "input_sha256": frame.sha256()} def make_variant(vid: str, threads: int) -> None: """Goldens of one LOAD-knob variant (``VARIANTS``): the base frame with the absent cameras zeroed and given their donor's calibration (``host.preprocess.assemble_frame``'s rule), run by the reference with the variant config.""" base, overrides, absent, tap_names, flag_present = VARIANTS[vid] scene, idx = FRAMES[base][:2] frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx) imgs, K, T, present = frame.imgs.copy(), frame.K.copy(), frame.T_cam_ego.copy(), frame.present.copy() for name in absent: i, d = C.CAMERAS.index(name), C.CAMERAS.index(C.DONOR[name]) imgs[0, i] = 0 K[0, i], T[0, i] = K[0, d], T[0, d] present[i] = False if flag_present: present[:] = True frame = MeteorFrame(imgs, K, T, frame.v0, present, frame.pose, dict(frame.meta)) ref = MeteorReference(threads=threads, cfg=C.default_config(**overrides)) taps = TapRegistry(include=list(tap_names)) t1 = time.time() ref.run_frame(frame, taps) dt = time.time() - t1 geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0]) arrays = store(taps.to_dict()) arrays.update({f"input.{k}": v for k, v in frame.feed().items()}) arrays["input.present"] = frame.present meta = {"generator": "code/scripts/make_goldens.py --variants", "bundle_version": __version__, "frame": vid, "base_frame": base, "absent": list(absent), "scene": str(scene), "index": idx, "onnx_sha256": ref.weights.sha256, "licence": FRAMES[base][4], "input_norm": ref.cfg.input_norm, "depth_mean_bins": ref.cfg.depth_mean_bins, "flag_present": bool(flag_present), "taps": "gate" if tap_names is GATE_TAPS else "outputs", "seconds": round(dt, 1), "lift": geom.stats()} path = save_goldens(OUT / vid / "taps.npz", arrays, meta) print(f"{vid}: {dt:.1f} s forward, {path.stat().st_size / 2 ** 20:.1f} MB", flush=True) def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--frames", nargs="*", default=list(FRAMES)) ap.add_argument("--variants", nargs="*", default=None, help="LOAD-knob variants (VARIANTS) instead of frames") ap.add_argument("--threads", type=int, default=4) a = ap.parse_args() t0 = time.time() if a.variants is not None: for vid in a.variants or list(VARIANTS): make_variant(vid, a.threads) print(f"done in {time.time() - t0:.0f} s") return ref = MeteorReference(threads=a.threads) report = json.loads((OUT / "report.json").read_text()) if (OUT / "report.json").is_file() else {} for fid in a.frames: scene, idx, golden, full, note = FRAMES[fid] if not Path(scene).is_dir(): print(f"skip {fid}: {scene} missing") continue frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx) taps = TapRegistry(include=["*"] if full else GATE_TAPS) t1 = time.time() out = ref.run_frame(frame, taps) dt = time.time() - t1 geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0]) arrays = store(taps.to_dict()) arrays.update({f"input.{k}": v for k, v in frame.feed().items()}) arrays["input.present"] = frame.present arrays.update({"geom.b0": geom.b0.astype(np.uint8), "geom.fr": geom.fr}) rows = {} if golden.is_file(): g = np.load(golden) same_feed = all(np.array_equal(g[f"in_{k}"], v) for k, v in frame.feed().items()) rows = compare(out, g) rows["_same_feed_as_ort_golden"] = bool(same_feed) meta = {"generator": "code/scripts/make_goldens.py", "bundle_version": __version__, "frame": fid, "scene": str(scene), "index": idx, "onnx_sha256": ref.weights.sha256, "licence": note, "input_norm": ref.cfg.input_norm, "depth_mean_bins": ref.cfg.depth_mean_bins, "taps": "all" if full else "gate", "ort_golden": str(golden), "seconds": round(dt, 1), "lift": geom.stats()} path = save_goldens(OUT / fid / "taps.npz", arrays, meta) report[fid] = {"meta": meta, "vs_ort": rows, "size_mb": round(path.stat().st_size / 2 ** 20, 1)} res = build_output(out, frame, PostConfig(), None) report[fid]["decode"] = {"boxes3d": len(res.boxes3d), "boxes2d": [len(b) for b in res.boxes2d], "unknown": len(res.unknown), "mode": int(res.plan["mode"]), "tl": res.traffic_light["state"]} worst = min((r.get("pcc", r.get("agreement", 1.0)) for k, r in rows.items() if isinstance(r, dict)), default=float("nan")) print(f"{fid}: {dt:.1f} s forward, {report[fid]['size_mb']} MB, worst vs ORT {worst:.10f}, " f"decode {report[fid]['decode']}", flush=True) if fid == "pandaset_019_f40": small = summary(res, frame, out) small["_doc"] = ("Decoded fp32 CPU reference (tt_meteor.reference, PCC 1.0 vs ONNX Runtime) on the shipped " "sample PandaSet 019 frame 40, stateless host decode (PostConfig defaults); " "code/scripts/make_goldens.py.") (_ROOT / "tests/goldens/pandaset_019_f40_reference.json").write_text( json.dumps(small, indent=1, default=float) + "\n") (OUT / "report.json").write_text(json.dumps(report, indent=1, default=float) + "\n") if "pandaset_019_f40" in a.frames: # the /predict body of the reference on the shipped sample request (fresh stream, as the smoke test sends it) request = load_sample(SHIP.parent / "pandaset_019_f40.json") body = MeteorReference(weights=ref.weights, threads=a.threads)(**request) d = body.to_dict() d["timing_ms"] = {} (SHIP.parent / "pandaset_019_f40.reference.json").write_text(json.dumps(d) + "\n") print("wrote samples/pandaset_019_f40.reference.json:", d["num_detections"], "detections, mode", d["plan"]["mode"]) print(f"done in {time.time() - t0:.0f} s") if __name__ == "__main__": main()