#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Goldens of the METEOR port from the bundle's fp32 CPU reference (``tt_meteor.reference``), cross-checked against
the research ONNX Runtime goldens of the same frames.
Run in the research venv (torch, onnx; no device)::
cd bundles/meteor-p150 && OMP_NUM_THREADS=4 PYTHONPATH=$PWD/code \\
/home/ubuntu/experiments/tt-models/tools/research-venv/bin/python code/scripts/make_goldens.py [--frames ...]
Writes (large files never go into the bundle):
- ``research/meteor/goldens//taps.npz``: the reference's taps (``ttaw.golden`` format, JSON ``__meta__``): every
tap for the shipped sample, the gate taps for the other frames; float tensors above 1 M elements as float16, the
rest float32, the 19 outputs in their ONNX dtypes, the feed (``input.*``) and the lift geometry tables.
- ``research/meteor/goldens/report.json``: the cross-check of every frame against its ORT golden (PCC, agreement).
- ``tests/goldens/pandaset_019_f40_reference.json``: the decoded reference result of the PandaSet sample (small; the
host / e2e tests and the card compare against it);
- ``samples/pandaset_019_f40.reference.json``: the ``/predict`` body of the reference on the PandaSet sample request
(``tests/test_e2e_device.py`` compares the served body with it).
Both under ``staging_samples_pandaset/`` (the PandaSet sample does not ship until the user approves it; once it is
swapped into ``code/tt_meteor/`` they are written there). The shipped synthetic sample's reference is written by
``code/scripts/make_synthetic_sample.py``.
- ``--variants``: ``research/meteor/goldens//taps.npz`` (gate taps) of the LOAD-knob variants of the graph
(``VARIANTS``: the D12 ``input_norm=imagenet`` option with ``depth_mean_bins=linear``, on the shipped sample with
CAM_FRONT_NARROW absent: zero image + the donor's calibration, so the ``present`` mask matters; plus its
``_unzeroed`` counter-example, the mask not applied) for ``tests/test_variants_device.py``. The imagenet reference is itself checked against ONNX Runtime
(``test_reference_cpu.py::test_imagenet_variant``).
"""
from __future__ import annotations
import argparse
import hashlib
import json
import sys
import time
from pathlib import Path
import numpy as np
BUNDLE = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(BUNDLE / "code"))
from tt_meteor import __version__ # noqa: E402
from tt_meteor.api import load_sample # noqa: E402
from tt_meteor.host.calib import lift_geometry # noqa: E402
from tt_meteor.host.postprocess import PostConfig # noqa: E402
from tt_meteor.host.preprocess import MeteorFrame, load_scene_frame, scene_dir_of # noqa: E402
from tt_meteor.host.result import build_output # noqa: E402
from tt_meteor.reference import config as C # noqa: E402
from tt_meteor.reference.model import OUTPUT_NAMES # noqa: E402
from tt_meteor.reference.pipeline import MeteorReference # noqa: E402
from tt_meteor.ttaw.golden import TapRegistry, save_goldens # noqa: E402
from tt_meteor.ttaw.metrics import pcc # noqa: E402
ROOT = Path("/home/ubuntu/experiments/tt-models")
R = ROOT / "research" / "meteor"
OUT = R / "goldens"
_PKG = BUNDLE / "code" / "tt_meteor"
_ROOT = _PKG if (_PKG / "samples" / "pandaset_019_f40.json").is_file() else BUNDLE / "staging_samples_pandaset"
SHIP = _ROOT / "samples" / "pandaset_019_f40"
GATE_TAPS = ["img.fpn", "seg2d.logits", "depth.logits", "ctx.painted", "lift.weight", "lift.bev", "bev.raw",
"bev.fused", "lane.pre", "det.feat", "det.hm_pre", "det.reg_pre", "traj.agent_delta", "ego.*",
"bev.fused_mean", "refiner.e2e_ln", "refiner.e2e_res", "lift.valid", "lift.grid"] + list(OUTPUT_NAMES)
EXACT_TAPS = {"lift.grid"} # kept float32 whatever their size (bit-exact geometry)
# frame id -> (scene dir or None, frame index, ORT golden, all taps?, licence note)
FRAMES = {
"pandaset_019_f40": (SHIP, 0, R / "public_data/golden/ps019/full_f0020.npz", True,
"PandaSet CC BY 4.0 + Terms (the shipped sample)"),
"pandaset_090_f40": (R / "public_data/inputs/ps090", 20, R / "public_data/golden/ps090/full_f0020.npz", False,
"PandaSet CC BY 4.0 + Terms"),
"nuscenes_0103_kf09": (R / "public_data/inputs/ns0103", 9, R / "public_data/golden/ns0103/full_f0009.npz", False,
"nuScenes CC BY-NC-SA 4.0: internal validation only, never shipped"),
"meteor_valday_f040": (ROOT / "assets/meteor/hf_meteor-demo-scenes/valday", 40, R / "golden/valday_f040_ort.npz",
False, "METEOR demo scenes: research / demonstration use only, never shipped"),
}
# variant id -> (base frame, MeteorConfig overrides, absent cameras, taps, present flags of the absent cameras)
# ``_unzeroed``: the same zero images but every camera flagged present, i.e. what the graph computes when the imagenet
# present mask is NOT applied (absent camera = the normalised zero image -mean / std): the counter-example that shows
# the device's absent-camera outputs come from the zeroed input (``tests/test_variants_device.py``).
_IMAGENET_LINEAR = {"input_norm": "imagenet", "depth_mean_bins": "linear"}
VARIANTS = {
"pandaset_019_f40_imagenet_linear": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",), GATE_TAPS,
False),
"pandaset_019_f40_imagenet_linear_unzeroed": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",),
list(OUTPUT_NAMES), True),
}
def sha(a) -> str:
return hashlib.sha256(np.ascontiguousarray(a).tobytes()).hexdigest()
def store(taps: dict) -> dict:
out = {}
for k, v in taps.items():
v = np.asarray(v)
if k in OUTPUT_NAMES or v.dtype in (np.bool_, np.uint8, np.float16) or v.dtype.kind in "iu":
out[k] = v
elif k in EXACT_TAPS:
out[k] = v.astype(np.float32)
elif v.size > (1 << 20):
out[k] = v.astype(np.float16)
else:
out[k] = v.astype(np.float32)
return out
def compare(out: dict, gold) -> dict:
rows = {}
for k in OUTPUT_NAMES:
a, b = np.asarray(out[k]), np.asarray(gold[k])
if b.dtype == np.uint8:
rows[k] = {"agreement": float((a == b).mean())}
else:
rows[k] = {"pcc": pcc(a.astype(np.float64), b.astype(np.float64)),
"max_abs": float(np.abs(a.astype(np.float64) - b.astype(np.float64)).max())}
return rows
def summary(res, frame: MeteorFrame, out: dict) -> dict:
return {"ego": np.asarray(out["ego"], np.float64).reshape(-1).tolist(),
"tl": np.asarray(out["tl"], np.float64).reshape(-1).tolist(),
"boxes3d": [b.to_dict() for b in sorted(res.boxes3d, key=lambda z: -z.score)],
"boxes2d": {cam: [b.to_dict() for b in bs] for cam, bs in zip(C.CAMERAS, res.boxes2d)},
"unknown": res.unknown, "plan_mode": int(res.plan["mode"]), "stationary_healthy": res.stationary_healthy,
"argmax_sha256": {k: sha(out[k]) for k in ("lane", "seg2d", "depth")},
"input_sha256": frame.sha256()}
def make_variant(vid: str, threads: int) -> None:
"""Goldens of one LOAD-knob variant (``VARIANTS``): the base frame with the absent cameras zeroed and given their
donor's calibration (``host.preprocess.assemble_frame``'s rule), run by the reference with the variant config."""
base, overrides, absent, tap_names, flag_present = VARIANTS[vid]
scene, idx = FRAMES[base][:2]
frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx)
imgs, K, T, present = frame.imgs.copy(), frame.K.copy(), frame.T_cam_ego.copy(), frame.present.copy()
for name in absent:
i, d = C.CAMERAS.index(name), C.CAMERAS.index(C.DONOR[name])
imgs[0, i] = 0
K[0, i], T[0, i] = K[0, d], T[0, d]
present[i] = False
if flag_present:
present[:] = True
frame = MeteorFrame(imgs, K, T, frame.v0, present, frame.pose, dict(frame.meta))
ref = MeteorReference(threads=threads, cfg=C.default_config(**overrides))
taps = TapRegistry(include=list(tap_names))
t1 = time.time()
ref.run_frame(frame, taps)
dt = time.time() - t1
geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0])
arrays = store(taps.to_dict())
arrays.update({f"input.{k}": v for k, v in frame.feed().items()})
arrays["input.present"] = frame.present
meta = {"generator": "code/scripts/make_goldens.py --variants", "bundle_version": __version__, "frame": vid,
"base_frame": base, "absent": list(absent), "scene": str(scene), "index": idx,
"onnx_sha256": ref.weights.sha256, "licence": FRAMES[base][4], "input_norm": ref.cfg.input_norm,
"depth_mean_bins": ref.cfg.depth_mean_bins, "flag_present": bool(flag_present),
"taps": "gate" if tap_names is GATE_TAPS else "outputs", "seconds": round(dt, 1),
"lift": geom.stats()}
path = save_goldens(OUT / vid / "taps.npz", arrays, meta)
print(f"{vid}: {dt:.1f} s forward, {path.stat().st_size / 2 ** 20:.1f} MB", flush=True)
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--frames", nargs="*", default=list(FRAMES))
ap.add_argument("--variants", nargs="*", default=None, help="LOAD-knob variants (VARIANTS) instead of frames")
ap.add_argument("--threads", type=int, default=4)
a = ap.parse_args()
t0 = time.time()
if a.variants is not None:
for vid in a.variants or list(VARIANTS):
make_variant(vid, a.threads)
print(f"done in {time.time() - t0:.0f} s")
return
ref = MeteorReference(threads=a.threads)
report = json.loads((OUT / "report.json").read_text()) if (OUT / "report.json").is_file() else {}
for fid in a.frames:
scene, idx, golden, full, note = FRAMES[fid]
if not Path(scene).is_dir():
print(f"skip {fid}: {scene} missing")
continue
frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx)
taps = TapRegistry(include=["*"] if full else GATE_TAPS)
t1 = time.time()
out = ref.run_frame(frame, taps)
dt = time.time() - t1
geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0])
arrays = store(taps.to_dict())
arrays.update({f"input.{k}": v for k, v in frame.feed().items()})
arrays["input.present"] = frame.present
arrays.update({"geom.b0": geom.b0.astype(np.uint8), "geom.fr": geom.fr})
rows = {}
if golden.is_file():
g = np.load(golden)
same_feed = all(np.array_equal(g[f"in_{k}"], v) for k, v in frame.feed().items())
rows = compare(out, g)
rows["_same_feed_as_ort_golden"] = bool(same_feed)
meta = {"generator": "code/scripts/make_goldens.py", "bundle_version": __version__, "frame": fid,
"scene": str(scene), "index": idx, "onnx_sha256": ref.weights.sha256, "licence": note,
"input_norm": ref.cfg.input_norm, "depth_mean_bins": ref.cfg.depth_mean_bins,
"taps": "all" if full else "gate", "ort_golden": str(golden), "seconds": round(dt, 1),
"lift": geom.stats()}
path = save_goldens(OUT / fid / "taps.npz", arrays, meta)
report[fid] = {"meta": meta, "vs_ort": rows, "size_mb": round(path.stat().st_size / 2 ** 20, 1)}
res = build_output(out, frame, PostConfig(), None)
report[fid]["decode"] = {"boxes3d": len(res.boxes3d), "boxes2d": [len(b) for b in res.boxes2d],
"unknown": len(res.unknown), "mode": int(res.plan["mode"]),
"tl": res.traffic_light["state"]}
worst = min((r.get("pcc", r.get("agreement", 1.0)) for k, r in rows.items() if isinstance(r, dict)),
default=float("nan"))
print(f"{fid}: {dt:.1f} s forward, {report[fid]['size_mb']} MB, worst vs ORT {worst:.10f}, "
f"decode {report[fid]['decode']}", flush=True)
if fid == "pandaset_019_f40":
small = summary(res, frame, out)
small["_doc"] = ("Decoded fp32 CPU reference (tt_meteor.reference, PCC 1.0 vs ONNX Runtime) on the shipped "
"sample PandaSet 019 frame 40, stateless host decode (PostConfig defaults); "
"code/scripts/make_goldens.py.")
(_ROOT / "tests/goldens/pandaset_019_f40_reference.json").write_text(
json.dumps(small, indent=1, default=float) + "\n")
(OUT / "report.json").write_text(json.dumps(report, indent=1, default=float) + "\n")
if "pandaset_019_f40" in a.frames:
# the /predict body of the reference on the shipped sample request (fresh stream, as the smoke test sends it)
request = load_sample(SHIP.parent / "pandaset_019_f40.json")
body = MeteorReference(weights=ref.weights, threads=a.threads)(**request)
d = body.to_dict()
d["timing_ms"] = {}
(SHIP.parent / "pandaset_019_f40.reference.json").write_text(json.dumps(d) + "\n")
print("wrote samples/pandaset_019_f40.reference.json:", d["num_detections"], "detections, mode",
d["plan"]["mode"])
print(f"done in {time.time() - t0:.0f} s")
if __name__ == "__main__":
main()