#!/usr/bin/env python3 # SPDX-License-Identifier: Apache-2.0 """Build the PandaSet sample of meteor-p150 from the research ship sample (PandaSet 019 frame 40, CC BY 4.0 + the PandaSet Dataset Terms; research/meteor/public_data/ship_sample/pandaset_019_f40, made by its make_ship_sample.py). It does NOT ship until the user approves PandaSet-derived samples (PLAN.md 6.3, research/DATASETS.md N5): by default (``--dest staging``) it is written to the git-ignored ``staging_samples_pandaset/`` (laid out like ``code/tt_meteor/``, where the tests find it, ``tests/paths.py``); ``--dest package`` writes it into ``code/tt_meteor/`` (the swap-in of ``staging_samples_pandaset/README.md``). The shipped sample is the synthetic ``samples/synthetic_8cam.json`` (``code/scripts/make_synthetic_sample.py``). Paths below are relative to the destination: - ``samples/pandaset_019_f40/``: the METEOR demo-scene layout (scenes.txt, manifest.json, the 8 camera JPEGs incl. the all-zero absent BACK_NARROW, ego_motion.npz, gt_boxes.json), LICENSE.txt, README.md, SHA256SUMS; - ``calib/pandaset_019.json``: the rig as a calibration preset (K at 768x432, T_ref_from_camera = camera optical frame -> ego; BACK_NARROW carries its donor's calibration, as the manifest does); - ``samples/pandaset_019_f40.json``: the request manifest (``model(**load_sample(path))``). Usage: python code/scripts/make_sample.py [--src ] [--dest staging|package] (numpy; no device) """ from __future__ import annotations import argparse import hashlib import json import shutil from pathlib import Path import numpy as np BUNDLE = Path(__file__).resolve().parents[2] PKG = BUNDLE / "code" / "tt_meteor" SRC = Path("/home/ubuntu/experiments/tt-models/research/meteor/public_data/ship_sample/pandaset_019_f40") CAMERAS = ("CAM_FRONT_WIDE", "CAM_FRONT_LEFT", "CAM_FRONT_RIGHT", "CAM_BACK_WIDE", "CAM_BACK_LEFT", "CAM_BACK_RIGHT", "CAM_FRONT_NARROW", "CAM_BACK_NARROW") SCENE = "pandaset_019_f40" def sha256(p: Path) -> str: return hashlib.sha256(p.read_bytes()).hexdigest() def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--src", type=Path, default=SRC) ap.add_argument("--dest", choices=("staging", "package"), default="staging") a = ap.parse_args() root = PKG if a.dest == "package" else BUNDLE / "staging_samples_pandaset" dst = root / "samples" / SCENE if dst.exists(): shutil.rmtree(dst) (dst / SCENE / "img").mkdir(parents=True) imgs = [f"{SCENE}/img/{p.name}" for p in sorted((a.src / SCENE / "img").iterdir())] for rel in ["scenes.txt", "LICENSE.txt", f"{SCENE}/manifest.json", f"{SCENE}/ego_motion.npz", f"{SCENE}/gt_boxes.json"] + imgs: shutil.copyfile(a.src / rel, dst / rel) m = json.loads((a.src / SCENE / "manifest.json").read_text()) ego = np.load(a.src / SCENE / "ego_motion.npz") readme = (a.src / "README.md").read_text() readme = readme.replace("## Run\n", "## Run (this bundle)\n\n```python\nfrom tt_meteor import METEOR, load_sample\n" "kwargs = load_sample('code/tt_meteor/samples/pandaset_019_f40.json')\n" "```\n\nThe CPU reference of the bundle (`tt_meteor.reference.pipeline.MeteorReference`) " "and the research tools read the scene layout directly:\n\n", 1) readme = readme.replace("## Expected outputs (`expected/`)", "## Expected outputs (research `expected/`, not " "copied here; the bundle's own reference output is " "`../pandaset_019_f40.reference.json`)", 1) (dst / "README.md").write_text(readme) files = sorted(p for p in dst.rglob("*") if p.is_file()) (dst / "SHA256SUMS").write_text("".join(f"{sha256(p)} {p.relative_to(dst).as_posix()}\n" for p in files)) cams = {} for c in CAMERAS: cc = m["cams"][c] cams[c] = {"intrinsics": cc["K"], "T_ref_from_camera": cc["T_ego_cam"], "image_size": [768, 432]} if not cc.get("present", True): cams[c]["absent"] = f"zero image + donor {cc.get('donor')} calibration (bevlane/dataset.py:131-145)" preset = {"frame_id": "base_link", "_source": f"research/meteor/public_data/ship_sample/{SCENE}/{SCENE}/manifest.json (PandaSet 019, " "CC BY 4.0 + PandaSet Dataset Terms; rig derivation: research/meteor/public_data/README.md)", "_ego_frame": m.get("ego_frame"), "cameras": cams} (root / "calib").mkdir(parents=True, exist_ok=True) (root / "calib" / "pandaset_019.json").write_text(json.dumps(preset, indent=1) + "\n") fr = m["frames"][0] req = {"images": {c: f"{SCENE}/{SCENE}/{fr['imgs'][c]}" for c in CAMERAS}, "calibration": {"preset": "pandaset_019"}, "ego_speed": float(ego["v0"][0]), "stream": {"id": "pandaset_019", "T_world_from_ego": dict(zip(("x", "y", "yaw"), (float(v) for v in ego["pose"][0])), z=0.0, roll=0.0, pitch=0.0)}, "_source": "PandaSet 019 frame 40 (Scale AI and Hesai, https://pandaset.org), CC BY 4.0 + PandaSet Dataset " f"Terms ({SCENE}/LICENSE.txt); METEOR 8-slot layout, BACK_NARROW absent (all-zero image)"} (root / "samples" / f"{SCENE}.json").write_text(json.dumps(req, indent=1) + "\n") print(f"wrote {dst} ({sum(p.stat().st_size for p in files) / 1e6:.2f} MB), calib/pandaset_019.json, " f"samples/{SCENE}.json") if __name__ == "__main__": main()