changh95 commited on
Commit
51defdc
·
verified ·
1 Parent(s): 2589074

tt-model push meteor-p150 (container)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +29 -0
  2. OPT_BASELINE.md +451 -0
  3. OPT_REPORT.md +113 -0
  4. PYTHON.md +165 -0
  5. README.md +199 -0
  6. SERVING.md +230 -0
  7. VERIFICATION_2026-10-10.md +182 -0
  8. build_info.json +48 -0
  9. code/PYTHON.md +165 -0
  10. code/conftest.py +59 -0
  11. code/models/common/lightweightmodule.py +12 -0
  12. code/scripts/README.md +18 -0
  13. code/scripts/bench.py +317 -0
  14. code/scripts/container_smoke.sh +137 -0
  15. code/scripts/device_stage_check.py +80 -0
  16. code/scripts/dump_device_outputs.py +44 -0
  17. code/scripts/fetch_samples.sh +25 -0
  18. code/scripts/make_demo.py +322 -0
  19. code/scripts/make_goldens.py +237 -0
  20. code/scripts/make_sample.py +97 -0
  21. code/scripts/make_synthetic_sample.py +508 -0
  22. code/scripts/precision_emulation.py +182 -0
  23. code/scripts/profile_ops.py +231 -0
  24. code/scripts/profile_summary.py +288 -0
  25. code/scripts/repro_grid_sample_eth.py +122 -0
  26. code/scripts/repro_resize2d_eth.py +223 -0
  27. code/scripts/run_public_frames.py +85 -0
  28. code/scripts/stress_frames.py +227 -0
  29. code/tt_meteor/__init__.py +40 -0
  30. code/tt_meteor/api.py +205 -0
  31. code/tt_meteor/calib/README.md +13 -0
  32. code/tt_meteor/calib/synthetic_8cam.json +464 -0
  33. code/tt_meteor/device.py +65 -0
  34. code/tt_meteor/host/__init__.py +16 -0
  35. code/tt_meteor/host/calib.py +199 -0
  36. code/tt_meteor/host/inputs.py +160 -0
  37. code/tt_meteor/host/outputs.py +111 -0
  38. code/tt_meteor/host/postprocess.py +462 -0
  39. code/tt_meteor/host/preprocess.py +189 -0
  40. code/tt_meteor/host/result.py +51 -0
  41. code/tt_meteor/host/temporal.py +180 -0
  42. code/tt_meteor/io.py +34 -0
  43. code/tt_meteor/reference/__init__.py +25 -0
  44. code/tt_meteor/reference/config.py +147 -0
  45. code/tt_meteor/reference/model.py +539 -0
  46. code/tt_meteor/reference/ort.py +139 -0
  47. code/tt_meteor/reference/pipeline.py +86 -0
  48. code/tt_meteor/reference/port_form.py +280 -0
  49. code/tt_meteor/reference/weights.py +236 -0
  50. code/tt_meteor/samples/README.md +31 -0
.gitattributes CHANGED
@@ -33,3 +33,32 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ code/tt_meteor/samples/synthetic_8cam/CAM_BACK_NARROW.png filter=lfs diff=lfs merge=lfs -text
37
+ code/tt_meteor/samples/synthetic_8cam/CAM_BACK_RIGHT.png filter=lfs diff=lfs merge=lfs -text
38
+ code/tt_meteor/samples/synthetic_8cam/CAM_BACK_WIDE.png filter=lfs diff=lfs merge=lfs -text
39
+ code/tt_meteor/samples/synthetic_8cam/CAM_FRONT_LEFT.png filter=lfs diff=lfs merge=lfs -text
40
+ code/tt_meteor/samples/synthetic_8cam/CAM_FRONT_NARROW.png filter=lfs diff=lfs merge=lfs -text
41
+ code/tt_meteor/samples/synthetic_8cam/CAM_FRONT_WIDE.png filter=lfs diff=lfs merge=lfs -text
42
+ image/blobs/sha256/067872b26b7173f0708cd110500b917d462d08711d6892501a9ea54a38b4c6f6 filter=lfs diff=lfs merge=lfs -text
43
+ image/blobs/sha256/17d0c1caf6bbb45eee78a04317ec8b3dcefb4c37ea065b82c29c20a183f8c765 filter=lfs diff=lfs merge=lfs -text
44
+ image/blobs/sha256/2e19a7ded24ad73423027b973b6c375f16dea19e2aa43cad23833870de4a0680 filter=lfs diff=lfs merge=lfs -text
45
+ image/blobs/sha256/32fa560530289e8b9adb68b3d37bd6fdc2c63479a79ce728a6779e63a861ebda filter=lfs diff=lfs merge=lfs -text
46
+ image/blobs/sha256/3570876206ec7924f1f40ec6d1a3d74c44e21f77220cee79851fd9ad9b5305a6 filter=lfs diff=lfs merge=lfs -text
47
+ image/blobs/sha256/385eb61defef064a82c5382744c1f00c925a71890ebc03dcbdb65ede3dfbd627 filter=lfs diff=lfs merge=lfs -text
48
+ image/blobs/sha256/3fe767711c183db63e0c1e745ecc00844fa349ac9e24f137b00342b90f7b2ac0 filter=lfs diff=lfs merge=lfs -text
49
+ image/blobs/sha256/5073f3c9466f53e104fc4b3166ce982d7e53f764697592f78c609016187947a4 filter=lfs diff=lfs merge=lfs -text
50
+ image/blobs/sha256/53d35e743cff2bc1f9e506e7ea215b8b1a0c33e7868f8381f1759747d01b33fa filter=lfs diff=lfs merge=lfs -text
51
+ image/blobs/sha256/725d484635e8379455b63418fdd70e920d1f74853e81240ce98bb67daee60dee filter=lfs diff=lfs merge=lfs -text
52
+ image/blobs/sha256/8f0daca2adb0ba572e616d0aec982fa0f8419d1ddcbac6e3695c42c168451800 filter=lfs diff=lfs merge=lfs -text
53
+ image/blobs/sha256/98c4455a98982b35380ec62f90d5cf0edff02eeb4b3bb18595d871fdeb0c0c67 filter=lfs diff=lfs merge=lfs -text
54
+ image/blobs/sha256/a225e43749bf82b0c24c6dc81f6cf207e35a01ed8445ec2ba5027d258919ea54 filter=lfs diff=lfs merge=lfs -text
55
+ image/blobs/sha256/b22f52e00ded7b565d8541479dbba1f62b001c32be7efbd3873da891633b36a1 filter=lfs diff=lfs merge=lfs -text
56
+ media/meteor_ns0103_f0009_NC_card_tt.jpg filter=lfs diff=lfs merge=lfs -text
57
+ media/meteor_ns0103_f0009_NC_heads_tt.jpg filter=lfs diff=lfs merge=lfs -text
58
+ media/meteor_ps019_f0020_bev_tt_vs_cpu.png filter=lfs diff=lfs merge=lfs -text
59
+ media/meteor_ps019_f0020_card_tt.jpg filter=lfs diff=lfs merge=lfs -text
60
+ media/meteor_ps019_f0020_heads_tt.jpg filter=lfs diff=lfs merge=lfs -text
61
+ media/meteor_ps019_seq_tt.gif filter=lfs diff=lfs merge=lfs -text
62
+ media/meteor_ps090_f0020_card_tt.jpg filter=lfs diff=lfs merge=lfs -text
63
+ media/meteor_ps090_f0020_heads_tt.jpg filter=lfs diff=lfs merge=lfs -text
64
+ media/meteor_synthetic_8cam_tt_vs_cpu.png filter=lfs diff=lfs merge=lfs -text
OPT_BASELINE.md ADDED
@@ -0,0 +1,451 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # meteor-p150 baseline on the p150 (before optimization)
2
+
3
+ Date 2026-10-10. Code is the baseline commit `e86c13a` (first correct port, no optimization; ETH default with the
4
+ hang fix of PORT_LOG.md section 12). The model code of `code/tt_meteor/` is unchanged since then
5
+ (`git diff e86c13a -- code/tt_meteor` is empty); this file's commit adds only the measurement scripts
6
+ (`code/scripts/bench.py` rewritten with p50 / p99, several inputs, AICLK and the agreement check; `profile_ops.py`;
7
+ `profile_summary.py`). tt-metal `44d66500520` + `patches/tt-metal-eth-dispatch.patch` +
8
+ `patches/tt-metal-reshape-rm-sys1419.patch`; ttaw 0.23.2. Weights `AutowareFoundation/meteor` @ `01a5f6d71df`
9
+ (tag `v1.0`), `meteor_v157c3Z.onnx`, run as released (`input_norm=onnx`, D12).
10
+
11
+ Inputs, batch 1 (one frame = 8 camera slots of 768×432 + calibration + ego speed):
12
+ - **sample**: `code/tt_meteor/samples/pandaset_019_f40.json`, frame 40 of PandaSet sequence 019 in METEOR's
13
+ 8-slot layout (7 camera JPEGs + the absent BACK_NARROW; preset `pandaset_019`; ego speed 10.88 m/s; CC BY 4.0 +
14
+ PandaSet Dataset Terms). It is the sample the port ships today. The release may move it to the git-ignored
15
+ `staging_samples_pandaset/` until the user approves shipping PandaSet data; `bench.py` finds it there too. Its
16
+ stored fp32 CPU reference gives the agreement check of every run;
17
+ - **synthetic**: eight 768×432 uint8 noise images made by `bench.py` (data generated by us), preset `pandaset_019`,
18
+ 10 m/s. The kind of input a release can always ship;
19
+ - **PandaSet 090 f40** and **nuScenes 0103 kf09**: public-dataset frames of two other rigs, as API requests built
20
+ from `research/meteor/public_data/inputs` (local only, never shipped; nuScenes is CC BY-NC-SA). Their camera
21
+ arrays, K, T and v0 equal the goldens' graph inputs exactly (checked on the host before the bench).
22
+
23
+ Configuration of every number: ETH dispatch, 1 CQ, 12×10 grid (`device.compute_with_storage_grid_size()` printed in
24
+ each log), the shipped precision policy (PLAN §9.3: fp32 weights, HiFi4 + fp32 accumulation; image branch and det
25
+ path in three bf16 terms, `METEOR_IMAGE_PRECISION=terms3`; BEV trunk bf16 activations), unless a row says otherwise.
26
+
27
+ ## Environment
28
+
29
+ | item | value |
30
+ |---|---|
31
+ | tt-metal | `44d66500520` (main) + `patches/tt-metal-eth-dispatch.patch` (sha256 `08d0ddf6…`) + `patches/tt-metal-reshape-rm-sys1419.patch` (sha256 `74878683…`, the SYS-1419 row-major reshape fix that ended the ETH hangs, PORT_LOG 12.7-12.8). Both are applied to the workspace tree (`git diff --stat`: 5 files) and shipped in `patches/`; `eth_patch: true` in every log |
32
+ | ttaw | 0.23.2, vendored from `common` `60b6dd7` (`code/tt_meteor/ttaw/VENDORED.json`; `vendor.py --check`: 0 differences, so no re-vendor and no gate re-run was needed) |
33
+ | device | Blackhole p150b, KMD 2.10.0, firmware bundle 19.13.1.0. ETH dispatch, 1 CQ, 12×10 = 120 cores, `l1_small_size` 32 KiB, `trace_region_size` 256 MiB. ONE trace (`frame`, the whole network), **4,678 programs per frame**, 1,131 programs in the program cache |
34
+ | AICLK | 1350 MHz median (min 1343) over 8,412 samples (one every 50 ms) during every bench and matrix loop (`ttaw.profiling.AiclkSampler`, sysfs); board power median 69-78 W (max 203 W), ASIC 64.5-71.3 °C. The profiler CSV header reads CHIP_FREQ 1350 MHz |
35
+ | numerics | HiFi4 on all 2,402 compute programs of a frame (profile) |
36
+ | host | AMD EPYC-Rome VM, 8 vCPUs, 90 GB RAM, Linux 6.8, shared with up to 8 other agents' jobs: load average 4.1-12.5 during the stage bench, 4.5-12.1 during the matrix; `OMP_NUM_THREADS=4`. Host rows move with that load (the sample's first plain loop ran at load 12.5 and has the worst e2e p99); device rows do not (trace p50 556.7-556.9 ms on every input and run) |
37
+
38
+ ## How to run
39
+
40
+ ```bash
41
+ ROOT=/home/ubuntu/experiments/tt-models; cd $ROOT/bundles/meteor-p150; source $ROOT/bin/tt-env.sh
42
+ export PYTHONPATH=$PWD/code:$PYTHONPATH HF_HUB_OFFLINE=1 TT_MODEL_WEIGHTS_REVISION=01a5f6d71df5ecbbb5853ec600825481d57b9c6b
43
+ # (every METEOR device job needs METEOR_DEVICE_OK=1 from the orchestrator; never arm tt-triage)
44
+ # accuracy gates (per-module PCC + end-to-end agreement vs the fp32 CPU reference), read-only, one job
45
+ $ROOT/bin/devrun -t 1800 -- env TTAW_GATES_READONLY=1 python -m pytest -q -s \
46
+ code/tt_meteor/tests/test_pcc_device.py code/tt_meteor/tests/test_e2e_device.py
47
+ # stage breakdown (API split, host_in / H2D / trace / b2b / D2H / unpack, e2e p50 / p99, AICLK, agreement)
48
+ $ROOT/bin/devrun -t 2400 -- python code/scripts/bench.py --iters 60 --inputs sample,synthetic,<name>=<manifest> --json <out>.json
49
+ # dispatch / CQ matrix: one process per configuration
50
+ $ROOT/bin/devrun -t 2700 -- python code/scripts/bench.py --inputs sample --iters 50 --dispatch worker --num-cqs 2 --json <out>.json
51
+ # rig switch (each call writes the other rig's lift tables)
52
+ $ROOT/bin/devrun -t 1500 -- python code/scripts/bench.py --inputs a=<rig A manifest>,b=<rig B manifest> --iters 5 --rig-switch 40 --json <out>.json
53
+ # device profile: one eager frame with module signposts + two signposted traced replays
54
+ $ROOT/bin/devrun -t 3600 -- python -m tracy -r -p -v --no-web-server --op-support-count 6000 \
55
+ -o $ROOT/generated/profiler/meteor_baseline code/scripts/profile_ops.py --rounds 2
56
+ tt-perf-report <ops_perf_results_*.csv> --start-signpost frame --end-signpost frame_end --tracing-mode --arch blackhole
57
+ python code/scripts/profile_summary.py $ROOT/generated/profiler/meteor_baseline --json s.json --md s.md
58
+ # ttnn-visualizer graph report of one eager frame (slow, never used for timing)
59
+ $ROOT/bin/devrun -t 2700 -- python code/scripts/profile_ops.py --graph-report $ROOT/generated/ttnn_reports/meteor_baseline
60
+ ttnn-visualizer --profiler-path $ROOT/generated/ttnn_reports/meteor_baseline/frame \
61
+ --performance-path $ROOT/generated/profiler/meteor_baseline/reports/2026_10_10_15_12_27
62
+ ```
63
+
64
+ - **One frame is 4,678 programs**, above the device profiler's default buffer of about 1,000 programs, so the
65
+ profile runs with `--op-support-count 6000` and flushes the buffer (`ttnn.ReadDeviceProfiler`) before and after
66
+ each section.
67
+ - The log warns of dropped markers at 15:05:53-15:06:02. That is the flush right after the compile frame (the
68
+ first eager frame, 806.6 s of JIT with the profiler build), before any measured window.
69
+ - The measured windows are complete: both traced rounds have 4,678 programs and kernel sums of 555.28 / 555.61 ms,
70
+ and the eager frame has the same op sequence as the trace, op for op (4,678 programs, kernel sum 554.86 ms).
71
+ - The job drivers and analysis scripts are in `logs/meteor/baseline/scripts/` of the workspace (`env.sh` holds the
72
+ common environment, `run_rest.sh` / `run_e.sh` / `run_f.sh` the job order).
73
+
74
+ ## Accuracy (baseline)
75
+
76
+ The model code, ttaw and weights are those of `e86c13a`, whose gates ran three times under ETH on 2026-10-10
77
+ (PORT_LOG 12.8, `logs/meteor/hangfix/suite_final{1,2,3}.log`: 15 passed each, read-only gates, identical values in
78
+ all three runs). `vendor.py --check` reports 0 differences and `code/tt_meteor/` is unchanged, so the frozen gates
79
+ were not re-run for this file.
80
+
81
+ | check | gate | baseline (device suite x3 under ETH, read-only gates: 15 passed each, identical values) |
82
+ |---|---|---|
83
+ | teacher-forced stage PCC (26 gates of `tests/test_pcc_device.py`, PandaSet 019 f40) | >= 0.99 | min 0.9999890 (`tf.risk`); `depth_mean` 0.99948 (reported) |
84
+ | teacher-forced lane / seg2d / depth argmax agreement | >= 0.99 | 0.9984 / 0.9965 / 0.9936 |
85
+ | chained frame (the served graph from the cameras), min output PCC: PandaSet 019 / 090, nuScenes 0103, valday #40 | >= 0.99 | 0.99989 / 0.99990 / 0.99991 / 0.99982 |
86
+ | chained lane / seg2d / depth agreement (worst frame) | >= 0.99 | 0.9963 / 0.9965 / 0.9925 |
87
+ | det3d strict recall / precision (served decode): sample, PandaSet 090, nuScenes, valday #40 | >= 0.95 / >= 0.95 | 1.0/1.0, 1.0/1.0, 1.0/1.0, **0.971 / 0.971** (34 / 35) |
88
+ | selected ego path deviation (max waypoint), worst frame | <= 0.3 m | 0.170 m (PandaSet 090) |
89
+ | sample `/predict` vs stored CPU-reference body (smoke compare): recall / precision / max \|Δscore\| | >= 0.95 / >= 0.95 / - | 1.0 / 1.0 / 0.0081 |
90
+ | imagenet + linear variant (`test_variants_device.py`), float outputs min PCC | >= 0.99 | 0.99988 |
91
+ | CPU reference vs ONNX Runtime on the shipped ONNX (the reference itself is right) | >= 0.9999 | min 0.99999990 (155 taps) |
92
+ | ETH stress (`code/scripts/stress_frames.py --mode rigs`: rig change every frame, bit-compare per rig) | no hang, 0 mismatches | 2000 frames clean |
93
+
94
+ - **Every bench run also checks its own served output.** The sample runs as a fresh stream and is compared with
95
+ its stored CPU reference with the container-smoke gates (`ttaw.server.smoke.compare_with_reference`):
96
+ 8 / 8 boxes matched, recall = precision = 1.0, max |Δscore| 0.0081, lane agreement 0.99770, selected path
97
+ deviation 0.108 m. Identical in ETH-1CQ, ETH-2CQ and WORKER-2CQ (WORKER lane agreement 0.99770 too).
98
+ - Host suite on the fake ttnn (the host test files, explicitly listed): **100 passed, 1 skipped**
99
+ (`test_reference_cpu.py`: no onnxruntime in tt-env), `logs/meteor/baseline/host_suite.log`.
100
+
101
+ ## Performance (warm, batch 1, p50 (p99) of 60 iterations per loop)
102
+
103
+ Each row of the API split is timed inside one `model(...)` call (wrappers around the parts of
104
+ `TtMETEOR.run_frame`); the `TraceRunner` rows come from a second loop with every stage synchronised on its own.
105
+ Images are decoded before timing (the camera driver hands METEOR decoded images).
106
+
107
+ | stage | sample (PandaSet 019) | synthetic | PandaSet 090 | nuScenes 0103 |
108
+ |---|---:|---:|---:|---:|
109
+ | host preprocess (`_prepare`: INTER_AREA to 768×432, K / T, METEOR's runtime pre-processing) | 11.2 (35.8) | 11.1 (13.2) | 9.3 (10.3) | 9.2 (10.1) |
110
+ | device stage of the API (`api.device`), of which: | 774.9 (883.9) | 735.1 (756.6) | 755.0 (787.2) | 765.9 (780.8) |
111
+ | … NCHW → NHWC camera rows on the host (`image_rows`) | 26.3 (31.9) | 23.1 (30.9) | 21.6 (32.0) | 22.1 (35.6) |
112
+ | … `TraceRunner` call: host tensor + H2D + replay + segmented D2H | 689.5 (711.4) | 646.6 (668.8) | 677.7 (694.4) | 675.9 (700.1) |
113
+ | … join of the 5 readback segments (`unpack`) | 15.9 (53.6) | 15.6 (16.1) | 15.6 (16.5) | 15.6 (16.4) |
114
+ | … device layouts → the 19 ONNX outputs (`outputs_from_device`, host) | 49.4 (95.9) | 47.8 (51.8) | 46.0 (52.4) | 45.7 (50.9) |
115
+ | host postprocess (METEOR's C++ decode rules + temporal state) | 129.9 (164.6) | 124.0 (130.7) | 130.0 (141.3) | 131.2 (143.1) |
116
+ | **e2e `model()`** | **918.2** (1071.8) | **875.7** (901.8) | **902.8** (927.8) | **914.4** (931.1) |
117
+ | `TraceRunner` stages: host tensor (`host_in`) | 1.17 (1.39) | 1.14 (1.22) | 1.17 (3.11) | 1.12 (1.17) |
118
+ | H2D (8 × 432 × 768 × 3 uint8 + v0, synchronised) | 66.5 (74.2) | 66.4 (84.2) | 66.4 (68.2) | 66.4 (67.6) |
119
+ | **device trace, one blocking replay** | **556.85** (559.34) | 556.87 (562.78) | 556.85 (559.24) | 556.68 (556.97) |
120
+ | **back-to-back replays, per frame** (20) | **556.75** | 556.79 | 556.82 | 556.57 |
121
+ | D2H (5 segments, 75.0 MB fp32) | 17.2 (63.5) | 17.8 (66.7) | 47.5 (70.5) | 46.7 (67.8) |
122
+ | `TraceRunner` e2e (call + unpack) | 658.3 (680.5) | 658.2 (670.2) | 688.3 (720.2) | 688.1 (708.4) |
123
+
124
+ - **Throughput.** Back-to-back replays give **1.80 frames/s** of device throughput (556.8 ms per frame); a
125
+ synchronous `model()` call gives 1.09-1.14 frames/s (875.7-918.2 ms).
126
+ - **The device time does not depend on the input** (one fixed-shape graph; the lift tables only change the
127
+ sampling positions): trace p50 556.7-556.9 ms on all four inputs, AICLK 1350 MHz throughout.
128
+ - **Where the time goes on the sample (918 ms):** device trace 557 ms (61 %); host decode 130 ms (14 %); host
129
+ conversion of the 75 MB readback (`unpack` + `outputs_from_device`) 65 ms (7 %); H2D 66 ms (7 %); host
130
+ NCHW → NHWC 26 ms; D2H 17-48 ms; host preprocess 11 ms.
131
+ - **D2H is bimodal** (17 ms or 46-48 ms for the same 75 MB, both within one process; p99 63-71 ms on every input).
132
+ The device and the bytes are the same, so it is a host-side effect of the shared host (cause not isolated;
133
+ UNVERIFIED: page faults or memory pressure while other agents run).
134
+ - **Load** with a warm JIT cache: 49.7 s (weights 0.3 s, graph build 2.6 s, warm-up 42.2 s: one eager frame 41.0 s
135
+ plus the capture 1.2 s). A process on a new dispatch configuration compiles its own kernels: ETH-2CQ 141.9 s,
136
+ WORKER-2CQ (11×10) 767.7 s.
137
+ - **Rig change** (`--rig-switch`; calls alternate between two calibrations, so every call writes the other rig's
138
+ `lift_grid` + 20 table chunks): PandaSet 090 ↔ nuScenes 0103, 40 calls (`rigswitch_eth-1cq.json`): the table
139
+ write takes **182.3 ms p50** (p99 203.9; 211 MB in 21 synchronised transfers, about 1.16 GB/s) and the
140
+ call 1128.5 ms p50 (p99 1203.1) instead of about 905 ms. The lift geometry itself is cached per rig on the host. A
141
+ fixed rig (the vehicle case) pays this once. (The stage bench's own `--rig-switch` loop alternated the sample and
142
+ the synthetic input, which share the `pandaset_019` preset, so it measured no rig change; it is not quoted.)
143
+
144
+ ## Dispatch / CQ matrix (D14)
145
+
146
+ Setup: the sample (identical frames in every row), one process per configuration, each its own devrun window
147
+ (14:29-14:51 UTC), 50 iterations per loop plus 20 back-to-back frames. The first row is the stage bench.
148
+
149
+ | configuration | grid | programs (cache) | b2b frame ms | trace p50 (p99) ms | H2D p50 / min ms | D2H p50 ms | runner e2e p50 ms | e2e p50 (p99) / min ms | host load |
150
+ |---|---|---:|---:|---:|---:|---:|---:|---:|---|
151
+ | **ETH-1CQ** (shipped; stage bench) | 12×10 | 1,131 | **556.75** | **556.85** (559.34) | 66.46 / 66.42 | 17.19 | 658.3 | 918.2 (1071.8) / 883.1 | 12.5 → 5.4 |
152
+ | ETH-2CQ | 12×10 | 1,131 | 561.37 | 561.52 (562.05) | 67.36 / 67.33 | 18.02 | 662.4 | 916.8 (941.8) / 881.4 | 5.1 → 4.5 |
153
+ | WORKER-2CQ | 11×10 | 1,161 | 578.85 | 579.04 (579.46) | 62.81 / 62.76 | 17.30 | 675.8 | 965.9 (1086.0) / 907.6 | 12.1 → 10.6 |
154
+ | WORKER-1CQ (PORT_LOG 12.6, measured at an earlier commit, for reference only) | 11×10 | | | 573.0 | | | | | |
155
+
156
+ - **ETH stays the default.** WORKER-2CQ replays the frame 22.1 ms slower (578.9 vs 556.8 ms, +4.0 %): on 11×10
157
+ ttnn partitions differently and builds 1,161 programs instead of 1,131. WORKER's H2D is 3.6 ms faster (62.8 vs
158
+ 66.4 ms), which does not make up the difference. WORKER also needs its own 13-minute JIT compile at first load.
159
+ - **1 CQ.** ETH-2CQ replays 4.6 ms slower than ETH-1CQ (561.4 vs 556.8 ms; ETH-2CQ runs without dispatch_s, PLAN
160
+ §0.2), and a synchronous request gains nothing from the second queue (e2e 916.8 vs 918.2 ms at a lower host load).
161
+ **Decision: 1 CQ stays pinned** (`DEVICE_DEFAULTS num_command_queues: 1`, `METEOR_NUM_CQS`, D14): the API and the
162
+ server serve one synchronous request at a time. A pipelined serve mode (opportunity 8) is the only way 2 CQs pay.
163
+ - The ETH-2CQ and WORKER-2CQ processes ran clean (rc 0, outputs identical to ETH-1CQ on the agreement check). Before
164
+ the SYS-1419 reshape fix, ETH-2CQ hung within 43 frames (PORT_LOG 12.5); with the fix it is a working option again.
165
+
166
+ ## Host vs device, and every host<->device transfer per forward
167
+
168
+ **Host** (METEOR's own runtimes define pre / post: there is no Autoware package):
169
+ - before the device: request parsing, the INTER_AREA resize to 768×432 (identity on the 768×432 sample), K scaled
170
+ per axis, T inverted, the absent narrow cameras as zero images with their donor's pose (9-11 ms); the lift geometry
171
+ and tables per calibration (cached per rig); the NCHW → NHWC transpose of the 8 cameras (22-26 ms);
172
+ - after the readback: the join of the segments (15.6 ms), `outputs_from_device` (device layouts → the 19 ONNX
173
+ outputs: views, transposes, uint32 → uint8, 46-49 ms), and METEOR's C++ decode rules
174
+ (`host.result.build_output`, 124-131 ms).
175
+
176
+ There is no mid-graph host round trip and no host fallback op: the whole network is ONE trace, replayed per call.
177
+
178
+ **H2D per frame** (`copy_host_to_device_tensor` into the persistent trace inputs, CQ0):
179
+
180
+ | tensor | shape, dtype, layout | bytes | pages × page size | when |
181
+ |---|---|---:|---|---|
182
+ | `imgs` | `[1, 1, 2654208, 3]` uint8 ROW_MAJOR (8 × 432 × 768 NHWC rows) | 7,962,624 | **2,654,208 × 3 B** | every frame: **66.4 ms (0.12 GB/s)** |
183
+ | `v0` (RT-dev param) | `[1, 1, 1, 1]` fp32 TILE | 4 KB (one tile) | | when the speed changes (unchanged values are not re-uploaded) |
184
+ | `lift_grid` | `[8, 400, 250, 2]` fp32 ROW_MAJOR | 6,400,000 | | when the rig changes |
185
+ | `lift_table_00..19` | 20 × `[1, 1, 40000, 64]` fp32 TILE | 20 × 10,240,000 | | when the rig changes; every chunk is followed by a device sync |
186
+
187
+ **D2H per frame:** one packed fp32 tensor `[18309, 1024]` (75.0 MB: `pack_outputs` of the 15 device outputs, the
188
+ argmaxes as uint32 → fp32) read as 5 row segments of at most 16.8 MB, 17.2 ms (4.4 GB/s) when the host is calm.
189
+
190
+ **The image upload is the transfer problem.** 8 MB goes up as 2.65 M pages of 3 B (one pixel per page): 0.12 GB/s.
191
+ The same path moves the 75 MB readback at 4.4 GB/s with 4 KB pages, so camera rows as pages (2,304 B) would take
192
+ about 2-3 ms (UNVERIFIED estimate from the measured D2H rate).
193
+
194
+ ## Grid usage
195
+
196
+ - **Nothing hard-codes 11×10 or 12×10** (VERIFY_PORT): on 11×10 the WORKER run builds its own program set (1,161 vs
197
+ 1,131) and gives the same served output.
198
+ - **1,916 of the 4,678 programs run on all 120 cores, and they hold 407.6 of the 555.3 ms** (73 %):
199
+ the DRAM-sliced convs and their slices, typecasts, the binary ops, the TILE reshapes, the lift grid_samples.
200
+ - 1,234 programs on 110-119 cores (65.8 ms): convs and their halo / move on height-sharded slices.
201
+ - 1,198 programs on 64-109 cores (51.0 ms), 220 on 16-63 cores (18.3 ms), 110 on 1-15 cores (12.6 ms). The costly
202
+ few-core programs:
203
+ - `head.ego_attn` pooling matmul `[1, 800, 96, 500] × [500, 16]` fp32 on **3 cores: 5.10 ms**;
204
+ - `head.risk_sample` `[1, 400000] × [400000, 96]` bf16 (a K = 400,000 reduction) on **3 cores: 4.81 ms**;
205
+ - the W-axis interpolation matmuls of `Resize2d` at 250 → 500 on 16-24 cores: `lift.resize` 2.96 ms,
206
+ `head.seg_refine` 2.89 ms, `bev.lane` 2.89 ms; `bev.traj` `[400·128, 250] × [250, 16]` on 4 cores 1.48 ms.
207
+ - **Every activation is DRAM-interleaved between ops** (C17 / C27 first-port layout). The convs run ttnn's
208
+ automatic DRAM slicing with height-sharded L1 slices (386 PaddedSlice + 386 SliceWrite programs).
209
+ - **L1 is empty between ops.** The ttnn-visualizer graph report of one eager frame (2,820 recorded ttnn operations,
210
+ 37,001 buffer records, all DRAM) shows no L1 buffer at any op boundary. DRAM peaks at 95.9 MiB per bank at the
211
+ lift's fp32 product `[8, 1, 100000, 96] × w` (the 307 MB fp32 camera products next to the 205 MB of fp32 lift
212
+ tables, the 8-camera fp32 feature maps of the three-term image branch and the 48.25 M weights).
213
+
214
+ ## What is already fused / traced
215
+
216
+ **Done:**
217
+ - **The whole network is one trace** (`frame`, 4,678 programs), replayed per call with no host work inside. Capture
218
+ takes 1.2 s; replay == eager is a gate of the port (M7).
219
+ - **Eagerly** (profiler, program cache hot), the same 4,678 programs take the same kernel time (554.86 vs 555.28 ms
220
+ traced), but the eager frame spans 806.1 ms: **251.3 ms of op-to-op gaps** (host dispatch of 4,678 ops, the most
221
+ before `img.depth` 67.4 ms and `img.resnet` 55.3 ms). Traced, the gaps add up to 2.74 ms (0.5 %).
222
+ - **Exact rewrites** (PORT_LOG section 2, all checked against the fp32 reference):
223
+ - BN folded into every conv (float64); the /255 input scale folded into the stem weights; tgate + tfuse merged
224
+ into one 1×1 conv (96 → 100); merged hm | reg heads (det and det2d); ctx + paint (21 → 96) as one K-split conv;
225
+ the MHAs in folded form (attention pools); traj's `reg_pre` input folded in;
226
+ - every Resize as separable fp32 interpolation matmuls (`F.interpolate` half-pixel incl. the edge clamp, also the
227
+ non-integer ones);
228
+ - the lift as two `grid_sample` gathers + a per-calibration bin-interpolation table + fp32 camera sums;
229
+ - pools as constant matmuls, `mean(fused)` computed once;
230
+ - argmaxes channel-major on the device; one packed readback in <= 16 MB segments.
231
+ - **Fused epilogues:** ReLU in every conv; fp32 conv outputs where a value feeds a threshold, a softmax, a clamp or
232
+ the planner.
233
+
234
+ **Not fused / not resident:**
235
+ - every activation round-trips DRAM between ops;
236
+ - **the three-term convs (E20 / E22)**: every conv of the image branch and the det path runs three bf16 convs
237
+ (`x_hi·W_hi`, `x_hi·W_lo`, `x_lo·W_hi`) plus a typecast pair to split the fp32 activation and two fp32 adds;
238
+ - **TILE reshape copies** between `[1, 1, N·H·W, C]` and `[1, H, W, C]` around every `Resize2d` and every conv
239
+ whose map width (500, 250, 48, ...) is not a multiple of 32: 435 copies, 128.5 ms;
240
+ - the stem's exact fp32 max pool is a chain of DRAM untilize / pad / strided slices / maximums / tilize (25.3 ms);
241
+ - the output tail untilizes, pads and reshapes every output into the packed rows (12.9 ms);
242
+ - no custom kernel, no megakernel.
243
+
244
+ ## Device profile (traced replay; kernel sum vs span, op-to-op gaps)
245
+
246
+ Traced replays, ETH-1CQ, the sample (`profile_ops.py --rounds 2`; stage labels from the eager frame's module
247
+ signposts, carried over op for op). `profile_summary.json` / `.md` hold every table below.
248
+
249
+ | | frame (round 1) | frame (round 2) |
250
+ |---|---:|---:|
251
+ | programs | **4,678** | 4,678 |
252
+ | device kernel sum | **555.28 ms** | 555.61 ms |
253
+ | span (first FW start → last FW end at 1350 MHz) | 558.02 ms (b2b 556.75) | 558.35 ms |
254
+ | op-to-op gaps | **2.74 ms (0.5 %)** | 2.74 ms |
255
+ | math fidelity | HiFi4 on all 2,402 compute programs | |
256
+
257
+ **The trace is kernel-time bound: dispatch is not a sink.** At 0.6 µs per op, removing every op-to-op gap would
258
+ save 2.7 ms; a megakernel pays only through the kernel time (DRAM round trips, layout copies) it removes.
259
+
260
+ **Compute vs everything else.** Data movement and layout ops, not math, dominate:
261
+ - TILE reshapes (23.1 %), typecasts (9.2 %), slicing (PaddedSlice / SliceWrite / Slice 13.1 %), tilize / untilize /
262
+ pad (6.5 %), halo / move / sharding (3.6 %), transpose / concat / permute (1.3 %): **315 ms of the 555 ms (57 %)
263
+ move or reformat data**;
264
+ - the convs (65.2 ms, 585 programs) run at 63 % of the HiFi4 FLOPs inside their kernels (height-sharded, weighted
265
+ mean, tt-perf-report) and 31 % (block-sharded); the matmuls at 7.9 % (weighted mean);
266
+ - tt-perf-report models the frame at 2.1 % of the DRAM roofline (11 GB/s).
267
+
268
+ Per block and module (traced kernel time):
269
+
270
+ | block / module | ops | kernel ms | share | where the time goes (traced kernel ms) |
271
+ |---|---:|---:|---:|---|
272
+ | **image branch** (8 cameras, terms3) | **2,621** | **268.27** | **48.3 %** | |
273
+ | `img.normalize` (uint8 → bf16) | 1 | 8.57 | 1.5 % | one Typecast of `[2654208, 3]` ROW_MAJOR (3-byte pages) |
274
+ | `img.stem` (7×7 s2 conv + exact fp32 max pool) | 44 | 42.51 | 7.7 % | max pool chain 25.3 (strided Slice 17.56, maximums 4.29, pad 2.15, untilize 0.86, tilize 0.40), PaddedSlice of the 3-channel input 13.85 (6 DRAM slices, padded to 8 channels), the 6 conv slices 0.99 + their SliceWrites 1.22 |
275
+ | `img.resnet` (ResNet-34, terms3) | 949 | 57.54 | 10.4 % | BinaryNg 15.26 (term sums, residuals), conv 13.52, ReshapeView 10.41, Typecast 7.53 |
276
+ | `img.fpn` (laterals, resizes, adds) | 75 | 25.03 | 4.5 % | matmul 8.24 (resizes), ReshapeView 5.82, BinaryNg 4.72, untilize 2.28 |
277
+ | `img.fpn_fuse` | 96 | 7.66 | 1.4 % | |
278
+ | `img.seg2d` (decoder + clamp + argmax) | 373 | 40.89 | 7.4 % | BinaryNg 9.69, conv 6.95, Typecast 6.21, matmul 5.60 |
279
+ | `img.depth` (head + argmax + softmax) | 511 | 43.56 | 7.8 % | BinaryNg 12.25, conv 9.13, Typecast 7.07, SliceWrite 6.10, a `[64, 165888]` fp32 TILE reshape 3.34 |
280
+ | `img.depth_mean`, `img.ctx_paint`, `img.det2d`, `img.tl` (+ the lift-input conversions) | 572 | 42.51 | 7.7 % | `img.det2d` 28.12 (terms3) |
281
+ | **lift** | **61** | **56.26** | **10.1 %** | GridSample 28.21 (two calls of 14.1 ms: `[8, 108, 192, 96 / 64]` bf16 RM maps, `[8, 400, 250, 2]` fp32 grid → `[8, 400, 250, C]`), weights / sums BinaryNg 6.92, tilize 4.25, typecast 2.84; `lift.resize` ×2 10.58 (matmul 4.88, reshape 1.91) |
282
+ | **BEV trunk** (800×500 and 400×250 maps) | **939** | **147.48** | **26.6 %** | ReshapeView 69.88 (104 copies), conv 20.84, matmul 13.41, BinaryNg 11.48, SliceWrite 5.93 |
283
+ | … `bev.lane` | 326 | 58.16 | 10.5 % | ReshapeView 27.11, conv 12.65, matmul 7.52 |
284
+ | … `bev.det` (terms3 det stem + head) | 348 | 48.56 | 8.7 % | ReshapeView 24.00, BinaryNg 6.73, conv 5.20, typecast 3.76 |
285
+ | … `bev.fuse`, `bev.traj`, `bev.occupancy`, `bev.stationary`, `bev.risk` | 265 | 40.76 | 7.3 % | `bev.fuse` 17.13 (reshape 6.66), `bev.traj` 10.36 |
286
+ | **heads + planner** | **1,012** | **70.34** | **12.7 %** | ReshapeView 24.73, matmul 17.42, BinaryNg 5.32, conv 4.23 |
287
+ | … `head.seg_refine` | 264 | 34.19 | 6.2 % | ReshapeView 14.71, matmul 6.05 |
288
+ | … `head.box_refine` (terms3) | 476 | 15.69 | 2.8 % | ReshapeView 6.43, BinaryNg 3.06 |
289
+ | … planner (`ego_stem`, `ego_mlp`, `ego_attn`, `sem`, `risk_sample`, `decoder_wp`, `e2e`) | 272 | 20.46 | 3.7 % | `ego_attn` 8.09 (3-core matmul 5.10), `risk_sample` 5.07 (3-core matmul 4.81), `ego_stem` 4.38 |
290
+ | **output tail** (`out.pack`) | **45** | **12.93** | **2.3 %** | ReshapeView 6.00, pad 3.40, untilize 2.50 |
291
+ | **total** | **4,678** | **555.28** | | |
292
+
293
+ | sink (op type), frame | ms | % | where |
294
+ |---|---:|---:|---|
295
+ | ReshapeView (435) | 128.52 | 23.1 | TILE copies `[1, 1, N·H·W, C]` ↔ `[1, H, W, C]` at W = 500 / 250 / 48: `[400000, 96]` → `[800, 500, 96]` bf16 10 × 1.65 ms, `[100000, 128]` ↔ `[400, 250, 128]` (bf16 and fp32) 31 copies 18.8 ms, `[10368, 256]` ↔ `[216, 48, 256]` 90 copies 9.8 ms (resnet layer 3-4), ...; by module: `bev.lane` 27.1, `bev.det` 24.0, `head.seg_refine` 14.7, `img.resnet` 10.4, `bev.fuse` 6.7, `head.box_refine` 6.4, `out.pack` 6.0, `img.fpn` 5.8 |
296
+ | BinaryNg (440) | 85.66 | 15.4 | the fp32 term sums of the three-term convs, residual adds, the stem max-pool maximums, lift weights and camera sums |
297
+ | Conv2d (585) | 65.17 | 11.7 | each image / det conv three times (terms3); 63 % (height-sharded) / 31 % (block-sharded) of the HiFi4 FLOPs inside the kernels |
298
+ | Halo + Move + interleaved ↔ sharded (1,580) | 19.86 | 3.6 | the conv glue of the DRAM-sliced convs |
299
+ | Matmul (131) | 54.31 | 9.8 | the planner's two 3-core matmuls 9.91; the rest (44.4) mostly the `Resize2d` interpolation matmuls (fp32, many on 16-48 cores), pools and the paint fold |
300
+ | Typecast (359) | 50.91 | 9.2 | fp32 → bf16 TILE (234 programs, 25.68 ms) and bf16 → fp32 (121, 16.58 ms): the term split of the three-term convs; uint8 → bf16 input 8.57 |
301
+ | PaddedSlice + SliceWrite + Slice (820) | 72.56 | 13.1 | ttnn's automatic DRAM slicing of the convs 49.3 (the stem's 3-channel input 13.85); the max-pool strided slices 17.56; lift camera slices 2.08 |
302
+ | GridSample (3) | 28.22 | 5.1 | the lift (2 × 14.1 ms) and the planner's risk sampling |
303
+ | Tilize / Untilize / Pad (194) | 36.34 | 6.5 | the output tail 6.4, lift sampled-row tilize 4.25, the stem max pool 3.4, resize and planner glue |
304
+
305
+ Top 10 device ops of the frame (traced):
306
+
307
+ | # | op | where | µs | cores |
308
+ |---:|---|---|---:|---:|
309
+ | 1 | GridSample | `lift`: ctx `[8, 108, 192, 96]` bf16 RM, `[8, 400, 250, 2]` fp32 grid → `[8, 400, 250, 96]` | 14,119 | 120 |
310
+ | 2 | GridSample | `lift`: depth probabilities `[8, 108, 192, 64]` → `[8, 400, 250, 64]` | 14,093 | 120 |
311
+ | 3 | Typecast | `img.normalize`: `[2654208, 3]` uint8 → bf16 ROW_MAJOR | 8,575 | 120 |
312
+ | 4 | Matmul | `head.ego_attn`: `[1, 800, 96, 500] × [500, 16]` fp32 (attention pooling) | 5,098 | **3** |
313
+ | 5 | Matmul | `head.risk_sample`: `[1, 400000] × [400000, 96]` bf16 | 4,809 | **3** |
314
+ | 6 | Slice | `img.stem` max pool: `[8, 218, 386, 64]` fp32 RM → `[8, 218, 192, 64]` (stride 2) | 3,936 | 120 |
315
+ | 7 | Slice | `img.stem` max pool (second column phase) | 3,914 | 120 |
316
+ | 8 | Slice | `img.stem` max pool (third column phase) | 3,910 | 120 |
317
+ | 9 | ReshapeView | `img.depth`: `[1, 1, 64, 165888]` → `[1, 64, 5184, 32]` fp32 TILE | 3,335 | 120 |
318
+ | 10 | Matmul | `lift.resize`: `[1, 400, 96, 250] × [250, 500]` fp32 (W axis ×2) | 2,961 | 24 |
319
+
320
+ Then the two other W-axis resize matmuls at 250 → 500 (2,890 / 2,885 µs on 16 cores), the six PaddedSlices of the
321
+ stem's 3-channel input (2,285-2,361 µs each), the max-pool pad (2,150 µs) and the output tail's `[400384, 9]` →
322
+ `[3519, 1024]` reshape (2,086 µs).
323
+
324
+ - Shapes in the CSV come from the first program of a cached program (binary_ng reuses one program across shapes),
325
+ so the per-module attribution uses the stage labels, not the CSV's shape columns. Ops a module issues inline after
326
+ its last call belong to the module signposted last (e.g. the depth softmax to `img.depth_mean`, the lift-input
327
+ conversions to `img.tl`).
328
+ - `tt_perf_report_frame.csv` has the per-op FLOPs and DRAM figures; `profile_extra.json` the per-module op-code
329
+ tables, the reshape shapes and the core-count histogram.
330
+
331
+ **The precision cost found in the port** (A/B only: the bench's agreement check on the sample against its stored
332
+ CPU reference with the container-smoke gates; the frozen gates were not run):
333
+
334
+ | policy | b2b frame ms | Δ vs shipped | sample check (smoke gates): matched, recall / precision, max \|Δscore\|, lane agreement, path deviation | frozen gates |
335
+ |---|---:|---:|---|---|
336
+ | **shipped: image branch and det path in three bf16 terms (fp32 activations), BEV trunk bf16** | **556.75** | – | 8 / 8, 1.0 / 1.0, 0.0081, 0.99770, 0.108 m | all pass (above) |
337
+ | image branch with bf16 activations (`METEOR_IMAGE_PRECISION=bf16`; det path still terms3), measured here (`precision_image-bf16.json`, 20 iterations) | **390.83** | **−165.9 ms (−29.8 %)** | 8 / 8, 1.0 / 1.0, 0.0046, 0.99781, 0.077 m (center error up to 0.37 m) | **fails** the depth argmax gate: 0.9797 < 0.99 (PORT_LOG E20, port measurement on the same frame; not re-run here) |
338
+ | det path in bf16 (`det_precision="bf16"`, no env knob; port A/B, PORT_LOG 11.4: trace 548.5 vs 507.4 ms before the ETH / reshape changes) | | −41 ms (port) | | **fails** strict det3d on valday #40: recall 0.943 < 0.95 (PORT_LOG E22) |
339
+
340
+ - **The three-term precision of the image branch is the largest single cost of the port: 165.9 ms per frame
341
+ (29.8 %)**, measured as a same-day A/B on the same build. Together with the det path (+41 ms at the port) the
342
+ three-term policy accounts for about 207 ms (37 %). Both cheaper policies fail a frozen gate (depth argmax
343
+ agreement, strict det3d recall), so the cost is real accuracy, not over-engineering. Item 2 targets the mechanism
344
+ (three convs, a typecast split and two fp32 adds per conv), not the precision.
345
+ - The bf16 run passes the sample's smoke comparison, which shows how loose that check is next to the frozen gates.
346
+
347
+ ## Ranked optimization opportunities
348
+
349
+ How to read the estimates:
350
+ - Gains are on the device frame (556.8 ms, 1.80 frames/s) unless marked e2e. Each one is the measured kernel time
351
+ of the programs the item removes, with a range for what replaces them.
352
+ - Items overlap, so the gains do not add up.
353
+ - Every item keeps the frozen gates above and the precision policy (PLAN §9.3), and lands as one commit with one
354
+ `METEOR_*` A/B knob (PLAN §5.1).
355
+
356
+ 1. **TILE reshape copies (trace −90..−120 ms; the largest item).**
357
+ - 435 `ReshapeView` programs take 128.5 ms (23.1 %). They are real copies because the map widths (500, 250,
358
+ 48, 125, ...) are not multiples of 32: `[1, 1, N·H·W, C]` ↔ `[1, H, W, C]` around every `Resize2d` and every
359
+ conv whose neighbour wants the other form. By module: `bev.lane` 27.1, `bev.det` 24.0, `head.seg_refine` 14.7,
360
+ `img.resnet` 10.4, `bev.fuse` 6.7, `head.box_refine` 6.4, `out.pack` 6.0, `img.fpn` 5.8 ms.
361
+ - Fixes, one knob each: keep every FeatureMap in one form per stage (the convs take the flattened form; give
362
+ `Resize2d` a flattened-input path); ROW_MAJOR views where a reshape is page-aligned (safe again under ETH with
363
+ the SYS-1419 patch); or pad the BEV width to 512 / 256 so the TILE reshape is a view.
364
+ - The same C17 / C24 item as in CenterPoint and BEVFormer (shared ttaw fix, re-vendor rule PLAN §5.1 step 5).
365
+ 2. **Three-term convs as one fused op (trace −90..−150 ms; keeps the precision).**
366
+ - Measured cost of the policy: 165.9 ms for the image branch (above) + about 41 ms for the det path.
367
+ - Mechanism today, per conv: a typecast pair splits the fp32 activation into `x_hi` / `x_lo` (359 typecasts,
368
+ 50.9 ms per frame in total, of which 42.3 ms are TILE fp32 ↔ bf16 conversions, mostly these splits), three
369
+ bf16 convs (585 convs, 65.2 ms), two fp32 adds (the image branch alone runs 282 BinaryNg programs, 61.9 ms).
370
+ - A fused terms conv (one program: hi / lo split in the reader / unpacker, the three products accumulated in the
371
+ fp32 dest, one output) removes the typecasts, the adds and two of three activation reads. Alternatively a true
372
+ fp32-activation conv path of TF32+ quality (I7 investigates why ttnn's own fp32 path is worse).
373
+ - Precision-preserving stepping stone: drop the `x_lo·W_hi` / `x_hi·W_lo` term per layer where the
374
+ `precision_emulation.py` gates allow (E20 found two-term activations insufficient as a whole; per-layer is
375
+ untested).
376
+ 3. **Input path and stem (trace −35..−45 ms; e2e −60 ms).**
377
+ - `img.normalize` 8.6 ms: a uint8 → bf16 typecast over 2.65 M ROW_MAJOR pages of 3 bytes.
378
+ - The stem conv's DRAM slicing reads the 3-channel input six times and pads it to 8 channels (PaddedSlice
379
+ 13.9 ms).
380
+ - The exact fp32 max pool is a DRAM chain (untilize, pad, 6 strided slices, 4 maximums, tilize): 25.3 ms. A
381
+ generic_op 3×3 / s2 fp32 max pool on the conv's height-sharded output, or the pool folded into the stem
382
+ output, takes it to about 1-3 ms (UNVERIFIED).
383
+ - H2D: upload whole camera rows (2,304-byte pages) or a pre-padded 8-channel layout instead of 3-byte pages:
384
+ 66.4 ms → about 2-5 ms at the measured D2H rate (e2e).
385
+ 4. **DRAM slicing and conv glue (trace −25..−40 ms).**
386
+ - 386 PaddedSlice + 386 SliceWrite programs (49.3 ms) and the halo / move / sharding glue (19.9 ms) around the
387
+ DRAM-sliced convs of the 8-camera image maps and the 800×500 BEV maps.
388
+ - Larger slices, explicit height-sharded inputs (L1 holds 171 MB for tensors; layer-1 maps of 8 × 108 × 192 × 64
389
+ fp32 are 42 MB), and the band streaming of item 9 (MK4).
390
+ 5. **Lift as one gather kernel (K5 / MK3; trace −30..−45 ms).**
391
+ - The lift takes 56.3 ms: two `grid_sample` calls of 14.1 ms that write `[8, 400, 250, C]` bf16 to DRAM
392
+ (256 MB), then tilize, fp32 typecast, the product with the 205 MB fp32 table, sums over bins and cameras,
393
+ divide, and the ×2 resize (10.6 ms).
394
+ - A kernel per BEV block reading the L1-resident camera maps (ctx + prob: 53 MB bf16, fits the 171 MB of L1),
395
+ with the per-calibration ELL weights (K slots) and fp32 accumulation, writes only `[400, 250, 96]`.
396
+ 6. **Few-core and fp32 matmuls (trace −12..−20 ms).**
397
+ - `head.ego_attn` `[800·96, 500] × [500, 16]` fp32 on 3 cores (5.10 ms) and `head.risk_sample` `[1, 400000] ×
398
+ [400000, 96]` on 3 cores (4.81 ms): put the long dimension into M or split K over cores with a final
399
+ reduction: well under 1 ms each.
400
+ - The W-axis `Resize2d` matmuls at 250 → 500 run on 16-24 cores (2.9 ms each, 3 calls); `bev.traj`'s
401
+ `[400·128, 250] × [250, 16]` on 4 cores (1.5 ms).
402
+ 7. **Host side (e2e −150..−220 ms; latency only).**
403
+ - Host decode 124-131 ms (`host_post_profile.json`, CPU-only cProfile on the golden outputs: 150 ms on the
404
+ loaded host): the temporal seg fusion 73 ms (log-softmax of `[9, 800, 500]` 25 ms, contiguous copies 26 ms)
405
+ and the 2D decode 69 ms (1,210 per-class `peak_mask` / `sigmoid32` calls on 8 cameras × 10 classes × 3
406
+ scales). Vectorise the 2D peaks over classes and cameras and work in place: about −80..−100 ms (UNVERIFIED).
407
+ - Readback conversion 61-65 ms (`unpack` 15.6 + `outputs_from_device` 46-49) and D2H 17-48 ms of 75 MB fp32:
408
+ bf16 / uint8 outputs packed in their final layouts (views only) halve the bytes and remove most of the
409
+ conversion: about −40..−60 ms.
410
+ - The host NCHW → NHWC transpose (22-26 ms): pre-processing can write NHWC directly (the INTER_AREA output
411
+ is HWC already): about −20 ms.
412
+ 8. **Throughput by pipelining (frames/s; latency unchanged).**
413
+ - The device allows 1.80 frames/s (556.8 ms b2b); a synchronous call gives 1.09-1.14 frames/s.
414
+ - A pipelined serve mode (2 CQs: frame k+1's host pre-processing and upload, and frame k−1's readback and
415
+ decode, overlapped with replay k) is bounded by max(device 561 ms on ETH-2CQ, host about 250-330 ms): about
416
+ 1.7-1.8 frames/s (UNMEASURED). It needs double-buffered outputs and reads on CQ1 (`TraceRunner` reads on CQ0
417
+ today), and the per-stream host state must stay ordered.
418
+ 9. **Megakernel and fusion candidates (D19, PLAN §5.2 / §5.3; SPEC §9 MK0-MK7).**
419
+ - The §5.2 gate is met through DRAM round trips and layout copies (57 % of the kernel time), not dispatch:
420
+ op-to-op gaps are 0.5 % of the frame (2.74 ms for 4,678 programs).
421
+ - Candidates, by measured block time:
422
+ - **MK4 band-streamed BEV trunk** (`bev.*` 147.5 ms + `head.seg_refine` / `box_refine` 49.9 ms = 197 ms, of
423
+ which 91 ms are reshape copies): 800×500 bands in L1, no reshapes, no DRAM slicing. First, after item 1;
424
+ - **MK1 / MK2 image encoder + heads** (268.3 ms, 48 %): L1-resident per camera group; item 2's fused terms conv
425
+ is the precondition;
426
+ - **MK3 lift** (56.3 ms; item 5);
427
+ - **MK0 input** (normalize + stem slicing + max pool: 51 ms; item 3);
428
+ - **MK6 planner** (20.5 ms, of which the two 3-core matmuls are 9.9 ms; item 6 first);
429
+ - **MK7 output tail** (12.9 ms of untilize / pad / reshape in `out.pack`, plus the host conversion of item 7).
430
+ - 2-4-op fusions to try first: the three-term conv triple + split + adds (item 2); conv + residual add + ReLU;
431
+ the fp32 → bf16 typecast folded into the producing conv's output dtype; untilize + pad + reshape of each output
432
+ into one pack kernel; the 4-op argmax chains.
433
+ - A single whole-model megakernel is not realistic: 4,678 programs, 190 heterogeneous convs, 48.25 M parameters,
434
+ 77 MB BEV maps (SPEC §9).
435
+
436
+ ## Raw logs, CSVs and reports
437
+
438
+ Paths are under `/home/ubuntu/experiments/tt-models/`. `L` = `logs/meteor/baseline/`.
439
+
440
+ | what | where |
441
+ |---|---|
442
+ | devrun jobs (`logs/devrun/history.log`, all rc 0, all with `METEOR_DEVICE_OK=1`, no tt-triage) | A 14:06:10-14:17:55 (`L/scripts/jobA_bench.sh`: stage bench, 4 inputs); B1 14:29:17-14:33:52 (`jobB_one.sh eth 2`); B2 14:36:03-14:51:10 (`jobB_one.sh worker 2`); C 14:51:59-15:13:28 (`jobC_profile.sh`: Tracy); D 15:26:43-15:27:46 (`jobD_visualizer.sh`); E 15:34:26-15:39:45 (`jobE_precision.sh`); F 15:41:19-15:43:36 (`jobF_rigswitch.sh`). Order: `L/scripts/run_rest.sh`, `run_e.sh`, `run_f.sh`; `L/run_all.rcs` |
443
+ | stage bench (ETH-1CQ) | `L/bench_eth-1cq.{json}`, `L/jobA_bench.log`; inputs `L/inputs/{pandaset_090_f40,nuscenes_0103_kf09}.json` (`L/scripts/make_manifests.py`) |
444
+ | dispatch matrix | `L/matrix_{eth-2cq,worker-2cq}.json`, `L/jobB_{eth-2cq,worker-2cq}.log` |
445
+ | rig switch | `L/rigswitch_eth-1cq.json`, `L/jobF_rigswitch.log` |
446
+ | precision cost | `L/precision_image-bf16.json`, `L/jobE_precision.log` |
447
+ | device profile | `generated/profiler/meteor_baseline/reports/2026_10_10_15_12_27/ops_perf_results_2026_10_10_15_12_27.csv` (+ `profile_log_device.csv.zst` (zstd-compressed 2026-10-10 to free disk; `zstd -d` restores the 6.0 GB CSV), `tracy_profile_log_host.tracy`; the ttnn-visualizer performance report); `L/profile_summary.{json,md}` (`code/scripts/profile_summary.py`); `L/profile_extra.json` (`L/scripts/profile_extra.py`: core histogram, per-module op codes, reshape shapes, typecast directions); `L/tt_perf_report_frame.{txt,csv}`, `L/tt_perf_report_summary_frame.{csv,png}`; `L/jobC_profile.log`, `L/profile_meta.json` |
448
+ | ttnn-visualizer graph report | `generated/ttnn_reports/meteor_baseline/frame/` (`db.sqlite`, `config.json`; `profile_ops.py --graph-report`, plain graph capture, 8.1 s, 52.7 MB), `L/jobD_visualizer.log`, `L/graph_report_meta.json` |
449
+ | host post-processing (CPU only, loaded host) | `L/host_post_profile.{json,log}` (`L/scripts/host_post_profile.py`) |
450
+ | host suite | `L/host_suite.log` (100 passed, 1 skipped) |
451
+ | gates of this code (`e86c13a`) | `logs/meteor/hangfix/suite_final{1,2,3}.log`, `values_final*`; the 2000-frame ETH stress `logs/meteor/hangfix/meteor_eth_final2000.*` |
OPT_REPORT.md ADDED
@@ -0,0 +1,113 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # meteor-p150 optimization report (p150, ETH dispatch, 12×10 grid)
2
+
3
+ **Status: baseline port; optimization pending.** This is the first public release: the functional port, measured in
4
+ [`OPT_BASELINE.md`](OPT_BASELINE.md) and published with the precision policy of the first release (fp32 weights and
5
+ biases, HiFi4, fp32 and packer L1 accumulation; the image branch and the 3D detection path with fp32 activations,
6
+ every conv as three bf16 terms; the BEV trunk with bf16 activations). No optimization round has run yet. The whole
7
+ network runs as ONE metal trace of **4,678 programs**, which replays in **556.85 ms** (back to back 556.75 ms,
8
+ 1.80 frames/s device-bound). A synchronous `model()` call takes 876-918 ms (p50, by input): around the replay come
9
+ the host decode (124-131 ms), the conversion of the 75 MB readback (61-65 ms), the image upload (66 ms) and the host
10
+ camera transpose (22-26 ms). The trace is kernel-bound (2.74 ms of op-to-op gaps), and 57 % of the kernel time moves
11
+ or reformats data rather than computing. The ranked plan below starts with the TILE reshape copies and the three-term
12
+ convs. Every round will be added to this file, newest first, with its commits, its numbers and its accuracy gates.
13
+
14
+ All numbers: `code/scripts/bench.py` (warm, batch 1, p50 of 60 iterations unless marked), ETH dispatch, 1 CQ,
15
+ 12×10 grid, AICLK 1350 MHz (minimum 1343), tt-metal `44d66500520` + `patches/tt-metal-eth-dispatch.patch` +
16
+ `patches/tt-metal-reshape-rm-sys1419.patch`, measured 2026-10-10 on the baseline commit `e86c13a` (ttaw 0.23.2;
17
+ OPT_BASELINE.md committed as `cc35027`). Device profile: `code/scripts/profile_ops.py` under the Tracy device profiler
18
+ (two traced replays). Accuracy gates: `OPT_BASELINE.md` "How to run".
19
+
20
+ ## Summary
21
+
22
+ | | baseline `e86c13a` | **this release** |
23
+ |---|---|---|
24
+ | device, one blocking replay of `frame` | 556.85 ms (p99 559.34) | same graph (no device code changed); re-measured **556.92 ms** (p99 557.5) on the PandaSet frame, 556.80 ms on the shipped sample |
25
+ | back-to-back replays | 556.75 ms per frame (1.80 frames/s) | re-measured **556.89 / 556.65 ms** |
26
+ | e2e `model()` p50 (p99): PandaSet 019 sample / synthetic noise / PandaSet 090 / nuScenes 0103 | 918.2 (1,071.8) / 875.7 (901.8) / 902.8 (927.8) / 914.4 (931.1) ms | PandaSet 019 935.9 (999.6) ms on a busier host (30 iterations); the new shipped synthetic sample 871.6 (942.1) ms; served `/predict` `timing_ms.total` 904-909 ms |
27
+ | host preprocess / H2D / D2H / readback conversion / postprocess (PandaSet 019) | 11.2 / 66.5 / 17.2 (17-48, bimodal) / 65.3 / 129.9 ms | unchanged |
28
+ | rig change (lift tables of a new calibration) | 182.3 ms table write; call 1,128.5 ms p50 | unchanged |
29
+ | device programs per frame | 4,678 (1,131 program-cache entries) | same trace |
30
+ | kernel sum / span (profile) | 555.28 / 558.02 ms; 2.74 ms of op-to-op gaps | |
31
+ | image branch / lift / BEV trunk / heads + planner / output tail (kernel ms) | 268.27 / 56.26 / 147.48 / 70.34 / 12.93 | |
32
+ | accuracy gates | every PLAN 2.13 gate green, frozen values (OPT_BASELINE.md "Accuracy") | the same frozen gates (values in VERIFICATION_2026-10-10.md) + the container smoke's check on the new shipped synthetic sample |
33
+
34
+ ## Steps (chronological; each row is one commit)
35
+
36
+ | step | commit | what | b2b frame | e2e (PandaSet 019 p50) | accuracy | revert switch |
37
+ |---|---|---|---|---|---|---|
38
+ | baseline | `e86c13a` | the verified functional port (VERIFY_PORT round 1 fixed; ETH default with the SYS-1419 reshape fix, ttaw 0.23.2) | 556.75 ms | 918.2 ms | every gate green, frozen values; 2000-frame ETH rig-switch stress clean | |
39
+ | release | the release commit (`git log`: "Release docs + demo media") | release changes, none on the device graph: the synthetic shipped sample (`samples/synthetic_8cam*`); the PandaSet sample moved unchanged to the git-ignored staging dir; `opencv-python-headless` declared for the image (host post-processing, PORT_LOG I3); the smoke test's sample expectation filled; `server/client.py --sample / --ego-speed`; `examples/quickstart.py` fixed (it called `model(path)`); `stress_frames.py --rigs shipped` (golden-free, for the container) | 556.89 ms | 935.9 ms (busier host) | frozen gates unchanged: 16 device tests passed, every stored value identical to the baseline's ETH runs; the new shipped-sample check 8 / 8 | |
40
+
41
+ ## Round 1
42
+
43
+ Not started. The first items are fixed by the baseline profile (below and OPT_BASELINE.md).
44
+
45
+ ## Known hangs (all rounds)
46
+
47
+ All hangs below happened before this baseline, during the port (PORT_LOG.md section 12;
48
+ `logs/meteor/hangfix/ETH_DISPATCH_HANG_ISSUE.md` of the workspace). None happened at or after `e86c13a`.
49
+
50
+ | when (UTC) | command | cause | status |
51
+ |---|---|---|---|
52
+ | 10-08 / 10-09 (4 jobs) | METEOR device suites and benches under ETH dispatch | a non-deterministic hang of the trace replay; tt-triage launched on the hung chip then hard-reset the server 4 times | triage is now refused by the workspace's device lock; root cause below |
53
+ | 10-10 04:17-05:38 (3 hangs) | `stress_frames.py --mode rigs` under ETH 1 CQ (frames 263, 248) and ETH 2 CQs (frame 43); WORKER 1,600 frames clean | ETH dispatch implicated; contained (no triage, no reset; the host stayed up) | bisected |
54
+ | 10-10 07:15-09:28 (4 hangs + repros) | segmented-trace bisect: lift trace -> `lift_rest` -> `Resize2d` (400×250 -> 800×500) -> its tail -> one op | **one row-major `ttnn.reshape` [800, 48000] -> [400000, 96] (dual-kernel, 192 / 384 B DRAM pages from both NOCs) hangs under ETH dispatch: Blackhole DRAM-arbiter issue SYS-1419**; reproduced standalone in plain ttnn (`code/scripts/repro_resize2d_eth.py`) | **fixed**: `patches/tt-metal-reshape-rm-sys1419.patch` (single-kernel path for small DRAM destination pages; 20,000 replays clean at 192 / 384 B) applied to the image's tt-metal; ttaw 0.23.2 keeps a TILE tail when the patch is absent |
55
+ | 10-10 11:4x-13:51 | after the fix: device suite ×3, a 2000-frame ETH rig-switch stress at `18e0fbd` and another at `e86c13a`, and the baseline's 7 device jobs | none | ETH is the default again |
56
+
57
+ ## Rejected / not kept
58
+
59
+ From the port and the baseline (PORT_LOG.md E20 / E22, OPT_BASELINE.md "precision cost"), kept here so the
60
+ optimization rounds do not repeat them blindly:
61
+
62
+ - **bf16 activations in the image branch** (`METEOR_IMAGE_PRECISION=bf16`): −165.9 ms per frame (−29.8 %), the
63
+ largest single saving measured, but it fails the depth argmax gate (agreement 0.9797 < 0.99): the 64 depth bins are
64
+ near-tied (13 % of the pixels have a top-2 logit margin < 0.02). The shipped sample's loose smoke comparison still
65
+ passes with it, which shows how much stricter the frozen gates are.
66
+ - **ttnn's own fp32-activation conv path**: worse than bf16 (depth agreement 0.9617): it is not TF32 quality.
67
+ - **Two-term activations with fp32 weights**: emulated 0.9940 on the depth argmax gate as a whole (per-layer two
68
+ terms are untested).
69
+ - **A bf16 det stem / box refiner** (`det_precision="bf16"`): −41 ms, but the strict det3d recall on the in-domain
70
+ METEOR frame falls to 0.943 < 0.95 (PORT_LOG E22).
71
+ - **The near-tie aware det3d gate** (E21): withdrawn; the strict PLAN gate is kept on every frame.
72
+ - **WORKER dispatch as the default**: 22.1 ms slower per frame (578.9 ms, 11×10) and a 13-minute JIT compile; it was
73
+ the default only while the ETH hang was open.
74
+ - **A second command queue**: ETH-2CQ replays 4.6 ms slower and a synchronous request gains nothing (D14).
75
+
76
+ ## Profile at the end (trace replay)
77
+
78
+ The baseline profile is the current one (OPT_BASELINE.md "Device profile"): ReshapeView 128.5 ms (23.1 %),
79
+ BinaryNg 85.7, PaddedSlice / SliceWrite / Slice 72.6, Conv2d 65.2, Matmul 54.3, Typecast 50.9, Tilize / Untilize / Pad
80
+ 36.3, GridSample 28.2 ms.
81
+
82
+ ## Remaining backlog (gains on the device frame, 556.8 ms, unless marked e2e)
83
+
84
+ The ranked opportunities of OPT_BASELINE.md; the estimates overlap and do not add up. Each item keeps the frozen
85
+ gates and the precision policy (PLAN §9.3) and lands as one commit with one `METEOR_*` A/B knob.
86
+
87
+ 1. **TILE reshape copies** (−90..−120 ms): 435 `ReshapeView` copies (128.5 ms) between `[1, 1, N·H·W, C]` and
88
+ `[1, H, W, C]` at map widths that are not multiples of 32 (500, 250, 48, ...). One layout per stage, a flattened
89
+ input path for `Resize2d`, ROW_MAJOR views where page-aligned (safe again under ETH with the SYS-1419 patch), or
90
+ padding the BEV width to 512 / 256. The same C17 / C24 item as CenterPoint and BEVFormer (shared ttaw fix).
91
+ 2. **Three-term convs as one fused op** (−90..−150 ms, precision kept): a conv kernel that splits hi / lo in the
92
+ reader and accumulates the three products in the fp32 dest removes 359 typecasts (50.9 ms), most of the 440
93
+ BinaryNg adds and two of three activation reads. Or a TF32+-quality fp32-activation conv (I7).
94
+ 3. **Input path and stem** (−35..−45 ms; e2e −60 ms): the uint8 -> bf16 typecast over 3-byte pages (8.6 ms), the
95
+ stem's 6 DRAM slices of the 3-channel input (13.9 ms), the exact fp32 max pool as a DRAM chain (25.3 ms; a
96
+ generic_op pool about 1-3 ms, UNVERIFIED), and the H2D of 2.65 M pages of 3 bytes (66.4 ms; whole camera rows
97
+ about 2-5 ms, UNVERIFIED).
98
+ 4. **DRAM slicing and conv glue** (−25..−40 ms): 386 PaddedSlice + 386 SliceWrite (49.3 ms) and the halo / sharding
99
+ glue (19.9 ms) of the DRAM-sliced convs.
100
+ 5. **Lift as one gather kernel** (−30..−45 ms): two `grid_sample` calls of 14.1 ms writing 256 MB, then tilize,
101
+ typecast, the product with the 205 MB fp32 table and the sums; a kernel per BEV block with L1-resident camera maps
102
+ and per-calibration ELL weights writes only `[400, 250, 96]`.
103
+ 6. **Few-core and fp32 matmuls** (−12..−20 ms): `head.ego_attn` (5.10 ms) and `head.risk_sample` (4.81 ms) run on 3
104
+ cores; the W-axis `Resize2d` matmuls at 250 -> 500 on 16-24 cores (2.9 ms each).
105
+ 7. **Host side** (e2e −150..−220 ms): the host decode (124-131 ms: the seg fusion and the 2D peak loops), the readback
106
+ conversion (61-65 ms) and D2H of 75 MB fp32 (bf16 / uint8 outputs in final layouts), the NCHW -> NHWC transpose
107
+ (22-26 ms).
108
+ 8. **Throughput by pipelining** (frames/s, latency unchanged): 2 CQs with double-buffered I/O, bounded by the
109
+ device's 1.8 frames/s.
110
+ 9. **Megakernels and fusions** (D19): MK4 band-streamed BEV trunk (197 ms of BEV + refiners, 91 ms of it reshapes)
111
+ first, then MK1 / MK2 image encoder + heads (after item 2), MK3 lift, MK0 input, MK6 planner, MK7 output tail.
112
+ Dispatch is not a sink (0.5 % of the frame), so a megakernel pays only through the DRAM round trips and layout
113
+ copies it removes.
PYTHON.md ADDED
@@ -0,0 +1,165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Python API: METEOR (TIER IV, AutowareFoundation/meteor) on Blackhole
2
+
3
+ Use this API from Python code (a pipeline, a notebook, a ROS 2 node wrapper). You do not need the HTTP server: the
4
+ API and the server share the decoders, the device graph and the post-processing, so the outputs and the speed are
5
+ the same.
6
+
7
+ ## Install
8
+
9
+ Install the package on top of an environment that already has `ttnn`: a tt-metal `python_env` at `44d66500520`
10
+ with `patches/tt-metal-eth-dispatch.patch` and `patches/tt-metal-reshape-rm-sys1419.patch` applied, or the tt-model
11
+ container. From the root of the model repository (the directory that holds `pyproject.toml`, `README.md` and
12
+ `code/`):
13
+
14
+ ```bash
15
+ pip install -e . # the Python API (numpy<2, pillow, pyyaml, onnx, huggingface_hub, safetensors, opencv-python-headless)
16
+ pip install -e ".[server,test]" # + the HTTP server and the tests
17
+ ```
18
+
19
+ The pip project is the repository's top-level `pyproject.toml`; it installs the package from `code/tt_meteor` (there
20
+ is no `pyproject.toml` inside `code/`, because the container build copies `code/` over the tt-metal tree). `ttnn`
21
+ and `torch` are not declared, so pip never replaces tt-metal's own build; tt-metal's `python_env` already has
22
+ OpenCV 4.8.1, which satisfies the `opencv-python-headless>=4.8,<4.12` requirement.
23
+
24
+ The package carries `tt_meteor.ttaw`, the shared code of the Autoware ports to Blackhole (device open, trace runner,
25
+ decoders, model base class, HTTP app), vendored at the version recorded in `code/tt_meteor/ttaw/VENDORED.json`
26
+ (0.23.2).
27
+
28
+ | You want to run | Extras |
29
+ |---|---|
30
+ | the Python API | none |
31
+ | the HTTP server (`tt_meteor.server.app`, see `SERVING.md`) | `server` |
32
+ | host tests (no device; the device tests are skipped): `TT_VISIBLE_DEVICES=none python -m pytest -q code/tt_meteor/tests` | `server,test` |
33
+ | device tests: `python -m pytest -q -s code/tt_meteor/tests/test_pcc_device.py code/tt_meteor/tests/test_e2e_device.py code/tt_meteor/tests/test_variants_device.py` (most of them need the fp32 goldens of the development workspace and skip without them) | `test` |
34
+
35
+ ## Quickstart
36
+
37
+ ```python
38
+ from tt_meteor import METEOR, load_sample
39
+
40
+ with METEOR.from_pretrained(device_id=0) as model:
41
+ out = model(**load_sample("code/tt_meteor/samples/synthetic_8cam.json")) # 8 cameras + calibration + ego speed + stream
42
+ print(out.to_dict()["plan"]["mode"], [d["label"] for d in out.to_dicts()])
43
+ ```
44
+
45
+ `examples/quickstart.py` runs the same snippet and writes the `/predict` JSON and a bird's-eye view
46
+ (`quickstart_bev.png`). The shipped sample is a synthetic test frame generated by this repository (see
47
+ `code/tt_meteor/samples/README.md`); feed your own rig's eight images and calibration for real use.
48
+
49
+ ## `METEOR.from_pretrained(...)`
50
+
51
+ ```python
52
+ METEOR.from_pretrained(
53
+ model_id=None, # HF repo or a local directory with the weights files; default AutowareFoundation/meteor
54
+ *,
55
+ revision=None, # default for the default repo: the validated commit 01a5f6d71df (tag v1.0)
56
+ variant=None, # load-time variant: "default" (the only one); default $METEOR_VARIANT
57
+ device_id=None, # chip to open; default $TT_DEVICE_ID or 0
58
+ device=None, # an already-opened ttnn device (tt_meteor.device.open_device); close() does not close it
59
+ dispatch=None, # "eth" (p150 target, 12x10 grid) | "worker" (A/B only, 11x10) | "auto"; default $METEOR_DISPATCH or "eth"
60
+ num_command_queues=None, # default $METEOR_NUM_CQS or 1 (D14: a second queue gains nothing for synchronous calls)
61
+ weights_dir=None, # explicit local weights directory; no Hub access
62
+ warmup_variants="default", # trace variants to capture now; see "Warm-up"
63
+ verbose=False,
64
+ **compile_params, # load-time knobs (below), e.g. input_norm="imagenet"
65
+ ) -> METEOR
66
+ ```
67
+
68
+ What it does: resolves the weights first, so a Hub problem never claims the chip (`weights_dir` >
69
+ `$METEOR_WEIGHTS_DIR` > a local `model_id` directory > the HF snapshot at the pinned revision, restricted to
70
+ `meteor_v157c3Z.onnx, meteor_v157.param.yaml, LICENSE, SHA256SUMS`, with an offline fallback to the cache), opens the
71
+ chip (ETH dispatch, 12×10, 1 CQ; the other open parameters are `DEVICE_DEFAULTS` in `tt_meteor/device.py`:
72
+ `l1_small_size` 32 KiB, `trace_region_size` 256 MiB, overridable with `METEOR_*`), reads the parameters of the ONNX
73
+ file by consuming node, builds the graph, then compiles and captures the metal trace. If ETH dispatch cannot open
74
+ (tt-metal without the patch), it warns and falls back to WORKER dispatch.
75
+
76
+ Load-time knobs (`compile_params` in lower case, or the environment variable `METEOR_<NAME>`; read once at load):
77
+
78
+ | knob | default | meaning |
79
+ |---|---|---|
80
+ | `input_norm` | `onnx` | `onnx` = the released graph's `/255` (ONNX Runtime parity; decision D12); `imagenet` = the trained ImageNet mean / std (upstream export defect, research/meteor SPEC section 10 risk 1); with `imagenet` an absent camera is zeroed after normalising, as in training |
81
+ | `depth_mean_bins` | `log` | `depth_mean` bin centres: log-spaced as exported (parity) or `linear` 1 + 1.25 b m (the trained bins) |
82
+ | `max_streams` | 16 | host temporal states kept (seg fusion, yaw tracks, mode hysteresis); the least recently used is dropped |
83
+ | `image_precision` | `terms3` | image branch with fp32 activations, every conv as three bf16 terms (needed by the depth / seg2d argmax gates); `bf16` is an A/B only: 166 ms faster per frame, but below the depth argmax gate (0.9797 < 0.99) |
84
+
85
+ ## Warm-up
86
+
87
+ The load runs one eager frame of the whole graph (it compiles every kernel and fills the program cache) and then
88
+ captures it as ONE metal trace, `frame` (4,678 programs). The first call is therefore as fast as the later ones.
89
+ With a warm JIT cache the load takes about 45-50 s (49.7 s measured: weights 0.3 s, graph build 2.6 s, eager frame
90
+ 41.0 s, capture 1.2 s). A cold kernel cache compiles for minutes (a new dispatch configuration on the build host:
91
+ ETH-2CQ 142 s, WORKER 11×10 768 s). `warmup_variants="none"` skips the capture (the first call then captures).
92
+
93
+ ## Call: `model(...)`
94
+
95
+ | Argument | Type | Description |
96
+ |---|---|---|
97
+ | `images` | list or mapping | the cameras `CAM_FRONT_WIDE`, `CAM_FRONT_LEFT`, `CAM_FRONT_RIGHT`, `CAM_BACK_WIDE`, `CAM_BACK_LEFT`, `CAM_BACK_RIGHT`, `CAM_FRONT_NARROW`, `CAM_BACK_NARROW`, any order: `CameraImage`s, dicts `{"camera", "image", "intrinsics", "T_ref_from_camera"}` or `{name: image}`. Raw, unrectified frames of at least 768×432; larger frames are resized with OpenCV INTER_AREA semantics (bit-exact numpy port) and K is scaled per axis. The six wide / corner cameras are required; a missing or all-zero narrow camera is **absent**: a zero image plus its donor's K and pose (`CAM_FRONT_WIDE` / `CAM_BACK_WIDE`), the trained 7-camera configuration |
98
+ | `calibration` | dict | `{"cameras": {name: {"intrinsics", "T_ref_from_camera"}}}` (K of the image as sent; camera optical frame -> base_link: x forward, y left, z up, origin on the road) or `{"preset": name}` (`code/tt_meteor/calib/`); inline calibration of a camera wins |
99
+ | `ego_speed` | float | m/s, the graph's `v0` (required) |
100
+ | `stream` | dict | optional: `{"id", "reset", "timestamp_s", "T_world_from_ego"}` (or `"pose": [x, y, yaw]`): METEOR's host temporal post-processing per stream id (BEV seg fusion needs the pose; yaw smoothing; plan-mode hysteresis). A new id or `reset=True` starts fresh |
101
+ | runtime params | keyword | host post-processing only (`METEOR.RUNTIME_PARAMS`, METEOR's C++ renderer defaults): 3D `det3d_threshold` 0.15, `det3d_topk` 64, `vehicle_threshold` 0.35, `vru_threshold` 0.15, `bev_nms_iou` 0.3, `bev_nms_containment` 0.6, `stationary_logit_threshold` 0.0; 2D `det2d_threshold` 0.30, `det2d_topk` 48, `det2d_hide` "7" (road paint); `unk2d` True, `unk2d_threshold` (= det2d), `ground_z` 0.0; plan `mode_hysteresis` 0.35, `straight_margin` 1.0; `seg_fuse` True, `thin_road_edge` True, `yaw_smoothing` True; `heads` False (adds the dense heads to `to_dict("npz")`) |
102
+
103
+ `tt_meteor.load_sample(path)` turns a sample manifest (`samples/<name>.json`: image paths relative to the file, a
104
+ preset, the ego speed and a stream) into these keyword arguments.
105
+
106
+ ### Input types
107
+
108
+ - Images: a path, PNG / JPEG bytes, a `PIL.Image`, a uint8 H×W×3 RGB array, or a float array in [0, 1].
109
+ - Transforms: 4×4 (or 3×4) matrices, `{"translation", "rotation_wxyz"}` (or `rotation_xyzw`), or Autoware
110
+ `{"x", "y", "z", "roll", "pitch", "yaw"}`. Intrinsics: 3×3 K, 3×4 P or `{"fx", "fy", "cx", "cy"}`.
111
+
112
+ ## Output
113
+
114
+ `model(...)` returns a `MeteorOutput` (`tt_meteor.Output`); `out.to_dict()` is the `/predict` JSON:
115
+
116
+ | key | content |
117
+ |---|---|
118
+ | `detections` (`out.to_dicts()`) | 3D BEV boxes sorted by score: `label` (VEHICLE / VRU), `label_id`, `score`, `center` [x, y] and `size` [length, width] in metres (base_link; METEOR predicts no z, height or velocity), `yaw` (rad, CCW from +x), `stationary`, `future` (6 × [x, y] at 0.5 .. 3 s, the agent's best mode), `future_mode` |
119
+ | `trajectory`, `columns` | the selected ego path, 6 × [x, y] at 0.5 s steps |
120
+ | `plan` | `mode`, `mode_probs` (softmax of the hysteresis-adjusted logits), `mode_logits`, `paths` (3 × 6 × [x, y]), `steer` (rad), `accel` (m/s²), `brake_prob`, `dt` |
121
+ | `detections_2d` | per camera: `label` (10 classes), `label_id`, `score`, `box_xyxy` (pixels at 768×432) |
122
+ | `unknown_obstacles` | [x, y] of 2D "obstacle" boxes placed on the ground plane (METEOR's `unk2d`) |
123
+ | `traffic_light` | `state` (none / green / yellow / red), `state_id`, `probs` |
124
+ | `lane`, `lane_classes` | the BEV lane map, uint8 [800, 500] as PNG (0.2 m cells, row 0 = +80 m ahead, column 0 = +50 m left; classes bg, road, sidewalk, crosswalk, laneline, stopline, road_edge, marking, parking), after the optional seg fusion and road-edge thinning |
125
+ | `stationary_head_healthy`, `meta` | METEOR's health check of the stationary head; `present` cameras, `v0`, `stream_id` |
126
+ | `heads` (`to_dict("npz")` with `heads=True`) | the dense outputs: `seg2d`, `depth` (bins), `depth_mean`, `occupancy` (class per voxel), `risk` (sigmoid), `stationary` |
127
+ | `timing_ms` | `preprocess`, `device` (tables + upload + replay + read + conversion), `postprocess`, `total` |
128
+
129
+ `out.path` is the selected path as an array; `out.lane` the lane map; `out.boxes3d` / `out.boxes2d` the box objects.
130
+
131
+ ## Lifetime and information
132
+
133
+ - `model.close()` releases the trace and the device tensors and closes the chip if the model opened it; idempotent.
134
+ `with` calls it for you; an unclosed model is closed when Python exits.
135
+ - `model.info`: weights (repo, tag, revision, path), device (dispatch, grid), variant, warm variants, warm-up times,
136
+ runtime parameter defaults, the load-time knobs and the trace description.
137
+ - Calls from several threads are safe: the device calls are serialised. One model per process per chip.
138
+ - **Rig changes.** The lift tables (205 MB of fp32 bin weights + the sampling grid) are built on the host per
139
+ calibration (cached per rig) and written to the chip only when the calibration changes: 182 ms for the write, so a
140
+ call with a new rig takes about 1.13 s instead of about 0.9 s. A fixed rig pays it once.
141
+
142
+ ## Speed
143
+
144
+ Warm, batch 1, ETH dispatch, 1 CQ, 12×10, AICLK 1350 MHz, p50 of 60 calls on a shared 8-core host
145
+ ([`OPT_BASELINE.md`](OPT_BASELINE.md); the device rows do not depend on the input):
146
+
147
+ | stage | ms |
148
+ |---|---:|
149
+ | device trace, one replay (the whole network) | 556.85 (back to back 556.75: 1.80 frames/s) |
150
+ | upload of the eight cameras (8 MB uint8) | 66.5 |
151
+ | readback (75 MB, 5 segments) + conversion to the 19 outputs | 17-48 + 65 |
152
+ | host pre-processing / camera transpose | 9-11 / 22-26 |
153
+ | host post-processing (METEOR's C++ decode rules, temporal state) | 124-131 |
154
+ | `model(...)` end to end | 876-918 (1.09-1.14 calls/s) |
155
+
156
+ This is the first release: optimization has not started ([`OPT_REPORT.md`](OPT_REPORT.md)).
157
+
158
+ ## Limits
159
+
160
+ - Batch 1 on the chip; one model per process; eight 768×432 camera slots in METEOR's order, a 400×250 lift grid at
161
+ 0.4 m and an 800×500 BEV at 0.2 m (fixed shapes of the released graph); at most 3 cameras per lift cell on the
162
+ rigs validated.
163
+ - The released graph is camera-only and single-frame (its temporal memory was baked out upstream); the temporal
164
+ behaviour of METEOR's runtime is host post-processing.
165
+ - Numerics: bf16 / fp32 on the chip; outputs differ slightly from the fp32 reference (README "Demo & Performances").
README.md ADDED
@@ -0,0 +1,199 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ tags:
3
+ - blackhole
4
+ - p150
5
+ - tt-dit-server
6
+ - tt-model-cache
7
+ - tt-model-container
8
+ - tenstorrent
9
+ - ttnn
10
+ - tt-metal
11
+ - tt-nn
12
+ - autoware
13
+ - autonomous-driving
14
+ - camera
15
+ - multi-view
16
+ - bird-eye-view
17
+ - multi-task
18
+ - e2e
19
+ - planning
20
+ - 3d-object-detection
21
+ - semantic-segmentation
22
+ - depth-estimation
23
+ - meteor
24
+ - tt-model-catalog
25
+ pipeline_tag: robotics
26
+ license: apache-2.0
27
+ license_link: https://huggingface.co/AutowareFoundation/meteor
28
+ base_model:
29
+ - AutowareFoundation/meteor
30
+ ---
31
+
32
+ # meteor-p150
33
+
34
+ METEOR, TIER IV's surround-view multi-task driving network released by the Autoware Foundation (`meteor_v157c3Z.onnx`, `DepthSegIPMNetV52`: a ResNet-34 + FPN encoder over eight cameras, image heads, a depth-gated IPM lift into one 96-channel 800×500 bird's-eye-view map at 0.2 m, BEV heads, an ego planner and three residual refiners; 48.25 M parameters, 2.58 TFLOP per frame), ported to one Tenstorrent Blackhole p150 with tt-nn. Eight camera images (768×432 RGB, raw and unrectified) with their intrinsics and extrinsics and the ego speed in; a BEV lane map, 3D boxes (vehicle / VRU) with stationary flags and agent futures, per-camera 2D boxes, 2D segmentation and depth, 3D occupancy, the ego-relevant traffic light, a risk field and three ego paths with steer / accel / brake out.
35
+ Weights: [AutowareFoundation/meteor `v1.0`](https://huggingface.co/AutowareFoundation/meteor/tree/01a5f6d71df5ecbbb5853ec600825481d57b9c6b) · Paper: none (GTC 2026 talk S81897; [tier4/METEOR README](https://github.com/tier4/METEOR/blob/dc193a81e9959de57b751e269127ae9b67e3f300/README.md)) · Autoware package: none (METEOR's own runtimes, [tier4/METEOR `deploy/cpp`](https://github.com/tier4/METEOR/tree/dc193a81e9959de57b751e269127ae9b67e3f300/deploy/cpp)) · Training code: [tier4/METEOR](https://github.com/tier4/METEOR/tree/dc193a81e9959de57b751e269127ae9b67e3f300) (`bevlane/train.py`) · Port: [`code/`](https://huggingface.co/changh95/meteor-p150/tree/main/code)
36
+
37
+ Runs on **p150** (mesh `P150`). Configuration: dispatch on the ETH cores, 1 command queue, 12×10 compute grid. The whole network runs on the chip as ONE metal trace per frame (`frame`, 4,678 programs: the image encoder and heads of the eight cameras, the lift, the BEV trunk and heads, the planner, the refiners and the output packing); the image resize, the lift geometry tables (once per calibration), METEOR's C++ decode rules and the temporal state run on the host. All numbers on this card were measured in this configuration.
38
+
39
+ Packaged and published with [tt-model-manager](https://github.com/tenstorrent/tt-model-manager) 0.1.0 (manifest schema 5.1).
40
+
41
+ ## Quickstart (Python)
42
+
43
+ Prerequisite: a tt-metal / ttnn environment at tt-metal [`44d66500520`](https://github.com/tenstorrent/tt-metal/commit/44d66500520fda9f2c7060c0f6b41ec48f7ab37e) with [`patches/tt-metal-eth-dispatch.patch`](patches/tt-metal-eth-dispatch.patch) and [`patches/tt-metal-reshape-rm-sys1419.patch`](patches/tt-metal-reshape-rm-sys1419.patch) applied (the second one fixes an intermittent hang of row-major reshapes under ETH dispatch on Blackhole, see Caveats). ttnn is not on PyPI.
44
+
45
+ ```bash
46
+ hf download changh95/meteor-p150 --exclude "image/*" --local-dir meteor-p150 && cd meteor-p150
47
+ pip install -e . # adds numpy<2, pillow, pyyaml, onnx, huggingface_hub, safetensors, opencv-python-headless; ttnn and torch come from tt-metal
48
+ pip install -e ".[server,test]" # optional: the HTTP server and the tests
49
+ ```
50
+
51
+ Run the snippet from the model repo root (the sample path is relative to it).
52
+
53
+ ```python
54
+ from tt_meteor import METEOR, load_sample
55
+
56
+ with METEOR.from_pretrained(device_id=0) as model: # weights -> your HF cache, trace captured
57
+ out = model(**load_sample("code/tt_meteor/samples/synthetic_8cam.json")) # 8 cameras + calibration + ego speed + stream
58
+
59
+ for d in out.to_dicts():
60
+ print(d["label"], d["score"], d["center"], d["size"], d["yaw"], d["stationary"])
61
+ body = out.to_dict()
62
+ print(body["plan"]["mode"], body["trajectory"], body["traffic_light"]["state"])
63
+ ```
64
+
65
+ - `from_pretrained` downloads [`AutowareFoundation/meteor`](https://huggingface.co/AutowareFoundation/meteor) at the pinned commit `01a5f6d71df` (tag `v1.0`; `meteor_v157c3Z.onnx`, 335 MB, plus `meteor_v157.param.yaml`, `LICENSE`, `SHA256SUMS`) to your HF cache, opens the chip, builds the graph and captures the metal trace. The first load on a machine compiles the kernels (minutes: the container's first boot on an empty JIT cache was ready after 799 s); later loads take about 45-50 s (49.7 s measured with a warm kernel cache: most of it is the eager warm-up frame of 4,678 programs).
66
+ - The first call is as fast as the later calls: the load runs the whole graph eagerly before it captures the trace.
67
+ - The shipped sample is a synthetic test frame (a rendered street seen by a generic eight-camera rig; see [`code/tt_meteor/samples/README.md`](code/tt_meteor/samples/README.md)); feed your own rig's eight images, calibration and speed for real use.
68
+ - The `with` block releases the trace and closes the chip. Without `with`, call `model.close()`.
69
+
70
+ | | |
71
+ |---|---|
72
+ | **Input** | `images`: the eight cameras `CAM_FRONT_WIDE`, `CAM_FRONT_LEFT`, `CAM_FRONT_RIGHT`, `CAM_BACK_WIDE`, `CAM_BACK_LEFT`, `CAM_BACK_RIGHT`, `CAM_FRONT_NARROW`, `CAM_BACK_NARROW`, any order (path, bytes, PIL, uint8 array; raw, unrectified, at least 768×432, resized with OpenCV INTER_AREA semantics). An absent narrow camera (omitted, or an all-zero image) gets a zero input and its donor's pose, as trained. `calibration`: per camera `intrinsics` (of the image as sent) and `T_ref_from_camera` (camera optical frame -> base_link), or a preset. `ego_speed` (m/s). Optional `stream`: `id`, `reset`, `T_world_from_ego` for METEOR's host temporal state (BEV seg fusion, yaw smoothing, plan-mode hysteresis). |
73
+ | **Options** | Host post-processing (METEOR's C++ renderer defaults): `det3d_threshold=0.15`, `det3d_topk=64`, `vehicle_threshold=0.35`, `vru_threshold=0.15`, `bev_nms_iou=0.3`, `bev_nms_containment=0.6`, `det2d_threshold=0.30`, `det2d_topk=48`, `det2d_hide="7"`, `unk2d=True`, `mode_hysteresis=0.35`, `straight_margin=1.0`, `seg_fuse=True`, `thin_road_edge=True`, `yaw_smoothing=True`, `heads=False`. Load time: `from_pretrained(device_id=0, dispatch="eth", weights_dir=None, device=None, input_norm="onnx", depth_mean_bins="log", max_streams=16)`. |
74
+ | **Output** | `MeteorOutput`: 3D BEV boxes in base_link (`center` [x, y], `size` [length, width], `yaw`, `score`, `label` VEHICLE / VRU, `stationary`, `future`: 6 × [x, y] at 0.5-3 s), the selected ego path (6 × [x, y]) and the plan (`mode`, `mode_probs`, 3 `paths`, `steer`, `accel`, `brake_prob`), 2D boxes per camera (10 classes), unknown obstacles, the traffic-light state, the BEV lane map (uint8 800×500, 9 classes), `timing_ms`. |
75
+ | **Methods** | `out.to_dict()` gives the `/predict` JSON (`out.to_dict("npz")` with `heads=True` adds seg2d, depth, depth_mean, occupancy, risk, stationary); `out.to_dicts()` one dict per 3D box; `out.path`, `out.lane`. |
76
+
77
+ - The API gives the same output as the HTTP server `/predict`: both share the decoders, the device trace and the host post-processing. One model uses one chip; calls from several threads are serialised.
78
+ - Full reference: [`code/PYTHON.md`](code/PYTHON.md). Runnable example: [`examples/quickstart.py`](examples/quickstart.py) (writes the `/predict` JSON and a bird's-eye view, `quickstart_bev.png`).
79
+
80
+ ## Serving (HTTP)
81
+
82
+ ```bash
83
+ tt-model pull changh95/meteor-p150 --with-weights
84
+ tt-model serve changh95/meteor-p150 # or with tt-cli: tt serve changh95/meteor-p150
85
+ python3 code/tt_meteor/server/client.py --sample code/tt_meteor/samples/synthetic_8cam.json --out req.json
86
+ curl -s localhost:20000/predict -H 'Content-Type: application/json' -d @req.json
87
+ tt model stop changh95/meteor-p150
88
+ ```
89
+
90
+ - The image does not contain the weights. `--with-weights` puts them in your HF cache.
91
+ - The server uses port 20000 (or the next free port). It is ready when the log shows `Application startup complete`.
92
+ - `POST /predict`: `images` (all eight cameras, base64 JPEG / PNG / `.npy`, rgb8; an absent narrow camera as an all-zero image), `calibration` (per camera `intrinsics` and `T_ref_from_camera`, or `{"preset": name}`), `ego_speed` (m/s, required); optional `stream` (`id`, `reset`, `T_world_from_ego`), `params`, `output_format`. For your own frame: `client.py --image CAM_FRONT_WIDE=f.jpg ... --calib rig.json --ego-speed 8.3`. Also `GET /health`, `GET /info`, `GET /v1/models` (stub). Contract: [`SERVING.md`](SERVING.md) section 3.
93
+
94
+ The shipped sample (a request after the warm-up, host server, 2026-10-10; trimmed):
95
+
96
+ ```json
97
+ {"model": "meteor-p150", "frame_id": "base_link", "num_detections": 8,
98
+ "detections": [
99
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9932, "center": [-12.86, 0.095], "size": [4.399, 1.742], "yaw": -0.0687, "stationary": false, "future_mode": 0, "future": [[-11.92, -0.09], [-10.74, -0.15], ...]},
100
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9872, "center": [-7.176, -3.215], "size": [3.965, 1.731], "yaw": 0.0183, "stationary": false, ...},
101
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9624, "center": [10.84, 0.019], "size": [4.334, 1.761], "yaw": -0.0189, "stationary": false, ...},
102
+ ...],
103
+ "trajectory": [[0.74, 0.91], [0.85, 1.54], [0.41, 2.34], [0.31, 3.08], [0.62, 3.82], [0.3, 4.45]], "columns": ["x", "y"],
104
+ "plan": {"mode": 0, "mode_probs": [0.233, 0.296, 0.471], "mode_logits": [...], "paths": [...], "steer": 0.0026, "accel": 0.021, "brake_prob": 0.131, "dt": 0.5},
105
+ "detections_2d": {"CAM_FRONT_WIDE": [{"label": "car", "label_id": 1, "score": 0.5239, "box_xyxy": [412.76, 211.02, 438.5, 230.29]}, ...], ...},
106
+ "unknown_obstacles": [],
107
+ "traffic_light": {"state": "red", "state_id": 3, "probs": [0.363, 0.140, 0.031, 0.465]},
108
+ "lane": {"format": "png", "key": "lane", "dtype": "uint8", "shape": [800, 500], "data": "iVBORw0KGgo..."},
109
+ "lane_classes": ["bg", "road", "sidewalk", "crosswalk", "laneline", "stopline", "road_edge", "marking", "parking"],
110
+ "stationary_head_healthy": true,
111
+ "meta": {"present": [true, true, true, true, true, true, true, true], "v0": 8.0, "stream_id": "synthetic_8cam"},
112
+ "timing_ms": {"decode": 83.95, "preprocess": 10.69, "device": 722.42, "postprocess": 86.08, "model_call": 820.65, "total": 908.67}}
113
+ ```
114
+
115
+ - Boxes, paths and the lane map are in base_link metres (x forward, y left, origin on the road below the ego reference point); METEOR predicts BEV boxes without z, height or velocity. `yaw` is counter-clockwise from +x; boxes are sorted by score.
116
+ - `trajectory` is the selected ego path (METEOR's renderer: straight preference and mode hysteresis per stream); `plan.paths` holds all three modes. The synthetic sample is a test pattern (a red light, a car 11 m ahead): its plan barely moves forward (under 1 m in 3 s, drifting 4.5 m to the left) and is not meaningful driving; the stored CPU reference has the same path.
117
+ - `lane` is a PNG of the 800×500 class map (row 0 = 80 m ahead, column 0 = 50 m to the left); `lane_classes` names its values.
118
+ - `meta` carries the present cameras, `v0` and the stream id; `/info` reports the device (ETH, 12×10), the camera order, the calibration presets and the load-time knobs.
119
+
120
+ ## Demo
121
+
122
+ | PandaSet 019, frame 40 (San Francisco; CC BY 4.0): TT output | nuScenes scene-0103, key-frame 9 (Boston; CC BY-NC-SA 4.0, non-commercial): TT output |
123
+ |:---:|:---:|
124
+ | ![PandaSet 019 frame 40: eight camera slots with 2D boxes, the bird's-eye view with lanes, risk, 3D boxes and the plan, the front view with 3D boxes and the planned path](media/meteor_ps019_f0020_card_tt.jpg) | ![nuScenes scene-0103 key-frame 9, non-commercial](media/meteor_ns0103_f0009_NC_card_tt.jpg) |
125
+
126
+ | TT vs the fp32 CPU reference: PandaSet 019 frame 40, bird's-eye view | The shipped synthetic sample: TT vs the stored CPU reference |
127
+ |:---:|:---:|
128
+ | ![TT vs CPU, bird's-eye view](media/meteor_ps019_f0020_bev_tt_vs_cpu.png) | ![The synthetic sample](media/meteor_synthetic_8cam_tt_vs_cpu.png) |
129
+
130
+ More: [PandaSet 019 frames 0-48 as a GIF](media/meteor_ps019_seq_tt.gif) (every 4th frame, real-time playback), [PandaSet 090 frame 40](media/meteor_ps090_f0020_card_tt.jpg), and the camera heads (2D segmentation overlay and depth per slot) of [PandaSet 019](media/meteor_ps019_f0020_heads_tt.jpg), [PandaSet 090](media/meteor_ps090_f0020_heads_tt.jpg) and [nuScenes scene-0103](media/meteor_ns0103_f0009_NC_heads_tt.jpg) (non-commercial). Every image shows outputs of this port on the p150, some next to the fp32 CPU reference; the BEV panels compare the plan with the logged ego path (white circles) and the 3D boxes with the dataset annotations (light blue). The public rigs map onto METEOR's eight slots as 6 real cameras, a virtual FRONT_NARROW centre crop of the front camera and an absent BACK_NARROW (zero image + the BACK_WIDE pose, as trained). Head and licence-plate regions of annotated people and vehicles are blurred. Sources, licences and changes: [`media/ATTRIBUTION.md`](media/ATTRIBUTION.md). The weights were trained on TIER IV's own rig in Japan only, so both datasets are out of domain: the detections show the domain gap, not the port.
131
+
132
+ ## Demo & Performances
133
+
134
+ Warm, batch 1, eight cameras. Accuracy: the TT output against the port's fp32 CPU reference of the same network (same weights, same pre- and post-processing: METEOR's C++ runtime rules), which matches ONNX Runtime on `meteor_v157c3Z.onnx` (min PCC 0.99999990 over 155 taps and outputs). The gates are the PLAN thresholds, frozen at the first green run and never loosened. Speed: `code/scripts/bench.py`, p50 (p99) of 60 iterations, on a shared 8-core host ([`OPT_BASELINE.md`](OPT_BASELINE.md)).
135
+
136
+ | Metric | Performance |
137
+ |---|---:|
138
+ | Per-stage PCC vs the CPU reference, teacher-forced (26 gates, PandaSet 019 frame 40) | every tap **≥ 0.999989** (gate 0.99); lane / seg2d / depth argmax agreement **0.9984 / 0.9965 / 0.9936** (gate 0.99) |
139
+ | The served graph from the cameras, 4 frames of 3 rigs (PandaSet 019 / 090, nuScenes scene-0103, an in-domain METEOR demo frame) | min output PCC **0.99982**; worst lane / seg2d / depth agreement **0.9963 / 0.9965 / 0.9925**; selected ego path within **0.170 m** (gate 0.3 m); 3D boxes strict recall / precision **1.0 / 1.0** on three frames, **0.971 / 0.971** (34 / 35) on the METEOR frame (gate 0.95) |
140
+ | 38 public frames (PandaSet 019 and 090: 13 frames each, nuScenes scene-0103: 12 key-frames) through the served graph | argmax agreement lane ≥ **0.9971**, seg2d ≥ **0.9962**, depth ≥ **0.9925**; dense-head PCC ≥ **0.9999** (hm, reg, stationary; risk ≥ 0.9996); all 19 outputs on one frame per sequence: PCC ≥ **0.9990** (`depth_mean`; the rest ≥ 0.99989); selected ego path **0.04-0.11 m** from the CPU path (mean distance), the same plan mode and traffic-light state on all 38 frames; 3D boxes: every TT box matches a CPU box (precision 1.0), **408 of 418** CPU boxes matched (0.976): 9 of the 10 misses are CPU boxes within 0.004 of the class threshold (0.3507-0.3538 vs 0.35, 0.1505-0.1513 vs 0.15), the 10th a pedestrian whose TT peak lies 2.1 m away |
141
+ | Planned path vs the logged ego path (ADE, frames with a full 3 s future; a sanity metric, METEOR has no paper metric) | PandaSet 019 **TT 0.598 m** / CPU 0.552 m; PandaSet 090 2.570 / 2.472 m; nuScenes scene-0103 1.846 / 1.885 m |
142
+ | The shipped synthetic sample vs its stored CPU reference (the container smoke's check) | **8 / 8** boxes (recall 1.000, precision 1.000), max \|score difference\| **0.0020**, lane map agreement **0.9987**, path ADE / FDE **0.018 / 0.030 m** |
143
+ | `input_norm=imagenet` + `depth_mean_bins=linear` (the load-time option, two absent cameras) vs its own CPU reference | every float output PCC ≥ **0.99988**; lane / seg2d / depth agreement 0.9980 / 0.9947 / 0.9934 |
144
+ | Device, one frame (`frame` replay, 4,678 programs) | **556.85 ms**; back to back **556.75 ms**: **1.80 frames/s** device-bound |
145
+ | Python `model()`, the PandaSet 019 frame (eight 768×432 JPEGs, decoded) | **918.2 ms** (1,071.8): preprocess 11.2, camera transpose 26.3, upload 66.5, device 556.9, readback 17.2 + conversion 65.3, postprocess 129.9 |
146
+ | Python `model()`, other inputs | synthetic noise 875.7 ms, PandaSet 090 902.8 ms, nuScenes 914.4 ms; the shipped sample **871.6 ms** (release re-check, 30 iterations) |
147
+ | A call with a new calibration (rig change) | lift-table write **182.3 ms**; call 1,128.5 ms |
148
+ | Served `/predict` on the host, shipped sample (uvicorn, loopback, after the first request) | `timing_ms.total` **904-909 ms** (decode of the 1.8 MB body 81-84, preprocess 10.6, device 720, postprocess 86-87); client round trip 0.92-0.93 s; boot to ready 44.6 s (warm kernel cache) |
149
+ | ETH-dispatch stress (`stress_frames.py --mode rigs`: a rig change, table write, upload, replay and bit-compared read every frame) | **2,000 frames** clean on the baseline code (4 public rigs); **200 frames** clean in this release (4 synthetic rigs, the golden-free mode), and 200 more inside the published container image (per-rig output hashes identical to the host run) |
150
+
151
+ All numbers in this table were measured with dispatch on the ETH cores, 1 command queue and a 12×10 compute grid on one p150, AICLK 1350 MHz. The device is kernel-bound (555.3 ms of kernels and 2.74 ms of op-to-op gaps for 4,678 programs); the image branch takes 268 ms of it (the three-term fp32 convs of the precision policy), and 57 % of the replay is layout, data movement and precision glue. The host rows move with the load of the shared host (a busier host added 18 ms to the PandaSet call in the release re-check). This is the first release: optimization has not started ([`OPT_REPORT.md`](OPT_REPORT.md) ranks the plan; [`OPT_BASELINE.md`](OPT_BASELINE.md) has every measurement). Verification: [`VERIFICATION_2026-10-10.md`](VERIFICATION_2026-10-10.md).
152
+
153
+ Box agreement: a CPU box counts as matched when a TT box of the same class lies within 1 m (0.5 m for the gated frames and the shipped sample), one to one. The public frames are out of domain for METEOR, so they test the port (the same network on new rigs, lift tables and scenes), not the model's accuracy. nuScenes dataset © Motional AD Inc., CC BY-NC-SA 4.0 and the nuScenes Terms of Use (https://www.nuscenes.org/terms-of-use); non-commercial: the raw data were used for local validation only and are not distributed here (only the `_NC` renders in `media/` are); Motional does not endorse this work. Cite: H. Caesar et al., *nuScenes: A Multimodal Dataset for Autonomous Driving*, CVPR 2020.
154
+
155
+ No GPU comparison: no GPU was available on the host where this port was built and measured, so this card makes no GPU speed claim. The reference rows are the port's own fp32 CPU reference on the same host (a correctness baseline, not a speed target). Upstream reports 67.4 ms median per eight-camera frame on a Jetson AGX Orin (TensorRT INT8 with the 2:4-sparse trunk, CUDA Graph, zero-copy input; tier4/METEOR README) and about 30 ms with TensorRT fp16 on a data-centre GPU (the upstream model card); both are other precisions and, on the Orin, a different lift (the optional plugin path uses a coarser 0.8 m lift grid with the calibration baked in), so they are not like-for-like. p150 power was not measured, so no efficiency comparison is made.
156
+
157
+ ## Caveats
158
+
159
+ - **Deployment status.** There is no Autoware package for METEOR: neither autoware_universe nor the Autoware launch / ansible artifacts contain it; the upstream card and README list an Autoware (ROS 2) node as future work. METEOR is a reference model with its own Python and C++ TensorRT runtimes ([tier4/METEOR](https://github.com/tier4/METEOR)), whose pre- and post-processing this port follows. It is not a ROS node and not a certified Autoware component: do not use it for safety-critical driving decisions (the upstream card: a research artefact that must not control a vehicle on public roads).
160
+ - **First release: optimization pending.** This is the functional baseline port; no optimization round has run yet ([`OPT_REPORT.md`](OPT_REPORT.md) ranks the plan: TILE reshape copies, fused three-term convs, the input path, the lift as one gather kernel).
161
+ - **ETH-dispatch hang, fixed.** During the port the model hung intermittently under ETH dispatch (about 1 in 300 frames). The cause was one row-major `ttnn.reshape` writing 192-384 B DRAM pages from both NOCs (the Blackhole DRAM-arbiter issue SYS-1419), reproduced standalone in plain ttnn. `patches/tt-metal-reshape-rm-sys1419.patch` (built into the image) routes small destination pages to a single-kernel path; with it, 20,000 replays of the repro and two 2,000-frame rig-switch stress runs of the whole model were clean. Without the patch, ttaw 0.23.2 switches that resize tail to a TILE reshape (also clean, 38 ms slower per frame). Both patches are needed for the published configuration.
162
+ - **Precision** (the first release's policy: fp32 weights and biases, HiFi4, fp32 dest and packer L1 accumulation). The image branch and the 3D detection path run with fp32 activations, every conv as three bf16 terms (bf16 activations fail the depth argmax gate: 0.9797 < 0.99); the BEV trunk runs bf16 activations; the lift sums, the planner and every value that feeds a threshold, a softmax or an argmax are fp32. Outputs differ slightly from the fp32 reference (agreement figures above); objects and plan modes near a threshold can flip. The three-term image branch costs 166 ms of the 557 ms frame.
163
+ - **Documented behaviour and deviations** (PORT_LOG.md section 2, [`SERVING.md`](SERVING.md)):
164
+ - The input normalisation is the released graph's `/255` (ONNX Runtime parity; decision D12), although METEOR was trained with ImageNet mean / std (an upstream export defect): `input_norm="imagenet"` (`METEOR_INPUT_NORM`) gives the trained normalisation, validated on the chip against its own CPU reference.
165
+ - `depth_mean` keeps the exported log-spaced bin centres (parity); the trained bins are linear, 1 + 1.25 b m: `depth_mean_bins="linear"`.
166
+ - The temporal state (seg fusion, yaw tracks, plan hysteresis) is kept per `stream.id` on the host and resets on a new id or `reset`; upstream resets only the seg accumulator on a pose jump.
167
+ - The released graph is single-frame and camera-only (its temporal memory and LiDAR input were baked out upstream).
168
+ - **Domain gap.** The weights were trained on TIER IV's eight-camera DRS rig in Japan (98 deg wide cameras, corner cameras pitched about 25 deg down, vertically squashed images). Other rigs work (the network takes K and T as inputs) but lose accuracy: on PandaSet and nuScenes the depth head is the main gap (only the camera with matching intrinsics gets METEOR-grade depth), 3D boxes are accurate but sparse, and the planner follows steady driving but not unseen intent (fp32 CPU reference on PandaSet 019: vehicles recall / precision 0.46 / 0.91 within 20 m of the LiDAR-supported annotations, planned-path ADE 0.55 m against the logged path vs 0.96 m on in-domain METEOR demo frames; the front camera's depth reads 1.65× the LiDAR range; on nuScenes scene-0103 the BEV drivable-area IoU is 0.39 against 0.70-0.74 on PandaSet).
169
+ - **Fixed shapes.** Eight 768×432 camera slots in METEOR's order, a 400×250 lift grid at 0.4 m, an 800×500 BEV at 0.2 m (±80 m × ±50 m); batch 1, one frame per request; requests are serialised on the chip. A new calibration costs about 0.2 s once (the lift tables, cached per rig).
170
+ - **Validation scope.** Agreement with the fp32 CPU reference on public data (PandaSet, nuScenes), on an in-domain METEOR demo frame and on the synthetic sample; no dataset-level metric exists for METEOR (no paper, no public split). The shipped sample is synthetic; a PandaSet sample is prepared but not shipped yet.
171
+ - Does not scale to multiple p150 in a mesh configuration. The build uses a 12×10 compute grid of Tensix cores: the dispatch functions move from one Tensix column to the ETH cores (`patches/tt-metal-eth-dispatch.patch`), so this build assumes that you do not need chip-to-chip ethernet communication.
172
+ - `dispatch="worker"` (server: `METEOR_DISPATCH=worker`) is an A/B opt-in. On a p150 it gives an 11×10 grid (578.9 ms per frame with 2 CQs; the gates passed under WORKER during the port). If ETH dispatch is not available (tt-metal without the patch), the model falls back to it with a warning. The numbers on this card do not apply to that mode.
173
+ - Not an OpenAI-compatible API; `GET /v1/models` is a stub so the tt-model ready card does not 404.
174
+
175
+ ## Licensing
176
+
177
+ - Weights: [AutowareFoundation/meteor](https://huggingface.co/AutowareFoundation/meteor) at tag `v1.0` (commit `01a5f6d71df5ecbbb5853ec600825481d57b9c6b`), Apache-2.0 per its model card and `LICENSE`. Not redistributed here: the package only points to them. The upstream card: trained with `bevlane/train.py` on TIER IV Co-MLOps DRS recordings in Japan, every supervision signal auto-generated (no human labels; adverse conditions added as NVIDIA Cosmos Transfer re-renderings of real scenes); the recordings are not public, and accuracy figures are measured on an internal split and not published. The AutowareFoundation/meteor-demo-scenes frames are for research and demonstration use only and are never redistributed or rendered here.
178
+ - Pre- and post-processing ported from METEOR's own runtimes, [tier4/METEOR](https://github.com/tier4/METEOR/tree/dc193a81e9959de57b751e269127ae9b67e3f300) `deploy/cpp` and `hf/onnx_smoke_test.py` (Apache-2.0); there is no Autoware package.
179
+ - Port and serving code (`code/`): Apache-2.0.
180
+ - Sample data: `code/tt_meteor/samples/synthetic_8cam*` and the preset `calib/synthetic_8cam.json` were generated by this repository (`code/scripts/make_synthetic_sample.py`; no third-party data), Apache-2.0. No dataset sample is shipped; nuScenes and the METEOR demo scenes never ship, and a PandaSet sample (CC BY 4.0) is prepared but not yet included.
181
+ - Demo media (`media/`; per-file sources, frames and changes: [`media/ATTRIBUTION.md`](media/ATTRIBUTION.md)):
182
+ - PandaSet renders (`media/meteor_ps*`): Contains data from PandaSet (Scale AI and Hesai), https://pandaset.org, licensed under CC BY 4.0 and the PandaSet Dataset Terms. Changes: camera JPEGs resized to 768x432 (and a centre crop for the virtual narrow camera), re-encoded, calibration converted to the METEOR layout; downscaled, heads and licence plates of annotated people and vehicles blurred; model outputs drawn. Scale AI and Hesai do not endorse this work. Cite: P. Xiao et al., *PandaSet: Advanced Sensor Suite Dataset for Autonomous Driving*, ITSC 2021.
183
+ - nuScenes renders (`media/*_NC.*`), **non-commercial, CC BY-NC-SA 4.0**: Rendered from the nuScenes dataset, © Motional AD Inc., CC BY-NC-SA 4.0 and the nuScenes Terms of Use (https://www.nuscenes.org/terms-of-use). Non-commercial use only; adaptations under the same license. Motional does not endorse this work. Cite: H. Caesar et al., *nuScenes: A Multimodal Dataset for Autonomous Driving*, CVPR 2020.
184
+ - The synthetic-sample render (`media/meteor_synthetic_8cam_tt_vs_cpu.png`): generated by this repository, Apache-2.0.
185
+
186
+ ## Provenance
187
+
188
+ These are the exact sources the container image was built from:
189
+
190
+ | component | built from |
191
+ | --- | --- |
192
+ | tt-metal | [`44d66500520fda9f2c7060c0f6b41ec48f7ab37e`](https://github.com/tenstorrent/tt-metal/commit/44d66500520fda9f2c7060c0f6b41ec48f7ab37e) + [`patches/tt-metal-eth-dispatch.patch`](patches/tt-metal-eth-dispatch.patch) + [`patches/tt-metal-reshape-rm-sys1419.patch`](patches/tt-metal-reshape-rm-sys1419.patch) (dirty tree: the image includes both patches; the image build asserts both markers) |
193
+ | weights | [`AutowareFoundation/meteor@01a5f6d71df5ecbbb5853ec600825481d57b9c6b`](https://huggingface.co/AutowareFoundation/meteor/tree/01a5f6d71df5ecbbb5853ec600825481d57b9c6b) (tag `v1.0`), files `meteor_v157c3Z.onnx` (sha256 `50397b4d…b1a7`), `meteor_v157.param.yaml`, `LICENSE`, `SHA256SUMS` |
194
+ | pre- / post-processing reference | no Autoware package; METEOR's runtimes, tier4/METEOR [`dc193a81e9959de57b751e269127ae9b67e3f300`](https://github.com/tier4/METEOR/tree/dc193a81e9959de57b751e269127ae9b67e3f300) (`deploy/cpp`, `hf/onnx_smoke_test.py`) |
195
+ | shared port code | `ttaw` 0.23.2 @ common `60b6dd7` (vendored in `code/tt_meteor/ttaw`, `VENDORED.json`) |
196
+ | `code/` digest (image) | `40de720599c6b5b9` (sha256, first 16 hex digits; `built.code_sha256` of `tt_kernel_manifest.json`) |
197
+ | image | `tt-model/meteor-p150:823dc1fbd4c7` (`sha256:823dc1fbd4c7ec836a6e74486495d2801a71d6fa4e18d41011fa873749962afc`) |
198
+ | base images | build stage `ghcr.io/tenstorrent/tt-metal/tt-metalium/ubuntu-22.04-dev-amd64:latest` @ `sha256:df9d279c7f85c17c6fad982d196802682d669cca1b7ced9cbaad8181339cd5fc`; runtime stage `docker.io/library/ubuntu:22.04` @ `sha256:5ec03bb3441e8b0bf3b4f9cd4629a1ae763010dc3035bb8da3ae6cf026486401` (tt-model's `FROM` tags float; these are the digests this build resolved, see [`build_info.json`](build_info.json)) |
199
+ | built | 2026-10-10T17:04:30+00:00 by tt-model 0.1.0 |
SERVING.md ADDED
@@ -0,0 +1,230 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Serving METEOR (TIER IV, AutowareFoundation/meteor) on Blackhole with tt-model-manager
2
+
3
+ This repo is the authoring source of the **tt-model container package** `changh95/meteor-p150`
4
+ (`kind: tt-dit-server`, schema 5.1): `tt-model.yaml` is the manifest, `code/` the port, and
5
+ `code/tt_meteor/server/app.py` the ASGI app uvicorn runs. Weights are a pinned pointer, never baked into the image.
6
+ The device open, the decoders, the Python-API contract and the HTTP contract come from `code/tt_meteor/ttaw/`, the
7
+ shared package of the Autoware ports vendored into this repo (`ttaw/VENDORED.json` records its version and file
8
+ hashes; it is not edited here).
9
+
10
+ | item | value |
11
+ |---|---|
12
+ | tt-metal tree | `/home/ubuntu/experiments/tt-models/tt-metal` (main `44d66500520`, `v0.80.0-dev20261006-78-g44d66500520`, + `patches/tt-metal-eth-dispatch.patch` + `patches/tt-metal-reshape-rm-sys1419.patch`; torch 2.11.0+cpu) |
13
+ | weights | `AutowareFoundation/meteor` @ `01a5f6d71df5ecbbb5853ec600825481d57b9c6b` (tag `v1.0`; files `meteor_v157c3Z.onnx, meteor_v157.param.yaml, LICENSE, SHA256SUMS`; Apache-2.0; public, ungated) |
14
+ | pre- / post-processing reference | **no Autoware package exists for METEOR**: METEOR's own runtimes define them, [tier4/METEOR](https://github.com/tier4/METEOR) @ `dc193a81e9959de57b751e269127ae9b67e3f300` (`deploy/cpp`: `realtime_main.cpp`, `decode.cpp`, `render.cpp`; `hf/onnx_smoke_test.py`) |
15
+ | app | `tt_meteor.server.app:app` (uvicorn; the ASGI lifespan does weights -> device -> graph -> trace capture) |
16
+ | device recipe | `ttnn.open_device(device_id, dispatch_core_config=DispatchCoreConfig(ETH), l1_small_size=32768, trace_region_size=256 MiB, num_command_queues=1)` (`DEVICE_DEFAULTS` in `code/tt_meteor/device.py`) |
17
+ | hardware | one Blackhole p150 (`hardware: p150`, `mesh_device: P150`, `TT_MESH_SHAPE=1x1`), 12×10 compute grid |
18
+
19
+ ## 1. Run on the HOST (hardware validation, no Docker)
20
+
21
+ The tt-metal `python_env` has ttnn and torch; put the HTTP stack in a side directory instead of installing into it.
22
+ On the shared workspace box every command that opens the chip goes through the device lock (`bin/devrun`).
23
+
24
+ ```bash
25
+ ROOT=/home/ubuntu/experiments/tt-models
26
+ source $ROOT/bin/tt-env.sh # TT_METAL_HOME, PYTHONPATH, tt-metal python_env
27
+ mkdir -p /tmp/meteor-http && uv pip install --python $TT_METAL_HOME/python_env/bin/python \
28
+ --target /tmp/meteor-http fastapi uvicorn onnx
29
+ cd $ROOT/bundles/meteor-p150
30
+ export PYTHONPATH=$PWD/code:$TT_METAL_HOME:$TT_METAL_HOME/ttnn:/tmp/meteor-http
31
+ export HF_MODEL=AutowareFoundation/meteor TT_MODEL_WEIGHTS_REVISION=01a5f6d71df5ecbbb5853ec600825481d57b9c6b TT_MESH_SHAPE=1x1
32
+ # (on the shared workspace box every METEOR device job also needs METEOR_DEVICE_OK=1 from the orchestrator)
33
+
34
+ # host tests (no device; the device tests are skipped; ttnn is the fake of common/tests/host: a real ttnn host
35
+ # tensor opens the chip, so only devrun jobs may build one)
36
+ PYTHONPATH=$ROOT/research/packaging/scripts:$PYTHONPATH TT_VISIBLE_DEVICES=none \
37
+ python -m pytest -q -p no:cacheprovider -p fake_ttnn_plugin code/tt_meteor/tests
38
+ # device tests and the server, under the lock
39
+ $ROOT/bin/devrun -t 2400 -- env TTAW_GATES_READONLY=1 python -m pytest -q -s code/tt_meteor/tests/test_pcc_device.py \
40
+ code/tt_meteor/tests/test_e2e_device.py code/tt_meteor/tests/test_variants_device.py
41
+ $ROOT/bin/devrun -t 1800 -- bash -c 'python -m uvicorn --host 127.0.0.1 --port 20000 --lifespan on tt_meteor.server.app:app & \
42
+ S=$!; python code/tt_meteor/server/smoke_test.py --url http://127.0.0.1:20000 --wait 1200; R=$?; kill -TERM $S; wait $S; exit $R'
43
+ ```
44
+
45
+ Boot log landmarks (they drive `tt-model serve`'s progress view, `boot_progress.py` TT_DIT_PHASES):
46
+ `Loading weights` -> `Opening device` -> `Warming up: capturing trace ...` -> `Warmup complete` -> uvicorn
47
+ `Application startup complete`. The warm-up runs one eager frame of the whole graph and captures it as one trace:
48
+ about 45-50 s with a warm JIT cache (most of it the eager frame of 4,678 programs); a cold JIT cache compiles kernels
49
+ for minutes (the first container boot on an empty JIT cache was ready after 799 s, 2026-10-10). Startup
50
+ failures raise and uvicorn exits non-zero (no CPU fallback). SIGTERM: the lifespan releases the traces and closes the
51
+ chip (within `tt-model stop`'s 120 s budget).
52
+
53
+ Offline override (no Hub): `METEOR_WEIGHTS_DIR=<dir holding meteor_v157c3Z.onnx, meteor_v157.param.yaml, LICENSE, SHA256SUMS>`.
54
+
55
+ ## 2. Package, serve, push (Docker)
56
+
57
+ Docker 29 on this box is **rootful** (`/var/lib/docker`), and the user's session may not carry the `docker` group:
58
+ wrap every docker-using command in `sg docker -c "..."` (no `docker-env.sh` is needed here; the reference bundles'
59
+ `docker-env.sh` was the authors' rootless-Docker box). Run every `tt-model` command **from this directory**
60
+ (`source.tt_metal` and `extra_code[].root: code` are CWD-relative) and point `--out` outside it.
61
+
62
+ ```bash
63
+ ROOT=/home/ubuntu/experiments/tt-models; cd $ROOT/bundles/meteor-p150
64
+ TTM=$HOME/.local/share/uv/tools/tt-model/bin/python
65
+
66
+ # offline validation (seconds; no docker, no device) -- must print VALID
67
+ $TTM -c "from tt_kernel.container_manifest import load_container_manifest as L; m=L('tt-model.yaml', check_sources=True); p=m.resolve_profile(); print('VALID', m.name, m.kind, p.hardware, p.mesh_device, m.weights_ref)"
68
+
69
+ # build (one at a time on the box: the absolute lock file): ~35 min cold, ~12 min after a code/ change (the C++
70
+ # build re-runs, ccache-warm, whenever code/ changes), ~1 min for a manifest- or lock-only change
71
+ flock $ROOT/.package.lock sg docker -c "tt-model package --container tt-model.yaml --out $ROOT/build" # log: ~/.cache/tt-model/build/meteor-p150.log
72
+ # provenance: the base-image digests this build resolved (tt-model's FROM tags float) -> build_info.json (published)
73
+ sg docker -c "python3 $ROOT/research/packaging/scripts/record_build_info.py --staged $ROOT/build/meteor-p150 --bundle ."
74
+
75
+ # serve + smoke + stop, all inside ONE device-lock window, once per serve profile (`tt-model profiles
76
+ # $ROOT/build/meteor-p150/tt_kernel_manifest.json` lists them; serve returns once READY and leaves the container
77
+ # running; the script's trap saves the container log and always runs `tt-model stop`, also when devrun's timeout
78
+ # fires: -k 150 > the 120 s grace). No token reaches the public weights download. Evidence: logs/smoke/ (--log-dir).
79
+ $ROOT/bin/devrun -t 3600 -k 150 -- env -u HF_TOKEN -u HUGGING_FACE_HUB_TOKEN \
80
+ sg docker -c "bash code/scripts/container_smoke.sh $ROOT/build/meteor-p150 20000 [--profile NAME]"
81
+
82
+ # publish: overlay the hand-written docs, the pip project and the build record onto the staged dir, then push
83
+ # (HF_TOKEN only from the environment)
84
+ rsync -a README.md SERVING.md OPT_BASELINE.md OPT_REPORT.md VERIFICATION_*.md tt-model.yaml pyproject.toml build_info.json \
85
+ examples media patches $ROOT/build/meteor-p150/
86
+ cp PYTHON.md $ROOT/build/meteor-p150/PYTHON.md # the top-level copy (its links are relative to the repo root)
87
+ tt-model push $ROOT/build/meteor-p150 --publish
88
+ ```
89
+
90
+ The smoke test fails unless `/info` reports ETH dispatch, the 12x10 grid and what the staged package pins for the
91
+ profile (`METEOR_DISPATCH`, `METEOR_NUM_CQS`, `METEOR_VARIANT`, the weights revision), and it compares the
92
+ served output of the shipped sample (`code/tt_meteor/samples/synthetic_8cam.json`) with its stored CPU-reference
93
+ output (`synthetic_8cam.reference.json`: 3D boxes within 0.5 m and 0.05 of score, recall and precision >= 0.95; lane
94
+ map agreement >= 0.99; selected ego path ADE <= 0.3 m, FDE <= 0.6 m) and checks the expected boxes per label.
95
+
96
+ For the first release of this bundle the container check also runs the rig-switch stress (the ETH-dispatch hang
97
+ history of PORT_LOG.md section 12): `code/scripts/stress_frames.py --mode rigs --rigs shipped` for >= 200 frames
98
+ inside the image, with `tt-model serve --print`'s device, cache and env arguments (no port) plus
99
+ `METEOR_WEIGHTS_DIR=/hf/hub/models--AutowareFoundation--meteor/snapshots/<revision>` (the script reads the ONNX
100
+ directly, without the server's HF resolve; `serve` has pre-fetched the snapshot):
101
+ `python -u /opt/tt-metal/scripts/stress_frames.py --mode rigs --rigs shipped --frames 200 --jsonl <mounted dir>/stress.jsonl`.
102
+
103
+ `serve` publishes the first free port from 20000 and exports into the container: `HF_MODEL=AutowareFoundation/meteor`,
104
+ `MESH_DEVICE=P150`, `TT_MESH_SHAPE=1x1`, `TT_MODEL_WEIGHTS_REVISION=01a5f6d71df5ecbbb5853ec600825481d57b9c6b`, then `serve.env`. The HF cache is
105
+ mounted at `/hf` (rw), the JIT cache at `/cache` (host `~/.cache/tt-model/meteor-p150/cache`), and the container
106
+ sees only its own chip (`--device /dev/tenstorrent/<n>`). tt-cli users: `tt serve changh95/meteor-p150` /
107
+ `tt model stop changh95/meteor-p150` (point tt at this tt-model with
108
+ `tt config set tools.override.tt-model ~/.local/bin/tt-model` or `TT_TOOL_BIN_TT_MODEL`).
109
+
110
+ What `push` does to the repo: `code/` and `image/` on the Hub become exactly the staged trees (`extra_code.paths` +
111
+ `models/common/lightweightmodule.py`); every top-level file of the staged dir is uploaded (the overlaid README
112
+ replaces the generated card; `pyproject.toml` and `build_info.json` arrive the same way, never through `code/`);
113
+ top-level files already on the Hub and not in the staged dir are kept.
114
+
115
+ ## 3. The request / response contract
116
+
117
+ | route | purpose |
118
+ |---|---|
119
+ | `GET /health` | `{"status": "ok" \| "starting" \| "error", "model", "device": {"dispatch", "grid", "cores", "device_id"}, "error"}`, always 200; `ok` only after warm-up |
120
+ | `GET /v1/health` | same as `/health` (tt-model's hint for non-chat packages points here) |
121
+ | `GET /info` | model, task, io, Autoware package + commit, weights {repo, tag, revision, path, license}, device (dispatch, grid), input limits (point formats, fields, camera order, calibration presets), labels, variant, warm variants, runtime params + defaults, compile params, warm-up and boot times |
122
+ | `GET /v1/models` | `{"object": "list", "data": [{"id": "AutowareFoundation/meteor", "object": "model", "owned_by": "changh95"}]}` -- a stub so OpenAI-shaped probes do not 404; NOT a chat API |
123
+ | `POST /predict` | one frame -> the model output (below) |
124
+
125
+ ### 3.1 Request (`application/json`; unknown fields are a 422)
126
+
127
+ | field | type | meaning |
128
+ |---|---|---|
129
+ | `images` | list | **all eight cameras** `[{"camera": "CAM_FRONT_WIDE", "data": <base64 PNG / JPEG / .npy>, "intrinsics", "T_ref_from_camera"}, ...]`, any order (reordered to `CAM_FRONT_WIDE, CAM_FRONT_LEFT, CAM_FRONT_RIGHT, CAM_BACK_WIDE, CAM_BACK_LEFT, CAM_BACK_RIGHT, CAM_FRONT_NARROW, CAM_BACK_NARROW`); RGB, raw and unrectified, at least 768×432 (larger frames are resized with OpenCV INTER_AREA semantics and K is scaled per axis). An absent narrow camera is sent as an **all-zero image**: the host gives it a zero input and its donor's K and pose (`CAM_FRONT_WIDE` / `CAM_BACK_WIDE`), the trained 7-camera configuration |
130
+ | `calibration` | object | `{"frame_id", "cameras": {name: {"intrinsics", "T_ref_from_camera", "image_size"}}}` (K of the image as sent; `T_ref_from_camera` = camera optical frame -> base_link: x forward, y left, z up, origin on the road) or `{"preset": "<name>"}` (presets in `/info`: `synthetic_8cam`, the rig of the shipped sample). Inline image calibration wins |
131
+ | `ego_speed` | number | **required**: the ego speed in m/s (the graph's `v0`) |
132
+ | `stream` | object | optional: `{"id": "default", "reset": false, "timestamp_s", "T_world_from_ego"}`: METEOR's host temporal post-processing per stream id (the BEV seg fusion needs `T_world_from_ego`; yaw smoothing; plan-mode hysteresis); a new id or `reset` starts fresh. Without `stream`, every request uses the stream `default` |
133
+ | `params` | object | per-request knobs (host post-processing only; METEOR's C++ renderer defaults): 3D `det3d_threshold` (0.15), `det3d_topk` (64), `vehicle_threshold` (0.35), `vru_threshold` (0.15), `bev_nms_iou` (0.3), `bev_nms_containment` (0.6), `stationary_logit_threshold` (0.0); 2D `det2d_threshold` (0.30), `det2d_topk` (48), `det2d_hide` ("7": road paint); unknown obstacles `unk2d` (true), `unk2d_threshold` (= det2d), `ground_z` (0.0); plan `mode_hysteresis` (0.35), `straight_margin` (1.0); `seg_fuse` (true), `thin_road_edge` (true), `yaw_smoothing` (true); `heads` (false: with `output_format: "npz"`, adds the dense heads) |
134
+ | `output_format` | str | `json` (default) or `npz` (adds lossless base64 NPZ arrays) |
135
+
136
+ `points`, `sweeps` and `inputs` of the common envelope are not used by this model (a request with `points` is a 400).
137
+ `server/client.py` builds a request: `--sample <manifest>` (the shipped sample: images, preset, ego speed, stream),
138
+ or `--image NAME=PATH` (×8) + `--calib <file>` / `--calib-preset <name>` + `--ego-speed <m/s>` [+ `--stream-id`,
139
+ `--param k=v`].
140
+
141
+ Transforms (`T_<to>_from_<from>`): a 4×4 (or 3×4) row-major matrix, `{"translation": [x, y, z], "rotation_wxyz": [w, x, y, z]}`
142
+ (or `rotation_xyzw`), or `{"x", "y", "z", "roll", "pitch", "yaw"}` (Autoware `sensor_kit_calibration.yaml`, tf2 RPY).
143
+ Intrinsics: 3×3 K, 3×4 P, or `{"fx", "fy", "cx", "cy"}`. Base64 may be standard or URL-safe, wrapped, with or without a
144
+ `data:` prefix.
145
+
146
+ ### 3.2 Response (200)
147
+
148
+ ```json
149
+ {"model": "meteor-p150", "frame_id": "base_link", "num_detections": 8,
150
+ "detections": [
151
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9932, "center": [-12.86, 0.095], "size": [4.399, 1.742], "yaw": -0.0687, "stationary": false, "future_mode": 0, "future": [[-11.92, -0.09], [-10.74, -0.15], ...]},
152
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9872, "center": [-7.176, -3.215], "size": [3.965, 1.731], "yaw": 0.0183, "stationary": false, ...},
153
+ {"label": "VEHICLE", "label_id": 0, "score": 0.9624, "center": [10.84, 0.019], "size": [4.334, 1.761], "yaw": -0.0189, "stationary": false, ...},
154
+ ...],
155
+ "trajectory": [[0.74, 0.91], [0.85, 1.54], [0.41, 2.34], [0.31, 3.08], [0.62, 3.82], [0.3, 4.45]], "columns": ["x", "y"],
156
+ "plan": {"mode": 0, "mode_probs": [0.233, 0.296, 0.471], "mode_logits": [...], "paths": [...], "steer": 0.0026, "accel": 0.021, "brake_prob": 0.131, "dt": 0.5},
157
+ "detections_2d": {"CAM_FRONT_WIDE": [{"label": "car", "label_id": 1, "score": 0.5239, "box_xyxy": [412.76, 211.02, 438.5, 230.29]}, ...], ...},
158
+ "unknown_obstacles": [],
159
+ "traffic_light": {"state": "red", "state_id": 3, "probs": [0.363, 0.140, 0.031, 0.465]},
160
+ "lane": {"format": "png", "key": "lane", "dtype": "uint8", "shape": [800, 500], "data": "iVBORw0KGgo..."},
161
+ "lane_classes": ["bg", "road", "sidewalk", "crosswalk", "laneline", "stopline", "road_edge", "marking", "parking"],
162
+ "stationary_head_healthy": true,
163
+ "meta": {"present": [true, true, true, true, true, true, true, true], "v0": 8.0, "stream_id": "synthetic_8cam"},
164
+ "timing_ms": {"decode": 83.95, "preprocess": 10.69, "device": 722.42, "postprocess": 86.08, "model_call": 820.65, "total": 908.67}}
165
+ ```
166
+
167
+ (The shipped synthetic sample, a request after the warm-up, host server, 2026-10-10; trimmed.)
168
+
169
+ | key | content |
170
+ |---|---|
171
+ | `detections` | 3D BEV boxes sorted by score: `label` (VEHICLE / VRU), `label_id`, `score`, `center` [x, y] and `size` [length, width] in metres (base_link: x forward, y left; METEOR predicts no z, height or velocity), `yaw` (rad, CCW from +x), `stationary`, `future` (6 × [x, y] at 0.5-3 s, the agent's best mode; METEOR's renderer draws vehicle futures only), `future_mode` |
172
+ | `trajectory`, `columns` | the selected ego path, 6 × [x, y] at 0.5 s (straight preference and per-stream mode hysteresis, METEOR's `render.cpp`) |
173
+ | `plan` | `mode`, `mode_probs` (softmax of the hysteresis-adjusted logits), `mode_logits`, `paths` (3 × 6 × [x, y]), `steer` (rad), `accel` (m/s²), `brake_prob`, `dt` |
174
+ | `detections_2d` | per camera: `label` (10 classes; road paint hidden by default), `label_id`, `score`, `box_xyxy` (pixels at 768×432) |
175
+ | `unknown_obstacles` | 2D "obstacle" boxes placed on the ground plane (`unk2d`) |
176
+ | `traffic_light` | the ego-relevant light: `state` (none / green / yellow / red), `state_id`, `probs` |
177
+ | `lane`, `lane_classes` | the BEV lane map as PNG (uint8 [800, 500], 0.2 m cells, row 0 = 80 m ahead, column 0 = 50 m to the left), after the seg fusion (with a stream pose) and the road-edge thinning |
178
+ | `stationary_head_healthy`, `meta` | METEOR's stationary-head health check; the present cameras, `v0`, the stream id |
179
+ | `heads` | only with `output_format: "npz"` and `params.heads: true`: seg2d, depth, depth_mean, occupancy, risk, stationary |
180
+
181
+ The synthetic sample is a test pattern (a red light and a car 11 m ahead): its plan barely moves forward (under 1 m in 3 s,
182
+ drifting 4.5 m to the left), not meaningful driving; the stored CPU reference has the same path.
183
+
184
+ `timing_ms`: `decode` (base64 + parsing), `preprocess`, `device` (H2D + trace + D2H), `postprocess`, `model_call`,
185
+ `total` (server side, after the body arrived).
186
+
187
+ ### 3.3 Errors
188
+
189
+ **400** undecodable / malformed input (bad base64, wrong row size, missing calibration, unknown or out-of-range
190
+ `params`, a missing required input), **422** schema violation (unknown field, wrong type), **503** while starting
191
+ (or when the boot failed), **500** `inference failed: <ExceptionType>: <message>` if the device call raises.
192
+
193
+ ### 3.4 Environment the app reads (lifespan only, never at import)
194
+
195
+ | var | set by | meaning / default |
196
+ |---|---|---|
197
+ | `HF_MODEL` | launcher (`weights.repo`) | weights repo id; default `AutowareFoundation/meteor` |
198
+ | `TT_MODEL_WEIGHTS_REVISION` | launcher (`weights.revision`) | pinned commit; `TT_WEIGHTS_REVISION` (`serve.env`) is the same for older tt-model |
199
+ | `METEOR_WEIGHTS_DIR` | you (offline) | local weights directory; overrides the Hub |
200
+ | `TT_MESH_SHAPE` | launcher (`runtime.mesh_shape_env`) | `1x1`; anything else -> RuntimeError at startup |
201
+ | `TT_DEVICE_ID` | you | chip to open, default 0 |
202
+ | `METEOR_DISPATCH` | `serve.env` | `eth` (default) \| `worker` (A/B only) \| `auto` (ETH if the patch is present) |
203
+ | `METEOR_NUM_CQS` | `serve.env` | `1` |
204
+ | `METEOR_VARIANT` | `serve.env` / profile | `default` |
205
+ | `METEOR_WARMUP` | you | JSON list of warm-up variants, `default` or `none` |
206
+ | `METEOR_TRACE_REGION`, `METEOR_L1_SMALL`, `METEOR_WORKER_L1_SIZE` | you | device-open overrides (validated values: `DEVICE_DEFAULTS` in `device.py`) |
207
+ | `METEOR_MAX_BODY_MB` | you | request size guard, default 256 |
208
+ | `TT_METAL_VISIBLE_DEVICES`, `MESH_DEVICE` | `serve.env` / launcher | `0`, `P150` (informational) |
209
+
210
+ Server-side pipeline: JSON -> `tt_meteor.io` decoders -> `METEOR.__call__` (host pre-processing of METEOR's runtime:
211
+ INTER_AREA resize, K scaling, extrinsics inversion, the absent-camera donor rule; the lift tables of a new
212
+ calibration, cached per rig -> H2D of the eight cameras and `v0` into persistent device inputs -> `execute_trace` of
213
+ the `frame` trace -> segmented D2H -> host post-processing ported from `deploy/cpp`: 3D peaks + rotated BEV NMS,
214
+ stationary flags, agent futures, 2D boxes, unknown obstacles, the plan with hysteresis, seg fusion) under one lock
215
+ -> `Output.to_dict()`.
216
+
217
+ ## 4. Caveats
218
+
219
+ - Batch 1, one frame per request; concurrent clients queue on the lock.
220
+ - Fixed shapes of the released graph: eight 768×432 camera slots in METEOR's order, a 400×250 lift grid at 0.4 m,
221
+ an 800×500 BEV at 0.2 m; at most 3 cameras per lift cell on the rigs validated. A new calibration costs about
222
+ 0.2 s once (the lift tables), then it is cached per rig.
223
+ - The request bodies are large (eight images as base64): about 1-4 MB for PNG / JPEG at 768×432.
224
+ - Warm-up captures every trace variant inside the lifespan, so READY means warm; the first cold boot pays the ttnn JIT.
225
+ - Weights are pinned by sha in `weights.revision`, exported by the launcher and repeated in `serve.env`;
226
+ `snapshot_download(..., revision=<sha>)` is a cache hit after `serve`'s pre-download and falls back to
227
+ `local_files_only=True` if the Hub is unreachable.
228
+ - The runtime image has no host C/C++ compiler: device kernels JIT-compile (sfpi ships in the image), host-side C
229
+ helpers must have a numpy fallback.
230
+ - `tt-model curl` and the ready card's `/v1/models` hint are OpenAI-shaped and are not this API; use the routes above.
VERIFICATION_2026-10-10.md ADDED
@@ -0,0 +1,182 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # meteor-p150: verification of the first release, 2026-10-10
2
+
3
+ ## Verdict
4
+
5
+ **PASS.** This is the light verification pass of the first publish (`research/PLAN.md` §6.3): the device suite
6
+ re-run with every frozen gate (read-only), the baseline numbers re-checked, the published configuration exercised end
7
+ to end (quickstart, host server + smoke test, a golden-free ETH stress), and every device-open path audited for ETH
8
+ dispatch. It was made by the agent that wrote the release docs, on the release tree (bundle commit `cc35027` plus the
9
+ release changes that this file is committed with). Against the baseline commit `e86c13a` (`OPT_BASELINE.md`,
10
+ committed as `cc35027`):
11
+
12
+ - **Accuracy:** the device suite passes (**16 passed** in one devrun job, 93 s). Every stored value
13
+ (`pcc_<frame>.json` of the 4 frames, `e2e_device.json`, `variants_device.json`) is **identical** to the baseline
14
+ code's three ETH suite runs (`logs/meteor/hangfix/values_final1/`), timings aside; the only new entry is the
15
+ shipped synthetic sample's check (8 / 8 boxes, max |score difference| 0.0020, lane agreement 0.9987, path ADE
16
+ 0.018 m), gated with the existing thresholds. 38 public frames through the served graph agree with the CPU goldens
17
+ (below).
18
+ - **Speed:** the device rows reproduce `OPT_BASELINE.md` to 0.15 ms (replay 556.92 vs 556.85 ms, back to back 556.89
19
+ vs 556.75 ms); the end-to-end PandaSet call is 1.9 % slower (935.9 vs 918.2 ms, p50) on a busier shared host.
20
+ - **Audit:** ETH dispatch, 1 CQ and the 12×10 grid on every device-open path; no fallback in any run; both tt-metal
21
+ patches present (`eth_patch: true`; the reshape-patch marker is asserted by the image build).
22
+ - **No numerics change of the device graph** since the verified port: `e86c13a..` touches no file under `tt/`, no
23
+ precision knob and no vendored ttaw file (`vendor.py --check`: 0 differences, ttaw 0.23.2 @ common `60b6dd7`). The
24
+ release changes are host-side and documented (OPT_REPORT.md "release" row). No gate threshold moved; the gate files
25
+ are byte-identical.
26
+ - **No hang, no `.device.FAULT`, tt-triage never armed**: all 6 device jobs exited with rc 0 (every METEOR device job
27
+ ran with the orchestrator's `METEOR_DEVICE_OK=1`).
28
+
29
+ The adversarial verification of the port is `VERIFY_PORT.md` (round 1, 2026-10-09: FAIL on two medium findings, V1
30
+ and V2, plus 7 low ones). Both mediums were fixed in PORT_LOG.md section 11 (E22, E23); "Review findings" below
31
+ lists where each finding stands. No second full verification round was run before this release.
32
+
33
+ ## Setup
34
+
35
+ | item | value |
36
+ |---|---|
37
+ | Hardware | one Blackhole p150b, ETH dispatch, 1 CQ, 12×10 compute grid (120 cores), as printed by every run (`dispatch eth`, `fallback null`, `eth_patch true`); AICLK 1350 MHz (median of 163 samples during the bench, minimum 1343), board power median 69 W |
38
+ | tt-metal | `44d66500520` with `patches/tt-metal-eth-dispatch.patch` (sha256 `08d0ddf6…`) and `patches/tt-metal-reshape-rm-sys1419.patch` (sha256 `74878683…`), both byte-identical to the workspace copies the tree was built with |
39
+ | Weights | `AutowareFoundation/meteor` @ `01a5f6d71df5ecbbb5853ec600825481d57b9c6b` (tag `v1.0`) from the HF cache at the pinned revision (`HF_HUB_OFFLINE=1`, no token) |
40
+ | Verified code | `cc35027` + the release changes (ttaw 0.23.2 @ common `60b6dd7`, `vendor.py --check`: 0 differences) |
41
+ | Baseline | `e86c13a` (`OPT_BASELINE.md`) |
42
+ | Workloads | the shipped sample `samples/synthetic_8cam` (8 PNGs, 768×432, all cameras present, preset `synthetic_8cam`); the staged PandaSet 019 frame 40 sample (not shipped); the 4 golden frames of the gates (research goldens, local); PandaSet 019 / 090 and nuScenes scene-0103 frames of `research/meteor/public_data` (38 frames, local, never shipped) |
43
+ | Bench | `code/scripts/bench.py --inputs shipped,sample --iters 30 --b2b 20` (p50 / p99 of 30 per loop), one run |
44
+ | Host | AMD EPYC-Rome VM, 8 vCPUs, `OMP_NUM_THREADS=4`, shared with other agents' jobs (load average 6.0 → 8.7 during the bench) |
45
+ | Logs | `logs/meteor/release/` of the workspace (`job_{media,suite,quickstart,serve,bench,stress}.{sh,log}`, `run_all.rcs`); devrun windows 16:13:59, 16:30:01, 16:35:42, 16:38:44, 16:45:31, 16:50:22 UTC (`logs/devrun/history.log`), all rc 0 |
46
+
47
+ ## Measured baseline vs re-check
48
+
49
+ | measurement | baseline `e86c13a` (`OPT_BASELINE.md`) | re-check (release tree) | change |
50
+ |---|---:|---:|---:|
51
+ | one blocking replay of `frame`, p50, PandaSet 019 | 556.85 ms (p99 559.34) | 556.92 ms (p99 557.5) | +0.07 ms |
52
+ | back-to-back replays, per frame | 556.75 ms | 556.89 ms | +0.14 ms |
53
+ | H2D (8 cameras + v0) | 66.5 ms | 66.6 ms | +0.1 ms |
54
+ | D2H (75 MB, 5 segments) | 17.2 ms (17-48 bimodal) | 46.4 ms | in the documented 17-48 ms range (host side) |
55
+ | host preprocess / camera transpose | 11.2 / 26.3 ms | 13.8 / 27.0 ms | host load |
56
+ | readback conversion (unpack + outputs) | 15.9 + 49.4 ms | 14.9 + 43.3 ms | host |
57
+ | host postprocess | 129.9 ms | 133.8 ms | host load |
58
+ | e2e `model()` PandaSet 019, p50 (p99) | 918.2 (1,071.8) ms | 935.9 (999.6) ms | +1.9 % |
59
+ | `from_pretrained` load, warm kernel cache | 49.7 s | 45.3 s (weights 0.3, build 2.3, warm-up 38.6 s) | −4.4 s |
60
+ | PandaSet 019 frame vs its stored CPU reference (bench check) | 8 / 8, recall / precision 1.0 / 1.0, max \|Δscore\| 0.0081, lane 0.99770, path 0.108 m | identical | 0 |
61
+
62
+ The device rows reproduce; the host rows move with the load of the shared host.
63
+
64
+ Not in the baseline, measured for the card on the release tree:
65
+
66
+ | measurement | value |
67
+ |---|---:|
68
+ | shipped sample through `model()` (bench, 30 iterations) | e2e **871.6 ms** (942.1); device 755.6, postprocess 95.2; replay 556.80 ms, back to back 556.65 ms |
69
+ | shipped sample vs its stored CPU reference (`test_e2e_device.py::test_shipped_sample_vs_stored_reference`, the bench check and the server smoke) | 8 / 8, recall 1.000, precision 1.000, max \|score difference\| 0.0020, centre error max 0.002 m, lane agreement 0.9987, path ADE / FDE 0.0177 / 0.0298 m, same plan mode |
70
+ | served `/predict`, host uvicorn, the smoke request (first request: includes the rig change of the new calibration) | `timing_ms.total` 1,226.0 ms (device 991.7), client round trip 1,250.7 ms |
71
+ | served `/predict`, three more requests of the same body (1.8 MB) | `timing_ms` decode 81.3-84.0 · preprocess 10.5-10.8 · device 719.5-722.4 · postprocess 86.1-87.1 · **total 904.0-908.7 ms**; client round trip 0.92-0.93 s |
72
+ | server boot to ready (warm kernel cache) | 44.6 s (`ready in 44.6 s`; warm-up 37.8 s) |
73
+ | served body vs the stored CPU reference (`server/smoke_test.py`, the container-smoke rule, `--expect VEHICLE:6,VRU:2`) | **PASS**: 8 / 8, recall 1.000, precision 1.000, dscore 0.0020; trajectory ADE 0.0177, FDE 0.0298; lane 0.9987; `/info` dispatch eth, grid 12x10, 1 CQ |
74
+ | `examples/quickstart.py` (from the repo root, default input) | rc 0: 8 boxes (6 VEHICLE, 2 VRU, scores 0.887-0.993), plan mode 0, traffic light red, 10 2D boxes; `quickstart.json` and `quickstart_bev.png` written |
75
+ | golden-free ETH stress, `stress_frames.py --mode rigs --rigs shipped --frames 200` (4 synthetic rigs, a rig change + table write + upload + replay + read every frame) | **200 frames clean**, 0 mismatches (4 stable per-rig digests); table write 164.6 ms, run 708.0 ms (max 784.8) |
76
+ | 38 public frames through the served graph (`code/scripts/run_public_frames.py`; every feed's sha256 equal to the CPU golden's) | 707-1,005 ms per frame (the first frame of each rig includes the table write) |
77
+
78
+ ## Accuracy gate results
79
+
80
+ `test_pcc_device.py` + `test_e2e_device.py` + `test_variants_device.py` in one pytest, one devrun job,
81
+ `TTAW_GATES_READONLY=1`: **16 passed** (93 s). The gate files are unchanged (`git diff --stat` empty for
82
+ `*.gates.json`); the new shipped-sample test reuses the existing `det3d.*`, `ego.path_dev_m` and `lane.agreement`
83
+ gates and froze nothing.
84
+
85
+ | gate | threshold | value (identical to the baseline's ETH runs) |
86
+ |---|---:|---:|
87
+ | teacher-forced stage PCC, 26 gates (PandaSet 019 f40) | ≥ 0.99 | min 0.9999890 (`tf.risk`); `depth_mean` 0.99948 (reported) |
88
+ | teacher-forced lane / seg2d / depth argmax agreement | ≥ 0.99 | 0.9984 / 0.9965 / 0.9936 |
89
+ | chained frame, min output PCC: PandaSet 019 / 090, nuScenes 0103, valday #40 | ≥ 0.99 | 0.99989 / 0.99990 / 0.99991 / 0.99982 |
90
+ | chained lane / seg2d / depth agreement, worst frame | ≥ 0.99 | 0.9963 / 0.9965 / 0.9925 |
91
+ | replay == eager | bit-identical | yes (19 outputs) |
92
+ | det3d strict recall / precision: PandaSet sample, PandaSet 090, nuScenes, valday #40 | ≥ 0.95 / ≥ 0.95 | 1.0/1.0, 1.0/1.0, 1.0/1.0, 0.971/0.971 (34 / 35) |
93
+ | det3d near-tie extras (recall_nt ≥ recall, ambiguous_frac) | ≥ 0.95, ≤ 0.10 | pass on every frame |
94
+ | selected ego path deviation, worst frame | ≤ 0.3 m | 0.170 m (PandaSet 090) |
95
+ | lane agreement after the host decode, worst frame | ≥ 0.99 | 0.9966 |
96
+ | PandaSet sample `/predict` vs its stored reference body | ≥ 0.95 / ≥ 0.95 | 8 / 8, max \|Δscore\| 0.0081 |
97
+ | **shipped synthetic sample** `/predict` vs its stored reference body (new) | the same gates | 8 / 8, max \|Δscore\| 0.0020, path 0.030 m, lane 0.9987 |
98
+ | API == `/predict` (shipped sample) | identical bodies | yes |
99
+ | `input_norm=imagenet` + `depth_mean_bins=linear` variant, float outputs min PCC | ≥ 0.99 | 0.99988 |
100
+
101
+ **Public frames** (not gated by the device suite; `research/meteor/public_data/scripts/compare_tt.py` with the
102
+ PLAN thresholds, the TT outputs written by `run_public_frames.py`; `logs/meteor/release/public_tt/compare/`):
103
+
104
+ | sequence | frames | lane / seg2d / depth agreement (min) | dense-head PCC min (hm / reg / stationary / risk) | all 19 outputs, PCC min (one frame) | path dev max | 3D boxes: CPU matched / TT precision | same mode / TL |
105
+ |---|---:|---|---|---:|---:|---|---|
106
+ | PandaSet 019 | 13 | 0.99761 / 0.99637 / 0.99330 | 0.99995 / 0.99995 / 0.99999 / 0.99962 | 0.99948 (`depth_mean`) | 0.069 m | 99 / 103, 1.0 | 13 / 13 |
107
+ | PandaSet 090 | 13 | 0.99708 / 0.99684 / 0.99483 | 0.99996 / 0.99996 / 0.99999 / 0.99982 | 0.99983 | 0.110 m | 69 / 71, 1.0 | 13 / 13 |
108
+ | nuScenes scene-0103 | 12 | 0.99764 / 0.99624 / 0.99253 | 0.99997 / 0.99993 / 0.99998 / 0.99962 | 0.99902 | 0.081 m | 240 / 244, 1.0 | 12 / 12 |
109
+
110
+ Every frame passes the argmax, PCC and path gates. 8 of the 38 frames fall below the 95 % "golden boxes matched"
111
+ check of `compare_tt.py` because they have few boxes (5-14) and miss one or two: of the 10 missed CPU boxes, 9 score
112
+ within 0.004 of their class threshold on the CPU (0.3507-0.3538 vs 0.35; 0.1505-0.1513 vs 0.15), so a tiny score
113
+ difference drops them; the 10th is a pedestrian in a nuScenes crowd whose TT peak lies 2.1 m away. The TT never
114
+ produces a box the CPU does not (precision 1.0 on every frame). This is the documented near-threshold behaviour of a
115
+ bf16 / fp32 port (README Caveats), not a gate of the port: the frozen det3d gates run on the four golden frames above.
116
+
117
+ ## Disclosed numerics changes
118
+
119
+ None on the device: the release does not touch `tt/`, the precision knobs or the vendored ttaw. The precision policy
120
+ is the port's (PLAN §9.3; PORT_LOG E20 / E22): fp32 weights and biases, HiFi4, fp32 dest and packer L1 accumulation;
121
+ the image branch and the 3D detection path with fp32 activations as three bf16-term convs; the BEV trunk in bf16.
122
+
123
+ Host-side changes of the release (no effect on any gate value, confirmed by the identical stored values):
124
+ - the shipped sample is now `samples/synthetic_8cam*` (generated, Apache-2.0); the PandaSet sample, its preset and
125
+ its small goldens moved unchanged to the git-ignored `staging_samples_pandaset/` (sha256-checked), where
126
+ `tests/paths.py` finds them (inside the workspace a missing staging dir fails those tests);
127
+ - `load_sample` resolves a preset from the `calib/` beside the manifest's `samples/` directory first (a staged sample
128
+ carries its own preset) and returns it inline; `server/smoke_test.py` sends a staged preset inline;
129
+ - `opencv-python-headless>=4.8,<4.12` declared in `pyproject.toml` and `runtime.packages` (+ an `import cv2` verify
130
+ line): the host post-processing needs OpenCV (PORT_LOG I3) and tt-metal's ttnn install does not bring it;
131
+ - `server/smoke_test.py`: the default sample and `DEFAULT_EXPECT = "VEHICLE:6,VRU:2"` (it was the unfilled template
132
+ string (the template placeholder `SMOKE_EXPECT` in double braces), which the expectation parser would have read as a required label: the container smoke
133
+ would have failed);
134
+ - `examples/quickstart.py` called `model(path)` (a path is the point-cloud argument: an `InputError`) and `len(out)`;
135
+ fixed to `model(**load_sample(path))`, and it now writes the `quickstart_bev.png` it promised;
136
+ - `server/client.py`: `--sample <manifest>` and `--ego-speed` (the vendored client cannot send the required
137
+ `ego_speed`);
138
+ - `stress_frames.py --rigs shipped[:N]`: rigs from the shipped sample, no golden needed (for the container check).
139
+
140
+ ## Review findings
141
+
142
+ - No test threshold was loosened, no gate file edited; the device tests compare with goldens written by the fp32
143
+ CPU reference before any device code existed (VERIFY_PORT 3(e)); the shipped sample's reference was written by the
144
+ same CPU reference from the PNG files exactly as the API reads them (`make_synthetic_sample.py finalise` asserts the
145
+ PNGs read back bit-exactly).
146
+ - The shipped sample was optimised against the CPU reference (object pixels only) so that its published boxes sit
147
+ far from the thresholds (scores 0.889-0.993 vs 0.35 / 0.15; the strongest other heatmap cell 0.179 vehicle /
148
+ 0.072 VRU). This makes the container smoke stable; it does not make the TT output agree better (the optimisation
149
+ never saw a TT output).
150
+ - Timing loops measure what they claim (`bench.py` synchronises each stage; the b2b row replays 20 frames back to
151
+ back); every bench run also checks its own output against the stored reference.
152
+ - Nothing hung; the stress jobs ran contained (devrun timeout, no triage, no operation timeout).
153
+ - VERIFY_PORT round 1 findings:
154
+ - V1 (det3d gate redefined after a failure): **fixed** by E22 (three-term det path): the strict PLAN gate is gated
155
+ on every frame again, valday #40 0.971 / 0.971; re-confirmed here.
156
+ - V2 (`input_norm=imagenet` never ran on the card): **fixed** by E23 (`test_variants_device.py`, also covering
157
+ `depth_mean_bins=linear`, L1); re-confirmed here.
158
+ - L1 (`IMAGE_PRECISION=bf16` exposed): the knob's documentation states it is below the depth gate (A/B only).
159
+ - L2 (`depth_mean` max abs error from bin flips) and L3 (centre errors up to 0.4 m on public frames): reported
160
+ metrics, unchanged.
161
+ - L4 (fp32 D2H, host transpose): on the optimization backlog (OPT_REPORT items 3 and 7).
162
+ - L5 (the API falls back to WORKER if the ETH open fails): unchanged by design (shared ttaw); the container smoke
163
+ fails unless `/info` reports eth / 12x10, and the tests open with `allow_fallback=False`.
164
+ - L6 (`_tables_key` set before the upload succeeded): fixed in the hang round (PORT_LOG 12.1 A3: the key is set
165
+ after the last chunk lands). `tt/terms.py` stays unwired (documented). I3 (cv2 in the image): fixed in this
166
+ release (declared + verify line). I4 (docs pointing at autoware_universe): fixed in this release (README,
167
+ SERVING, card, pyproject point at tier4/METEOR).
168
+ - L7: cosmetic.
169
+
170
+ ## p150 ETH-dispatch compliance
171
+
172
+ | path | dispatch / CQs / grid |
173
+ |---|---|
174
+ | Python API `METEOR.from_pretrained()` | `DEVICE_DEFAULTS` (`device.py`): no dispatch override, so the vendored `ttaw.device` default ETH; 1 CQ; `l1_small_size` 32 KiB, `trace_region_size` 256 MiB. Every run of this pass printed `dispatch eth`, `grid 12x10`, `cores 120`, `fallback null`, `eth_patch true` |
175
+ | HTTP server (`METEOR_DISPATCH` default) | the same defaults; the host server run reported `/info` device `eth`, `12x10`, 1 CQ, `fallback null` |
176
+ | `tt-model.yaml` serve env (every serve profile) | one profile (`default`): `METEOR_DISPATCH: "eth"`, `METEOR_NUM_CQS: "1"`, `METEOR_VARIANT: "default"`; `check_bundle.py` refuses any other dispatch |
177
+ | image `verify:` | the ETH-patch marker in `topology.cpp`, `ttaw.device.eth_dispatch_patch_present()`, the reshape-patch marker (`dual_kernel_min_dram_dest_page_bytes`) in `reshape_rm_program_factory.cpp`, `import cv2`, the shipped sample and preset present |
178
+ | container smoke per serve profile (`/info` asserted: eth, 12x10, pins) | `server/smoke_test.py` hard-codes `EXPECT_DISPATCH, EXPECT_GRID = "eth", "12x10"` and checks the staged package's pins; run at packaging |
179
+ | gate tests (`code/conftest.py`) | `open_device(allow_fallback=False)`: a failing ETH open is an error, never a silent WORKER run |
180
+ | bench / profile / stress / public-frame scripts | `bench.py` and `profile_ops.py`: `--dispatch` default None = the API default (eth); `stress_frames.py`: `--dispatch eth` default with `allow_fallback=False`; `run_public_frames.py`: the API default, reports the device (eth, 12x10); the repro scripts default to eth |
181
+ | no hard-coded grid | `grep` for `CoreGrid`, `CoreRange`, `grid_x`, `core_grid` in `tt/` finds nothing (VERIFY_PORT 3(b)); WORKER 11×10 builds its own program set (OPT_BASELINE "Grid usage") |
182
+ | vendored `ttaw` | 0.23.2 @ common `60b6dd7` (`VENDORED.json`, `source_dirty: false`); `vendor.py --check`: 0 differences; `check_bundle.py` drift check: clean |
build_info.json ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema": "ttaw-build-info/1",
3
+ "bundle": "meteor-p150",
4
+ "recorded_at": "2026-10-10T17:14:45Z",
5
+ "image": {
6
+ "tag": "tt-model/meteor-p150:823dc1fbd4c7",
7
+ "digest": "sha256:823dc1fbd4c7ec836a6e74486495d2801a71d6fa4e18d41011fa873749962afc",
8
+ "built_at": "2026-10-10T17:04:30+00:00",
9
+ "code_sha256": "40de720599c6b5b9be0687f81b391539cf3e57dfb5614eab5a7bc721791748fc",
10
+ "size_bytes": 4265173718,
11
+ "layers": 23
12
+ },
13
+ "base_images": {
14
+ "build": {
15
+ "stage": "prep",
16
+ "ref": "ghcr.io/tenstorrent/tt-metal/tt-metalium/ubuntu-22.04-dev-amd64:latest",
17
+ "digest": "sha256:df9d279c7f85c17c6fad982d196802682d669cca1b7ced9cbaad8181339cd5fc",
18
+ "source": "docker build log",
19
+ "local_tag_matches": false,
20
+ "local_id": "sha256:3fd1e6013e658c65df1bc6b084543e7174d854c15af97373e493df0c1787c393",
21
+ "created": "2026-10-07T01:17:29.186036377Z",
22
+ "note": "the local tag moved after this build; the digest above is the one the image was built from"
23
+ },
24
+ "runtime": {
25
+ "stage": "runtime",
26
+ "ref": "docker.io/library/ubuntu:22.04",
27
+ "digest": "sha256:5ec03bb3441e8b0bf3b4f9cd4629a1ae763010dc3035bb8da3ae6cf026486401",
28
+ "source": "docker build log",
29
+ "local_tag_matches": null,
30
+ "note": "not in the local image store (BuildKit pulled it into its cache)"
31
+ }
32
+ },
33
+ "tt_metal": {
34
+ "sha": "44d66500520fda9f2c7060c0f6b41ec48f7ab37e",
35
+ "describe": "v0.80.0-dev20261006-78-g44d6650052-dirty",
36
+ "dirty": true,
37
+ "scm_version": "0.65.2.dev11169+g44d66500520",
38
+ "mode": "local",
39
+ "remote": "https://github.com/tenstorrent/tt-metal.git",
40
+ "branch": "main",
41
+ "pushed": true
42
+ },
43
+ "tools": {
44
+ "tt_model": "0.1.0",
45
+ "docker": "29.8.2"
46
+ },
47
+ "notes": "tt-model's FROM tags float (build: tt-metalium dev image :latest, runtime: ubuntu:<version>); these are the digests this image was built from. A moved build base costs a cold C++ build."
48
+ }
code/PYTHON.md ADDED
@@ -0,0 +1,165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Python API: METEOR (TIER IV, AutowareFoundation/meteor) on Blackhole
2
+
3
+ Use this API from Python code (a pipeline, a notebook, a ROS 2 node wrapper). You do not need the HTTP server: the
4
+ API and the server share the decoders, the device graph and the post-processing, so the outputs and the speed are
5
+ the same.
6
+
7
+ ## Install
8
+
9
+ Install the package on top of an environment that already has `ttnn`: a tt-metal `python_env` at `44d66500520`
10
+ with `patches/tt-metal-eth-dispatch.patch` and `patches/tt-metal-reshape-rm-sys1419.patch` applied, or the tt-model
11
+ container. From the root of the model repository (the directory that holds `pyproject.toml`, `README.md` and
12
+ `code/`):
13
+
14
+ ```bash
15
+ pip install -e . # the Python API (numpy<2, pillow, pyyaml, onnx, huggingface_hub, safetensors, opencv-python-headless)
16
+ pip install -e ".[server,test]" # + the HTTP server and the tests
17
+ ```
18
+
19
+ The pip project is the repository's top-level `pyproject.toml`; it installs the package from `code/tt_meteor` (there
20
+ is no `pyproject.toml` inside `code/`, because the container build copies `code/` over the tt-metal tree). `ttnn`
21
+ and `torch` are not declared, so pip never replaces tt-metal's own build; tt-metal's `python_env` already has
22
+ OpenCV 4.8.1, which satisfies the `opencv-python-headless>=4.8,<4.12` requirement.
23
+
24
+ The package carries `tt_meteor.ttaw`, the shared code of the Autoware ports to Blackhole (device open, trace runner,
25
+ decoders, model base class, HTTP app), vendored at the version recorded in `code/tt_meteor/ttaw/VENDORED.json`
26
+ (0.23.2).
27
+
28
+ | You want to run | Extras |
29
+ |---|---|
30
+ | the Python API | none |
31
+ | the HTTP server (`tt_meteor.server.app`, see `SERVING.md`) | `server` |
32
+ | host tests (no device; the device tests are skipped): `TT_VISIBLE_DEVICES=none python -m pytest -q code/tt_meteor/tests` | `server,test` |
33
+ | device tests: `python -m pytest -q -s code/tt_meteor/tests/test_pcc_device.py code/tt_meteor/tests/test_e2e_device.py code/tt_meteor/tests/test_variants_device.py` (most of them need the fp32 goldens of the development workspace and skip without them) | `test` |
34
+
35
+ ## Quickstart
36
+
37
+ ```python
38
+ from tt_meteor import METEOR, load_sample
39
+
40
+ with METEOR.from_pretrained(device_id=0) as model:
41
+ out = model(**load_sample("code/tt_meteor/samples/synthetic_8cam.json")) # 8 cameras + calibration + ego speed + stream
42
+ print(out.to_dict()["plan"]["mode"], [d["label"] for d in out.to_dicts()])
43
+ ```
44
+
45
+ `examples/quickstart.py` runs the same snippet and writes the `/predict` JSON and a bird's-eye view
46
+ (`quickstart_bev.png`). The shipped sample is a synthetic test frame generated by this repository (see
47
+ `code/tt_meteor/samples/README.md`); feed your own rig's eight images and calibration for real use.
48
+
49
+ ## `METEOR.from_pretrained(...)`
50
+
51
+ ```python
52
+ METEOR.from_pretrained(
53
+ model_id=None, # HF repo or a local directory with the weights files; default AutowareFoundation/meteor
54
+ *,
55
+ revision=None, # default for the default repo: the validated commit 01a5f6d71df (tag v1.0)
56
+ variant=None, # load-time variant: "default" (the only one); default $METEOR_VARIANT
57
+ device_id=None, # chip to open; default $TT_DEVICE_ID or 0
58
+ device=None, # an already-opened ttnn device (tt_meteor.device.open_device); close() does not close it
59
+ dispatch=None, # "eth" (p150 target, 12x10 grid) | "worker" (A/B only, 11x10) | "auto"; default $METEOR_DISPATCH or "eth"
60
+ num_command_queues=None, # default $METEOR_NUM_CQS or 1 (D14: a second queue gains nothing for synchronous calls)
61
+ weights_dir=None, # explicit local weights directory; no Hub access
62
+ warmup_variants="default", # trace variants to capture now; see "Warm-up"
63
+ verbose=False,
64
+ **compile_params, # load-time knobs (below), e.g. input_norm="imagenet"
65
+ ) -> METEOR
66
+ ```
67
+
68
+ What it does: resolves the weights first, so a Hub problem never claims the chip (`weights_dir` >
69
+ `$METEOR_WEIGHTS_DIR` > a local `model_id` directory > the HF snapshot at the pinned revision, restricted to
70
+ `meteor_v157c3Z.onnx, meteor_v157.param.yaml, LICENSE, SHA256SUMS`, with an offline fallback to the cache), opens the
71
+ chip (ETH dispatch, 12×10, 1 CQ; the other open parameters are `DEVICE_DEFAULTS` in `tt_meteor/device.py`:
72
+ `l1_small_size` 32 KiB, `trace_region_size` 256 MiB, overridable with `METEOR_*`), reads the parameters of the ONNX
73
+ file by consuming node, builds the graph, then compiles and captures the metal trace. If ETH dispatch cannot open
74
+ (tt-metal without the patch), it warns and falls back to WORKER dispatch.
75
+
76
+ Load-time knobs (`compile_params` in lower case, or the environment variable `METEOR_<NAME>`; read once at load):
77
+
78
+ | knob | default | meaning |
79
+ |---|---|---|
80
+ | `input_norm` | `onnx` | `onnx` = the released graph's `/255` (ONNX Runtime parity; decision D12); `imagenet` = the trained ImageNet mean / std (upstream export defect, research/meteor SPEC section 10 risk 1); with `imagenet` an absent camera is zeroed after normalising, as in training |
81
+ | `depth_mean_bins` | `log` | `depth_mean` bin centres: log-spaced as exported (parity) or `linear` 1 + 1.25 b m (the trained bins) |
82
+ | `max_streams` | 16 | host temporal states kept (seg fusion, yaw tracks, mode hysteresis); the least recently used is dropped |
83
+ | `image_precision` | `terms3` | image branch with fp32 activations, every conv as three bf16 terms (needed by the depth / seg2d argmax gates); `bf16` is an A/B only: 166 ms faster per frame, but below the depth argmax gate (0.9797 < 0.99) |
84
+
85
+ ## Warm-up
86
+
87
+ The load runs one eager frame of the whole graph (it compiles every kernel and fills the program cache) and then
88
+ captures it as ONE metal trace, `frame` (4,678 programs). The first call is therefore as fast as the later ones.
89
+ With a warm JIT cache the load takes about 45-50 s (49.7 s measured: weights 0.3 s, graph build 2.6 s, eager frame
90
+ 41.0 s, capture 1.2 s). A cold kernel cache compiles for minutes (a new dispatch configuration on the build host:
91
+ ETH-2CQ 142 s, WORKER 11×10 768 s). `warmup_variants="none"` skips the capture (the first call then captures).
92
+
93
+ ## Call: `model(...)`
94
+
95
+ | Argument | Type | Description |
96
+ |---|---|---|
97
+ | `images` | list or mapping | the cameras `CAM_FRONT_WIDE`, `CAM_FRONT_LEFT`, `CAM_FRONT_RIGHT`, `CAM_BACK_WIDE`, `CAM_BACK_LEFT`, `CAM_BACK_RIGHT`, `CAM_FRONT_NARROW`, `CAM_BACK_NARROW`, any order: `CameraImage`s, dicts `{"camera", "image", "intrinsics", "T_ref_from_camera"}` or `{name: image}`. Raw, unrectified frames of at least 768×432; larger frames are resized with OpenCV INTER_AREA semantics (bit-exact numpy port) and K is scaled per axis. The six wide / corner cameras are required; a missing or all-zero narrow camera is **absent**: a zero image plus its donor's K and pose (`CAM_FRONT_WIDE` / `CAM_BACK_WIDE`), the trained 7-camera configuration |
98
+ | `calibration` | dict | `{"cameras": {name: {"intrinsics", "T_ref_from_camera"}}}` (K of the image as sent; camera optical frame -> base_link: x forward, y left, z up, origin on the road) or `{"preset": name}` (`code/tt_meteor/calib/`); inline calibration of a camera wins |
99
+ | `ego_speed` | float | m/s, the graph's `v0` (required) |
100
+ | `stream` | dict | optional: `{"id", "reset", "timestamp_s", "T_world_from_ego"}` (or `"pose": [x, y, yaw]`): METEOR's host temporal post-processing per stream id (BEV seg fusion needs the pose; yaw smoothing; plan-mode hysteresis). A new id or `reset=True` starts fresh |
101
+ | runtime params | keyword | host post-processing only (`METEOR.RUNTIME_PARAMS`, METEOR's C++ renderer defaults): 3D `det3d_threshold` 0.15, `det3d_topk` 64, `vehicle_threshold` 0.35, `vru_threshold` 0.15, `bev_nms_iou` 0.3, `bev_nms_containment` 0.6, `stationary_logit_threshold` 0.0; 2D `det2d_threshold` 0.30, `det2d_topk` 48, `det2d_hide` "7" (road paint); `unk2d` True, `unk2d_threshold` (= det2d), `ground_z` 0.0; plan `mode_hysteresis` 0.35, `straight_margin` 1.0; `seg_fuse` True, `thin_road_edge` True, `yaw_smoothing` True; `heads` False (adds the dense heads to `to_dict("npz")`) |
102
+
103
+ `tt_meteor.load_sample(path)` turns a sample manifest (`samples/<name>.json`: image paths relative to the file, a
104
+ preset, the ego speed and a stream) into these keyword arguments.
105
+
106
+ ### Input types
107
+
108
+ - Images: a path, PNG / JPEG bytes, a `PIL.Image`, a uint8 H×W×3 RGB array, or a float array in [0, 1].
109
+ - Transforms: 4×4 (or 3×4) matrices, `{"translation", "rotation_wxyz"}` (or `rotation_xyzw`), or Autoware
110
+ `{"x", "y", "z", "roll", "pitch", "yaw"}`. Intrinsics: 3×3 K, 3×4 P or `{"fx", "fy", "cx", "cy"}`.
111
+
112
+ ## Output
113
+
114
+ `model(...)` returns a `MeteorOutput` (`tt_meteor.Output`); `out.to_dict()` is the `/predict` JSON:
115
+
116
+ | key | content |
117
+ |---|---|
118
+ | `detections` (`out.to_dicts()`) | 3D BEV boxes sorted by score: `label` (VEHICLE / VRU), `label_id`, `score`, `center` [x, y] and `size` [length, width] in metres (base_link; METEOR predicts no z, height or velocity), `yaw` (rad, CCW from +x), `stationary`, `future` (6 × [x, y] at 0.5 .. 3 s, the agent's best mode), `future_mode` |
119
+ | `trajectory`, `columns` | the selected ego path, 6 × [x, y] at 0.5 s steps |
120
+ | `plan` | `mode`, `mode_probs` (softmax of the hysteresis-adjusted logits), `mode_logits`, `paths` (3 × 6 × [x, y]), `steer` (rad), `accel` (m/s²), `brake_prob`, `dt` |
121
+ | `detections_2d` | per camera: `label` (10 classes), `label_id`, `score`, `box_xyxy` (pixels at 768×432) |
122
+ | `unknown_obstacles` | [x, y] of 2D "obstacle" boxes placed on the ground plane (METEOR's `unk2d`) |
123
+ | `traffic_light` | `state` (none / green / yellow / red), `state_id`, `probs` |
124
+ | `lane`, `lane_classes` | the BEV lane map, uint8 [800, 500] as PNG (0.2 m cells, row 0 = +80 m ahead, column 0 = +50 m left; classes bg, road, sidewalk, crosswalk, laneline, stopline, road_edge, marking, parking), after the optional seg fusion and road-edge thinning |
125
+ | `stationary_head_healthy`, `meta` | METEOR's health check of the stationary head; `present` cameras, `v0`, `stream_id` |
126
+ | `heads` (`to_dict("npz")` with `heads=True`) | the dense outputs: `seg2d`, `depth` (bins), `depth_mean`, `occupancy` (class per voxel), `risk` (sigmoid), `stationary` |
127
+ | `timing_ms` | `preprocess`, `device` (tables + upload + replay + read + conversion), `postprocess`, `total` |
128
+
129
+ `out.path` is the selected path as an array; `out.lane` the lane map; `out.boxes3d` / `out.boxes2d` the box objects.
130
+
131
+ ## Lifetime and information
132
+
133
+ - `model.close()` releases the trace and the device tensors and closes the chip if the model opened it; idempotent.
134
+ `with` calls it for you; an unclosed model is closed when Python exits.
135
+ - `model.info`: weights (repo, tag, revision, path), device (dispatch, grid), variant, warm variants, warm-up times,
136
+ runtime parameter defaults, the load-time knobs and the trace description.
137
+ - Calls from several threads are safe: the device calls are serialised. One model per process per chip.
138
+ - **Rig changes.** The lift tables (205 MB of fp32 bin weights + the sampling grid) are built on the host per
139
+ calibration (cached per rig) and written to the chip only when the calibration changes: 182 ms for the write, so a
140
+ call with a new rig takes about 1.13 s instead of about 0.9 s. A fixed rig pays it once.
141
+
142
+ ## Speed
143
+
144
+ Warm, batch 1, ETH dispatch, 1 CQ, 12×10, AICLK 1350 MHz, p50 of 60 calls on a shared 8-core host
145
+ ([`OPT_BASELINE.md`](../OPT_BASELINE.md); the device rows do not depend on the input):
146
+
147
+ | stage | ms |
148
+ |---|---:|
149
+ | device trace, one replay (the whole network) | 556.85 (back to back 556.75: 1.80 frames/s) |
150
+ | upload of the eight cameras (8 MB uint8) | 66.5 |
151
+ | readback (75 MB, 5 segments) + conversion to the 19 outputs | 17-48 + 65 |
152
+ | host pre-processing / camera transpose | 9-11 / 22-26 |
153
+ | host post-processing (METEOR's C++ decode rules, temporal state) | 124-131 |
154
+ | `model(...)` end to end | 876-918 (1.09-1.14 calls/s) |
155
+
156
+ This is the first release: optimization has not started ([`OPT_REPORT.md`](../OPT_REPORT.md)).
157
+
158
+ ## Limits
159
+
160
+ - Batch 1 on the chip; one model per process; eight 768×432 camera slots in METEOR's order, a 400×250 lift grid at
161
+ 0.4 m and an 800×500 BEV at 0.2 m (fixed shapes of the released graph); at most 3 cameras per lift cell on the
162
+ rigs validated.
163
+ - The released graph is camera-only and single-frame (its temporal memory was baked out upstream); the temporal
164
+ behaviour of METEOR's runtime is host post-processing.
165
+ - Numerics: bf16 / fp32 on the chip; outputs differ slightly from the fp32 reference (README "Demo & Performances").
code/conftest.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """pytest fixtures of meteor-p150 (no dependency on tt-metal's own conftest).
3
+
4
+ - ``device`` (session): one chip opened like the published numbers (``tt_meteor.device.open_device``:
5
+ ETH dispatch, 12x10 grid, the port's validated sizes and CQs). ``--device-id N`` or ``TT_DEVICE_ID`` selects
6
+ the chip (default 0); ``METEOR_DISPATCH=worker`` is the A/B opt-in, and the other ``METEOR_*``
7
+ variables apply as for the server. A failing ETH open is an error here, never a silent WORKER fallback: gates
8
+ are only valid on the published setup.
9
+ - Tests marked ``device`` are skipped when ttnn is missing or ``TT_VISIBLE_DEVICES=none`` (host-only runs).
10
+ On the shared workspace box run them through the lock and name the test files:
11
+ ``bin/devrun -t 1800 -- python -m pytest -q -s code/tt_meteor/tests/test_pcc_device.py``.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import gc
16
+ import importlib.util
17
+ import os
18
+
19
+ import pytest
20
+
21
+
22
+ def pytest_addoption(parser):
23
+ parser.addoption("--device-id", action="store", default=None, help="chip id (default $TT_DEVICE_ID or 0)")
24
+
25
+
26
+ def pytest_configure(config):
27
+ config.addinivalue_line("markers", "device: needs a Tenstorrent chip (skipped on host-only runs)")
28
+
29
+
30
+ def _no_device_reason():
31
+ if os.environ.get("TT_VISIBLE_DEVICES", "").lower() == "none":
32
+ return "TT_VISIBLE_DEVICES=none (host-only run)"
33
+ if importlib.util.find_spec("ttnn") is None:
34
+ return "ttnn is not installed"
35
+ return None
36
+
37
+
38
+ def pytest_collection_modifyitems(config, items):
39
+ reason = _no_device_reason()
40
+ if reason:
41
+ skip = pytest.mark.skip(reason=reason)
42
+ for item in items:
43
+ if "device" in item.keywords:
44
+ item.add_marker(skip)
45
+
46
+
47
+ @pytest.fixture(autouse=True)
48
+ def _gc_between_tests():
49
+ gc.collect()
50
+
51
+
52
+ @pytest.fixture(scope="session")
53
+ def device(request):
54
+ from tt_meteor.device import close_device, open_device
55
+
56
+ cli = request.config.getoption("--device-id")
57
+ dev = open_device(int(cli) if cli is not None else None, allow_fallback=False)
58
+ yield dev
59
+ close_device(dev)
code/models/common/lightweightmodule.py ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-FileCopyrightText: © 2023 Tenstorrent USA, Inc.
2
+
3
+ # SPDX-License-Identifier: Apache-2.0
4
+
5
+
6
+ class LightweightModule:
7
+ """Torch modules add a surprising amount of host overhead for attribute
8
+ access and method calls. This class is a lightweight alternative that
9
+ just wraps a forward function for now."""
10
+
11
+ def __call__(self, *args, **kwargs):
12
+ return self.forward(*args, **kwargs)
code/scripts/README.md ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # code/scripts
2
+
3
+ | script | purpose | device |
4
+ |---|---|---|
5
+ | `fetch_samples.sh` | downloads the public sample data that is not redistributable (sha256-checked) | no |
6
+ | `bench.py` | stage breakdown of warm forwards on several inputs (API split, host_in / H2D / trace / b2b / D2H, e2e p50 / p99, AICLK, the smoke agreement check; `--dispatch` / `--num-cqs` for the D14 matrix, `--rig-switch`): the card / OPT numbers | yes, via `bin/devrun` |
7
+ | `profile_ops.py` | Tracy + device profile: one eager frame with module signposts and `--rounds` signposted traced replays (`python -m tracy -r -p --op-support-count 6000 ...`); `--graph-report DIR` writes a ttnn-visualizer graph report of one eager frame instead | yes, via `bin/devrun` |
8
+ | `profile_summary.py` | summary of a `profile_ops.py` ops CSV: programs, kernel sum, span, op-to-op gaps, per-module / per-op-code tables (eager labels carried to the trace op for op), top ops | no |
9
+ | `container_smoke.sh` | serve ONE serve profile of the built package (`--profile NAME`; default profile otherwise), run `server/smoke_test.py` against it (asserts ETH dispatch, the 12x10 grid and the profile's pins; compares with the stored CPU reference), keep the evidence (container log, `/info`, the `/predict` output and the result, in `logs/smoke/` of the repo or `--log-dir DIR`), always stop it; one `bin/devrun -t 3600 -k 150` window per profile | yes |
10
+ | `make_synthetic_sample.py` | the shipped sample `samples/synthetic_8cam*` (generated, Apache-2.0): `render` (a ray-cast street seen by a generic METEOR-like rig, preset `calib/synthetic_8cam.json`), `optimise` (gradient ascent on the object pixels through the fp32 CPU reference until the 3D heads report the objects with margin), `finalise` (PNGs, request manifest, the stored CPU-reference `/predict` body, the feed sha256 golden) | no (research venv) |
11
+ | `make_sample.py` | builds the PandaSet 019 frame 40 sample (METEOR demo-scene layout, request manifest, preset `pandaset_019`) from the research ship sample; `--dest staging` (default: the git-ignored `staging_samples_pandaset/`, not shipped until the user approves PandaSet samples) or `--dest package` | no |
12
+ | `make_goldens.py` | CPU-reference goldens (taps + outputs of 4 frames under `research/meteor/goldens/`, never shipped), the small decoded golden `tests/goldens/pandaset_019_f40_reference.json` and the reference body `samples/pandaset_019_f40.reference.json` of the PandaSet sample (in the staging dir while it is not shipped); cross-checked against the research ONNX Runtime goldens (research venv) | no |
13
+ | `stress_frames.py` | contained stress of the `frame` trace (the ETH-dispatch hang investigation): `--mode replay` / `rigs` / `segrigs`; every read bit-compared per rig; `--rigs shipped[:N]` needs no golden (rigs from the shipped sample: what the container check runs) | yes, via `bin/devrun` |
14
+ | `run_public_frames.py` | the served model on the public-dataset frames of `research/meteor/public_data` (feed sha256 checked), TT outputs written in the CPU-golden layout for `compare_tt.py` and `make_demo.py` (development workspace only) | yes, via `bin/devrun` |
15
+ | `make_demo.py` | the card's `media/` from TT outputs: cards, camera heads and the animation of the public frames (research renderer: privacy blur, attribution footers), the TT-vs-CPU bird's-eye view, the shipped-sample render, `media/ATTRIBUTION.md` | no (research venv) |
16
+ | `dump_device_outputs.py`, `device_stage_check.py`, `precision_emulation.py`, `repro_*_eth.py` | port tools: raw device outputs of golden frames, eager stage checks, CPU emulation of precision policies, the standalone ETH-hang repros (PORT_LOG 12.7) | partly |
17
+
18
+ Everything here ships in `code/` on the Hub and inside the image (`source.extra_code` lists `scripts`).
code/scripts/bench.py ADDED
@@ -0,0 +1,317 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Stage bench of warm forwards: the numbers OPT_BASELINE.md, OPT_REPORT.md and the card quote.
4
+
5
+ bin/devrun -t 1800 -- python code/scripts/bench.py --iters 60 --inputs sample,synthetic --json out.json
6
+ bin/devrun -t 1800 -- python code/scripts/bench.py --dispatch worker --num-cqs 2 --json ... # the D14 matrix
7
+
8
+ Inputs (``--inputs``, comma separated; each one is a request of ``model(...)``, batch 1):
9
+
10
+ - ``sample``: the PandaSet 019 frame 40 sample ``pandaset_019_f40.json`` (the baseline's sample: the git-ignored
11
+ ``staging_samples_pandaset/`` copy until the user approves shipping it, else ``samples/``);
12
+ - ``shipped``: the shipped synthetic sample ``samples/synthetic_8cam.json`` (``make_synthetic_sample.py``);
13
+ - ``synthetic``: eight 768x432 uint8 noise images made here (seeded; data generated by us), the ``pandaset_019``
14
+ calibration (inline from the staged preset; the ``synthetic_8cam`` preset when it is absent), ego speed 10 m/s;
15
+ - ``<name>=<manifest.json>`` or a manifest path: any ``load_sample`` manifest (e.g. a public-dataset frame).
16
+
17
+ Images are decoded when the request is loaded, before timing (the camera driver hands METEOR decoded images). Per
18
+ input, after ``--warm`` calls (``ttaw.profiling.StageBench``: p50 / p99 / mean / min / max in ms):
19
+
20
+ 1. **plain**: ``model(...)`` exactly as a client calls it: ``e2e`` and the API's own ``timing_ms``
21
+ (``api.preprocess`` / ``api.device`` / ``api.postprocess``), with ``run_frame`` split by wrapping its parts:
22
+ ``dev.tables`` (rig check, and the lift-table write when the rig changed), ``dev.image_rows`` (NCHW -> NHWC rows),
23
+ ``dev.runner`` (upload + replay + segmented read), ``dev.unpack`` (join of the readback segments),
24
+ ``dev.outputs`` (``outputs_from_device``: device layouts -> the 19 ONNX outputs on the host);
25
+ 2. **stages**: ``ttaw.profiling.bench_trace_runner`` on the ``frame`` variant: ``host_in`` (numpy -> ttnn host
26
+ tensor), ``h2d`` (upload + synchronize), ``trace`` (one replay + synchronize), ``d2h`` (segmented read), ``post``
27
+ (``unpack``), ``e2e`` (runner call + unpack) and ``b2b`` (back-to-back replays: the device time per frame).
28
+
29
+ ``--rig-switch N``: N calls alternating between the first two inputs (each call writes its rig's lift tables).
30
+ An agreement check of every input with a stored CPU reference (``<stem>.reference.json``, fresh stream, the container
31
+ smoke's ``compare_with_reference``) runs first, so every configuration's numbers come with its accuracy. AICLK /
32
+ power / temperature are sampled during the loops (``AiclkSampler``).
33
+ """
34
+ from __future__ import annotations
35
+
36
+ import argparse
37
+ import json
38
+ import os
39
+ import platform
40
+ import subprocess
41
+ import sys
42
+ import time
43
+ from pathlib import Path
44
+ from typing import Any, Dict, List, Optional, Tuple
45
+
46
+ import numpy as np
47
+
48
+ HERE = Path(__file__).resolve()
49
+ CODE = HERE.parents[1]
50
+ sys.path.insert(0, str(CODE))
51
+ SAMPLES = CODE / "tt_meteor" / "samples"
52
+ STAGING = CODE.parent / "staging_samples_pandaset"
53
+ ROOT = Path(os.environ.get("TT_MODELS_ROOT", "/home/ubuntu/experiments/tt-models"))
54
+ SAMPLE_NAME = "pandaset_019_f40.json"
55
+
56
+
57
+ # ------------------------------------------------------------------------------------------------- inputs
58
+ def sample_path() -> Path:
59
+ for d in (SAMPLES, STAGING / "samples", STAGING):
60
+ if (d / SAMPLE_NAME).is_file():
61
+ return d / SAMPLE_NAME
62
+ return SAMPLES / SAMPLE_NAME
63
+
64
+
65
+ def synthetic_request(seed: int = 1) -> Dict[str, Any]:
66
+ from tt_meteor.reference import config as C
67
+
68
+ rng = np.random.default_rng(seed)
69
+ images = {cam: rng.integers(0, 256, size=(C.IMG_H, C.IMG_W, 3), dtype=np.uint8) for cam in C.CAMERAS}
70
+ staged = [d / "pandaset_019.json" for d in (CODE / "tt_meteor" / "calib", STAGING / "calib") if
71
+ (d / "pandaset_019.json").is_file()]
72
+ calib = json.loads(staged[0].read_text()) if staged else {"preset": "synthetic_8cam"}
73
+ return {"images": images, "calibration": calib, "ego_speed": 10.0,
74
+ "stream": {"id": "synthetic", "pose": [0.0, 0.0, 0.0]}}
75
+
76
+
77
+ def load_input(spec: str) -> Tuple[str, Dict[str, Any], Optional[Path]]:
78
+ from tt_meteor import load_sample
79
+
80
+ if spec == "synthetic":
81
+ return "synthetic", synthetic_request(), None
82
+ if spec == "sample":
83
+ path = sample_path()
84
+ return "sample", load_sample(path), path
85
+ if spec == "shipped":
86
+ path = SAMPLES / "synthetic_8cam.json"
87
+ return "shipped", load_sample(path), path
88
+ name, _, p = spec.partition("=")
89
+ path = Path(p or name)
90
+ return (name if p else path.stem), load_sample(path), path
91
+
92
+
93
+ def find_reference(manifest: Optional[Path]) -> Optional[Path]:
94
+ if manifest is None:
95
+ return None
96
+ p = manifest.parent / f"{manifest.stem}.reference.json"
97
+ return p if p.is_file() else None
98
+
99
+
100
+ # --------------------------------------------------------------------------------------------- environment
101
+ def _git(*args: str) -> str:
102
+ try:
103
+ return subprocess.run(["git", *args], capture_output=True, text=True,
104
+ env=dict(os.environ, GIT_OPTIONAL_LOCKS="0")).stdout.strip()
105
+ except OSError:
106
+ return ""
107
+
108
+
109
+ def environment() -> Dict[str, Any]:
110
+ env: Dict[str, Any] = {"host": platform.node(), "python": platform.python_version(),
111
+ "loadavg_start": os.getloadavg(), "cpus": os.cpu_count(),
112
+ "omp_num_threads": os.environ.get("OMP_NUM_THREADS"),
113
+ "env": {k: v for k, v in os.environ.items() if k.startswith("METEOR_")}}
114
+ try:
115
+ env["cpu_model"] = next(line.split(":", 1)[1].strip() for line in open("/proc/cpuinfo")
116
+ if line.startswith("model name"))
117
+ except (OSError, StopIteration):
118
+ pass
119
+ tm = ROOT / "tt-metal"
120
+ env["tt_metal_commit"] = _git("-C", str(tm), "rev-parse", "HEAD")
121
+ try:
122
+ vend = json.loads((CODE / "tt_meteor/ttaw/VENDORED.json").read_text())
123
+ env["ttaw"] = vend.get("version")
124
+ env["ttaw_vendored"] = vend.get("source_commit")
125
+ except (OSError, ValueError):
126
+ pass
127
+ env["bundle_commit"] = _git("-C", str(CODE.parent), "rev-parse", "HEAD")
128
+ env["bundle_dirty"] = bool(_git("-C", str(CODE.parent), "status", "--porcelain", "code"))
129
+ return env
130
+
131
+
132
+ def cache_entries(device) -> Optional[int]:
133
+ fn = getattr(device, "num_program_cache_entries", None)
134
+ try:
135
+ return int(fn()) if fn else None
136
+ except Exception: # noqa: BLE001
137
+ return None
138
+
139
+
140
+ # ------------------------------------------------------------------------------------------------- measure
141
+ class Split:
142
+ """Times the parts of ``TtMETEOR.run_frame`` (module docstring) while ``bench`` is set."""
143
+
144
+ def __init__(self, model):
145
+ import tt_meteor.tt.model as tmodel
146
+
147
+ self.bench = None
148
+ tt = model.tt
149
+ self._undo = []
150
+
151
+ def timed(name, fn):
152
+ def wrapper(*a, **k):
153
+ if self.bench is None:
154
+ return fn(*a, **k)
155
+ t0 = time.perf_counter()
156
+ try:
157
+ return fn(*a, **k)
158
+ finally:
159
+ self.bench.add(name, (time.perf_counter() - t0) * 1e3)
160
+ return wrapper
161
+
162
+ for attr, name in (("set_tables", "dev.tables"), ("image_rows", "dev.image_rows"), ("unpack", "dev.unpack"),
163
+ ("runner", "dev.runner")):
164
+ orig = getattr(tt, attr)
165
+ setattr(tt, attr, timed(name, orig) if attr != "runner" else _CallTimer(orig, self, name))
166
+ self._undo.append(lambda tt=tt, attr=attr, orig=orig: setattr(tt, attr, orig))
167
+ orig_out = tmodel.outputs_from_device
168
+ tmodel.outputs_from_device = timed("dev.outputs", orig_out)
169
+ self._undo.append(lambda: setattr(tmodel, "outputs_from_device", orig_out))
170
+
171
+ def remove(self) -> None:
172
+ for fn in reversed(self._undo):
173
+ fn()
174
+
175
+
176
+ class _CallTimer:
177
+ """Stands in for the TraceRunner inside ``run_frame``: times ``runner(...)``, forwards every attribute."""
178
+
179
+ def __init__(self, inner, split: Split, name: str):
180
+ self._inner, self._split, self._name = inner, split, name
181
+
182
+ def __call__(self, *a, **k):
183
+ if self._split.bench is None:
184
+ return self._inner(*a, **k)
185
+ t0 = time.perf_counter()
186
+ try:
187
+ return self._inner(*a, **k)
188
+ finally:
189
+ self._split.bench.add(self._name, (time.perf_counter() - t0) * 1e3)
190
+
191
+ def __getattr__(self, item):
192
+ return getattr(self._inner, item)
193
+
194
+
195
+ def agreement(model, name: str, req: Dict[str, Any], ref_path: Path) -> Dict[str, Any]:
196
+ sys.path.insert(0, str(CODE / "tt_meteor" / "ttaw" / "server"))
197
+ import smoke as ttaw_smoke # noqa: E402
198
+
199
+ fresh = dict(req)
200
+ fresh["stream"] = dict(req.get("stream") or {}, id=f"agreement-{name}-{time.time_ns()}")
201
+ body = model(**fresh).to_dict()
202
+ ref = json.loads(ref_path.read_text())
203
+ metrics, failures = ttaw_smoke.compare_with_reference(body, ref)
204
+ path_dev = float(np.linalg.norm(np.asarray(body["trajectory"]) - np.asarray(ref["trajectory"]), axis=-1).max())
205
+ return {"reference": str(ref_path), "metrics": metrics, "failures": failures, "ego.path_dev_m": path_dev,
206
+ "pass": not failures}
207
+
208
+
209
+ def bench_input(model, split: Split, name: str, req: Dict[str, Any], iters: int, warm: int,
210
+ b2b: int) -> Dict[str, Any]:
211
+ from tt_meteor.host.inputs import prepare_request
212
+ from tt_meteor.ttaw.profiling import AiclkSampler, StageBench, bench_trace_runner
213
+
214
+ res: Dict[str, Any] = {"loadavg_start": os.getloadavg()}
215
+ for _ in range(warm):
216
+ model(**req)
217
+ plain = StageBench(f"meteor {name} plain")
218
+ with AiclkSampler(interval_s=0.05) as clk_p:
219
+ split.bench = plain
220
+ try:
221
+ for _ in range(iters):
222
+ t0 = time.perf_counter()
223
+ out = model(**req)
224
+ plain.add("e2e", (time.perf_counter() - t0) * 1e3)
225
+ for k, v in out.timing_ms.items():
226
+ plain.add(f"api.{k}", v)
227
+ finally:
228
+ split.bench = None
229
+ res["plain"] = plain.summary()
230
+ res["aiclk_plain"] = clk_p.summary()
231
+ res["boxes3d"] = len(out.boxes3d)
232
+ print(plain.table(), flush=True)
233
+ frame = prepare_request(req["images"], req["calibration"], req["ego_speed"], req.get("stream"))
234
+ model.tt.set_tables(model.calib_cache.get(frame.K, frame.T_cam_ego))
235
+ tt = model.tt
236
+ with AiclkSampler(interval_s=0.05) as clk_s:
237
+ stages = bench_trace_runner(model.runner, "frame", {"imgs": tt.image_rows(frame.imgs)},
238
+ params=tt.frame_params(frame), iters=iters, warmup=warm,
239
+ name=f"meteor {name} stages", post=tt.unpack, b2b_iters=b2b)
240
+ res["stages"] = stages.summary()
241
+ res["aiclk_stages"] = clk_s.summary()
242
+ print(stages.table(), flush=True)
243
+ res["loadavg_end"] = os.getloadavg()
244
+ return res
245
+
246
+
247
+ def rig_switch(model, split: Split, reqs: List[Tuple[str, Dict[str, Any]]], n: int) -> Dict[str, Any]:
248
+ from tt_meteor.ttaw.profiling import StageBench
249
+
250
+ bench = StageBench("meteor rig switch")
251
+ split.bench = bench
252
+ try:
253
+ for i in range(n):
254
+ name, req = reqs[i % 2]
255
+ t0 = time.perf_counter()
256
+ out = model(**req)
257
+ bench.add("e2e", (time.perf_counter() - t0) * 1e3)
258
+ for k, v in out.timing_ms.items():
259
+ bench.add(f"api.{k}", v)
260
+ finally:
261
+ split.bench = None
262
+ print(bench.table(), flush=True)
263
+ return {"inputs": [r[0] for r in reqs[:2]], "calls": n, **{"stages": bench.summary()}}
264
+
265
+
266
+ def main() -> int:
267
+ ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
268
+ ap.add_argument("--inputs", default="sample", help="comma list: sample | synthetic | <name>=<manifest> | path")
269
+ ap.add_argument("--iters", type=int, default=60)
270
+ ap.add_argument("--warm", type=int, default=3)
271
+ ap.add_argument("--b2b", type=int, default=20, help="back-to-back replays of the device-only row")
272
+ ap.add_argument("--rig-switch", type=int, default=0, help="calls alternating between the first two inputs")
273
+ ap.add_argument("--dispatch", default=None, choices=["eth", "worker"])
274
+ ap.add_argument("--num-cqs", type=int, default=None, choices=[1, 2])
275
+ ap.add_argument("--no-agreement", dest="agreement", action="store_false")
276
+ ap.add_argument("--json")
277
+ a = ap.parse_args()
278
+ from tt_meteor import METEOR
279
+
280
+ inputs = [load_input(s.strip()) for s in a.inputs.split(",") if s.strip()]
281
+ res: Dict[str, Any] = {"args": vars(a), "environment": environment(),
282
+ "inputs": {n: str(p) if p else "synthetic" for n, _, p in inputs}}
283
+ t_load = time.perf_counter()
284
+ with METEOR.from_pretrained(dispatch=a.dispatch, num_command_queues=a.num_cqs) as model:
285
+ res["load_s"] = time.perf_counter() - t_load
286
+ res["warmup_ms"] = model.warmup_ms
287
+ res["config"] = model.device_info
288
+ res["program_cache_entries"] = cache_entries(model.device)
289
+ res["trace"] = model.runner.describe()
290
+ print("config:", json.dumps(res["config"]), "| load %.1f s" % res["load_s"], "| programs:",
291
+ res["program_cache_entries"], flush=True)
292
+ if a.agreement:
293
+ res["agreement"] = {}
294
+ for name, req, path in inputs:
295
+ ref = find_reference(path)
296
+ if ref is not None:
297
+ res["agreement"][name] = agreement(model, name, req, ref)
298
+ print("agreement", name, json.dumps(res["agreement"][name], default=str), flush=True)
299
+ split = Split(model)
300
+ res["per_input"] = {}
301
+ for name, req, _ in inputs:
302
+ res["per_input"][name] = bench_input(model, split, name, req, a.iters, a.warm, a.b2b)
303
+ if a.rig_switch and len(inputs) >= 2:
304
+ res["rig_switch"] = rig_switch(model, split, [(n, r) for n, r, _ in inputs], a.rig_switch)
305
+ split.remove()
306
+ res["program_cache_entries_end"] = cache_entries(model.device)
307
+ res["loadavg_end"] = os.getloadavg()
308
+ text = json.dumps(res, indent=1, default=str)
309
+ if a.json:
310
+ Path(a.json).parent.mkdir(parents=True, exist_ok=True)
311
+ Path(a.json).write_text(text + "\n")
312
+ print(text)
313
+ return 0
314
+
315
+
316
+ if __name__ == "__main__":
317
+ sys.exit(main())
code/scripts/container_smoke.sh ADDED
@@ -0,0 +1,137 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ # Serve ONE serve profile of the BUILT container package, run the smoke test against it, keep the evidence, and always
4
+ # stop it -- in one device-lock window. Run it once per serve profile (`tt-model profiles
5
+ # <staged dir>/tt_kernel_manifest.json` lists them; without --profile the package's default profile is served):
6
+ #
7
+ # ROOT=/home/ubuntu/experiments/tt-models
8
+ # $ROOT/bin/devrun -t 3600 -k 150 -- env -u HF_TOKEN -u HUGGING_FACE_HUB_TOKEN sg docker -c \
9
+ # "bash code/scripts/container_smoke.sh $ROOT/build/meteor-p150 [port] [--profile NAME] [--log-dir DIR]"
10
+ #
11
+ # The smoke test FAILS unless /info reports ETH dispatch and the 12x10 grid and runs what the package pins for the
12
+ # profile (dispatch, CQs, variant, weights revision); it also compares the output with the stored CPU reference of the
13
+ # sample when one exists (code/tt_meteor/server/smoke_test.py). Exit code: the smoke test's (0 = PASS); 1 when
14
+ # serve fails, 2 on a usage error.
15
+ #
16
+ # Evidence, kept whatever the outcome (--log-dir, default: logs/smoke/ of this repo, next to code/), named
17
+ # <name>[-<profile>]-<UTC time>.*:
18
+ # .container.log the container's whole log (boot, requests, shutdown), followed while the container stops
19
+ # .info.json GET /info as soon as the server is READY (device: dispatch, grid, cores; pins; versions)
20
+ # .smoke.json the smoke test's /predict output (the SMOKE_OUT environment variable overrides this path)
21
+ # .result.json profile, port, exit code, start / end times
22
+ #
23
+ # `tt-model serve` returns once the server is READY and leaves the container running, so the stop must happen before
24
+ # the lock is released, also when devrun's timeout TERMs this script: the EXIT trap saves the log, then stops the
25
+ # container cleanly with SIGTERM (120 s grace, hence devrun -k 150), never `docker kill`, which would leave the chip
26
+ # dirty. Needs docker access (sg docker) and python3.
27
+ set -u
28
+ usage() { echo "usage: container_smoke.sh <staged package dir> [port] [--profile NAME] [--log-dir DIR]" >&2; }
29
+ PROFILE=""
30
+ LOG_DIR=""
31
+ POSITIONAL=()
32
+ while [ $# -gt 0 ]; do
33
+ case "$1" in
34
+ --profile) [ $# -ge 2 ] || { usage; exit 2; }; PROFILE="$2"; shift 2 ;;
35
+ --profile=*) PROFILE="${1#--profile=}"; shift ;;
36
+ --log-dir) [ $# -ge 2 ] || { usage; exit 2; }; LOG_DIR="$2"; shift 2 ;;
37
+ --log-dir=*) LOG_DIR="${1#--log-dir=}"; shift ;;
38
+ -h|--help) usage; exit 0 ;;
39
+ -*) echo "container_smoke.sh: unknown option $1" >&2; usage; exit 2 ;;
40
+ *) POSITIONAL+=("$1"); shift ;;
41
+ esac
42
+ done
43
+ [ ${#POSITIONAL[@]} -ge 1 ] && [ ${#POSITIONAL[@]} -le 2 ] || { usage; exit 2; }
44
+ STAGED="${POSITIONAL[0]}"
45
+ PORT="${POSITIONAL[1]:-20000}"
46
+ MANIFEST="$STAGED/tt_kernel_manifest.json"
47
+ [ -f "$MANIFEST" ] || { echo "container_smoke.sh: $MANIFEST not found (run tt-model package first)" >&2; exit 2; }
48
+ HERE="$(cd "$(dirname "$0")" && pwd)"
49
+ PROFILE_ARGS=()
50
+ [ -n "$PROFILE" ] && PROFILE_ARGS=(--profile "$PROFILE")
51
+
52
+ # The package name and the served profile (tt-model's rule: --profile, else default_profile, else the first serve
53
+ # profile), hence the container name tt-model gives it: tt-model-<name>-<profile> (tt_kernel/container.py).
54
+ read -r NAME PROFILE_NAME < <(python3 - "$MANIFEST" "$PROFILE" <<'EOF'
55
+ import json, sys
56
+ m = json.load(open(sys.argv[1]))
57
+ c = m.get("container") or {}
58
+ profiles = c.get("serve_profiles") or [{}]
59
+ print(m.get("name") or "model", sys.argv[2] or c.get("default_profile") or profiles[0].get("name") or "default")
60
+ EOF
61
+ )
62
+ [ -n "${NAME:-}" ] || { echo "container_smoke.sh: cannot read the package name from $MANIFEST" >&2; exit 2; }
63
+ CONTAINER="tt-model-$NAME-$PROFILE_NAME"
64
+
65
+ [ -n "$LOG_DIR" ] || LOG_DIR="$(cd "$HERE/../.." && pwd)/logs/smoke"
66
+ if ! mkdir -p "$LOG_DIR" 2>/dev/null || [ ! -w "$LOG_DIR" ]; then
67
+ echo "container_smoke.sh: cannot write $LOG_DIR; keeping the evidence in ${TMPDIR:-/tmp}" >&2
68
+ LOG_DIR="${TMPDIR:-/tmp}"
69
+ fi
70
+ LOG_DIR="$(cd "$LOG_DIR" && pwd)"
71
+ STARTED="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
72
+ STEM="$LOG_DIR/$NAME${PROFILE:+-$PROFILE}-$(date -u +%Y%m%dT%H%M%SZ)"
73
+ SMOKE_JSON="${SMOKE_OUT:-$STEM.smoke.json}"
74
+
75
+ fetch() { # fetch URL FILE: one GET with a 30 s timeout; the body goes to FILE
76
+ python3 - "$1" "$2" <<'EOF'
77
+ import sys, urllib.request
78
+ try:
79
+ with urllib.request.urlopen(sys.argv[1], timeout=30) as r:
80
+ body = r.read()
81
+ except Exception as e: # noqa: BLE001 -- best effort: the smoke test reports the server's state
82
+ sys.exit(f"container_smoke.sh: GET {sys.argv[1]} failed: {e}")
83
+ open(sys.argv[2], "wb").write(body)
84
+ EOF
85
+ }
86
+
87
+ CHILD=""
88
+ STOPPED=0
89
+ cleanup() {
90
+ local rc=$?
91
+ [ "$STOPPED" = 1 ] && return
92
+ STOPPED=1
93
+ if [ -n "$CHILD" ]; then kill -TERM "$CHILD" 2>/dev/null; wait "$CHILD" 2>/dev/null; fi
94
+ # The log is lost with the container: follow it (whole history, then the shutdown lines) while it stops.
95
+ local logger=""
96
+ if docker inspect "$CONTAINER" >/dev/null 2>&1; then
97
+ docker logs --follow "$CONTAINER" > "$STEM.container.log" 2>&1 &
98
+ logger=$!
99
+ else
100
+ tt-model logs "${PROFILE_ARGS[@]}" "$MANIFEST" > "$STEM.container.log" 2>&1 || true
101
+ fi
102
+ tt-model stop "${PROFILE_ARGS[@]}" "$MANIFEST" || true
103
+ if [ -n "$logger" ]; then
104
+ for _ in $(seq 1 30); do kill -0 "$logger" 2>/dev/null || break; sleep 1; done
105
+ kill "$logger" 2>/dev/null
106
+ wait "$logger" 2>/dev/null
107
+ fi
108
+ python3 - "$STEM.result.json" "$NAME" "$PROFILE_NAME" "$PORT" "$rc" "$STARTED" "$CONTAINER" "$MANIFEST" <<'EOF'
109
+ import datetime, json, sys
110
+ out, name, profile, port, rc, started, container, manifest = sys.argv[1:]
111
+ ended = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
112
+ json.dump({"bundle": name, "profile": profile, "port": int(port) if port.isdigit() else port, "rc": int(rc),
113
+ "result": "PASS" if rc == "0" else "FAIL",
114
+ "started": started, "ended": ended, "container": container, "manifest": manifest}, open(out, "w"), indent=1)
115
+ EOF
116
+ echo "container_smoke.sh: rc=$rc; evidence in $STEM.*"
117
+ }
118
+ trap cleanup EXIT
119
+ trap 'exit 143' TERM
120
+ trap 'exit 130' INT
121
+
122
+ # Each step runs in the background and is waited for: bash defers a trap while a FOREGROUND command runs, so a TERM
123
+ # sent to this script alone (devrun's timeout signals the whole process group) would otherwise wait for the step.
124
+ step() {
125
+ "$@" &
126
+ CHILD=$!
127
+ wait "$CHILD"
128
+ local rc=$?
129
+ CHILD=""
130
+ return "$rc"
131
+ }
132
+
133
+ # --port and --profile BEFORE the target: options after it are passed through to the container (tt-model cli rule)
134
+ step tt-model serve --port "$PORT" "${PROFILE_ARGS[@]}" "$MANIFEST" || exit 1
135
+ step fetch "http://127.0.0.1:$PORT/info" "$STEM.info.json"
136
+ step python3 "$HERE/../tt_meteor/server/smoke_test.py" --url "http://127.0.0.1:$PORT" --wait 600 \
137
+ --manifest "$MANIFEST" "${PROFILE_ARGS[@]}" --out "$SMOKE_JSON"
code/scripts/device_stage_check.py ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Bring-up check of one device stage, eagerly, against the CPU-reference goldens (needs one p150; run under devrun).
3
+
4
+ bin/devrun -t 1800 -- python bundles/meteor-p150/code/scripts/device_stage_check.py --stage image \\
5
+ --json logs/meteor/stage_image.json
6
+
7
+ Stages (teacher-forced from ``research/meteor/goldens/<frame>/taps.npz``): ``image`` (the uint8 cameras -> FPN and the
8
+ image heads), ``lift`` (golden ctx / prob -> lift.bev, bev.raw), ``bev`` (golden bev.raw -> trunk and BEV heads),
9
+ ``head`` (golden fused / lane.pre / det / risk -> planner, refiners, output tail). Prints and saves the PCC of every tap
10
+ it produces; no gate here (the gates live in ``tests/test_pcc_device.py``, on trace replays). ``--runs 2`` runs the
11
+ stage twice and checks the two eager outputs are bit-identical (PLAN 4.4 bring-up: eager x2 before any trace).
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import argparse
16
+ import json
17
+ import os
18
+ import sys
19
+ import time
20
+ from pathlib import Path
21
+
22
+ import numpy as np
23
+
24
+ HERE = Path(__file__).resolve()
25
+ sys.path.insert(0, str(HERE.parents[1]))
26
+
27
+ GOLDENS = Path(os.environ.get("METEOR_GOLDENS", "/home/ubuntu/experiments/tt-models/research/meteor/goldens"))
28
+
29
+
30
+ def main() -> int:
31
+ ap = argparse.ArgumentParser()
32
+ ap.add_argument("--stage", required=True, choices=["image", "lift", "bev", "head"])
33
+ ap.add_argument("--frame", default="pandaset_019_f40")
34
+ ap.add_argument("--runs", type=int, default=1)
35
+ ap.add_argument("--json", default=None)
36
+ args = ap.parse_args()
37
+
38
+ from tt_meteor.device import close_device, describe_device, open_device
39
+ from tt_meteor.tt.debug import STAGES, compare_taps_to_golden
40
+ from tt_meteor.tt.params import MeteorParams
41
+
42
+ gold = np.load(GOLDENS / args.frame / "taps.npz")
43
+ t0 = time.perf_counter()
44
+ params = MeteorParams.load()
45
+ dev = open_device(allow_fallback=False)
46
+ info = describe_device(dev)
47
+ print("device:", {k: info[k] for k in ("dispatch", "grid", "num_command_queues")}, flush=True)
48
+ rows = {}
49
+ try:
50
+ stage = STAGES[args.stage](dev, params)
51
+ prev = None
52
+ for run in range(args.runs):
53
+ t1 = time.perf_counter()
54
+ taps = stage.run(gold)
55
+ dt = time.perf_counter() - t1
56
+ print(f"run {run}: {dt:.1f} s", flush=True)
57
+ if prev is not None:
58
+ for k in taps:
59
+ same = np.array_equal(np.asarray(taps[k]), np.asarray(prev[k]), equal_nan=True)
60
+ rows.setdefault("eager_x2_bit_identical", {})[k] = bool(same)
61
+ if not same:
62
+ print(f" NOT bit-identical across eager runs: {k}", flush=True)
63
+ prev = taps
64
+ rows.update(compare_taps_to_golden(prev, gold))
65
+ stage.release()
66
+ finally:
67
+ close_device(dev)
68
+ for k, v in rows.items():
69
+ if k != "eager_x2_bit_identical":
70
+ print(f" {k:24s} {v}")
71
+ print(f"total {time.perf_counter() - t0:.1f} s")
72
+ if args.json:
73
+ Path(args.json).parent.mkdir(parents=True, exist_ok=True)
74
+ Path(args.json).write_text(json.dumps({"stage": args.stage, "frame": args.frame, "device": info,
75
+ "rows": rows}, indent=1, default=str))
76
+ return 0
77
+
78
+
79
+ if __name__ == "__main__":
80
+ sys.exit(main())
code/scripts/dump_device_outputs.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Run golden frames' graph inputs through the served model and save the 19 raw device outputs (npz) for offline
4
+ analysis against the goldens (e.g. which boxes a set-based gate misses and why).
5
+
6
+ bin/devrun -t 900 -- python code/scripts/dump_device_outputs.py --out DIR meteor_valday_f040 ...
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import os
12
+ from pathlib import Path
13
+
14
+ import numpy as np
15
+
16
+ from tt_meteor import METEOR
17
+ from tt_meteor.host.preprocess import MeteorFrame
18
+
19
+ GOLDENS = Path(os.environ.get("METEOR_GOLDENS", "/home/ubuntu/experiments/tt-models/research/meteor/goldens"))
20
+
21
+
22
+ def main() -> None:
23
+ ap = argparse.ArgumentParser()
24
+ ap.add_argument("frames", nargs="+")
25
+ ap.add_argument("--out", required=True)
26
+ ap.add_argument("--repeat", type=int, default=2, help="runs per frame (outputs must be bit-identical)")
27
+ a = ap.parse_args()
28
+ out = Path(a.out)
29
+ out.mkdir(parents=True, exist_ok=True)
30
+ with METEOR.from_pretrained() as model:
31
+ for name in a.frames:
32
+ with np.load(GOLDENS / name / "taps.npz") as z:
33
+ g = {k: z[k] for k in ("input.imgs", "input.K", "input.T_cam_ego", "input.v0", "input.present")}
34
+ fr = MeteorFrame(g["input.imgs"], g["input.K"], g["input.T_cam_ego"], g["input.v0"], g["input.present"])
35
+ runs = [{k: np.array(v) for k, v in model._forward({"frame": fr}).items()} for _ in range(a.repeat)]
36
+ for r in runs[1:]:
37
+ for k in r:
38
+ assert np.array_equal(r[k], runs[0][k]), f"{name}: {k} differs between runs"
39
+ np.savez(out / f"{name}.npz", **runs[0])
40
+ print(f"{name}: saved {len(runs[0])} outputs, {a.repeat} runs bit-identical")
41
+
42
+
43
+ if __name__ == "__main__":
44
+ main()
code/scripts/fetch_samples.sh ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ # Download the public sample data that meteor-p150 may NOT redistribute (license unstated or non-commercial)
4
+ # into ${METEOR_SAMPLES:-$HOME/.cache/tt_meteor/samples}, sha256-checked. Tests and benchmarks read it from there.
5
+ # Nothing downloaded here is executed; parse it with numpy / sqlite3 only.
6
+ set -euo pipefail
7
+ DEST="${METEOR_SAMPLES:-$HOME/.cache/tt_meteor/samples}"
8
+ mkdir -p "$DEST"
9
+
10
+ fetch() { # fetch <url> <sha256> <file name>
11
+ local url="$1" sum="$2" name="$3"
12
+ if [ -f "$DEST/$name" ] && echo "$sum $DEST/$name" | sha256sum -c --status; then
13
+ echo "ok $name"; return
14
+ fi
15
+ curl -fL --retry 3 -o "$DEST/$name.part" "$url"
16
+ echo "$sum $DEST/$name.part" | sha256sum -c --status || { echo "sha256 mismatch: $name" >&2; exit 1; }
17
+ mv "$DEST/$name.part" "$DEST/$name"; echo "fetched $name"
18
+ }
19
+
20
+ # Example (Autoware demo rosbag; pinned in autoware/ansible/roles/demo_artifacts/tasks/main.yaml:48-53):
21
+ # fetch https://autoware-files.s3.us-west-2.amazonaws.com/recordings/bags/demos/sample-rosbag.zip \
22
+ # 5f9d36353393b3d249212153c19049822b1298db56512aa045b4f7f6fc37cf88 sample-rosbag.zip
23
+ # METEOR demo scenes (research / demonstration use only) and nuScenes-derived inputs (CC BY-NC-SA 4.0) are never shipped;
24
+ # fetch AutowareFoundation/meteor-demo-scenes at the pinned revision bd5f94dc79a0fbbd1836ad5e7198b1feb1b614d3 (research/meteor/SPEC.md section 7)
25
+ echo "samples in $DEST"
code/scripts/make_demo.py ADDED
@@ -0,0 +1,322 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """The card's demo media (``media/``) from outputs of this port on the p150. Development tool of the workspace (CPU
4
+ only; research venv: numpy, OpenCV, PIL), run after the device jobs that produce the TT outputs:
5
+
6
+ PY=/home/ubuntu/experiments/tt-models/tools/research-venv/bin/python
7
+ $PY code/scripts/make_demo.py --tt-public logs/public_tt --tt-sample logs/meteor/shipped_sample_tt.json \\
8
+ [--compare logs/public_tt/compare]
9
+
10
+ - ``--tt-public``: ``code/scripts/run_public_frames.py``'s output (TT outputs of the public frames in the CPU-golden
11
+ layout). Renders, with the research renderer ``research/meteor/public_data/scripts/render_media.py`` (privacy
12
+ blur, dataset attribution in every footer), the card and the camera-head images of PandaSet 019 frame 40,
13
+ PandaSet 090 frame 40 and nuScenes scene-0103 key-frame 9 (non-commercial, labelled ``_NC``), the PandaSet 019
14
+ animation, and a bird's-eye TT-vs-CPU comparison of PandaSet 019 frame 40 (no camera pixels).
15
+ - ``--tt-sample``: the TT ``/predict`` body of the shipped synthetic sample (``tests/test_e2e_device.py`` keeps it);
16
+ renders the eight synthetic cameras with the TT 2D boxes and the TT and CPU-reference bird's-eye views side by side.
17
+ - ``media/ATTRIBUTION.md``: every file, its source frames, licence and changes.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import base64
23
+ import io
24
+ import json
25
+ import math
26
+ import os
27
+ import sys
28
+ from pathlib import Path
29
+
30
+ import numpy as np
31
+
32
+ BUNDLE = Path(__file__).resolve().parents[2]
33
+ ROOT = Path(os.environ.get("TT_MODELS_ROOT", "/home/ubuntu/experiments/tt-models"))
34
+ PUBLIC = ROOT / "research" / "meteor" / "public_data"
35
+ sys.path.insert(0, str(PUBLIC / "scripts"))
36
+ sys.path.insert(0, str(ROOT / "research" / "meteor" / "scripts"))
37
+ MEDIA = BUNDLE / "media"
38
+ PKG = BUNDLE / "code" / "tt_meteor"
39
+ TT_LABEL = "Tenstorrent p150 output: meteor-p150 port (Blackhole, ETH dispatch 12x10, 1 CQ, one metal trace)"
40
+ CPU_LABEL = "fp32 CPU reference (the same network, ONNX Runtime parity)"
41
+ CARDS = [("ps019", 20), ("ps090", 20), ("ns0103", 9)]
42
+
43
+
44
+ def _rm():
45
+ import render_media as rm # research renderer (privacy blur, footers)
46
+
47
+ return rm
48
+
49
+
50
+ def tag_of(gid: str, i: int, F) -> str:
51
+ return f"meteor_{gid}_f{i:04d}" + ("_NC" if F.dataset == "nuScenes" else "")
52
+
53
+
54
+ def save_jpeg(path: Path, rgb: np.ndarray, q: int = 88) -> None:
55
+ from PIL import Image
56
+
57
+ Image.fromarray(rgb).save(path, quality=q, optimize=True)
58
+
59
+
60
+ def save_png(path: Path, rgb: np.ndarray) -> None:
61
+ from PIL import Image
62
+
63
+ Image.fromarray(rgb).save(path, optimize=True)
64
+
65
+
66
+ def public_renders(tt_dir: Path, entries: list) -> None:
67
+ rm = _rm()
68
+ for gid, i in CARDS:
69
+ F = rm.Frame(gid, i, str(tt_dir / gid))
70
+ tag = tag_of(gid, i, F)
71
+ lic = "CC BY 4.0 + PandaSet Dataset Terms" if F.dataset == "PandaSet" else "**CC BY-NC-SA 4.0, non-commercial**"
72
+ rgb, nb = rm.card(F, TT_LABEL)
73
+ save_jpeg(MEDIA / f"{tag}_card_tt.jpg", rgb)
74
+ entries.append((f"{tag}_card_tt.jpg", F.dataset, F.desc, lic, "TT outputs: 8-slot camera mosaic with 2D boxes; "
75
+ "BEV lanes / risk / 3D boxes / futures / plan vs the logged path; front view with projected "
76
+ "3D boxes and the planned path", rgb.shape))
77
+ hd = rm.heads(F, TT_LABEL)
78
+ save_jpeg(MEDIA / f"{tag}_heads_tt.jpg", hd)
79
+ entries.append((f"{tag}_heads_tt.jpg", F.dataset, F.desc, lic, "TT outputs: per-camera seg2d overlay and "
80
+ "depth (linear bins), 8 slots", hd.shape))
81
+ print("rendered", tag, flush=True)
82
+ frames = sorted(int(p.stem[1:]) for p in (tt_dir / "ps019").glob("f[0-9][0-9][0-9][0-9].json"))
83
+ p = MEDIA / "meteor_ps019_seq_tt.gif"
84
+ Fs, dur = rm.animation("ps019", frames, str(p), TT_LABEL, golden_dir=str(tt_dir / "ps019"))
85
+ from PIL import Image
86
+
87
+ with Image.open(p) as im:
88
+ shape = (im.height, im.width, 3)
89
+ entries.append((p.name, "PandaSet", f"{Fs[0].desc} ... {Fs[-1].desc} (every 4th frame, {len(frames)} frames, "
90
+ f"{dur} ms each: real time)", "CC BY 4.0 + PandaSet Dataset Terms", "TT outputs: the card as an "
91
+ "animation (128-colour GIF)", shape))
92
+ print("rendered", p.name, flush=True)
93
+
94
+
95
+ def _text(draw, xy, s, size=12, fill=(235, 235, 235), bold=False, anchor="la"):
96
+ _rm().text(draw, xy, s, size, fill, bold, anchor)
97
+
98
+
99
+ def bev_compare(tt_dir: Path, compare: dict, entries: list) -> None:
100
+ """PandaSet 019 frame 40: CPU BEV | TT BEV | where the two lane maps differ (no camera pixels)."""
101
+ import cv2
102
+ from PIL import Image, ImageDraw
103
+
104
+ rm = _rm()
105
+ gid, i = "ps019", 20
106
+ Fc = rm.Frame(gid, i, None)
107
+ Ft = rm.Frame(gid, i, str(tt_dir / gid))
108
+ w, h = 300, 450
109
+ a, b = rm.bev_panel(Fc, w, h), rm.bev_panel(Ft, w, h)
110
+ view_f, view_r, yh = 50.0, 25.0, 25.0
111
+ r0, r1 = int((80 - view_f) / 0.2), int((80 + view_r) / 0.2)
112
+ c0, c1 = int((50 - yh) / 0.2), int((50 + yh) / 0.2)
113
+ lc, lt = Fc.out["lane"][0][r0:r1, c0:c1], Ft.out["lane"][0][r0:r1, c0:c1]
114
+ diff = np.full(lc.shape + (3,), 30, np.uint8)
115
+ diff[lc > 0] = (70, 70, 75)
116
+ diff[lc != lt] = (255, 60, 60)
117
+ diff = cv2.resize(diff, (w, h), interpolation=cv2.INTER_NEAREST)
118
+ H = 44 + 16 + h + 92
119
+ canvas = np.full((H, 3 * w + 4 * 8, 3), rm.BG, np.uint8)
120
+ for k, img in enumerate((a, b, diff)):
121
+ rm.paste(canvas, img, 8 + k * (w + 8), 60)
122
+ im = Image.fromarray(canvas)
123
+ d = ImageDraw.Draw(im)
124
+ _text(d, (8, 6), f"METEOR v1.0 on {Fc.desc}: Tenstorrent p150 vs the fp32 CPU reference (bird's-eye view)", 14, bold=True)
125
+ _text(d, (8, 26), "Pixel-free BEV panels: lane classes, risk heat, 3D boxes, futures, the 3 ego paths (selected: green); "
126
+ "white circles: logged path", 10, fill=rm.TEXT2)
127
+ for k, lab in enumerate(("fp32 CPU reference", "Tenstorrent p150 (this port)", "lane cells that differ (red)")):
128
+ _text(d, (8 + k * (w + 8), 44), lab, 12, bold=True)
129
+ r = compare.get(f"{gid}/f{i:04d}", {})
130
+ lines = []
131
+ if r:
132
+ lines.append(f"Agreement on this frame: lane {r['lane_agree']:.4f}, seg2d {r['seg2d_agree']:.4f}, depth "
133
+ f"{r['depth_agree']:.4f} (argmax maps); hm / reg / stationary / risk PCC {r['hm_pcc']:.5f} / "
134
+ f"{r['reg_pcc']:.5f} / {r['stationary_pcc']:.5f} / {r['risk_pcc']:.5f};")
135
+ lines.append(f"3D boxes matched {r['det3d_matched']:.3f} (same class, <= 1 m); selected ego path mean deviation "
136
+ f"{r['ego_path_dev_m']:.3f} m, same mode: {r['ego_mode_equal']}; traffic light equal: {r['tl_equal']}.")
137
+ lines += ["Contains data from PandaSet (Scale AI and Hesai), https://pandaset.org, CC BY 4.0 + PandaSet Dataset Terms "
138
+ "(outputs only; no camera pixels).", "Scale AI and Hesai do not endorse this work."]
139
+ for k, line in enumerate(lines):
140
+ _text(d, (8, 60 + h + 8 + 15 * k), line, 10, fill=rm.TEXT1 if k < 2 else rm.TEXT2)
141
+ out = np.asarray(im)
142
+ path = MEDIA / "meteor_ps019_f0020_bev_tt_vs_cpu.png"
143
+ save_png(path, out)
144
+ entries.append((path.name, "PandaSet", Fc.desc, "CC BY 4.0 + PandaSet Dataset Terms", "BEV of the fp32 CPU "
145
+ "reference and of the TT output side by side, and the lane cells that differ (no camera pixels)",
146
+ out.shape))
147
+ print("rendered", path.name, flush=True)
148
+
149
+
150
+ # ------------------------------------------------------------------------------------------- shipped sample
151
+ def _decode_lane(body: dict) -> np.ndarray:
152
+ from PIL import Image
153
+
154
+ lane = body["lane"]
155
+ a = np.asarray(Image.open(io.BytesIO(base64.b64decode(lane["data"]))))
156
+ return a.reshape(lane["shape"])
157
+
158
+
159
+ def body_bev(body: dict, w: int, h: int, view_f=50.0, view_r=25.0, yh=25.0) -> np.ndarray:
160
+ import cv2
161
+
162
+ rm = _rm()
163
+ lane = _decode_lane(body)
164
+ r0, r1 = int((80 - view_f) / 0.2), int((80 + view_r) / 0.2)
165
+ c0, c1 = int((50 - yh) / 0.2), int((50 + yh) / 0.2)
166
+ img = cv2.resize(np.asarray(rm.LANE_RGB, np.uint8)[lane[r0:r1, c0:c1]], (w, h), interpolation=cv2.INTER_NEAREST)
167
+ s = h / (view_f + view_r)
168
+
169
+ def px(x, y):
170
+ return (yh - y) * s, (view_f - x) * s
171
+
172
+ for dd in range(-int(view_r // 10) * 10, int(view_f) + 1, 10):
173
+ yy = int(round((view_f - dd) * s))
174
+ cv2.line(img, (0, yy), (w, yy), (55, 55, 60), 1)
175
+ if dd:
176
+ cv2.putText(img, f"{dd}m", (3, yy - 3), cv2.FONT_HERSHEY_SIMPLEX, 0.32, (150, 150, 150), 1, cv2.LINE_AA)
177
+ for det in body["detections"]:
178
+ (x, y), (L, W), yaw = det["center"], det["size"], det["yaw"]
179
+ col = rm.STAT_RGB if det.get("stationary") else (rm.VEH_RGB if det["label_id"] == 0 else rm.VRU_RGB)
180
+ cors = rm.box_corners(x, y, 0, L, W, 0, yaw)[:4]
181
+ p = np.array([px(cx, cy) for cx, cy, _ in cors]).round().astype(np.int32)
182
+ cv2.polylines(img, [p], True, col, 2, cv2.LINE_AA)
183
+ cx, cy = px(x, y)
184
+ fx, fy = px(x + 0.5 * L * math.cos(yaw), y + 0.5 * L * math.sin(yaw))
185
+ cv2.line(img, (int(cx), int(cy)), (int(fx), int(fy)), col, 2, cv2.LINE_AA)
186
+ if det.get("future") and not det.get("stationary") and det["label_id"] == 0:
187
+ q = np.array([[cx, cy]] + [px(a, b) for a, b in det["future"]]).round().astype(np.int32)
188
+ cv2.polylines(img, [q], False, col, 1, cv2.LINE_AA)
189
+ ex, ey = px(0, 0)
190
+ sel = body["plan"]["mode"]
191
+ for k, path in enumerate(body["plan"]["paths"]):
192
+ q = np.array([[ex, ey]] + [px(a, b) for a, b in path]).round().astype(np.int32)
193
+ cv2.polylines(img, [q], False, rm.PLAN_RGB if k == sel else (0, 150, 60), 2 if k == sel else 1, cv2.LINE_AA)
194
+ if k == sel:
195
+ for pt in q[1:]:
196
+ cv2.circle(img, tuple(int(v) for v in pt), 3, rm.PLAN_RGB, -1, cv2.LINE_AA)
197
+ tri = np.array([[ex, ey - 9], [ex - 6, ey + 7], [ex + 6, ey + 7]]).round().astype(np.int32)
198
+ cv2.fillPoly(img, [tri], (255, 255, 255))
199
+ return img
200
+
201
+
202
+ def sample_render(tt_body: dict, entries: list) -> None:
203
+ import cv2
204
+ from PIL import Image, ImageDraw
205
+
206
+ rm = _rm()
207
+ ref = json.loads((PKG / "samples" / "synthetic_8cam.reference.json").read_text())
208
+ spec = json.loads((PKG / "samples" / "synthetic_8cam.json").read_text())
209
+ tw, th, gap, x0, top, lab = 233, 131, 6, 4, 44, 14
210
+ bw, bh = 300, 450
211
+ y_bot = top + 2 * (lab + th) + 12
212
+ H = y_bot + 18 + bh + 70
213
+ canvas = np.full((H, 960, 3), rm.BG, np.uint8)
214
+ for k, cam in enumerate(rm.TILE_ORDER):
215
+ img = np.asarray(Image.open(PKG / "samples" / spec["images"][cam]).convert("RGB")).copy()
216
+ for det in tt_body["detections_2d"].get(cam, []):
217
+ a, b, c, e = (int(round(v)) for v in det["box_xyxy"])
218
+ cv2.rectangle(img, (a, b), (c, e), rm.DET10_RGB[det["label_id"] % 10], 2)
219
+ img = cv2.resize(img, (tw, th), interpolation=cv2.INTER_AREA)
220
+ rm.paste(canvas, img, x0 + (k % 4) * (tw + gap), top + (k // 4) * (lab + th) + lab)
221
+ rm.paste(canvas, body_bev(ref, bw, bh), x0, y_bot + 18)
222
+ rm.paste(canvas, body_bev(tt_body, bw, bh), x0 + bw + 8, y_bot + 18)
223
+ im = Image.fromarray(canvas)
224
+ d = ImageDraw.Draw(im)
225
+ _text(d, (8, 6), "METEOR v1.0 on the shipped synthetic sample (samples/synthetic_8cam): Tenstorrent p150 vs the fp32 CPU "
226
+ "reference", 14, bold=True)
227
+ _text(d, (8, 26), "Synthetic test frame generated by this repository (Apache-2.0): a ray-cast street; the object "
228
+ "pixels were optimised against the CPU reference", 10, fill=rm.WARN)
229
+ for k, cam in enumerate(rm.TILE_ORDER):
230
+ n2d = len(tt_body["detections_2d"].get(cam, []))
231
+ _text(d, (x0 + (k % 4) * (tw + gap), top + (k // 4) * (lab + th) + 1), f"{cam[4:]} (TT 2D boxes: {n2d})",
232
+ 10, fill=rm.TEXT2)
233
+ _text(d, (x0, y_bot), "BEV: fp32 CPU reference (stored)", 12, bold=True)
234
+ _text(d, (x0 + bw + 8, y_bot), "BEV: Tenstorrent p150", 12, bold=True)
235
+ ix = x0 + 2 * (bw + 8) + 4
236
+ lines = [f"3D boxes: TT {tt_body['num_detections']}, reference {ref['num_detections']}",
237
+ f"plan mode: TT {tt_body['plan']['mode']} (p={tt_body['plan']['mode_probs'][tt_body['plan']['mode']]:.2f}), "
238
+ f"reference {ref['plan']['mode']}",
239
+ "path end (m): TT ({:.2f}, {:.2f}), reference ({:.2f}, {:.2f})".format(*tt_body["trajectory"][-1],
240
+ *ref["trajectory"][-1]),
241
+ f"traffic light: TT {tt_body['traffic_light']['state']}, reference {ref['traffic_light']['state']}",
242
+ f"2D boxes: TT {sum(len(v) for v in tt_body['detections_2d'].values())}, reference "
243
+ f"{sum(len(v) for v in ref['detections_2d'].values())}"]
244
+ lane_t, lane_r = _decode_lane(tt_body), _decode_lane(ref)
245
+ lines.append(f"lane map agreement: {float((lane_t == lane_r).mean()):.4f}")
246
+ by_label: dict = {}
247
+ for det in tt_body["detections"]:
248
+ by_label.setdefault(det["label"], []).append(round(det["score"], 2))
249
+ for k, v in sorted(by_label.items()):
250
+ lines.append(f"TT {k} scores: " + ", ".join(f"{x:.2f}" for x in v))
251
+ for k, line in enumerate(lines):
252
+ _text(d, (ix, y_bot + 22 + 17 * k), line, 11)
253
+ ly = y_bot + 22 + 17 * len(lines) + 10
254
+ for name, colr in (("vehicle", rm.VEH_RGB), ("VRU", rm.VRU_RGB), ("stationary", rm.STAT_RGB), ("plan", rm.PLAN_RGB)):
255
+ d.rectangle([ix, ly + 3, ix + 8, ly + 11], outline=colr, width=2)
256
+ _text(d, (ix + 12, ly), name, 10, fill=rm.TEXT2)
257
+ ly += 15
258
+ _text(d, (8, H - 18), "Data generated by this repository (code/scripts/make_synthetic_sample.py), Apache-2.0. "
259
+ "No third-party pixels.", 10, fill=rm.TEXT2)
260
+ out = np.asarray(im)
261
+ path = MEDIA / "meteor_synthetic_8cam_tt_vs_cpu.png"
262
+ save_png(path, out)
263
+ entries.append((path.name, "synthetic (this repository)", "code/tt_meteor/samples/synthetic_8cam", "Apache-2.0",
264
+ "the shipped sample's eight cameras with the TT 2D boxes; BEV of the stored CPU reference and of "
265
+ "the TT output", out.shape))
266
+ print("rendered", path.name, flush=True)
267
+
268
+
269
+ def write_attribution(entries: list) -> None:
270
+ rm = _rm()
271
+ C = rm.C
272
+ lines = ["# media/: sources and licences", "",
273
+ "Every model output drawn here was computed by **this port on one Tenstorrent Blackhole p150** (ETH "
274
+ "dispatch, 12x10 grid, 1 command queue, the `frame` metal trace), unless a panel says \"CPU reference\" "
275
+ "(the port's fp32 CPU reference of the same network, which matches ONNX Runtime on "
276
+ "`meteor_v157c3Z.onnx`). Renderer: `code/scripts/make_demo.py` (camera mosaics, heads and animation through "
277
+ "`research/meteor/public_data/scripts/render_media.py` of the development workspace).", "",
278
+ "METEOR was trained on TIER IV recordings only (no paper, no public training split). PandaSet and nuScenes "
279
+ "are out of its training domain (another camera rig, country and ISP), so the detections show the domain "
280
+ "gap, not the port. No METEOR demo-scene frame appears in any image.", "",
281
+ "| file | data | frames | content | licence | size (px) |", "|---|---|---|---|---|---|"]
282
+ for f, ds, frames, lic, content, shape in entries:
283
+ lines.append(f"| `{f}` | {ds} | {frames} | {content} | {lic} | {shape[1]}x{shape[0]} |")
284
+ lines += ["", "Changes to the dataset images: resized to the 768x432 METEOR inputs (and a centre crop for the "
285
+ "virtual FRONT_NARROW camera), downscaled for display (<= 960 px wide), heads of annotated pedestrians "
286
+ "(<= 30 m) and plate areas of annotated vehicles (<= 25 m) blurred, model outputs drawn. The BACK_NARROW "
287
+ "tile is black because that camera is absent (the model gets a zero image plus the BACK_WIDE pose, its "
288
+ "trained 7-camera configuration).", "",
289
+ "Attribution texts (also printed in each image footer):", "",
290
+ "- PandaSet: " + C.PS_ATTRIBUTION,
291
+ "- nuScenes (files marked `_NC`, **non-commercial**): " + C.NS_ATTRIBUTION,
292
+ "- Synthetic sample: data generated by this repository (`code/scripts/make_synthetic_sample.py`), "
293
+ "Apache-2.0.", "",
294
+ "Rules (research/DATASETS.md section 2): the nuScenes renders are **non-commercial (CC BY-NC-SA 4.0, "
295
+ "ShareAlike)** and keep that label next to the image wherever they are shown; the PandaSet renders are "
296
+ "CC BY 4.0 + the PandaSet Dataset Terms (do not use them to identify people; no use of the licensors' "
297
+ "names or logos beyond the attribution). No raw dataset file is in this directory."]
298
+ (MEDIA / "ATTRIBUTION.md").write_text("\n".join(lines) + "\n")
299
+
300
+
301
+ def main() -> None:
302
+ ap = argparse.ArgumentParser()
303
+ ap.add_argument("--tt-public", type=Path, required=True)
304
+ ap.add_argument("--tt-sample", type=Path, help="TT /predict body of the shipped sample (omit: no sample render)")
305
+ ap.add_argument("--compare", type=Path, help="compare_tt.py JSON files (<id>.json) of the TT run vs the goldens")
306
+ a = ap.parse_args()
307
+ MEDIA.mkdir(exist_ok=True)
308
+ compare = {}
309
+ if a.compare:
310
+ for p in sorted(a.compare.glob("*.json")):
311
+ for r in json.loads(p.read_text()).get("rows", []):
312
+ compare[f"{p.stem}/f{r['frame']:04d}"] = r
313
+ entries: list = []
314
+ public_renders(a.tt_public, entries)
315
+ bev_compare(a.tt_public, compare, entries)
316
+ if a.tt_sample:
317
+ sample_render(json.loads(a.tt_sample.read_text()), entries)
318
+ write_attribution(entries)
319
+
320
+
321
+ if __name__ == "__main__":
322
+ main()
code/scripts/make_goldens.py ADDED
@@ -0,0 +1,237 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Goldens of the METEOR port from the bundle's fp32 CPU reference (``tt_meteor.reference``), cross-checked against
4
+ the research ONNX Runtime goldens of the same frames.
5
+
6
+ Run in the research venv (torch, onnx; no device)::
7
+
8
+ cd bundles/meteor-p150 && OMP_NUM_THREADS=4 PYTHONPATH=$PWD/code \\
9
+ /home/ubuntu/experiments/tt-models/tools/research-venv/bin/python code/scripts/make_goldens.py [--frames ...]
10
+
11
+ Writes (large files never go into the bundle):
12
+
13
+ - ``research/meteor/goldens/<frame>/taps.npz``: the reference's taps (``ttaw.golden`` format, JSON ``__meta__``): every
14
+ tap for the shipped sample, the gate taps for the other frames; float tensors above 1 M elements as float16, the
15
+ rest float32, the 19 outputs in their ONNX dtypes, the feed (``input.*``) and the lift geometry tables.
16
+ - ``research/meteor/goldens/report.json``: the cross-check of every frame against its ORT golden (PCC, agreement).
17
+ - ``tests/goldens/pandaset_019_f40_reference.json``: the decoded reference result of the PandaSet sample (small; the
18
+ host / e2e tests and the card compare against it);
19
+ - ``samples/pandaset_019_f40.reference.json``: the ``/predict`` body of the reference on the PandaSet sample request
20
+ (``tests/test_e2e_device.py`` compares the served body with it).
21
+ Both under ``staging_samples_pandaset/`` (the PandaSet sample does not ship until the user approves it; once it is
22
+ swapped into ``code/tt_meteor/`` they are written there). The shipped synthetic sample's reference is written by
23
+ ``code/scripts/make_synthetic_sample.py``.
24
+ - ``--variants``: ``research/meteor/goldens/<variant>/taps.npz`` (gate taps) of the LOAD-knob variants of the graph
25
+ (``VARIANTS``: the D12 ``input_norm=imagenet`` option with ``depth_mean_bins=linear``, on the shipped sample with
26
+ CAM_FRONT_NARROW absent: zero image + the donor's calibration, so the ``present`` mask matters; plus its
27
+ ``_unzeroed`` counter-example, the mask not applied) for ``tests/test_variants_device.py``. The imagenet reference is itself checked against ONNX Runtime
28
+ (``test_reference_cpu.py::test_imagenet_variant``).
29
+ """
30
+ from __future__ import annotations
31
+
32
+ import argparse
33
+ import hashlib
34
+ import json
35
+ import sys
36
+ import time
37
+ from pathlib import Path
38
+
39
+ import numpy as np
40
+
41
+ BUNDLE = Path(__file__).resolve().parents[2]
42
+ sys.path.insert(0, str(BUNDLE / "code"))
43
+
44
+ from tt_meteor import __version__ # noqa: E402
45
+ from tt_meteor.api import load_sample # noqa: E402
46
+ from tt_meteor.host.calib import lift_geometry # noqa: E402
47
+ from tt_meteor.host.postprocess import PostConfig # noqa: E402
48
+ from tt_meteor.host.preprocess import MeteorFrame, load_scene_frame, scene_dir_of # noqa: E402
49
+ from tt_meteor.host.result import build_output # noqa: E402
50
+ from tt_meteor.reference import config as C # noqa: E402
51
+ from tt_meteor.reference.model import OUTPUT_NAMES # noqa: E402
52
+ from tt_meteor.reference.pipeline import MeteorReference # noqa: E402
53
+ from tt_meteor.ttaw.golden import TapRegistry, save_goldens # noqa: E402
54
+ from tt_meteor.ttaw.metrics import pcc # noqa: E402
55
+
56
+ ROOT = Path("/home/ubuntu/experiments/tt-models")
57
+ R = ROOT / "research" / "meteor"
58
+ OUT = R / "goldens"
59
+ _PKG = BUNDLE / "code" / "tt_meteor"
60
+ _ROOT = _PKG if (_PKG / "samples" / "pandaset_019_f40.json").is_file() else BUNDLE / "staging_samples_pandaset"
61
+ SHIP = _ROOT / "samples" / "pandaset_019_f40"
62
+ GATE_TAPS = ["img.fpn", "seg2d.logits", "depth.logits", "ctx.painted", "lift.weight", "lift.bev", "bev.raw",
63
+ "bev.fused", "lane.pre", "det.feat", "det.hm_pre", "det.reg_pre", "traj.agent_delta", "ego.*",
64
+ "bev.fused_mean", "refiner.e2e_ln", "refiner.e2e_res", "lift.valid", "lift.grid"] + list(OUTPUT_NAMES)
65
+ EXACT_TAPS = {"lift.grid"} # kept float32 whatever their size (bit-exact geometry)
66
+ # frame id -> (scene dir or None, frame index, ORT golden, all taps?, licence note)
67
+ FRAMES = {
68
+ "pandaset_019_f40": (SHIP, 0, R / "public_data/golden/ps019/full_f0020.npz", True,
69
+ "PandaSet CC BY 4.0 + Terms (the shipped sample)"),
70
+ "pandaset_090_f40": (R / "public_data/inputs/ps090", 20, R / "public_data/golden/ps090/full_f0020.npz", False,
71
+ "PandaSet CC BY 4.0 + Terms"),
72
+ "nuscenes_0103_kf09": (R / "public_data/inputs/ns0103", 9, R / "public_data/golden/ns0103/full_f0009.npz", False,
73
+ "nuScenes CC BY-NC-SA 4.0: internal validation only, never shipped"),
74
+ "meteor_valday_f040": (ROOT / "assets/meteor/hf_meteor-demo-scenes/valday", 40, R / "golden/valday_f040_ort.npz",
75
+ False, "METEOR demo scenes: research / demonstration use only, never shipped"),
76
+ }
77
+
78
+ # variant id -> (base frame, MeteorConfig overrides, absent cameras, taps, present flags of the absent cameras)
79
+ # ``_unzeroed``: the same zero images but every camera flagged present, i.e. what the graph computes when the imagenet
80
+ # present mask is NOT applied (absent camera = the normalised zero image -mean / std): the counter-example that shows
81
+ # the device's absent-camera outputs come from the zeroed input (``tests/test_variants_device.py``).
82
+ _IMAGENET_LINEAR = {"input_norm": "imagenet", "depth_mean_bins": "linear"}
83
+ VARIANTS = {
84
+ "pandaset_019_f40_imagenet_linear": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",), GATE_TAPS,
85
+ False),
86
+ "pandaset_019_f40_imagenet_linear_unzeroed": ("pandaset_019_f40", _IMAGENET_LINEAR, ("CAM_FRONT_NARROW",),
87
+ list(OUTPUT_NAMES), True),
88
+ }
89
+
90
+
91
+ def sha(a) -> str:
92
+ return hashlib.sha256(np.ascontiguousarray(a).tobytes()).hexdigest()
93
+
94
+
95
+ def store(taps: dict) -> dict:
96
+ out = {}
97
+ for k, v in taps.items():
98
+ v = np.asarray(v)
99
+ if k in OUTPUT_NAMES or v.dtype in (np.bool_, np.uint8, np.float16) or v.dtype.kind in "iu":
100
+ out[k] = v
101
+ elif k in EXACT_TAPS:
102
+ out[k] = v.astype(np.float32)
103
+ elif v.size > (1 << 20):
104
+ out[k] = v.astype(np.float16)
105
+ else:
106
+ out[k] = v.astype(np.float32)
107
+ return out
108
+
109
+
110
+ def compare(out: dict, gold) -> dict:
111
+ rows = {}
112
+ for k in OUTPUT_NAMES:
113
+ a, b = np.asarray(out[k]), np.asarray(gold[k])
114
+ if b.dtype == np.uint8:
115
+ rows[k] = {"agreement": float((a == b).mean())}
116
+ else:
117
+ rows[k] = {"pcc": pcc(a.astype(np.float64), b.astype(np.float64)),
118
+ "max_abs": float(np.abs(a.astype(np.float64) - b.astype(np.float64)).max())}
119
+ return rows
120
+
121
+
122
+ def summary(res, frame: MeteorFrame, out: dict) -> dict:
123
+ return {"ego": np.asarray(out["ego"], np.float64).reshape(-1).tolist(),
124
+ "tl": np.asarray(out["tl"], np.float64).reshape(-1).tolist(),
125
+ "boxes3d": [b.to_dict() for b in sorted(res.boxes3d, key=lambda z: -z.score)],
126
+ "boxes2d": {cam: [b.to_dict() for b in bs] for cam, bs in zip(C.CAMERAS, res.boxes2d)},
127
+ "unknown": res.unknown, "plan_mode": int(res.plan["mode"]), "stationary_healthy": res.stationary_healthy,
128
+ "argmax_sha256": {k: sha(out[k]) for k in ("lane", "seg2d", "depth")},
129
+ "input_sha256": frame.sha256()}
130
+
131
+
132
+ def make_variant(vid: str, threads: int) -> None:
133
+ """Goldens of one LOAD-knob variant (``VARIANTS``): the base frame with the absent cameras zeroed and given their
134
+ donor's calibration (``host.preprocess.assemble_frame``'s rule), run by the reference with the variant config."""
135
+ base, overrides, absent, tap_names, flag_present = VARIANTS[vid]
136
+ scene, idx = FRAMES[base][:2]
137
+ frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx)
138
+ imgs, K, T, present = frame.imgs.copy(), frame.K.copy(), frame.T_cam_ego.copy(), frame.present.copy()
139
+ for name in absent:
140
+ i, d = C.CAMERAS.index(name), C.CAMERAS.index(C.DONOR[name])
141
+ imgs[0, i] = 0
142
+ K[0, i], T[0, i] = K[0, d], T[0, d]
143
+ present[i] = False
144
+ if flag_present:
145
+ present[:] = True
146
+ frame = MeteorFrame(imgs, K, T, frame.v0, present, frame.pose, dict(frame.meta))
147
+ ref = MeteorReference(threads=threads, cfg=C.default_config(**overrides))
148
+ taps = TapRegistry(include=list(tap_names))
149
+ t1 = time.time()
150
+ ref.run_frame(frame, taps)
151
+ dt = time.time() - t1
152
+ geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0])
153
+ arrays = store(taps.to_dict())
154
+ arrays.update({f"input.{k}": v for k, v in frame.feed().items()})
155
+ arrays["input.present"] = frame.present
156
+ meta = {"generator": "code/scripts/make_goldens.py --variants", "bundle_version": __version__, "frame": vid,
157
+ "base_frame": base, "absent": list(absent), "scene": str(scene), "index": idx,
158
+ "onnx_sha256": ref.weights.sha256, "licence": FRAMES[base][4], "input_norm": ref.cfg.input_norm,
159
+ "depth_mean_bins": ref.cfg.depth_mean_bins, "flag_present": bool(flag_present),
160
+ "taps": "gate" if tap_names is GATE_TAPS else "outputs", "seconds": round(dt, 1),
161
+ "lift": geom.stats()}
162
+ path = save_goldens(OUT / vid / "taps.npz", arrays, meta)
163
+ print(f"{vid}: {dt:.1f} s forward, {path.stat().st_size / 2 ** 20:.1f} MB", flush=True)
164
+
165
+
166
+ def main() -> None:
167
+ ap = argparse.ArgumentParser()
168
+ ap.add_argument("--frames", nargs="*", default=list(FRAMES))
169
+ ap.add_argument("--variants", nargs="*", default=None, help="LOAD-knob variants (VARIANTS) instead of frames")
170
+ ap.add_argument("--threads", type=int, default=4)
171
+ a = ap.parse_args()
172
+ t0 = time.time()
173
+ if a.variants is not None:
174
+ for vid in a.variants or list(VARIANTS):
175
+ make_variant(vid, a.threads)
176
+ print(f"done in {time.time() - t0:.0f} s")
177
+ return
178
+ ref = MeteorReference(threads=a.threads)
179
+ report = json.loads((OUT / "report.json").read_text()) if (OUT / "report.json").is_file() else {}
180
+ for fid in a.frames:
181
+ scene, idx, golden, full, note = FRAMES[fid]
182
+ if not Path(scene).is_dir():
183
+ print(f"skip {fid}: {scene} missing")
184
+ continue
185
+ frame = load_scene_frame(scene_dir_of(scene) if (Path(scene) / "scenes.txt").is_file() else scene, idx)
186
+ taps = TapRegistry(include=["*"] if full else GATE_TAPS)
187
+ t1 = time.time()
188
+ out = ref.run_frame(frame, taps)
189
+ dt = time.time() - t1
190
+ geom = lift_geometry(frame.K, frame.T_cam_ego, points=ref.weights.lift_ground_points()[0])
191
+ arrays = store(taps.to_dict())
192
+ arrays.update({f"input.{k}": v for k, v in frame.feed().items()})
193
+ arrays["input.present"] = frame.present
194
+ arrays.update({"geom.b0": geom.b0.astype(np.uint8), "geom.fr": geom.fr})
195
+ rows = {}
196
+ if golden.is_file():
197
+ g = np.load(golden)
198
+ same_feed = all(np.array_equal(g[f"in_{k}"], v) for k, v in frame.feed().items())
199
+ rows = compare(out, g)
200
+ rows["_same_feed_as_ort_golden"] = bool(same_feed)
201
+ meta = {"generator": "code/scripts/make_goldens.py", "bundle_version": __version__, "frame": fid,
202
+ "scene": str(scene), "index": idx, "onnx_sha256": ref.weights.sha256, "licence": note,
203
+ "input_norm": ref.cfg.input_norm, "depth_mean_bins": ref.cfg.depth_mean_bins,
204
+ "taps": "all" if full else "gate", "ort_golden": str(golden), "seconds": round(dt, 1),
205
+ "lift": geom.stats()}
206
+ path = save_goldens(OUT / fid / "taps.npz", arrays, meta)
207
+ report[fid] = {"meta": meta, "vs_ort": rows, "size_mb": round(path.stat().st_size / 2 ** 20, 1)}
208
+ res = build_output(out, frame, PostConfig(), None)
209
+ report[fid]["decode"] = {"boxes3d": len(res.boxes3d), "boxes2d": [len(b) for b in res.boxes2d],
210
+ "unknown": len(res.unknown), "mode": int(res.plan["mode"]),
211
+ "tl": res.traffic_light["state"]}
212
+ worst = min((r.get("pcc", r.get("agreement", 1.0)) for k, r in rows.items() if isinstance(r, dict)),
213
+ default=float("nan"))
214
+ print(f"{fid}: {dt:.1f} s forward, {report[fid]['size_mb']} MB, worst vs ORT {worst:.10f}, "
215
+ f"decode {report[fid]['decode']}", flush=True)
216
+ if fid == "pandaset_019_f40":
217
+ small = summary(res, frame, out)
218
+ small["_doc"] = ("Decoded fp32 CPU reference (tt_meteor.reference, PCC 1.0 vs ONNX Runtime) on the shipped "
219
+ "sample PandaSet 019 frame 40, stateless host decode (PostConfig defaults); "
220
+ "code/scripts/make_goldens.py.")
221
+ (_ROOT / "tests/goldens/pandaset_019_f40_reference.json").write_text(
222
+ json.dumps(small, indent=1, default=float) + "\n")
223
+ (OUT / "report.json").write_text(json.dumps(report, indent=1, default=float) + "\n")
224
+ if "pandaset_019_f40" in a.frames:
225
+ # the /predict body of the reference on the shipped sample request (fresh stream, as the smoke test sends it)
226
+ request = load_sample(SHIP.parent / "pandaset_019_f40.json")
227
+ body = MeteorReference(weights=ref.weights, threads=a.threads)(**request)
228
+ d = body.to_dict()
229
+ d["timing_ms"] = {}
230
+ (SHIP.parent / "pandaset_019_f40.reference.json").write_text(json.dumps(d) + "\n")
231
+ print("wrote samples/pandaset_019_f40.reference.json:", d["num_detections"], "detections, mode",
232
+ d["plan"]["mode"])
233
+ print(f"done in {time.time() - t0:.0f} s")
234
+
235
+
236
+ if __name__ == "__main__":
237
+ main()
code/scripts/make_sample.py ADDED
@@ -0,0 +1,97 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Build the PandaSet sample of meteor-p150 from the research ship sample (PandaSet 019 frame 40, CC BY 4.0 + the
4
+ PandaSet Dataset Terms; research/meteor/public_data/ship_sample/pandaset_019_f40, made by its make_ship_sample.py).
5
+ It does NOT ship until the user approves PandaSet-derived samples (PLAN.md 6.3, research/DATASETS.md N5): by default
6
+ (``--dest staging``) it is written to the git-ignored ``staging_samples_pandaset/`` (laid out like ``code/tt_meteor/``,
7
+ where the tests find it, ``tests/paths.py``); ``--dest package`` writes it into ``code/tt_meteor/`` (the swap-in of
8
+ ``staging_samples_pandaset/README.md``). The shipped sample is the synthetic ``samples/synthetic_8cam.json``
9
+ (``code/scripts/make_synthetic_sample.py``). Paths below are relative to the destination:
10
+
11
+ - ``samples/pandaset_019_f40/``: the METEOR demo-scene layout (scenes.txt, manifest.json, the 8 camera
12
+ JPEGs incl. the all-zero absent BACK_NARROW, ego_motion.npz, gt_boxes.json), LICENSE.txt, README.md, SHA256SUMS;
13
+ - ``calib/pandaset_019.json``: the rig as a calibration preset (K at 768x432, T_ref_from_camera =
14
+ camera optical frame -> ego; BACK_NARROW carries its donor's calibration, as the manifest does);
15
+ - ``samples/pandaset_019_f40.json``: the request manifest (``model(**load_sample(path))``).
16
+
17
+ Usage: python code/scripts/make_sample.py [--src <ship sample dir>] [--dest staging|package] (numpy; no device)
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import argparse
22
+ import hashlib
23
+ import json
24
+ import shutil
25
+ from pathlib import Path
26
+
27
+ import numpy as np
28
+
29
+ BUNDLE = Path(__file__).resolve().parents[2]
30
+ PKG = BUNDLE / "code" / "tt_meteor"
31
+ SRC = Path("/home/ubuntu/experiments/tt-models/research/meteor/public_data/ship_sample/pandaset_019_f40")
32
+ CAMERAS = ("CAM_FRONT_WIDE", "CAM_FRONT_LEFT", "CAM_FRONT_RIGHT", "CAM_BACK_WIDE", "CAM_BACK_LEFT", "CAM_BACK_RIGHT",
33
+ "CAM_FRONT_NARROW", "CAM_BACK_NARROW")
34
+ SCENE = "pandaset_019_f40"
35
+
36
+
37
+ def sha256(p: Path) -> str:
38
+ return hashlib.sha256(p.read_bytes()).hexdigest()
39
+
40
+
41
+ def main() -> None:
42
+ ap = argparse.ArgumentParser()
43
+ ap.add_argument("--src", type=Path, default=SRC)
44
+ ap.add_argument("--dest", choices=("staging", "package"), default="staging")
45
+ a = ap.parse_args()
46
+ root = PKG if a.dest == "package" else BUNDLE / "staging_samples_pandaset"
47
+ dst = root / "samples" / SCENE
48
+ if dst.exists():
49
+ shutil.rmtree(dst)
50
+ (dst / SCENE / "img").mkdir(parents=True)
51
+ imgs = [f"{SCENE}/img/{p.name}" for p in sorted((a.src / SCENE / "img").iterdir())]
52
+ for rel in ["scenes.txt", "LICENSE.txt", f"{SCENE}/manifest.json", f"{SCENE}/ego_motion.npz",
53
+ f"{SCENE}/gt_boxes.json"] + imgs:
54
+ shutil.copyfile(a.src / rel, dst / rel)
55
+ m = json.loads((a.src / SCENE / "manifest.json").read_text())
56
+ ego = np.load(a.src / SCENE / "ego_motion.npz")
57
+ readme = (a.src / "README.md").read_text()
58
+ readme = readme.replace("## Run\n", "## Run (this bundle)\n\n```python\nfrom tt_meteor import METEOR, load_sample\n"
59
+ "kwargs = load_sample('code/tt_meteor/samples/pandaset_019_f40.json')\n"
60
+ "```\n\nThe CPU reference of the bundle (`tt_meteor.reference.pipeline.MeteorReference`) "
61
+ "and the research tools read the scene layout directly:\n\n", 1)
62
+ readme = readme.replace("## Expected outputs (`expected/`)", "## Expected outputs (research `expected/`, not "
63
+ "copied here; the bundle's own reference output is "
64
+ "`../pandaset_019_f40.reference.json`)", 1)
65
+ (dst / "README.md").write_text(readme)
66
+ files = sorted(p for p in dst.rglob("*") if p.is_file())
67
+ (dst / "SHA256SUMS").write_text("".join(f"{sha256(p)} {p.relative_to(dst).as_posix()}\n" for p in files))
68
+
69
+ cams = {}
70
+ for c in CAMERAS:
71
+ cc = m["cams"][c]
72
+ cams[c] = {"intrinsics": cc["K"], "T_ref_from_camera": cc["T_ego_cam"], "image_size": [768, 432]}
73
+ if not cc.get("present", True):
74
+ cams[c]["absent"] = f"zero image + donor {cc.get('donor')} calibration (bevlane/dataset.py:131-145)"
75
+ preset = {"frame_id": "base_link",
76
+ "_source": f"research/meteor/public_data/ship_sample/{SCENE}/{SCENE}/manifest.json (PandaSet 019, "
77
+ "CC BY 4.0 + PandaSet Dataset Terms; rig derivation: research/meteor/public_data/README.md)",
78
+ "_ego_frame": m.get("ego_frame"),
79
+ "cameras": cams}
80
+ (root / "calib").mkdir(parents=True, exist_ok=True)
81
+ (root / "calib" / "pandaset_019.json").write_text(json.dumps(preset, indent=1) + "\n")
82
+ fr = m["frames"][0]
83
+ req = {"images": {c: f"{SCENE}/{SCENE}/{fr['imgs'][c]}" for c in CAMERAS},
84
+ "calibration": {"preset": "pandaset_019"},
85
+ "ego_speed": float(ego["v0"][0]),
86
+ "stream": {"id": "pandaset_019", "T_world_from_ego": dict(zip(("x", "y", "yaw"), (float(v) for v in
87
+ ego["pose"][0])),
88
+ z=0.0, roll=0.0, pitch=0.0)},
89
+ "_source": "PandaSet 019 frame 40 (Scale AI and Hesai, https://pandaset.org), CC BY 4.0 + PandaSet Dataset "
90
+ f"Terms ({SCENE}/LICENSE.txt); METEOR 8-slot layout, BACK_NARROW absent (all-zero image)"}
91
+ (root / "samples" / f"{SCENE}.json").write_text(json.dumps(req, indent=1) + "\n")
92
+ print(f"wrote {dst} ({sum(p.stat().st_size for p in files) / 1e6:.2f} MB), calib/pandaset_019.json, "
93
+ f"samples/{SCENE}.json")
94
+
95
+
96
+ if __name__ == "__main__":
97
+ main()
code/scripts/make_synthetic_sample.py ADDED
@@ -0,0 +1,508 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """The shipped sample of meteor-p150: a synthetic eight-camera frame rendered here (data generated by this repository,
4
+ Apache-2.0; no third-party pixels or calibration). CPU only, research venv (torch, OpenCV)::
5
+
6
+ cd bundles/meteor-p150; PY=/home/ubuntu/experiments/tt-models/tools/research-venv/bin/python
7
+ OMP_NUM_THREADS=6 PYTHONPATH=$PWD/code $PY code/scripts/make_synthetic_sample.py render # ~3 min
8
+ OMP_NUM_THREADS=6 PYTHONPATH=$PWD/code $PY code/scripts/make_synthetic_sample.py optimise # ~15 s per step
9
+ OMP_NUM_THREADS=6 PYTHONPATH=$PWD/code $PY code/scripts/make_synthetic_sample.py finalise # PNGs + reference
10
+
11
+ (``--work`` holds the intermediate arrays, default ``logs/synthetic_sample/``; ``optimise`` resumes from its
12
+ ``images_f32.npy``.) The shipped sample came from ``render``, ``optimise --steps 80`` (stopped after 28 steps to drop
13
+ the pedestrian beyond 20 m from the targets), ``optimise --steps 60 --lr 0.015`` and ``finalise``.
14
+
15
+ A generic METEOR-like rig (``calib/synthetic_8cam.json``: the eight slots in METEOR's order, a 98 deg wide camera
16
+ front and back, four corner cameras pitched 25 deg down, two 30.4 deg narrow cameras; fy = 0.87 fx like METEOR's
17
+ vertically squashed training images; mounted 1.9 m above the road) looks at a procedural street scene that is
18
+ ray-cast per pixel (2x2 supersampled, INTER_AREA down to 768x432): a straight two-lane road with lane lines and a
19
+ stop line, kerbs, pavements, building facades, box-shaped vehicles and pedestrians. All eight cameras are present.
20
+ Plain renders give the network too little to detect (the strongest 3D heatmap cell was 0.22 < the 0.35 vehicle
21
+ threshold), so ``optimise`` then changes the object pixels only (gradient ascent through the fp32 CPU reference: image
22
+ encoder, lift, BEV detector, box refiner; the image rounded to uint8 by a straight-through estimator) until every
23
+ target object within range (:data:`DROP_FAR`) clears :data:`GOAL` at its cell and nothing else clears :data:`CEIL`:
24
+ a stable smoke reference whose published boxes sit far from the thresholds.
25
+
26
+ Writes:
27
+
28
+ - ``code/tt_meteor/calib/synthetic_8cam.json``: the rig as a calibration preset (served in ``/info``);
29
+ - ``code/tt_meteor/samples/synthetic_8cam/<CAM>.png``: the eight 768x432 RGB images (lossless, so every decoder
30
+ gives the same pixels);
31
+ - ``code/tt_meteor/samples/synthetic_8cam.json``: the request manifest (``model(**load_sample(path))``);
32
+ - ``code/tt_meteor/samples/synthetic_8cam.reference.json``: the ``/predict`` body of the fp32 CPU reference
33
+ (``tt_meteor.reference.pipeline.MeteorReference``, PCC 1.0 vs ONNX Runtime) on that request as a fresh stream,
34
+ the container smoke's stored reference (``--no-reference`` skips it: 40-60 s of CPU);
35
+ - ``code/tt_meteor/tests/goldens/synthetic_8cam_inputs.json``: sha256 of the graph feed (``imgs``, ``K``,
36
+ ``T_cam_ego``, ``v0``) the request produces, so a host test notices any change of the decoding / resize path.
37
+ """
38
+ from __future__ import annotations
39
+
40
+ import argparse
41
+ import hashlib
42
+ import json
43
+ import math
44
+ import sys
45
+ import time
46
+ from pathlib import Path
47
+
48
+ import numpy as np
49
+
50
+ BUNDLE = Path(__file__).resolve().parents[2]
51
+ sys.path.insert(0, str(BUNDLE / "code"))
52
+ PKG = BUNDLE / "code" / "tt_meteor"
53
+ NAME = "synthetic_8cam"
54
+ W, H, SS = 768, 432, 2 # network input size, supersampling factor
55
+ EGO_SPEED = 8.0 # m/s
56
+ SEED = 20261010
57
+
58
+ # slot -> (x, y, z of the camera in base_link, yaw deg CCW from +x, pitch deg down, horizontal FOV deg)
59
+ RIG = {
60
+ "CAM_FRONT_WIDE": (2.10, 0.00, 1.90, 0.0, 2.0, 98.0),
61
+ "CAM_FRONT_LEFT": (1.90, 0.85, 1.90, 60.0, 25.0, 98.0),
62
+ "CAM_FRONT_RIGHT": (1.90, -0.85, 1.90, -60.0, 25.0, 98.0),
63
+ "CAM_BACK_WIDE": (-0.90, 0.00, 1.90, 180.0, 2.0, 98.0),
64
+ "CAM_BACK_LEFT": (-0.70, 0.85, 1.90, 120.0, 25.0, 98.0),
65
+ "CAM_BACK_RIGHT": (-0.70, -0.85, 1.90, -120.0, 25.0, 98.0),
66
+ "CAM_FRONT_NARROW": (2.10, 0.00, 1.88, 0.0, 1.0, 30.4),
67
+ "CAM_BACK_NARROW": (-0.90, 0.00, 1.88, 180.0, 2.0, 30.4),
68
+ }
69
+ FY_OVER_FX = 0.87 # METEOR's training images are squashed vertically (2880x1860 -> 768x432)
70
+
71
+ # the street (base_link: x forward, y left, z up, origin on the road below the rear axle; left-hand traffic, ego in
72
+ # the left lane): lane edges and lines (y, half width, kind)
73
+ ROAD_L, ROAD_R = 1.75, -5.25 # carriageway edges (ego lane 1.75 .. -1.75, oncoming lane -1.75 .. -5.25)
74
+ KERB = 0.15 # kerb height
75
+ WALK_L, WALK_R = 5.0, -8.5 # pavement outer edges = building facades
76
+ # boxes: (x, y, yaw deg, length, width, height, rgb, kind)
77
+ OBJECTS = [
78
+ (11.0, 0.0, 0.0, 4.4, 1.8, 1.5, (180, 30, 35), "car"), # car ahead in the ego lane
79
+ (19.0, -3.5, 180.0, 4.6, 1.85, 1.55, (40, 70, 160), "car"), # oncoming cars
80
+ (29.0, -3.4, 180.0, 4.5, 1.8, 1.5, (150, 150, 155), "car"),
81
+ (-7.0, -3.5, 180.0, 4.4, 1.8, 1.5, (200, 170, 40), "car"),
82
+ (42.0, 0.1, 0.0, 8.5, 2.4, 3.0, (225, 225, 220), "truck"), # truck further ahead
83
+ (-13.0, 0.0, 0.0, 4.3, 1.75, 1.45, (220, 220, 225), "car"), # car behind
84
+ (6.0, 3.2, 0.0, 4.2, 1.75, 1.45, (60, 60, 65), "car"), # parked car on the left (on the pavement edge)
85
+ (-4.0, -6.8, 90.0, 0.5, 0.6, 1.7, (40, 40, 90), "ped"), # pedestrian on the right pavement
86
+ (9.0, 4.0, 0.0, 0.5, 0.6, 1.75, (120, 40, 40), "ped"), # pedestrian on the left pavement
87
+ (22.0, -7.2, 0.0, 0.5, 0.6, 1.65, (30, 90, 60), "ped"),
88
+ ]
89
+ STOP_LINE_X = 34.0 # stop line across the ego lane
90
+ CROSSWALK_X = (36.0, 40.0) # zebra across the road
91
+
92
+
93
+ def object_parts():
94
+ """Every object as oriented boxes: (centre x, y, z, yaw deg, half extents (l, w, h), rgb, material)."""
95
+ out = []
96
+ for (bx, by, yaw, L, Wd, Hh, col, kind) in OBJECTS:
97
+ z0 = KERB if (by > ROAD_L or by < ROAD_R) else 0.0 # on the pavement or on the road
98
+ c, s = math.cos(math.radians(yaw)), math.sin(math.radians(yaw))
99
+
100
+ def at(dx, dy, dz, half, rgb, mat, yaw_=yaw):
101
+ out.append((bx + c * dx - s * dy, by + s * dx + c * dy, z0 + dz, yaw_, half, rgb, mat))
102
+
103
+ if kind == "car":
104
+ hb = 0.55 * Hh # lower body 0.25 .. 0.25 + hb
105
+ at(0, 0, 0.25 + hb / 2, (L / 2, Wd / 2, hb / 2), col, "paint")
106
+ at(-0.15 * L, 0, 0.25 + hb + (Hh - 0.25 - hb) / 2, (0.28 * L, 0.44 * Wd, (Hh - 0.25 - hb) / 2),
107
+ (55, 65, 80), "glass") # cabin
108
+ at(-0.15 * L, 0, Hh - 0.02, (0.26 * L, 0.42 * Wd, 0.03), col, "paint") # roof
109
+ for sx in (0.32, -0.32):
110
+ for sy in (1, -1):
111
+ at(sx * L, sy * (Wd / 2 - 0.12), 0.32, (0.32, 0.13, 0.32), (22, 22, 24), "rubber")
112
+ for sy in (1, -1): # lamps
113
+ at(L / 2 - 0.02, sy * 0.33 * Wd, 0.25 + 0.7 * hb, (0.04, 0.16, 0.06), (240, 240, 215), "lamp")
114
+ at(-L / 2 + 0.02, sy * 0.33 * Wd, 0.25 + 0.7 * hb, (0.04, 0.16, 0.06), (210, 25, 25), "lamp")
115
+ elif kind == "truck":
116
+ cab = 1.9
117
+ at(L / 2 - cab / 2, 0, 0.35 + (Hh - 0.6 - 0.35) / 2, (cab / 2, Wd / 2, (Hh - 0.6 - 0.35) / 2), (40, 90, 160), "paint")
118
+ at(L / 2 - 0.05, 0, Hh - 1.3, (0.06, 0.45 * Wd, 0.35), (55, 65, 80), "glass")
119
+ at(-cab / 2, 0, 0.6 + (Hh - 0.6) / 2, ((L - cab) / 2 - 0.05, Wd / 2, (Hh - 0.6) / 2), col, "paint")
120
+ for sx in (0.38, 0.0, -0.33):
121
+ for sy in (1, -1):
122
+ at(sx * L, sy * (Wd / 2 - 0.15), 0.45, (0.45, 0.15, 0.45), (22, 22, 24), "rubber")
123
+ else: # pedestrian
124
+ at(0, 0, 0.42, (0.12, 0.2, 0.42), (40, 40, 55), "cloth")
125
+ at(0, 0, 0.84 + (Hh - 1.07) / 2, (0.14, 0.24, (Hh - 1.07) / 2), col, "cloth")
126
+ at(0, 0, Hh - 0.115, (0.1, 0.09, 0.115), (215, 170, 140), "skin")
127
+ return out
128
+
129
+
130
+ def shadow(px: np.ndarray, py: np.ndarray) -> np.ndarray:
131
+ """Soft contact shadows of the objects on the ground: a multiplicative factor in (0, 1]."""
132
+ f = np.ones(px.shape, np.float32)
133
+ for (bx, by, yaw, L, Wd, Hh, col, kind) in OBJECTS:
134
+ c, s = math.cos(math.radians(yaw)), math.sin(math.radians(yaw))
135
+ dx, dy = px - bx, py - by
136
+ lx, ly = np.abs(c * dx + s * dy) - L / 2, np.abs(-s * dx + c * dy) - Wd / 2
137
+ out = np.hypot(np.maximum(lx, 0), np.maximum(ly, 0)) + np.minimum(np.maximum(lx, ly), 0)
138
+ f *= (1.0 - 0.75 * np.clip(1.0 - (out + 0.15) / 0.6, 0, 1)).astype(np.float32)
139
+ return f
140
+
141
+
142
+ def rotation(yaw_deg: float, pitch_deg: float) -> np.ndarray:
143
+ """camera optical frame (x right, y down, z forward) -> base_link rotation."""
144
+ p, y = math.radians(pitch_deg), math.radians(yaw_deg)
145
+ f = np.array([math.cos(p) * math.cos(y), math.cos(p) * math.sin(y), -math.sin(p)])
146
+ r = np.array([math.sin(y), -math.cos(y), 0.0])
147
+ d = np.cross(f, r)
148
+ return np.stack([r, d, f], axis=1)
149
+
150
+
151
+ def intrinsics(hfov_deg: float, w: int = W, h: int = H) -> np.ndarray:
152
+ fx = (w / 2.0) / math.tan(math.radians(hfov_deg) / 2.0)
153
+ return np.array([[fx, 0.0, w / 2.0], [0.0, FY_OVER_FX * fx, h / 2.0], [0.0, 0.0, 1.0]])
154
+
155
+
156
+ def value_noise(x: np.ndarray, y: np.ndarray, cell: float, seed: int) -> np.ndarray:
157
+ """Smooth hash noise in [0, 1) on a ground-plane lattice of ``cell`` metres."""
158
+ gx, gy = np.floor(x / cell), np.floor(y / cell)
159
+ fx, fy = x / cell - gx, y / cell - gy
160
+
161
+ def h(ix, iy):
162
+ v = (ix.astype(np.int64) * 73856093) ^ (iy.astype(np.int64) * 19349663) ^ seed
163
+ v = (v * 2654435761) & 0xFFFFFFFF
164
+ return (v % 10007) / 10007.0
165
+
166
+ sx, sy = fx * fx * (3 - 2 * fx), fy * fy * (3 - 2 * fy)
167
+ a, b, c, d = h(gx, gy), h(gx + 1, gy), h(gx, gy + 1), h(gx + 1, gy + 1)
168
+ return (a * (1 - sx) + b * sx) * (1 - sy) + (c * (1 - sx) + d * sx) * sy
169
+
170
+
171
+ def ground_colour(px: np.ndarray, py: np.ndarray, dist: np.ndarray) -> np.ndarray:
172
+ n = 0.6 * value_noise(px, py, 0.35, SEED) + 0.4 * value_noise(px, py, 2.0, SEED + 1)
173
+ rgb = np.empty(px.shape + (3,), np.float32)
174
+ road = (py <= ROAD_L) & (py >= ROAD_R)
175
+ asphalt = 78 + 34 * n
176
+ rgb[...] = np.stack([asphalt, asphalt, asphalt + 4], -1)[...]
177
+ walk = ~road
178
+ tile = 150 + 30 * n + 12 * ((np.floor(px / 0.6) + np.floor(py / 0.6)) % 2)
179
+ rgb[walk] = np.stack([tile, tile - 6, tile - 14], -1)[walk]
180
+ # markings on the carriageway: edge lines (solid), centre line (dashed 5 m / 5 m), stop line, zebra
181
+ white = np.zeros(px.shape, bool)
182
+ white |= road & (np.abs(py - (ROAD_L - 0.2)) < 0.075)
183
+ white |= road & (np.abs(py - (ROAD_R + 0.2)) < 0.075)
184
+ white |= (np.abs(py + 1.75) < 0.075) & (np.mod(px, 10.0) < 5.0)
185
+ white |= (py <= 1.6) & (py >= -1.75) & (np.abs(px - STOP_LINE_X) < 0.225)
186
+ zebra = road & (px >= CROSSWALK_X[0]) & (px <= CROSSWALK_X[1]) & (np.mod(py - ROAD_R, 0.9) < 0.45)
187
+ white |= zebra
188
+ wv = 215 + 25 * n
189
+ rgb[white] = np.stack([wv, wv, wv - 5], -1)[white]
190
+ rgb *= shadow(px, py)[..., None]
191
+ # haze with distance
192
+ fog = np.clip(dist / 120.0, 0, 0.6)[..., None]
193
+ return rgb * (1 - fog) + np.array([170, 180, 195], np.float32) * fog
194
+
195
+
196
+ def render_camera(R: np.ndarray, t: np.ndarray, K: np.ndarray):
197
+ """Ray-cast the street for one camera; returns (uint8 (H*SS, W*SS, 3), bool object mask (H*SS, W*SS))."""
198
+ w, h = W * SS, H * SS
199
+ Ks = K.copy()
200
+ Ks[:2] *= SS
201
+ u, v = np.meshgrid(np.arange(w) + 0.5, np.arange(h) + 0.5)
202
+ rays_c = np.stack([(u - Ks[0, 2]) / Ks[0, 0], (v - Ks[1, 2]) / Ks[1, 1], np.ones_like(u)], -1)
203
+ d = rays_c @ R.T
204
+ d /= np.linalg.norm(d, axis=-1, keepdims=True)
205
+ best = np.full(d.shape[:2], np.inf)
206
+ rgb = np.zeros(d.shape, np.float32)
207
+ obj = np.zeros(d.shape[:2], bool)
208
+ # sky
209
+ up = np.clip(d[..., 2], -1, 1)
210
+ sky = np.stack([120 + 60 * (1 - up), 160 + 50 * (1 - up), 225 + 20 * (1 - up)], -1)
211
+ rgb[...] = sky
212
+ # ground (z = 0; pavements at kerb height)
213
+ with np.errstate(divide="ignore", invalid="ignore"):
214
+ for z0, keep in ((0.0, lambda py: (py <= ROAD_L) & (py >= ROAD_R)),
215
+ (KERB, lambda py: (py > ROAD_L) | (py < ROAD_R))):
216
+ tg = (z0 - t[2]) / d[..., 2]
217
+ ok = (tg > 0) & np.isfinite(tg)
218
+ px, py = t[0] + tg * d[..., 0], t[1] + tg * d[..., 1]
219
+ ok &= keep(py) & (py <= WALK_L) & (py >= WALK_R)
220
+ ok &= tg < best
221
+ col = ground_colour(px, py, tg)
222
+ rgb[ok], best[ok] = col[ok], tg[ok]
223
+ # kerb faces (vertical planes y = ROAD_L / ROAD_R, 0 .. KERB)
224
+ for yk in (ROAD_L, ROAD_R):
225
+ tk = (yk - t[1]) / d[..., 1]
226
+ zk = t[2] + tk * d[..., 2]
227
+ ok = (tk > 0) & (zk >= 0) & (zk <= KERB) & (tk < best)
228
+ rgb[ok], best[ok] = np.array([185, 180, 170], np.float32), tk[ok]
229
+ # facades (y = WALK_L / WALK_R), 6-14 m tall blocks with windows
230
+ for yw, sd in ((WALK_L, 3), (WALK_R, 4)):
231
+ tw = (yw - t[1]) / d[..., 1]
232
+ xw, zw = t[0] + tw * d[..., 0], t[2] + tw * d[..., 2]
233
+ block = np.floor(xw / 12.0)
234
+ hb = 6 + 8 * value_noise(block * 12.0, np.zeros_like(block), 12.0, SEED + sd)
235
+ ok = (tw > 0) & (zw >= KERB) & (zw <= hb) & (tw < best)
236
+ base = 0.55 + 0.45 * value_noise(block * 12.0, np.full_like(block, 5.0), 12.0, SEED + 10 + sd)
237
+ wall = np.stack([200 * base, 170 * base, 140 * base], -1)
238
+ win = (np.mod(xw, 3.0) > 0.9) & (np.mod(zw - 0.8, 3.2) < 1.6) & (zw > 3.0)
239
+ wall[win] = np.array([70, 90, 115], np.float32)
240
+ door = (zw < 2.4) & (np.mod(xw, 12.0) > 5.0) & (np.mod(xw, 12.0) < 6.6)
241
+ wall[door] = np.array([60, 45, 35], np.float32)
242
+ fog = np.clip(tw / 120.0, 0, 0.6)[..., None]
243
+ wall = wall * (1 - fog) + np.array([170, 180, 195], np.float32) * fog
244
+ rgb[ok], best[ok] = wall[ok], tw[ok]
245
+ # objects: every part is an oriented box (slab test in the part frame)
246
+ light = np.array([0.4, 0.3, 0.87])
247
+ light /= np.linalg.norm(light)
248
+ for (cx, cy, cz, yaw, half, col, mat) in object_parts():
249
+ c, s = math.cos(math.radians(yaw)), math.sin(math.radians(yaw))
250
+ Rb = np.array([[c, -s, 0], [s, c, 0], [0, 0, 1.0]])
251
+ o = Rb.T @ (t - np.array([cx, cy, cz]))
252
+ dl = d @ Rb
253
+ half = np.asarray(half, np.float64)
254
+ with np.errstate(divide="ignore", invalid="ignore"):
255
+ t1 = (-half - o) / dl
256
+ t2 = (half - o) / dl
257
+ tmin = np.nanmax(np.minimum(t1, t2), axis=-1)
258
+ tmax = np.nanmin(np.maximum(t1, t2), axis=-1)
259
+ hit = (tmax >= tmin) & (tmin > 0) & (tmin < best)
260
+ if not hit.any():
261
+ continue
262
+ p = o + tmin[..., None] * dl # hit point, part frame
263
+ ax = np.argmax(np.abs(p) / half, axis=-1)
264
+ nrm_l = np.zeros_like(p)
265
+ np.put_along_axis(nrm_l, ax[..., None], np.sign(np.take_along_axis(p, ax[..., None], -1)), -1)
266
+ nrm = nrm_l @ Rb.T
267
+ lam = np.clip(nrm @ light, 0, 1)
268
+ base = np.broadcast_to(np.array(col, np.float32), p.shape).copy()
269
+ if mat == "glass": # sky reflection on upward-tilted views
270
+ refl = np.clip(0.35 + 0.65 * np.abs(nrm[..., 2]), 0, 1)
271
+ shade = 0.5 + 0.3 * refl
272
+ elif mat == "paint":
273
+ shade = 0.35 + 0.65 * lam + 0.08 * (value_noise(p[..., 0], p[..., 2], 0.4, SEED + 7) - 0.5)
274
+ else:
275
+ shade = 0.4 + 0.6 * lam
276
+ cshade = base * shade[..., None]
277
+ rgb[hit], best[hit] = cshade[hit], tmin[hit]
278
+ obj |= hit
279
+ obj &= np.isfinite(best)
280
+ return np.clip(rgb, 0, 255).astype(np.uint8), obj
281
+
282
+
283
+ def build_rig():
284
+ cams = {}
285
+ for name, (x, y, z, yaw, pitch, fov) in RIG.items():
286
+ R = rotation(yaw, pitch)
287
+ T = np.eye(4)
288
+ T[:3, :3], T[:3, 3] = R, (x, y, z)
289
+ cams[name] = {"intrinsics": intrinsics(fov).round(4).tolist(), "T_ref_from_camera": T.round(9).tolist(),
290
+ "image_size": [W, H]}
291
+ return cams
292
+
293
+
294
+ def hm_cell(x: float, y: float):
295
+ """base_link (x, y) -> (row, col) of the 400x250 detection grid (rows x = 79.8 .. -79.8, cols y = 49.8 .. -49.8)."""
296
+ return int(round((79.8 - x) / 0.4)), int(round((49.8 - y) / 0.4))
297
+
298
+
299
+ # what the optimisation asks of the 3D heads: every OBJECTS entry is a target (vehicle: car / truck, VRU: ped)
300
+ GOAL = {0: 0.70, 1: 0.50} # sigmoid at the target cell (thresholds: vehicle 0.35, VRU 0.15)
301
+ CEIL = {0: 0.20, 1: 0.07} # sigmoid ceiling everywhere else (outside 1.6 m of a target of that class)
302
+ DROP_FAR = {0: 40.0, 1: 20.0} # targets beyond this range are scenery (pushed below the ceiling instead)
303
+
304
+
305
+ def targets():
306
+ out = []
307
+ for (bx, by, yaw, L, Wd, Hh, col, kind) in OBJECTS:
308
+ cls = 0 if kind in ("car", "truck") else 1
309
+ if math.hypot(bx, by) <= DROP_FAR[cls]:
310
+ out.append((cls, bx, by))
311
+ return out
312
+
313
+
314
+ def differentiable_hm(net, x, K, T):
315
+ """The 3D heatmap logits [1, 2, 400, 250] of the fp32 reference from float images x [8, 3, 432, 768] in [0, 1]
316
+ (the graph's /255 input normalisation; the same modules as ``MeteorNet.forward``, with gradients)."""
317
+ f = net.image_encoder(x)
318
+ seg = net.seg2d_head(f)
319
+ _, dprob = net.depth_head(f)
320
+ ctx = net.context(f, seg)
321
+ raw = net.lift(ctx, dprob, K, T)
322
+ _, hm_pre, reg_pre = net.det_stem(raw)
323
+ hm, _ = net.box_refiner(hm_pre, reg_pre)
324
+ return hm
325
+
326
+
327
+ def stage_render(work: Path) -> None:
328
+ import cv2
329
+
330
+ t0 = time.time()
331
+ cams = build_rig()
332
+ imgs, masks = [], []
333
+ for name in RIG:
334
+ c = cams[name]
335
+ T = np.asarray(c["T_ref_from_camera"])
336
+ img, obj = render_camera(T[:3, :3], T[:3, 3], np.asarray(c["intrinsics"]))
337
+ imgs.append(cv2.resize(img, (W, H), interpolation=cv2.INTER_AREA))
338
+ m = cv2.resize(obj.astype(np.uint8) * 255, (W, H), interpolation=cv2.INTER_AREA) > 0
339
+ masks.append(cv2.dilate(m.astype(np.uint8), np.ones((9, 9), np.uint8)) > 0)
340
+ work.mkdir(parents=True, exist_ok=True)
341
+ np.save(work / "base_u8.npy", np.stack(imgs)) # [8, H, W, 3] RGB
342
+ np.save(work / "mask.npy", np.stack(masks)) # [8, H, W]
343
+ print(f"render: {time.time() - t0:.0f} s, object pixels {np.stack(masks).mean():.3%}", flush=True)
344
+
345
+
346
+ def _load_net(weights_dir, threads):
347
+ import torch
348
+
349
+ from tt_meteor.reference.pipeline import MeteorReference
350
+ from tt_meteor.reference.weights import MeteorWeights
351
+
352
+ torch.set_num_threads(threads)
353
+ return MeteorReference(weights=MeteorWeights(weights_dir), threads=threads) # plain tensors: no weight grads
354
+
355
+
356
+ def _calib_tensors():
357
+ import torch
358
+
359
+ from tt_meteor.host.preprocess import invert_extrinsics
360
+
361
+ cams = build_rig()
362
+ K = np.stack([np.asarray(cams[n]["intrinsics"], np.float32) for n in RIG])[None]
363
+ T = np.stack([invert_extrinsics(cams[n]["T_ref_from_camera"]) for n in RIG])[None]
364
+ return torch.from_numpy(K), torch.from_numpy(T)
365
+
366
+
367
+ def stage_optimise(work: Path, weights_dir, threads: int, steps: int, lr: float) -> None:
368
+ """Gradient ascent on the object pixels (and only those) until every target clears its goal on the uint8-rounded
369
+ images and nothing else clears its ceiling."""
370
+ import torch
371
+
372
+ ref = _load_net(weights_dir, threads)
373
+ K, T = _calib_tensors()
374
+ base = torch.from_numpy(np.load(work / "base_u8.npy").astype(np.float32) / 255.0).permute(0, 3, 1, 2)
375
+ mask = torch.from_numpy(np.load(work / "mask.npy")).float()[:, None]
376
+ start = work / "images_f32.npy"
377
+ x0 = torch.from_numpy(np.load(start)) if start.is_file() else base.clone()
378
+ delta = (x0 - base).clone().requires_grad_(True)
379
+ opt = torch.optim.Adam([delta], lr=lr)
380
+ tg = targets()
381
+ logit = lambda p: math.log(p / (1 - p))
382
+ keep = torch.ones(1, 2, 400, 250, dtype=torch.bool)
383
+ rr, cc = torch.meshgrid(torch.arange(400), torch.arange(250), indexing="ij")
384
+ for cls, bx, by in tg:
385
+ r, c = hm_cell(bx, by)
386
+ keep[0, cls] &= (rr - r) ** 2 + (cc - c) ** 2 > 16 # 1.6 m around a target of its class
387
+ ceil = torch.tensor([logit(CEIL[0]), logit(CEIL[1])]).reshape(1, 2, 1, 1)
388
+ log = open(work / "optimise.log", "a")
389
+ for step in range(steps):
390
+ t0 = time.time()
391
+ x = (base + delta * mask).clamp(0, 1)
392
+ xq = x + ((x * 255).round() / 255 - x).detach() # straight-through uint8 rounding
393
+ hm = differentiable_hm(ref.net, xq, K, T)
394
+ z = []
395
+ loss = torch.zeros(())
396
+ for cls, bx, by in tg:
397
+ r, c = hm_cell(bx, by)
398
+ zt = hm[0, cls, r - 1:r + 2, c - 1:c + 2].max() # the 3x3 peak at the target
399
+ z.append(float(torch.sigmoid(zt)))
400
+ loss = loss + torch.nn.functional.softplus(logit(GOAL[cls]) + 0.3 - zt) * 2.0
401
+ over = torch.relu(hm - ceil)[keep]
402
+ loss = loss + (over ** 2).sum() * 0.5
403
+ opt.zero_grad()
404
+ loss.backward()
405
+ opt.step()
406
+ with torch.no_grad():
407
+ delta.clamp_(-0.6, 0.6)
408
+ n_over = int((over > 0).sum())
409
+ ok = all(p >= GOAL[cls] for p, (cls, _, _) in zip(z, tg)) and n_over == 0
410
+ line = (f"step {step}: loss {float(loss):.4f}, targets {[round(p, 3) for p in z]}, cells over the ceiling "
411
+ f"{n_over}, max other {float(torch.sigmoid(hm.detach()[keep]).max()):.3f}, {time.time() - t0:.0f} s")
412
+ print(line, flush=True)
413
+ log.write(line + "\n")
414
+ log.flush()
415
+ np.save(work / "images_f32.npy", x.detach().numpy())
416
+ if ok:
417
+ print("all targets met", flush=True)
418
+ break
419
+
420
+
421
+ def stage_finalise(work: Path, weights_dir, threads: int, reference: bool) -> None:
422
+ import cv2
423
+
424
+ t0 = time.time()
425
+ cams = build_rig()
426
+ preset = {"frame_id": "base_link",
427
+ "_source": "generated by code/scripts/make_synthetic_sample.py (a generic METEOR-like 8-camera rig of "
428
+ "round numbers; no third-party data), Apache-2.0",
429
+ "_rig": {k: dict(zip(("x", "y", "z", "yaw_deg", "pitch_down_deg", "hfov_deg"), v)) for k, v in RIG.items()},
430
+ "cameras": cams}
431
+ (PKG / "calib" / f"{NAME}.json").write_text(json.dumps(preset, indent=1) + "\n")
432
+ src = work / "images_f32.npy"
433
+ if src.is_file():
434
+ imgs = np.round(np.load(src) * 255).clip(0, 255).astype(np.uint8).transpose(0, 2, 3, 1)
435
+ else:
436
+ imgs = np.load(work / "base_u8.npy")
437
+ out = PKG / "samples" / NAME
438
+ out.mkdir(parents=True, exist_ok=True)
439
+ for name, img in zip(RIG, imgs):
440
+ cv2.imwrite(str(out / f"{name}.png"), np.ascontiguousarray(img[:, :, ::-1]), [cv2.IMWRITE_PNG_COMPRESSION, 9])
441
+ req = {"images": {name: f"{NAME}/{name}.png" for name in RIG},
442
+ "calibration": {"preset": NAME},
443
+ "ego_speed": EGO_SPEED,
444
+ "stream": {"id": NAME, "reset": True, "T_world_from_ego": {"x": 0.0, "y": 0.0, "z": 0.0, "roll": 0.0,
445
+ "pitch": 0.0, "yaw": 0.0}},
446
+ "_source": "synthetic street scene rendered and optimised by code/scripts/make_synthetic_sample.py (data "
447
+ "generated by this repository, Apache-2.0); calibration preset calib/synthetic_8cam.json; all "
448
+ "eight cameras present"}
449
+ (PKG / "samples" / f"{NAME}.json").write_text(json.dumps(req, indent=1) + "\n")
450
+ size = sum(p.stat().st_size for p in out.iterdir()) / 1e6
451
+ print(f"wrote samples/{NAME}/ ({size:.2f} MB), samples/{NAME}.json, calib/{NAME}.json", flush=True)
452
+
453
+ from tt_meteor.api import load_sample
454
+ from tt_meteor.host.inputs import prepare_request
455
+
456
+ kw = load_sample(PKG / "samples" / f"{NAME}.json")
457
+ frame = prepare_request(kw["images"], kw["calibration"], kw["ego_speed"], kw.get("stream"))
458
+ assert np.array_equal(frame.imgs[0].transpose(0, 2, 3, 1), imgs), "the PNGs do not read back bit-exactly"
459
+ feed_sha = {k: hashlib.sha256(np.ascontiguousarray(v).tobytes()).hexdigest() for k, v in frame.feed().items()}
460
+ (PKG / "tests" / "goldens" / f"{NAME}_inputs.json").write_text(json.dumps(
461
+ {"_doc": f"sha256 of the graph feed of samples/{NAME}.json (code/scripts/make_synthetic_sample.py)",
462
+ "input_sha256": feed_sha, "present": frame.present.astype(int).tolist()}, indent=1) + "\n")
463
+ if not reference:
464
+ return
465
+ ref = _load_net(weights_dir, threads)
466
+ t1 = time.time()
467
+ body = ref(**kw).to_dict()
468
+ body["timing_ms"] = {}
469
+ (PKG / "samples" / f"{NAME}.reference.json").write_text(json.dumps(body) + "\n")
470
+ labels = {}
471
+ for d in body["detections"]:
472
+ labels[d["label"]] = labels.get(d["label"], 0) + 1
473
+ n2d = sum(len(v) for v in body["detections_2d"].values())
474
+ print(f"reference ({time.time() - t1:.0f} s): {body['num_detections']} 3D boxes {labels} scores "
475
+ f"{[round(d['score'], 3) for d in body['detections']]}, {n2d} 2D boxes, mode {body['plan']['mode']} "
476
+ f"(probs {[round(v, 3) for v in body['plan']['mode_probs']]}), TL {body['traffic_light']['state']}, "
477
+ f"path end {[round(v, 2) for v in body['trajectory'][-1]]}; total {time.time() - t0:.0f} s", flush=True)
478
+
479
+
480
+ def default_weights_dir():
481
+ snaps = Path.home() / ".cache/huggingface/hub/models--AutowareFoundation--meteor/snapshots"
482
+ for cand in sorted(snaps.glob("*")):
483
+ if (cand / "meteor_v157c3Z.onnx").is_file():
484
+ return cand
485
+ return Path("/home/ubuntu/experiments/tt-models/assets/meteor/hf_meteor")
486
+
487
+
488
+ def main() -> None:
489
+ ap = argparse.ArgumentParser()
490
+ ap.add_argument("stage", choices=("render", "optimise", "finalise", "all"))
491
+ ap.add_argument("--work", type=Path, default=BUNDLE / "logs" / "synthetic_sample")
492
+ ap.add_argument("--threads", type=int, default=4)
493
+ ap.add_argument("--steps", type=int, default=60)
494
+ ap.add_argument("--lr", type=float, default=0.02)
495
+ ap.add_argument("--no-reference", action="store_true")
496
+ ap.add_argument("--weights-dir", type=Path, default=None)
497
+ a = ap.parse_args()
498
+ wd = a.weights_dir or default_weights_dir()
499
+ if a.stage in ("render", "all"):
500
+ stage_render(a.work)
501
+ if a.stage in ("optimise", "all"):
502
+ stage_optimise(a.work, wd, a.threads, a.steps, a.lr)
503
+ if a.stage in ("finalise", "all"):
504
+ stage_finalise(a.work, wd, a.threads, not a.no_reference)
505
+
506
+
507
+ if __name__ == "__main__":
508
+ main()
code/scripts/precision_emulation.py ADDED
@@ -0,0 +1,182 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """CPU emulation of the device's activation precision on the class decisions (PORT_LOG.md E20; no device).
3
+
4
+ cd bundles/meteor-p150 && OMP_NUM_THREADS=4 PYTHONPATH=$PWD/code \\
5
+ /home/ubuntu/experiments/tt-models/tools/research-venv/bin/python code/scripts/precision_emulation.py image bb tt Bt
6
+ ... precision_emulation.py bev
7
+ ... precision_emulation.py det meteor_valday_f040 fff bff fbf ffb bbb nbb bFb nFF
8
+ ... precision_emulation.py absent ff ft fT TT tt bb
9
+
10
+ ``image``: the CPU reference's image branch with every conv / resize output rounded to bf16 (``b``), TF32 (``t``: 10
11
+ explicit mantissa bits, what the device's fp32 conv path computes: its unpacker rounds fp32 activations to TF32) or
12
+ kept fp32 (``f``), per (encoder, heads) pair; ``B`` = the stem output alone in bf16 (``ttnn.max_pool2d`` is bf16
13
+ only). Prints the depth / seg2d argmax agreement with the fp32 reference (PLAN 2.13 gate: >= 0.99) and the depth
14
+ top-2 logit margins. ``bev``: the lane decoder + seg refiner from the golden raw BEV with bf16 activations (the convs
15
+ whose outputs the device keeps fp32 stay fp32) -> lane agreement. ``det FRAME RDB ...`` (PORT_LOG.md E22): the det
16
+ stem + merged heads (``D``) and the box refiner (``B``) from the golden raw BEV (``R``) of FRAME, each letter one of
17
+ ``f`` (fp32), ``b`` (bf16 conv / resize outputs), ``t`` (TF32), ``F`` (fp32 convs = three bf16 terms, TF32 resizes:
18
+ the device's terms3 path), and for ``R`` also ``n`` (bf16 after 0.3 % relative noise: the chained device raw) -> the
19
+ strict det3d recall / precision of the decoded boxes vs the golden outputs (``tests/box_agreement.py``) and the hm
20
+ error. ``absent CR ...`` (PORT_LOG.md E23 / I7): the image branch on an absent camera (the all-zero normalised input
21
+ of ``input_norm=imagenet``) with conv outputs ``C`` and resize outputs ``R`` rounded as above, plus ``T`` (TF32 by
22
+ TRUNCATION) -> seg2d / depth agreement with the variant golden's absent camera and with the device's argmax maps
23
+ (``logs/meteor/variants_device_absent_argmax.npz``, written by ``tests/test_variants_device.py``).
24
+ Goldens: ``research/meteor/goldens``.
25
+ Measured (PandaSet 019 f40): image bb 0.9819 / 0.9908, bt 0.9856 / 0.9948, tb 0.9913 / 0.9917, tt 0.9981 / 0.9988,
26
+ Bt 0.9874 / 0.9951; bev lane 0.99966 (bf16) / 0.99996 (fp32). det (valday #40, strict recall / precision): fff
27
+ 1.0 / 1.0, bff 1.0 / 1.0, fbf 0.971 / 1.0, ffb 1.0 / 1.0, bbb 0.914 / 1.0, nbb 0.914 / 1.0, bFb 1.0 / 1.0, nFF
28
+ 1.0 / 1.0 (nuScenes nbb 0.821 / 0.885, nFF 1.0 / 1.0): the bf16 det stem is the loss. absent (seg2d / depth vs the
29
+ reference; vs the device): ff 1.0 / 1.0; fT 0.9975 / 0.9993; tt 0.9966 / 0.9981; TT 0.9856 / 0.9901 (0.9927 / 0.9933
30
+ vs the device); bb 0.9724 / 0.9800; the device 0.9832 / 0.9913 (camera 6): it behaves like TF32-truncated conv outputs.
31
+ """
32
+ from __future__ import annotations
33
+
34
+ import os
35
+ import sys
36
+ from pathlib import Path
37
+
38
+ import numpy as np
39
+
40
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
41
+ GOLD = Path(os.environ.get("METEOR_GOLDENS", "/home/ubuntu/experiments/tt-models/research/meteor/goldens"))
42
+
43
+ FP32_OUT_BEV = {"dec/out/out.3", "lane_branch/lane_branch.6", "refiner/seg/out/out.3"}
44
+
45
+
46
+ def rnd(t, m):
47
+ import torch
48
+
49
+ if m == "f":
50
+ return t
51
+ if m == "T": # TF32 by truncation
52
+ a = t.contiguous().numpy().view(np.uint32) & np.uint32(0xFFFFE000)
53
+ return torch.from_numpy(a.view(np.float32).copy())
54
+ if m == "b":
55
+ return t.to(torch.bfloat16).to(torch.float32)
56
+ a = t.contiguous().numpy().view(np.uint32).astype(np.uint64)
57
+ a = ((a + 0x1000) & 0xFFFFE000).astype(np.uint32) # round to 10 explicit mantissa bits (TF32)
58
+ return torch.from_numpy(a.view(np.float32))
59
+
60
+
61
+ def main() -> None:
62
+ import torch
63
+
64
+ from tt_meteor.reference.model import MeteorNet
65
+ from tt_meteor.reference.weights import MeteorWeights
66
+
67
+ torch.set_num_threads(int(os.environ.get("OMP_NUM_THREADS", "4")))
68
+ what = sys.argv[1] if len(sys.argv) > 1 else "image"
69
+ net = MeteorNet(MeteorWeights(), check_graph=False)
70
+ z = np.load(GOLD / "pandaset_019_f40" / "taps.npz")
71
+ cur = {"m": "b", "stem": None}
72
+ conv0, resize0 = net.conv, net.resize
73
+
74
+ def conv(x, module, relu=False):
75
+ y = conv0(x, module, relu)
76
+ if what == "bev" and module in FP32_OUT_BEV:
77
+ return y
78
+ if module == "stem/stem.0" and cur["stem"]:
79
+ return rnd(y, cur["stem"])
80
+ return rnd(y, cur["m"])
81
+
82
+ if what == "det":
83
+ return det(net, sys.argv[2], sys.argv[3:])
84
+ if what == "absent":
85
+ return absent(net, sys.argv[2:])
86
+ net.conv = conv
87
+ net.resize = lambda x, hw: rnd(resize0(x, hw), cur["m"])
88
+ with torch.inference_mode():
89
+ if what == "bev":
90
+ raw = rnd(torch.from_numpy(z["bev.raw"].astype(np.float32)), "b")
91
+ for m in ("b", "f"):
92
+ cur["m"] = m
93
+ ll = net.seg_refiner(net.lane_decoder(raw))
94
+ lane = torch.argmax(ll, dim=1).numpy().astype(np.uint8)
95
+ print(f"bev {m}: lane agreement {(lane == z['lane']).mean():.5f}")
96
+ return
97
+ x = net.normalize(torch.from_numpy(z["input.imgs"]))
98
+ for cfg in sys.argv[2:] or ["bb", "tt"]:
99
+ enc, head = cfg[0], cfg[1]
100
+ cur["stem"] = "b" if enc == "B" else None
101
+ cur["m"] = "t" if enc == "B" else enc
102
+ f = net.image_encoder(x)
103
+ cur["m"], cur["stem"] = head, None
104
+ dl, _ = net.depth_head(f)
105
+ seg = net.seg2d_head(f)
106
+ d = torch.argmax(dl.reshape(1, 8, 64, 108, 192), dim=2).numpy()
107
+ s = torch.argmax(torch.clamp(seg, -30, 30).reshape(1, 8, 21, 108, 192), dim=2).numpy()
108
+ print(f"encoder {enc} heads {head}: depth {(d == z['depth']).mean():.4f} seg2d {(s == z['seg2d']).mean():.4f}",
109
+ flush=True)
110
+ srt = np.sort(z["depth.logits"].astype(np.float32), axis=1)
111
+ margin = srt[:, -1] - srt[:, -2]
112
+ print("depth top-2 margin quantiles (0.05, 0.1, 0.25, 0.5):",
113
+ np.round(np.quantile(margin, [0.05, 0.1, 0.25, 0.5]), 4).tolist())
114
+
115
+
116
+ def det(net, frame: str, configs) -> None:
117
+ """``det`` mode (module docstring)."""
118
+ import torch
119
+
120
+ from tt_meteor.host.postprocess import PostConfig
121
+ from tt_meteor.tests.box_agreement import box_agreement, decode_3d
122
+
123
+ g = np.load(GOLD / frame / "taps.npz")
124
+ conv0, resize0 = net.conv, net.resize
125
+ cur = {"m": "f"}
126
+ conv_mode = lambda m: "f" if m == "F" else m # noqa: E731
127
+ resize_mode = lambda m: "t" if m == "F" else m # noqa: E731
128
+ net.conv = lambda x, module, relu=False: rnd(conv0(x, module, relu), conv_mode(cur["m"]))
129
+ net.resize = lambda x, hw: rnd(resize0(x, hw), resize_mode(cur["m"]))
130
+ cfg = PostConfig()
131
+ ref = decode_3d({"hm": g["hm"], "reg": g["reg"]}, cfg)
132
+ torch.manual_seed(0)
133
+ for c in configs or ["bbb", "bFF"]:
134
+ r_m, d_m, b_m = c
135
+ with torch.inference_mode():
136
+ raw = torch.from_numpy(g["bev.raw"].astype(np.float32))
137
+ raw = rnd(raw + torch.randn_like(raw) * raw.abs() * 0.003, "b") if r_m == "n" else rnd(raw, r_m)
138
+ cur["m"] = d_m
139
+ _, hm_pre, reg_pre = net.det_stem(raw)
140
+ cur["m"] = b_m
141
+ hm, reg = net.box_refiner(hm_pre, reg_pre)
142
+ r = box_agreement(decode_3d({"hm": hm.numpy(), "reg": reg.numpy()}, cfg), ref, cfg)
143
+ err = np.abs(hm.numpy() - g["hm"]).ravel()
144
+ print(f"{frame} raw {r_m} det {d_m} refiner {b_m}: recall {r['recall']:.4f} precision {r['precision']:.4f} "
145
+ f"hm max abs {err.max():.4f}", flush=True)
146
+
147
+
148
+ def absent(net, configs) -> None:
149
+ """``absent`` mode (module docstring)."""
150
+ import torch
151
+
152
+ g = np.load(GOLD / "pandaset_019_f40_imagenet_linear" / "taps.npz")
153
+ cam = int(np.flatnonzero(~np.asarray(g["input.present"], bool))[0])
154
+ ref_s, ref_d = g["seg2d"][0][cam], g["depth"][0][cam]
155
+ dev_path = Path(os.environ.get("TT_MODELS_ROOT", "/home/ubuntu/experiments/tt-models")) / "logs" / "meteor" / \
156
+ "variants_device_absent_argmax.npz"
157
+ dev = None
158
+ if dev_path.is_file():
159
+ with np.load(dev_path) as d:
160
+ i = int(np.flatnonzero(d["absent"] == cam)[0])
161
+ dev = {"seg2d": d["seg2d"][i], "depth": d["depth"][i]}
162
+ conv0, resize0 = net.conv, net.resize
163
+ cur = {"c": "f", "r": "f"}
164
+ net.conv = lambda x, module, relu=False: rnd(conv0(x, module, relu), cur["c"])
165
+ net.resize = lambda x, hw: rnd(resize0(x, hw), cur["r"])
166
+ x = torch.zeros(1, 3, *g["input.imgs"].shape[-2:]) # imagenet: absent = zero AFTER normalising
167
+ for c in configs or ["ff", "TT"]:
168
+ cur["c"], cur["r"] = c[0], c[1]
169
+ with torch.inference_mode():
170
+ f = net.image_encoder(x)
171
+ dl, _ = net.depth_head(f)
172
+ seg = net.seg2d_head(f)
173
+ s = torch.argmax(torch.clamp(seg, -30, 30), dim=1).numpy()[0]
174
+ d = torch.argmax(dl.reshape(1, -1, *s.shape), dim=1).numpy()[0]
175
+ msg = f"absent cam {cam} conv {c[0]} resize {c[1]}: seg2d {(s == ref_s).mean():.4f} depth {(d == ref_d).mean():.4f}"
176
+ if dev is not None:
177
+ msg += f" | vs device seg2d {(s == dev['seg2d']).mean():.4f} depth {(d == dev['depth']).mean():.4f}"
178
+ print(msg, flush=True)
179
+
180
+
181
+ if __name__ == "__main__":
182
+ main()
code/scripts/profile_ops.py ADDED
@@ -0,0 +1,231 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Device profile of meteor-p150: one eager frame with module signposts and ``--rounds`` traced replays of the
4
+ ``frame`` trace, between Tracy signposts (OPT_BASELINE.md).
5
+
6
+ ROOT=/home/ubuntu/experiments/tt-models
7
+ $ROOT/bin/devrun -t 3600 -- python -m tracy -r -p -v --no-web-server --op-support-count 6000 \\
8
+ -o $ROOT/generated/profiler/meteor_baseline code/scripts/profile_ops.py --rounds 2
9
+ tt-perf-report <ops_perf_results_*.csv> --start-signpost frame --end-signpost frame_end --tracing-mode
10
+ python code/scripts/profile_summary.py $ROOT/generated/profiler/meteor_baseline --json s.json --md s.md
11
+
12
+ ttnn-visualizer memory + graph report of one eager frame (slow; never used for timing):
13
+
14
+ $ROOT/bin/devrun -t 3600 -- python code/scripts/profile_ops.py --graph-report $ROOT/generated/ttnn_reports/meteor_baseline
15
+ ttnn-visualizer --profiler-path $ROOT/generated/ttnn_reports/meteor_baseline/frame \\
16
+ --performance-path $ROOT/generated/profiler/meteor_baseline/reports/<ts>
17
+
18
+ The model is loaded like the API but without the trace capture (``warmup_variants="none"``), the rig's lift tables
19
+ are written, and one eager frame compiles every program. Then, with the device profiler buffer flushed
20
+ (``ttnn.ReadDeviceProfiler``) before and after each section:
21
+
22
+ - ``eager_frame`` .. ``eager_frame_end``: one eager frame (program cache hot) with a signpost per module (``img.*``,
23
+ ``lift``, ``lift.resize``, ``bev.*``, ``head.*``, ``out.pack``). Ops issued inline after a module call (argmaxes,
24
+ the depth softmax, the ``to_rm`` conversions, planner adds) belong to the module signposted last. An eager frame
25
+ BEFORE the capture is the documented order (``TtMETEOR.run_frame_eager``: no allocation next to a live trace);
26
+ - the trace is captured (``TtMETEOR.capture``) and replayed once (not profiled);
27
+ - ``frame`` .. ``frame_end`` (and ``frame2`` ... for ``--rounds``): one replay each, the served device path.
28
+
29
+ Through the identical op order the replay's ops inherit the eager attribution (``profile_summary.py``). One frame is
30
+ about 2,000 programs, above the profiler's default buffer of ~1,000, hence ``--op-support-count 6000``. The output
31
+ folder must be absolute and outside the bundle (PLAN.md 5.1).
32
+ """
33
+ from __future__ import annotations
34
+
35
+ import argparse
36
+ import contextlib
37
+ import json
38
+ import os
39
+ import sys
40
+ import threading
41
+ import time
42
+ from pathlib import Path
43
+ from typing import Callable, List
44
+
45
+ HERE = Path(__file__).resolve()
46
+ sys.path.insert(0, str(HERE.parent))
47
+ sys.path.insert(0, str(HERE.parents[1]))
48
+
49
+ IMAGE_PARTS = (("normalize", "img.normalize"), ("stem_conv", "img.stem"), ("stages", "img.resnet"),
50
+ ("fuse", "img.fpn_fuse"), ("seg_logits", "img.seg2d"), ("depth_logits", "img.depth"),
51
+ ("depth_mean", "img.depth_mean"), ("context", "img.ctx_paint"), ("det2d", "img.det2d"),
52
+ ("tl", "img.tl"))
53
+ BEV_PARTS = ("fuse", "lane", "det", "feat_bf16", "occupancy", "stationary", "traj", "risk")
54
+ HEAD_PARTS = ("ego_stem", "ego_mlp", "ego_attn", "sem", "risk_sample", "decoder_wp", "e2e", "seg_refine",
55
+ "box_refine")
56
+
57
+
58
+ class _Signposted:
59
+ """A callable stand-in that emits a signpost (optionally one after the call too), then calls the inner object."""
60
+
61
+ def __init__(self, inner, label: str, signpost, after: str = ""):
62
+ self.inner, self._label, self._signpost, self._after = inner, label, signpost, after
63
+
64
+ def __call__(self, *args, **kwargs):
65
+ if self._label:
66
+ self._signpost(self._label)
67
+ out = self.inner(*args, **kwargs)
68
+ if self._after:
69
+ self._signpost(self._after)
70
+ return out
71
+
72
+ def __getattr__(self, name):
73
+ return getattr(self.inner, name)
74
+
75
+
76
+ def module_signposts(tt, signpost) -> Callable[[], None]:
77
+ """Wrap the stage modules of a ``TtMETEOR`` with signposting stand-ins; returns the function that restores
78
+ them."""
79
+ undo: List[Callable[[], None]] = []
80
+
81
+ def swap(obj, attr, label, after=""):
82
+ had = attr in obj.__dict__
83
+ old = obj.__dict__.get(attr)
84
+ inner = getattr(obj, attr)
85
+ setattr(obj, attr, _Signposted(inner, label, signpost, after))
86
+ undo.append(lambda: setattr(obj, attr, old) if had else obj.__dict__.pop(attr, None))
87
+
88
+ img = tt.image
89
+ for attr, label in IMAGE_PARTS:
90
+ swap(img, attr, label)
91
+ lat0 = img.lat[0]
92
+ img.lat[0] = _Signposted(lat0, "img.fpn", signpost)
93
+ undo.append(lambda: img.lat.__setitem__(0, lat0))
94
+ swap(tt.lift, "resize", "lift.resize")
95
+ swap(tt, "lift", "lift")
96
+ for attr in BEV_PARTS:
97
+ swap(tt.bev, attr, f"bev.{attr}")
98
+ for attr in HEAD_PARTS:
99
+ swap(tt.head, attr, f"head.{attr}")
100
+ swap(tt, "head", "", after="out.pack")
101
+
102
+ def restore() -> None:
103
+ for fn in reversed(undo):
104
+ fn()
105
+
106
+ return restore
107
+
108
+
109
+ def rss_guard(limit_gb: float, stop: threading.Event) -> None:
110
+ """The graph capture lives in host RAM: never let it take the shared host down."""
111
+ page = os.sysconf("SC_PAGE_SIZE")
112
+ while not stop.wait(1.0):
113
+ with open("/proc/self/statm") as fh:
114
+ rss = int(fh.read().split()[1]) * page
115
+ if rss > limit_gb * 2 ** 30:
116
+ print(f"graph capture: host RSS {rss / 2 ** 30:.1f} GB > {limit_gb} GB, abort", flush=True)
117
+ os._exit(3)
118
+
119
+
120
+ def main() -> int:
121
+ ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
122
+ ap.add_argument("--input", default="sample", help="bench.py input spec (sample | synthetic | name=manifest)")
123
+ ap.add_argument("--rounds", type=int, default=2)
124
+ ap.add_argument("--no-eager", dest="eager", action="store_false")
125
+ ap.add_argument("--graph-report", default=None, help="write a ttnn-visualizer graph report of one eager frame to "
126
+ "<dir>/frame instead of profiling")
127
+ ap.add_argument("--max-rss-gb", type=float, default=24.0)
128
+ ap.add_argument("--dispatch", default=None, choices=["eth", "worker"])
129
+ ap.add_argument("--num-cqs", type=int, default=None)
130
+ ap.add_argument("--json", default=None, help="write the run's metadata (config, trace description)")
131
+ a = ap.parse_args()
132
+ import ttnn
133
+
134
+ from bench import cache_entries, load_input
135
+ from tt_meteor import METEOR
136
+ from tt_meteor.host.inputs import prepare_request
137
+ from tt_meteor.ttaw.profiling import read_device_profiler, signpost
138
+
139
+ name, req, path = load_input(a.input)
140
+ t0 = time.perf_counter()
141
+ model = METEOR.from_pretrained(dispatch=a.dispatch, num_command_queues=a.num_cqs, warmup_variants="none")
142
+ meta = {"config": model.device_info, "input": str(path or name), "load_s": round(time.perf_counter() - t0, 1)}
143
+ print("config:", json.dumps(meta["config"]), flush=True)
144
+ dev, tt = model.device, model.tt
145
+ sync = lambda: ttnn.synchronize_device(dev) # noqa: E731
146
+ graph_out = None
147
+ try:
148
+ frame = prepare_request(req["images"], req["calibration"], req["ego_speed"], req.get("stream"))
149
+ geom = model.calib_cache.get(frame.K, frame.T_cam_ego)
150
+ t1 = time.perf_counter()
151
+ tt.run_frame_eager(frame, geom) # compile + constants
152
+ sync()
153
+ meta["first_eager_s"] = round(time.perf_counter() - t1, 1)
154
+ meta["program_cache_entries"] = cache_entries(dev)
155
+ print("first eager frame %.1f s, programs %s" % (meta["first_eager_s"], meta["program_cache_entries"]),
156
+ flush=True)
157
+ if a.graph_report:
158
+ out_dir = Path(a.graph_report).resolve() / "frame"
159
+ out_dir.mkdir(parents=True, exist_ok=True)
160
+ json_path = out_dir / "graph_capture.json"
161
+ stop = threading.Event()
162
+ threading.Thread(target=rss_guard, args=(a.max_rss_gb, stop), daemon=True).start()
163
+ prev = ttnn.CONFIG.enable_fast_runtime_mode
164
+ ttnn.CONFIG.enable_fast_runtime_mode = False
165
+ t1 = time.perf_counter()
166
+ try:
167
+ ttnn.graph.begin_graph_capture()
168
+ tt.run_frame_eager(frame, geom)
169
+ sync()
170
+ ttnn.graph.end_graph_capture_to_file(str(json_path))
171
+ finally:
172
+ ttnn.CONFIG.enable_fast_runtime_mode = prev
173
+ stop.set()
174
+ meta["graph_capture_s"] = round(time.perf_counter() - t1, 1)
175
+ meta["graph_json_mb"] = round(json_path.stat().st_size / 1e6, 1)
176
+ print("graph captured:", meta["graph_capture_s"], "s,", meta["graph_json_mb"], "MB", flush=True)
177
+ graph_out = (out_dir, json_path)
178
+ else:
179
+ read_device_profiler(dev) # drop the load / compile data
180
+ if a.eager:
181
+ restore = module_signposts(tt, signpost)
182
+ try:
183
+ signpost("eager_frame")
184
+ tt.run_frame_eager(frame, geom)
185
+ sync()
186
+ signpost("eager_frame_end")
187
+ finally:
188
+ restore()
189
+ read_device_profiler(dev)
190
+ t1 = time.perf_counter()
191
+ tt.capture()
192
+ meta["capture_s"] = round(time.perf_counter() - t1, 1)
193
+ tt.run_frame(frame, geom) # one served frame (tables current, inputs uploaded)
194
+ sync()
195
+ read_device_profiler(dev)
196
+ for k in range(1, a.rounds + 1):
197
+ tag = "" if k == 1 else str(k)
198
+ signpost(f"frame{tag}")
199
+ tt.runner.replay("frame")
200
+ sync()
201
+ signpost(f"frame{tag}_end")
202
+ read_device_profiler(dev)
203
+ meta["trace"] = tt.describe()
204
+ meta["program_cache_entries_end"] = cache_entries(dev)
205
+ finally:
206
+ model.close()
207
+ if graph_out is not None:
208
+ out_dir, json_path = graph_out
209
+ t1 = time.perf_counter()
210
+ try:
211
+ db = ttnn.graph_report.import_report(json_path, out_dir)
212
+ finally:
213
+ json_path.unlink(missing_ok=True)
214
+ ttnn.save_config_to_json_file(out_dir / "config.json")
215
+ cfg = json.loads((out_dir / "config.json").read_text())
216
+ cfg.update({"enable_fast_runtime_mode": False, "enable_graph_report": True,
217
+ "enable_detailed_buffer_report": False, "root_report_path": str(out_dir.parent),
218
+ "report_name": out_dir.name})
219
+ (out_dir / "config.json").write_text(json.dumps(cfg, indent=4) + "\n")
220
+ meta["graph_report"] = {"dir": str(out_dir), "db": str(db), "import_s": round(time.perf_counter() - t1, 1),
221
+ "db_mb": round(Path(db).stat().st_size / 1e6, 1)}
222
+ print("graph report:", json.dumps(meta["graph_report"]), flush=True)
223
+ if a.json:
224
+ Path(a.json).parent.mkdir(parents=True, exist_ok=True)
225
+ Path(a.json).write_text(json.dumps(meta, indent=1, default=str) + "\n")
226
+ print(json.dumps(meta, indent=1, default=str))
227
+ return 0
228
+
229
+
230
+ if __name__ == "__main__":
231
+ sys.exit(main())
code/scripts/profile_summary.py ADDED
@@ -0,0 +1,288 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Summary of a ``profile_ops.py`` device profile (the ``ops_perf_results_<ts>.csv`` of ``python -m tracy -r``).
4
+
5
+ python code/scripts/profile_summary.py $ROOT/generated/profiler/meteor_baseline --json out.json --md out.md
6
+
7
+ - per traced segment (the ``frame`` signpost range, and ``frame2`` ... of later rounds):
8
+ programs, kernel sum, FW sum, op-to-op gaps (sum over the segment, its first op excluded) and the span (first FW
9
+ start to last FW end, at the CHIP_FREQ of ``profile_log_device.csv``, else 1350 MHz), per op code;
10
+ - per module stage: the eager run's module signposts (``img.normalize`` ... ``out.pack``) label the eager op sequence;
11
+ when it equals the traced segment's op sequence op for op, the labels are carried over to the traced ops, and the
12
+ table reports the traced kernel times per stage, per block (``img`` / ``lift`` / ``bev`` / ``head`` / ``out``)
13
+ and per (stage, op code);
14
+ - the top ops by traced kernel time and the largest op-to-op gaps.
15
+
16
+ csv + json only (no ttnn): runs anywhere.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import argparse
21
+ import csv
22
+ import glob
23
+ import json
24
+ import os
25
+ import sys
26
+ from collections import OrderedDict, defaultdict
27
+ from pathlib import Path
28
+ from typing import Any, Dict, List, Optional, Tuple
29
+
30
+ SEGMENTS = ("frame",)
31
+ KERNEL = "DEVICE KERNEL DURATION [ns]"
32
+ FW = "DEVICE FW DURATION [ns]"
33
+ GAP = "OP TO OP LATENCY [ns]"
34
+
35
+
36
+ def find_csv(path: Path) -> Path:
37
+ if path.is_file():
38
+ return path
39
+ found = sorted(glob.glob(str(path / "**" / "ops_perf_results*.csv"), recursive=True), key=os.path.getmtime)
40
+ if not found:
41
+ raise FileNotFoundError(f"no ops_perf_results*.csv under {path}")
42
+ return Path(found[-1])
43
+
44
+
45
+ def chip_freq_mhz(csv_path: Path, root: Path) -> Optional[float]:
46
+ cands = [csv_path.parent / "profile_log_device.csv", root / ".logs" / "profile_log_device.csv"]
47
+ cands += [Path(p) for p in glob.glob(str(root / "**" / "profile_log_device.csv"), recursive=True)]
48
+ for c in cands:
49
+ try:
50
+ head = c.read_text(errors="ignore").splitlines()[0]
51
+ except (OSError, IndexError):
52
+ continue
53
+ for part in head.split(","):
54
+ if "CHIP_FREQ" in part and ":" in part:
55
+ try:
56
+ return float(part.split(":")[1].strip())
57
+ except ValueError:
58
+ pass
59
+ return None
60
+
61
+
62
+ def num(row: Dict[str, str], key: str) -> float:
63
+ v = (row.get(key) or "").strip()
64
+ try:
65
+ return float(v) if v else 0.0
66
+ except ValueError:
67
+ return 0.0
68
+
69
+
70
+ def is_signpost(row: Dict[str, str]) -> bool:
71
+ return (row.get("OP TYPE") or "").strip() == "signpost"
72
+
73
+
74
+ def window(rows: List[Dict[str, str]], name: str) -> Optional[List[Tuple[str, Dict[str, str]]]]:
75
+ """Rows between the first signpost ``name`` and the next ``name_end`` as ``(label, row)``; ``label`` is the
76
+ latest inner signpost (``""`` before the first one). None when the signpost is absent."""
77
+ out, on, label = [], False, ""
78
+ for r in rows:
79
+ if is_signpost(r):
80
+ code = r.get("OP CODE", "")
81
+ if on and code == f"{name}_end":
82
+ return out
83
+ if not on and code == name:
84
+ on = True
85
+ continue
86
+ if on:
87
+ label = code
88
+ continue
89
+ if on:
90
+ out.append((label, r))
91
+ return out if on else None
92
+
93
+
94
+ def tensor_desc(row: Dict[str, str], prefix: str) -> str:
95
+ """``INPUT_0`` / ``OUTPUT_0`` columns -> ``[W, Z, Y, X] LAYOUT DTYPE MEMORY`` (logical dims), '' when absent."""
96
+ dims = []
97
+ for d in "WZYX":
98
+ v = (row.get(f"{prefix}_{d}_PAD[LOGICAL]") or "").strip()
99
+ if not v:
100
+ return ""
101
+ dims.append(v.split("[")[1].rstrip("]") if "[" in v else v)
102
+ mem = (row.get(f"{prefix}_MEMORY") or "").replace("DEV_1_", "").replace("DEV_0_", "")
103
+ return f"[{','.join(dims)}] {row.get(f'{prefix}_LAYOUT', '')} {row.get(f'{prefix}_DATATYPE', '')} {mem}".strip()
104
+
105
+
106
+ def io_desc(row: Dict[str, str], prefix: str, n: int = 3) -> str:
107
+ if (row.get(prefix + "S") or "").strip(): # older reports: one INPUTS / OUTPUTS column
108
+ text = " ".join(row[prefix + "S"].split())
109
+ return text if len(text) <= 160 else text[:157] + "..."
110
+ parts = [tensor_desc(row, f"{prefix}_{i}") for i in range(n)]
111
+ return "; ".join(p for p in parts if p)
112
+
113
+
114
+ def group_of(label: str) -> str:
115
+ """``img.resnet`` -> ``img``, ``bev.fuse`` -> ``bev``, ``lift.resize`` -> ``lift``: the block of a stage."""
116
+ return label.partition(".")[0]
117
+
118
+
119
+ def summarize(ops: List[Dict[str, str]], freq: float) -> Dict[str, Any]:
120
+ kernel = [num(r, KERNEL) / 1e3 for r in ops]
121
+ fw = [num(r, FW) / 1e3 for r in ops]
122
+ gaps = [num(r, GAP) / 1e3 for r in ops]
123
+ starts = [num(r, "DEVICE FW START CYCLE") for r in ops]
124
+ ends = [num(r, "DEVICE FW END CYCLE") for r in ops]
125
+ valid = [s for s in starts if s > 0]
126
+ span = (max(ends) - min(valid)) / freq if ops and valid and max(ends) > 0 else None
127
+ by_op: Dict[str, List[float]] = defaultdict(lambda: [0, 0.0])
128
+ fid: Dict[str, int] = defaultdict(int)
129
+ for r, k in zip(ops, kernel):
130
+ by_op[r.get("OP CODE", "?")][0] += 1
131
+ by_op[r.get("OP CODE", "?")][1] += k
132
+ f = (r.get("MATH FIDELITY") or "").strip()
133
+ if f:
134
+ fid[f] += 1
135
+ total = sum(kernel) or 1.0
136
+ traced = sum(1 for r in ops if (r.get("METAL TRACE ID") or "").strip())
137
+ return {"ops": len(ops), "traced_ops": traced, "kernel_us": round(sum(kernel), 1), "fw_us": round(sum(fw), 1),
138
+ "op2op_us": round(sum(gaps[1:]), 1), "span_us": None if span is None else round(span, 1),
139
+ "by_op": [{"op": op, "count": int(c), "kernel_us": round(t, 1), "share": round(t / total, 4)}
140
+ for op, (c, t) in sorted(by_op.items(), key=lambda kv: -kv[1][1])],
141
+ "fidelity": dict(fid)}
142
+
143
+
144
+ def main() -> int:
145
+ ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
146
+ ap.add_argument("source", type=Path, help="tracy output dir (-o) or an ops_perf_results CSV")
147
+ ap.add_argument("--freq-mhz", type=float, default=None)
148
+ ap.add_argument("--top", type=int, default=15)
149
+ ap.add_argument("--json", type=Path, default=None)
150
+ ap.add_argument("--md", type=Path, default=None)
151
+ a = ap.parse_args()
152
+ path = find_csv(a.source)
153
+ with open(path, newline="") as f:
154
+ rows = list(csv.DictReader(f))
155
+ root = a.source if a.source.is_dir() else path.parent
156
+ freq = a.freq_mhz or chip_freq_mhz(path, root) or 1350.0
157
+ res: Dict[str, Any] = {"csv": str(path), "freq_mhz": freq, "rows": len(rows),
158
+ "signposts": [r.get("OP CODE") for r in rows if is_signpost(r)], "segments": OrderedDict(),
159
+ "eager": OrderedDict()}
160
+ traced_ops: Dict[str, List[Dict[str, str]]] = {}
161
+ for k in ("", "2", "3"):
162
+ for seg in SEGMENTS:
163
+ w = window(rows, f"{seg}{k}")
164
+ if w is None:
165
+ continue
166
+ ops = [r for _, r in w]
167
+ traced_ops[f"{seg}{k}"] = ops
168
+ res["segments"][f"{seg}{k}"] = summarize(ops, freq)
169
+ steady = [s for s in SEGMENTS if s in res["segments"]]
170
+ if steady:
171
+ seg = res["segments"]
172
+ res["steady_frame"] = {key: round(sum(seg[s][key] or 0 for s in steady), 1)
173
+ for key in ("ops", "kernel_us", "fw_us", "op2op_us", "span_us")}
174
+ res["steady_frame"]["segments"] = steady
175
+ # eager windows with stage labels; carry the labels over to the traced ops when the sequences match
176
+ labels: Dict[str, List[str]] = {}
177
+ groups: "OrderedDict[str, Dict[str, Any]]" = OrderedDict()
178
+ for seg in SEGMENTS:
179
+ w = window(rows, f"eager_{seg}")
180
+ if w is None:
181
+ continue
182
+ ops = [r for _, r in w]
183
+ res["eager"][seg] = summarize(ops, freq)
184
+ eager_codes = [r.get("OP CODE") for r in ops]
185
+ traced_codes = [r.get("OP CODE") for r in traced_ops.get(seg, [])]
186
+ match = eager_codes == traced_codes
187
+ res["eager"][seg]["sequence_matches_trace"] = match
188
+ if not match:
189
+ res["eager"][seg]["mismatch"] = {"eager_ops": len(eager_codes), "traced_ops": len(traced_codes),
190
+ "first_diff": next((i for i, (x, y) in enumerate(
191
+ zip(eager_codes, traced_codes)) if x != y), None)}
192
+ labels[seg] = [lab or f"{seg[:3]}.begin" for lab, _ in w]
193
+ # per stage: eager kernel sums (always) and traced kernel sums (when aligned)
194
+ stages: "OrderedDict[str, Dict[str, Any]]" = OrderedDict()
195
+ for i, (lab, r) in enumerate(w):
196
+ st = stages.setdefault(lab or f"{seg[:3]}.begin", {"ops": 0, "eager_kernel_us": 0.0, "eager_op2op_us": 0.0,
197
+ "kernel_us": 0.0, "op2op_us": 0.0,
198
+ "by_op": defaultdict(float)})
199
+ st["ops"] += 1
200
+ st["eager_kernel_us"] += num(r, KERNEL) / 1e3
201
+ st["eager_op2op_us"] += num(r, GAP) / 1e3 if i else 0.0
202
+ src = traced_ops[seg][i] if match else r
203
+ st["by_op"][src.get("OP CODE", "?")] += num(src, KERNEL) / 1e3
204
+ if match:
205
+ st["kernel_us"] += num(src, KERNEL) / 1e3
206
+ st["op2op_us"] += num(src, GAP) / 1e3 if i else 0.0
207
+ g = group_of(lab or f"{seg[:3]}.begin")
208
+ go = groups.setdefault(g, {"ops": 0, "kernel_us": 0.0, "by_op": defaultdict(lambda: [0, 0.0])})
209
+ go["ops"] += 1
210
+ go["kernel_us"] += num(src, KERNEL) / 1e3
211
+ go["by_op"][src.get("OP CODE", "?")][0] += 1
212
+ go["by_op"][src.get("OP CODE", "?")][1] += num(src, KERNEL) / 1e3
213
+ go["traced"] = match
214
+ for v in stages.values():
215
+ v["by_op"] = {op: round(t, 1) for op, t in sorted(v["by_op"].items(), key=lambda kv: -kv[1])}
216
+ res["eager"][seg]["stages"] = {k: {kk: round(vv, 1) if isinstance(vv, float) else vv for kk, vv in v.items()}
217
+ for k, v in stages.items()}
218
+ res["groups"] = {g: {"ops": v["ops"], "kernel_us": round(v["kernel_us"], 1), "traced": v.get("traced", False),
219
+ "by_op": [{"op": op, "count": c, "kernel_us": round(t, 1)}
220
+ for op, (c, t) in sorted(v["by_op"].items(), key=lambda kv: -kv[1][1])]}
221
+ for g, v in groups.items()}
222
+ # top ops by traced kernel time (all segments of the first round)
223
+ flat = []
224
+ for seg in SEGMENTS:
225
+ for i, r in enumerate(traced_ops.get(seg, [])):
226
+ lab = labels.get(seg, [None] * (i + 1))[i] if (seg in labels and res["eager"].get(seg, {}).get(
227
+ "sequence_matches_trace")) else None
228
+ flat.append({"segment": seg, "index": i, "stage": lab, "op": r.get("OP CODE"),
229
+ "kernel_us": round(num(r, KERNEL) / 1e3, 1), "fw_us": round(num(r, FW) / 1e3, 1),
230
+ "op2op_us": round(num(r, GAP) / 1e3, 2), "cores": r.get("CORE COUNT"),
231
+ "fidelity": r.get("MATH FIDELITY"), "pm_ideal_us": round(num(r, "PM IDEAL [ns]") / 1e3, 1),
232
+ "dram_bw_util": r.get("DRAM BW UTIL (%)"), "fpu_util": r.get("PM FPU UTIL (%)"),
233
+ "inputs": io_desc(r, "INPUT"), "outputs": io_desc(r, "OUTPUT", 1)})
234
+ res["top_ops"] = sorted(flat, key=lambda d: -d["kernel_us"])[: a.top]
235
+ res["top_gaps"] = sorted(flat, key=lambda d: -d["op2op_us"])[: a.top]
236
+ if a.json:
237
+ a.json.parent.mkdir(parents=True, exist_ok=True)
238
+ a.json.write_text(json.dumps(res, indent=1) + "\n")
239
+ lines = [f"csv `{path}` ({len(rows)} rows), CHIP_FREQ {freq:.0f} MHz", "",
240
+ "| segment | programs | kernel us | FW us | op-to-op us | span us |", "|---|---:|---:|---:|---:|---:|"]
241
+ for k, v in res["segments"].items():
242
+ lines.append(f"| {k} | {v['ops']} | {v['kernel_us']} | {v['fw_us']} | {v['op2op_us']} | {v['span_us']} |")
243
+ if "steady_frame" in res:
244
+ s = res["steady_frame"]
245
+ lines.append(f"| **steady frame** ({' + '.join(s['segments'])}) | {s['ops']} | {s['kernel_us']} | {s['fw_us']} | "
246
+ f"{s['op2op_us']} | {s['span_us']} |")
247
+ for seg, e in res["eager"].items():
248
+ lines += ["", f"stages of `{seg}` (labels from the eager run; traced times "
249
+ f"{'aligned op for op' if e['sequence_matches_trace'] else 'NOT aligned: eager times only'})", "",
250
+ "| stage | ops | traced kernel us | traced op-to-op us | eager kernel us | eager op-to-op us | "
251
+ "top op codes (us) |",
252
+ "|---|---:|---:|---:|---:|---:|---|"]
253
+ for name, st in e["stages"].items():
254
+ tops = ", ".join(f"{op} {t}" for op, t in list(st["by_op"].items())[:4])
255
+ lines.append(f"| {name} | {st['ops']} | {st['kernel_us']} | {st['op2op_us']} | {st['eager_kernel_us']} | "
256
+ f"{st['eager_op2op_us']} | {tops} |")
257
+ if res.get("groups"):
258
+ lines += ["", "stage groups (traced kernel time when the eager sequence matches the trace)", "",
259
+ "| group | ops | kernel us | top op codes (us) |", "|---|---:|---:|---|"]
260
+ for g, v in res["groups"].items():
261
+ tops = ", ".join(f"{d['op']} {d['count']}x {d['kernel_us']}" for d in v["by_op"][:5])
262
+ lines.append(f"| {g} | {v['ops']} | {v['kernel_us']} | {tops} |")
263
+ for seg in SEGMENTS:
264
+ if seg in res["segments"]:
265
+ lines += ["", f"op codes of `{seg}` (traced)", "", "| op code | count | kernel us | share |",
266
+ "|---|---:|---:|---:|"]
267
+ for b in res["segments"][seg]["by_op"][:12]:
268
+ lines.append(f"| {b['op']} | {b['count']} | {b['kernel_us']} | {b['share']:.1%} |")
269
+ lines += ["", f"top {a.top} ops by traced kernel time", "",
270
+ "| # | segment | stage | op | kernel us | cores | fidelity | DRAM BW % | FPU % | inputs | output |",
271
+ "|---:|---|---|---|---:|---:|---|---:|---:|---|---|"]
272
+ for i, d in enumerate(res["top_ops"], 1):
273
+ lines.append(f"| {i} | {d['segment']}[{d['index']}] | {d['stage']} | {d['op']} | {d['kernel_us']} | {d['cores']} | "
274
+ f"{d['fidelity']} | {d['dram_bw_util']} | {d['fpu_util']} | {d['inputs']} | {d['outputs']} |")
275
+ lines += ["", f"largest {a.top} op-to-op gaps (traced)", "", "| # | segment | stage | op | gap us | kernel us |",
276
+ "|---:|---|---|---|---:|---:|"]
277
+ for i, d in enumerate(res["top_gaps"], 1):
278
+ lines.append(f"| {i} | {d['segment']}[{d['index']}] | {d['stage']} | {d['op']} | {d['op2op_us']} | {d['kernel_us']} |")
279
+ text = "\n".join(lines) + "\n"
280
+ if a.md:
281
+ a.md.parent.mkdir(parents=True, exist_ok=True)
282
+ a.md.write_text(text)
283
+ print(text)
284
+ return 0
285
+
286
+
287
+ if __name__ == "__main__":
288
+ sys.exit(main())
code/scripts/repro_grid_sample_eth.py ADDED
@@ -0,0 +1,122 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Minimal standalone repro: ``ttnn.grid_sample`` at the METEOR lift shapes, replayed in a trace loop (no METEOR code).
4
+
5
+ bin/devrun -t 900 -- python -u repro_grid_sample_eth.py --dispatch eth --iters 2000 --jsonl out.jsonl
6
+ bin/devrun -t 900 -- python -u repro_grid_sample_eth.py --dispatch worker ... # A/B
7
+
8
+ One trace = ``--per-trace`` x (grid_sample of x64 [8, 108, 192, 64] and of x96 [8, 108, 192, 96], bf16 ROW_MAJOR,
9
+ with one fp32 grid [8, 400, 250, 2]; bilinear, zeros, align_corners=False; DRAM outputs [8, 400, 250, C]).
10
+ The grid is ``--grid file.npy`` (e.g. a real METEOR lift grid: normalised coords in [-2, 2], ~half out of image)
11
+ or uniform random in [-1.2, 1.2]. Each replay is followed by ``synchronize_device`` and a fsync'd JSON heartbeat;
12
+ every ``--check-every`` replays the outputs are read back and hashed (must not change). faulthandler dumps the Python
13
+ stack when a replay stalls for ``--stall-s``. Nothing arms tt-triage or an operation timeout.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import faulthandler
19
+ import hashlib
20
+ import json
21
+ import os
22
+ import sys
23
+ import time
24
+
25
+ import numpy as np
26
+
27
+
28
+ def log(fh, **rec):
29
+ rec["t"] = round(time.time(), 3)
30
+ line = json.dumps(rec)
31
+ print(line, flush=True)
32
+ if fh:
33
+ fh.write(line + "\n")
34
+ fh.flush()
35
+ os.fsync(fh.fileno())
36
+
37
+
38
+ def main() -> int:
39
+ ap = argparse.ArgumentParser()
40
+ ap.add_argument("--dispatch", choices=["eth", "worker"], default="eth")
41
+ ap.add_argument("--cqs", type=int, default=1)
42
+ ap.add_argument("--iters", type=int, default=2000)
43
+ ap.add_argument("--per-trace", type=int, default=1, help="grid_sample pairs per trace")
44
+ ap.add_argument("--grid", help=".npy fp32 [8, 400, 250, 2] grid (default: random)")
45
+ ap.add_argument("--check-every", type=int, default=100)
46
+ ap.add_argument("--stall-s", type=float, default=60.0)
47
+ ap.add_argument("--jsonl")
48
+ a = ap.parse_args()
49
+ faulthandler.enable()
50
+ fh = open(a.jsonl, "a") if a.jsonl else None
51
+ faulthandler.dump_traceback_later(600, repeat=True)
52
+
53
+ import torch
54
+ import ttnn
55
+
56
+ core = ttnn.DispatchCoreType.ETH if a.dispatch == "eth" else ttnn.DispatchCoreType.WORKER
57
+ dev = ttnn.open_device(device_id=0, dispatch_core_config=ttnn.DispatchCoreConfig(core),
58
+ num_command_queues=a.cqs, l1_small_size=32768, trace_region_size=64 << 20)
59
+ rc = 0
60
+ try:
61
+ g = dev.compute_with_storage_grid_size()
62
+ log(fh, ev="open", dispatch=a.dispatch, cqs=a.cqs, grid=f"{g.x}x{g.y}", per_trace=a.per_trace,
63
+ grid_file=a.grid)
64
+ rng = np.random.default_rng(0)
65
+ if a.grid:
66
+ grid_np = np.load(a.grid).astype(np.float32).reshape(8, 400, 250, 2)
67
+ else:
68
+ grid_np = rng.uniform(-1.2, 1.2, (8, 400, 250, 2)).astype(np.float32)
69
+ mk = lambda arr, dt: ttnn.from_torch(torch.from_numpy(arr), dtype=dt, layout=ttnn.ROW_MAJOR_LAYOUT, # noqa: E731
70
+ device=dev, memory_config=ttnn.DRAM_MEMORY_CONFIG)
71
+ x64 = mk(rng.standard_normal((8, 108, 192, 64)).astype(np.float32), ttnn.bfloat16)
72
+ x96 = mk(rng.standard_normal((8, 108, 192, 96)).astype(np.float32), ttnn.bfloat16)
73
+ grid = mk(grid_np, ttnn.float32)
74
+
75
+ def body():
76
+ outs = []
77
+ for _ in range(a.per_trace):
78
+ for x in (x64, x96):
79
+ outs.append(ttnn.grid_sample(x, grid, mode="bilinear", padding_mode="zeros",
80
+ align_corners=False, memory_config=ttnn.DRAM_MEMORY_CONFIG))
81
+ return outs
82
+
83
+ warm = body() # compile; then free, capture
84
+ ttnn.synchronize_device(dev)
85
+ for t in warm:
86
+ ttnn.deallocate(t)
87
+ tid = ttnn.begin_trace_capture(dev, cq_id=0)
88
+ outs = body()
89
+ ttnn.end_trace_capture(dev, tid, cq_id=0)
90
+ ttnn.synchronize_device(dev)
91
+ log(fh, ev="captured")
92
+ ref = None
93
+ for i in range(a.iters):
94
+ faulthandler.dump_traceback_later(a.stall_s, repeat=True)
95
+ t0 = time.perf_counter()
96
+ ttnn.execute_trace(dev, tid, cq_id=0, blocking=False)
97
+ ttnn.synchronize_device(dev)
98
+ rec = {"ev": "iter", "i": i, "ms": round((time.perf_counter() - t0) * 1e3, 2)}
99
+ if i % a.check_every == 0 or i == a.iters - 1:
100
+ h = hashlib.sha256()
101
+ for t in outs:
102
+ h.update(ttnn.to_torch(t).float().numpy().tobytes())
103
+ d = h.hexdigest()[:16]
104
+ rec["digest"] = d
105
+ ref = ref or d
106
+ if d != ref:
107
+ rec["MISMATCH"] = ref
108
+ rc = 1
109
+ log(fh, **rec)
110
+ faulthandler.cancel_dump_traceback_later()
111
+ log(fh, ev="done", iters=a.iters, rc=rc)
112
+ ttnn.release_trace(dev, tid)
113
+ finally:
114
+ faulthandler.dump_traceback_later(300, repeat=True)
115
+ ttnn.close_device(dev)
116
+ faulthandler.cancel_dump_traceback_later()
117
+ log(fh, ev="closed")
118
+ return rc
119
+
120
+
121
+ if __name__ == "__main__":
122
+ sys.exit(main())
code/scripts/repro_resize2d_eth.py ADDED
@@ -0,0 +1,223 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Minimal standalone repro (plain ttnn, no METEOR code): the separable bilinear x2 resize of the METEOR lift,
4
+ captured in one trace and replayed in a tight loop. On a Blackhole p150 with ETH dispatch the replay intermittently
5
+ never completes (logs/meteor/hangfix/ETH_DISPATCH_HANG_ISSUE.md).
6
+
7
+ bin/devrun -t 600 -- python -u repro_resize2d_eth.py --dispatch eth --iters 3000 --jsonl out.jsonl
8
+ bin/devrun -t 600 -- python -u repro_resize2d_eth.py --dispatch worker ... # A/B
9
+
10
+ Op chain (= ttaw.ops.upsample.Resize2d, in (400, 250) -> out (800, 500), C = 96, N = 1, fp32 operands, HiFi4 + fp32
11
+ accumulation), on an input x [1, 1, 100000, 96] bf16 TILE in DRAM:
12
+
13
+ RM -> reshape [1, 400, 250, 96] -> TILE -> typecast fp32 -> transpose(-2, -1) [1, 400, 96, 250]
14
+ -> matmul A_w^T [250, 500] -> transpose(-2, -1) [1, 400, 500, 96] -> RM -> reshape [1, 1, 400, 48000] -> TILE
15
+ -> matmul A_h [800, 400] @ [1, 1, 400, 48000] -> [1, 1, 800, 48000] fp32 -> typecast bf16 -> RM
16
+ -> reshape [1, 1, 400000, 96] -> TILE
17
+
18
+ ``--part w`` stops after the W pass, ``--part h`` runs only the H-pass matmul (on a persistent [1, 1, 400, 48000]
19
+ fp32 TILE input). Each replay is followed by ``synchronize_device`` and a fsync'd JSON heartbeat; outputs are hashed
20
+ every ``--check-every`` replays; faulthandler dumps the stack when a replay stalls. No tt-triage, no op timeout.
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import argparse
25
+ import faulthandler
26
+ import hashlib
27
+ import json
28
+ import os
29
+ import sys
30
+ import time
31
+
32
+ import numpy as np
33
+
34
+ H, W, H2, W2 = 400, 250, 800, 500
35
+
36
+
37
+ def interp(n_in: int, n_out: int) -> np.ndarray:
38
+ """1-D bilinear, align_corners=False (half-pixel, clamped at 0), as F.interpolate."""
39
+ s = n_in / n_out
40
+ src = np.maximum((np.arange(n_out) + 0.5) * s - 0.5, 0.0)
41
+ i0 = np.minimum(np.floor(src).astype(np.int64), n_in - 1)
42
+ i1 = np.minimum(i0 + 1, n_in - 1)
43
+ lam = src - i0
44
+ a = np.zeros((n_out, n_in))
45
+ np.add.at(a, (np.arange(n_out), i0), 1.0 - lam)
46
+ np.add.at(a, (np.arange(n_out), i1), lam)
47
+ return a.astype(np.float32)
48
+
49
+
50
+ def log(fh, **rec):
51
+ rec["t"] = round(time.time(), 3)
52
+ line = json.dumps(rec)
53
+ print(line, flush=True)
54
+ if fh:
55
+ fh.write(line + "\n")
56
+ fh.flush()
57
+ os.fsync(fh.fileno())
58
+
59
+
60
+ def main() -> int:
61
+ ap = argparse.ArgumentParser()
62
+ ap.add_argument("--dispatch", choices=["eth", "worker"], default="eth")
63
+ ap.add_argument("--cqs", type=int, default=1)
64
+ ap.add_argument("--iters", type=int, default=3000)
65
+ ap.add_argument("--per-trace", type=int, default=1, help="resize chains per trace")
66
+ ap.add_argument("--part", choices=["full", "w", "h", "w_rm", "wh", "tail", "mid", "mid_h", "h_tail",
67
+ "t_typecast", "t_untilize", "t_reshape", "t_tilize", "t_reshape_tile"], default="full",
68
+ help="w: up to the W-pass transpose; w_rm: + RM/reshape/TILE; wh: + H matmul (no tail); "
69
+ "tail: typecast/RM/reshape/TILE of a persistent z; mid: RM/reshape/TILE of a persistent "
70
+ "[1,400,500,C] y; mid_h: mid + H matmul; h_tail: H matmul + tail; t_*: ONE tail op on a "
71
+ "persistent input (typecast fp32->bf16 TILE [800, 500C]; untilize bf16 [800, 500C]; RM "
72
+ "reshape [800, 500C] -> [400000, C]; tilize RM [400000, C])")
73
+ ap.add_argument("--channels", type=int, default=96, help="C (the H-pass matmul N is 500 * C; 96 -> 48000)")
74
+ ap.add_argument("--dtype", choices=["float32", "bfloat16"], default="float32", help="matmul operand dtype")
75
+ ap.add_argument("--fidelity", choices=["HiFi4", "HiFi3", "HiFi2", "LoFi"], default="HiFi4")
76
+ ap.add_argument("--no-fp32-acc", action="store_true", help="fp32_dest_acc_en=False")
77
+ ap.add_argument("--dest-cols", type=int, default=0,
78
+ help="t_reshape: destination last dim (default C: [800, 500C] -> [400000, C]); dest page = 2 B x it")
79
+ ap.add_argument("--reshape", choices=["rm", "tile", "tail_tile"], default="rm",
80
+ help="rm: TILE -> RM -> reshape -> TILE (as Resize2d); tile: ttnn.reshape on the TILE tensor")
81
+ ap.add_argument("--check-every", type=int, default=200)
82
+ ap.add_argument("--stall-s", type=float, default=60.0)
83
+ ap.add_argument("--jsonl")
84
+ a = ap.parse_args()
85
+ faulthandler.enable()
86
+ fh = open(a.jsonl, "a") if a.jsonl else None
87
+ faulthandler.dump_traceback_later(600, repeat=True)
88
+
89
+ import torch
90
+ import ttnn
91
+
92
+ core = ttnn.DispatchCoreType.ETH if a.dispatch == "eth" else ttnn.DispatchCoreType.WORKER
93
+ dev = ttnn.open_device(device_id=0, dispatch_core_config=ttnn.DispatchCoreConfig(core),
94
+ num_command_queues=a.cqs, l1_small_size=32768, trace_region_size=64 << 20)
95
+ rc = 0
96
+ try:
97
+ g = dev.compute_with_storage_grid_size()
98
+ C = a.channels
99
+ dt = ttnn.float32 if a.dtype == "float32" else ttnn.bfloat16
100
+ log(fh, ev="open", dispatch=a.dispatch, cqs=a.cqs, grid=f"{g.x}x{g.y}", part=a.part, per_trace=a.per_trace,
101
+ channels=C, n=W2 * C, dtype=a.dtype, fidelity=a.fidelity, fp32_acc=not a.no_fp32_acc,
102
+ reshape=a.reshape, dest_cols=a.dest_cols or None)
103
+ cfg = ttnn.init_device_compute_kernel_config(dev.arch(), math_fidelity=getattr(ttnn.MathFidelity, a.fidelity),
104
+ fp32_dest_acc_en=not a.no_fp32_acc, packer_l1_acc=False,
105
+ math_approx_mode=False)
106
+ to_dev = lambda arr, dt, lay: ttnn.from_torch(torch.from_numpy(np.ascontiguousarray(arr)), dtype=dt, # noqa
107
+ layout=lay, device=dev, memory_config=ttnn.DRAM_MEMORY_CONFIG)
108
+ rng = np.random.default_rng(0)
109
+ aw_t = to_dev(interp(W, W2).T, dt, ttnn.TILE_LAYOUT) # [250, 500]
110
+ ah = to_dev(interp(H, H2), dt, ttnn.TILE_LAYOUT) # [800, 400]
111
+ x_in = to_dev(rng.standard_normal((1, 1, H * W, C)).astype(np.float32), ttnn.bfloat16, ttnn.TILE_LAYOUT)
112
+ y_in = to_dev(rng.standard_normal((1, 1, H, W2 * C)).astype(np.float32), dt, ttnn.TILE_LAYOUT)
113
+
114
+ z_in = to_dev(rng.standard_normal((1, 1, H2, W2 * C)).astype(np.float32), dt, ttnn.TILE_LAYOUT) \
115
+ if a.part == "tail" else None
116
+ yw_in = to_dev(rng.standard_normal((1, H, W2, C)).astype(np.float32), dt, ttnn.TILE_LAYOUT) \
117
+ if a.part in ("mid", "mid_h") else None
118
+
119
+ one = None
120
+ if a.part.startswith("t_"):
121
+ z32 = rng.standard_normal((1, 1, H2, W2 * C)).astype(np.float32)
122
+ one = {"t_typecast": lambda: to_dev(z32, ttnn.float32, ttnn.TILE_LAYOUT),
123
+ "t_untilize": lambda: to_dev(z32, ttnn.bfloat16, ttnn.TILE_LAYOUT),
124
+ "t_reshape": lambda: to_dev(z32, ttnn.bfloat16, ttnn.ROW_MAJOR_LAYOUT),
125
+ "t_tilize": lambda: to_dev(z32.reshape(1, 1, H2 * W2, C), ttnn.bfloat16, ttnn.ROW_MAJOR_LAYOUT),
126
+ "t_reshape_tile": lambda: to_dev(z32, ttnn.bfloat16, ttnn.TILE_LAYOUT),
127
+ }[a.part]()
128
+
129
+ def tail(z):
130
+ if z.dtype != ttnn.bfloat16:
131
+ z = ttnn.typecast(z, ttnn.bfloat16)
132
+ if a.reshape in ("tile", "tail_tile"):
133
+ return ttnn.reshape(z, (1, 1, H2 * W2, C))
134
+ z = ttnn.to_layout(z, ttnn.ROW_MAJOR_LAYOUT)
135
+ z = ttnn.reshape(z, (1, 1, H2 * W2, C))
136
+ return ttnn.to_layout(z, ttnn.TILE_LAYOUT)
137
+
138
+ def mid(y):
139
+ if a.reshape == "tile":
140
+ return ttnn.reshape(y, (1, 1, H, W2 * C))
141
+ y = ttnn.to_layout(y, ttnn.ROW_MAJOR_LAYOUT)
142
+ y = ttnn.reshape(y, (1, 1, H, W2 * C))
143
+ return ttnn.to_layout(y, ttnn.TILE_LAYOUT)
144
+
145
+ def chain():
146
+ if a.part == "t_typecast":
147
+ return ttnn.typecast(one, ttnn.bfloat16)
148
+ if a.part == "t_untilize":
149
+ return ttnn.to_layout(one, ttnn.ROW_MAJOR_LAYOUT)
150
+ if a.part == "t_reshape":
151
+ d = a.dest_cols or C
152
+ return ttnn.reshape(one, (1, 1, H2 * W2 * C // d, d))
153
+ if a.part == "t_reshape_tile":
154
+ return ttnn.reshape(one, (1, 1, H2 * W2, C))
155
+ if a.part == "t_tilize":
156
+ return ttnn.to_layout(one, ttnn.TILE_LAYOUT)
157
+ if a.part == "h":
158
+ return ttnn.matmul(ah, y_in, compute_kernel_config=cfg)
159
+ if a.part == "h_tail":
160
+ return tail(ttnn.matmul(ah, y_in, compute_kernel_config=cfg))
161
+ if a.part == "tail":
162
+ return tail(z_in)
163
+ if a.part == "mid":
164
+ return mid(yw_in)
165
+ if a.part == "mid_h":
166
+ return ttnn.matmul(ah, mid(yw_in), compute_kernel_config=cfg)
167
+ x = ttnn.to_layout(x_in, ttnn.ROW_MAJOR_LAYOUT)
168
+ x = ttnn.reshape(x, (1, H, W, C))
169
+ x = ttnn.to_layout(x, ttnn.TILE_LAYOUT)
170
+ if dt != ttnn.bfloat16:
171
+ x = ttnn.typecast(x, dt)
172
+ xt = ttnn.transpose(x, -2, -1)
173
+ y = ttnn.matmul(xt, aw_t, compute_kernel_config=cfg)
174
+ y = ttnn.transpose(y, -2, -1)
175
+ if a.part == "w":
176
+ return y
177
+ y = mid(y)
178
+ if a.part == "w_rm":
179
+ return y
180
+ z = ttnn.matmul(ah, y, compute_kernel_config=cfg)
181
+ if a.part == "wh":
182
+ return z
183
+ return tail(z)
184
+
185
+ warm = [chain() for _ in range(a.per_trace)] # compile, then capture
186
+ ttnn.synchronize_device(dev)
187
+ del warm
188
+ tid = ttnn.begin_trace_capture(dev, cq_id=0)
189
+ outs = [chain() for _ in range(a.per_trace)]
190
+ ttnn.end_trace_capture(dev, tid, cq_id=0)
191
+ ttnn.synchronize_device(dev)
192
+ log(fh, ev="captured")
193
+ ref = None
194
+ for i in range(a.iters):
195
+ faulthandler.dump_traceback_later(a.stall_s, repeat=True)
196
+ t0 = time.perf_counter()
197
+ ttnn.execute_trace(dev, tid, cq_id=0, blocking=False)
198
+ ttnn.synchronize_device(dev)
199
+ rec = {"ev": "iter", "i": i, "ms": round((time.perf_counter() - t0) * 1e3, 2)}
200
+ if i % a.check_every == 0 or i == a.iters - 1:
201
+ h = hashlib.sha256()
202
+ for t in outs:
203
+ h.update(ttnn.to_torch(t).float().numpy().tobytes())
204
+ d = h.hexdigest()[:16]
205
+ rec["digest"] = d
206
+ ref = ref or d
207
+ if d != ref:
208
+ rec["MISMATCH"] = ref
209
+ rc = 1
210
+ log(fh, **rec)
211
+ faulthandler.cancel_dump_traceback_later()
212
+ log(fh, ev="done", iters=a.iters, rc=rc)
213
+ ttnn.release_trace(dev, tid)
214
+ finally:
215
+ faulthandler.dump_traceback_later(300, repeat=True)
216
+ ttnn.close_device(dev)
217
+ faulthandler.cancel_dump_traceback_later()
218
+ log(fh, ev="closed")
219
+ return rc
220
+
221
+
222
+ if __name__ == "__main__":
223
+ sys.exit(main())
code/scripts/run_public_frames.py ADDED
@@ -0,0 +1,85 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Run the served model (``METEOR.from_pretrained``: the ``frame`` trace, ETH 12x10) on the public-dataset frames of
4
+ ``research/meteor/public_data`` and write the TT outputs in the layout of the CPU goldens (``fNNNN.npz`` compact +
5
+ ``fNNNN.json`` decoded, ``run_public_reference.save_frame``), so ``compare_tt.py`` scores them against the CPU
6
+ goldens and ``make_demo.py`` renders them. Development tool of the workspace (needs the research scripts and the
7
+ converted public inputs, which never ship; nuScenes is CC BY-NC-SA: local validation only).
8
+
9
+ bin/devrun -t 1800 -- python -u code/scripts/run_public_frames.py --out logs/public_tt [--ids ps019,ps090,ns0103]
10
+
11
+ Every frame's graph feed is checked against the golden's ``input_sha256`` first (the same pixels, K, T and v0 as the
12
+ CPU reference). Frames are processed per sequence (one rig each: the lift tables are written once per sequence).
13
+ Also writes ``<out>/run.json`` (versions, device, per-frame seconds) and, with ``--full`` frames, the 19 outputs.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import glob
19
+ import json
20
+ import os
21
+ import sys
22
+ import time
23
+ from pathlib import Path
24
+
25
+ import numpy as np
26
+
27
+ ROOT = Path(os.environ.get("TT_MODELS_ROOT", "/home/ubuntu/experiments/tt-models"))
28
+ PUBLIC = ROOT / "research" / "meteor" / "public_data"
29
+ sys.path.insert(0, str(PUBLIC / "scripts"))
30
+ sys.path.insert(0, str(ROOT / "research" / "meteor" / "scripts"))
31
+
32
+
33
+ def main() -> None:
34
+ ap = argparse.ArgumentParser()
35
+ ap.add_argument("--out", required=True, type=Path)
36
+ ap.add_argument("--ids", default="ps019,ps090,ns0103")
37
+ ap.add_argument("--full", default="ps019:20,ps090:20,ns0103:9", help="id:frame pairs that also get full_fNNNN.npz")
38
+ a = ap.parse_args()
39
+ import me_public_common as C # noqa: F401 (research decode, metrics)
40
+ import meteor_io
41
+ from run_public_reference import save_frame
42
+
43
+ from tt_meteor import METEOR, __version__
44
+ from tt_meteor.host.preprocess import MeteorFrame
45
+
46
+ full = {tuple(p.split(":")) for p in a.full.split(",") if p}
47
+ jobs = []
48
+ for gid in a.ids.split(","):
49
+ gdir = PUBLIC / "golden" / gid
50
+ frames = sorted(int(Path(f).stem[1:]) for f in glob.glob(str(gdir / "f[0-9][0-9][0-9][0-9].json")))
51
+ jobs.append((gid, frames))
52
+ run = {"bundle_version": __version__, "frames": {}, "started": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())}
53
+ t_load = time.time()
54
+ with METEOR.from_pretrained() as model:
55
+ run["load_s"] = round(time.time() - t_load, 1)
56
+ run["device"] = {k: model.device_info.get(k) for k in ("dispatch", "grid", "num_command_queues", "cores")}
57
+ print("device", run["device"], "load", run["load_s"], "s", flush=True)
58
+ for gid, frames in jobs:
59
+ root = PUBLIC / "inputs" / gid
60
+ sd = meteor_io.scene_dir_of(str(root)) if hasattr(meteor_io, "scene_dir_of") else \
61
+ str(root / (root / "scenes.txt").read_text().split()[0])
62
+ man = meteor_io.load_manifest(sd)
63
+ ego_gt = dict(np.load(os.path.join(sd, "ego_motion.npz")))
64
+ present = np.asarray(man.get("present_mask", [1] * 8), bool)
65
+ out_dir = a.out / gid
66
+ out_dir.mkdir(parents=True, exist_ok=True)
67
+ for i in frames:
68
+ feed = meteor_io.load_frame(sd, i, man)
69
+ gold = json.loads((PUBLIC / "golden" / gid / f"f{i:04d}.json").read_text())
70
+ sha = {k: C.sha256_array(v) for k, v in feed.items()}
71
+ if sha != gold["input_sha256"]:
72
+ raise SystemExit(f"{gid} f{i}: the feed differs from the golden's input_sha256")
73
+ fr = MeteorFrame(feed["imgs"], feed["K"], feed["T_cam_ego"], feed["v0"], present)
74
+ t0 = time.time()
75
+ o = {k: np.asarray(v) for k, v in model._forward({"frame": fr}).items()}
76
+ dt = time.time() - t0
77
+ save_frame(str(out_dir), i, o, feed, man, ego_gt, full=(gid, str(i)) in full)
78
+ run["frames"][f"{gid}/f{i:04d}"] = round(dt, 3)
79
+ print(f"{gid} f{i:04d}: {dt * 1e3:.0f} ms", flush=True)
80
+ run["finished"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
81
+ (a.out / "run.json").write_text(json.dumps(run, indent=1) + "\n")
82
+
83
+
84
+ if __name__ == "__main__":
85
+ main()
code/scripts/stress_frames.py ADDED
@@ -0,0 +1,227 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Contained stress runs of the ``frame`` trace (the METEOR hang investigation, RESET_INVESTIGATION.md section 6).
4
+
5
+ METEOR_DEVICE_OK=1 bin/devrun -t 900 -- python -u code/scripts/stress_frames.py --mode replay --frames 200
6
+ METEOR_DEVICE_OK=1 bin/devrun -t 1500 -- python -u code/scripts/stress_frames.py --mode rigs --frames 100
7
+
8
+ ``--mode replay`` (R1): the model is built with the sample rig's tables and cameras as its warm-up data, captured,
9
+ then the frame trace is replayed ``--frames`` times with NO upload at all, reading back every ``--read-every``-th
10
+ replay. ``--mode rigs`` (R2): the model is built like the API (zero tables), then every frame cycles to the next of
11
+ the golden rigs: table write (when the rig changes) + camera upload + replay + segmented readback.
12
+
13
+ Every read is checked: bit-identical to the first read of the same rig (corruption / non-determinism shows up as a
14
+ mismatch) and agreement with the CPU golden (``lane`` argmax, ``hm`` PCC).
15
+
16
+ ``--rigs shipped[:N]`` needs no golden (for the container, which has none): N rigs (default 4) made from the shipped
17
+ synthetic sample, rig k with every camera raised by 5k cm (a different calibration, so different lift tables, and the
18
+ same cameras); only the bit-identity per rig is checked. Inside the container image (``/opt/tt-metal``)::
19
+
20
+ python -u /opt/tt-metal/scripts/stress_frames.py --mode rigs --rigs shipped --frames 200 One heartbeat line per frame goes to
21
+ stdout (run with ``python -u``) and one JSON line to ``--jsonl`` (fsync'd); ``faulthandler`` dumps the Python stack
22
+ when no frame finishes for ``--stall-s`` seconds (a hang leaves evidence of where it stopped, without touching the
23
+ device). Nothing here arms tt-triage or an operation timeout.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import argparse
28
+ import faulthandler
29
+ import hashlib
30
+ import json
31
+ import os
32
+ import sys
33
+ import time
34
+ from pathlib import Path
35
+
36
+ import numpy as np
37
+
38
+ GOLDENS = Path(os.environ.get("METEOR_GOLDENS", "/home/ubuntu/experiments/tt-models/research/meteor/goldens"))
39
+ RIGS = ["pandaset_019_f40", "pandaset_090_f40", "nuscenes_0103_kf09", "meteor_valday_f040"]
40
+ INPUT_KEYS = ("input.imgs", "input.K", "input.T_cam_ego", "input.v0", "input.present")
41
+ CHECK_KEYS = ("lane", "hm", "ego")
42
+
43
+
44
+ def log(fh, **rec) -> None:
45
+ rec["t"] = round(time.time(), 3)
46
+ line = json.dumps(rec, default=str)
47
+ print(line, flush=True)
48
+ if fh is not None:
49
+ fh.write(line + "\n")
50
+ fh.flush()
51
+ os.fsync(fh.fileno())
52
+
53
+
54
+ def digest(out) -> str:
55
+ h = hashlib.sha256()
56
+ for k in sorted(out):
57
+ a = np.ascontiguousarray(out[k])
58
+ h.update(k.encode())
59
+ h.update(a.tobytes())
60
+ return h.hexdigest()[:16]
61
+
62
+
63
+ def checks(out, gold) -> dict:
64
+ from tt_meteor.ttaw.metrics import pcc
65
+
66
+ lane = float((np.asarray(out["lane"]).reshape(-1) == np.asarray(gold["lane"]).reshape(-1)).mean())
67
+ hm = float(pcc(np.asarray(out["hm"], np.float64), np.asarray(gold["hm"], np.float64)))
68
+ return {"lane_agree": round(lane, 5), "hm_pcc": round(hm, 6)}
69
+
70
+
71
+ def main() -> int:
72
+ ap = argparse.ArgumentParser()
73
+ ap.add_argument("--mode", choices=["replay", "rigs", "segrigs"], required=True)
74
+ ap.add_argument("--frames", type=int, default=100)
75
+ ap.add_argument("--read-every", type=int, default=1)
76
+ ap.add_argument("--rigs", default=",".join(RIGS))
77
+ ap.add_argument("--dispatch", default="eth", choices=["eth", "worker"])
78
+ ap.add_argument("--open-cqs", type=int, default=None, help="HW command queues to open (default: DEVICE_DEFAULTS)")
79
+ ap.add_argument("--runner-cqs", type=int, default=None, help="CQs the TraceRunner uses (default: as opened)")
80
+ ap.add_argument("--lift-reps", type=int, default=1,
81
+ help="segrigs: replay the lift segment(s) this many times per frame (sync + heartbeat after each)")
82
+ ap.add_argument("--split-lift", nargs="?", const=True, default=False,
83
+ help="segrigs: lift as lift_gs (grid_samples) + lift_rest; 'fine': lift_rest as 4 op groups")
84
+ ap.add_argument("--no-tables", action="store_true", help="segrigs: one rig only, tables written once")
85
+ ap.add_argument("--stall-s", type=float, default=120.0)
86
+ ap.add_argument("--jsonl")
87
+ a = ap.parse_args()
88
+
89
+ faulthandler.enable()
90
+ fh = open(a.jsonl, "a") if a.jsonl else None
91
+ rigs = a.rigs.split(",")
92
+ faulthandler.dump_traceback_later(900, repeat=True) # build / warm-up / capture
93
+ log(fh, ev="start", mode=a.mode, frames=a.frames, read_every=a.read_every, rigs=rigs, pid=os.getpid())
94
+
95
+ import ttnn
96
+
97
+ from tt_meteor.device import close_device, describe_device, open_device
98
+ from tt_meteor.host.calib import lift_geometry
99
+ from tt_meteor.host.preprocess import MeteorFrame
100
+ from tt_meteor.tt.lift import lift_tables
101
+ from tt_meteor.tt.model import TtMETEOR
102
+ from tt_meteor.tt.params import MeteorParams
103
+ from tt_meteor.tt.unpack import outputs_from_device
104
+
105
+ t0 = time.perf_counter()
106
+ params = MeteorParams.load()
107
+ data = {}
108
+ if len(rigs) == 1 and rigs[0].startswith("shipped"):
109
+ from tt_meteor.api import load_sample
110
+ from tt_meteor.host.inputs import SAMPLES_DIR, prepare_request
111
+
112
+ n = int(rigs[0].partition(":")[2] or 4)
113
+ kw = load_sample(SAMPLES_DIR / "synthetic_8cam.json")
114
+ base = prepare_request(kw["images"], kw["calibration"], kw["ego_speed"], None)
115
+ rigs = [f"shipped{k}" for k in range(n)]
116
+ for k, name in enumerate(rigs):
117
+ shift = np.eye(4, dtype=np.float32)
118
+ shift[2, 3] = -0.05 * k # ego -> ego lowered by 5k cm = every camera raised by 5k cm
119
+ T = (base.T_cam_ego.astype(np.float64) @ shift).astype(np.float32)
120
+ fr = MeteorFrame(base.imgs, base.K, T, base.v0, base.present)
121
+ data[name] = (fr, lift_geometry(fr.K, fr.T_cam_ego, points=params.ground_points()), None)
122
+ for name in ([] if data else rigs):
123
+ with np.load(GOLDENS / name / "taps.npz") as z:
124
+ g = {k: z[k] for k in INPUT_KEYS + CHECK_KEYS}
125
+ fr = MeteorFrame(g["input.imgs"], g["input.K"], g["input.T_cam_ego"], g["input.v0"], g["input.present"])
126
+ geom = lift_geometry(fr.K, fr.T_cam_ego, points=params.ground_points())
127
+ data[name] = (fr, geom, g)
128
+ log(fh, ev="host_ready", s=round(time.perf_counter() - t0, 1), keys={n: d[1].key for n, d in data.items()})
129
+
130
+ dev = open_device(dispatch=a.dispatch, allow_fallback=False, num_command_queues=a.open_cqs)
131
+ rc = 0
132
+ try:
133
+ log(fh, ev="device_open", info=describe_device(dev))
134
+ first = rigs[0]
135
+ fr0, geom0, _ = data[first]
136
+ if a.mode == "replay":
137
+ warm = {"imgs": TtMETEOR.image_rows(fr0.imgs), "v0": float(np.asarray(fr0.v0).reshape(-1)[0]),
138
+ **lift_tables(geom0)}
139
+ tt = TtMETEOR(dev, params, warmup=warm, tables_key=geom0.key, num_command_queues=a.runner_cqs)
140
+ elif a.mode == "segrigs":
141
+ tt = TtMETEOR(dev, params, segmented=True, split_lift=a.split_lift, num_command_queues=a.runner_cqs)
142
+ else:
143
+ tt = TtMETEOR(dev, params, num_command_queues=a.runner_cqs)
144
+ t1 = time.perf_counter()
145
+ tt.capture()
146
+ ttnn.synchronize_device(dev)
147
+ log(fh, ev="captured", s=round(time.perf_counter() - t1, 1), timings=tt.runner.timings_ms,
148
+ program_cache_entries=tt.describe().get("program_cache_entries"))
149
+ faulthandler.dump_traceback_later(a.stall_s, repeat=True)
150
+ ref_digest = {}
151
+ mismatches = 0
152
+ for i in range(a.frames):
153
+ faulthandler.dump_traceback_later(a.stall_s, repeat=True) # re-armed per frame: fires only on a stall
154
+ ts = time.perf_counter()
155
+ if a.mode == "replay":
156
+ name = first
157
+ tt.runner.replay("frame")
158
+ ttnn.synchronize_device(dev)
159
+ rec = {"ev": "frame", "i": i, "rig": name, "replay_ms": round((time.perf_counter() - ts) * 1e3, 1)}
160
+ if i % a.read_every == 0 or i == a.frames - 1:
161
+ tr = time.perf_counter()
162
+ out = outputs_from_device(tt.unpack(tt.runner.read("frame")))
163
+ rec["read_ms"] = round((time.perf_counter() - tr) * 1e3, 1)
164
+ else:
165
+ out = None
166
+ elif a.mode == "segrigs":
167
+ name = rigs[0] if a.no_tables else rigs[i % len(rigs)]
168
+ fr, geom, _ = data[name]
169
+ tw = time.perf_counter()
170
+ wrote = tt.set_tables(geom)
171
+ rec = {"ev": "frame", "i": i, "rig": name, "tables": wrote,
172
+ "tables_ms": round((time.perf_counter() - tw) * 1e3, 1)}
173
+ r = tt.runner
174
+ for k, stage in enumerate(tt.segments):
175
+ ta = time.perf_counter()
176
+ print(f"[seg] i={i} {stage} start", flush=True)
177
+ if k == 0:
178
+ r.run("seg_image", inputs={"imgs": tt.image_rows(fr.imgs)}, params=tt.frame_params(fr))
179
+ else:
180
+ r.run(f"seg_{stage}")
181
+ ttnn.synchronize_device(dev)
182
+ if stage.startswith(("lift", "lr_")):
183
+ for rep in range(1, a.lift_reps):
184
+ print(f"[seg] i={i} {stage} rep {rep} start", flush=True)
185
+ r.run(f"seg_{stage}")
186
+ ttnn.synchronize_device(dev)
187
+ rec[f"{stage}_ms"] = round((time.perf_counter() - ta) * 1e3, 1)
188
+ ego = np.array(r.read("seg_head"), copy=True)
189
+ out = {"ego": ego}
190
+ else:
191
+ name = rigs[i % len(rigs)]
192
+ fr, geom, _ = data[name]
193
+ tw = time.perf_counter()
194
+ wrote = tt.set_tables(geom)
195
+ tables_ms = (time.perf_counter() - tw) * 1e3
196
+ tr = time.perf_counter()
197
+ out = tt.run_frame(fr, geom)
198
+ rec = {"ev": "frame", "i": i, "rig": name, "tables": wrote, "tables_ms": round(tables_ms, 1),
199
+ "run_ms": round((time.perf_counter() - tr) * 1e3, 1)}
200
+ if out is not None:
201
+ d = digest(out)
202
+ rec["digest"] = d
203
+ if name not in ref_digest:
204
+ ref_digest[name] = d
205
+ if "lane" in out and data[name][2] is not None:
206
+ rec.update(checks(out, data[name][2]))
207
+ elif d != ref_digest[name]:
208
+ mismatches += 1
209
+ rec["MISMATCH_vs"] = ref_digest[name]
210
+ if "lane" in out and data[name][2] is not None:
211
+ rec.update(checks(out, data[name][2]))
212
+ log(fh, **rec)
213
+ faulthandler.cancel_dump_traceback_later()
214
+ log(fh, ev="done", frames=a.frames, mismatches=mismatches, digests=ref_digest,
215
+ s=round(time.perf_counter() - t0, 1))
216
+ rc = 1 if mismatches else 0
217
+ tt.release()
218
+ finally:
219
+ faulthandler.dump_traceback_later(300, repeat=True)
220
+ close_device(dev)
221
+ faulthandler.cancel_dump_traceback_later()
222
+ log(fh, ev="closed", rc=rc)
223
+ return rc
224
+
225
+
226
+ if __name__ == "__main__":
227
+ sys.exit(main())
code/tt_meteor/__init__.py ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """METEOR (TIER IV, AutowareFoundation/meteor) on a Tenstorrent Blackhole p150 (tt-nn), packaged as changh95/meteor-p150.
3
+
4
+ Python API (see PYTHON.md)::
5
+
6
+ from tt_meteor import METEOR, load_sample
7
+
8
+ with METEOR.from_pretrained(device_id=0) as model:
9
+ out = model(**load_sample("code/tt_meteor/samples/synthetic_8cam.json"))
10
+
11
+ Importing the package has no side effects (no device, no ttnn / torch import, no network); the names below load
12
+ on first use. The HTTP server is ``tt_meteor.server.app:app`` (SERVING.md). ``tt_meteor.ttaw`` is the
13
+ vendored shared package of the Autoware ports (``ttaw/VENDORED.json`` records its version and file hashes; never
14
+ edit it here).
15
+ """
16
+
17
+ __version__ = "0.1.0"
18
+ __all__ = ["METEOR", "Output", "open_device", "load_points", "load_image", "load_sample", "__version__"]
19
+
20
+ _LAZY = {
21
+ "METEOR": (".api", "METEOR"),
22
+ "Output": (".api", "Output"),
23
+ "open_device": (".device", "open_device"),
24
+ "load_points": (".io", "load_points"),
25
+ "load_image": (".io", "load_image"),
26
+ "load_sample": (".api", "load_sample"),
27
+ }
28
+
29
+
30
+ def __getattr__(name):
31
+ if name in _LAZY:
32
+ import importlib
33
+
34
+ module, attr = _LAZY[name]
35
+ return getattr(importlib.import_module(module, __name__), attr)
36
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
37
+
38
+
39
+ def __dir__():
40
+ return sorted(list(globals()) + __all__)
code/tt_meteor/api.py ADDED
@@ -0,0 +1,205 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Python API: METEOR (TIER IV, AutowareFoundation/meteor) on one Tenstorrent Blackhole p150.
3
+
4
+ from tt_meteor import METEOR, load_sample
5
+
6
+ with METEOR.from_pretrained(device_id=0) as model: # weights -> HF cache, device open, traces captured
7
+ out = model(**load_sample("code/tt_meteor/samples/synthetic_8cam.json")) # 8 cameras + calibration + speed
8
+ print(out.to_dict()) # the same JSON as POST /predict
9
+
10
+ The contract shared by every bundle of the Autoware collection is the vendored ``ttaw.api_base.ModelBase``
11
+ (BUNDLE_CONVENTIONS.md section 8): ``from_pretrained`` resolves the pinned weights before it claims the chip, opens it
12
+ (ETH dispatch, 12x10), builds the graph and captures every trace variant in ``warmup_variants``, so the first call is
13
+ as fast as the later ones; calls are serialised by a lock (one chip, batch 1) and fill ``timing_ms``; ``close()`` is
14
+ idempotent, also runs at interpreter exit, and closes the chip only if the model opened it. The HTTP server
15
+ (``tt_meteor.server.app``) calls this class, so ``/predict`` and ``model(...)`` agree bit for bit.
16
+
17
+ Inputs per call (``_prepare``; host code in ``tt_meteor.host``, METEOR's own runtime semantics: there is no Autoware
18
+ package):
19
+
20
+ - ``images``: the cameras ``CAM_FRONT_WIDE``, ``CAM_FRONT_LEFT``, ``CAM_FRONT_RIGHT``, ``CAM_BACK_WIDE``,
21
+ ``CAM_BACK_LEFT``, ``CAM_BACK_RIGHT``, ``CAM_FRONT_NARROW``, ``CAM_BACK_NARROW`` (any order; raw, unrectified frames
22
+ of at least 768x432, resized with OpenCV INTER_AREA semantics); the two narrow cameras may be absent (zero image +
23
+ donor pose, as trained);
24
+ - ``calibration``: per camera ``intrinsics`` (of the image as sent) and ``T_ref_from_camera`` (camera optical frame ->
25
+ ego / base_link), or ``{"preset": name}``;
26
+ - ``ego_speed``: m/s (the graph's ``v0``);
27
+ - ``stream``: ``{"id", "reset", "timestamp_s", "T_world_from_ego"}`` (or ``"pose": [x, y, yaw]``): the host temporal
28
+ post-processing per stream (BEV seg fusion needs the pose; yaw smoothing; mode hysteresis).
29
+
30
+ Runtime params (``RUNTIME_PARAMS``): every knob of METEOR's C++ renderer (``host.postprocess.PostConfig``). Load-time
31
+ knobs: :data:`KNOBS` (``METEOR_*``): ``INPUT_NORM`` (``onnx`` = the released /255, D12; ``imagenet`` = the trained
32
+ normalisation), ``DEPTH_MEAN_BINS`` (``log`` as exported; ``linear``), ``MAX_STREAMS``.
33
+
34
+ The device graph is ``tt_meteor.tt.model.TtMETEOR``: the whole network in ONE trace (``frame``), replayed per call;
35
+ the host half (``_build_host``, ``_prepare``, ``_postprocess``) is what the host tests exercise. Importing this module
36
+ has no side effects (device code is imported in ``_build``).
37
+ """
38
+ from __future__ import annotations
39
+
40
+ from collections import OrderedDict
41
+ from typing import Any, Dict, Mapping, Optional
42
+
43
+ from . import io as tio
44
+ from .device import DEVICE_DEFAULTS
45
+ from .host.outputs import LABELS as _LABELS
46
+ from .host.outputs import MeteorOutput
47
+ from .reference.config import CAMERAS
48
+ from .ttaw.api_base import ModelBase
49
+ from .ttaw.knobs import Knob, Knobs
50
+
51
+ __all__ = ["METEOR", "Output", "KNOBS", "load_sample"]
52
+
53
+ # The result class of this model (the multi-task MeteorOutput of host.outputs; BUNDLE_CONVENTIONS 7.4 "per head").
54
+ Output = MeteorOutput
55
+
56
+ # Load-time switches (read once at build; env METEOR_<NAME>). Each optimization adds its knob here (A/B switch).
57
+ KNOBS = Knobs("METEOR", [
58
+ Knob("INPUT_NORM", "onnx", "input normalisation: the released graph's /255 (D12 default, ORT parity) or the "
59
+ "trained ImageNet mean/std (SPEC section 10 risk 1)", choices=("onnx", "imagenet")),
60
+ Knob("DEPTH_MEAN_BINS", "log", "depth_mean bin centres: log-spaced as exported (parity) or linear 1 + 1.25 b "
61
+ "(the trained bins, SPEC section 10 risk 2)", choices=("log", "linear")),
62
+ Knob("MAX_STREAMS", 16, "host temporal states kept (seg fusion, yaw tracks, mode); the least recent is dropped"),
63
+ Knob("IMAGE_PRECISION", "terms3", "image branch (ResNet-34, FPN, image heads): terms3 = fp32 activations, every "
64
+ "conv as three bf16 terms (needed by the depth / seg2d argmax gates, PORT_LOG.md "
65
+ "E20); bf16 = bf16 activations (A/B only: below the depth argmax gate, "
66
+ "0.9797 < 0.99)", choices=("terms3", "bf16")),
67
+ ])
68
+
69
+
70
+ def load_sample(path: Any, *, with_stream: bool = True) -> Dict[str, Any]:
71
+ """``model(**load_sample("samples/<name>.json"))``: cameras + calibration preset + ego speed (+ stream)."""
72
+ from .host.inputs import load_sample as _load
73
+
74
+ return _load(path, with_stream=with_stream)
75
+
76
+
77
+ class METEOR(ModelBase):
78
+ """METEOR (TIER IV, AutowareFoundation/meteor) on one Blackhole p150. Create it with :meth:`from_pretrained`."""
79
+
80
+ MODEL_NAME = "meteor-p150"
81
+ ENV_PREFIX = "METEOR" # prefix of the environment knobs (SERVING.md section 3.4)
82
+ DEFAULT_REPO = "AutowareFoundation/meteor"
83
+ DEFAULT_TAG = "v1.0" # the AutowareFoundation release tag (no Autoware ansible pin exists for METEOR)
84
+ # the commit DEFAULT_TAG points to (pinned: tags can move)
85
+ DEFAULT_REVISION = "01a5f6d71df5ecbbb5853ec600825481d57b9c6b"
86
+ ALLOW_PATTERNS = ["meteor_v157c3Z.onnx", "meteor_v157.param.yaml", "LICENSE", "SHA256SUMS"]
87
+ VARIANTS = ["default"] # load-time: selects weights files and trace shapes
88
+ DEFAULT_VARIANT = "default"
89
+ INPUT_KIND = "multicam" # lidar | camera | multicam | lidar+multicam | planner
90
+ CAMERA_ORDER = CAMERAS # the network's fixed camera order (meteor_v157.param.yaml:10)
91
+ POINT_FIELDS = tio.DEFAULT_POINT_FIELDS # unused (camera model); kept for the shared /info schema
92
+ LABELS = list(_LABELS) # 3D box classes (label_id 0 vehicle, 1 VRU)
93
+ # Per-request knobs: name -> (type, min, max, default). Host-side post-processing only (SPEC section 9, RT-host;
94
+ # render.cpp:345-376 defaults).
95
+ RUNTIME_PARAMS = {
96
+ "det3d_threshold": (float, 0.0, 1.0, 0.15),
97
+ "det3d_topk": (int, 1, 1024, 64),
98
+ "vehicle_threshold": (float, 0.0, 1.0, 0.35),
99
+ "vru_threshold": (float, 0.0, 1.0, 0.15),
100
+ "bev_nms_iou": (float, 0.0, 1.0, 0.3),
101
+ "bev_nms_containment": (float, 0.0, 1.0, 0.6),
102
+ "stationary_logit_threshold": (float, None, None, 0.0),
103
+ "det2d_threshold": (float, 0.0, 1.0, 0.30),
104
+ "det2d_topk": (int, 1, 1024, 48),
105
+ "det2d_hide": (str, None, None, "7"),
106
+ "unk2d": (bool, None, None, True),
107
+ "unk2d_threshold": (float, 0.0, 1.0, None),
108
+ "ground_z": (float, -5.0, 5.0, 0.0),
109
+ "mode_hysteresis": (float, 0.0, 10.0, 0.35),
110
+ "straight_margin": (float, 0.0, 10.0, 1.0),
111
+ "seg_fuse": (bool, None, None, True),
112
+ "thin_road_edge": (bool, None, None, True),
113
+ "yaw_smoothing": (bool, None, None, True),
114
+ "heads": (bool, None, None, False),
115
+ }
116
+ EXTRA_INPUTS = ("ego_speed",)
117
+ DEVICE_DEFAULTS = DEVICE_DEFAULTS # validated open parameters (device.py)
118
+
119
+ # ---- port-specific hooks (called by ModelBase; keep host work out of _forward) ---------------------------
120
+ def _build(self) -> None:
121
+ """Weights (``reference.weights.MeteorWeights``: the ONNX parameters by consuming node) -> the device graph of
122
+ ``tt_meteor.tt`` (``TtMETEOR``: the whole graph as ONE ``TraceRunner`` variant ``frame``, no capture yet)."""
123
+ self._build_host()
124
+ from .host.calib import CalibrationCache
125
+ from .reference.weights import MeteorWeights
126
+ from .tt.model import TtMETEOR
127
+ from .tt.params import MeteorParams
128
+
129
+ params = MeteorParams(MeteorWeights(self.weights_path), self.cfg)
130
+ self.calib_cache = CalibrationCache(points=params.ground_points())
131
+ self.tt = TtMETEOR(self.device, params, num_command_queues=self.device_info.get("num_command_queues"),
132
+ image_precision=str(self.knobs.IMAGE_PRECISION))
133
+ self.runner = self.tt.runner
134
+
135
+ def _warm_one(self, variant: Any) -> None:
136
+ """Warm-up of the ``frame`` variant (compiles every program) and its capture."""
137
+ self.runner.capture()
138
+
139
+ def _build_host(self) -> None:
140
+ """The host half of the model: load-time knobs and the per-stream host temporal states (no device, no
141
+ weights; the host tests call it directly)."""
142
+ from .host.postprocess import PostConfig
143
+ from .reference.config import default_config
144
+
145
+ params = {k.upper(): v for k, v in getattr(self, "compile_params", {}).items() if k.upper() in KNOBS.knobs}
146
+ self.knobs = KNOBS.read(**params)
147
+ self.cfg = default_config(input_norm=str(self.knobs.INPUT_NORM),
148
+ depth_mean_bins=str(self.knobs.DEPTH_MEAN_BINS))
149
+ self.post = PostConfig()
150
+ self._streams: "OrderedDict[str, Any]" = OrderedDict()
151
+
152
+ def _stream_state(self, stream: Optional[Mapping[str, Any]]):
153
+ from .host.temporal import StreamState
154
+
155
+ sid = str((stream or {}).get("id", "default"))
156
+ st = self._streams.get(sid)
157
+ if st is None:
158
+ st = self._streams[sid] = StreamState()
159
+ while len(self._streams) > int(self.knobs.MAX_STREAMS):
160
+ self._streams.popitem(last=False)
161
+ elif (stream or {}).get("reset"):
162
+ st.reset()
163
+ self._streams.move_to_end(sid)
164
+ return sid, st
165
+
166
+ def _prepare(self, points: Any = None, *, images: Any = None, calibration: Any = None, stream: Any = None,
167
+ ego_speed: Any = None, **_: Any) -> Dict[str, Any]:
168
+ """Cameras + calibration + ego speed -> the graph feed (``host.inputs.prepare_request``)."""
169
+ from .host.inputs import prepare_request
170
+
171
+ if points is not None:
172
+ raise tio.InputError("this model takes eight camera images (+ calibration, ego_speed), not a point cloud")
173
+ frame = prepare_request(images, calibration, ego_speed, stream)
174
+ sid, state = self._stream_state(stream)
175
+ return {"frame": frame, "stream_id": sid, "state": state}
176
+
177
+ def _forward(self, prepared: Dict[str, Any]) -> Any:
178
+ """Upload (cameras, v0; the lift tables when the rig changed) + replay of the ``frame`` trace + one packed
179
+ read -> the 19 outputs (``tt.unpack``: ONNX names / layouts / dtypes)."""
180
+ frame = prepared["frame"]
181
+ geom = self.calib_cache.get(frame.K, frame.T_cam_ego)
182
+ return self.tt.run_frame(frame, geom)
183
+
184
+ def _postprocess(self, raw: Any, prepared: Dict[str, Any], params: Dict[str, Any]) -> Output:
185
+ """METEOR's C++ host decode on the 19 outputs (``host.result.build_output``) with this stream's state."""
186
+ from .host.result import build_output
187
+
188
+ hide = tuple(int(v) for v in str(params.get("det2d_hide", "7")).replace(" ", "").split(",") if v != "")
189
+ knobs = {k: v for k, v in params.items() if k not in ("det2d_hide", "heads")}
190
+ cfg = self.post.with_params(det2d_hide=hide, **knobs)
191
+ return build_output(raw, prepared["frame"], cfg, prepared["state"], heads=bool(params.get("heads")),
192
+ model=self.MODEL_NAME, meta={"stream_id": prepared["stream_id"]})
193
+
194
+ def _release(self) -> None:
195
+ """Release the traces and persistent device tensors; also called when ``from_pretrained`` fails half-way."""
196
+ runner = getattr(self, "runner", None)
197
+ if runner is not None:
198
+ runner.release()
199
+
200
+ def extra_info(self) -> Dict[str, Any]:
201
+ runner = getattr(self, "runner", None)
202
+ info: Dict[str, Any] = {"knobs": self.knobs.as_dict()} if hasattr(self, "knobs") else {}
203
+ if runner is not None:
204
+ info["trace"] = runner.describe()
205
+ return info
code/tt_meteor/calib/README.md ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Calibration presets
2
+
3
+ `<sensor_kit>.json` files in the calibration envelope of SERVING.md section 3.3 (`{"frame_id", "cameras": {name:
4
+ {"intrinsics", "T_ref_from_camera", "image_size"}}, "T_ref_from_lidar"}`), e.g. the camera rig of the public sample
5
+ data. The server lists them in `/info` (`calibration_presets`) and a request may say `"calibration": {"preset":
6
+ "<name>"}` instead of sending the matrices. Cite the source file and commit of every number.
7
+
8
+ | preset | rig | source |
9
+ |---|---|---|
10
+ | `synthetic_8cam.json` | the generated rig of the shipped sample `samples/synthetic_8cam.json`: METEOR's eight slots, K at 768x432 (98 deg wide front / back and corner cameras, 30.4 deg narrow cameras, principal point at the image centre, fy = 0.87 fx), `T_ref_from_camera` = camera optical frame -> base_link (x forward, y left, z up, origin on the road), mounted 1.9 m high, corner cameras at +-60 / +-120 deg yaw pitched 25 deg down; `_rig` lists the round numbers | `code/scripts/make_synthetic_sample.py` (no third-party data), Apache-2.0 |
11
+
12
+ The PandaSet 019 rig (`pandaset_019.json`, CC BY 4.0 + PandaSet Dataset Terms) belongs to the staged PandaSet sample
13
+ and is not shipped yet (`staging_samples_pandaset/calib/`).
code/tt_meteor/calib/synthetic_8cam.json ADDED
@@ -0,0 +1,464 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "frame_id": "base_link",
3
+ "_source": "generated by code/scripts/make_synthetic_sample.py (a generic METEOR-like 8-camera rig of round numbers; no third-party data), Apache-2.0",
4
+ "_rig": {
5
+ "CAM_FRONT_WIDE": {
6
+ "x": 2.1,
7
+ "y": 0.0,
8
+ "z": 1.9,
9
+ "yaw_deg": 0.0,
10
+ "pitch_down_deg": 2.0,
11
+ "hfov_deg": 98.0
12
+ },
13
+ "CAM_FRONT_LEFT": {
14
+ "x": 1.9,
15
+ "y": 0.85,
16
+ "z": 1.9,
17
+ "yaw_deg": 60.0,
18
+ "pitch_down_deg": 25.0,
19
+ "hfov_deg": 98.0
20
+ },
21
+ "CAM_FRONT_RIGHT": {
22
+ "x": 1.9,
23
+ "y": -0.85,
24
+ "z": 1.9,
25
+ "yaw_deg": -60.0,
26
+ "pitch_down_deg": 25.0,
27
+ "hfov_deg": 98.0
28
+ },
29
+ "CAM_BACK_WIDE": {
30
+ "x": -0.9,
31
+ "y": 0.0,
32
+ "z": 1.9,
33
+ "yaw_deg": 180.0,
34
+ "pitch_down_deg": 2.0,
35
+ "hfov_deg": 98.0
36
+ },
37
+ "CAM_BACK_LEFT": {
38
+ "x": -0.7,
39
+ "y": 0.85,
40
+ "z": 1.9,
41
+ "yaw_deg": 120.0,
42
+ "pitch_down_deg": 25.0,
43
+ "hfov_deg": 98.0
44
+ },
45
+ "CAM_BACK_RIGHT": {
46
+ "x": -0.7,
47
+ "y": -0.85,
48
+ "z": 1.9,
49
+ "yaw_deg": -120.0,
50
+ "pitch_down_deg": 25.0,
51
+ "hfov_deg": 98.0
52
+ },
53
+ "CAM_FRONT_NARROW": {
54
+ "x": 2.1,
55
+ "y": 0.0,
56
+ "z": 1.88,
57
+ "yaw_deg": 0.0,
58
+ "pitch_down_deg": 1.0,
59
+ "hfov_deg": 30.4
60
+ },
61
+ "CAM_BACK_NARROW": {
62
+ "x": -0.9,
63
+ "y": 0.0,
64
+ "z": 1.88,
65
+ "yaw_deg": 180.0,
66
+ "pitch_down_deg": 2.0,
67
+ "hfov_deg": 30.4
68
+ }
69
+ },
70
+ "cameras": {
71
+ "CAM_FRONT_WIDE": {
72
+ "intrinsics": [
73
+ [
74
+ 333.8061,
75
+ 0.0,
76
+ 384.0
77
+ ],
78
+ [
79
+ 0.0,
80
+ 290.4113,
81
+ 216.0
82
+ ],
83
+ [
84
+ 0.0,
85
+ 0.0,
86
+ 1.0
87
+ ]
88
+ ],
89
+ "T_ref_from_camera": [
90
+ [
91
+ 0.0,
92
+ -0.034899497,
93
+ 0.999390827,
94
+ 2.1
95
+ ],
96
+ [
97
+ -1.0,
98
+ -0.0,
99
+ 0.0,
100
+ 0.0
101
+ ],
102
+ [
103
+ 0.0,
104
+ -0.999390827,
105
+ -0.034899497,
106
+ 1.9
107
+ ],
108
+ [
109
+ 0.0,
110
+ 0.0,
111
+ 0.0,
112
+ 1.0
113
+ ]
114
+ ],
115
+ "image_size": [
116
+ 768,
117
+ 432
118
+ ]
119
+ },
120
+ "CAM_FRONT_LEFT": {
121
+ "intrinsics": [
122
+ [
123
+ 333.8061,
124
+ 0.0,
125
+ 384.0
126
+ ],
127
+ [
128
+ 0.0,
129
+ 290.4113,
130
+ 216.0
131
+ ],
132
+ [
133
+ 0.0,
134
+ 0.0,
135
+ 1.0
136
+ ]
137
+ ],
138
+ "T_ref_from_camera": [
139
+ [
140
+ 0.866025404,
141
+ -0.211309131,
142
+ 0.453153894,
143
+ 1.9
144
+ ],
145
+ [
146
+ -0.5,
147
+ -0.365998151,
148
+ 0.784885567,
149
+ 0.85
150
+ ],
151
+ [
152
+ 0.0,
153
+ -0.906307787,
154
+ -0.422618262,
155
+ 1.9
156
+ ],
157
+ [
158
+ 0.0,
159
+ 0.0,
160
+ 0.0,
161
+ 1.0
162
+ ]
163
+ ],
164
+ "image_size": [
165
+ 768,
166
+ 432
167
+ ]
168
+ },
169
+ "CAM_FRONT_RIGHT": {
170
+ "intrinsics": [
171
+ [
172
+ 333.8061,
173
+ 0.0,
174
+ 384.0
175
+ ],
176
+ [
177
+ 0.0,
178
+ 290.4113,
179
+ 216.0
180
+ ],
181
+ [
182
+ 0.0,
183
+ 0.0,
184
+ 1.0
185
+ ]
186
+ ],
187
+ "T_ref_from_camera": [
188
+ [
189
+ -0.866025404,
190
+ -0.211309131,
191
+ 0.453153894,
192
+ 1.9
193
+ ],
194
+ [
195
+ -0.5,
196
+ 0.365998151,
197
+ -0.784885567,
198
+ -0.85
199
+ ],
200
+ [
201
+ 0.0,
202
+ -0.906307787,
203
+ -0.422618262,
204
+ 1.9
205
+ ],
206
+ [
207
+ 0.0,
208
+ 0.0,
209
+ 0.0,
210
+ 1.0
211
+ ]
212
+ ],
213
+ "image_size": [
214
+ 768,
215
+ 432
216
+ ]
217
+ },
218
+ "CAM_BACK_WIDE": {
219
+ "intrinsics": [
220
+ [
221
+ 333.8061,
222
+ 0.0,
223
+ 384.0
224
+ ],
225
+ [
226
+ 0.0,
227
+ 290.4113,
228
+ 216.0
229
+ ],
230
+ [
231
+ 0.0,
232
+ 0.0,
233
+ 1.0
234
+ ]
235
+ ],
236
+ "T_ref_from_camera": [
237
+ [
238
+ 0.0,
239
+ 0.034899497,
240
+ -0.999390827,
241
+ -0.9
242
+ ],
243
+ [
244
+ 1.0,
245
+ -0.0,
246
+ 0.0,
247
+ 0.0
248
+ ],
249
+ [
250
+ 0.0,
251
+ -0.999390827,
252
+ -0.034899497,
253
+ 1.9
254
+ ],
255
+ [
256
+ 0.0,
257
+ 0.0,
258
+ 0.0,
259
+ 1.0
260
+ ]
261
+ ],
262
+ "image_size": [
263
+ 768,
264
+ 432
265
+ ]
266
+ },
267
+ "CAM_BACK_LEFT": {
268
+ "intrinsics": [
269
+ [
270
+ 333.8061,
271
+ 0.0,
272
+ 384.0
273
+ ],
274
+ [
275
+ 0.0,
276
+ 290.4113,
277
+ 216.0
278
+ ],
279
+ [
280
+ 0.0,
281
+ 0.0,
282
+ 1.0
283
+ ]
284
+ ],
285
+ "T_ref_from_camera": [
286
+ [
287
+ 0.866025404,
288
+ 0.211309131,
289
+ -0.453153894,
290
+ -0.7
291
+ ],
292
+ [
293
+ 0.5,
294
+ -0.365998151,
295
+ 0.784885567,
296
+ 0.85
297
+ ],
298
+ [
299
+ 0.0,
300
+ -0.906307787,
301
+ -0.422618262,
302
+ 1.9
303
+ ],
304
+ [
305
+ 0.0,
306
+ 0.0,
307
+ 0.0,
308
+ 1.0
309
+ ]
310
+ ],
311
+ "image_size": [
312
+ 768,
313
+ 432
314
+ ]
315
+ },
316
+ "CAM_BACK_RIGHT": {
317
+ "intrinsics": [
318
+ [
319
+ 333.8061,
320
+ 0.0,
321
+ 384.0
322
+ ],
323
+ [
324
+ 0.0,
325
+ 290.4113,
326
+ 216.0
327
+ ],
328
+ [
329
+ 0.0,
330
+ 0.0,
331
+ 1.0
332
+ ]
333
+ ],
334
+ "T_ref_from_camera": [
335
+ [
336
+ -0.866025404,
337
+ 0.211309131,
338
+ -0.453153894,
339
+ -0.7
340
+ ],
341
+ [
342
+ 0.5,
343
+ 0.365998151,
344
+ -0.784885567,
345
+ -0.85
346
+ ],
347
+ [
348
+ 0.0,
349
+ -0.906307787,
350
+ -0.422618262,
351
+ 1.9
352
+ ],
353
+ [
354
+ 0.0,
355
+ 0.0,
356
+ 0.0,
357
+ 1.0
358
+ ]
359
+ ],
360
+ "image_size": [
361
+ 768,
362
+ 432
363
+ ]
364
+ },
365
+ "CAM_FRONT_NARROW": {
366
+ "intrinsics": [
367
+ [
368
+ 1413.3548,
369
+ 0.0,
370
+ 384.0
371
+ ],
372
+ [
373
+ 0.0,
374
+ 1229.6187,
375
+ 216.0
376
+ ],
377
+ [
378
+ 0.0,
379
+ 0.0,
380
+ 1.0
381
+ ]
382
+ ],
383
+ "T_ref_from_camera": [
384
+ [
385
+ 0.0,
386
+ -0.017452406,
387
+ 0.999847695,
388
+ 2.1
389
+ ],
390
+ [
391
+ -1.0,
392
+ -0.0,
393
+ 0.0,
394
+ 0.0
395
+ ],
396
+ [
397
+ 0.0,
398
+ -0.999847695,
399
+ -0.017452406,
400
+ 1.88
401
+ ],
402
+ [
403
+ 0.0,
404
+ 0.0,
405
+ 0.0,
406
+ 1.0
407
+ ]
408
+ ],
409
+ "image_size": [
410
+ 768,
411
+ 432
412
+ ]
413
+ },
414
+ "CAM_BACK_NARROW": {
415
+ "intrinsics": [
416
+ [
417
+ 1413.3548,
418
+ 0.0,
419
+ 384.0
420
+ ],
421
+ [
422
+ 0.0,
423
+ 1229.6187,
424
+ 216.0
425
+ ],
426
+ [
427
+ 0.0,
428
+ 0.0,
429
+ 1.0
430
+ ]
431
+ ],
432
+ "T_ref_from_camera": [
433
+ [
434
+ 0.0,
435
+ 0.034899497,
436
+ -0.999390827,
437
+ -0.9
438
+ ],
439
+ [
440
+ 1.0,
441
+ -0.0,
442
+ 0.0,
443
+ 0.0
444
+ ],
445
+ [
446
+ 0.0,
447
+ -0.999390827,
448
+ -0.034899497,
449
+ 1.88
450
+ ],
451
+ [
452
+ 0.0,
453
+ 0.0,
454
+ 0.0,
455
+ 1.0
456
+ ]
457
+ ],
458
+ "image_size": [
459
+ 768,
460
+ 432
461
+ ]
462
+ }
463
+ }
464
+ }
code/tt_meteor/device.py ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Opening the Blackhole p150 the way every published number of meteor-p150 was measured.
3
+
4
+ The implementation is the vendored ``ttaw.device`` (C01): dispatch on the idle ETH cores
5
+ (``ttnn.DispatchCoreConfig(ttnn.DispatchCoreType.ETH)``), which gives a 12x10 = 120-core compute grid on a p150
6
+ (11x10 with WORKER dispatch, the A/B switch). ETH dispatch needs ``patches/tt-metal-eth-dispatch.patch`` on tt-metal
7
+ 44d6650 (the container image is built from a patched tree); if the ETH open fails, a ``RuntimeWarning`` is issued and
8
+ WORKER dispatch is used, unless ``allow_fallback=False``. Never hard-code the grid: use :func:`compute_grid` or
9
+ ``device.compute_with_storage_grid_size()``.
10
+
11
+ This module binds it to the validated open parameters of this port, :data:`DEVICE_DEFAULTS` (the model class uses
12
+ the same dict), which the ``METEOR_DISPATCH``, ``_NUM_CQS``, ``_L1_SMALL``, ``_TRACE_REGION`` and
13
+ ``_WORKER_L1_SIZE`` variables and ``TT_DEVICE_ID`` override per process (SERVING.md section 3.4). Importing it has
14
+ no side effects (``ttnn`` is imported when a device is opened).
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import contextlib
19
+ import dataclasses
20
+ from typing import Any, Dict, Iterator, Optional
21
+
22
+ from .ttaw.device import DeviceConfig, close_device, compute_grid, core_grid, describe_device, full_core_range_set
23
+
24
+ __all__ = ["ENV_PREFIX", "DEVICE_DEFAULTS", "DeviceConfig", "device_config", "open_device", "device_session",
25
+ "close_device", "describe_device", "compute_grid", "core_grid", "full_core_range_set"]
26
+
27
+ ENV_PREFIX = "METEOR"
28
+
29
+ # Validated device-open parameters of this port (fill per model; the card's numbers are measured with them).
30
+ DEVICE_DEFAULTS: Dict[str, Any] = {
31
+ # ETH dispatch (12x10) needs BOTH local tt-metal patches (patches/): the ETH-dispatch patch, and the reshape patch
32
+ # (Blackhole SYS-1419: the dual-kernel ROW_MAJOR reshape into small DRAM pages hangs this model under ETH dispatch
33
+ # without it; PORT_LOG.md section 12). ttaw 0.23.1's Resize2d TILE tail protects METEOR on its own as well.
34
+ # METEOR_DISPATCH=worker (11x10) is the A/B opt-in.
35
+ "num_command_queues": 1, # 1, or 2 when the input upload (CQ1) overlaps the trace (CQ0)
36
+ "l1_small_size": 32768, # L1_SMALL bytes per core (conv / pool config tensors)
37
+ # DRAM bytes for the traces of all captured variants: the whole graph is one trace of ~2,000+ programs (many
38
+ # DRAM-sliced 800x500 convs); the device tests also capture four debug variants. Measured in PORT_LOG.md.
39
+ "trace_region_size": 256 << 20,
40
+ }
41
+
42
+
43
+ def device_config(**overrides: Any) -> DeviceConfig:
44
+ """:data:`DEVICE_DEFAULTS` < the ``METEOR_*`` / ``TT_DEVICE_ID`` environment < explicit non-None ``overrides``
45
+ (``device_id``, ``dispatch``, ``num_command_queues``, ``l1_small_size``, ``trace_region_size``,
46
+ ``worker_l1_size``, ``allow_fallback``): the same resolution as ``METEOR.from_pretrained``."""
47
+ config = DeviceConfig.from_env(ENV_PREFIX, **DEVICE_DEFAULTS)
48
+ return dataclasses.replace(config, **{k: v for k, v in overrides.items() if v is not None})
49
+
50
+
51
+ def open_device(device_id: Optional[int] = None, *, dispatch: Optional[str] = None, **overrides: Any):
52
+ """Open one chip like the published numbers: ETH dispatch (``dispatch="worker"`` is the A/B opt-in) and this
53
+ port's sizes. Close it with :func:`close_device`, or use :func:`device_session`."""
54
+ return device_config(device_id=device_id, dispatch=dispatch, **overrides).open()
55
+
56
+
57
+ @contextlib.contextmanager
58
+ def device_session(device_id: Optional[int] = None, **overrides: Any) -> Iterator[Any]:
59
+ """``with device_session() as dev:`` opens with :func:`open_device` and always closes, also when the body
60
+ raises."""
61
+ device = open_device(device_id, **overrides)
62
+ try:
63
+ yield device
64
+ finally:
65
+ close_device(device)
code/tt_meteor/host/__init__.py ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Host pre- and post-processing of METEOR's own runtimes (no Autoware package exists; tier4/METEOR @ dc193a81e995
3
+ ``hf/onnx_smoke_test.py``, ``deploy/cpp/realtime_main.cpp``, ``decode.cpp``, ``render.cpp``; research/meteor/SPEC.md
4
+ sections 2, 3 and 5), shared by the Python API, the HTTP server and the CPU reference. numpy (+ the vendored ``ttaw``;
5
+ OpenCV only for the rotated-rectangle intersection and the seg-fusion warp, as the C++).
6
+
7
+ - ``preprocess.py`` frames -> the graph feed (INTER_AREA via ``ttaw.image_area``, K per-axis scaling, T inversion,
8
+ absent-camera donors); METEOR's demo-scene reader.
9
+ - ``inputs.py`` API inputs (cameras + calibration / presets + ego speed + stream) -> ``MeteorFrame``.
10
+ - ``calib.py`` the lift geometry of a calibration (grid, validity, depth bins, gather tables): RT-dev tables.
11
+ - ``postprocess.py`` 3D boxes + rotated NMS, stationary flags, agent futures, 2D boxes, unknown obstacles, plan,
12
+ road-edge thinning (stateless, ``PostConfig`` = the RT-host knobs).
13
+ - ``temporal.py`` per-stream state: BEV seg fusion, yaw smoothing, mode hysteresis.
14
+ - ``outputs.py`` ``MeteorOutput`` (the ``/predict`` JSON).
15
+ - ``result.py`` outputs of the network + frame (+ stream state) -> ``MeteorOutput``.
16
+ """
code/tt_meteor/host/calib.py ADDED
@@ -0,0 +1,199 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Lift geometry of the deployed 0.4 m lift from the calibration (host side; RT-dev tables of the TT port).
3
+
4
+ K and T_cam_ego are consumed by the graph ONLY through the lift (one Reshape each, ONNX nodes 265 / 280; SPEC
5
+ section 9), so everything the lift needs besides the image features is a function of the calibration and can be
6
+ computed once per rig and uploaded as persistent device tensors (PLAN.md 2.13 "at calibration: lift geometry";
7
+ RT-dev). This module computes it exactly as the graph does (float32, the graph's operation order; ONNX nodes 265-390;
8
+ SPEC section 4.3), for the 8 cameras x 100,000 ground points of the 400 x 250 grid:
9
+
10
+ - ``grid`` [8, 100000, 2]: the normalised sampling position (u / 767 * 2 - 1, v / 431 * 2 - 1, clipped to +-2) on the
11
+ 108 x 192 feature maps (align_corners=False, zero padding);
12
+ - ``valid`` [8, 100000]: z > 0.5, 0 <= u < 768, 0 <= v < 432, range < 90 m;
13
+ - ``b0`` / ``fr`` [8, 100000]: the LINEAR depth bin of the range, b = clip((d - 1) / 1.25, 0, 62.9999), b0 = floor(b),
14
+ fr = b - b0 (the lift interpolates prob[b0] and prob[b0 + 1]);
15
+ - the pair statistics (pairs per camera, cameras per cell, k_max) that size the gather-form tables of a later lift
16
+ kernel: :func:`ell_tables` packs the valid pairs per cell into ``K_SLOTS`` slots (k_max is 3 on every rig seen so far;
17
+ SPEC section 9 recommends 4 with a host assert).
18
+
19
+ The projection ``T @ [x, y, 0, 1]`` is evaluated as a float32 fused-multiply-add chain (``x T0 -> fma(y, T1) ->
20
+ fma(0, T2) -> fma(1, T3)``), which reproduces ONNX Runtime's MatMul (and torch's) bit for bit on this host; the rest is
21
+ elementwise float32 in graph order. ``tests/test_host_calib.py`` checks grid / valid against the ORT taps exactly.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import hashlib
26
+ import threading
27
+ from collections import OrderedDict
28
+ from dataclasses import dataclass
29
+ from typing import Any, Dict, Optional
30
+
31
+ import numpy as np
32
+
33
+ from ..reference import config as C
34
+
35
+ __all__ = ["LiftGeometry", "lift_geometry", "lift_ground_points", "ell_tables", "EllTables", "K_SLOTS",
36
+ "CalibrationCache", "calibration_key"]
37
+
38
+ F32 = np.float32
39
+ K_SLOTS = 4 # cameras per BEV lift cell a table slot set can hold (k_max = 3 measured on all rigs; SPEC 9)
40
+
41
+
42
+ def lift_ground_points() -> np.ndarray:
43
+ """The [4, 100000] homogeneous ground points of the 400 x 250 lift grid: x = linspace(79.8, -79.8, 400) per row,
44
+ y = linspace(49.8, -49.8, 250) per column, z = 0, w = 1 (V52.__init__, model.py:5269-5282). The graph's constant
45
+ ``onnx::Expand_1643`` holds torch's float32 ``linspace`` (its two half-ranges are computed from either end), which
46
+ numpy's ``linspace`` misses by up to 7.6e-6 at 132,900 entries, so the values come from torch here (checked equal
47
+ to the constant in ``tests/test_host_calib.py``); the TT port passes the constant read from the ONNX instead."""
48
+ import torch
49
+
50
+ xs = torch.linspace(79.8, -79.8, C.LIFT_H, dtype=torch.float32).numpy()
51
+ ys = torch.linspace(49.8, -49.8, C.LIFT_W, dtype=torch.float32).numpy()
52
+ gx, gy = np.meshgrid(xs, ys, indexing="ij")
53
+ n = C.LIFT_H * C.LIFT_W
54
+ return np.stack([gx.ravel(), gy.ravel(), np.zeros(n, F32), np.ones(n, F32)], 0)
55
+
56
+
57
+ def _fma32(a: np.ndarray, b: np.ndarray, c: np.ndarray) -> np.ndarray:
58
+ """float32 fused multiply-add (the float32 product is exact in float64; one rounding of the sum)."""
59
+ return (a.astype(np.float64) * b.astype(np.float64) + c.astype(np.float64)).astype(F32)
60
+
61
+
62
+ @dataclass(frozen=True)
63
+ class LiftGeometry:
64
+ """Per (camera, ground point) lift geometry of one calibration (see the module docstring)."""
65
+
66
+ grid: np.ndarray # [8, N, 2] float32
67
+ valid: np.ndarray # [8, N] bool
68
+ b0: np.ndarray # [8, N] int64 in [0, 62]
69
+ fr: np.ndarray # [8, N] float32 in [0, 1)
70
+ dist: np.ndarray # [8, N] float32 range from the camera (metres)
71
+ key: str = "" # calibration hash
72
+
73
+ @property
74
+ def pairs_per_camera(self) -> np.ndarray:
75
+ return self.valid.sum(axis=1)
76
+
77
+ @property
78
+ def cameras_per_cell(self) -> np.ndarray:
79
+ return self.valid.sum(axis=0)
80
+
81
+ def stats(self) -> Dict[str, Any]:
82
+ k = self.cameras_per_cell
83
+ return {"pairs_total": int(self.valid.sum()),
84
+ "pairs_per_camera": dict(zip(C.CAMERAS, self.pairs_per_camera.tolist())),
85
+ "cells_per_k": np.bincount(k, minlength=C.N_CAMS + 1).tolist(), "k_max": int(k.max()),
86
+ "cells_covered": int((k > 0).sum()),
87
+ "b0_range": ([int(self.b0[self.valid].min()), int(self.b0[self.valid].max())]
88
+ if self.valid.any() else [])}
89
+
90
+ def bin_weights(self) -> np.ndarray:
91
+ """The dense [8, N, 64] float32 bin-interpolation table of the functional lift (PLAN 2.13 "a bin-interpolation
92
+ weight table for the fallback"): (1 - fr) at b0, fr at b0 + 1, zero elsewhere and for invalid pairs, so that
93
+ w = valid * (0.05 + sum_b prob_s[b] * table[b]). (b0 + 1 <= 63 always: the bin index is clipped to 62.9999.)"""
94
+ n, p = self.b0.shape
95
+ t = np.zeros((n, p, C.DEPTH_BINS), F32)
96
+ ii, jj = np.nonzero(self.valid)
97
+ t[ii, jj, self.b0[ii, jj]] = F32(1.0) - self.fr[ii, jj]
98
+ t[ii, jj, self.b0[ii, jj] + 1] = self.fr[ii, jj]
99
+ return t
100
+
101
+
102
+ def calibration_key(K: Any, T_cam_ego: Any) -> str:
103
+ h = hashlib.sha256()
104
+ h.update(np.ascontiguousarray(np.asarray(K, F32).reshape(C.N_CAMS, 3, 3)).tobytes())
105
+ h.update(np.ascontiguousarray(np.asarray(T_cam_ego, F32).reshape(C.N_CAMS, 4, 4)).tobytes())
106
+ return h.hexdigest()[:16]
107
+
108
+
109
+ def lift_geometry(K: Any, T_cam_ego: Any, points: Optional[np.ndarray] = None) -> LiftGeometry:
110
+ """The graph's lift geometry for ``K`` [(1,) 8, 3, 3] (at 768x432) and ``T_cam_ego`` [(1,) 8, 4, 4] (ego ->
111
+ camera); ``points`` overrides the ground points (the TT port passes the ONNX constant)."""
112
+ Km = np.asarray(K, F32).reshape(C.N_CAMS, 3, 3)
113
+ T = np.asarray(T_cam_ego, F32).reshape(C.N_CAMS, 4, 4)
114
+ P = lift_ground_points() if points is None else np.asarray(points, F32).reshape(4, -1)
115
+ pc = []
116
+ for r in range(3): # rows x, y, z of T @ [px, py, 0, 1] (the 4th is unused)
117
+ acc = T[:, r, 0, None] * P[0][None]
118
+ for c in range(1, 4):
119
+ acc = _fma32(T[:, r, c, None], P[c][None], acc)
120
+ pc.append(acc)
121
+ x, y, z = pc
122
+ zc = np.maximum(z, F32(C.LIFT_MIN_Z))
123
+ u = (Km[:, 0, 0, None] * x) / zc + Km[:, 0, 2, None]
124
+ v = (Km[:, 1, 1, None] * y) / zc + Km[:, 1, 2, None]
125
+ dist = np.sqrt((x * x + y * y) + z * z)
126
+ valid = ((z > F32(C.LIFT_MIN_Z)) & (u >= F32(0)) & (u < F32(C.IMG_W)) & (v >= F32(0)) & (v < F32(C.IMG_H))
127
+ & (dist < F32(C.LIFT_MAX_DIST)))
128
+ gu = np.clip((u / F32(C.IMG_W - 1)) * F32(2) - F32(1), F32(-2), F32(2))
129
+ gv = np.clip((v / F32(C.IMG_H - 1)) * F32(2) - F32(1), F32(-2), F32(2))
130
+ b = np.clip((dist - F32(C.DEPTH_MIN)) / F32(C.DEPTH_STEP), F32(0), F32(C.LIFT_BIN_CLIP))
131
+ b0 = np.floor(b).astype(np.int64)
132
+ fr = (b - b0.astype(F32)).astype(F32)
133
+ return LiftGeometry(np.stack([gu, gv], axis=-1), valid, b0, fr, dist.astype(F32), calibration_key(Km, T))
134
+
135
+
136
+ @dataclass(frozen=True)
137
+ class EllTables:
138
+ """Gather-form lift tables, ``K_SLOTS`` slots per lift cell (unused slots have ``valid`` 0): camera, normalised
139
+ grid position, depth bin and fraction of every valid (camera, cell) pair, in camera order."""
140
+
141
+ cam: np.ndarray # [N, K_SLOTS] int32
142
+ grid: np.ndarray # [N, K_SLOTS, 2] float32
143
+ b0: np.ndarray # [N, K_SLOTS] int32
144
+ fr: np.ndarray # [N, K_SLOTS] float32
145
+ valid: np.ndarray # [N, K_SLOTS] bool
146
+ k_max: int
147
+
148
+
149
+ def ell_tables(geom: LiftGeometry, k_slots: int = K_SLOTS) -> EllTables:
150
+ """Pack the valid pairs of every cell into ``k_slots`` slots; raises if a cell is seen by more cameras (the
151
+ static table shape would overflow: SPEC section 9 K_SLOTS)."""
152
+ valid = geom.valid # [8, N]
153
+ k = valid.sum(axis=0)
154
+ if int(k.max()) > k_slots:
155
+ raise ValueError(f"{int((k > k_slots).sum())} lift cells are seen by more than K_SLOTS={k_slots} cameras "
156
+ f"(k_max {int(k.max())}): this rig needs larger tables")
157
+ n = valid.shape[1]
158
+ slot = np.cumsum(valid, axis=0) - 1 # slot of each valid pair in its cell, cameras in order
159
+ cam = np.zeros((n, k_slots), np.int32)
160
+ grid = np.zeros((n, k_slots, 2), F32)
161
+ b0 = np.zeros((n, k_slots), np.int32)
162
+ fr = np.zeros((n, k_slots), F32)
163
+ ok = np.zeros((n, k_slots), bool)
164
+ cc, pp = np.nonzero(valid)
165
+ ss = slot[cc, pp]
166
+ cam[pp, ss] = cc
167
+ grid[pp, ss] = geom.grid[cc, pp]
168
+ b0[pp, ss] = geom.b0[cc, pp]
169
+ fr[pp, ss] = geom.fr[cc, pp]
170
+ ok[pp, ss] = True
171
+ return EllTables(cam, grid, b0, fr, ok, int(k.max()))
172
+
173
+
174
+ class CalibrationCache:
175
+ """A small thread-safe LRU of lift geometries keyed by the calibration hash (a rig keeps its calibration, so the
176
+ tables are built once per rig; the demo data has 2 rigs, the public samples 2 more). ``points``: the ground points
177
+ (the TT port passes the graph's constant, ``MeteorWeights.lift_ground_points``; default: torch's linspace)."""
178
+
179
+ def __init__(self, maxsize: int = 8, points: Optional[np.ndarray] = None):
180
+ self.points = None if points is None else np.asarray(points, F32).reshape(4, -1)
181
+ self.maxsize = int(maxsize)
182
+ self._items: "OrderedDict[str, LiftGeometry]" = OrderedDict()
183
+ self._lock = threading.Lock()
184
+ self.hits = self.misses = 0
185
+
186
+ def get(self, K: Any, T_cam_ego: Any) -> LiftGeometry:
187
+ key = calibration_key(K, T_cam_ego)
188
+ with self._lock:
189
+ if key in self._items:
190
+ self._items.move_to_end(key)
191
+ self.hits += 1
192
+ return self._items[key]
193
+ geom = lift_geometry(K, T_cam_ego, points=self.points)
194
+ with self._lock:
195
+ self.misses += 1
196
+ self._items[key] = geom
197
+ while len(self._items) > self.maxsize:
198
+ self._items.popitem(last=False)
199
+ return geom
code/tt_meteor/host/inputs.py ADDED
@@ -0,0 +1,160 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Input handling shared by the Python API, the HTTP server and the CPU reference: up to eight cameras + calibration +
3
+ ego speed + stream -> one :class:`~.preprocess.MeteorFrame`.
4
+
5
+ Accepted ``images`` (Python API): a list of ``ttaw.io.CameraImage`` (what the server decodes ``/predict`` into), a list
6
+ of dicts ``{"camera": name, "image": path | bytes | array | PIL, "intrinsics": ..., "T_ref_from_camera": ...}``, or a
7
+ mapping ``{camera name: image}``. Calibration not given inline comes from ``calibration["cameras"][name]``
8
+ (``intrinsics`` of the image AS SENT, ``T_ref_from_camera`` = camera optical frame -> base_link (METEOR's ego frame:
9
+ x forward, y left, z up, origin on the ground), any ``ttaw.io.parse_transform`` spelling) or from a preset
10
+ (``{"preset": name}`` -> ``calib/<name>.json``). The six wide / corner cameras are required; a missing
11
+ ``CAM_FRONT_NARROW`` / ``CAM_BACK_NARROW`` is an absent camera (zero image + its donor's K and pose, as trained:
12
+ ``config.DONOR``). ``ego_speed`` (m/s) is the graph's ``v0``. ``stream`` = ``{"id", "reset", "timestamp_s",
13
+ "T_world_from_ego"}``: the pose drives the host temporal post-processing (BEV seg fusion, yaw tracks).
14
+
15
+ :func:`load_sample` reads the bundle's sample manifests (``samples/<name>.json``: image paths relative to the file, a
16
+ calibration preset, the ego speed and an optional stream) into the keyword arguments of ``model(...)``. A preset is
17
+ looked up in the ``calib/`` directory beside the manifest's ``samples/`` directory first (a staged sample carries its
18
+ own preset), then in the package's ``calib/``; it is returned inline.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ from pathlib import Path
24
+ from typing import Any, Dict, Mapping, Optional, Tuple
25
+
26
+ import numpy as np
27
+
28
+ from ..reference import config as C
29
+ from ..ttaw import io as tio
30
+ from .preprocess import MeteorFrame, assemble_frame, pose_from_transform
31
+
32
+ __all__ = ["load_cameras", "prepare_request", "load_sample", "SAMPLES_DIR", "CALIB_DIR", "stream_pose"]
33
+
34
+ PKG_DIR = Path(__file__).resolve().parents[1]
35
+ SAMPLES_DIR = PKG_DIR / "samples"
36
+ CALIB_DIR = PKG_DIR / "calib"
37
+ REQUIRED = tuple(c for c in C.CAMERAS if c not in C.DONOR)
38
+
39
+
40
+ def _camera_from(entry: Any, name: Optional[str], cams_cal: Mapping[str, Any]) -> "tio.CameraImage":
41
+ if isinstance(entry, tio.CameraImage):
42
+ cam = entry
43
+ elif isinstance(entry, Mapping):
44
+ if "camera" not in entry and name is None:
45
+ raise tio.InputError("every image needs a 'camera' name")
46
+ name = str(entry.get("camera", name))
47
+ if "data" in entry: # the /predict spelling (base64)
48
+ return tio.decode_cameras([dict(entry, camera=name)], {"cameras": dict(cams_cal)}, order=None,
49
+ require_calibration=True)[0]
50
+ src = entry.get("image", entry.get("path"))
51
+ if src is None:
52
+ raise tio.InputError(f"camera {name!r}: no 'image' / 'path' / 'data'")
53
+ cam = tio.CameraImage(name=name, image=tio.load_image(src), intrinsics=entry.get("intrinsics"),
54
+ T_ref_from_camera=entry.get("T_ref_from_camera", entry.get("extrinsics")),
55
+ timestamp_s=entry.get("timestamp_s"))
56
+ else:
57
+ if name is None:
58
+ raise tio.InputError("an image without a camera name: pass dicts with 'camera' or a {name: image} map")
59
+ cam = tio.CameraImage(name=str(name), image=tio.load_image(entry))
60
+ cal = dict(cams_cal.get(cam.name) or {})
61
+ k = cam.intrinsics if cam.intrinsics is not None else cal.get("intrinsics")
62
+ t = cam.T_ref_from_camera if cam.T_ref_from_camera is not None else cal.get(
63
+ "T_ref_from_camera", cal.get("extrinsics"))
64
+ if k is None or t is None:
65
+ raise tio.InputError(f"camera {cam.name!r} needs 'intrinsics' and 'T_ref_from_camera' (inline or in "
66
+ "calibration.cameras)")
67
+ return tio.CameraImage(name=cam.name, image=cam.image,
68
+ intrinsics=tio.parse_intrinsics(k, field=f"{cam.name}.intrinsics"),
69
+ T_ref_from_camera=tio.parse_transform(t, field=f"{cam.name}.T_ref_from_camera"),
70
+ distortion=cam.distortion, timestamp_s=cam.timestamp_s)
71
+
72
+
73
+ def resolve_calibration(calibration: Optional[Mapping[str, Any]]) -> Dict[str, Any]:
74
+ calibration = dict(calibration or {})
75
+ if "preset" in calibration:
76
+ calibration = tio.resolve_calibration(calibration, CALIB_DIR)
77
+ return calibration
78
+
79
+
80
+ def load_cameras(images: Any, calibration: Optional[Mapping[str, Any]]) -> Dict[str, "tio.CameraImage"]:
81
+ """``images`` (+ ``calibration``) -> {camera name: CameraImage with K and T_ref_from_camera} (the present ones)."""
82
+ if images is None:
83
+ raise tio.InputError(f"this model needs 'images': the cameras {list(C.CAMERAS)}")
84
+ cams_cal = dict(resolve_calibration(calibration).get("cameras") or {})
85
+ if isinstance(images, Mapping):
86
+ cams = [_camera_from(v, k, cams_cal) for k, v in images.items()]
87
+ else:
88
+ cams = [_camera_from(e, None, cams_cal) for e in images]
89
+ by_name: Dict[str, Any] = {}
90
+ for c in cams:
91
+ if c.name in by_name:
92
+ raise tio.InputError(f"camera {c.name!r} given twice")
93
+ by_name[c.name] = c
94
+ missing = [n for n in REQUIRED if n not in by_name]
95
+ extra = [n for n in by_name if n not in C.CAMERAS]
96
+ if missing or extra:
97
+ raise tio.InputError(f"cameras must be {list(C.CAMERAS)} (the narrow ones may be absent); missing {missing}, "
98
+ f"unexpected {extra}")
99
+ for c in by_name.values():
100
+ a = np.asarray(c.image)
101
+ if a.ndim != 3 or a.shape[2] != 3 or a.dtype != np.uint8:
102
+ raise tio.InputError(f"camera {c.name!r}: expected an (H, W, 3) uint8 image, got {a.shape} {a.dtype}")
103
+ if a.shape[0] < C.IMG_H or a.shape[1] < C.IMG_W:
104
+ raise tio.InputError(f"camera {c.name!r}: {a.shape[1]}x{a.shape[0]} is smaller than 768x432")
105
+ return by_name
106
+
107
+
108
+ def stream_pose(stream: Optional[Mapping[str, Any]]) -> Optional[Tuple[float, float, float]]:
109
+ """(x, y, yaw) from ``stream["T_world_from_ego"]`` (or ``stream["pose"]`` = [x, y, yaw]); None without one."""
110
+ if not stream:
111
+ return None
112
+ if stream.get("pose") is not None:
113
+ p = [float(v) for v in stream["pose"]]
114
+ if len(p) != 3:
115
+ raise tio.InputError("stream.pose must be [x, y, yaw]")
116
+ return p[0], p[1], p[2]
117
+ t = stream.get("T_world_from_ego")
118
+ if t is None:
119
+ return None
120
+ return pose_from_transform(tio.parse_transform(t, field="stream.T_world_from_ego"))
121
+
122
+
123
+ def prepare_request(images: Any, calibration: Optional[Mapping[str, Any]], ego_speed: Any,
124
+ stream: Optional[Mapping[str, Any]] = None, *, inverse: str = "float32") -> MeteorFrame:
125
+ """The API inputs -> the network input of one frame (``MeteorFrame``), with the ego pose of ``stream``."""
126
+ if ego_speed is None:
127
+ raise tio.InputError("this model needs 'ego_speed' (m/s, the graph's v0)")
128
+ try:
129
+ v0 = float(ego_speed)
130
+ except (TypeError, ValueError):
131
+ raise tio.InputError(f"ego_speed must be a number (m/s), got {ego_speed!r}") from None
132
+ if not np.isfinite(v0):
133
+ raise tio.InputError("ego_speed must be finite")
134
+ cams = load_cameras(images, calibration)
135
+ try:
136
+ frame = assemble_frame({n: c.image for n, c in cams.items()}, {n: c.intrinsics for n, c in cams.items()},
137
+ {n: c.T_ref_from_camera for n, c in cams.items()}, v0, inverse=inverse,
138
+ pose=stream_pose(stream))
139
+ except ValueError as e:
140
+ raise tio.InputError(str(e)) from None
141
+ frame.meta["cameras"] = sorted(cams)
142
+ return frame
143
+
144
+
145
+ def load_sample(path: Any, *, with_stream: bool = True) -> Dict[str, Any]:
146
+ """A sample manifest ``{"images": {camera: relative path}, "calibration": {...} | {"preset": name},
147
+ "ego_speed": m/s, "stream": {...}}`` -> the keyword arguments of ``model(...)``."""
148
+ p = Path(path)
149
+ spec = json.loads(p.read_text())
150
+ calib = dict(spec.get("calibration") or {})
151
+ beside = p.parent.parent / "calib"
152
+ if "preset" in calib and beside.resolve() != CALIB_DIR.resolve() and (beside / f"{calib['preset']}.json").is_file():
153
+ calib = tio.resolve_calibration(calib, beside)
154
+ calib = resolve_calibration(calib)
155
+ images = [{"camera": name, "image": str((p.parent / rel).resolve())} for name, rel in spec["images"].items()]
156
+ cams = [_camera_from(e, None, dict(calib.get("cameras") or {})) for e in images]
157
+ out: Dict[str, Any] = {"images": cams, "calibration": calib, "ego_speed": float(spec["ego_speed"])}
158
+ if with_stream and spec.get("stream"):
159
+ out["stream"] = dict(spec["stream"])
160
+ return out
code/tt_meteor/host/outputs.py ADDED
@@ -0,0 +1,111 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """``MeteorOutput``: the multi-task result of one frame, what ``model(...)`` returns and ``/predict`` serialises.
3
+
4
+ BUNDLE_CONVENTIONS.md section 7.4 leaves METEOR's payload "per head"; this port's JSON (``to_dict()``):
5
+
6
+ ========================== ==================================================================================
7
+ key content
8
+ ========================== ==================================================================================
9
+ ``detections`` 3D BEV boxes sorted by score: ``label`` (VEHICLE / VRU), ``label_id``, ``score``,
10
+ ``center`` [x, y] and ``size`` [length, width] in metres (ego frame, x forward,
11
+ y left; METEOR predicts no z, height or velocity), ``yaw`` (rad, CCW from +x),
12
+ ``stationary``, ``future`` (6 x [x, y] at 0.5 .. 3 s, the best agent mode) and
13
+ ``future_mode``
14
+ ``trajectory``, ``columns`` the selected ego path, 6 x [x, y] at 0.5 s steps (``ttaw.server.smoke`` compares it)
15
+ ``plan`` ``mode``, ``mode_probs`` (softmax of the hysteresis-adjusted logits), ``mode_logits``,
16
+ ``paths`` (3 x 6 x [x, y]), ``steer`` (rad), ``accel`` (m/s^2), ``brake_prob``
17
+ ``detections_2d`` per camera: ``label``, ``label_id``, ``score``, ``box_xyxy`` (pixels at 768x432)
18
+ ``unknown_obstacles`` [x, y] of 2D "obs" boxes placed on the ground plane (``render.cpp`` unk2d)
19
+ ``traffic_light`` ``state`` (none / green / yellow / red), ``probs``
20
+ ``lane`` the BEV lane map, uint8 [800, 500] (PNG; 0.2 m cells, row 0 = +80 m, col 0 = +50 m),
21
+ after the optional seg fusion and road-edge thinning; ``lane_classes`` names its values
22
+ ``heads`` (npz only) with ``output_format="npz"``: the dense outputs ``seg2d``, ``depth`` (bins),
23
+ ``depth_mean``, ``occupancy`` (class per voxel), ``risk`` (sigmoid), ``stationary``
24
+ ========================== ==================================================================================
25
+ """
26
+ from __future__ import annotations
27
+
28
+ from dataclasses import dataclass, field
29
+ from typing import Any, Dict, List
30
+
31
+ import numpy as np
32
+
33
+ from ..reference import config as C
34
+ from ..ttaw.io import encode_array, encode_png, to_jsonable
35
+ from .postprocess import Box2D, Box3D
36
+
37
+ __all__ = ["MeteorOutput", "LABELS"]
38
+
39
+ LABELS = ("VEHICLE", "VRU")
40
+
41
+
42
+ @dataclass
43
+ class MeteorOutput:
44
+ boxes3d: List[Box3D]
45
+ boxes2d: List[List[Box2D]]
46
+ unknown: List[Dict[str, Any]]
47
+ plan: Dict[str, Any]
48
+ traffic_light: Dict[str, Any]
49
+ lane: np.ndarray
50
+ stationary_healthy: bool = True
51
+ heads: Dict[str, np.ndarray] = field(default_factory=dict)
52
+ model: str = "meteor-p150"
53
+ frame_id: str = "base_link"
54
+ timing_ms: Dict[str, float] = field(default_factory=dict)
55
+ meta: Dict[str, Any] = field(default_factory=dict)
56
+
57
+ def __post_init__(self) -> None:
58
+ self.boxes3d = sorted(self.boxes3d, key=lambda b: -b.score)
59
+
60
+ # ---- views ----------------------------------------------------------------------------------------------------
61
+ def to_dicts(self) -> List[Dict[str, Any]]:
62
+ """The 3D detection list (``detections`` of :meth:`to_dict`)."""
63
+ out = []
64
+ for b in self.boxes3d:
65
+ d = {"label": LABELS[b.cls], "label_id": int(b.cls), "score": float(b.score),
66
+ "center": [float(b.x), float(b.y)], "size": [float(b.l), float(b.w)], "yaw": float(b.yaw),
67
+ "stationary": bool(b.stationary), "future_mode": int(b.future_mode)}
68
+ if b.future is not None:
69
+ d["future"] = np.asarray(b.future, np.float64).tolist()
70
+ out.append(d)
71
+ return out
72
+
73
+ def detections_2d(self) -> Dict[str, List[Dict[str, Any]]]:
74
+ return {cam: [{"label": b.label, "label_id": int(b.cls), "score": float(b.score), "box_xyxy": b.xyxy()}
75
+ for b in boxes] for cam, boxes in zip(C.CAMERAS, self.boxes2d)}
76
+
77
+ @property
78
+ def path(self) -> np.ndarray:
79
+ return np.asarray(self.plan["path"], np.float64)
80
+
81
+ def to_dict(self, output_format: str = "json") -> Dict[str, Any]:
82
+ if output_format not in ("json", "npz"):
83
+ raise ValueError(f"output_format must be 'json' or 'npz', not {output_format!r}")
84
+ plan = {"mode": int(self.plan["mode"]), "mode_probs": np.asarray(self.plan["mode_probs"]).tolist(),
85
+ "mode_logits": np.asarray(self.plan["mode_logits"]).tolist(),
86
+ "paths": np.asarray(self.plan["paths"]).tolist(), "steer": float(self.plan["steer"]),
87
+ "accel": float(self.plan["accel"]), "brake_prob": float(self.plan["brake_prob"]),
88
+ "dt": float(self.plan.get("dt", C.EGO_DT))}
89
+ body: Dict[str, Any] = {
90
+ "model": self.model, "frame_id": self.frame_id, "num_detections": len(self.boxes3d),
91
+ "detections": self.to_dicts(),
92
+ "trajectory": self.path.tolist(), "columns": ["x", "y"],
93
+ "plan": plan,
94
+ "detections_2d": self.detections_2d(),
95
+ "unknown_obstacles": [{k: (float(v) if isinstance(v, (float, np.floating)) else v) for k, v in u.items()}
96
+ for u in self.unknown],
97
+ "traffic_light": {"state": self.traffic_light["state"], "state_id": int(self.traffic_light["state_id"]),
98
+ "probs": np.asarray(self.traffic_light["probs"]).tolist()},
99
+ "lane": encode_png(np.asarray(self.lane, np.uint8), key="lane"),
100
+ "lane_classes": list(C.LANE_CLASSES),
101
+ "stationary_head_healthy": bool(self.stationary_healthy),
102
+ "meta": to_jsonable(self.meta),
103
+ "timing_ms": {k: float(v) for k, v in self.timing_ms.items()},
104
+ }
105
+ if output_format == "npz":
106
+ body["heads"] = {k: encode_array(np.asarray(v), fmt="npz", key=k) for k, v in self.heads.items()}
107
+ return body
108
+
109
+ def __repr__(self) -> str:
110
+ return (f"MeteorOutput(boxes3d={len(self.boxes3d)}, boxes2d={sum(len(b) for b in self.boxes2d)}, "
111
+ f"mode={self.plan.get('mode')}, tl={self.traffic_light.get('state')!r})")
code/tt_meteor/host/postprocess.py ADDED
@@ -0,0 +1,462 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Host post-processing of METEOR's C++ runtime (tier4/METEOR @ dc193a81e995 ``deploy/cpp/decode.cpp`` + ``render.cpp``;
3
+ research/meteor/SPEC.md section 5). numpy (OpenCV for the rotated-rectangle intersection when installed).
4
+
5
+ Per frame, stateless (the temporal parts are in ``host.temporal``):
6
+
7
+ - 3D boxes (``decode.cpp:27-57`` via ``render.cpp:383-395``): sigmoid(hm) > 0.15 at 3x3 peaks (no strictly greater
8
+ neighbour), best 64 by score; x = 80 - (r + reg0) 0.4, y = 50 - (c + reg1) 0.4, l = e^reg2, w = e^reg3,
9
+ yaw = atan2(reg4, reg5) (float32 as the C++); vehicle > 0.35, VRU > 0.15, x in [-(80 - 2), 80 - 2]; greedy
10
+ same-class BEV NMS, suppressed if rotated IoU > 0.3 or intersection / smaller area > 0.6 (``render.cpp:283-319``).
11
+ - Stationary flag (``render.cpp:265-276, 590-610``): stationary logit at the box cell > 0.0 when the head is healthy
12
+ (std >= 0.05 and range >= 0.2 over the map), else the best traj mode's last waypoint |wp6| < 0.5 m.
13
+ - Agent futures (``render.cpp:631-655``): the traj mode argmax at the box cell, waypoints = centre + (dx, dy)_t at
14
+ t = 0.5 .. 3 s (the renderer draws vehicles only; every box carries its future here, ``kind`` tells them apart).
15
+ - 2D boxes per camera and stride (``decode.cpp:59-108``): logits clipped to +-50, sigmoid > 0.30 at 3x3 peaks (on the
16
+ clipped plane), best 48 per camera and scale; cx = (c + reg1) s, cy = (r + reg0) s, w = e^reg2 s, h = e^reg3 s;
17
+ class 7 (road paint) hidden (``METEOR_2D_HIDE``).
18
+ - Unknown obstacles (``render.cpp:657-729``): 2D class 0 boxes >= TH_UNK placed on the ground plane through the
19
+ bottom-centre ray (range 1.5-60 m), else back-projected with the median ``depth_mean`` of a 5x5 window (1.5-50 m);
20
+ |x| <= 60, |y| <= 25; greedy de-duplication within 1.5 m by score.
21
+ - Plan (``render.cpp:432-452``): mode logits + 0.35 on the previous mode, argmax, straight preference (mode 0 unless
22
+ the margin is >= 1.0); the selected path, softmax confidences of the (hysteresis-adjusted) logits, controls.
23
+ - BEV seg road-edge thinning (``render.cpp:252-262``), traffic light argmax, risk = sigmoid, occupancy argmax, metric
24
+ depth of a bin index (linear 1 + 1.25 b as trained, SPEC section 10 risk 2).
25
+
26
+ Ties: ``std::sort`` of the C++ is not stable; this module sorts candidates stably in (class, row, col) order, so equal
27
+ scores (saturated sigmoids) keep the scan order (documented policy, a set-based gate is insensitive to it).
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import math
32
+ from dataclasses import dataclass, field, replace
33
+ from typing import Any, Dict, List, Optional, Sequence, Tuple
34
+
35
+ import numpy as np
36
+
37
+ from ..reference import config as C
38
+
39
+ __all__ = ["PostConfig", "Box3D", "Box2D", "sigmoid32", "peak_mask", "decode_boxes_3d", "filter_boxes_3d",
40
+ "rotated_intersection", "bev_nms", "stationary_healthy", "annotate_boxes", "decode_boxes_2d",
41
+ "ground_point", "unknown_obstacles", "select_mode", "plan_from_ego", "thin_road_edge", "decode_frame",
42
+ "metric_depth", "tl_state", "occupancy_classes"]
43
+
44
+ F32 = np.float32
45
+
46
+
47
+ @dataclass(frozen=True)
48
+ class PostConfig:
49
+ """Every host knob of METEOR's runtime with its default (RT-host; ``render.cpp:345-376``, ``param.yaml:36-40``)."""
50
+
51
+ det3d_threshold: float = 0.15 # decode pre-threshold (render.cpp:388)
52
+ det3d_topk: int = 64
53
+ vehicle_threshold: float = 0.35 # render.cpp:391
54
+ vru_threshold: float = 0.15
55
+ range_margin: float = 2.0 # x in [-(XR - 2), XF - 2]
56
+ bev_nms_iou: float = 0.3 # METEOR_BEV_NMS_IOU
57
+ bev_nms_containment: float = 0.6 # render.cpp:316
58
+ stationary_logit_threshold: float = 0.0 # METEOR_STAT_LOGIT_THRESH
59
+ stationary_health_std: float = 0.05
60
+ stationary_health_range: float = 0.2
61
+ stationary_fallback_m: float = 0.5
62
+ det2d_threshold: float = 0.30 # METEOR_TH2D (C++ and param.yaml; Python orin_render 0.50)
63
+ det2d_topk: int = 48
64
+ det2d_clip: float = 50.0
65
+ det2d_hide: Tuple[int, ...] = (7,) # METEOR_2D_HIDE (road_paint)
66
+ unk2d: bool = True # METEOR_UNK2D
67
+ unk2d_threshold: Optional[float] = None # METEOR_UNK2D_TH (default: det2d_threshold)
68
+ ground_z: float = 0.0 # METEOR_GROUND_Z
69
+ unk2d_ground_range: Tuple[float, float] = (1.5, 60.0)
70
+ unk2d_depth_range: Tuple[float, float] = (1.5, 50.0)
71
+ unk2d_max_x: float = 60.0
72
+ unk2d_max_y: float = 25.0
73
+ unk2d_dedupe_m: float = 1.5
74
+ mode_hysteresis: float = 0.35 # render.cpp:436
75
+ straight_margin: float = 1.0 # render.cpp:441
76
+ thin_road_edge: bool = True # METEOR_NO_THIN = 0
77
+ seg_fuse: bool = True # METEOR_SEG_FUSE (host.temporal)
78
+ yaw_smoothing: bool = True # render.cpp:396-421 (host.temporal)
79
+ risk_gain: float = 1.0 # METEOR_RISK_GAIN (display only)
80
+
81
+ def with_params(self, **params: Any) -> "PostConfig":
82
+ return replace(self, **{k: v for k, v in params.items() if v is not None})
83
+
84
+ @property
85
+ def th_unk(self) -> float:
86
+ return self.det2d_threshold if self.unk2d_threshold is None else self.unk2d_threshold
87
+
88
+
89
+ @dataclass
90
+ class Box3D:
91
+ """One BEV box (ego frame, metres, radians CCW from +x); ``cell`` is the 400x250 detection-grid cell."""
92
+
93
+ cls: int
94
+ score: float
95
+ x: float
96
+ y: float
97
+ l: float
98
+ w: float
99
+ yaw: float
100
+ cell: Tuple[int, int]
101
+ stationary: bool = False
102
+ future_mode: int = -1
103
+ future: Optional[np.ndarray] = None # [6, 2] absolute waypoints (ego frame)
104
+
105
+ @property
106
+ def label(self) -> str:
107
+ return C.DET3D_CLASSES[self.cls]
108
+
109
+ def to_dict(self) -> Dict[str, Any]:
110
+ d = {"cls": self.cls, "label": self.label, "score": float(self.score), "x": float(self.x), "y": float(self.y),
111
+ "l": float(self.l), "w": float(self.w), "yaw": float(self.yaw), "cell": [int(v) for v in self.cell],
112
+ "stationary": bool(self.stationary), "future_mode": int(self.future_mode)}
113
+ if self.future is not None:
114
+ d["future"] = np.asarray(self.future, np.float64).round(4).tolist()
115
+ return d
116
+
117
+
118
+ @dataclass
119
+ class Box2D:
120
+ cls: int
121
+ score: float
122
+ cx: float
123
+ cy: float
124
+ w: float
125
+ h: float
126
+ stride: int
127
+
128
+ @property
129
+ def label(self) -> str:
130
+ return C.DET2D_CLASSES[self.cls]
131
+
132
+ def xyxy(self) -> List[float]:
133
+ return [self.cx - self.w / 2, self.cy - self.h / 2, self.cx + self.w / 2, self.cy + self.h / 2]
134
+
135
+ def to_dict(self) -> Dict[str, Any]:
136
+ return {"cls": self.cls, "label": self.label, "score": float(self.score), "cx": float(self.cx),
137
+ "cy": float(self.cy), "w": float(self.w), "h": float(self.h), "stride": int(self.stride)}
138
+
139
+
140
+ def sigmoid32(x: Any) -> np.ndarray:
141
+ """``1.0f / (1.0f + std::exp(-x))`` in float32 (decode.cpp ``sigmoidf``)."""
142
+ x = np.asarray(x, F32)
143
+ with np.errstate(over="ignore"):
144
+ return (F32(1.0) / (F32(1.0) + np.exp(-x))).astype(F32)
145
+
146
+
147
+ def peak_mask(plane: np.ndarray) -> np.ndarray:
148
+ """True where no in-bounds 3x3 neighbour is strictly greater (``decode.cpp`` isPeak)."""
149
+ p = np.pad(np.asarray(plane, F32), 1, constant_values=-np.inf)
150
+ h, w = plane.shape
151
+ m = np.full(plane.shape, -np.inf, F32)
152
+ for dr in range(3):
153
+ for dc in range(3):
154
+ m = np.maximum(m, p[dr:dr + h, dc:dc + w])
155
+ return plane >= m
156
+
157
+
158
+ def _candidates(planes: np.ndarray, thresh: float, topk: int, clip: Optional[float] = None):
159
+ """(score, cls, row, col) of the thresholded peaks of [C, H, W] logits, best ``topk`` (stable by scan order)."""
160
+ out = []
161
+ for c in range(planes.shape[0]):
162
+ v = np.asarray(planes[c], F32)
163
+ if clip is not None:
164
+ v = np.clip(v, F32(-clip), F32(clip))
165
+ p = sigmoid32(v)
166
+ rr, cc = np.nonzero((p > F32(thresh)) & peak_mask(v))
167
+ for r, q in zip(rr.tolist(), cc.tolist()):
168
+ out.append((float(p[r, q]), c, r, q))
169
+ out.sort(key=lambda z: -z[0])
170
+ return out[:topk]
171
+
172
+
173
+ def decode_boxes_3d(hm: np.ndarray, reg: np.ndarray, *, thresh: float = 0.15, topk: int = 64) -> List[Box3D]:
174
+ """``decode_boxes`` (decode.cpp:27-57) on hm [2, 400, 250] logits and reg [6, 400, 250]; float32 box math."""
175
+ hm = np.asarray(hm, F32).reshape(-1, C.DET_H, C.DET_W)
176
+ reg = np.asarray(reg, F32).reshape(6, C.DET_H, C.DET_W)
177
+ res = F32(C.DET_RES)
178
+ boxes = []
179
+ for sc, c, r, q in _candidates(hm, thresh, topk):
180
+ R = reg[:, r, q]
181
+ x = F32(80.0) - (F32(r) + R[0]) * res
182
+ y = F32(50.0) - (F32(q) + R[1]) * res
183
+ boxes.append(Box3D(c, sc, float(x), float(y), float(np.exp(R[2])), float(np.exp(R[3])),
184
+ float(np.arctan2(R[4], R[5])), (r, q)))
185
+ return boxes
186
+
187
+
188
+ def filter_boxes_3d(boxes: Sequence[Box3D], cfg: PostConfig = PostConfig()) -> List[Box3D]:
189
+ """Class thresholds (vehicle > 0.35, VRU > 0.15) and the grid-range clip (``render.cpp:388-394``)."""
190
+ xr = C.BEV_H * C.BEV_RES - C.BEV_XF
191
+ out = []
192
+ for b in boxes:
193
+ th = cfg.vehicle_threshold if b.cls == 0 else cfg.vru_threshold
194
+ if F32(b.score) > F32(th) and -(xr - cfg.range_margin) <= b.x <= C.BEV_XF - cfg.range_margin:
195
+ out.append(b)
196
+ return out
197
+
198
+
199
+ def _rect(b: Box3D):
200
+ return ((float(b.x), float(b.y)), (float(b.l), float(b.w)), float(F32(b.yaw * 180.0 / math.pi)))
201
+
202
+
203
+ def rotated_intersection(a: Box3D, b: Box3D) -> float:
204
+ """Area of the intersection of two rotated BEV rectangles: ``cv::rotatedRectangleIntersection`` + ``convexHull`` +
205
+ ``contourArea`` as ``render.cpp:283-307`` (OpenCV when installed, else exact convex clipping)."""
206
+ try:
207
+ import cv2
208
+ except ImportError: # pragma: no cover - the image and both dev envs have OpenCV
209
+ cv2 = None
210
+ if cv2 is not None:
211
+ r, pts = cv2.rotatedRectangleIntersection(_rect(a), _rect(b))
212
+ if r == cv2.INTERSECT_NONE or pts is None or len(pts) < 3:
213
+ return 0.0
214
+ return float(cv2.contourArea(cv2.convexHull(pts)))
215
+ from ..ttaw.nms import bev_box_corners, clip_convex, polygon_area
216
+ pa = bev_box_corners(np.array([a.x]), np.array([a.y]), np.array([a.l]), np.array([a.w]), np.array([a.yaw]))[0]
217
+ pb = bev_box_corners(np.array([b.x]), np.array([b.y]), np.array([b.l]), np.array([b.w]), np.array([b.yaw]))[0]
218
+ poly = clip_convex(pa, pb)
219
+ return float(polygon_area(poly)) if len(poly) >= 3 else 0.0
220
+
221
+
222
+ def bev_nms(boxes: Sequence[Box3D], iou_threshold: float = 0.3, containment: float = 0.6) -> List[Box3D]:
223
+ """``bevBoxNms`` (render.cpp:309-319): score order, greedy, same class only; a box is dropped if its rotated IoU
224
+ with a kept box exceeds ``iou_threshold`` or intersection / smaller area exceeds ``containment``."""
225
+ order = sorted(boxes, key=lambda b: -b.score)
226
+ kept: List[Box3D] = []
227
+ for d in order:
228
+ ok = True
229
+ for k in kept:
230
+ if k.cls != d.cls:
231
+ continue
232
+ inter = rotated_intersection(d, k)
233
+ if inter <= 0.0:
234
+ continue
235
+ ad, ak = float(d.l) * float(d.w), float(k.l) * float(k.w)
236
+ if inter / max(ad + ak - inter, 1e-6) > iou_threshold or inter / max(min(ad, ak), 1e-6) > containment:
237
+ ok = False
238
+ break
239
+ if ok:
240
+ kept.append(d)
241
+ return kept
242
+
243
+
244
+ def stationary_healthy(stat: np.ndarray, cfg: PostConfig = PostConfig()) -> bool:
245
+ """``stationaryHealthy`` (render.cpp:265-276): std >= 0.05 and max - min >= 0.2 over the finite values (double)."""
246
+ s = np.asarray(stat, np.float64).ravel()
247
+ s = s[np.isfinite(s)]
248
+ if s.size == 0:
249
+ return False
250
+ mean = s.sum() / s.size
251
+ var = (s * s).sum() / s.size - mean * mean
252
+ return bool(math.sqrt(max(var, 0.0)) >= cfg.stationary_health_std
253
+ and (s.max() - s.min()) >= cfg.stationary_health_range)
254
+
255
+
256
+ def _cell(b: Box3D) -> Tuple[int, int]:
257
+ return int((C.BEV_XF - b.x) / C.DET_RES), int((C.BEV_YH - b.y) / C.DET_RES)
258
+
259
+
260
+ def annotate_boxes(boxes: Sequence[Box3D], stationary: Optional[np.ndarray], traj: Optional[np.ndarray],
261
+ cfg: PostConfig = PostConfig()) -> bool:
262
+ """Stationary flag and agent future of every box, in place; returns whether the stationary head is healthy."""
263
+ stat = None if stationary is None else np.asarray(stationary, F32).reshape(C.DET_H, C.DET_W)
264
+ tr = None if traj is None else np.asarray(traj, F32).reshape(C.TRAJ_C, C.DET_H, C.DET_W)
265
+ ok = stat is not None and stationary_healthy(stat, cfg)
266
+ for b in boxes:
267
+ r0, c0 = _cell(b)
268
+ inside = 0 <= r0 < C.DET_H and 0 <= c0 < C.DET_W
269
+ fut = None
270
+ if tr is not None and inside:
271
+ v = tr[:, r0, c0]
272
+ kb = int(np.argmax(v[36:39])) # first max (strict > scan)
273
+ fut = v[kb * 12:kb * 12 + 12].reshape(6, 2)
274
+ b.future_mode = kb
275
+ b.future = fut.astype(np.float64) + np.array([b.x, b.y])
276
+ if ok and inside:
277
+ b.stationary = bool(stat[r0, c0] > F32(cfg.stationary_logit_threshold))
278
+ elif fut is not None:
279
+ b.stationary = bool(math.hypot(float(fut[5, 0]), float(fut[5, 1])) < cfg.stationary_fallback_m)
280
+ else:
281
+ b.stationary = False
282
+ return ok
283
+
284
+
285
+ def decode_boxes_2d(outputs: Dict[str, np.ndarray], cfg: PostConfig = PostConfig(), *,
286
+ keep_hidden: bool = False) -> List[List[Box2D]]:
287
+ """``decode_boxes2d_ms`` (decode.cpp:59-108) over the three strides -> 8 lists of boxes (pixels at 768x432);
288
+ hidden classes (``det2d_hide``) are removed unless ``keep_hidden``."""
289
+ res: List[List[Box2D]] = [[] for _ in range(C.N_CAMS)]
290
+ for si, s in enumerate(C.DET2D_STRIDES):
291
+ hm = np.asarray(outputs[f"hm2d_s{si}"], F32).reshape(C.N_CAMS, len(C.DET2D_CLASSES), -1)
292
+ h, w = C.FEAT_H * 4 // s, C.FEAT_W * 4 // s
293
+ hm = hm.reshape(C.N_CAMS, -1, h, w)
294
+ reg = np.asarray(outputs[f"reg2d_s{si}"], F32).reshape(C.N_CAMS, 4, h, w)
295
+ sf = F32(s)
296
+ for n in range(C.N_CAMS):
297
+ for sc, c, r, q in _candidates(hm[n], cfg.det2d_threshold, cfg.det2d_topk, clip=cfg.det2d_clip):
298
+ if c in cfg.det2d_hide and not keep_hidden:
299
+ continue
300
+ R = reg[n, :, r, q]
301
+ res[n].append(Box2D(c, sc, float((F32(q) + R[1]) * sf), float((F32(r) + R[0]) * sf),
302
+ float(np.exp(R[2]) * sf), float(np.exp(R[3]) * sf), s))
303
+ return res
304
+
305
+
306
+ def ground_point(u: float, v: float, Kc: np.ndarray, T: np.ndarray,
307
+ ground_z: float = 0.0) -> Optional[Tuple[float, float]]:
308
+ """``groundPoint`` (render.cpp:321-336): where the ray through pixel (u, v) meets the plane z = ground_z (ego
309
+ frame); ``T`` = T_cam_ego (ego -> camera). None if the ray points up or behind."""
310
+ Kc = np.asarray(Kc, np.float64).reshape(3, 3)
311
+ T = np.asarray(T, np.float64).reshape(4, 4)
312
+ dc = np.array([(u - Kc[0, 2]) / Kc[0, 0], (v - Kc[1, 2]) / Kc[1, 1], 1.0])
313
+ R, t = T[:3, :3], T[:3, 3]
314
+ c = -(R.T @ t)
315
+ d = R.T @ dc
316
+ if d[2] >= -1e-6:
317
+ return None
318
+ lam = (ground_z - c[2]) / d[2]
319
+ if lam <= 0:
320
+ return None
321
+ return float(c[0] + lam * d[0]), float(c[1] + lam * d[1])
322
+
323
+
324
+ def unknown_obstacles(boxes2d: Sequence[Sequence[Box2D]], depth_mean: Optional[np.ndarray], K: np.ndarray,
325
+ T_cam_ego: np.ndarray, cfg: PostConfig = PostConfig()) -> List[Dict[str, float]]:
326
+ """2D 'obs' boxes lifted to the BEV (``render.cpp:657-729``) -> [{x, y, score, range}] after de-duplication."""
327
+ K = np.asarray(K, np.float64).reshape(C.N_CAMS, 3, 3)
328
+ T = np.asarray(T_cam_ego, np.float64).reshape(C.N_CAMS, 4, 4)
329
+ dm = None if depth_mean is None else np.asarray(depth_mean, np.float64).reshape(C.N_CAMS, -1, C.FEAT_W // 2)
330
+ cand = []
331
+ for i in range(min(C.N_CAMS, len(boxes2d))):
332
+ for b in boxes2d[i]:
333
+ if b.cls != 0 or b.score < cfg.th_unk:
334
+ continue
335
+ gp = ground_point(b.cx, b.cy + 0.5 * b.h, K[i], T[i], cfg.ground_z)
336
+ lo, hi = cfg.unk2d_ground_range
337
+ if gp is not None and lo < math.hypot(*gp) < hi:
338
+ x, y, d = gp[0], gp[1], math.hypot(*gp)
339
+ else:
340
+ if dm is None:
341
+ continue
342
+ dmh, dmw = dm.shape[1], dm.shape[2]
343
+ u = min(max(int(b.cx / C.IMG_W * dmw), 0), dmw - 1)
344
+ v = min(max(int((b.cy + 0.35 * b.h) / C.IMG_H * dmh), 0), dmh - 1)
345
+ win = np.sort(dm[i, max(0, v - 2):min(dmh, v + 3), max(0, u - 2):min(dmw, u + 3)].ravel())
346
+ d = float(win[win.size // 2])
347
+ lo, hi = cfg.unk2d_depth_range
348
+ if not (lo < d < hi):
349
+ continue
350
+ pc = np.array([(b.cx - K[i, 0, 2]) / K[i, 0, 0] * d, (b.cy - K[i, 1, 2]) / K[i, 1, 1] * d, d])
351
+ pe = T[i, :3, :3].T @ (pc - T[i, :3, 3])
352
+ x, y = float(pe[0]), float(pe[1])
353
+ if abs(x) > cfg.unk2d_max_x or abs(y) > cfg.unk2d_max_y:
354
+ continue
355
+ cand.append({"x": x, "y": y, "score": float(b.score), "range": float(d), "camera": C.CAMERAS[i]})
356
+ cand.sort(key=lambda p: -p["score"])
357
+ kept: List[Dict[str, float]] = []
358
+ for p in cand:
359
+ if all((p["x"] - k["x"]) ** 2 + (p["y"] - k["y"]) ** 2 > cfg.unk2d_dedupe_m ** 2 for k in kept):
360
+ kept.append(p)
361
+ return kept
362
+
363
+
364
+ def select_mode(ego: np.ndarray, prev_mode: Optional[int] = None,
365
+ cfg: PostConfig = PostConfig()) -> Tuple[int, np.ndarray]:
366
+ """``render.cpp:432-445``: float32 mode logits + hysteresis on the previous mode, first argmax, straight preference.
367
+ Returns (mode, the adjusted logits)."""
368
+ lg = np.asarray(ego, F32).reshape(-1)[36:39].copy()
369
+ if prev_mode is not None and 0 <= prev_mode < C.EGO_MODES:
370
+ lg[prev_mode] = lg[prev_mode] + F32(cfg.mode_hysteresis)
371
+ k = 0
372
+ for i in range(1, C.EGO_MODES):
373
+ if lg[i] > lg[k]:
374
+ k = i
375
+ if k != 0 and (lg[k] - lg[0]) < F32(cfg.straight_margin):
376
+ k = 0
377
+ return k, lg
378
+
379
+
380
+ def plan_from_ego(ego: np.ndarray, prev_mode: Optional[int] = None, cfg: PostConfig = PostConfig()) -> Dict[str, Any]:
381
+ """The plan of ``render.cpp:432-452``: selected mode and path, the three paths, confidences (softmax of the
382
+ adjusted logits, double), controls (steer rad, accel m/s^2, brake logit and probability)."""
383
+ e = np.asarray(ego, F32).reshape(-1)
384
+ k, lg = select_mode(e, prev_mode, cfg)
385
+ mx = float(lg.max())
386
+ pr = np.exp(lg.astype(np.float64) - mx)
387
+ pr /= pr.sum()
388
+ paths = e[:36].reshape(C.EGO_MODES, C.EGO_STEPS, 2)
389
+ brake = float(e[41])
390
+ return {"mode": int(k), "path": paths[k].astype(np.float64), "paths": paths.astype(np.float64),
391
+ "mode_logits": e[36:39].astype(np.float64), "mode_probs": pr, "steer": float(e[39]),
392
+ "accel": float(e[40]), "brake_logit": brake, "brake_prob": 1.0 / (1.0 + math.exp(-brake)),
393
+ "dt": C.EGO_DT}
394
+
395
+
396
+ def thin_road_edge(lane: np.ndarray) -> np.ndarray:
397
+ """``thin_road_edge`` (render.cpp:252-262): keep only road-edge (6) pixels next to drivable (1, 3, 4, 5) pixels
398
+ that are not drivable themselves; every other edge pixel becomes background. Returns a new map."""
399
+ lane = np.asarray(lane, np.uint8)
400
+ edge = lane == 6
401
+ if not edge.any():
402
+ return lane.copy()
403
+ drv = np.isin(lane, (1, 3, 4, 5))
404
+ p = np.pad(drv, 1)
405
+ h, w = lane.shape
406
+ dil = np.zeros_like(drv)
407
+ for dr in range(3):
408
+ for dc in range(3):
409
+ dil |= p[dr:dr + h, dc:dc + w]
410
+ out = lane.copy()
411
+ out[edge] = 0
412
+ out[edge & dil & ~drv] = 6
413
+ return out
414
+
415
+
416
+ def metric_depth(bins: np.ndarray, kind: str = "linear") -> np.ndarray:
417
+ """Depth in metres of a bin index map: ``linear`` 1 + 1.25 b (the trained bins, SPEC 10 risk 2) or ``log`` (the
418
+ exported depth_mean centres)."""
419
+ return C.depth_bin_centres(kind)[np.asarray(bins, np.int64)]
420
+
421
+
422
+ def tl_state(tl: np.ndarray) -> Dict[str, Any]:
423
+ t = np.asarray(tl, np.float64).reshape(-1)
424
+ p = np.exp(t - t.max())
425
+ p /= p.sum()
426
+ k = int(np.argmax(t))
427
+ return {"state": C.TL_CLASSES[k], "state_id": k, "probs": p}
428
+
429
+
430
+ def occupancy_classes(occ: np.ndarray) -> np.ndarray:
431
+ """[1, 10, 16, 200, 200] logits -> uint8 [16, 200, 200] class per voxel (first max)."""
432
+ return np.argmax(np.asarray(occ).reshape(len(C.OCC_CLASSES), C.OCC_Z_BINS, 200, 200), axis=0).astype(np.uint8)
433
+
434
+
435
+ @dataclass
436
+ class FrameDecode:
437
+ """The stateless decode of one frame's outputs."""
438
+
439
+ boxes3d: List[Box3D]
440
+ stationary_healthy: bool
441
+ boxes2d: List[List[Box2D]]
442
+ unknown: List[Dict[str, float]]
443
+ plan: Dict[str, Any]
444
+ traffic_light: Dict[str, Any]
445
+ lane: np.ndarray
446
+ extras: Dict[str, Any] = field(default_factory=dict)
447
+
448
+
449
+ def decode_frame(outputs: Dict[str, np.ndarray], K: Optional[np.ndarray] = None, T_cam_ego: Optional[np.ndarray] = None,
450
+ cfg: PostConfig = PostConfig(), *, prev_mode: Optional[int] = None) -> FrameDecode:
451
+ """Every stateless decode of ``render.cpp`` on the 19 outputs (numpy, ONNX shapes). Seg fusion, yaw smoothing and
452
+ the mode history are ``host.temporal``'s."""
453
+ boxes = decode_boxes_3d(outputs["hm"], outputs["reg"], thresh=cfg.det3d_threshold, topk=cfg.det3d_topk)
454
+ boxes = bev_nms(filter_boxes_3d(boxes, cfg), cfg.bev_nms_iou, cfg.bev_nms_containment)
455
+ healthy = annotate_boxes(boxes, outputs.get("stationary"), outputs.get("traj"), cfg)
456
+ b2d = decode_boxes_2d(outputs, cfg)
457
+ unk: List[Dict[str, float]] = []
458
+ if cfg.unk2d and K is not None and T_cam_ego is not None:
459
+ unk = unknown_obstacles(b2d, outputs.get("depth_mean"), K, T_cam_ego, cfg)
460
+ lane = np.asarray(outputs["lane"], np.uint8).reshape(C.BEV_H, C.BEV_W)
461
+ return FrameDecode(boxes, healthy, b2d, unk, plan_from_ego(outputs["ego"], prev_mode, cfg),
462
+ tl_state(outputs["tl"]), lane)
code/tt_meteor/host/preprocess.py ADDED
@@ -0,0 +1,189 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Host pre-processing of METEOR's runtimes: eight camera frames + calibration + ego speed -> the graph's feed.
3
+
4
+ The contract (research/meteor/SPEC.md sections 2-3; ``meteor_v157.param.yaml``; tier4/METEOR ``hf/onnx_smoke_test.py:
5
+ 34-56``, the stated reference for feeding the graph, and ``deploy/cpp/realtime_main.cpp:123-140``):
6
+
7
+ - ``imgs`` uint8 [1, 8, 3, 432, 768], RGB, CHW, cameras in the training order (``config.CAMERAS``). A frame of another
8
+ size is resized with OpenCV ``INTER_AREA`` (``ttaw.image_area.meteor_resize``, bit-exact), anisotropically if the
9
+ aspect differs; frames are raw and unrectified (training never undistorted).
10
+ - ``K`` float32 [1, 8, 3, 3] at 768x432: the intrinsics of the image as captured, row 0 x 768 / w0 and row 1 x 432 / h0
11
+ (``extract_gt.py:128-130``); only fx, fy, cx, cy reach the graph.
12
+ - ``T_cam_ego`` float32 [1, 8, 4, 4] = inv(T_ego_cam): ``np.linalg.inv`` on float32 like ``onnx_smoke_test.py:46`` (the
13
+ research goldens' feed, byte-identical; ``inverse="float64"`` gives the C++ runtime's double inverse cast to float).
14
+ - ``v0`` float32 [1]: the ego speed in m/s.
15
+ - An absent camera (the 7-camera rig, or a dropped slot) is an all-zero image plus the DONOR's K and pose
16
+ (``bevlane/dataset.py:17-21``: CAM_BACK_NARROW <- CAM_BACK_WIDE, CAM_FRONT_NARROW <- CAM_FRONT_WIDE). With the
17
+ released ``/255`` input a uint8 zero image is exactly the trained "absent" input; with ``input_norm="imagenet"`` the
18
+ camera must be zeroed after normalising (``present`` mask, :func:`normalize_images`).
19
+
20
+ :func:`load_scene_frame` reads METEOR's demo-scene layout (``manifest.json``, ``img/NNNN_<CAM>.jpg``,
21
+ ``ego_motion.npz``), which the public-dataset samples reuse; it reproduces ``research/meteor/scripts/meteor_io.py``'s
22
+ feed byte for byte (``tests/test_host_preprocess.py`` checks the sha256 of the shipped sample against its golden).
23
+ :func:`assemble_frame` is the API path (images of any size + calibration + speed).
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import hashlib
28
+ import json
29
+ import math
30
+ import os
31
+ from dataclasses import dataclass, field
32
+ from pathlib import Path
33
+ from typing import Any, Dict, Mapping, Optional, Sequence, Tuple
34
+
35
+ import numpy as np
36
+
37
+ from ..reference import config as C
38
+ from ..ttaw.image_area import meteor_resize, scale_intrinsics
39
+ from ..ttaw.io import load_image
40
+
41
+ __all__ = ["MeteorFrame", "assemble_frame", "load_scene_frame", "load_manifest", "scene_dir_of", "invert_extrinsics",
42
+ "prepare_camera", "normalize_images", "pose_from_transform"]
43
+
44
+ F32 = np.float32
45
+
46
+
47
+ @dataclass
48
+ class MeteorFrame:
49
+ """One network input (the ONNX feed) plus what the host post-processing needs."""
50
+
51
+ imgs: np.ndarray # uint8 [1, 8, 3, 432, 768] RGB
52
+ K: np.ndarray # float32 [1, 8, 3, 3] at 768x432
53
+ T_cam_ego: np.ndarray # float32 [1, 8, 4, 4] ego -> camera
54
+ v0: np.ndarray # float32 [1] m/s
55
+ present: np.ndarray # bool [8]
56
+ pose: Optional[Tuple[float, float, float]] = None # ego (x, y, yaw) in a world frame (seg fusion, yaw tracks)
57
+ meta: Dict[str, Any] = field(default_factory=dict)
58
+
59
+ def feed(self) -> Dict[str, np.ndarray]:
60
+ """The ONNX input dict (``imgs``, ``K``, ``T_cam_ego``, ``v0``)."""
61
+ return {"imgs": self.imgs, "K": self.K, "T_cam_ego": self.T_cam_ego, "v0": self.v0}
62
+
63
+ def sha256(self) -> Dict[str, str]:
64
+ return {k: hashlib.sha256(np.ascontiguousarray(v).tobytes()).hexdigest() for k, v in self.feed().items()}
65
+
66
+
67
+ def invert_extrinsics(T_ego_cam: Any, inverse: str = "float32") -> np.ndarray:
68
+ """camera -> ego (4x4) to the graph's ego -> camera, float32. ``float32``: ``np.linalg.inv`` of the float32 matrix
69
+ (``onnx_smoke_test.py:46``, the research feed); ``float64``: double inverse cast to float (``realtime_main.cpp:
70
+ 123-128``)."""
71
+ if inverse == "float32":
72
+ return np.linalg.inv(np.asarray(T_ego_cam, F32).reshape(4, 4)).astype(F32)
73
+ if inverse == "float64":
74
+ return np.linalg.inv(np.asarray(T_ego_cam, np.float64).reshape(4, 4)).astype(F32)
75
+ raise ValueError(f"inverse must be 'float32' or 'float64', not {inverse!r}")
76
+
77
+
78
+ def prepare_camera(image: Any, K: Optional[Any] = None) -> Tuple[np.ndarray, Optional[np.ndarray], Tuple[int, int]]:
79
+ """One frame (anything ``ttaw.io.load_image`` takes; RGB) -> (uint8 [3, 432, 768] RGB CHW, K at 768x432 or None,
80
+ the source (h, w)). INTER_AREA when the size differs (``ttaw.image_area``); K is scaled per axis."""
81
+ rgb = load_image(image)
82
+ src_hw = (int(rgb.shape[0]), int(rgb.shape[1]))
83
+ if src_hw != (C.IMG_H, C.IMG_W):
84
+ if src_hw[0] < C.IMG_H or src_hw[1] < C.IMG_W:
85
+ raise ValueError(f"camera frame {src_hw[1]}x{src_hw[0]} is smaller than METEOR's 768x432 input "
86
+ "(INTER_AREA up-scaling is not supported)")
87
+ rgb = meteor_resize(rgb)
88
+ k = None if K is None else scale_intrinsics(K, src_hw)
89
+ return np.ascontiguousarray(rgb.transpose(2, 0, 1)), k, src_hw
90
+
91
+
92
+ def pose_from_transform(T_world_from_ego: Any) -> Tuple[float, float, float]:
93
+ """(x, y, yaw) of a 4x4 world <- ego transform (METEOR's ``ego_motion.npz`` ``pose`` convention)."""
94
+ T = np.asarray(T_world_from_ego, np.float64).reshape(4, 4)
95
+ return float(T[0, 3]), float(T[1, 3]), float(math.atan2(T[1, 0], T[0, 0]))
96
+
97
+
98
+ def assemble_frame(images: Mapping[str, Any], intrinsics: Mapping[str, Any], T_ego_cam: Mapping[str, Any],
99
+ v0: float, *, present: Optional[Mapping[str, bool]] = None, inverse: str = "float32",
100
+ pose: Optional[Tuple[float, float, float]] = None) -> MeteorFrame:
101
+ """The API path: per camera name an image (any size; None = absent), the intrinsics of that image and its
102
+ camera -> ego transform; ``v0`` in m/s. Absent cameras get a zero image and their donor's K and pose; a camera
103
+ without a donor (only the two narrow cameras have one) must be present. A narrow camera sent as an all-zero image
104
+ is absent too (METEOR's data convention, ``absent_zero.jpg``): its calibration is replaced by the donor's, as in
105
+ training (``bevlane/dataset.py:131-145``)."""
106
+ imgs = np.zeros((1, C.N_CAMS, 3, C.IMG_H, C.IMG_W), np.uint8)
107
+ K = np.zeros((1, C.N_CAMS, 3, 3), F32)
108
+ T = np.zeros((1, C.N_CAMS, 4, 4), F32)
109
+ flags = np.zeros(C.N_CAMS, bool)
110
+ src_sizes: Dict[str, Any] = {}
111
+ cams: Dict[str, Tuple[np.ndarray, np.ndarray]] = {}
112
+ for name in C.CAMERAS:
113
+ img = images.get(name)
114
+ ok = img is not None and (present is None or bool(present.get(name, True)))
115
+ if not ok:
116
+ continue
117
+ chw, k, src_hw = prepare_camera(img, intrinsics.get(name))
118
+ if name in C.DONOR and not chw.any():
119
+ continue # an all-zero narrow camera is METEOR's "absent" convention: zero + donor pose
120
+ if k is None or T_ego_cam.get(name) is None:
121
+ raise ValueError(f"{name}: intrinsics and T_ego_cam are required for a present camera")
122
+ i = C.CAMERAS.index(name)
123
+ imgs[0, i] = chw
124
+ flags[i] = True
125
+ cams[name] = (k, np.asarray(T_ego_cam[name], np.float64).reshape(4, 4))
126
+ src_sizes[name] = list(src_hw)
127
+ donors: Dict[str, str] = {}
128
+ for i, name in enumerate(C.CAMERAS):
129
+ src = name if flags[i] else C.DONOR.get(name)
130
+ if src is None or src not in cams:
131
+ raise ValueError(f"{name} is absent and has no present donor camera (donors: {C.DONOR})")
132
+ if src != name:
133
+ donors[name] = src
134
+ k, t = cams[src]
135
+ K[0, i] = np.asarray(k, F32)
136
+ T[0, i] = invert_extrinsics(t, inverse)
137
+ return MeteorFrame(imgs, K, T, np.array([v0], F32), flags, pose,
138
+ {"source_hw": src_sizes, "donors": donors, "inverse": inverse})
139
+
140
+
141
+ def scene_dir_of(root: os.PathLike) -> Path:
142
+ """``<root>/scenes.txt`` -> ``<root>/<first scene>`` (METEOR demo-root layout)."""
143
+ root = Path(root)
144
+ return root / (root / "scenes.txt").read_text().split()[0]
145
+
146
+
147
+ def load_manifest(scene_dir: os.PathLike) -> Dict[str, Any]:
148
+ return json.loads((Path(scene_dir) / "manifest.json").read_text())
149
+
150
+
151
+ def load_scene_frame(scene_dir: os.PathLike, frame: int, manifest: Optional[Mapping[str, Any]] = None, *,
152
+ inverse: str = "float32") -> MeteorFrame:
153
+ """Frame ``frame`` of a METEOR demo-scene directory (``manifest.json`` cams[c].K at 768x432 and T_ego_cam,
154
+ frames[i].imgs[c]; ``ego_motion.npz`` v0 / pose / wp / valid), as ``meteor_io.load_frame`` feeds the graph."""
155
+ scene_dir = Path(scene_dir)
156
+ m = manifest or load_manifest(scene_dir)
157
+ fr = m["frames"][frame]
158
+ imgs = np.zeros((1, C.N_CAMS, 3, C.IMG_H, C.IMG_W), np.uint8)
159
+ K = np.zeros((1, C.N_CAMS, 3, 3), F32)
160
+ T = np.zeros((1, C.N_CAMS, 4, 4), F32)
161
+ for i, cam in enumerate(C.CAMERAS):
162
+ chw, _, _ = prepare_camera(scene_dir / fr["imgs"][cam])
163
+ imgs[0, i] = chw
164
+ K[0, i] = np.asarray(m["cams"][cam]["K"], F32)
165
+ T[0, i] = invert_extrinsics(m["cams"][cam]["T_ego_cam"], inverse)
166
+ present = np.asarray(m.get("present_mask", [1] * C.N_CAMS), bool).reshape(C.N_CAMS)
167
+ ego = np.load(scene_dir / "ego_motion.npz")
168
+ v0 = float(ego["v0"][frame])
169
+ pose = tuple(float(v) for v in ego["pose"][frame]) if "pose" in ego.files else None
170
+ meta: Dict[str, Any] = {"scene": m.get("scene", scene_dir.name), "frame": int(frame)}
171
+ for key in ("wp", "valid"):
172
+ if key in ego.files:
173
+ meta[key] = np.asarray(ego[key][frame]).tolist()
174
+ return MeteorFrame(imgs, K, T, np.array([v0], F32), present, pose, meta)
175
+
176
+
177
+ def normalize_images(imgs_u8: np.ndarray, mode: str = "onnx", present: Optional[Sequence[bool]] = None) -> np.ndarray:
178
+ """Host oracle of the device input normalisation: uint8 [1, 8, 3, H, W] -> float32 [8, 3, H, W]; ``onnx`` = x / 255
179
+ (as released), ``imagenet`` = (x / 255 - mean) / std with absent cameras zeroed afterwards (D12)."""
180
+ x = np.asarray(imgs_u8).astype(F32) / F32(255.0)
181
+ if mode == "imagenet":
182
+ mean = np.asarray(C.IMAGENET_MEAN, F32).reshape(1, 1, 3, 1, 1)
183
+ std = np.asarray(C.IMAGENET_STD, F32).reshape(1, 1, 3, 1, 1)
184
+ x = (x - mean) / std
185
+ if present is not None:
186
+ x = x * np.asarray(present, F32).reshape(1, -1, 1, 1, 1)
187
+ elif mode != "onnx":
188
+ raise ValueError(f"input_norm must be one of {C.INPUT_NORMS}, not {mode!r}")
189
+ return x.reshape(C.N_CAMS, 3, C.IMG_H, C.IMG_W)
code/tt_meteor/host/result.py ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """The 19 network outputs of one frame (+ the frame, + the stream's host state) -> :class:`~.outputs.MeteorOutput`,
3
+ in the order of ``render.cpp``'s ``compose_frame``: 3D decode -> class filter -> BEV NMS -> yaw smoothing ->
4
+ stationary / futures; 2D decode -> unknown obstacles; mode selection with hysteresis; lane -> seg fusion -> road-edge
5
+ thinning. Shared by the API (``_postprocess``) and the CPU reference, so both apply exactly the same host code."""
6
+ from __future__ import annotations
7
+
8
+ from typing import Any, Dict, Mapping, Optional
9
+
10
+ import numpy as np
11
+
12
+ from ..reference import config as C
13
+ from .outputs import MeteorOutput
14
+ from .postprocess import PostConfig, decode_frame, occupancy_classes, plan_from_ego, sigmoid32, thin_road_edge
15
+ from .preprocess import MeteorFrame
16
+ from .temporal import StreamState
17
+
18
+ __all__ = ["build_output", "dense_heads"]
19
+
20
+
21
+ def dense_heads(outputs: Mapping[str, np.ndarray]) -> Dict[str, np.ndarray]:
22
+ """The dense per-head arrays of ``output_format="npz"``."""
23
+ return {"seg2d": np.asarray(outputs["seg2d"], np.uint8).reshape(C.N_CAMS, C.FEAT_H, C.FEAT_W),
24
+ "depth": np.asarray(outputs["depth"], np.uint8).reshape(C.N_CAMS, C.FEAT_H, C.FEAT_W),
25
+ "depth_mean": np.asarray(outputs["depth_mean"], np.float16).reshape(C.N_CAMS, C.FEAT_H // 2, C.FEAT_W // 2),
26
+ "occupancy": occupancy_classes(outputs["occ"]),
27
+ "risk": sigmoid32(np.asarray(outputs["risk"]).reshape(C.DET_H, C.DET_W)).astype(np.float16),
28
+ "stationary": np.asarray(outputs["stationary"], np.float32).reshape(C.DET_H, C.DET_W)}
29
+
30
+
31
+ def build_output(outputs: Mapping[str, np.ndarray], frame: MeteorFrame, cfg: PostConfig = PostConfig(),
32
+ state: Optional[StreamState] = None, *, heads: bool = False, model: str = "meteor-p150",
33
+ timing_ms: Optional[Dict[str, float]] = None, meta: Optional[Dict[str, Any]] = None) -> MeteorOutput:
34
+ dec = decode_frame(outputs, frame.K, frame.T_cam_ego, cfg)
35
+ lane = dec.lane
36
+ plan = dec.plan
37
+ if state is not None:
38
+ if cfg.yaw_smoothing:
39
+ state.yaw.apply(dec.boxes3d)
40
+ plan = plan_from_ego(outputs["ego"], state.mode.prev, cfg)
41
+ state.mode.prev = int(plan["mode"])
42
+ if cfg.seg_fuse:
43
+ lane = state.seg.apply(lane, outputs["lane_logit"], frame.pose)
44
+ state.frames += 1
45
+ if cfg.thin_road_edge:
46
+ lane = thin_road_edge(lane)
47
+ out_meta = {"present": [bool(v) for v in frame.present], "v0": float(frame.v0[0])}
48
+ out_meta.update(meta or {})
49
+ return MeteorOutput(dec.boxes3d, dec.boxes2d, dec.unknown, plan, dec.traffic_light, lane, dec.stationary_healthy,
50
+ dense_heads(outputs) if heads else {}, model=model, timing_ms=dict(timing_ms or {}),
51
+ meta=out_meta)
code/tt_meteor/host/temporal.py ADDED
@@ -0,0 +1,180 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Host-side temporal state of METEOR's runtime (the deployed graph itself is single-frame; SPEC section 3 end).
3
+
4
+ Three pieces of state live in ``render.cpp``, all host post-processing (RT-host), kept here per ``stream.id``:
5
+
6
+ - **BEV seg fusion** (``seg_fuse``, ``render.cpp:131-198``): the 9-class log-softmax of ``lane_logit`` is accumulated
7
+ with decay 0.7, the previous accumulator warped into the current ego frame by the pose change (nearest,
8
+ ``cv::warpAffine`` with ``WARP_INVERSE_MAP``, 5 px/m about row 400 / col 250); the fused argmax replaces the lane
9
+ class on rows >= 175 (x <= 45 m) except where the current class is thin (3 crosswalk, 4 laneline, 5 stopline,
10
+ 6 road edge). The state resets itself on a pose jump (|tx| + |ty| >= 10 m in the previous frame, or |dyaw| >= 0.5).
11
+ - **Yaw smoothing** (``render.cpp:396-421``): every box matched to the nearest previous box within 2.5 m has its yaw
12
+ flipped by pi if it turned by more than pi / 2 (no 180 deg flips) and is blended 0.4 towards the new value.
13
+ - **Mode hysteresis** (``render.cpp:432-442``): the previously selected ego mode gets +0.35 on its logit.
14
+
15
+ Upstream resets only the seg accumulator (on a pose jump); the yaw tracks and the previous mode have no reset. This
16
+ port resets all three on ``stream.reset`` / a new stream id (SPEC section 3, "our design"). Seg fusion needs the ego
17
+ pose (``stream.T_world_from_ego`` -> (x, y, yaw)); without a pose it is skipped and its state cleared, as the C++ does.
18
+ The warp uses OpenCV (as the C++); the log-softmax is the C++'s float32 loop order (max, sum of exp, log).
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import math
23
+ from dataclasses import dataclass, field
24
+ from typing import Any, Dict, List, Optional, Tuple
25
+
26
+ import numpy as np
27
+
28
+ from ..reference import config as C
29
+ from .postprocess import Box3D, PostConfig, select_mode
30
+
31
+ __all__ = ["SegFusion", "YawTracks", "ModeHistory", "StreamState", "warp_matrix", "log_softmax9"]
32
+
33
+ F32 = np.float32
34
+ THIN = (3, 4, 5, 6)
35
+ SEG_DECAY = 0.7
36
+ SEG_ROW0 = 175 # rows >= 175 (x <= 45 m) are replaced by the fused class
37
+ YAW_GATE2 = 6.25 # 2.5 m squared
38
+ YAW_EMA = 0.4
39
+
40
+
41
+ def warp_matrix(pose: Tuple[float, float, float], prev: Tuple[float, float, float]) -> Optional[np.ndarray]:
42
+ """``warpM`` (render.cpp:149-160): the inverse-map affine of the 800x500 grid from the previous pose to the
43
+ current one, float32 [2, 3]; None on a pose jump (|tx| + |ty| >= 10 m or |dyaw| >= 0.5 rad)."""
44
+ cp, sp = math.cos(prev[2]), math.sin(prev[2])
45
+ dx, dy = pose[0] - prev[0], pose[1] - prev[1]
46
+ tx, ty = cp * dx + sp * dy, -sp * dx + cp * dy
47
+ dyaw = pose[2] - prev[2]
48
+ if abs(tx) + abs(ty) >= 10.0 or abs(dyaw) >= 0.5:
49
+ return None
50
+ c, s2 = math.cos(dyaw), math.sin(dyaw)
51
+ return np.array([[c, s2, -250 * c - 400 * s2 - 5 * ty + 250],
52
+ [-s2, c, 250 * s2 - 400 * c - 5 * tx + 400]], F32)
53
+
54
+
55
+ def log_softmax9(logit: np.ndarray) -> np.ndarray:
56
+ """[9, H, W] logits -> [H, W, 9] float32 log-probabilities, the C++ loop: mx = max_k, se = sum_k exp(v - mx) in
57
+ class order (float), lse = mx + log(se), lp = v - lse."""
58
+ v = np.asarray(logit, F32)
59
+ mx = v.max(axis=0)
60
+ se = np.zeros(mx.shape, F32)
61
+ for k in range(v.shape[0]):
62
+ se = se + np.exp(v[k] - mx)
63
+ lse = mx + np.log(se)
64
+ return np.ascontiguousarray((v - lse[None]).transpose(1, 2, 0))
65
+
66
+
67
+ class SegFusion:
68
+ """The BEV seg temporal accumulator of one stream (``seg_fuse`` with logits)."""
69
+
70
+ def __init__(self) -> None:
71
+ self.acc: Optional[np.ndarray] = None # [800, 500, 9] float32 log-probabilities
72
+ self.pose: Optional[Tuple[float, float, float]] = None
73
+
74
+ def reset(self) -> None:
75
+ self.acc, self.pose = None, None
76
+
77
+ def apply(self, lane: np.ndarray, lane_logit: np.ndarray, pose: Optional[Tuple[float, float, float]]) -> np.ndarray:
78
+ """Fused lane map (uint8 [800, 500]; a new array) from this frame's argmax and logits [9, 800, 500]."""
79
+ lane = np.array(np.asarray(lane, np.uint8).reshape(C.BEV_H, C.BEV_W))
80
+ if pose is None:
81
+ self.reset()
82
+ return lane
83
+ import cv2
84
+
85
+ lp = log_softmax9(np.asarray(lane_logit).reshape(len(C.LANE_CLASSES), C.BEV_H, C.BEV_W))
86
+ if self.acc is not None:
87
+ M = warp_matrix(pose, self.pose)
88
+ if M is not None:
89
+ warped = np.concatenate(
90
+ [cv2.warpAffine(np.ascontiguousarray(self.acc[:, :, k:k + 3]), M, (C.BEV_W, C.BEV_H),
91
+ flags=cv2.INTER_NEAREST | cv2.WARP_INVERSE_MAP, borderMode=cv2.BORDER_CONSTANT,
92
+ borderValue=(0, 0, 0)) for k in range(0, 9, 3)], axis=2)
93
+ lp = lp + F32(SEG_DECAY) * warped
94
+ self.acc = lp
95
+ self.pose = (float(pose[0]), float(pose[1]), float(pose[2]))
96
+ sub = lp[SEG_ROW0:]
97
+ best = np.zeros(sub.shape[:2], np.int64) # first max with a strict > scan (class order)
98
+ bv = sub[:, :, 0].copy()
99
+ for k in range(1, sub.shape[2]):
100
+ better = sub[:, :, k] > bv
101
+ best[better] = k
102
+ bv[better] = sub[:, :, k][better]
103
+ cur = lane[SEG_ROW0:]
104
+ keep = np.isin(cur, THIN)
105
+ cur[~keep] = best[~keep].astype(np.uint8)
106
+ return lane
107
+
108
+
109
+ class YawTracks:
110
+ """The yaw-smoothing tracks of one stream (``g_yawTracks``): the previous frame's boxes (x, y, yaw)."""
111
+
112
+ def __init__(self) -> None:
113
+ self.tracks: List[Tuple[float, float, float]] = []
114
+
115
+ def reset(self) -> None:
116
+ self.tracks = []
117
+
118
+ def apply(self, boxes: List[Box3D]) -> List[Box3D]:
119
+ """Smooth ``boxes`` (in place, NMS order) against the previous frame and make them the new tracks."""
120
+ nxt = []
121
+ for b in boxes:
122
+ best, bd = None, 1e30
123
+ bx, by = F32(b.x), F32(b.y)
124
+ for t in self.tracks:
125
+ ddx, ddy = bx - F32(t[0]), by - F32(t[1])
126
+ dd = float(ddx * ddx + ddy * ddy) # float arithmetic, compared in double
127
+ if dd < YAW_GATE2 and dd < bd:
128
+ best, bd = t, dd
129
+ if best is not None:
130
+ py = F32(best[2])
131
+ yaw = F32(b.yaw)
132
+ dy = F32(math.fmod(float(yaw) - float(py) + math.pi, 2 * math.pi))
133
+ if dy < 0:
134
+ dy = F32(float(dy) + 2 * math.pi)
135
+ dy = F32(float(dy) - math.pi)
136
+ if abs(float(dy)) > math.pi / 2:
137
+ yaw = F32(float(yaw) + (math.pi if dy < 0 else -math.pi))
138
+ dy = F32(math.fmod(float(yaw) - float(py) + math.pi, 2 * math.pi))
139
+ if dy < 0:
140
+ dy = F32(float(dy) + 2 * math.pi)
141
+ dy = F32(float(dy) - math.pi)
142
+ b.yaw = float(py + F32(YAW_EMA) * dy)
143
+ nxt.append((float(F32(b.x)), float(F32(b.y)), float(F32(b.yaw))))
144
+ self.tracks = nxt
145
+ return boxes
146
+
147
+
148
+ class ModeHistory:
149
+ """The previously selected ego mode of one stream (``g_prevMode``)."""
150
+
151
+ def __init__(self) -> None:
152
+ self.prev: Optional[int] = None
153
+
154
+ def reset(self) -> None:
155
+ self.prev = None
156
+
157
+ def select(self, ego: np.ndarray, cfg: PostConfig = PostConfig()) -> int:
158
+ k, _ = select_mode(ego, self.prev, cfg)
159
+ self.prev = k
160
+ return k
161
+
162
+
163
+ @dataclass
164
+ class StreamState:
165
+ """All host temporal state of one ``stream.id``."""
166
+
167
+ seg: SegFusion = field(default_factory=SegFusion)
168
+ yaw: YawTracks = field(default_factory=YawTracks)
169
+ mode: ModeHistory = field(default_factory=ModeHistory)
170
+ frames: int = 0
171
+
172
+ def reset(self) -> None:
173
+ self.seg.reset()
174
+ self.yaw.reset()
175
+ self.mode.reset()
176
+ self.frames = 0
177
+
178
+ def describe(self) -> Dict[str, Any]:
179
+ return {"frames": self.frames, "seg_pose": self.seg.pose, "yaw_tracks": len(self.yaw.tracks),
180
+ "prev_mode": self.mode.prev}
code/tt_meteor/io.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Input decoding and output encoding shared by the Python API and the HTTP server of meteor-p150.
3
+
4
+ The implementation is the vendored ``ttaw.io`` (C08): point clouds (``bin`` / ``npy`` / ``npz`` / ``pcd`` /
5
+ ``list``), images, cameras + calibration, transforms, named arrays and the output encoders; numpy only, data
6
+ parsing only, and every client mistake raises :class:`InputError`, which the server maps to HTTP 400. This module
7
+ re-exports it with the Autoware input layout of this model bound as the default point fields, so ``model(points)``
8
+ and ``POST /predict`` decode the same bytes into the same array.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from typing import Any, Mapping, Optional, Sequence
13
+
14
+ from .ttaw import io as _io
15
+ from .ttaw.io import * # noqa: F401,F403 (the decoders and encoders: ttaw/API.md section 9)
16
+ from .ttaw.io import MAX_POINTS_DEFAULT, PointCloud
17
+
18
+ __all__ = list(_io.__all__) # the same public names; the three below are rebound to this model's layout
19
+
20
+ DEFAULT_POINT_FIELDS = tuple(["x", "y", "z", "intensity"]) # the Autoware input layout of this model
21
+
22
+
23
+ def decode_points(spec: Mapping[str, Any], *, max_points: int = MAX_POINTS_DEFAULT, max_bytes: Optional[int] = None,
24
+ default_fields: Sequence[str] = DEFAULT_POINT_FIELDS) -> PointCloud:
25
+ """The ``points`` envelope of ``/predict`` -> :class:`PointCloud` (``ttaw.io.decode_points`` with this model's
26
+ default fields)."""
27
+ return _io.decode_points(spec, max_points=max_points, max_bytes=max_bytes, default_fields=default_fields)
28
+
29
+
30
+ def load_points(source: Any, *, fields: Optional[Sequence[str]] = None, fmt: Optional[str] = None,
31
+ frame_id: str = "base_link", default_fields: Sequence[str] = DEFAULT_POINT_FIELDS) -> PointCloud:
32
+ """Python-API point input (path, bytes + ``fmt``, (N, C) array or tensor, structured array, envelope,
33
+ :class:`PointCloud`) -> :class:`PointCloud` (``ttaw.io.load_points`` with this model's default fields)."""
34
+ return _io.load_points(source, fields=fields, fmt=fmt, frame_id=frame_id, default_fields=default_fields)
code/tt_meteor/reference/__init__.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """CPU fp32 reference of METEOR (TIER IV, AutowareFoundation/meteor v1.0, ``meteor_v157c3Z.onnx``): the ground truth
3
+ every PCC / output-agreement gate compares against (PLAN.md 0.3 item 1; research/meteor/SPEC.md section 6).
4
+
5
+ Layout (importable without ttnn; torch / onnx are imported inside the functions that need them):
6
+
7
+ - ``config.py`` the compile-time constants of the deployed graph (cameras, sizes, bins, crops, class lists) and
8
+ the load-time options (``input_norm``, ``depth_mean_bins``), each with its source.
9
+ - ``weights.py`` ``MeteorWeights``: the ONNX parsed as data with the vendored ``ttaw.weights`` (parameters
10
+ addressed by their consuming node; conv attributes from the graph; BN already folded by the
11
+ export), plus a graph check that the file is the deployed network.
12
+ - ``model.py`` ``MeteorNet``: the whole graph in torch fp32 (a node-by-node transliteration) with per-module taps.
13
+ - ``port_form.py`` the TT port's exact rewrites (input scale fold, merged heads, paint fold, tgate + tfuse merge,
14
+ traj_stem without the zero history channels, the lift from host tables) with CPU proofs.
15
+ - ``pipeline.py`` ``MeteorReference``: the end-to-end frame (network + the shared host pre / post-processing),
16
+ stateless ``run_frame`` and the API-equivalent ``__call__``.
17
+ - ``ort.py`` the ONNX Runtime oracle with any internal tensor as a named tap (research and tests only;
18
+ onnxruntime is NOT a runtime dependency of the image).
19
+
20
+ There is no Autoware package for METEOR: the pre- and post-processing reproduced in ``tt_meteor.host`` are METEOR's own
21
+ runtimes (tier4/METEOR @ dc193a81e9959de57b751e269127ae9b67e3f300, ``deploy/cpp``).
22
+
23
+ The ``to_dict()`` of this reference's output on the shipped sample is stored as
24
+ ``samples/<sample stem>.reference.json``: ``server/smoke_test.py`` compares the served output with it.
25
+ """
code/tt_meteor/reference/config.py ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Compile-time constants of the METEOR port (``meteor_v157c3Z.onnx``, AutowareFoundation/meteor@v1.0).
3
+
4
+ Every value here is either a static shape of the deployed graph or a constant baked into it; the source of each is
5
+ the ONNX node it comes from (``research/meteor/SPEC.md``, cited as ``S:<line>``) or METEOR's runtime
6
+ (``tier4/METEOR`` @ ``dc193a81e995``, ``deploy/cpp``). Changing one of them means a different graph, a new trace and
7
+ a new image (PLAN.md section 0.4, class COMPILE). Host post-processing defaults (class RT-host) live in
8
+ ``host/postprocess.py`` (``PostConfig``); load-time options (``input_norm``, ``depth_mean_bins``) in
9
+ :class:`MeteorConfig`.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import math
14
+ from dataclasses import dataclass, field
15
+ from typing import Tuple
16
+
17
+ __all__ = [
18
+ "ONNX_FILE", "ONNX_SHA256", "PARAM_YAML", "HF_REPO", "HF_TAG", "HF_REVISION", "METEOR_UPSTREAM_COMMIT",
19
+ "CAMERAS", "N_CAMS", "IMG_H", "IMG_W", "FEAT_H", "FEAT_W", "FPN_C", "CTX_C", "BEV_C", "BEV_H", "BEV_W",
20
+ "BEV_RES", "BEV_XF", "BEV_YH", "LIFT_H", "LIFT_W", "LIFT_RES", "DET_H", "DET_W", "DET_RES", "DEPTH_BINS",
21
+ "DEPTH_MIN", "DEPTH_STEP", "DEPTH_LOG_MAX", "LANE_CLASSES", "SEG2D_CLASSES", "DET2D_CLASSES", "DET3D_CLASSES",
22
+ "OCC_CLASSES", "TL_CLASSES", "DET2D_STRIDES", "PAINT_CLASSES", "OCC_CROP", "RISK_CROP", "EGO_MODES", "EGO_STEPS",
23
+ "EGO_DT", "EGO_DIM", "TRAJ_C", "DONOR", "IMAGENET_MEAN", "IMAGENET_STD", "INPUT_NORMS", "DEPTH_MEAN_BINS",
24
+ "MeteorConfig", "default_config", "depth_bin_centres",
25
+ ]
26
+
27
+ # ---- artifact pins (S:99-118) ---------------------------------------------------------------------------------------
28
+ ONNX_FILE = "meteor_v157c3Z.onnx"
29
+ ONNX_SHA256 = "50397b4df76d33df2a71e8063b38d164399255008e1f0758c2a83675a200b1a7"
30
+ PARAM_YAML = "meteor_v157.param.yaml"
31
+ HF_REPO = "AutowareFoundation/meteor"
32
+ HF_TAG = "v1.0"
33
+ HF_REVISION = "01a5f6d71df5ecbbb5853ec600825481d57b9c6b"
34
+ METEOR_UPSTREAM_COMMIT = "dc193a81e9959de57b751e269127ae9b67e3f300" # tier4/METEOR, defines pre/post (no Autoware pkg)
35
+
36
+ # ---- cameras and image branch (param.yaml:10, S:136-140) ------------------------------------------------------------
37
+ CAMERAS: Tuple[str, ...] = ("CAM_FRONT_WIDE", "CAM_FRONT_LEFT", "CAM_FRONT_RIGHT", "CAM_BACK_WIDE", "CAM_BACK_LEFT",
38
+ "CAM_BACK_RIGHT", "CAM_FRONT_NARROW", "CAM_BACK_NARROW")
39
+ N_CAMS = 8
40
+ IMG_H, IMG_W = 432, 768 # network input per camera (the graph normalises u by 767, v by 431)
41
+ FEAT_H, FEAT_W = 108, 192 # stride-4 FPN map
42
+ FPN_C = 160 # FPN width
43
+ CTX_C = 96 # painted context channels lifted to the BEV
44
+ # the absent-camera donor rule of training (bevlane/dataset.py:17-21): zero image + the donor's K and pose
45
+ DONOR = {"CAM_BACK_NARROW": "CAM_BACK_WIDE", "CAM_FRONT_NARROW": "CAM_FRONT_WIDE"}
46
+
47
+ # ---- BEV (model.py:34-47, 82-89, 877-879; S:155-160) ----------------------------------------------------------------
48
+ BEV_C = 96
49
+ BEV_H, BEV_W = 800, 500 # rows: x = +80 .. -80 m, cols: y = +50 .. -50 m
50
+ BEV_RES = 0.2
51
+ BEV_XF, BEV_YH = 80.0, 50.0 # forward extent, half width
52
+ LIFT_H, LIFT_W = 400, 250 # 0.4 m lift grid (V52, LIFT_DIV 2): x 79.8 .. -79.8, y 49.8 .. -49.8
53
+ LIFT_RES = 0.4
54
+ DET_H, DET_W = 400, 250 # 3D detection / traj / stationary / risk grids
55
+ DET_RES = 0.4
56
+
57
+ # ---- depth bins (S:325-336; SPEC section 10 risk 2) -----------------------------------------------------------------
58
+ DEPTH_BINS = 64
59
+ DEPTH_MIN, DEPTH_STEP = 1.0, 1.25 # LINEAR bins of the lift and the trained head: 1.0 + 1.25 * b metres
60
+ DEPTH_LOG_MAX = 4.378896236419678 # ln(79.75) as the float32 constant of /Div_1: depth_mean's log-spaced centres
61
+ LIFT_MIN_Z = 0.5 # camera z clamp and validity threshold
62
+ LIFT_MAX_DIST = 90.0 # validity: Euclidean range from the camera < 90 m
63
+ LIFT_BIN_CLIP = 62.9999 # clip of the continuous bin index (/net/Clip_4, float32 62.99990081787109)
64
+ LIFT_W_EPS = 0.05 # + 0.05 on the interpolated bin probability (/net/Add_10)
65
+ LIFT_DEN_EPS = 1e-4 # max(sum_n w, 1e-4) (/net/Clip_6)
66
+
67
+ # ---- heads -----------------------------------------------------------------------------------------------------
68
+ LANE_CLASSES = ("bg", "road", "sidewalk", "crosswalk", "laneline", "stopline", "road_edge", "marking", "parking")
69
+ SEG2D_CLASSES = ("bg", "obstacle", "car", "truck", "bus", "motorcycle", "bicycle", "pedestrian", "marking",
70
+ "traffic_light", "sign", "road", "sidewalk", "lane_stop_line", "crosswalk", "unused", "wall",
71
+ "building", "vegetation", "sky", "pole") # comlops-21cls-autolabel-2504.csv (0, 15 unused)
72
+ DET2D_CLASSES = ("obs", "car", "truck", "bus", "bicycle", "motorcycle", "pedestrian", "road_paint", "traffic_light",
73
+ "traffic_sign") # comlops-instance-2510.csv
74
+ DET3D_CLASSES = ("vehicle", "vru")
75
+ OCC_CLASSES = ("free", "obstacle", "vehicle", "two_wheeler", "pedestrian", "road", "sidewalk", "vegetation",
76
+ "building_wall", "pole_sign_light") # extract_occ.py:4-8; 16 z-bins of 0.4 m from -1 m
77
+ OCC_Z_BINS = 16
78
+ TL_CLASSES = ("none", "green", "yellow", "red")
79
+ DET2D_STRIDES = (4, 8, 16)
80
+ PAINT_CLASSES = (2, 3, 4, 5, 6, 7, 8, 13) # seg2d classes painted into the context (/net/Gather indices; checked)
81
+ SEG2D_CLAMP = 30.0 # +-30 clamp, ONLY before the seg2d argmax (the paint softmax is unclamped)
82
+ OCC_CROP = ((200, 600), (50, 450)) # rows, cols of the RAW BEV (/net/Slice_8, Slice_9)
83
+ RISK_CROP = ((200, 600), (125, 375)) # rows, cols of the FUSED BEV (/net/Slice_11, Slice_12)
84
+ TGATE_SCALE = 4.0 # fused = bev + tfuse(bev * 4 * softmax(tgate)[0]) (/net/Mul_12)
85
+
86
+ # ---- planner (S:315, 4.4) --------------------------------------------------------------------------------------
87
+ EGO_MODES, EGO_STEPS, EGO_DT = 3, 6, 0.5
88
+ EGO_DIM = 42 # [0:36] 3 paths x 6 x (x, y); [36:39] mode logits; 39 steer, 40 accel, 41 brake
89
+ TRAJ_C = 39 # per-cell agent futures: 3 modes x 6 x (dx, dy) + 3 mode logits
90
+ V0_SCALE = 15.0 # kin_delta input [v0 / 15, 0, ..., 0] (7 wide)
91
+ KIN_DIM = 7
92
+ DEC_SPEED_MAX = 25.0 # decoupled decoder: v = min(softplus(.), 25)
93
+ RISK_X_HALF, RISK_Y_HALF = 40.0, 25.0 # risk grid +-40 m x +-25 m at 0.2 m (400 x 250)
94
+ E2E_LN_EPS = 1e-5 # LayerNormalization(139) epsilon (float32 9.99999974738e-06)
95
+ E2E_SCALE = 3.0 # ego[:36] += 3 * tanh(mlp(LN(...)))
96
+ SEG_REFINE_CLAMP, SEG_REFINE_SCALE = 20.0, 8.0 # lane_logit = clamp(lane_pre, +-20) + 8 tanh(res / 8)
97
+ BOX_HM_CLAMP, BOX_REG_CLAMP, BOX_REFINE_SCALE = 15.0, 20.0, 6.0
98
+
99
+ # ---- input normalisation (D12; S:209-229, section 10 risk 1) ---------------------------------------------------------
100
+ IMAGENET_MEAN = (0.485, 0.456, 0.406)
101
+ IMAGENET_STD = (0.229, 0.224, 0.225)
102
+ INPUT_NORMS = ("onnx", "imagenet") # onnx: x / 255 as released (default); imagenet: (x / 255 - mean) / std
103
+ DEPTH_MEAN_BINS = ("log", "linear") # depth_mean bin centres: log as exported (default, parity) or linear (trained)
104
+
105
+
106
+ def depth_bin_centres(kind: str = "log"):
107
+ """The 64 bin centres of ``depth_mean`` in metres: ``log`` = exp(r * ln(79.75) / 63) in float32 exactly as the
108
+ graph (/Mul_1 -> /Exp), ``linear`` = 1 + 1.25 r (the bins the head and the lift were trained with)."""
109
+ import numpy as np
110
+
111
+ r = np.arange(DEPTH_BINS, dtype=np.float32)
112
+ if kind == "log":
113
+ step = np.float32(DEPTH_LOG_MAX) / np.float32(DEPTH_BINS - 1)
114
+ return np.exp(r * step + np.float32(0.0)).astype(np.float32)
115
+ if kind == "linear":
116
+ return (np.float32(DEPTH_MIN) + np.float32(DEPTH_STEP) * r).astype(np.float32)
117
+ raise ValueError(f"depth_mean bins must be one of {DEPTH_MEAN_BINS}, not {kind!r}")
118
+
119
+
120
+ @dataclass(frozen=True)
121
+ class MeteorConfig:
122
+ """Load-time options of the reference (and of the TT port, ``METEOR_INPUT_NORM`` / ``METEOR_DEPTH_MEAN_BINS``).
123
+
124
+ ``input_norm="onnx"`` reproduces the released graph (only /255; D12 default, validated against ORT);
125
+ ``"imagenet"`` adds the training normalisation after /255 (SPEC section 10 risk 1; an in-memory patched ONNX is
126
+ its oracle). ``depth_mean_bins="log"`` keeps ONNX parity, ``"linear"`` gives metric depth as trained."""
127
+
128
+ input_norm: str = "onnx"
129
+ depth_mean_bins: str = "log"
130
+ cameras: Tuple[str, ...] = field(default=CAMERAS)
131
+
132
+ def check(self) -> "MeteorConfig":
133
+ if self.input_norm not in INPUT_NORMS:
134
+ raise ValueError(f"input_norm must be one of {INPUT_NORMS}, not {self.input_norm!r}")
135
+ if self.depth_mean_bins not in DEPTH_MEAN_BINS:
136
+ raise ValueError(f"depth_mean_bins must be one of {DEPTH_MEAN_BINS}, not {self.depth_mean_bins!r}")
137
+ if tuple(self.cameras) != CAMERAS:
138
+ raise ValueError("the camera order is fixed by training (param.yaml:10)")
139
+ return self
140
+
141
+
142
+ def default_config(**overrides) -> MeteorConfig:
143
+ return MeteorConfig(**overrides).check()
144
+
145
+
146
+ # sanity: the log centres span 1 .. 79.75 m (S:183)
147
+ assert abs(math.exp(DEPTH_LOG_MAX) - 79.75) < 1e-4
code/tt_meteor/reference/model.py ADDED
@@ -0,0 +1,539 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Pure-PyTorch fp32 re-implementation of ``meteor_v157c3Z.onnx`` with per-module taps.
3
+
4
+ A node-by-node transliteration of the deployed graph (2,021 ONNX nodes; ``research/meteor/SPEC.md`` sections 4.2-4.4,
5
+ cited ``S:<line>``), written from the graph itself (no upstream code is imported or executed): every convolution runs
6
+ with the weights and the attributes of its ONNX node (``reference.weights``), every non-conv constant (attention
7
+ pools, LayerNorm, gates, pooling matrices, lift ground points) is read from its consuming node, and the arithmetic
8
+ keeps the graph's operation order where it decides a rounding or a discrete choice (lift projection, validity and
9
+ depth-bin selection, risk sampling, decoupled decoder, depth_mean). Shape glue (Shape / Gather / Concat / Reshape of
10
+ shapes, ~470 nodes) disappears. ``MeteorNet.forward`` returns the 19 graph outputs with the ONNX names, dtypes and
11
+ shapes; ``taps=`` records the intermediates below (names shared with ``reference.ort.ORT_TAPS``, the device gates
12
+ and the goldens).
13
+
14
+ Module boundaries (what a TT stage computes) and taps:
15
+
16
+ ========================== =========================== ====================================================
17
+ tap shape graph tensor / meaning
18
+ ========================== =========================== ====================================================
19
+ ``img.input`` [8, 3, 432, 768] uint8 / 255 (+ ImageNet mean/std with input_norm=imagenet)
20
+ ``img.stem`` [8, 64, 108, 192] 7x7 s2 + ReLU + maxpool 3x3 s2
21
+ ``img.layer1..4`` [8, 64..512, ...] ResNet-34 stages (BasicBlocks, BN folded)
22
+ ``img.fpn`` [8, 160, 108, 192] lat1..4 + bilinear (x2, x4, 14x24 -> 108x192) + fuse 3x3 ReLU
23
+ ``seg2d.logits`` [8, 21, 108, 192] SegHeadED raw logits (the PointPainting input, unclamped)
24
+ ``seg2d.clamped`` [8, 21, 108, 192] nan_to_num + clamp +-30 (the seg2d argmax input only)
25
+ ``depth.logits`` / ``.prob`` [8, 64, 108, 192] depth head 1x1 / softmax over the 64 linear bins
26
+ ``ctx.painted`` [8, 96, 108, 192] ctx 1x1 + paint_proj(softmax(seg)[2,3,4,5,6,7,8,13])
27
+ ``lift.grid`` [8, 100000, 1, 2] normalised (u, v) of the 400x250 ground points per camera
28
+ ``lift.valid`` [8, 100000] bool z > 0.5, inside the image, range < 90 m
29
+ ``lift.weight`` [8, 1, 100000] (interpolated bin probability + 0.05) * valid
30
+ ``lift.bev`` [1, 96, 400, 250] sum_n ctx_s w / max(sum_n w, 1e-4)
31
+ ``bev.raw`` [1, 96, 800, 500] bilinear x2 of lift.bev
32
+ ``bev.gate`` [1, 1, 800, 500] 4 softmax(tgate(raw))[0]
33
+ ``bev.fused`` [1, 96, 800, 500] raw + tfuse(raw * gate)
34
+ ``lane.pre`` [1, 9, 800, 500] lane decoder + lane_branch on channels 4..6
35
+ ``det.feat`` [1, 128, 400, 250] det stem (ED) output
36
+ ``det.hm_pre/.reg_pre`` [1, 2 | 6, 400, 250] heads before the box refiner
37
+ ``occ.feat`` [1, 192, 200, 200] occ stem on raw rows 200:600, cols 50:450
38
+ ``traj.stem / .feat`` [1, 128 | 256, 400, 250] traj stem; cat(stem, det.feat) + agent_delta
39
+ ``traj.agent_delta`` [1, 256, 1, 1] agent attention pool -> 1x1 conv
40
+ ``stat.feat`` [1, 48, 400, 250] delta_stat convs on avgpool2(|raw|)
41
+ ``risk.feat`` [1, 64, 400, 250] risk ConvBlocks on fused rows 200:600, cols 125:375
42
+ ``ego.*`` [1, 42] (and others) planner chain: stem, mlp, attn, plus_attn, sem_in, sem,
43
+ plus_sem, kin, plus_kin, risk_sample, after_risk,
44
+ dec_head, dec_wp, after_dec
45
+ ``bev.fused_mean`` [1, 96] global mean of the fused BEV (computed once; S:368-369)
46
+ ``refiner.seg_res`` [1, 9, 800, 500] seg refiner residual (before 8 tanh(x / 8))
47
+ ``refiner.box_res`` [1, 8, 400, 250] box refiner residual (before 6 tanh(x / 6))
48
+ ``refiner.e2e_ln / _res`` [1, 139] / [1, 36] e2e LayerNorm output / 3 tanh(mlp)
49
+ ========================== =========================== ====================================================
50
+
51
+ The 19 outputs are tapped under their ONNX names (``lane``, ``depth``, ``seg2d``, ``hm`` ... ``depth_mean``).
52
+ """
53
+ from __future__ import annotations
54
+
55
+ from typing import Any, Dict, Optional
56
+
57
+ import numpy as np
58
+
59
+ from ..ttaw.golden import NULL_TAPS, TapRegistry
60
+ from . import config as C
61
+ from .config import MeteorConfig, default_config
62
+ from .weights import MeteorWeights
63
+
64
+ __all__ = ["MeteorNet", "OUTPUT_NAMES", "child"]
65
+
66
+ OUTPUT_NAMES = ("lane", "depth", "seg2d", "hm", "reg", "hm2d_s0", "hm2d_s1", "hm2d_s2", "reg2d_s0", "reg2d_s1",
67
+ "reg2d_s2", "ego", "occ", "traj", "stationary", "tl", "risk", "lane_logit", "depth_mean")
68
+
69
+
70
+ def _torch():
71
+ import torch
72
+
73
+ return torch
74
+
75
+
76
+ def child(module: str, index: int) -> str:
77
+ """ONNX scope of child ``index`` of a Sequential: ``child("seg_head/d1/d1.3", 0)`` = ``seg_head/d1/d1.3/d1.3.0``."""
78
+ return f"{module}/{module.rsplit('/', 1)[-1]}.{index}"
79
+
80
+
81
+ class MeteorNet:
82
+ """The network as fp32 torch functions over the ONNX weights (no autograd, no nn.Module state)."""
83
+
84
+ def __init__(self, weights: MeteorWeights, cfg: Optional[MeteorConfig] = None, *, check_graph: bool = True):
85
+ torch = _torch()
86
+ self.cfg = (cfg or default_config()).check()
87
+ self.weights = weights
88
+ if check_graph:
89
+ weights.check_graph()
90
+ f32 = lambda a: torch.from_numpy(np.array(a, dtype=np.float32)) # own writable copy
91
+ self._conv: Dict[str, Any] = {}
92
+ for module, p in weights.iter_convs():
93
+ pads = tuple(int(v) for v in p.pads)
94
+ if len(pads) != 4 or pads[:2] != pads[2:] or p.group != 1 or tuple(p.dilations) != (1, 1):
95
+ raise ValueError(f"{module}: unsupported conv attributes {p}")
96
+ self._conv[module] = (f32(p.weight), None if p.bias is None else f32(p.bias),
97
+ tuple(int(s) for s in p.strides), pads[:2])
98
+ self._gemm: Dict[str, Any] = {}
99
+ for module in weights.gemm_modules():
100
+ g = weights.gemm(module)
101
+ if g.trans_a or not g.trans_b or g.alpha != 1.0 or g.beta != 1.0:
102
+ raise ValueError(f"{module}: unsupported Gemm attributes")
103
+ self._gemm[module] = (f32(g.weight), f32(g.bias))
104
+ self._pool = {name: f32(weights.pool_matrix(name)) for name in
105
+ ("ego_stem_w", "ego_stem_h", "tl_w", "tl_h", "fused_pool_w", "fused_pool_h", "sem_w", "sem_h",
106
+ "det_pool_w", "det_pool_h", "fused_mean_w", "fused_mean_h", "e2e_mean_w", "e2e_mean_h")}
107
+ self._mha = {name: weights.mha(name) for name in ("ego_attn", "agent_attn")}
108
+ self._mha_t = {name: (f32(m.in_weight), f32(m.in_bias), f32(m.out_weight), f32(m.out_bias))
109
+ for name, m in self._mha.items()}
110
+ self._queries = {"ego": f32(weights.queries("ego")), "agent": f32(weights.queries("agent"))}
111
+ g, b, eps = weights.layer_norm()
112
+ self._ln = (f32(g), f32(b), float(eps))
113
+ self._risk_gate = f32(np.asarray(weights.risk_gate(), np.float32).reshape(1))
114
+ self._dec_gate = f32(np.asarray(weights.dec_gate(), np.float32).reshape(1))
115
+ self._ground = f32(weights.lift_ground_points()) # [1, 4, 100000]
116
+ zc = weights.zero_constants()
117
+ self._hist = f32(zc["hist"]) # [1, 96, 800, 500] zeros (baked history)
118
+ self._mask = f32(zc["mask"]) # [1, 1, 800, 500] zeros
119
+ self._paint = torch.tensor(weights.paint_classes(), dtype=torch.long)
120
+ self._depth_centres = f32(C.depth_bin_centres(self.cfg.depth_mean_bins)).reshape(1, 1, C.DEPTH_BINS, 1, 1)
121
+
122
+ # ---- primitives ------------------------------------------------------------------------------------------------
123
+ def conv(self, x, module: str, relu: bool = False):
124
+ F = _torch().nn.functional
125
+ w, b, stride, pad = self._conv[module]
126
+ y = F.conv2d(x, w, b, stride=stride, padding=pad)
127
+ return F.relu(y) if relu else y
128
+
129
+ def block(self, x, module: str):
130
+ """ConvBlock = (3x3 conv, BN (folded), ReLU) x 2: children ``.0`` and ``.3`` (S:318)."""
131
+ return self.conv(self.conv(x, child(module, 0), True), child(module, 3), True)
132
+
133
+ def down(self, x, module: str):
134
+ """stride-2 3x3 conv + ReLU (child ``.0``) then a ConvBlock (child ``.3``): the encoder steps of the EDs."""
135
+ return self.block(self.conv(x, child(module, 0), True), child(module, 3))
136
+
137
+ def gemm(self, x, module: str):
138
+ w, b = self._gemm[module]
139
+ return _torch().addmm(b, x, w.t())
140
+
141
+ @staticmethod
142
+ def resize(x, hw):
143
+ """ONNX Resize linear / half_pixel to a target size = F.interpolate(bilinear, align_corners=False)."""
144
+ F = _torch().nn.functional
145
+ return F.interpolate(x, size=tuple(int(v) for v in hw), mode="bilinear", align_corners=False)
146
+
147
+ def pool2(self, x, w_name: str, h_name: str):
148
+ """The exported adaptive pool / mean: (H_mat @ (x @ W_mat)) on the last two dims (S:356-366)."""
149
+ torch = _torch()
150
+ return torch.matmul(self._pool[h_name], torch.matmul(x, self._pool[w_name]))
151
+
152
+ @staticmethod
153
+ def nan_to_num(x):
154
+ """The graph's IsNaN / IsInf / Where chain (ONNX nodes 190-213 and 233-255): NaN -> 0, +inf -> 30,
155
+ -inf -> -30; the identity for finite values."""
156
+ torch = _torch()
157
+ x = torch.where(torch.isnan(x), torch.zeros_like(x), x)
158
+ x = torch.where(torch.isinf(x) & (x > 0), torch.full_like(x, 30.0), x)
159
+ return torch.where(torch.isinf(x) & (x < 0), torch.full_like(x, -30.0), x)
160
+
161
+ def mha_pool(self, name: str, keys, queries):
162
+ """``nn.MultiheadAttention(q=learned queries, k=v=keys)`` as exported (torch 2.1, need_weights=False):
163
+ q and k|v projections, q / sqrt(head_dim), softmax over the 400 keys, out-projection.
164
+ ``keys`` [L, 1, E] (sequence first), ``queries`` [1, Q, E] -> [1, Q, E]."""
165
+ torch = _torch()
166
+ m = self._mha[name]
167
+ w_in, b_in, w_out, b_out = self._mha_t[name]
168
+ e, h = m.embed_dim, m.num_heads
169
+ hd = e // h
170
+ q_in = queries.transpose(0, 1) # [Q, 1, E]
171
+ q = b_in[:e] + torch.matmul(q_in, w_in[:e].t()) # /Add_4 = bias + MatMul
172
+ kv = b_in[e:] + torch.matmul(keys, w_in[e:].t()) # [L, 1, 2E]
173
+ k, v = kv[..., :e], kv[..., e:]
174
+ nq, nk = q.shape[0], k.shape[0]
175
+ qh = q.reshape(nq, h, hd).transpose(0, 1) # [H, Q, hd]
176
+ kh = k.reshape(nk, h, hd).transpose(0, 1) # [H, L, hd]
177
+ vh = v.reshape(nk, h, hd).transpose(0, 1)
178
+ att = torch.softmax(torch.matmul(qh / m.scale_divisor, kh.transpose(1, 2)), dim=-1)
179
+ o = torch.matmul(att, vh).transpose(0, 1).reshape(nq, e) # [Q, E]
180
+ o = torch.addmm(b_out, o, w_out.t()) # out-projection (Gemm)
181
+ return o.reshape(nq, 1, e).transpose(0, 1) # [1, Q, E]
182
+
183
+ # ---- stages ----------------------------------------------------------------------------------------------------
184
+ def normalize(self, imgs_u8, present=None):
185
+ """uint8 [1, 8, 3, 432, 768] -> [8, 3, 432, 768] float32: /255 (``onnx``) or (x/255 - mean)/std with absent
186
+ cameras zeroed AFTER normalisation (``imagenet``, SPEC section 2 imgs row)."""
187
+ torch = _torch()
188
+ x = imgs_u8.to(torch.float32) / 255.0
189
+ if self.cfg.input_norm == "imagenet":
190
+ mean = torch.tensor(C.IMAGENET_MEAN, dtype=torch.float32).reshape(1, 1, 3, 1, 1)
191
+ std = torch.tensor(C.IMAGENET_STD, dtype=torch.float32).reshape(1, 1, 3, 1, 1)
192
+ x = (x - mean) / std
193
+ if present is not None:
194
+ keep = torch.as_tensor(np.asarray(present, np.float32)).reshape(1, -1, 1, 1, 1)
195
+ x = x * keep
196
+ return x.reshape(C.N_CAMS, 3, C.IMG_H, C.IMG_W)
197
+
198
+ def image_encoder(self, x, taps: TapRegistry = NULL_TAPS):
199
+ """ResNet-34 (torchvision layout, BN folded) + FPN -> f [8, 160, 108, 192] (S:301-302)."""
200
+ F = _torch().nn.functional
201
+ y = F.max_pool2d(self.conv(x, "stem/stem.0", True), kernel_size=3, stride=2, padding=1)
202
+ taps.tap("img.stem", y)
203
+ feats = []
204
+ for li, nb in enumerate((3, 4, 6, 3), start=1):
205
+ for bi in range(nb):
206
+ p = f"layer{li}/layer{li}.{bi}"
207
+ ds = f"{p}/downsample/downsample.0"
208
+ idt = self.conv(y, ds) if ds in self._conv else y
209
+ y = F.relu(self.conv(self.conv(y, f"{p}/conv1", True), f"{p}/conv2") + idt)
210
+ feats.append(taps.tap(f"img.layer{li}", y))
211
+ c1, c2, c3, c4 = feats
212
+ hw = (C.FEAT_H, C.FEAT_W)
213
+ s = self.conv(c1, "lat1") + self.resize(self.conv(c2, "lat2"), hw)
214
+ s = s + self.resize(self.conv(c3, "lat3"), hw)
215
+ s = s + self.resize(self.conv(c4, "lat4"), hw) # 14x24 -> 108x192: non-integer (S:302)
216
+ return taps.tap("img.fpn", self.conv(s, "fuse/fuse.0", True))
217
+
218
+ def seg2d_head(self, f, taps: TapRegistry = NULL_TAPS):
219
+ """SegHeadED -> raw logits [8, 21, 108, 192] (S:303)."""
220
+ d1 = self.down(f, "seg_head/d1")
221
+ d2 = self.down(d1, "seg_head/d2")
222
+ m1 = self.block(d1 + self.resize(self.conv(d2, "seg_head/u1"), d1.shape[-2:]), "seg_head/m1")
223
+ u2 = self.resize(self.conv(m1, "seg_head/u2"), f.shape[-2:])
224
+ y = self.conv(self.conv(self.conv(f, "seg_head/skip") + u2, "seg_head/out/out.0", True), "seg_head/out/out.3")
225
+ return taps.tap("seg2d.logits", y)
226
+
227
+ def depth_head(self, f, taps: TapRegistry = NULL_TAPS):
228
+ """4 ConvBlocks 160->128->128->96->64, 1x1 -> 64 bins, softmax over the bins (S:304)."""
229
+ torch = _torch()
230
+ y = f
231
+ for i in range(4):
232
+ y = self.block(y, f"depth_head/depth_head.{i}")
233
+ logits = taps.tap("depth.logits", self.conv(y, "depth_head/depth_head.4"))
234
+ return logits, taps.tap("depth.prob", torch.softmax(logits, dim=1))
235
+
236
+ def det2d_heads(self, f):
237
+ """det2d stem + d8 / d16 pyramid, 1x1 heads at strides 4 / 8 / 16 (S:305) -> the six [1, 8, k, h, w] outputs."""
238
+ s = self.block(f, "det2d_stem")
239
+ d8 = self.down(s, "det2d_d8")
240
+ d16 = self.down(d8, "det2d_d16")
241
+ out = {}
242
+ for i, (x, sfx) in enumerate(((s, ""), (d8, "8"), (d16, "16"))):
243
+ for kind, ch in (("hm2d", 10), ("reg2d", 4)):
244
+ y = self.conv(x, f"{kind}_head{sfx}")
245
+ out[f"{kind}_s{i}"] = y.reshape(1, C.N_CAMS, ch, y.shape[-2], y.shape[-1])
246
+ return out
247
+
248
+ def tl_head(self, f):
249
+ """cat(f[cam 0], f[cam 6]) -> s2 conv + CB -> s2 conv + CB -> global mean (MatMul consts) -> Linear (S:306)."""
250
+ torch = _torch()
251
+ x = torch.cat([f[0:1], f[6:7]], dim=1)
252
+ x = self.block(self.conv(x, "tl_head/tl_head.0", True), "tl_head/tl_head.3")
253
+ x = self.block(self.conv(x, "tl_head/tl_head.4", True), "tl_head/tl_head.7")
254
+ x = self.pool2(x, "tl_w", "tl_h").flatten(1)
255
+ return self.gemm(x, "tl_fc")
256
+
257
+ def context(self, f, seg_logits, taps: TapRegistry = NULL_TAPS):
258
+ """ctx 1x1 + paint_proj(softmax(nan_to_num(raw seg logits))[paint classes]) (S:307; UNCLAMPED logits)."""
259
+ torch = _torch()
260
+ p = torch.softmax(self.nan_to_num(seg_logits), dim=1).index_select(1, self._paint)
261
+ return taps.tap("ctx.painted", self.conv(f, "ctx") + self.conv(p, "paint_proj"))
262
+
263
+ def lift(self, ctx, prob, K, T_cam_ego, taps: TapRegistry = NULL_TAPS):
264
+ """Depth-gated IPM lift to the 0.4 m grid, then bilinear x2 to the raw BEV (S:333-351, ONNX nodes 261-421)."""
265
+ torch = _torch()
266
+ F = torch.nn.functional
267
+ n = C.N_CAMS
268
+ T = T_cam_ego.reshape(n, 4, 4).to(torch.float32)
269
+ Km = K.reshape(n, 3, 3).to(torch.float32)
270
+ pc = torch.matmul(T, self._ground) # [8, 4, 100000]
271
+ x, y, z = pc[:, 0], pc[:, 1], pc[:, 2]
272
+ zc = torch.clamp(z, min=C.LIFT_MIN_Z)
273
+ u = Km[:, 0, 0].unsqueeze(-1) * x / zc + Km[:, 0, 2].unsqueeze(-1)
274
+ v = Km[:, 1, 1].unsqueeze(-1) * y / zc + Km[:, 1, 2].unsqueeze(-1)
275
+ dist = torch.sqrt(x * x + y * y + z * z)
276
+ valid = ((z > C.LIFT_MIN_Z) & (u >= 0.0) & (u < float(C.IMG_W)) & (v >= 0.0) & (v < float(C.IMG_H))
277
+ & (dist < C.LIFT_MAX_DIST))
278
+ taps.tap("lift.valid", valid)
279
+ gu = torch.clamp(u / float(C.IMG_W - 1) * 2.0 - 1.0, -2.0, 2.0)
280
+ gv = torch.clamp(v / float(C.IMG_H - 1) * 2.0 - 1.0, -2.0, 2.0)
281
+ grid = taps.tap("lift.grid", torch.stack([gu, gv], dim=-1).unsqueeze(2)) # [8, 100000, 1, 2]
282
+ ctx_s = F.grid_sample(ctx, grid, mode="bilinear", padding_mode="zeros", align_corners=False).squeeze(3)
283
+ prob_s = F.grid_sample(prob, grid, mode="bilinear", padding_mode="zeros", align_corners=False).squeeze(3)
284
+ b = torch.clamp((dist - C.DEPTH_MIN) / C.DEPTH_STEP, 0.0, float(np.float32(C.LIFT_BIN_CLIP)))
285
+ b0 = torch.floor(b).to(torch.int64)
286
+ fr = (b - b0.to(torch.float32)).unsqueeze(1) # [8, 1, 100000]
287
+ g0 = torch.gather(prob_s, 1, b0.unsqueeze(1))
288
+ g1 = torch.gather(prob_s, 1, torch.clamp(b0 + 1, max=C.DEPTH_BINS - 1).unsqueeze(1))
289
+ w = (g0 * (1.0 - fr) + g1 * fr + C.LIFT_W_EPS) * valid.unsqueeze(1).to(torch.float32)
290
+ taps.tap("lift.weight", w)
291
+ wn = w.reshape(1, n, 1, -1)
292
+ num = (ctx_s.reshape(1, n, C.CTX_C, -1) * wn).sum(dim=1) # [1, 96, 100000]
293
+ den = torch.clamp(wn.sum(dim=1), min=C.LIFT_DEN_EPS) # [1, 1, 100000]
294
+ bev04 = taps.tap("lift.bev", (num / den).reshape(1, C.CTX_C, C.LIFT_H, C.LIFT_W))
295
+ return taps.tap("bev.raw", self.resize(bev04, (C.BEV_H, C.BEV_W)))
296
+
297
+ def temporal_fuse(self, raw, taps: TapRegistry = NULL_TAPS):
298
+ """No-history temporal fuse: fused = raw + tfuse(raw * 4 softmax(tgate(raw))[0]) (S:308)."""
299
+ torch = _torch()
300
+ gate = taps.tap("bev.gate", torch.softmax(self.conv(raw, "tgate_slim"), dim=1)[:, 0:1] * C.TGATE_SCALE)
301
+ t = self.block(self.conv(raw * gate, "tfuse3_slim/tfuse3_slim.0", True), "tfuse3_slim/tfuse3_slim.3")
302
+ return taps.tap("bev.fused", raw + t)
303
+
304
+ def lane_decoder(self, raw, taps: TapRegistry = NULL_TAPS):
305
+ """LaneDecED + lane_branch added to channels 4..6 -> lane_pre [1, 9, 800, 500] (S:309)."""
306
+ torch = _torch()
307
+ skip = self.block(raw, "dec/skip")
308
+ d1 = self.down(raw, "dec/d1")
309
+ d2 = self.down(d1, "dec/d2")
310
+ m1 = self.block(d1 + self.resize(self.conv(d2, "dec/u1"), d1.shape[-2:]), "dec/m1")
311
+ u2 = self.resize(self.conv(m1, "dec/u2"), skip.shape[-2:])
312
+ out = self.conv(self.conv(skip + u2, "dec/out/out.0", True), "dec/out/out.3")
313
+ lb = self.conv(self.conv(self.conv(raw, "lane_branch/lane_branch.0", True), "lane_branch/lane_branch.3", True),
314
+ "lane_branch/lane_branch.6")
315
+ return taps.tap("lane.pre", torch.cat([out[:, :4], out[:, 4:7] + lb, out[:, 7:]], dim=1))
316
+
317
+ def det_stem(self, raw, taps: TapRegistry = NULL_TAPS):
318
+ """_DetStemED on the raw BEV -> det_feat [1, 128, 400, 250], hm_pre, reg_pre (S:310)."""
319
+ s = self.down(raw, "det_stem/stem")
320
+ d8 = self.down(s, "det_stem/d8")
321
+ feat = self.conv(s + self.resize(self.conv(d8, "det_stem/u"), s.shape[-2:]), "det_stem/m/m.0", True)
322
+ taps.tap("det.feat", feat)
323
+ hm = taps.tap("det.hm_pre", self.conv(feat, "hm_head"))
324
+ return feat, hm, taps.tap("det.reg_pre", self.conv(feat, "reg_head"))
325
+
326
+ def occupancy(self, raw, taps: TapRegistry = NULL_TAPS):
327
+ """occ stem on the raw BEV crop rows 200:600, cols 50:450 -> 1x1 160 -> view [1, 10, 16, 200, 200] (S:311)."""
328
+ (r0, r1), (c0, c1) = C.OCC_CROP
329
+ x = self.down(raw[:, :, r0:r1, c0:c1], "occ_stem")
330
+ taps.tap("occ.feat", x)
331
+ y = self.conv(x, "occ_head")
332
+ return y.reshape(1, len(C.OCC_CLASSES), C.OCC_Z_BINS, y.shape[-2], y.shape[-1])
333
+
334
+ def agent_forecast(self, raw, fused, det_feat, reg_pre, taps: TapRegistry = NULL_TAPS):
335
+ """traj_stem on cat(fused, mot = (raw - hist) * mask = 0), agent attention pool, traj head (S:312)."""
336
+ torch = _torch()
337
+ mot = (raw - self._hist) * self._mask # the baked-out history: all zeros
338
+ stem = self.block(self.conv(torch.cat([fused, mot], dim=1), "traj_stem/traj_stem.0", True),
339
+ "traj_stem/traj_stem.3")
340
+ del mot
341
+ taps.tap("traj.stem", stem)
342
+ reg46 = reg_pre[:, 4:6]
343
+ keys = self.pool2(det_feat, "det_pool_w", "det_pool_h").reshape(1, det_feat.shape[1], -1).permute(2, 0, 1)
344
+ att = self.mha_pool("agent_attn", keys, self._queries["agent"]) # [1, 4, 128]
345
+ delta = self.conv(att.mean(dim=1).reshape(1, -1, 1, 1), "agent_delta")
346
+ taps.tap("traj.agent_delta", delta)
347
+ feat = taps.tap("traj.feat", torch.cat([stem, det_feat], dim=1) + delta)
348
+ return self.conv(torch.cat([feat, reg46], dim=1), "traj_head")
349
+
350
+ def stationary(self, raw, taps: TapRegistry = NULL_TAPS):
351
+ """delta_stat on avgpool2(|raw - 0|) (S:313)."""
352
+ F = _torch().nn.functional
353
+ x = F.avg_pool2d((raw - self._hist).abs(), kernel_size=2, stride=2)
354
+ x = self.conv(self.conv(x, "delta_stat/delta_stat.1", True), "delta_stat/delta_stat.4", True)
355
+ taps.tap("stat.feat", x)
356
+ return self.conv(x, "delta_stat/delta_stat.7")
357
+
358
+ def risk(self, fused, taps: TapRegistry = NULL_TAPS):
359
+ """risk head on the fused BEV crop rows 200:600, cols 125:375 (S:314)."""
360
+ (r0, r1), (c0, c1) = C.RISK_CROP
361
+ x = self.block(self.block(fused[:, :, r0:r1, c0:c1], "risk_head/risk_head.0"), "risk_head/risk_head.1")
362
+ taps.tap("risk.feat", x)
363
+ return self.conv(x, "risk_head/risk_head.2")
364
+
365
+ def planner(self, fused, lane_pre, hm_pre, risk, v0, taps: TapRegistry = NULL_TAPS):
366
+ """Ego planner up to the e2e refiner (S:315): ego_stem + MLP, + ego_attn, + sem_ego, + kin, risk-aware
367
+ mode logits, decoupled decoder. Returns ``(ego_after_dec [1, 42], fused_mean [1, 96])``."""
368
+ torch = _torch()
369
+ F = torch.nn.functional
370
+ with taps.scope("ego"):
371
+ e = self.block(self.conv(fused, "ego_stem/ego_stem.0", True), "ego_stem/ego_stem.3") # s4
372
+ e = self.block(self.conv(e, "ego_stem/ego_stem.4", True), "ego_stem/ego_stem.7")
373
+ e = self.block(self.conv(e, "ego_stem/ego_stem.8", True), "ego_stem/ego_stem.11")
374
+ e = self.conv(e, "ego_stem/ego_stem.12", True) # [1, 256, 25, 16]
375
+ e = taps.tap("stem", self.pool2(e, "ego_stem_w", "ego_stem_h").flatten(1))
376
+ v0 = v0.reshape(1, 1).to(torch.float32)
377
+ h = torch.cat([e, v0], dim=1)
378
+ for i in (0, 2, 4):
379
+ h = F.relu(self.gemm(h, f"ego_mlp/ego_mlp.{i}"))
380
+ ego = taps.tap("mlp", self.gemm(h, "ego_mlp/ego_mlp.6"))
381
+ keys = self.pool2(fused, "fused_pool_w", "fused_pool_h").reshape(1, fused.shape[1], -1).permute(2, 0, 1)
382
+ att = self.mha_pool("ego_attn", keys, self._queries["ego"]) # [1, 3, 96]
383
+ ego = taps.tap("plus_attn", ego + taps.tap("attn", self.gemm(att.flatten(1), "ego_delta")))
384
+ sem_in = torch.cat([F.avg_pool2d(torch.softmax(lane_pre, dim=1), kernel_size=2, stride=2),
385
+ torch.sigmoid(hm_pre)], dim=1)
386
+ taps.tap("sem_in", sem_in)
387
+ s = self.conv(self.conv(self.conv(sem_in, "sem_ego/sem_ego.0", True), "sem_ego/sem_ego.3", True),
388
+ "sem_ego/sem_ego.6", True)
389
+ s = self.pool2(s, "sem_w", "sem_h").flatten(1)
390
+ s = self.gemm(F.relu(self.gemm(s, "sem_ego/sem_ego.11")), "sem_ego/sem_ego.13")
391
+ ego = taps.tap("plus_sem", ego + taps.tap("sem", s))
392
+ kin_in = torch.zeros(1, C.KIN_DIM, dtype=torch.float32)
393
+ kin_in[0, 0] = v0.reshape(()) / C.V0_SCALE
394
+ ego = taps.tap("plus_kin", ego + taps.tap("kin", self.gemm(kin_in, "kin_delta")))
395
+ # risk-aware mode logits: sample sigmoid(risk) (border padding) at each path's 6 waypoints (S:315)
396
+ paths = ego[:, :36].reshape(1, C.EGO_MODES, C.EGO_STEPS, 2)
397
+ gx = -paths[..., 1] / C.RISK_Y_HALF
398
+ gy = (C.RISK_X_HALF - paths[..., 0]) / C.RISK_X_HALF - 1.0
399
+ samp = F.grid_sample(torch.sigmoid(risk), torch.stack([gx, gy], dim=-1), mode="bilinear",
400
+ padding_mode="border", align_corners=False) # [1, 1, 3, 6]
401
+ samp = taps.tap("risk_sample", samp.mean(dim=3).squeeze(1)) # [1, 3]
402
+ ego = torch.cat([ego[:, :36], ego[:, 36:39] - self._risk_gate * samp, ego[:, 39:]], dim=1)
403
+ taps.tap("after_risk", ego)
404
+ # decoupled decoder on mean(fused): heading phi and speed v = min(softplus, 25) -> cumulative waypoints
405
+ fmean = self.pool2(fused, "fused_mean_w", "fused_mean_h").flatten(1)
406
+ dh = taps.tap("dec_head", self.gemm(fmean, "dec_head")).reshape(1, C.EGO_MODES, 12)
407
+ phi, vr = dh[..., :6], dh[..., 6:]
408
+ sp = torch.clamp(vr, min=0.0) + torch.log(1.0 + torch.exp(-vr.abs()))
409
+ step = torch.clamp(sp, max=C.DEC_SPEED_MAX) * 0.5
410
+ wp = taps.tap("dec_wp", torch.stack([torch.cumsum(step * torch.cos(phi), dim=2),
411
+ torch.cumsum(step * torch.sin(phi), dim=2)], dim=-1))
412
+ e36 = ego[:, :36].reshape(1, C.EGO_MODES, C.EGO_STEPS, 2)
413
+ e36 = e36 + torch.tanh(self._dec_gate) * (wp - e36)
414
+ ego = taps.tap("after_dec", torch.cat([e36.reshape(1, 36), ego[:, 36:]], dim=1))
415
+ return ego, fmean
416
+
417
+ @staticmethod
418
+ def _range_channel(h: int, w: int):
419
+ """The refiners' constant range channel: row i -> (80 - (i + 0.5) * (160 / h)) / 80, float32 in graph order
420
+ (Range + 0.5, Reciprocal(h) * 160, Mul, 80 - ., / 80)."""
421
+ torch = _torch()
422
+ step = (1.0 / torch.tensor(float(h), dtype=torch.float32)) * 160.0
423
+ rows = (80.0 - (torch.arange(h, dtype=torch.float32) + 0.5) * step) / 80.0
424
+ return rows.reshape(1, 1, h, 1).expand(1, 1, h, w)
425
+
426
+ def seg_refiner(self, lane_pre, taps: TapRegistry = NULL_TAPS):
427
+ """BEVSegRefiner: lane_logit = clamp(lane_pre, +-20) + 8 tanh(res / 8) (S:316)."""
428
+ torch = _torch()
429
+ lc = torch.clamp(lane_pre, -C.SEG_REFINE_CLAMP, C.SEG_REFINE_CLAMP)
430
+ x = torch.cat([lc, self._range_channel(lc.shape[-2], lc.shape[-1])], dim=1)
431
+ p = "refiner/seg"
432
+ stem = self.conv(x, f"{p}/stem/stem.0", True)
433
+ d1 = self.down(stem, f"{p}/d1")
434
+ d2 = self.down(d1, f"{p}/d2")
435
+ d3 = self.down(d2, f"{p}/d3")
436
+ d4 = self.down(d3, f"{p}/d4")
437
+ m3 = self.block(d3 + self.resize(self.conv(d4, f"{p}/u4"), d3.shape[-2:]), f"{p}/m3")
438
+ m2 = self.block(d2 + self.resize(self.conv(m3, f"{p}/u3"), d2.shape[-2:]), f"{p}/m2")
439
+ m1 = self.block(d1 + self.resize(self.conv(m2, f"{p}/u2"), d1.shape[-2:]), f"{p}/m1")
440
+ o = stem + self.resize(self.conv(m1, f"{p}/u1"), stem.shape[-2:])
441
+ res = taps.tap("refiner.seg_res", self.conv(self.conv(o, f"{p}/out/out.0", True), f"{p}/out/out.3"))
442
+ return lc + torch.tanh(res / C.SEG_REFINE_SCALE) * C.SEG_REFINE_SCALE
443
+
444
+ def box_refiner(self, hm_pre, reg_pre, taps: TapRegistry = NULL_TAPS):
445
+ """BEVBoxRefiner: hm, reg = clamp(.) + 6 tanh(res / 6) (S:317)."""
446
+ torch = _torch()
447
+ hc = torch.clamp(hm_pre, -C.BOX_HM_CLAMP, C.BOX_HM_CLAMP)
448
+ rc = torch.clamp(reg_pre, -C.BOX_REG_CLAMP, C.BOX_REG_CLAMP)
449
+ x = torch.cat([hc, rc, self._range_channel(hc.shape[-2], hc.shape[-1])], dim=1)
450
+ p = "refiner/box"
451
+ stem = self.conv(x, f"{p}/stem/stem.0", True)
452
+ d1 = self.down(stem, f"{p}/d1")
453
+ d2 = self.down(d1, f"{p}/d2")
454
+ d3 = self.down(d2, f"{p}/d3")
455
+ m2 = self.block(d2 + self.resize(self.conv(d3, f"{p}/u3"), d2.shape[-2:]), f"{p}/m2")
456
+ m1 = self.block(d1 + self.resize(self.conv(m2, f"{p}/u2"), d1.shape[-2:]), f"{p}/m1")
457
+ o = stem + self.resize(self.conv(m1, f"{p}/u1"), stem.shape[-2:])
458
+ res = taps.tap("refiner.box_res", self.conv(self.conv(o, f"{p}/out/out.0", True), f"{p}/out/out.3"))
459
+ r = torch.tanh(res / C.BOX_REFINE_SCALE) * C.BOX_REFINE_SCALE
460
+ return hc + r[:, :2], rc + r[:, 2:]
461
+
462
+ def e2e_refiner(self, ego, v0, fused_mean, taps: TapRegistry = NULL_TAPS):
463
+ """E2ERefiner: ego[:36] += 3 tanh(mlp(LN([ego, v0, mean(fused)]))) (S:318; LN at the logical width 139)."""
464
+ torch = _torch()
465
+ F = torch.nn.functional
466
+ g, b, eps = self._ln
467
+ x = torch.cat([ego, v0.reshape(1, 1).to(torch.float32), fused_mean], dim=1)
468
+ h = taps.tap("refiner.e2e_ln", F.layer_norm(x, (x.shape[-1],), g, b, eps))
469
+ h = F.relu(self.gemm(h, "refiner/e2e/mlp/mlp.0"))
470
+ h = F.relu(self.gemm(h, "refiner/e2e/mlp/mlp.2"))
471
+ r = taps.tap("refiner.e2e_res", torch.tanh(self.gemm(h, "refiner/e2e/mlp/mlp.4")) * C.E2E_SCALE)
472
+ return torch.cat([ego[:, :36] + r, ego[:, 36:]], dim=1)
473
+
474
+ def depth_mean(self, depth_logits):
475
+ """avgpool2 of the 64 logits -> softmax -> argmax -> +-2-bin window -> sum p c / max(sum p, 1e-6) -> fp16
476
+ (ONNX nodes 1946-2014; bin centres per ``cfg.depth_mean_bins``)."""
477
+ torch = _torch()
478
+ F = torch.nn.functional
479
+ lp = F.avg_pool2d(depth_logits, kernel_size=2, stride=2)
480
+ p = torch.softmax(lp.reshape(1, C.N_CAMS, C.DEPTH_BINS, lp.shape[-2], lp.shape[-1]), dim=2)
481
+ am = torch.argmax(p, dim=2, keepdim=True)
482
+ r = torch.arange(C.DEPTH_BINS, dtype=torch.int64).reshape(1, 1, C.DEPTH_BINS, 1, 1)
483
+ pm = p * ((r - am).abs() <= 2).to(torch.float32)
484
+ num = (pm * self._depth_centres).sum(dim=2)
485
+ den = torch.clamp(pm.sum(dim=2), min=1e-6)
486
+ return (num / den).to(torch.float16)
487
+
488
+ # ---- whole graph ------------------------------------------------------------------------------------------------
489
+ def forward(self, imgs, K, T_cam_ego, v0, *, present=None, taps: TapRegistry = NULL_TAPS) -> Dict[str, Any]:
490
+ """The 19 outputs of the graph (torch tensors, ONNX dtypes) for one frame.
491
+
492
+ ``imgs`` uint8 [1, 8, 3, 432, 768] RGB, ``K`` [1, 8, 3, 3] at 768x432, ``T_cam_ego`` [1, 8, 4, 4], ``v0`` [1];
493
+ numpy or torch. ``present`` (8 flags) only matters for ``input_norm="imagenet"``."""
494
+ torch = _torch()
495
+ t = lambda a, dt: a.to(dt) if isinstance(a, torch.Tensor) else torch.from_numpy(np.ascontiguousarray(a)).to(dt)
496
+ imgs = t(imgs, torch.uint8)
497
+ K = t(K, torch.float32)
498
+ T = t(T_cam_ego, torch.float32)
499
+ v0 = t(v0, torch.float32).reshape(1)
500
+ out: Dict[str, Any] = {}
501
+ with torch.inference_mode():
502
+ x = taps.tap("img.input", self.normalize(imgs, present))
503
+ f = self.image_encoder(x, taps)
504
+ del x
505
+ seg = self.seg2d_head(f, taps)
506
+ seg_c = taps.tap("seg2d.clamped", torch.clamp(self.nan_to_num(seg), -C.SEG2D_CLAMP, C.SEG2D_CLAMP))
507
+ dlog, dprob = self.depth_head(f, taps)
508
+ ctx = self.context(f, seg, taps)
509
+ out.update(self.det2d_heads(f))
510
+ out["tl"] = self.tl_head(f)
511
+ del f
512
+ raw = self.lift(ctx, dprob, K, T, taps)
513
+ del ctx, dprob
514
+ fused = self.temporal_fuse(raw, taps)
515
+ lane_pre = self.lane_decoder(raw, taps)
516
+ det_feat, hm_pre, reg_pre = self.det_stem(raw, taps)
517
+ out["occ"] = self.occupancy(raw, taps)
518
+ out["traj"] = self.agent_forecast(raw, fused, det_feat, reg_pre, taps)
519
+ out["stationary"] = self.stationary(raw, taps)
520
+ del raw, det_feat
521
+ out["risk"] = self.risk(fused, taps)
522
+ ego, fmean = self.planner(fused, lane_pre, hm_pre, out["risk"], v0, taps)
523
+ taps.tap("bev.fused_mean", fmean)
524
+ out["lane_logit"] = self.seg_refiner(lane_pre, taps)
525
+ out["hm"], out["reg"] = self.box_refiner(hm_pre, reg_pre, taps)
526
+ e2e_mean = self.pool2(fused, "e2e_mean_w", "e2e_mean_h").flatten(1)
527
+ out["ego"] = self.e2e_refiner(ego, v0, e2e_mean, taps)
528
+ del fused
529
+ out["lane"] = torch.argmax(out["lane_logit"], dim=1).to(torch.uint8)
530
+ dl5 = dlog.reshape(1, C.N_CAMS, C.DEPTH_BINS, C.FEAT_H, C.FEAT_W)
531
+ out["depth"] = torch.argmax(dl5, dim=2).to(torch.uint8)
532
+ out["seg2d"] = torch.argmax(seg_c.reshape(1, C.N_CAMS, len(C.SEG2D_CLASSES), C.FEAT_H, C.FEAT_W),
533
+ dim=2).to(torch.uint8)
534
+ out["depth_mean"] = self.depth_mean(dlog)
535
+ for name in OUTPUT_NAMES:
536
+ taps.tap(name, out[name])
537
+ return {name: out[name] for name in OUTPUT_NAMES}
538
+
539
+ __call__ = forward
code/tt_meteor/reference/ort.py ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """ONNX Runtime oracle of the deployed graph: the 19 outputs plus any internal tensor as a named tap.
3
+
4
+ Only for tests and golden generation in the research venv (onnxruntime is not a runtime dependency of the bundle,
5
+ BUNDLE_CONVENTIONS.md section 13). The graph is modified in memory only (extra graph outputs; for
6
+ ``input_norm="imagenet"`` the training normalisation inserted after ``/Div``, the oracle of SPEC section 10 risk 1,
7
+ re-implemented from ``research/meteor/scripts/norm_ab_test.py``); the downloaded file is never written.
8
+ ORT on this graph is bit-reproducible at a fixed thread count (SPEC section 6).
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import os
13
+ from typing import Any, Dict, Iterable, Mapping, Optional, Sequence
14
+
15
+ import numpy as np
16
+
17
+ from . import config as C
18
+
19
+ __all__ = ["ORT_TAPS", "OrtOracle", "available"]
20
+
21
+ # reference tap name -> ONNX tensor name (shapes as in the graph; see reference/model.py)
22
+ ORT_TAPS: Dict[str, str] = {
23
+ "img.input": "/net/Reshape_output_0",
24
+ "img.stem": "/net/stem/stem.3/MaxPool_output_0",
25
+ "img.layer1": "/net/layer1/layer1.2/relu_1/Relu_output_0",
26
+ "img.layer2": "/net/layer2/layer2.3/relu_1/Relu_output_0",
27
+ "img.layer3": "/net/layer3/layer3.5/relu_1/Relu_output_0",
28
+ "img.layer4": "/net/layer4/layer4.2/relu_1/Relu_output_0",
29
+ "img.fpn": "/net/fuse/fuse.2/Relu_output_0",
30
+ "seg2d.logits": "/net/seg_head/out/out.3/Conv_output_0",
31
+ "seg2d.clamped": "/net/Clip_output_0",
32
+ "depth.logits": "/net/depth_head/depth_head.4/Conv_output_0",
33
+ "depth.prob": "/net/Softmax_output_0",
34
+ "ctx.painted": "/net/Add_3_output_0",
35
+ "lift.grid": "/net/Unsqueeze_6_output_0",
36
+ "lift.valid": "/net/And_8_output_0",
37
+ "lift.weight": "/net/Mul_10_output_0",
38
+ "lift.bev": "/net/Reshape_5_output_0",
39
+ "bev.raw": "/net/Resize_3_output_0",
40
+ "bev.gate": "/net/Mul_12_output_0",
41
+ "bev.fused": "/net/Add_11_output_0",
42
+ "lane.pre": "/net/Concat_7_output_0",
43
+ "det.feat": "/net/det_stem/m/m.2/Relu_output_0",
44
+ "det.hm_pre": "/net/hm_head/Conv_output_0",
45
+ "det.reg_pre": "/net/reg_head/Conv_output_0",
46
+ "occ.feat": "/net/occ_stem/occ_stem.3/occ_stem.3.5/Relu_output_0",
47
+ "traj.stem": "/net/traj_stem/traj_stem.3/traj_stem.3.5/Relu_output_0",
48
+ "traj.agent_delta": "/net/agent_delta/Conv_output_0",
49
+ "traj.feat": "/net/Add_15_output_0",
50
+ "stat.feat": "/net/delta_stat/delta_stat.6/Relu_output_0",
51
+ "risk.feat": "/net/risk_head/risk_head.1/risk_head.1.5/Relu_output_0",
52
+ "ego.stem": "/net/Flatten_output_0",
53
+ "ego.mlp": "/net/ego_mlp/ego_mlp.6/Gemm_output_0",
54
+ "ego.attn": "/net/ego_delta/Gemm_output_0",
55
+ "ego.plus_attn": "/net/Add_13_output_0",
56
+ "ego.sem_in": "/net/Concat_23_output_0",
57
+ "ego.sem": "/net/sem_ego/sem_ego.13/Gemm_output_0",
58
+ "ego.plus_sem": "/net/Add_14_output_0",
59
+ "ego.kin": "/net/kin_delta/Gemm_output_0",
60
+ "ego.plus_kin": "/net/Add_17_output_0",
61
+ "ego.risk_sample": "/net/Squeeze_2_output_0",
62
+ "ego.after_risk": "/net/ScatterND_1_output_0",
63
+ "ego.dec_head": "/net/dec_head/Gemm_output_0",
64
+ "ego.dec_wp": "/net/Concat_36_output_0",
65
+ "ego.after_dec": "/net/ScatterND_2_output_0",
66
+ "bev.fused_mean": "/net/Flatten_3_output_0",
67
+ "refiner.seg_res": "/net/refiner/seg/out/out.3/Conv_output_0",
68
+ "refiner.box_res": "/net/refiner/box/out/out.3/Conv_output_0",
69
+ "refiner.e2e_ln": "/net/refiner/e2e/norm/LayerNormalization_output_0",
70
+ "refiner.e2e_res": "/net/refiner/e2e/Mul_output_0",
71
+ }
72
+
73
+
74
+ def available() -> bool:
75
+ try:
76
+ import onnxruntime # noqa: F401 (probe only)
77
+ except ImportError:
78
+ return False
79
+ return True
80
+
81
+
82
+ def _insert_imagenet(model) -> None:
83
+ """(x / 255 - mean) / std right after ``/Div`` (in memory; SPEC section 10 risk 1)."""
84
+ from onnx import helper, numpy_helper
85
+
86
+ g = model.graph
87
+ div = next(n for n in g.node if n.name == "/Div")
88
+ out = div.output[0]
89
+ mean = np.array(C.IMAGENET_MEAN, np.float32).reshape(1, 1, 3, 1, 1)
90
+ std = np.array(C.IMAGENET_STD, np.float32).reshape(1, 1, 3, 1, 1)
91
+ g.initializer.extend([numpy_helper.from_array(mean, "tt_meteor_mean"),
92
+ numpy_helper.from_array(std, "tt_meteor_std")])
93
+ for n in g.node:
94
+ if n is not div:
95
+ for i, x in enumerate(n.input):
96
+ if x == out:
97
+ n.input[i] = "tt_meteor_norm"
98
+ idx = list(g.node).index(div)
99
+ g.node.insert(idx + 1, helper.make_node("Sub", [out, "tt_meteor_mean"], ["tt_meteor_sub"], name="/tt_meteor/Sub"))
100
+ g.node.insert(idx + 2, helper.make_node("Div", ["tt_meteor_sub", "tt_meteor_std"], ["tt_meteor_norm"],
101
+ name="/tt_meteor/Div"))
102
+
103
+
104
+ class OrtOracle:
105
+ """``OrtOracle(onnx_path, taps=[...], threads=4)``; ``oracle(feed) -> {name: array}`` with the 19 outputs under
106
+ their ONNX names and the requested taps under their reference names (``ORT_TAPS``; raw ONNX tensor names are
107
+ accepted too)."""
108
+
109
+ def __init__(self, onnx_path: os.PathLike, taps: Iterable[str] = (), *, threads: int = 4,
110
+ input_norm: str = "onnx"):
111
+ import onnx
112
+ import onnxruntime as ort
113
+
114
+ self.taps = {t: ORT_TAPS.get(t, t) for t in taps}
115
+ model = onnx.load(str(onnx_path))
116
+ if input_norm == "imagenet":
117
+ _insert_imagenet(model)
118
+ elif input_norm != "onnx":
119
+ raise ValueError(f"input_norm {input_norm!r}")
120
+ have = {o.name for o in model.graph.output}
121
+ for name in self.taps.values():
122
+ if name not in have:
123
+ model.graph.output.append(onnx.helper.make_empty_tensor_value_info(name))
124
+ so = ort.SessionOptions()
125
+ so.intra_op_num_threads = int(threads)
126
+ so.inter_op_num_threads = 1
127
+ self.session = ort.InferenceSession(model.SerializeToString(), sess_options=so,
128
+ providers=["CPUExecutionProvider"])
129
+ del model
130
+ self.output_names = [o.name for o in self.session.get_outputs()]
131
+ self.version = ort.__version__
132
+ self.threads = int(threads)
133
+
134
+ def __call__(self, feed: Mapping[str, np.ndarray], *, outputs: Optional[Sequence[str]] = None) -> Dict[str, Any]:
135
+ inv = {v: k for k, v in self.taps.items()}
136
+ feed = {k: np.asarray(feed[k]) for k in ("imgs", "K", "T_cam_ego", "v0")}
137
+ res = self.session.run(outputs, feed)
138
+ names = list(outputs) if outputs is not None else self.output_names
139
+ return {inv.get(n, n): v for n, v in zip(names, res)}
code/tt_meteor/reference/pipeline.py ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """End-to-end fp32 CPU reference of METEOR's runtime on the released graph: the oracle of every TT gate.
3
+
4
+ Data flow per frame (SPEC sections 2-5):
5
+
6
+ eight cameras + calibration + ego speed -> host.inputs / host.preprocess -> MeteorFrame (the ONNX feed)
7
+ -> reference.model.MeteorNet (torch fp32, per-module taps) -> the 19 outputs
8
+ -> host.result.build_output (3D / 2D decode, NMS, stationary, futures, unk2d, plan; per-stream seg fusion,
9
+ yaw smoothing, mode hysteresis) -> host.outputs.MeteorOutput (the /predict JSON)
10
+
11
+ :meth:`MeteorReference.run_frame` is the network alone (a ``MeteorFrame`` -> numpy outputs, optional taps: what the
12
+ goldens hold). :meth:`MeteorReference.__call__` takes the same inputs as ``api.METEOR.__call__`` (``images``,
13
+ ``calibration``, ``ego_speed``, ``stream``, runtime params) and keeps the host state per stream id, so the served
14
+ model and this reference can be compared request by request.
15
+
16
+ torch is imported when the reference is built. ``threads`` caps torch's intra-op threads (the workspace rule: <= 4).
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import time
21
+ from typing import Any, Dict, Mapping, Optional
22
+
23
+ import numpy as np
24
+
25
+ from ..host.inputs import prepare_request
26
+ from ..host.outputs import MeteorOutput
27
+ from ..host.postprocess import PostConfig
28
+ from ..host.preprocess import MeteorFrame
29
+ from ..host.result import build_output
30
+ from ..host.temporal import StreamState
31
+ from ..ttaw.golden import NULL_TAPS, TapRegistry
32
+ from .config import MeteorConfig, default_config
33
+ from .model import OUTPUT_NAMES, MeteorNet
34
+ from .weights import MeteorWeights
35
+
36
+ __all__ = ["MeteorReference", "to_numpy_outputs"]
37
+
38
+
39
+ def to_numpy_outputs(out: Mapping[str, Any]) -> Dict[str, np.ndarray]:
40
+ return {k: (v.numpy() if hasattr(v, "numpy") else np.asarray(v)) for k, v in out.items()}
41
+
42
+
43
+ class MeteorReference:
44
+ """``MeteorReference(weights_dir=None, cfg=None, threads=4)``: weights (``reference.weights``), the torch network
45
+ and per-stream host state."""
46
+
47
+ def __init__(self, weights_dir: Optional[Any] = None, cfg: Optional[MeteorConfig] = None, *,
48
+ threads: Optional[int] = 4, post: PostConfig = PostConfig(), weights: Optional[MeteorWeights] = None):
49
+ import torch
50
+
51
+ if threads:
52
+ torch.set_num_threads(int(threads))
53
+ self.cfg = (cfg or default_config()).check()
54
+ self.weights = weights or MeteorWeights(weights_dir)
55
+ self.net = MeteorNet(self.weights, self.cfg)
56
+ self.post = post
57
+ self.streams: Dict[str, StreamState] = {}
58
+
59
+ def run_frame(self, frame: MeteorFrame, taps: TapRegistry = NULL_TAPS) -> Dict[str, np.ndarray]:
60
+ """The 19 outputs (numpy, ONNX names / dtypes / shapes) of one frame."""
61
+ out = self.net(frame.imgs, frame.K, frame.T_cam_ego, frame.v0, present=frame.present, taps=taps)
62
+ return to_numpy_outputs({k: out[k] for k in OUTPUT_NAMES})
63
+
64
+ def stream(self, stream: Optional[Mapping[str, Any]]) -> StreamState:
65
+ sid = str((stream or {}).get("id", "default"))
66
+ st = self.streams.get(sid)
67
+ if st is None:
68
+ st = self.streams[sid] = StreamState()
69
+ elif (stream or {}).get("reset"):
70
+ st.reset()
71
+ return st
72
+
73
+ def __call__(self, images: Any = None, *, calibration: Optional[Mapping[str, Any]] = None, ego_speed: Any = None,
74
+ stream: Optional[Mapping[str, Any]] = None, output_heads: bool = False, **params: Any) -> MeteorOutput:
75
+ t0 = time.perf_counter()
76
+ frame = prepare_request(images, calibration, ego_speed, stream)
77
+ t1 = time.perf_counter()
78
+ outputs = self.run_frame(frame)
79
+ t2 = time.perf_counter()
80
+ cfg = self.post.with_params(**params)
81
+ res = build_output(outputs, frame, cfg, self.stream(stream), heads=output_heads)
82
+ t3 = time.perf_counter()
83
+ res.timing_ms.update({"preprocess": 1e3 * (t1 - t0), "device": 1e3 * (t2 - t1), "postprocess": 1e3 * (t3 - t2),
84
+ "total": 1e3 * (t3 - t0)})
85
+ res.meta["backend"] = "cpu-reference"
86
+ return res
code/tt_meteor/reference/port_form.py ADDED
@@ -0,0 +1,280 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """The exact rewrites the TT port applies to the deployed graph (PLAN.md 2.13; SPEC sections 3 and 8.4), with torch
3
+ emulations proven against the literal graph (``tests/test_reference_cpu.py::test_port_form_rewrites``, teacher-forced
4
+ from the ONNX Runtime taps). Each is exact in real arithmetic; in float32 they differ from the graph only by
5
+ rounding (measured max |difference| is recorded in PORT_LOG.md section 3).
6
+
7
+ ======================== ===========================================================================================
8
+ rewrite what changes
9
+ ======================== ===========================================================================================
10
+ ``stem_input_fold`` the input normalisation folds into the stem conv: ``onnx``: W / 255 on raw uint8 values
11
+ (zero padding is preserved under scaling); ``imagenet``: W / (255 std_c) on
12
+ ``x_u8 - 255 mean_c`` (the mean stays explicit: zero padding of the normalised image is
13
+ ``x_u8 = 255 mean_c``, so it cannot become a bias; SPEC section 3 item 3)
14
+ ``tgate_tfuse_merged`` tgate (96 -> 4) and tfuse.0 (96 -> 96) as one 96 -> 100 1x1 conv: W (s x) = s (W x) for
15
+ the per-pixel gate s = 4 softmax(tgate)[0], so the gate scales the conv output before its
16
+ bias and ReLU (SPEC 8.4)
17
+ ``merged_heads`` hm_head + reg_head = one 1x1 128 -> 8; hm2d + reg2d per stride = one 1x1 -> 14
18
+ ``paint_conv21`` the paint gather (seg classes 2..8, 13) + paint_proj 8 -> 96 = one 1x1 21 -> 96 conv on
19
+ softmax(raw seg logits) with zero columns for the unpainted classes
20
+ ``traj_stem_live`` traj_stem.0 on cat(fused, mot = 0) = the conv of its first 96 input channels on fused
21
+ ``lift_from_tables`` the lift with the calibration-only geometry from ``host.calib`` (grid, validity and the
22
+ dense bin-interpolation table): w = valid (0.05 + sum_b prob_s[b] table[b]); the device's
23
+ functional fallback (per-camera ``grid_sample`` + a bin contraction + the camera sum)
24
+ ``lane_pre_merged`` dec.out.3 (64 -> 9) and lane_branch.6 (64 -> 3, added to channels 4..6) as one 1x1 conv over
25
+ cat(dec.out.0 output, lane_branch.3 output) (a K-split conv, rows 4..6 of the second half)
26
+ ``traj_head_fold`` traj_head on cat(traj.feat, reg_pre[4:6]) with reg_pre = reg_head(det.feat): the reg path
27
+ folds into a 1x1 conv on det.feat (W_t[:, 256:] W_reg[4:6]; bias W_t[:, 256:] b_reg[4:6])
28
+ ``mha_fold`` the attention pools with learned (constant) queries: q is a constant, so the key projection
29
+ folds into the scores (S = G^T keys^T + c), the value projection, the out-projection and the
30
+ consumer (ego_delta Gemm / mean over queries + agent_delta conv) fold into one matrix on the
31
+ attention-weighted keys Z = softmax(S) keys (sum_k softmax = 1 moves every bias out)
32
+ ``range_bias_map`` the refiners' constant range channel through their stem conv (+ the conv bias) as a constant
33
+ per-pixel bias map: stem(cat(x, range)) = conv(x; W[:, :-1]) + B (zero padding included)
34
+ ``sem_fc_nhwc`` sem_ego.11 with its 1280 inputs re-ordered from (c, i, j) to (i, j, c) (the device's NHWC
35
+ flatten of the 5 x 4 pooled map)
36
+ ======================== ===========================================================================================
37
+ """
38
+ from __future__ import annotations
39
+
40
+ from dataclasses import dataclass
41
+ from typing import Dict, Tuple
42
+
43
+ import numpy as np
44
+
45
+ from . import config as C
46
+ from .weights import MeteorWeights
47
+
48
+ __all__ = ["stem_input_fold", "tgate_tfuse_merged", "merged_heads", "paint_conv21", "traj_stem_live",
49
+ "fused_from_merged", "lift_from_tables", "lane_pre_merged", "traj_head_fold", "MhaFold", "mha_fold",
50
+ "mha_from_fold", "range_channel", "range_bias_map", "sem_fc_nhwc", "REFINER_STEMS"]
51
+
52
+ F32 = np.float32
53
+
54
+
55
+ def stem_input_fold(w: MeteorWeights, input_norm: str = "onnx") -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
56
+ """(W', b, offset): ``conv(normalised x, W) == conv(x_u8 - offset_c, W')`` with W' = W / (255 std_c), offset =
57
+ 255 mean_c (``imagenet``) or W / 255, offset 0 (``onnx``). Folded in float64, rounded once."""
58
+ p = w.conv("stem/stem.0")
59
+ wt = np.asarray(p.weight, np.float64)
60
+ if input_norm == "onnx":
61
+ scale = np.full(3, 255.0)
62
+ offset = np.zeros(3)
63
+ elif input_norm == "imagenet":
64
+ scale = 255.0 * np.asarray(C.IMAGENET_STD, np.float64)
65
+ offset = 255.0 * np.asarray(C.IMAGENET_MEAN, np.float64)
66
+ else:
67
+ raise ValueError(input_norm)
68
+ return (wt / scale.reshape(1, 3, 1, 1)).astype(F32), np.asarray(p.bias, F32), offset.astype(F32)
69
+
70
+
71
+ def tgate_tfuse_merged(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
72
+ """(W [100, 96, 1, 1], b_gate [4], b_tfuse [96]): rows 0..3 = tgate, rows 4..99 = tfuse3_slim.0 (no bias: it is
73
+ added after the gate scaling)."""
74
+ g, t = w.conv("tgate_slim"), w.conv("tfuse3_slim/tfuse3_slim.0")
75
+ return (np.concatenate([g.weight, t.weight], axis=0).astype(F32), np.asarray(g.bias, F32),
76
+ np.asarray(t.bias, F32))
77
+
78
+
79
+ def fused_from_merged(conv_out, b_gate, b_tfuse):
80
+ """The merged conv's output [1, 100, H, W] -> (gate [1, 1, H, W], relu(gate * W_t x + b_t) [1, 96, H, W])."""
81
+ import torch
82
+
83
+ gate = torch.softmax(conv_out[:, :4] + torch.as_tensor(b_gate).reshape(1, 4, 1, 1), dim=1)[:, 0:1] * C.TGATE_SCALE
84
+ t0 = torch.relu(gate * conv_out[:, 4:] + torch.as_tensor(b_tfuse).reshape(1, -1, 1, 1))
85
+ return gate, t0
86
+
87
+
88
+ def merged_heads(w: MeteorWeights) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
89
+ """``det``: hm (2) | reg (6) on det_feat; ``det2d_s0/s1/s2``: hm2d (10) | reg2d (4) per stride."""
90
+ out = {}
91
+ for name, parts in (("det", ("hm_head", "reg_head")), ("det2d_s0", ("hm2d_head", "reg2d_head")),
92
+ ("det2d_s1", ("hm2d_head8", "reg2d_head8")), ("det2d_s2", ("hm2d_head16", "reg2d_head16"))):
93
+ ps = [w.conv(m) for m in parts]
94
+ out[name] = (np.concatenate([p.weight for p in ps], 0).astype(F32),
95
+ np.concatenate([p.bias for p in ps], 0).astype(F32))
96
+ return out
97
+
98
+
99
+ def paint_conv21(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray]:
100
+ """paint_proj [96, 8, 1, 1] scattered into [96, 21, 1, 1] at the painted seg classes (zeros elsewhere)."""
101
+ p = w.conv("paint_proj")
102
+ full = np.zeros((p.weight.shape[0], len(C.SEG2D_CLASSES), 1, 1), F32)
103
+ full[:, list(w.paint_classes())] = p.weight
104
+ return full, np.asarray(p.bias, F32)
105
+
106
+
107
+ def traj_stem_live(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray]:
108
+ """traj_stem.0 restricted to the fused-BEV input channels (the 96 ``mot`` channels are the baked-out zeros)."""
109
+ p = w.conv("traj_stem/traj_stem.0")
110
+ return np.ascontiguousarray(p.weight[:, :C.BEV_C]).astype(F32), np.asarray(p.bias, F32)
111
+
112
+
113
+ def lift_from_tables(ctx, prob, geom) -> "object":
114
+ """The lift with host tables (``host.calib.LiftGeometry``): ctx [8, 96, 108, 192], prob [8, 64, 108, 192] ->
115
+ lift.bev [1, 96, 400, 250]. Per camera: ``grid_sample`` of ctx and prob at the table grid (bilinear, zeros),
116
+ w = valid (0.05 + sum_b prob_s table), camera sums, the clamped division."""
117
+ import torch
118
+ F = torch.nn.functional
119
+
120
+ grid = torch.from_numpy(np.ascontiguousarray(geom.grid)).unsqueeze(2) # [8, N, 1, 2]
121
+ ctx_s = F.grid_sample(ctx, grid, mode="bilinear", padding_mode="zeros", align_corners=False).squeeze(3)
122
+ prob_s = F.grid_sample(prob, grid, mode="bilinear", padding_mode="zeros", align_corners=False).squeeze(3)
123
+ table = torch.from_numpy(geom.bin_weights()).permute(0, 2, 1) # [8, 64, N]
124
+ valid = torch.from_numpy(geom.valid.astype(F32)).unsqueeze(1) # [8, 1, N]
125
+ w_ = valid * ((prob_s * table).sum(dim=1, keepdim=True) + C.LIFT_W_EPS) # [8, 1, N]
126
+ num = (ctx_s * w_).sum(dim=0, keepdim=True) # [1, 96, N]
127
+ den = torch.clamp(w_.sum(dim=0, keepdim=True), min=C.LIFT_DEN_EPS)
128
+ return (num / den).reshape(1, C.CTX_C, C.LIFT_H, C.LIFT_W)
129
+
130
+
131
+ # ---- rewrites of the device graph (M2+) -----------------------------------------------------------------------------
132
+
133
+ def lane_pre_merged(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray]:
134
+ """(W [9, 128, 1, 1], b [9]): ``lane_pre = conv1x1(cat(x_dec, x_lb); W) + b`` with x_dec the dec.out.0 output and
135
+ x_lb the lane_branch.3 output (64 channels each); rows 4..6 of the second half carry lane_branch.6, the bias is
136
+ dec.out.3's plus lane_branch.6's on channels 4..6 (``model.lane_decoder``: out[:, 4:7] + lb)."""
137
+ o, lb = w.conv("dec/out/out.3"), w.conv("lane_branch/lane_branch.6")
138
+ c, k = o.weight.shape[0], o.weight.shape[1]
139
+ wt = np.zeros((c, k + lb.weight.shape[1], 1, 1), F32)
140
+ wt[:, :k] = o.weight
141
+ wt[4:7, k:] = lb.weight
142
+ b = np.asarray(o.bias, F32).copy()
143
+ b[4:7] = b[4:7] + np.asarray(lb.bias, F32)
144
+ return wt, b
145
+
146
+
147
+ def traj_head_fold(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
148
+ """(W_feat [39, 256, 1, 1], W_det [39, 128, 1, 1], b [39]): ``traj = conv(feat; W_feat) + conv(det_feat; W_det) +
149
+ b`` equals traj_head(cat(feat, reg_pre[:, 4:6])) with reg_pre = reg_head(det_feat) (both 1x1: the reg path is
150
+ linear in det_feat). Folded in float64, rounded once."""
151
+ t, r = w.conv("traj_head"), w.conv("reg_head")
152
+ wt = np.asarray(t.weight, np.float64)[:, :, 0, 0] # [39, 258]
153
+ nf = wt.shape[1] - 2
154
+ w2 = wt[:, nf:] # [39, 2]
155
+ wr = np.asarray(r.weight, np.float64)[4:6, :, 0, 0] # [2, 128]
156
+ w_det = w2 @ wr
157
+ b = np.asarray(t.bias, np.float64) + w2 @ np.asarray(r.bias, np.float64)[4:6]
158
+ return (wt[:, :nf, None, None].astype(F32), w_det[:, :, None, None].astype(F32), b.astype(F32))
159
+
160
+
161
+ @dataclass(frozen=True)
162
+ class MhaFold:
163
+ """An attention pool with constant queries in folded form: for keys K [L, E] (one token per row),
164
+ ``S = G^T K^T + c`` [R, L] with R = queries x heads (row r = query i * heads + head h), ``att = softmax(S)`` over
165
+ the keys, ``Z = att K`` [R, E], and the consumer output ``vec(Z) V + d0`` ([R * E] x [R * E, out])."""
166
+
167
+ G: np.ndarray # [E, R] float32
168
+ c: np.ndarray # [R] float32
169
+ V: np.ndarray # [R * E, out] float32
170
+ d0: np.ndarray # [out] float32
171
+ queries: int
172
+ heads: int
173
+
174
+ @property
175
+ def rows(self) -> int:
176
+ return self.queries * self.heads
177
+
178
+ @property
179
+ def embed(self) -> int:
180
+ return int(self.G.shape[0])
181
+
182
+
183
+ def mha_fold(w: MeteorWeights, name: str) -> MhaFold:
184
+ """``ego_attn`` (3 queries, consumer ``ego_delta`` Gemm 288 -> 42 on the flattened [3, 96] output) or
185
+ ``agent_attn`` (4 queries, mean over the queries, consumer ``agent_delta`` 1x1 conv 128 -> 256): see
186
+ :class:`MhaFold`. ``model.mha_pool``: q = b_q + queries W_q^T, k|v = b_kv + keys W_kv^T, per head
187
+ softmax((q / div) k^T) v, out = b_o + o W_o^T. With q constant, the scores are keys . (W_k,h^T q_i,h / div) +
188
+ q_i,h . b_k,h / div; since the attention weights sum to 1, o_i,h = W_v,h Z_i,h + b_v,h and the consumer is linear,
189
+ so everything after the softmax is one matrix on Z. Float64, rounded once."""
190
+ m = w.mha(name)
191
+ qn = {"ego_attn": "ego", "agent_attn": "agent"}[name]
192
+ queries = np.asarray(w.queries(qn), np.float64)[0] # [Q, E]
193
+ e, h = m.embed_dim, m.num_heads
194
+ hd = e // h
195
+ nq = queries.shape[0]
196
+ win = np.asarray(m.in_weight, np.float64)
197
+ bin_ = np.asarray(m.in_bias, np.float64)
198
+ q = queries @ win[:e].T + bin_[:e] # [Q, E]
199
+ wk, bk, wv, bv = win[e:2 * e], bin_[e:2 * e], win[2 * e:], bin_[2 * e:]
200
+ wo, bo = np.asarray(m.out_weight, np.float64), np.asarray(m.out_bias, np.float64)
201
+ div = float(np.float32(m.scale_divisor))
202
+ rows = nq * h
203
+ G = np.zeros((e, rows))
204
+ c = np.zeros(rows)
205
+ U = []
206
+ for hh in range(h):
207
+ sl = slice(hh * hd, (hh + 1) * hd)
208
+ U.append(wo[:, sl] @ wv[sl]) # [E, E]: out-proj of head hh's values
209
+ const_o = bo + wo @ bv # attention output bias of every query
210
+ for i in range(nq):
211
+ for hh in range(h):
212
+ sl = slice(hh * hd, (hh + 1) * hd)
213
+ qs = q[i, sl] / div
214
+ G[:, i * h + hh] = wk[sl].T @ qs
215
+ c[i * h + hh] = qs @ bk[sl]
216
+ if name == "ego_attn":
217
+ g = w.gemm("ego_delta")
218
+ wd, bd = np.asarray(g.weight, np.float64), np.asarray(g.bias, np.float64) # [42, Q * E]
219
+ mats = [[wd[:, i * e:(i + 1) * e] @ U[hh] for hh in range(h)] for i in range(nq)]
220
+ d0 = bd + sum(wd[:, i * e:(i + 1) * e] @ const_o for i in range(nq))
221
+ else:
222
+ p = w.conv("agent_delta")
223
+ wa = np.asarray(p.weight, np.float64)[:, :, 0, 0] # [256, 128]
224
+ mats = [[wa @ U[hh] / nq for hh in range(h)] for _ in range(nq)]
225
+ d0 = np.asarray(p.bias, np.float64) + wa @ const_o
226
+ out = mats[0][0].shape[0]
227
+ V = np.zeros((rows * e, out))
228
+ for i in range(nq):
229
+ for hh in range(h):
230
+ r = i * h + hh
231
+ V[r * e:(r + 1) * e] = mats[i][hh].T
232
+ return MhaFold(G.astype(F32), c.astype(F32), V.astype(F32), d0.astype(F32), nq, h)
233
+
234
+
235
+ def mha_from_fold(fold: MhaFold, keys):
236
+ """torch emulation of the folded attention pool: keys [L, E] -> the consumer output [out] (CPU proofs)."""
237
+ import torch
238
+
239
+ k = torch.as_tensor(keys, dtype=torch.float32)
240
+ s = torch.as_tensor(fold.G).T @ k.T + torch.as_tensor(fold.c)[:, None] # [R, L]
241
+ z = torch.softmax(s, dim=-1) @ k # [R, E]
242
+ return z.reshape(-1) @ torch.as_tensor(fold.V) + torch.as_tensor(fold.d0)
243
+
244
+
245
+ # the refiners' stems: module -> (input channels without the range channel, output channels)
246
+ REFINER_STEMS = {"refiner/seg/stem/stem.0": (9, 48), "refiner/box/stem/stem.0": (8, 32)}
247
+
248
+
249
+ def range_channel(h: int, w_: int) -> np.ndarray:
250
+ """The refiners' constant range channel [h, w] in float32, in the graph's operation order
251
+ (``model.MeteorNet._range_channel``)."""
252
+ from .model import MeteorNet
253
+
254
+ return MeteorNet._range_channel(h, w_).reshape(h, w_).numpy().astype(F32)
255
+
256
+
257
+ def range_bias_map(w: MeteorWeights, module: str, h: int, w_: int) -> np.ndarray:
258
+ """[h * w, Cout] float32 (NHWC rows): the stem conv of the constant range channel (its last input channel, zero
259
+ padding of the graph) plus the conv bias, computed in float64: ``stem(cat(x, range)) = conv(x; W[:, :-1]) + B``."""
260
+ import torch
261
+ F = torch.nn.functional
262
+
263
+ p = w.conv(module)
264
+ wt = torch.from_numpy(np.asarray(p.weight, np.float64)[:, -1:]) # [Cout, 1, 3, 3]
265
+ r = torch.from_numpy(range_channel(h, w_).astype(np.float64)).reshape(1, 1, h, w_)
266
+ pad = int(p.pads[0])
267
+ y = F.conv2d(r, wt, torch.from_numpy(np.asarray(p.bias, np.float64)), stride=tuple(int(s) for s in p.strides),
268
+ padding=pad)
269
+ return np.ascontiguousarray(y[0].permute(1, 2, 0).reshape(-1, y.shape[1]).numpy().astype(F32))
270
+
271
+
272
+ def sem_fc_nhwc(w: MeteorWeights) -> Tuple[np.ndarray, np.ndarray]:
273
+ """sem_ego.11 (Gemm 1280 -> 256 on ``flatten`` of the [64, 5, 4] pooled map, index c * 20 + p) with its inputs
274
+ re-ordered to the NHWC flatten p * 64 + c: (W [256, 1280], b [256])."""
275
+ g = w.gemm("sem_ego/sem_ego.11")
276
+ wt = np.asarray(g.weight, F32)
277
+ out, k = wt.shape
278
+ c = 64
279
+ npos = k // c
280
+ return np.ascontiguousarray(wt.reshape(out, c, npos).transpose(0, 2, 1).reshape(out, k)), np.asarray(g.bias, F32)
code/tt_meteor/reference/weights.py ADDED
@@ -0,0 +1,236 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ """Weights of ``meteor_v157c3Z.onnx`` (AutowareFoundation/meteor@v1.0), read the way the TT port reads them.
3
+
4
+ The ONNX is parsed as data with the vendored ``ttaw.weights.OnnxWeights`` (C04): nothing is executed, and every
5
+ parameter is addressed through the graph node that consumes it, so convolution attributes (strides, pads, dilations,
6
+ groups) come from the deployed graph itself, never from re-typed constants. Module names are the ONNX scopes without
7
+ the ``/net/`` prefix and the op suffix, e.g. ``seg_head/d1/d1.3/d1.3.0`` for ``/net/seg_head/d1/d1.3/d1.3.0/Conv``.
8
+
9
+ BatchNorm: the export folded every BatchNorm2d of the network into its convolution (SPEC section 4.2, "Every BN is
10
+ folded into the conv in the ONNX"; the stem equals ``meteor_v157.pt`` conv1 + BN to 2e-8, S:113), and the graph has
11
+ no BatchNormalization node left (checked by :meth:`MeteorWeights.check_graph`). The port therefore uses the conv
12
+ weights as stored and folds nothing itself: BN folding happens exactly where the TT port folds it, i.e. nowhere.
13
+ The exact rewrites the port applies (the /255 or ImageNet scale into the stem, merged sibling heads, the paint
14
+ gather as a 21->96 1x1 conv, tgate + tfuse merge, the zero ``mot`` channels of traj_stem dropped) are proven on the
15
+ CPU in ``reference.port_form``.
16
+
17
+ Besides convolutions and Gemms the graph carries a few non-conv constants the reference reads by consuming node:
18
+ the attention-pool parameters (``ego_attn``, ``agent_attn``: packed in-projections sliced by ``Slice`` nodes, learned
19
+ queries expanded by ``Expand``), the e2e LayerNorm, the two learned gates (``risk_gate`` -> ``/net/Mul_19``,
20
+ ``dec_gate`` -> ``/net/Tanh``), the constant adaptive-pooling matrices (``MatMul`` operands), the 0.4 m lift ground
21
+ points (``onnx::Expand_1643`` [1, 4, 100000]) and the painted seg2d classes (``/net/Gather`` indices).
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import os
26
+ from dataclasses import dataclass
27
+ from pathlib import Path
28
+ from typing import Dict, Iterator, List, Optional, Tuple
29
+
30
+ import numpy as np
31
+
32
+ from ..ttaw.weights import ConvParams, GemmParams, OnnxWeights, file_sha256
33
+ from . import config as C
34
+
35
+ __all__ = ["MeteorWeights", "MHAParams", "find_onnx", "conv_node", "gemm_node", "POOL_NODES"]
36
+
37
+ NET = "/net/"
38
+
39
+
40
+ def conv_node(module: str) -> str:
41
+ return f"{NET}{module}/Conv"
42
+
43
+
44
+ def gemm_node(module: str) -> str:
45
+ return f"{NET}{module}/Gemm"
46
+
47
+
48
+ # constant adaptive-pool / mean matrices: name -> (MatMul node, constant slot). Windows follow
49
+ # adaptive_avg_pool2d: [floor(i n / m), ceil((i + 1) n / m)) (export_onnx.py:53-104; S:356-366)
50
+ POOL_NODES: Dict[str, Tuple[str, int]] = {
51
+ "ego_stem_w": ("/net/ego_stem/ego_stem.15/MatMul", 1), # [16, 1] global mean over 25x16
52
+ "ego_stem_h": ("/net/ego_stem/ego_stem.15/MatMul_1", 0), # [1, 25]
53
+ "tl_w": ("/net/tl_head/tl_head.8/MatMul", 1), # [48, 1] global mean over 27x48
54
+ "tl_h": ("/net/tl_head/tl_head.8/MatMul_1", 0), # [1, 27]
55
+ "fused_pool_w": ("/net/MatMul_1", 1), # [500, 16] fused 800x500 -> 25x16 (ego_attn keys)
56
+ "fused_pool_h": ("/net/MatMul_2", 0), # [25, 800]
57
+ "sem_w": ("/net/sem_ego/sem_ego.9/MatMul", 1), # [16, 4] 25x16 -> 5x4
58
+ "sem_h": ("/net/sem_ego/sem_ego.9/MatMul_1", 0), # [5, 25]
59
+ "det_pool_w": ("/net/MatMul_3", 1), # [250, 16] det_feat 400x250 -> 25x16 (agent keys)
60
+ "det_pool_h": ("/net/MatMul_4", 0), # [25, 400]
61
+ "fused_mean_w": ("/net/MatMul_5", 1), # [500, 1] mean(fused) for dec_head
62
+ "fused_mean_h": ("/net/MatMul_6", 0), # [1, 800]
63
+ "e2e_mean_w": ("/net/refiner/e2e/MatMul", 1), # [500, 1] the same mean for the e2e refiner
64
+ "e2e_mean_h": ("/net/refiner/e2e/MatMul_1", 0), # [1, 800]
65
+ }
66
+
67
+
68
+ @dataclass(frozen=True)
69
+ class MHAParams:
70
+ """``nn.MultiheadAttention`` exported by torch 2.1: packed in-projection (q | k | v rows), out-projection."""
71
+
72
+ in_weight: np.ndarray # [3E, E]
73
+ in_bias: np.ndarray # [3E]
74
+ out_weight: np.ndarray # [E, E] (Gemm transB=1: y = x W^T + b)
75
+ out_bias: np.ndarray # [E]
76
+ num_heads: int
77
+ scale_divisor: float # q is divided by sqrt(E / heads) (the graph's Div constant)
78
+
79
+ @property
80
+ def embed_dim(self) -> int:
81
+ return int(self.out_weight.shape[0])
82
+
83
+
84
+ def find_onnx(weights_dir: Optional[os.PathLike] = None) -> Path:
85
+ """The ONNX file in ``weights_dir`` (a snapshot of AutowareFoundation/meteor) or, for workspace runs,
86
+ ``$METEOR_WEIGHTS_DIR`` or ``assets/meteor/hf_meteor`` of the workspace."""
87
+ cands: List[Path] = []
88
+ if weights_dir is not None:
89
+ cands.append(Path(weights_dir))
90
+ if os.environ.get("METEOR_WEIGHTS_DIR"):
91
+ cands.append(Path(os.environ["METEOR_WEIGHTS_DIR"]))
92
+ here = Path(__file__).resolve()
93
+ for parent in here.parents:
94
+ if (parent / "assets" / "meteor" / "hf_meteor").is_dir():
95
+ cands.append(parent / "assets" / "meteor" / "hf_meteor")
96
+ break
97
+ for c in cands:
98
+ p = c if c.suffix == ".onnx" else c / C.ONNX_FILE
99
+ if p.is_file():
100
+ return p
101
+ raise FileNotFoundError(f"{C.ONNX_FILE} not found in {[str(c) for c in cands]}")
102
+
103
+
104
+ class MeteorWeights:
105
+ """The deployed graph's parameters, addressed by consuming node (see the module docstring)."""
106
+
107
+ def __init__(self, path: Optional[os.PathLike] = None, *, verify_sha256: bool = False):
108
+ self.path = find_onnx(path) if (path is None or Path(path).is_dir()) else Path(path)
109
+ if verify_sha256 and file_sha256(self.path) != C.ONNX_SHA256:
110
+ raise ValueError(f"{self.path} is not the pinned {C.ONNX_FILE} (sha256 {C.ONNX_SHA256[:12]}...)")
111
+ self.onnx = OnnxWeights(self.path)
112
+
113
+ # ---- generic access ------------------------------------------------------------------------------------------
114
+ @property
115
+ def sha256(self) -> str:
116
+ return self.onnx.sha256
117
+
118
+ def conv(self, module: str) -> ConvParams:
119
+ return self.onnx.conv(conv_node(module))
120
+
121
+ def gemm(self, module: str) -> GemmParams:
122
+ return self.onnx.gemm(gemm_node(module))
123
+
124
+ def conv_modules(self) -> List[str]:
125
+ """Every Conv of the graph by module name, in graph order (190)."""
126
+ return [n.name[len(NET):-len("/Conv")] for n in self.onnx.nodes("Conv")]
127
+
128
+ def gemm_modules(self) -> List[str]:
129
+ return [n.name[len(NET):-len("/Gemm")] for n in self.onnx.nodes("Gemm")]
130
+
131
+ def iter_convs(self) -> Iterator[Tuple[str, ConvParams]]:
132
+ for m in self.conv_modules():
133
+ yield m, self.conv(m)
134
+
135
+ def const(self, node: str, slot: int) -> np.ndarray:
136
+ """The constant feeding input ``slot`` of ``node`` (a full node name, e.g. ``/net/Mul_19``)."""
137
+ return self.onnx.param(node, slot)
138
+
139
+ # ---- named non-conv parameters ---------------------------------------------------------------------------------
140
+ def pool_matrix(self, name: str) -> np.ndarray:
141
+ node, slot = POOL_NODES[name]
142
+ return self.const(node, slot)
143
+
144
+ def mha(self, name: str) -> MHAParams:
145
+ """``ego_attn`` (E 96, 4 heads, 3 queries) or ``agent_attn`` (E 128, 4 heads, 4 queries)."""
146
+ p = f"{NET}{name}/"
147
+ w_q = self.const(p + "Slice_1", 0) # the packed [3E, E] in_proj_weight, sliced into q and k|v
148
+ w_kv = self.const(p + "Slice_2", 0)
149
+ b_q = self.const(p + "Slice_3", 0)
150
+ b_kv = self.const(p + "Slice_4", 0)
151
+ if w_q is not w_kv and not np.array_equal(w_q, w_kv):
152
+ raise ValueError(f"{name}: the q and k|v slices do not read one packed in-projection")
153
+ if not np.array_equal(b_q, b_kv):
154
+ raise ValueError(f"{name}: the q and k|v bias slices do not read one packed bias")
155
+ out = self.gemm(name)
156
+ div = float(self.const(p + "Div_1", 1))
157
+ heads = int(self.const(p + "Div", 1)) # head dim = E // heads (the graph's /Div by 4)
158
+ e = int(out.weight.shape[0])
159
+ if abs(div - np.float32(np.sqrt(e / heads))) > 1e-6:
160
+ raise ValueError(f"{name}: q scale divisor {div} != sqrt({e}/{heads})")
161
+ return MHAParams(np.asarray(w_q, np.float32), np.asarray(b_q, np.float32), np.asarray(out.weight, np.float32),
162
+ np.asarray(out.bias, np.float32), heads, div)
163
+
164
+ def queries(self, name: str) -> np.ndarray:
165
+ """Learned attention-pool queries: ``ego`` [1, 3, 96] (``/net/Expand_1``), ``agent`` [1, 4, 128]
166
+ (``/net/Expand_2``)."""
167
+ node = {"ego": "/net/Expand_1", "agent": "/net/Expand_2"}[name]
168
+ return self.const(node, 0)
169
+
170
+ def layer_norm(self) -> Tuple[np.ndarray, np.ndarray, float]:
171
+ node = self.onnx.node(f"{NET}refiner/e2e/norm/LayerNormalization")
172
+ return (self.const(node.name, 1), self.const(node.name, 2), float(node.attrs.get("epsilon", 1e-5)))
173
+
174
+ def risk_gate(self) -> float:
175
+ """``net.risk_gate`` (2.2981195, S:120): mode logits -= risk_gate * mean_t sigmoid(risk)(path_t)."""
176
+ return float(self.const("/net/Mul_19", 0).reshape(()))
177
+
178
+ def dec_gate(self) -> float:
179
+ """``net.dec_gate`` (0.0336921, S:120): ego[:36] += tanh(dec_gate) * (wp_dec - ego[:36])."""
180
+ return float(self.const("/net/Tanh", 0).reshape(()))
181
+
182
+ def lift_ground_points(self) -> np.ndarray:
183
+ """The constant [1, 4, 100000] homogeneous ground points (x, y, 0, 1) of the 400 x 250 lift grid."""
184
+ return self.const("/net/Expand", 0)
185
+
186
+ def paint_classes(self) -> Tuple[int, ...]:
187
+ return tuple(int(i) for i in self.const("/net/Gather", 1).reshape(-1))
188
+
189
+ def zero_constants(self) -> Dict[str, np.ndarray]:
190
+ """The baked-out history (``mot``) constants that the graph subtracts / multiplies (both all-zero)."""
191
+ return {"hist": self.const("/net/Sub_5", 1), "mask": self.const("/net/Mul_14", 1)}
192
+
193
+ # ---- graph check ----------------------------------------------------------------------------------------------
194
+ def check_graph(self) -> Dict[str, int]:
195
+ """Fail loudly unless the file is the graph this port reproduces (ops, attributes, constants)."""
196
+ o = self.onnx
197
+ problems: List[str] = []
198
+ ops: Dict[str, int] = {}
199
+ for n in o.nodes():
200
+ ops[n.op_type] = ops.get(n.op_type, 0) + 1
201
+ expect = {"Conv": 190, "Gemm": 15, "GridSample": 3, "Resize": 16, "Softmax": 7, "ArgMax": 4,
202
+ "LayerNormalization": 1, "AveragePool": 3, "MaxPool": 1, "GatherElements": 2, "CumSum": 2}
203
+ for op, cnt in expect.items():
204
+ if ops.get(op, 0) != cnt:
205
+ problems.append(f"{ops.get(op, 0)} {op} nodes, expected {cnt}")
206
+ if ops.get("BatchNormalization", 0):
207
+ problems.append("BatchNormalization nodes present (the reference assumes a fully folded export)")
208
+ for n in o.nodes("Resize"):
209
+ a = n.attrs
210
+ if a.get("mode") != "linear" or a.get("coordinate_transformation_mode") != "half_pixel":
211
+ problems.append(f"{n.name}: Resize {a.get('mode')}/{a.get('coordinate_transformation_mode')}")
212
+ for n in o.nodes("GridSample"):
213
+ want = "border" if n.name == "/net/GridSample_2" else "zeros"
214
+ a = n.attrs
215
+ if (a.get("mode"), a.get("padding_mode"), int(a.get("align_corners", 0))) != ("bilinear", want, 0):
216
+ problems.append(f"{n.name}: GridSample {a}")
217
+ for n in o.nodes("ArgMax"):
218
+ if int(n.attrs.get("select_last_index", 0)) != 0:
219
+ problems.append(f"{n.name}: select_last_index")
220
+ if self.paint_classes() != C.PAINT_CLASSES:
221
+ problems.append(f"paint classes {self.paint_classes()} != {C.PAINT_CLASSES}")
222
+ g, b, eps = self.layer_norm()
223
+ if g.shape != (139,) or abs(eps - C.E2E_LN_EPS) > 1e-9:
224
+ problems.append(f"e2e LayerNorm {g.shape} eps {eps}")
225
+ z = self.zero_constants()
226
+ if any(np.any(v != 0) for v in z.values()):
227
+ problems.append("the baked history constants are not all zero")
228
+ pts = self.lift_ground_points()
229
+ if pts.shape != (1, 4, C.LIFT_H * C.LIFT_W):
230
+ problems.append(f"lift points {pts.shape}")
231
+ for name, (node, slot) in POOL_NODES.items():
232
+ if not o.has(o.node(node).inputs[slot]):
233
+ problems.append(f"pool matrix {name}: {node} slot {slot} is not constant")
234
+ if problems:
235
+ raise ValueError(f"{self.path.name} is not the graph this port reproduces: " + "; ".join(problems))
236
+ return ops
code/tt_meteor/samples/README.md ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Sample inputs shipped with meteor-p150
2
+
3
+ Only REDISTRIBUTABLE data goes here (Apache-2.0 / MIT / CC-BY with attribution). Data whose license is unstated or
4
+ non-commercial (the Autoware demo rosbag, nuScenes, Argoverse 2, ...) is NOT shipped: `code/scripts/fetch_samples.sh`
5
+ downloads it (sha256-checked) for the tests and benchmarks.
6
+
7
+ Never use the `.bin` suffix here (or `.pt`, `.pth`, `.ckpt`, `.safetensors`): tt-model's staging silently drops
8
+ those suffixes from `code/` (`CODE_IGNORE`, tt-model-manager `src/tt_kernel/build.py:328-332`). Store point clouds as
9
+ `.npy` / `.npz` / `.pcd`.
10
+
11
+ Next to each sample, `<stem>.reference.json` is the `/predict` body of the fp32 CPU reference (`tt_meteor.reference`,
12
+ `Output.to_dict()`) on it; `server/smoke_test.py` compares the served output with it (the smoke gate for synthetic
13
+ samples, which may give no detections). When the output depends on the serve profile or the model variant, store
14
+ one per profile or variant as `<stem>.<profile>.reference.json` / `<stem>.<variant>.reference.json`. Regenerate it
15
+ whenever the reference, the weights or the post-processing changes.
16
+
17
+ | file | content | source | license |
18
+ |---|---|---|---|
19
+ | `synthetic_8cam.json` | the request manifest: the eight camera PNGs (all present), calibration preset `synthetic_8cam`, `ego_speed` 8.0 m/s, stream id + ego pose; `tt_meteor.load_sample(path)` -> `model(**kwargs)`; `server/client.py --sample` / `server/smoke_test.py` build the `/predict` body from it | generated by `code/scripts/make_synthetic_sample.py` | Apache-2.0 (data generated by this repository) |
20
+ | `synthetic_8cam/<CAM>.png` | eight 768x432 RGB images (lossless PNG, 1.4 MB in total): a ray-cast street (a straight two-lane road with lane lines, a stop line and a zebra crossing, kerbs, pavements, facades, box-shaped vehicles and pedestrians) seen by a generic METEOR-like rig; the pixels of the objects (only those) were then optimised against the fp32 CPU reference (gradient ascent through the image encoder, the lift, the BEV detector and the box refiner, 88 steps) until the 3D heads report six vehicles and two pedestrians with large margins (scores 0.889-0.993 against the thresholds 0.35 / 0.15) and nothing else (strongest other heatmap cell 0.179 vehicle / 0.072 VRU), so bf16 numerics cannot flip a published box. A test pattern, not a photograph: the objects carry noisy texture | as above | Apache-2.0 |
21
+ | `synthetic_8cam.reference.json` | the `/predict` body of the fp32 CPU reference (`tt_meteor.reference.pipeline.MeteorReference`, PCC 1.0 vs ONNX Runtime) on the request above as a fresh stream (`timing_ms` emptied): 8 boxes (6 VEHICLE, 2 VRU), 10 2D boxes, plan mode 0 (the car ahead and a red light: the selected path brakes), traffic light red; `server/smoke_test.py` (the container smoke, `--expect VEHICLE:6,VRU:2` by default) and `tests/test_e2e_device.py` compare the served body with it | `code/scripts/make_synthetic_sample.py finalise` | Apache-2.0 |
22
+
23
+ The calibration preset `../calib/synthetic_8cam.json` is the same generated rig (round numbers of our own: METEOR's
24
+ slot layout, 98 deg wide cameras front and back, corner cameras pitched 25 deg down, 30.4 deg narrow cameras, fy =
25
+ 0.87 fx like METEOR's vertically squashed training images, 1.9 m above the road).
26
+
27
+ A real-world sample (PandaSet 019 frame 40, CC BY 4.0 + PandaSet Dataset Terms, METEOR's 8-slot layout) is prepared
28
+ but **not shipped** until the redistribution of PandaSet-derived samples is approved; the device tests read it from
29
+ the git-ignored `staging_samples_pandaset/` of the development checkout (`tests/paths.py`). METEOR's own demo scenes
30
+ (`AutowareFoundation/meteor-demo-scenes`, research / demonstration use only) and nuScenes (CC BY-NC-SA 4.0) are never
31
+ shipped: `code/scripts/fetch_samples.sh` notes where to fetch them.