#!/usr/bin/env python3 # SPDX-License-Identifier: Apache-2.0 """Device profile of meteor-p150: one eager frame with module signposts and ``--rounds`` traced replays of the ``frame`` trace, between Tracy signposts (OPT_BASELINE.md). ROOT=/home/ubuntu/experiments/tt-models $ROOT/bin/devrun -t 3600 -- python -m tracy -r -p -v --no-web-server --op-support-count 6000 \\ -o $ROOT/generated/profiler/meteor_baseline code/scripts/profile_ops.py --rounds 2 tt-perf-report --start-signpost frame --end-signpost frame_end --tracing-mode python code/scripts/profile_summary.py $ROOT/generated/profiler/meteor_baseline --json s.json --md s.md ttnn-visualizer memory + graph report of one eager frame (slow; never used for timing): $ROOT/bin/devrun -t 3600 -- python code/scripts/profile_ops.py --graph-report $ROOT/generated/ttnn_reports/meteor_baseline ttnn-visualizer --profiler-path $ROOT/generated/ttnn_reports/meteor_baseline/frame \\ --performance-path $ROOT/generated/profiler/meteor_baseline/reports/ The model is loaded like the API but without the trace capture (``warmup_variants="none"``), the rig's lift tables are written, and one eager frame compiles every program. Then, with the device profiler buffer flushed (``ttnn.ReadDeviceProfiler``) before and after each section: - ``eager_frame`` .. ``eager_frame_end``: one eager frame (program cache hot) with a signpost per module (``img.*``, ``lift``, ``lift.resize``, ``bev.*``, ``head.*``, ``out.pack``). Ops issued inline after a module call (argmaxes, the depth softmax, the ``to_rm`` conversions, planner adds) belong to the module signposted last. An eager frame BEFORE the capture is the documented order (``TtMETEOR.run_frame_eager``: no allocation next to a live trace); - the trace is captured (``TtMETEOR.capture``) and replayed once (not profiled); - ``frame`` .. ``frame_end`` (and ``frame2`` ... for ``--rounds``): one replay each, the served device path. Through the identical op order the replay's ops inherit the eager attribution (``profile_summary.py``). One frame is about 2,000 programs, above the profiler's default buffer of ~1,000, hence ``--op-support-count 6000``. The output folder must be absolute and outside the bundle (PLAN.md 5.1). """ from __future__ import annotations import argparse import contextlib import json import os import sys import threading import time from pathlib import Path from typing import Callable, List HERE = Path(__file__).resolve() sys.path.insert(0, str(HERE.parent)) sys.path.insert(0, str(HERE.parents[1])) IMAGE_PARTS = (("normalize", "img.normalize"), ("stem_conv", "img.stem"), ("stages", "img.resnet"), ("fuse", "img.fpn_fuse"), ("seg_logits", "img.seg2d"), ("depth_logits", "img.depth"), ("depth_mean", "img.depth_mean"), ("context", "img.ctx_paint"), ("det2d", "img.det2d"), ("tl", "img.tl")) BEV_PARTS = ("fuse", "lane", "det", "feat_bf16", "occupancy", "stationary", "traj", "risk") HEAD_PARTS = ("ego_stem", "ego_mlp", "ego_attn", "sem", "risk_sample", "decoder_wp", "e2e", "seg_refine", "box_refine") class _Signposted: """A callable stand-in that emits a signpost (optionally one after the call too), then calls the inner object.""" def __init__(self, inner, label: str, signpost, after: str = ""): self.inner, self._label, self._signpost, self._after = inner, label, signpost, after def __call__(self, *args, **kwargs): if self._label: self._signpost(self._label) out = self.inner(*args, **kwargs) if self._after: self._signpost(self._after) return out def __getattr__(self, name): return getattr(self.inner, name) def module_signposts(tt, signpost) -> Callable[[], None]: """Wrap the stage modules of a ``TtMETEOR`` with signposting stand-ins; returns the function that restores them.""" undo: List[Callable[[], None]] = [] def swap(obj, attr, label, after=""): had = attr in obj.__dict__ old = obj.__dict__.get(attr) inner = getattr(obj, attr) setattr(obj, attr, _Signposted(inner, label, signpost, after)) undo.append(lambda: setattr(obj, attr, old) if had else obj.__dict__.pop(attr, None)) img = tt.image for attr, label in IMAGE_PARTS: swap(img, attr, label) lat0 = img.lat[0] img.lat[0] = _Signposted(lat0, "img.fpn", signpost) undo.append(lambda: img.lat.__setitem__(0, lat0)) swap(tt.lift, "resize", "lift.resize") swap(tt, "lift", "lift") for attr in BEV_PARTS: swap(tt.bev, attr, f"bev.{attr}") for attr in HEAD_PARTS: swap(tt.head, attr, f"head.{attr}") swap(tt, "head", "", after="out.pack") def restore() -> None: for fn in reversed(undo): fn() return restore def rss_guard(limit_gb: float, stop: threading.Event) -> None: """The graph capture lives in host RAM: never let it take the shared host down.""" page = os.sysconf("SC_PAGE_SIZE") while not stop.wait(1.0): with open("/proc/self/statm") as fh: rss = int(fh.read().split()[1]) * page if rss > limit_gb * 2 ** 30: print(f"graph capture: host RSS {rss / 2 ** 30:.1f} GB > {limit_gb} GB, abort", flush=True) os._exit(3) def main() -> int: ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) ap.add_argument("--input", default="sample", help="bench.py input spec (sample | synthetic | name=manifest)") ap.add_argument("--rounds", type=int, default=2) ap.add_argument("--no-eager", dest="eager", action="store_false") ap.add_argument("--graph-report", default=None, help="write a ttnn-visualizer graph report of one eager frame to " "/frame instead of profiling") ap.add_argument("--max-rss-gb", type=float, default=24.0) ap.add_argument("--dispatch", default=None, choices=["eth", "worker"]) ap.add_argument("--num-cqs", type=int, default=None) ap.add_argument("--json", default=None, help="write the run's metadata (config, trace description)") a = ap.parse_args() import ttnn from bench import cache_entries, load_input from tt_meteor import METEOR from tt_meteor.host.inputs import prepare_request from tt_meteor.ttaw.profiling import read_device_profiler, signpost name, req, path = load_input(a.input) t0 = time.perf_counter() model = METEOR.from_pretrained(dispatch=a.dispatch, num_command_queues=a.num_cqs, warmup_variants="none") meta = {"config": model.device_info, "input": str(path or name), "load_s": round(time.perf_counter() - t0, 1)} print("config:", json.dumps(meta["config"]), flush=True) dev, tt = model.device, model.tt sync = lambda: ttnn.synchronize_device(dev) # noqa: E731 graph_out = None try: frame = prepare_request(req["images"], req["calibration"], req["ego_speed"], req.get("stream")) geom = model.calib_cache.get(frame.K, frame.T_cam_ego) t1 = time.perf_counter() tt.run_frame_eager(frame, geom) # compile + constants sync() meta["first_eager_s"] = round(time.perf_counter() - t1, 1) meta["program_cache_entries"] = cache_entries(dev) print("first eager frame %.1f s, programs %s" % (meta["first_eager_s"], meta["program_cache_entries"]), flush=True) if a.graph_report: out_dir = Path(a.graph_report).resolve() / "frame" out_dir.mkdir(parents=True, exist_ok=True) json_path = out_dir / "graph_capture.json" stop = threading.Event() threading.Thread(target=rss_guard, args=(a.max_rss_gb, stop), daemon=True).start() prev = ttnn.CONFIG.enable_fast_runtime_mode ttnn.CONFIG.enable_fast_runtime_mode = False t1 = time.perf_counter() try: ttnn.graph.begin_graph_capture() tt.run_frame_eager(frame, geom) sync() ttnn.graph.end_graph_capture_to_file(str(json_path)) finally: ttnn.CONFIG.enable_fast_runtime_mode = prev stop.set() meta["graph_capture_s"] = round(time.perf_counter() - t1, 1) meta["graph_json_mb"] = round(json_path.stat().st_size / 1e6, 1) print("graph captured:", meta["graph_capture_s"], "s,", meta["graph_json_mb"], "MB", flush=True) graph_out = (out_dir, json_path) else: read_device_profiler(dev) # drop the load / compile data if a.eager: restore = module_signposts(tt, signpost) try: signpost("eager_frame") tt.run_frame_eager(frame, geom) sync() signpost("eager_frame_end") finally: restore() read_device_profiler(dev) t1 = time.perf_counter() tt.capture() meta["capture_s"] = round(time.perf_counter() - t1, 1) tt.run_frame(frame, geom) # one served frame (tables current, inputs uploaded) sync() read_device_profiler(dev) for k in range(1, a.rounds + 1): tag = "" if k == 1 else str(k) signpost(f"frame{tag}") tt.runner.replay("frame") sync() signpost(f"frame{tag}_end") read_device_profiler(dev) meta["trace"] = tt.describe() meta["program_cache_entries_end"] = cache_entries(dev) finally: model.close() if graph_out is not None: out_dir, json_path = graph_out t1 = time.perf_counter() try: db = ttnn.graph_report.import_report(json_path, out_dir) finally: json_path.unlink(missing_ok=True) ttnn.save_config_to_json_file(out_dir / "config.json") cfg = json.loads((out_dir / "config.json").read_text()) cfg.update({"enable_fast_runtime_mode": False, "enable_graph_report": True, "enable_detailed_buffer_report": False, "root_report_path": str(out_dir.parent), "report_name": out_dir.name}) (out_dir / "config.json").write_text(json.dumps(cfg, indent=4) + "\n") meta["graph_report"] = {"dir": str(out_dir), "db": str(db), "import_s": round(time.perf_counter() - t1, 1), "db_mb": round(Path(db).stat().st_size / 1e6, 1)} print("graph report:", json.dumps(meta["graph_report"]), flush=True) if a.json: Path(a.json).parent.mkdir(parents=True, exist_ok=True) Path(a.json).write_text(json.dumps(meta, indent=1, default=str) + "\n") print(json.dumps(meta, indent=1, default=str)) return 0 if __name__ == "__main__": sys.exit(main())