#!/usr/bin/env python3 # SPDX-License-Identifier: Apache-2.0 """Summary of a ``profile_ops.py`` device profile (the ``ops_perf_results_.csv`` of ``python -m tracy -r``). python code/scripts/profile_summary.py $ROOT/generated/profiler/meteor_baseline --json out.json --md out.md - per traced segment (the ``frame`` signpost range, and ``frame2`` ... of later rounds): programs, kernel sum, FW sum, op-to-op gaps (sum over the segment, its first op excluded) and the span (first FW start to last FW end, at the CHIP_FREQ of ``profile_log_device.csv``, else 1350 MHz), per op code; - per module stage: the eager run's module signposts (``img.normalize`` ... ``out.pack``) label the eager op sequence; when it equals the traced segment's op sequence op for op, the labels are carried over to the traced ops, and the table reports the traced kernel times per stage, per block (``img`` / ``lift`` / ``bev`` / ``head`` / ``out``) and per (stage, op code); - the top ops by traced kernel time and the largest op-to-op gaps. csv + json only (no ttnn): runs anywhere. """ from __future__ import annotations import argparse import csv import glob import json import os import sys from collections import OrderedDict, defaultdict from pathlib import Path from typing import Any, Dict, List, Optional, Tuple SEGMENTS = ("frame",) KERNEL = "DEVICE KERNEL DURATION [ns]" FW = "DEVICE FW DURATION [ns]" GAP = "OP TO OP LATENCY [ns]" def find_csv(path: Path) -> Path: if path.is_file(): return path found = sorted(glob.glob(str(path / "**" / "ops_perf_results*.csv"), recursive=True), key=os.path.getmtime) if not found: raise FileNotFoundError(f"no ops_perf_results*.csv under {path}") return Path(found[-1]) def chip_freq_mhz(csv_path: Path, root: Path) -> Optional[float]: cands = [csv_path.parent / "profile_log_device.csv", root / ".logs" / "profile_log_device.csv"] cands += [Path(p) for p in glob.glob(str(root / "**" / "profile_log_device.csv"), recursive=True)] for c in cands: try: head = c.read_text(errors="ignore").splitlines()[0] except (OSError, IndexError): continue for part in head.split(","): if "CHIP_FREQ" in part and ":" in part: try: return float(part.split(":")[1].strip()) except ValueError: pass return None def num(row: Dict[str, str], key: str) -> float: v = (row.get(key) or "").strip() try: return float(v) if v else 0.0 except ValueError: return 0.0 def is_signpost(row: Dict[str, str]) -> bool: return (row.get("OP TYPE") or "").strip() == "signpost" def window(rows: List[Dict[str, str]], name: str) -> Optional[List[Tuple[str, Dict[str, str]]]]: """Rows between the first signpost ``name`` and the next ``name_end`` as ``(label, row)``; ``label`` is the latest inner signpost (``""`` before the first one). None when the signpost is absent.""" out, on, label = [], False, "" for r in rows: if is_signpost(r): code = r.get("OP CODE", "") if on and code == f"{name}_end": return out if not on and code == name: on = True continue if on: label = code continue if on: out.append((label, r)) return out if on else None def tensor_desc(row: Dict[str, str], prefix: str) -> str: """``INPUT_0`` / ``OUTPUT_0`` columns -> ``[W, Z, Y, X] LAYOUT DTYPE MEMORY`` (logical dims), '' when absent.""" dims = [] for d in "WZYX": v = (row.get(f"{prefix}_{d}_PAD[LOGICAL]") or "").strip() if not v: return "" dims.append(v.split("[")[1].rstrip("]") if "[" in v else v) mem = (row.get(f"{prefix}_MEMORY") or "").replace("DEV_1_", "").replace("DEV_0_", "") return f"[{','.join(dims)}] {row.get(f'{prefix}_LAYOUT', '')} {row.get(f'{prefix}_DATATYPE', '')} {mem}".strip() def io_desc(row: Dict[str, str], prefix: str, n: int = 3) -> str: if (row.get(prefix + "S") or "").strip(): # older reports: one INPUTS / OUTPUTS column text = " ".join(row[prefix + "S"].split()) return text if len(text) <= 160 else text[:157] + "..." parts = [tensor_desc(row, f"{prefix}_{i}") for i in range(n)] return "; ".join(p for p in parts if p) def group_of(label: str) -> str: """``img.resnet`` -> ``img``, ``bev.fuse`` -> ``bev``, ``lift.resize`` -> ``lift``: the block of a stage.""" return label.partition(".")[0] def summarize(ops: List[Dict[str, str]], freq: float) -> Dict[str, Any]: kernel = [num(r, KERNEL) / 1e3 for r in ops] fw = [num(r, FW) / 1e3 for r in ops] gaps = [num(r, GAP) / 1e3 for r in ops] starts = [num(r, "DEVICE FW START CYCLE") for r in ops] ends = [num(r, "DEVICE FW END CYCLE") for r in ops] valid = [s for s in starts if s > 0] span = (max(ends) - min(valid)) / freq if ops and valid and max(ends) > 0 else None by_op: Dict[str, List[float]] = defaultdict(lambda: [0, 0.0]) fid: Dict[str, int] = defaultdict(int) for r, k in zip(ops, kernel): by_op[r.get("OP CODE", "?")][0] += 1 by_op[r.get("OP CODE", "?")][1] += k f = (r.get("MATH FIDELITY") or "").strip() if f: fid[f] += 1 total = sum(kernel) or 1.0 traced = sum(1 for r in ops if (r.get("METAL TRACE ID") or "").strip()) return {"ops": len(ops), "traced_ops": traced, "kernel_us": round(sum(kernel), 1), "fw_us": round(sum(fw), 1), "op2op_us": round(sum(gaps[1:]), 1), "span_us": None if span is None else round(span, 1), "by_op": [{"op": op, "count": int(c), "kernel_us": round(t, 1), "share": round(t / total, 4)} for op, (c, t) in sorted(by_op.items(), key=lambda kv: -kv[1][1])], "fidelity": dict(fid)} def main() -> int: ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) ap.add_argument("source", type=Path, help="tracy output dir (-o) or an ops_perf_results CSV") ap.add_argument("--freq-mhz", type=float, default=None) ap.add_argument("--top", type=int, default=15) ap.add_argument("--json", type=Path, default=None) ap.add_argument("--md", type=Path, default=None) a = ap.parse_args() path = find_csv(a.source) with open(path, newline="") as f: rows = list(csv.DictReader(f)) root = a.source if a.source.is_dir() else path.parent freq = a.freq_mhz or chip_freq_mhz(path, root) or 1350.0 res: Dict[str, Any] = {"csv": str(path), "freq_mhz": freq, "rows": len(rows), "signposts": [r.get("OP CODE") for r in rows if is_signpost(r)], "segments": OrderedDict(), "eager": OrderedDict()} traced_ops: Dict[str, List[Dict[str, str]]] = {} for k in ("", "2", "3"): for seg in SEGMENTS: w = window(rows, f"{seg}{k}") if w is None: continue ops = [r for _, r in w] traced_ops[f"{seg}{k}"] = ops res["segments"][f"{seg}{k}"] = summarize(ops, freq) steady = [s for s in SEGMENTS if s in res["segments"]] if steady: seg = res["segments"] res["steady_frame"] = {key: round(sum(seg[s][key] or 0 for s in steady), 1) for key in ("ops", "kernel_us", "fw_us", "op2op_us", "span_us")} res["steady_frame"]["segments"] = steady # eager windows with stage labels; carry the labels over to the traced ops when the sequences match labels: Dict[str, List[str]] = {} groups: "OrderedDict[str, Dict[str, Any]]" = OrderedDict() for seg in SEGMENTS: w = window(rows, f"eager_{seg}") if w is None: continue ops = [r for _, r in w] res["eager"][seg] = summarize(ops, freq) eager_codes = [r.get("OP CODE") for r in ops] traced_codes = [r.get("OP CODE") for r in traced_ops.get(seg, [])] match = eager_codes == traced_codes res["eager"][seg]["sequence_matches_trace"] = match if not match: res["eager"][seg]["mismatch"] = {"eager_ops": len(eager_codes), "traced_ops": len(traced_codes), "first_diff": next((i for i, (x, y) in enumerate( zip(eager_codes, traced_codes)) if x != y), None)} labels[seg] = [lab or f"{seg[:3]}.begin" for lab, _ in w] # per stage: eager kernel sums (always) and traced kernel sums (when aligned) stages: "OrderedDict[str, Dict[str, Any]]" = OrderedDict() for i, (lab, r) in enumerate(w): st = stages.setdefault(lab or f"{seg[:3]}.begin", {"ops": 0, "eager_kernel_us": 0.0, "eager_op2op_us": 0.0, "kernel_us": 0.0, "op2op_us": 0.0, "by_op": defaultdict(float)}) st["ops"] += 1 st["eager_kernel_us"] += num(r, KERNEL) / 1e3 st["eager_op2op_us"] += num(r, GAP) / 1e3 if i else 0.0 src = traced_ops[seg][i] if match else r st["by_op"][src.get("OP CODE", "?")] += num(src, KERNEL) / 1e3 if match: st["kernel_us"] += num(src, KERNEL) / 1e3 st["op2op_us"] += num(src, GAP) / 1e3 if i else 0.0 g = group_of(lab or f"{seg[:3]}.begin") go = groups.setdefault(g, {"ops": 0, "kernel_us": 0.0, "by_op": defaultdict(lambda: [0, 0.0])}) go["ops"] += 1 go["kernel_us"] += num(src, KERNEL) / 1e3 go["by_op"][src.get("OP CODE", "?")][0] += 1 go["by_op"][src.get("OP CODE", "?")][1] += num(src, KERNEL) / 1e3 go["traced"] = match for v in stages.values(): v["by_op"] = {op: round(t, 1) for op, t in sorted(v["by_op"].items(), key=lambda kv: -kv[1])} res["eager"][seg]["stages"] = {k: {kk: round(vv, 1) if isinstance(vv, float) else vv for kk, vv in v.items()} for k, v in stages.items()} res["groups"] = {g: {"ops": v["ops"], "kernel_us": round(v["kernel_us"], 1), "traced": v.get("traced", False), "by_op": [{"op": op, "count": c, "kernel_us": round(t, 1)} for op, (c, t) in sorted(v["by_op"].items(), key=lambda kv: -kv[1][1])]} for g, v in groups.items()} # top ops by traced kernel time (all segments of the first round) flat = [] for seg in SEGMENTS: for i, r in enumerate(traced_ops.get(seg, [])): lab = labels.get(seg, [None] * (i + 1))[i] if (seg in labels and res["eager"].get(seg, {}).get( "sequence_matches_trace")) else None flat.append({"segment": seg, "index": i, "stage": lab, "op": r.get("OP CODE"), "kernel_us": round(num(r, KERNEL) / 1e3, 1), "fw_us": round(num(r, FW) / 1e3, 1), "op2op_us": round(num(r, GAP) / 1e3, 2), "cores": r.get("CORE COUNT"), "fidelity": r.get("MATH FIDELITY"), "pm_ideal_us": round(num(r, "PM IDEAL [ns]") / 1e3, 1), "dram_bw_util": r.get("DRAM BW UTIL (%)"), "fpu_util": r.get("PM FPU UTIL (%)"), "inputs": io_desc(r, "INPUT"), "outputs": io_desc(r, "OUTPUT", 1)}) res["top_ops"] = sorted(flat, key=lambda d: -d["kernel_us"])[: a.top] res["top_gaps"] = sorted(flat, key=lambda d: -d["op2op_us"])[: a.top] if a.json: a.json.parent.mkdir(parents=True, exist_ok=True) a.json.write_text(json.dumps(res, indent=1) + "\n") lines = [f"csv `{path}` ({len(rows)} rows), CHIP_FREQ {freq:.0f} MHz", "", "| segment | programs | kernel us | FW us | op-to-op us | span us |", "|---|---:|---:|---:|---:|---:|"] for k, v in res["segments"].items(): lines.append(f"| {k} | {v['ops']} | {v['kernel_us']} | {v['fw_us']} | {v['op2op_us']} | {v['span_us']} |") if "steady_frame" in res: s = res["steady_frame"] lines.append(f"| **steady frame** ({' + '.join(s['segments'])}) | {s['ops']} | {s['kernel_us']} | {s['fw_us']} | " f"{s['op2op_us']} | {s['span_us']} |") for seg, e in res["eager"].items(): lines += ["", f"stages of `{seg}` (labels from the eager run; traced times " f"{'aligned op for op' if e['sequence_matches_trace'] else 'NOT aligned: eager times only'})", "", "| stage | ops | traced kernel us | traced op-to-op us | eager kernel us | eager op-to-op us | " "top op codes (us) |", "|---|---:|---:|---:|---:|---:|---|"] for name, st in e["stages"].items(): tops = ", ".join(f"{op} {t}" for op, t in list(st["by_op"].items())[:4]) lines.append(f"| {name} | {st['ops']} | {st['kernel_us']} | {st['op2op_us']} | {st['eager_kernel_us']} | " f"{st['eager_op2op_us']} | {tops} |") if res.get("groups"): lines += ["", "stage groups (traced kernel time when the eager sequence matches the trace)", "", "| group | ops | kernel us | top op codes (us) |", "|---|---:|---:|---|"] for g, v in res["groups"].items(): tops = ", ".join(f"{d['op']} {d['count']}x {d['kernel_us']}" for d in v["by_op"][:5]) lines.append(f"| {g} | {v['ops']} | {v['kernel_us']} | {tops} |") for seg in SEGMENTS: if seg in res["segments"]: lines += ["", f"op codes of `{seg}` (traced)", "", "| op code | count | kernel us | share |", "|---|---:|---:|---:|"] for b in res["segments"][seg]["by_op"][:12]: lines.append(f"| {b['op']} | {b['count']} | {b['kernel_us']} | {b['share']:.1%} |") lines += ["", f"top {a.top} ops by traced kernel time", "", "| # | segment | stage | op | kernel us | cores | fidelity | DRAM BW % | FPU % | inputs | output |", "|---:|---|---|---|---:|---:|---|---:|---:|---|---|"] for i, d in enumerate(res["top_ops"], 1): lines.append(f"| {i} | {d['segment']}[{d['index']}] | {d['stage']} | {d['op']} | {d['kernel_us']} | {d['cores']} | " f"{d['fidelity']} | {d['dram_bw_util']} | {d['fpu_util']} | {d['inputs']} | {d['outputs']} |") lines += ["", f"largest {a.top} op-to-op gaps (traced)", "", "| # | segment | stage | op | gap us | kernel us |", "|---:|---|---|---|---:|---:|"] for i, d in enumerate(res["top_gaps"], 1): lines.append(f"| {i} | {d['segment']}[{d['index']}] | {d['stage']} | {d['op']} | {d['op2op_us']} | {d['kernel_us']} |") text = "\n".join(lines) + "\n" if a.md: a.md.parent.mkdir(parents=True, exist_ok=True) a.md.write_text(text) print(text) return 0 if __name__ == "__main__": sys.exit(main())