[PATCH 2/2] perf jevents: Add NVIDIA Tegra410 uncore PCIe metrics

From: Chun-Tse Shao

Date: Mon Sep 28 2026 - 13:51:55 EST


Add PCIe bandwidth metrics for the NVIDIA Tegra410 SoC. The formulas
follow Documentation/admin-guide/perf/nvidia-tegra410-pmu.rst:

AVG_RD_BANDWIDTH_IN_GBPS = RD_BYTES / ELAPSED_TIME_IN_NS
AVG_WR_BANDWIDTH_IN_GBPS = WR_BYTES / ELAPSED_TIME_IN_NS

There is one PCIE PMU per socket and PCIe Root Complex (RC), named
nvidia_pcie_pmu_<socket-id>_rc_<pcie-rc-id>, with RCs 0 to 5 in each
socket. The metrics sum the rd_bytes/wr_bytes events of both sockets:

lpm_pcie_bw, lpm_pcie_rd_bw, lpm_pcie_wr_bw: all RCs.
lpm_pcie_bw_<rc>, lpm_pcie_rd_bw_<rc>, lpm_pcie_wr_bw_<rc>: one RC.
lpm_pcie_loc_rd_bw_<rc>, lpm_pcie_loc_wr_bw_<rc>: one RC to local
CMEM, GMEM, PCIe peer and CXL memory.
lpm_pcie_rem_rd_bw_<rc>, lpm_pcie_rem_wr_bw_<rc>: one RC to remote
memory.

The per RC metrics are in the lpm_pcie_rc<rc>_bw metric groups.

A socket's event counts as 0 if its PMU isn't present (has_event() is
0), e.g. on a single socket system, or if it isn't counted in the
aggregation (source_count() is 0), e.g. the other socket's event with
--per-socket. Otherwise the metric fails to parse or is nan.

Signed-off-by: Chun-Tse Shao <ctshao@xxxxxxxxxx>
Assisted-by: Gemini:gemini-3.1-pro-preview
---
tools/perf/pmu-events/arm64_metrics.py | 68 ++++++++++++++++++++++++--
1 file changed, 64 insertions(+), 4 deletions(-)

diff --git a/tools/perf/pmu-events/arm64_metrics.py b/tools/perf/pmu-events/arm64_metrics.py
index b0d0651fec12..1ca3d8ddb4b7 100755
--- a/tools/perf/pmu-events/arm64_metrics.py
+++ b/tools/perf/pmu-events/arm64_metrics.py
@@ -2,9 +2,10 @@
# SPDX-License-Identifier: (LGPL-2.1 OR BSD-2-Clause)
import argparse
import os
-from typing import Optional
-from metric import (aggr_nr, d_ratio, Event, JsonEncodeMetric,
- JsonEncodeMetricGroupDescriptions, LoadEvents, Metric, MetricGroup)
+from typing import List, Optional, Union
+from metric import (aggr_nr, d_ratio, has_event, source_count, Event, Expression,
+ JsonEncodeMetric, JsonEncodeMetricGroupDescriptions, LoadEvents, Metric,
+ MetricGroup, Select)
from common_metrics import Cycles

# Global command line arguments.
@@ -67,7 +68,66 @@ def NvidiaT410Uncore() -> Optional[MetricGroup]:
rd_cum_outs - rd_cum_outs, "1ns"),
], description="NVIDIA Tegra410 DDR latency")

- return MetricGroup("lpm_uncore", [DdrBw(), DdrLat()],
+ def PcieBw() -> MetricGroup:
+ # PCIE PMU, one per socket and PCIe Root Complex (RC):
+ # AVG_{RD,WR}_BANDWIDTH = {RD,WR}_BYTES / ELAPSED_TIME
+ loc_dst = "dst_loc_cmem=0x1,dst_loc_gmem=0x1,dst_loc_pcie_p2p=0x1,dst_loc_pcie_cxl=0x1"
+ rem_dst = "dst_rem=0x1"
+
+ def SocketSum(rc: str, event: str) -> Expression:
+ # Sum the event over both sockets. A PMU counts as 0 if it isn't
+ # present (has_event() is 0) or isn't counted in the aggregation,
+ # e.g. the other socket with --per-socket (source_count() is 0).
+ def Count(e: Event) -> Expression:
+ return Select(Select(e, source_count(e), 0), has_event(e), 0)
+
+ return (Count(Event(f"nvidia_pcie_pmu_0_{rc}/{event}/")) +
+ Count(Event(f"nvidia_pcie_pmu_1_{rc}/{event}/")))
+
+ # nvidia_pcie_pmu_<socket-id>_rc matches the PMUs of all RCs in the socket.
+ rd = SocketSum("rc", "rd_bytes")
+ wr = SocketSum("rc", "wr_bytes")
+ metrics: List[Union[Metric, MetricGroup]] = [
+ Metric("lpm_pcie_bw", "Total PCIe bandwidth across all Root Complexes",
+ d_ratio(rd + wr, interval_sec), mb_per_sec),
+ Metric("lpm_pcie_rd_bw", "Total PCIe read bandwidth across all Root Complexes",
+ d_ratio(rd, interval_sec), mb_per_sec),
+ Metric("lpm_pcie_wr_bw", "Total PCIe write bandwidth across all Root Complexes",
+ d_ratio(wr, interval_sec), mb_per_sec),
+ ]
+ for i in range(6):
+ rc = f"rc_{i}"
+ rc_rd = SocketSum(rc, "rd_bytes")
+ rc_wr = SocketSum(rc, "wr_bytes")
+ loc_rd = SocketSum(rc, f"rd_bytes,{loc_dst}")
+ loc_wr = SocketSum(rc, f"wr_bytes,{loc_dst}")
+ rem_rd = SocketSum(rc, f"rd_bytes,{rem_dst}")
+ rem_wr = SocketSum(rc, f"wr_bytes,{rem_dst}")
+ metrics.append(MetricGroup(f"lpm_pcie_rc{i}_bw", [
+ Metric(f"lpm_pcie_bw_{i}", f"PCIe total bandwidth on Root Complex {i}",
+ d_ratio(rc_rd + rc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rd_bw_{i}", f"PCIe read bandwidth on Root Complex {i}",
+ d_ratio(rc_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_wr_bw_{i}", f"PCIe write bandwidth on Root Complex {i}",
+ d_ratio(rc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_loc_rd_bw_{i}",
+ f"PCIe local read bandwidth on Root Complex {i}",
+ d_ratio(loc_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_loc_wr_bw_{i}",
+ f"PCIe local write bandwidth on Root Complex {i}",
+ d_ratio(loc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rem_rd_bw_{i}",
+ f"PCIe remote read bandwidth on Root Complex {i}",
+ d_ratio(rem_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rem_wr_bw_{i}",
+ f"PCIe remote write bandwidth on Root Complex {i}",
+ d_ratio(rem_wr, interval_sec), mb_per_sec),
+ ], description=f"PCIe bandwidth on Root Complex {i}"))
+
+ return MetricGroup("lpm_pcie_bw", metrics,
+ description="NVIDIA Tegra410 PCIe Root Complex bandwidth")
+
+ return MetricGroup("lpm_uncore", [DdrBw(), DdrLat(), PcieBw()],
description="NVIDIA Tegra410 uncore metrics")


--
2.56.0.rc1.315.gc6ed9934b7-goog