* [PATCH 1/2] perf jevents: Add NVIDIA Tegra410 uncore DDR metrics
2026-09-28 17:51 [PATCH 0/2] perf jevents: Add NVIDIA Tegra410 uncore DDR and PCIe metrics Chun-Tse Shao
@ 2026-09-28 17:51 ` Chun-Tse Shao
2026-09-28 17:51 ` [PATCH 2/2] perf jevents: Add NVIDIA Tegra410 uncore PCIe metrics Chun-Tse Shao
1 sibling, 0 replies; 3+ messages in thread
From: Chun-Tse Shao @ 2026-09-28 17:51 UTC (permalink / raw)
To: acme, namhyung, irogers
Cc: peterz, mingo, mark.rutland, alexander.shishkin, jolsa,
adrian.hunter, james.clark, bwicaksono, linux-perf-users,
linux-kernel, Chun-Tse Shao
Add DDR bandwidth and latency metrics for the NVIDIA Tegra410 SoC. The
formulas follow Documentation/admin-guide/perf/nvidia-tegra410-pmu.rst:
AVG_MEM_READ_BANDWIDTH_IN_GBPS = MEM_BYTES_RD / ELAPSED_TIME_IN_NS
AVG_MEM_WRITE_BANDWIDTH_IN_GBPS = MEM_BYTES_WR / ELAPSED_TIME_IN_NS
FREQ_IN_GHZ = CYCLES / ELAPSED_TIME_IN_NS
AVG_LATENCY_IN_CYCLES = RD_CUM_OUTS / RD_REQ
AVERAGE_LATENCY_IN_NS = AVG_LATENCY_IN_CYCLES / FREQ_IN_GHZ
The bandwidth metrics use the mem_bytes_rd/mem_bytes_wr events of the
Unified Coherence Fabric (UCF) PMU with source/destination filters:
lpm_ddr_bw, lpm_ddr_rd_bw, lpm_ddr_wr_bw: all sources to local CMEM.
lpm_ddr_loc_rd_bw, lpm_ddr_loc_wr_bw: local CPU and non-CPU sources to
local CMEM.
lpm_ddr_rem_rd_bw, lpm_ddr_rem_wr_bw: local CPU and non-CPU sources to
remote memory.
The latency metrics use the CPU Memory (CMEM) latency PMU:
lpm_ddr_lat, lpm_ddr_rd_lat: average DDR read latency in ns.
lpm_ddr_wr_lat: the PMU only measures reads, so this is always 0.
There is one CMEM latency PMU per socket and perf sums the counts of the
PMUs in an aggregation, e.g. of both sockets in the default aggregation
mode. Use aggr_nr() to divide the summed cycles by the number of PMUs,
otherwise the frequency is too high and the latency too low by that
factor.
The events are sysfs events of the nvidia_ucf_pmu_<socket-id> and
nvidia_cmem_latency_pmu_<socket-id> PMUs rather than json events, so add
the "nvidia" PMU prefix to the prefixes that skip the named json event
check in metric.py.
pmu-events/Build only runs arm64_metrics.py for the "arm" vendor models,
so also run it for the "nvidia" vendor models to generate the Tegra410
metrics.
Signed-off-by: Chun-Tse Shao <ctshao@google.com>
Assisted-by: Gemini:gemini-3.1-pro-preview
---
tools/perf/pmu-events/Build | 2 +-
tools/perf/pmu-events/arm64_metrics.py | 66 +++++++++++++++++++++++++-
tools/perf/pmu-events/metric.py | 2 +-
3 files changed, 66 insertions(+), 4 deletions(-)
diff --git a/tools/perf/pmu-events/Build b/tools/perf/pmu-events/Build
index 1ac067be9d34..8dc8cfcb8b04 100644
--- a/tools/perf/pmu-events/Build
+++ b/tools/perf/pmu-events/Build
@@ -78,7 +78,7 @@ endif
ifeq ($(JEVENTS_ARCH),$(filter $(JEVENTS_ARCH),arm64 all))
# Generate ARM Json
-ARMS := $(shell ls -d pmu-events/arch/arm64/arm/*|grep -v cmn)
+ARMS := $(shell ls -d pmu-events/arch/arm64/arm/* pmu-events/arch/arm64/nvidia/*|grep -v cmn)
ARM_METRICS = $(foreach x,$(ARMS),$(OUTPUT)$(x)/extra-metrics.json)
ARM_METRICGROUPS = $(foreach x,$(ARMS),$(OUTPUT)$(x)/extra-metricgroups.json)
GEN_JSON += $(ARM_METRICS) $(ARM_METRICGROUPS)
diff --git a/tools/perf/pmu-events/arm64_metrics.py b/tools/perf/pmu-events/arm64_metrics.py
index 4ecda96d11fa..b0d0651fec12 100755
--- a/tools/perf/pmu-events/arm64_metrics.py
+++ b/tools/perf/pmu-events/arm64_metrics.py
@@ -2,14 +2,75 @@
# SPDX-License-Identifier: (LGPL-2.1 OR BSD-2-Clause)
import argparse
import os
-from metric import (JsonEncodeMetric, JsonEncodeMetricGroupDescriptions, LoadEvents,
- MetricGroup)
+from typing import Optional
+from metric import (aggr_nr, d_ratio, Event, JsonEncodeMetric,
+ JsonEncodeMetricGroupDescriptions, LoadEvents, Metric, MetricGroup)
from common_metrics import Cycles
# Global command line arguments.
_args = None
+def NvidiaT410Uncore() -> Optional[MetricGroup]:
+ """Uncore metrics for the NVIDIA Tegra410 SoC.
+
+ The formulas follow Documentation/admin-guide/perf/nvidia-tegra410-pmu.rst.
+ """
+ assert _args is not None
+ if _args.vendor != "nvidia" or _args.model != "t410":
+ return None
+
+ interval_sec = Event("duration_time")
+ mb_per_sec = f"{1/1e6}MB/s"
+
+ def DdrBw() -> MetricGroup:
+ # Unified Coherence Fabric (UCF) PMU:
+ # AVG_MEM_{READ,WRITE}_BANDWIDTH = MEM_BYTES_{RD,WR} / ELAPSED_TIME
+ loc_src = "src_loc_cpu=0x1,src_loc_noncpu=0x1"
+ rd = Event("nvidia_ucf_pmu/mem_bytes_rd,dst_loc_cmem=0x1/")
+ wr = Event("nvidia_ucf_pmu/mem_bytes_wr,dst_loc_cmem=0x1/")
+ loc_rd = Event(f"nvidia_ucf_pmu/mem_bytes_rd,{loc_src},dst_loc_cmem=0x1/")
+ loc_wr = Event(f"nvidia_ucf_pmu/mem_bytes_wr,{loc_src},dst_loc_cmem=0x1/")
+ rem_rd = Event(f"nvidia_ucf_pmu/mem_bytes_rd,{loc_src},dst_rem=0x1/")
+ rem_wr = Event(f"nvidia_ucf_pmu/mem_bytes_wr,{loc_src},dst_rem=0x1/")
+ return MetricGroup("lpm_ddr_bw", [
+ Metric("lpm_ddr_bw", "Total DDR bandwidth",
+ d_ratio(rd + wr, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_rd_bw", "DDR read bandwidth",
+ d_ratio(rd, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_wr_bw", "DDR write bandwidth",
+ d_ratio(wr, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_loc_rd_bw", "Local DDR read bandwidth",
+ d_ratio(loc_rd, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_loc_wr_bw", "Local DDR write bandwidth",
+ d_ratio(loc_wr, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_rem_rd_bw", "Remote DDR read bandwidth",
+ d_ratio(rem_rd, interval_sec), mb_per_sec),
+ Metric("lpm_ddr_rem_wr_bw", "Remote DDR write bandwidth",
+ d_ratio(rem_wr, interval_sec), mb_per_sec),
+ ], description="NVIDIA Tegra410 DDR bandwidth")
+
+ def DdrLat() -> MetricGroup:
+ # CPU Memory (CMEM) latency PMU, one per socket:
+ # AVERAGE_LATENCY_IN_NS = (RD_CUM_OUTS / RD_REQ) / (CYCLES / ELAPSED_TIME_IN_NS)
+ rd_cum_outs = Event("nvidia_cmem_latency_pmu/rd_cum_outs/")
+ rd_req = Event("nvidia_cmem_latency_pmu/rd_req/")
+ cycles = Event("nvidia_cmem_latency_pmu/cycles/")
+ # perf sums the counts of the PMUs in an aggregation, e.g. both sockets
+ # by default, so divide cycles by their number to get the frequency.
+ rd_lat = d_ratio(1e9 * interval_sec * rd_cum_outs * aggr_nr(cycles), rd_req * cycles)
+ return MetricGroup("lpm_ddr_lat", [
+ Metric("lpm_ddr_lat", "Average DDR read latency in ns", rd_lat, "1ns"),
+ Metric("lpm_ddr_rd_lat", "DDR read latency in ns", rd_lat, "1ns"),
+ # The CMEM latency PMU only measures reads, write latency is always 0.
+ Metric("lpm_ddr_wr_lat", "DDR write latency in ns",
+ rd_cum_outs - rd_cum_outs, "1ns"),
+ ], description="NVIDIA Tegra410 DDR latency")
+
+ return MetricGroup("lpm_uncore", [DdrBw(), DdrLat()],
+ description="NVIDIA Tegra410 uncore metrics")
+
+
def main() -> None:
global _args
@@ -37,6 +98,7 @@ def main() -> None:
all_metrics = MetricGroup("", [
Cycles(),
+ NvidiaT410Uncore(),
])
if _args.metricgroups:
diff --git a/tools/perf/pmu-events/metric.py b/tools/perf/pmu-events/metric.py
index f5d81eccd5c3..6fb69024d1fd 100644
--- a/tools/perf/pmu-events/metric.py
+++ b/tools/perf/pmu-events/metric.py
@@ -91,7 +91,7 @@ def CheckEveryEvent(*names: str) -> None:
name = name[:name.find(':')]
elif '/' in name:
name = name[:name.find('/')]
- if any([name.startswith(x) for x in ['amd', 'arm', 'cpu', 'msr', 'power', 'cha', 'uncore']]):
+ if any([name.startswith(x) for x in ['amd', 'arm', 'cpu', 'msr', 'power', 'cha', 'uncore', 'nvidia']]):
continue
if name not in all_events_all_models:
raise ValueError(f"Is {name} a named json event?")
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply [flat|nested] 3+ messages in thread* [PATCH 2/2] perf jevents: Add NVIDIA Tegra410 uncore PCIe metrics
2026-09-28 17:51 [PATCH 0/2] perf jevents: Add NVIDIA Tegra410 uncore DDR and PCIe metrics Chun-Tse Shao
2026-09-28 17:51 ` [PATCH 1/2] perf jevents: Add NVIDIA Tegra410 uncore DDR metrics Chun-Tse Shao
@ 2026-09-28 17:51 ` Chun-Tse Shao
1 sibling, 0 replies; 3+ messages in thread
From: Chun-Tse Shao @ 2026-09-28 17:51 UTC (permalink / raw)
To: acme, namhyung, irogers
Cc: peterz, mingo, mark.rutland, alexander.shishkin, jolsa,
adrian.hunter, james.clark, bwicaksono, linux-perf-users,
linux-kernel, Chun-Tse Shao
Add PCIe bandwidth metrics for the NVIDIA Tegra410 SoC. The formulas
follow Documentation/admin-guide/perf/nvidia-tegra410-pmu.rst:
AVG_RD_BANDWIDTH_IN_GBPS = RD_BYTES / ELAPSED_TIME_IN_NS
AVG_WR_BANDWIDTH_IN_GBPS = WR_BYTES / ELAPSED_TIME_IN_NS
There is one PCIE PMU per socket and PCIe Root Complex (RC), named
nvidia_pcie_pmu_<socket-id>_rc_<pcie-rc-id>, with RCs 0 to 5 in each
socket. The metrics sum the rd_bytes/wr_bytes events of both sockets:
lpm_pcie_bw, lpm_pcie_rd_bw, lpm_pcie_wr_bw: all RCs.
lpm_pcie_bw_<rc>, lpm_pcie_rd_bw_<rc>, lpm_pcie_wr_bw_<rc>: one RC.
lpm_pcie_loc_rd_bw_<rc>, lpm_pcie_loc_wr_bw_<rc>: one RC to local
CMEM, GMEM, PCIe peer and CXL memory.
lpm_pcie_rem_rd_bw_<rc>, lpm_pcie_rem_wr_bw_<rc>: one RC to remote
memory.
The per RC metrics are in the lpm_pcie_rc<rc>_bw metric groups.
A socket's event counts as 0 if its PMU isn't present (has_event() is
0), e.g. on a single socket system, or if it isn't counted in the
aggregation (source_count() is 0), e.g. the other socket's event with
--per-socket. Otherwise the metric fails to parse or is nan.
Signed-off-by: Chun-Tse Shao <ctshao@google.com>
Assisted-by: Gemini:gemini-3.1-pro-preview
---
tools/perf/pmu-events/arm64_metrics.py | 68 ++++++++++++++++++++++++--
1 file changed, 64 insertions(+), 4 deletions(-)
diff --git a/tools/perf/pmu-events/arm64_metrics.py b/tools/perf/pmu-events/arm64_metrics.py
index b0d0651fec12..1ca3d8ddb4b7 100755
--- a/tools/perf/pmu-events/arm64_metrics.py
+++ b/tools/perf/pmu-events/arm64_metrics.py
@@ -2,9 +2,10 @@
# SPDX-License-Identifier: (LGPL-2.1 OR BSD-2-Clause)
import argparse
import os
-from typing import Optional
-from metric import (aggr_nr, d_ratio, Event, JsonEncodeMetric,
- JsonEncodeMetricGroupDescriptions, LoadEvents, Metric, MetricGroup)
+from typing import List, Optional, Union
+from metric import (aggr_nr, d_ratio, has_event, source_count, Event, Expression,
+ JsonEncodeMetric, JsonEncodeMetricGroupDescriptions, LoadEvents, Metric,
+ MetricGroup, Select)
from common_metrics import Cycles
# Global command line arguments.
@@ -67,7 +68,66 @@ def NvidiaT410Uncore() -> Optional[MetricGroup]:
rd_cum_outs - rd_cum_outs, "1ns"),
], description="NVIDIA Tegra410 DDR latency")
- return MetricGroup("lpm_uncore", [DdrBw(), DdrLat()],
+ def PcieBw() -> MetricGroup:
+ # PCIE PMU, one per socket and PCIe Root Complex (RC):
+ # AVG_{RD,WR}_BANDWIDTH = {RD,WR}_BYTES / ELAPSED_TIME
+ loc_dst = "dst_loc_cmem=0x1,dst_loc_gmem=0x1,dst_loc_pcie_p2p=0x1,dst_loc_pcie_cxl=0x1"
+ rem_dst = "dst_rem=0x1"
+
+ def SocketSum(rc: str, event: str) -> Expression:
+ # Sum the event over both sockets. A PMU counts as 0 if it isn't
+ # present (has_event() is 0) or isn't counted in the aggregation,
+ # e.g. the other socket with --per-socket (source_count() is 0).
+ def Count(e: Event) -> Expression:
+ return Select(Select(e, source_count(e), 0), has_event(e), 0)
+
+ return (Count(Event(f"nvidia_pcie_pmu_0_{rc}/{event}/")) +
+ Count(Event(f"nvidia_pcie_pmu_1_{rc}/{event}/")))
+
+ # nvidia_pcie_pmu_<socket-id>_rc matches the PMUs of all RCs in the socket.
+ rd = SocketSum("rc", "rd_bytes")
+ wr = SocketSum("rc", "wr_bytes")
+ metrics: List[Union[Metric, MetricGroup]] = [
+ Metric("lpm_pcie_bw", "Total PCIe bandwidth across all Root Complexes",
+ d_ratio(rd + wr, interval_sec), mb_per_sec),
+ Metric("lpm_pcie_rd_bw", "Total PCIe read bandwidth across all Root Complexes",
+ d_ratio(rd, interval_sec), mb_per_sec),
+ Metric("lpm_pcie_wr_bw", "Total PCIe write bandwidth across all Root Complexes",
+ d_ratio(wr, interval_sec), mb_per_sec),
+ ]
+ for i in range(6):
+ rc = f"rc_{i}"
+ rc_rd = SocketSum(rc, "rd_bytes")
+ rc_wr = SocketSum(rc, "wr_bytes")
+ loc_rd = SocketSum(rc, f"rd_bytes,{loc_dst}")
+ loc_wr = SocketSum(rc, f"wr_bytes,{loc_dst}")
+ rem_rd = SocketSum(rc, f"rd_bytes,{rem_dst}")
+ rem_wr = SocketSum(rc, f"wr_bytes,{rem_dst}")
+ metrics.append(MetricGroup(f"lpm_pcie_rc{i}_bw", [
+ Metric(f"lpm_pcie_bw_{i}", f"PCIe total bandwidth on Root Complex {i}",
+ d_ratio(rc_rd + rc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rd_bw_{i}", f"PCIe read bandwidth on Root Complex {i}",
+ d_ratio(rc_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_wr_bw_{i}", f"PCIe write bandwidth on Root Complex {i}",
+ d_ratio(rc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_loc_rd_bw_{i}",
+ f"PCIe local read bandwidth on Root Complex {i}",
+ d_ratio(loc_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_loc_wr_bw_{i}",
+ f"PCIe local write bandwidth on Root Complex {i}",
+ d_ratio(loc_wr, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rem_rd_bw_{i}",
+ f"PCIe remote read bandwidth on Root Complex {i}",
+ d_ratio(rem_rd, interval_sec), mb_per_sec),
+ Metric(f"lpm_pcie_rem_wr_bw_{i}",
+ f"PCIe remote write bandwidth on Root Complex {i}",
+ d_ratio(rem_wr, interval_sec), mb_per_sec),
+ ], description=f"PCIe bandwidth on Root Complex {i}"))
+
+ return MetricGroup("lpm_pcie_bw", metrics,
+ description="NVIDIA Tegra410 PCIe Root Complex bandwidth")
+
+ return MetricGroup("lpm_uncore", [DdrBw(), DdrLat(), PcieBw()],
description="NVIDIA Tegra410 uncore metrics")
--
2.56.0.rc1.315.gc6ed9934b7-goog
^ permalink raw reply [flat|nested] 3+ messages in thread