mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Ian Rogers <irogers@google.com>
To: Peter Zijlstra <peterz@infradead.org>,
	Ingo Molnar <mingo@redhat.com>,
	 Arnaldo Carvalho de Melo <acme@kernel.org>,
	Namhyung Kim <namhyung@kernel.org>, Jiri Olsa <jolsa@kernel.org>,
	 Ian Rogers <irogers@google.com>,
	Adrian Hunter <adrian.hunter@intel.com>,
	 James Clark <james.clark@linaro.org>,
	Thomas Falcon <thomas.falcon@intel.com>,
	 Alice Rogers <alice.mei.rogers@gmail.com>,
	Changbin Du <changbin.du@huawei.com>,
	 Tengda Wu <wutengda@huaweicloud.com>, tanze <tanze@kylinos.cn>,
	 Athira Rajeev <atrajeev@linux.ibm.com>,
	Dapeng Mi <dapeng1.mi@linux.intel.com>,
	 linux-kernel@vger.kernel.org, linux-perf-users@vger.kernel.org
Subject: [PATCH v1 04/13] perf python: Lazily resolve sample callchains
Date: Fri,  2 Oct 2026 11:26:13 -0700	[thread overview]
Message-ID: <20261002182624.3259797-5-irogers@google.com> (raw)
In-Reply-To: <20261002182624.3259797-1-irogers@google.com>

Previously, pyrf_event__new() eagerly resolved the sample's address
location and callchain (including DWARF/FP unwinding and symbol lookup)
for every sample with a callchain, even when the Python callback never
accessed event.callchain (such as in ttimechart.py).

Move callchain resolution to pyrf_sample_event__get_callchain() so that
callchains are only unwound and resolved when event.callchain is
accessed, reusing the sample's cached pevent->al via
pyrf_sample_event__resolve_al().

Because process_events() borrows the event and sample during the
callback, callchains are never copied in the common case. Only when a
callback retains a reference to the event beyond its return
(Py_REFCNT(pyevent) > 1), the callchain was not already resolved during
the callback, and the callchain is synthesized or LBR-merged (not backed
by event_copy) does pyrf_event__copy() allocate a copy of
sample->callchain.

Assisted-by: Antigravity:gemini-3.1-pro
Signed-off-by: Ian Rogers <irogers@google.com>
---
 tools/perf/util/python.c | 160 +++++++++++++++++++--------------------
 1 file changed, 79 insertions(+), 81 deletions(-)

diff --git a/tools/perf/util/python.c b/tools/perf/util/python.c
index a40e89c1b329..251b7d517ad5 100644
--- a/tools/perf/util/python.c
+++ b/tools/perf/util/python.c
@@ -100,7 +100,7 @@ struct pyrf_event {
 	struct addr_location al;
 	/** @al_resolved: True when machine__resolve been called. */
 	bool al_resolved;
-	/** @callchain: Resolved callchain, eagerly computed if requested. */
+	/** @callchain: Resolved callchain, lazily computed. */
 	PyObject *callchain;
 	/** @brstack: Resolved branch stack, eagerly computed if requested. */
 	PyObject *brstack;
@@ -727,28 +727,19 @@ get_tracepoint_field(struct pyrf_event *pevent, PyObject *attr_name)
 
 static int pyrf_sample_event__resolve_al(struct pyrf_event *pevent)
 {
-	struct evsel *evsel = pevent->sample->evsel ?: pevent->evsel;
-	struct evlist *evlist = evsel ? evsel->evlist : NULL;
-	struct perf_session *session = evlist ? evlist__session(evlist) : NULL;
-	struct machine *machine;
-
 	if (pevent->al_resolved)
 		return 0;
 
-	if (!session)
-		return -1;
-
 	/*
-	 * Use pevent->machine, which pyrf_event__new() initializes either from
-	 * the machine resolved by perf_session (machines__find_for_cpumode(),
-	 * preserving DEFAULT_GUEST_KERNEL_ID == 0 for default guests) or via
-	 * sample.machine_pid when pyrf_event__new() is called with a NULL
-	 * machine.
+	 * Use pevent->machine, which pyrf_event__new() initializes while the
+	 * session callback is active and pyrf_event__copy() clears after
+	 * resolving pevent->al if a reference is retained beyond the callback.
 	 */
-	machine = pevent->machine ? pevent->machine : &session->machines.host;
+	if (!pevent->machine)
+		return -1;
 
 	addr_location__init(&pevent->al);
-	if (machine__resolve(machine, &pevent->al, pevent->sample) < 0) {
+	if (machine__resolve(pevent->machine, &pevent->al, pevent->sample) < 0) {
 		addr_location__exit(&pevent->al);
 		return -1;
 	}
@@ -924,23 +915,15 @@ static PyObject *pyrf_sample_event__srccode(PyObject *self, PyObject *args)
 static PyObject *pyrf_sample_event__insn(PyObject *self, PyObject *args __maybe_unused)
 {
 	struct pyrf_event *pevent = (void *)self;
-	struct thread *thread;
-	struct machine *machine;
 
-	if (pyrf_sample_event__resolve_al(pevent) < 0)
-		Py_RETURN_NONE;
-
-	thread = pevent->al.thread;
-
-	if (!thread || !thread__maps(thread))
-		Py_RETURN_NONE;
-
-	machine = maps__machine(thread__maps(thread));
-	if (!machine)
-		Py_RETURN_NONE;
+	if (pevent->sample->ip && !pevent->sample->insn_len) {
+		if (pyrf_sample_event__resolve_al(pevent) < 0 ||
+		    !pevent->al.thread || !pevent->machine)
+			Py_RETURN_NONE;
 
-	if (pevent->sample->ip && !pevent->sample->insn_len)
-		perf_sample__fetch_insn(pevent->sample, thread, machine);
+		perf_sample__fetch_insn(pevent->sample, pevent->al.thread,
+					pevent->machine);
+	}
 
 	if (!pevent->sample->insn_len)
 		Py_RETURN_NONE;
@@ -1097,10 +1080,53 @@ static PyTypeObject pyrf_callchain__type = {
 	.tp_as_sequence	= &pyrf_callchain__sequence_methods,
 };
 
+static int pyrf_sample_event__resolve_callchain(struct pyrf_event *pevent)
+{
+	struct callchain_cursor *cursor;
+	struct pyrf_callchain *pchain;
+	struct callchain_cursor_node *node;
+
+	if (pevent->callchain || !pevent->sample->callchain || !pevent->machine)
+		return 0;
+
+	if (pyrf_sample_event__resolve_al(pevent) < 0)
+		return 0;
+
+	cursor = get_tls_callchain_cursor();
+	if (thread__resolve_callchain(pevent->al.thread, cursor, pevent->sample,
+				      NULL, NULL, PERF_MAX_STACK_DEPTH) != 0)
+		return 0;
+
+	callchain_cursor_commit(cursor);
+	pchain = PyObject_New(struct pyrf_callchain, &pyrf_callchain__type);
+	if (!pchain)
+		return -ENOMEM;
+
+	pchain->nr_frames = cursor->nr;
+	pchain->frames = calloc(pchain->nr_frames, sizeof(*pchain->frames));
+	if (!pchain->frames) {
+		Py_DECREF(pchain);
+		PyErr_NoMemory();
+		return -ENOMEM;
+	}
+	for (u64 i = 0; i < pchain->nr_frames; i++) {
+		node = callchain_cursor_current(cursor);
+		pchain->frames[i].ip = node->ip;
+		pchain->frames[i].map = map__get(node->ms.map);
+		pchain->frames[i].sym = node->ms.sym;
+		callchain_cursor_advance(cursor);
+	}
+	pevent->callchain = (PyObject *)pchain;
+	return 0;
+}
+
 static PyObject *pyrf_sample_event__get_callchain(PyObject *self, void *closure __maybe_unused)
 {
 	struct pyrf_event *pevent = (void *)self;
 
+	if (pyrf_sample_event__resolve_callchain(pevent) < 0)
+		return NULL;
+
 	if (!pevent->callchain)
 		Py_RETURN_NONE;
 
@@ -1646,13 +1672,31 @@ static int pyrf_event__copy(struct pyrf_event *pevent)
 	size_t copy_size = orig_event->header.size;
 	int err = -EINVAL;
 
+	/*
+	 * If a session callback retained a reference to pevent, resolve its
+	 * address location and callchain now while pevent->machine is still
+	 * live and thread->maps reflects the point in time of the sample.
+	 */
+	if (pevent->machine && pevent->evsel) {
+		pyrf_sample_event__resolve_al(pevent);
+		if (pyrf_sample_event__resolve_callchain(pevent) < 0) {
+			pevent->machine = NULL;
+			if (!pevent->event_copy) {
+				pevent->event = &zero_event;
+				pevent->sample = &pevent->sample_storage;
+			}
+			return -ENOMEM;
+		}
+	}
+	pevent->machine = NULL;
+
 	if (pevent->event_copy)
 		return 0;
 
 	/*
 	 * Clear borrowed pointers immediately so that even if copying fails,
-	 * pevent does not retain dangling pointers to the caller's stack or
-	 * ring buffer.
+	 * pevent does not retain dangling pointers to the caller's stack,
+	 * ring buffer, or session.
 	 */
 	pevent->event = &zero_event;
 	pevent->sample = &pevent->sample_storage;
@@ -1759,7 +1803,7 @@ static PyObject *pyrf_event__new(const union perf_event *event, struct evsel *ev
 	perf_sample__init(&pevent->sample_storage, /*all=*/true);
 	pevent->sample = sample_arg ?: &pevent->sample_storage;
 	pevent->evsel = evsel ? evsel__get(evsel) : NULL;
-	pevent->machine = machine;
+	pevent->machine = NULL;
 	pevent->callchain = NULL;
 	pevent->brstack = NULL;
 	pevent->al_resolved = false;
@@ -1777,11 +1821,8 @@ static PyObject *pyrf_event__new(const union perf_event *event, struct evsel *ev
 		return NULL;
 	}
 
-	if (!evsel) {
-		if (!pevent->machine && session)
-			pevent->machine = &session->machines.host;
+	if (!evsel)
 		return (PyObject *)pevent;
-	}
 
 	sample = pevent->sample;
 	if (session && session->evlist && perf_guest && sample->id) {
@@ -1802,49 +1843,6 @@ static PyObject *pyrf_event__new(const union perf_event *event, struct evsel *ev
 	pevent->machine = machine;
 	if (machine && machine->pid > 0 && !sample->machine_pid)
 		sample->machine_pid = machine->pid;
-	if (machine && sample->callchain) {
-		struct addr_location al;
-		struct callchain_cursor *cursor;
-		u64 i;
-		struct pyrf_callchain *pchain;
-
-		addr_location__init(&al);
-		if (machine__resolve(machine, &al, sample) >= 0) {
-			cursor = get_tls_callchain_cursor();
-			if (thread__resolve_callchain(al.thread, cursor, sample,
-						      NULL, NULL, PERF_MAX_STACK_DEPTH) == 0) {
-				callchain_cursor_commit(cursor);
-
-				pchain = PyObject_New(struct pyrf_callchain, &pyrf_callchain__type);
-				if (!pchain) {
-					addr_location__exit(&al);
-					Py_DECREF(pevent);
-					return NULL;
-				}
-				pchain->nr_frames = cursor->nr;
-				pchain->frames = calloc(pchain->nr_frames,
-							sizeof(*pchain->frames));
-				if (!pchain->frames) {
-					Py_DECREF(pchain);
-					addr_location__exit(&al);
-					Py_DECREF(pevent);
-					return PyErr_NoMemory();
-				}
-				struct callchain_cursor_node *node;
-
-				for (i = 0; i < pchain->nr_frames; i++) {
-					node = callchain_cursor_current(cursor);
-					pchain->frames[i].ip = node->ip;
-					pchain->frames[i].map =
-						map__get(node->ms.map);
-					pchain->frames[i].sym = node->ms.sym;
-					callchain_cursor_advance(cursor);
-				}
-				pevent->callchain = (PyObject *)pchain;
-			}
-			addr_location__exit(&al);
-		}
-	}
 	if (sample->branch_stack) {
 		struct branch_stack *bs = sample->branch_stack;
 		struct branch_entry *entries = perf_sample__branch_entries(sample);
-- 
2.56.0.rc1.315.gc6ed9934b7-goog


  parent reply	other threads:[~2026-10-02 18:26 UTC|newest]

Thread overview: 19+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-02 18:26 [PATCH v1 00/13] perf timechart/list/treport: Interactive Textual TUIs and perf python session improvements Ian Rogers
2026-10-02 18:26 ` [PATCH v1 01/13] perf session: Don't flush remaining events once processing is done Ian Rogers
2026-10-02 18:26 ` [PATCH v1 02/13] perf python: Quietly stop processing events when a callback raises Ian Rogers
2026-10-02 18:26 ` [PATCH v1 03/13] perf python: Lazily copy events and samples from process_events Ian Rogers
2026-10-02 18:26 ` Ian Rogers [this message]
2026-10-02 18:26 ` [PATCH v1 05/13] perf python: Release the GIL while processing session events Ian Rogers
2026-10-02 18:26 ` [PATCH v1 06/13] perf list: Add a --tui option to launch ilist Ian Rogers
2026-10-02 18:26 ` [PATCH v1 07/13] perf test: Add a test for the ilist script Ian Rogers
2026-10-02 18:26 ` [PATCH v1 08/13] perf treport: Show the profile while it loads Ian Rogers
2026-10-03  9:31   ` Arnaldo Carvalho de Melo
2026-10-02 18:26 ` [PATCH v1 09/13] perf test: Add a test for the treport script Ian Rogers
2026-10-03  9:35   ` Arnaldo Carvalho de Melo
2026-10-02 18:26 ` [PATCH v1 10/13] perf timechart: Add an interactive --tui mode Ian Rogers
2026-10-02 18:26 ` [PATCH v1 11/13] perf test: Add a test for perf timechart --tui Ian Rogers
2026-10-02 18:26 ` [PATCH v1 12/13] perf timechart: Add a --live mode to the TUI Ian Rogers
2026-10-02 18:26 ` [PATCH v1 13/13] perf test: Test perf timechart --live Ian Rogers
2026-10-03  9:10 ` [PATCH v1 00/13] perf timechart/list/treport: Interactive Textual TUIs and perf python session improvements Arnaldo Carvalho de Melo
2026-10-03  9:12   ` Arnaldo Carvalho de Melo
2026-10-03  9:15     ` Arnaldo Carvalho de Melo

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261002182624.3259797-5-irogers@google.com \
    --to=irogers@google.com \
    --cc=acme@kernel.org \
    --cc=adrian.hunter@intel.com \
    --cc=alice.mei.rogers@gmail.com \
    --cc=atrajeev@linux.ibm.com \
    --cc=changbin.du@huawei.com \
    --cc=dapeng1.mi@linux.intel.com \
    --cc=james.clark@linaro.org \
    --cc=jolsa@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-perf-users@vger.kernel.org \
    --cc=mingo@redhat.com \
    --cc=namhyung@kernel.org \
    --cc=peterz@infradead.org \
    --cc=tanze@kylinos.cn \
    --cc=thomas.falcon@intel.com \
    --cc=wutengda@huaweicloud.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®