* [PATCH perf-tools-next v2 1/4] perf ftrace: Optimise __cmd_ftrace() stack frame with strbuf streaming
2026-10-06 23:27 [PATCH perf-tools-next v2 0/4] perf ftrace: Support inlined functions and display enhancements in function graph tracer Aaron Tomlin
@ 2026-10-06 23:27 ` Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 2/4] perf ftrace: Support filtering return address comments Aaron Tomlin
` (2 subsequent siblings)
3 siblings, 0 replies; 5+ messages in thread
From: Aaron Tomlin @ 2026-10-06 23:27 UTC (permalink / raw)
To: peterz, mingo, acme, namhyung
Cc: mark.rutland, alexander.shishkin, jolsa, irogers, adrian.hunter,
atomlin, linux-perf-users, linux-kernel
In __cmd_ftrace(), incoming trace data from trace_pipe was historically
buffered in a fixed 4096-byte array allocated directly on the stack
frame. Furthermore, reads were written directly to stdout via fwrite()
in arbitrary chunk sizes, rather than as discrete, well-delimited
trace lines.
Refactor __cmd_ftrace() to employ dynamic heap allocation for the read
buffer alongside a dynamic struct strbuf to accumulate streamed
characters into whole lines. Pre-allocating the strbuf ensures that
line processing incurs zero heap reallocations during steady-state
polling, while the stack frame of __cmd_ftrace() is reduced by over 96%
(down to ~160 bytes).
Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
tools/perf/builtin-ftrace.c | 52 ++++++++++++++++++++++++++++++-------
1 file changed, 43 insertions(+), 9 deletions(-)
diff --git a/tools/perf/builtin-ftrace.c b/tools/perf/builtin-ftrace.c
index 6017493ff179..4e77a0b2513f 100644
--- a/tools/perf/builtin-ftrace.c
+++ b/tools/perf/builtin-ftrace.c
@@ -41,7 +41,10 @@
#include "util/units.h"
#include "util/parse-sublevel-options.h"
+#include "util/strbuf.h"
+
#define DEFAULT_TRACER "function_graph"
+#define TRACE_BUF_SIZE 4096
static volatile sig_atomic_t workload_exec_errno;
static volatile sig_atomic_t done;
@@ -738,7 +741,8 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
{
char *trace_file;
int trace_fd;
- char buf[4096];
+ char *buf = NULL;
+ struct strbuf linebuf = STRBUF_INIT;
struct pollfd pollfd = {
.events = POLLIN,
};
@@ -767,6 +771,13 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
setup_pager();
+ buf = malloc(TRACE_BUF_SIZE);
+ if (!buf) {
+ pr_err("failed to allocate trace buffer\n");
+ goto out_reset;
+ }
+ strbuf_init(&linebuf, 512);
+
trace_file = get_tracing_instance_file("trace_pipe");
if (!trace_file) {
pr_err("failed to open trace_pipe\n");
@@ -810,11 +821,19 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
break;
if (pollfd.revents & POLLIN) {
- int n = read(trace_fd, buf, sizeof(buf));
- if (n < 0)
- break;
- if (fwrite(buf, n, 1, stdout) != 1)
+ int n = read(trace_fd, buf, TRACE_BUF_SIZE);
+
+ if (n <= 0)
break;
+ for (int i = 0; i < n; i++) {
+ if (buf[i] == '\n') {
+ fprintf(stdout, "%s\n", linebuf.buf);
+ strbuf_setlen(&linebuf, 0);
+ } else {
+ if (strbuf_addch(&linebuf, buf[i]) < 0)
+ goto out_close_fd;
+ }
+ }
/* flush output since stdout is in full buffering mode due to pager */
fflush(stdout);
}
@@ -823,7 +842,7 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
write_tracing_file("tracing_on", "0");
if (workload_exec_errno) {
- const char *emsg = str_error_r(workload_exec_errno, buf, sizeof(buf));
+ const char *emsg = str_error_r(workload_exec_errno, buf, TRACE_BUF_SIZE);
/* flush stdout first so below error msg appears at the end. */
fflush(stdout);
pr_err("workload failed: %s\n", emsg);
@@ -832,14 +851,29 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
/* read remaining buffer contents */
while (true) {
- int n = read(trace_fd, buf, sizeof(buf));
+ int n = read(trace_fd, buf, TRACE_BUF_SIZE);
+
if (n <= 0)
break;
- if (fwrite(buf, n, 1, stdout) != 1)
- break;
+ for (int i = 0; i < n; i++) {
+ if (buf[i] == '\n') {
+ fprintf(stdout, "%s\n", linebuf.buf);
+ strbuf_setlen(&linebuf, 0);
+ } else {
+ if (strbuf_addch(&linebuf, buf[i]) < 0)
+ goto out_close_fd;
+ }
+ }
+ }
+
+ if (linebuf.len > 0) {
+ fprintf(stdout, "%s\n", linebuf.buf);
+ strbuf_setlen(&linebuf, 0);
}
out_close_fd:
+ free(buf);
+ strbuf_release(&linebuf);
close(trace_fd);
out_reset:
exit_tracing_instance();
--
2.55.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH perf-tools-next v2 2/4] perf ftrace: Support filtering return address comments
2026-10-06 23:27 [PATCH perf-tools-next v2 0/4] perf ftrace: Support inlined functions and display enhancements in function graph tracer Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 1/4] perf ftrace: Optimise __cmd_ftrace() stack frame with strbuf streaming Aaron Tomlin
@ 2026-10-06 23:27 ` Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 3/4] perf ftrace: Support display of inlined functions in function graph tracer Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 4/4] perf ftrace: Support omitting execution duration " Aaron Tomlin
3 siblings, 0 replies; 5+ messages in thread
From: Aaron Tomlin @ 2026-10-06 23:27 UTC (permalink / raw)
To: peterz, mingo, acme, namhyung
Cc: mark.rutland, alexander.shishkin, jolsa, irogers, adrian.hunter,
atomlin, linux-perf-users, linux-kernel
When tracing with the Linux kernel's function graph tracer, enabling
funcgraph-retaddr records the return address of each function entry as
a comment in the trace stream (e.g. /* <-load_elf_phdrs+0x6c/0xb0 */).
While valuable for tracing execution origins, these trailing comments
can add considerable visual clutter to the call graph, particularly
when examining deeply nested call trees or when synthesising higher-level
abstractions.
Introduce support for filtering out return address comments from
function graph trace output. Provide both a standalone --filter-retaddr
flag and a --graph-opts filter-retaddr sub-option.
Key aspects of this implementation:
1. Return address parsing and filtering
Implement ftrace_parse_retaddr() and ftrace_filter_retaddr() in
tools/perf/util/ftrace.c to robustly match return address
comments across diverse output styles (standard format, return
values, and bracketed addresses) and strip them in-place with
memmove(3).
2. Command-line interface
Expose --filter-retaddr and --graph-opts filter-retaddr in
tools/perf/builtin-ftrace.c, updating the documentation in
tools/perf/Documentation/perf-ftrace.txt accordingly.
3. Automated verification
Add a dedicated unit test suite ("Ftrace return address
processing") in tools/perf/tests/ftrace.c to validate comment
parsing and string filtering across all supported formats.
Assisted-by: Antigravity:gemini-3.1-pro
Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
tools/perf/Documentation/perf-ftrace.txt | 5 ++
tools/perf/builtin-ftrace.c | 17 ++++-
tools/perf/tests/Build | 1 +
tools/perf/tests/builtin-test.c | 1 +
tools/perf/tests/ftrace.c | 80 ++++++++++++++++++++++++
tools/perf/tests/tests.h | 1 +
tools/perf/util/Build | 1 +
tools/perf/util/ftrace.c | 68 ++++++++++++++++++++
tools/perf/util/ftrace.h | 5 ++
9 files changed, 178 insertions(+), 1 deletion(-)
create mode 100644 tools/perf/tests/ftrace.c
create mode 100644 tools/perf/util/ftrace.c
diff --git a/tools/perf/Documentation/perf-ftrace.txt b/tools/perf/Documentation/perf-ftrace.txt
index 3f3808e513fe..ee126604a8ca 100644
--- a/tools/perf/Documentation/perf-ftrace.txt
+++ b/tools/perf/Documentation/perf-ftrace.txt
@@ -127,6 +127,7 @@ OPTIONS for 'perf ftrace trace'
- retval - Show function return value.
- retval-hex - Show function return value in hexadecimal format.
- retaddr - Show function return address.
+ - filter-retaddr - Filter out function return address comments from output.
- nosleep-time - Measure on-CPU time only for function_graph tracer.
- noirqs - Ignore functions that happen inside interrupt.
- verbose - Show process names, PIDs, timestamps, etc.
@@ -134,6 +135,10 @@ OPTIONS for 'perf ftrace trace'
- depth=<n> - Set max depth for function graph tracer to follow.
- tail - Print function name at the end.
+--filter-retaddr::
+ Filter out function return address comments (`/* <-caller+offset */`)
+ from function_graph output.
+
OPTIONS for 'perf ftrace latency'
---------------------------------
diff --git a/tools/perf/builtin-ftrace.c b/tools/perf/builtin-ftrace.c
index 4e77a0b2513f..ddb567781125 100644
--- a/tools/perf/builtin-ftrace.c
+++ b/tools/perf/builtin-ftrace.c
@@ -827,6 +827,8 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
break;
for (int i = 0; i < n; i++) {
if (buf[i] == '\n') {
+ if (ftrace->filter_retaddr)
+ ftrace_filter_retaddr(linebuf.buf);
fprintf(stdout, "%s\n", linebuf.buf);
strbuf_setlen(&linebuf, 0);
} else {
@@ -857,6 +859,8 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
break;
for (int i = 0; i < n; i++) {
if (buf[i] == '\n') {
+ if (ftrace->filter_retaddr)
+ ftrace_filter_retaddr(linebuf.buf);
fprintf(stdout, "%s\n", linebuf.buf);
strbuf_setlen(&linebuf, 0);
} else {
@@ -867,6 +871,8 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
}
if (linebuf.len > 0) {
+ if (ftrace->filter_retaddr)
+ ftrace_filter_retaddr(linebuf.buf);
fprintf(stdout, "%s\n", linebuf.buf);
strbuf_setlen(&linebuf, 0);
}
@@ -1729,6 +1735,7 @@ static int parse_graph_tracer_opts(const struct option *opt,
{
int ret;
struct perf_ftrace *ftrace = (struct perf_ftrace *) opt->value;
+ int filter_retaddr = -1;
struct sublevel_option graph_tracer_opts[] = {
{ .name = "args", .value_ptr = &ftrace->graph_args },
{ .name = "retval", .value_ptr = &ftrace->graph_retval },
@@ -1740,6 +1747,7 @@ static int parse_graph_tracer_opts(const struct option *opt,
{ .name = "thresh", .value_ptr = &ftrace->graph_thresh },
{ .name = "depth", .value_ptr = &ftrace->graph_depth },
{ .name = "tail", .value_ptr = &ftrace->graph_tail },
+ { .name = "filter-retaddr", .value_ptr = &filter_retaddr },
{ .name = NULL, }
};
@@ -1750,6 +1758,11 @@ static int parse_graph_tracer_opts(const struct option *opt,
if (ret)
return ret;
+ if (filter_retaddr != -1) {
+ ftrace->filter_retaddr = (filter_retaddr != 0);
+ ftrace->filter_retaddr_set = true;
+ }
+
return 0;
}
@@ -1825,12 +1838,14 @@ int cmd_ftrace(int argc, const char **argv)
OPT_CALLBACK('g', "nograph-funcs", &ftrace.nograph_funcs, "func",
"Set nograph filter on given functions", parse_filter_func),
OPT_CALLBACK(0, "graph-opts", &ftrace, "options",
- "Graph tracer options, available options: args,retval,retval-hex,retaddr,nosleep-time,noirqs,verbose,thresh=<n>,depth=<n>",
+ "Graph tracer options, available options: args,retval,retval-hex,retaddr,filter-retaddr,nosleep-time,noirqs,verbose,thresh=<n>,depth=<n>",
parse_graph_tracer_opts),
OPT_CALLBACK('m', "buffer-size", &ftrace.percpu_buffer_size, "size",
"Size of per cpu buffer, needs to use a B, K, M or G suffix.", parse_buffer_size),
OPT_BOOLEAN(0, "inherit", &ftrace.inherit,
"Trace children processes"),
+ OPT_BOOLEAN_SET(0, "filter-retaddr", &ftrace.filter_retaddr, &ftrace.filter_retaddr_set,
+ "Filter out funcgraph return address comments"),
OPT_INTEGER('D', "delay", &ftrace.target.initial_delay,
"Number of milliseconds to wait before starting tracing after program start"),
OPT_PARENT(common_options),
diff --git a/tools/perf/tests/Build b/tools/perf/tests/Build
index 395f2866094a..136138932fcb 100644
--- a/tools/perf/tests/Build
+++ b/tools/perf/tests/Build
@@ -72,6 +72,7 @@ perf-test-y += hwmon_pmu.o
perf-test-y += tool_pmu.o
perf-test-y += subcmd-help.o
perf-test-y += kallsyms-split.o
+perf-test-y += ftrace.o
ifeq ($(SRCARCH),$(filter $(SRCARCH),x86 arm arm64 powerpc riscv))
perf-test-$(CONFIG_DWARF_UNWIND) += dwarf-unwind.o
diff --git a/tools/perf/tests/builtin-test.c b/tools/perf/tests/builtin-test.c
index d2f594921e25..9663f48afb92 100644
--- a/tools/perf/tests/builtin-test.c
+++ b/tools/perf/tests/builtin-test.c
@@ -137,6 +137,7 @@ static struct test_suite *generic_tests[] = {
&suite__perf_hooks,
&suite__unit_number__scnprint,
&suite__mem2node,
+ &suite__ftrace,
&suite__time_utils,
&suite__jit_write_elf,
&suite__pfm,
diff --git a/tools/perf/tests/ftrace.c b/tools/perf/tests/ftrace.c
new file mode 100644
index 000000000000..596cecee844e
--- /dev/null
+++ b/tools/perf/tests/ftrace.c
@@ -0,0 +1,80 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/compiler.h>
+#include <linux/kernel.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include "tests.h"
+#include "debug.h"
+#include "util/ftrace.h"
+
+static int test_parse_retaddr(void)
+{
+ char sym[128];
+ u64 offset = 0;
+
+ /* Standard format */
+ TEST_ASSERT_VAL("parse standard retaddr",
+ ftrace_parse_retaddr("kernel_read() { /* <-load_elf_phdrs+0x6c/0xb0 */",
+ sym, sizeof(sym), &offset));
+ TEST_ASSERT_EQUAL("sym name", strcmp(sym, "load_elf_phdrs"), 0);
+ TEST_ASSERT_VAL("offset", offset == 0x6c);
+
+ /* With return value */
+ TEST_ASSERT_VAL("parse retaddr with ret",
+ ftrace_parse_retaddr(
+ "__cond_resched(); /* <-load_elf_phdrs+0x6c/0xb0 ret=0x0 */",
+ sym, sizeof(sym), &offset));
+ TEST_ASSERT_EQUAL("sym name", strcmp(sym, "load_elf_phdrs"), 0);
+ TEST_ASSERT_VAL("offset", offset == 0x6c);
+
+ /* Bracketed address */
+ TEST_ASSERT_VAL("parse bracketed retaddr",
+ ftrace_parse_retaddr("foo() { /* <-[0xffffffff818534c8] bar+0x20/0x40 */",
+ sym, sizeof(sym), &offset));
+ TEST_ASSERT_EQUAL("sym name", strcmp(sym, "bar"), 0);
+ TEST_ASSERT_VAL("offset", offset == 0x20);
+
+ /* No return address */
+ TEST_ASSERT_VAL("no retaddr",
+ !ftrace_parse_retaddr("do_filp_open();", sym, sizeof(sym), &offset));
+
+ return TEST_OK;
+}
+
+static int test_filter_retaddr(void)
+{
+ char line1[] = "kernel_read() { /* <-load_elf_phdrs+0x6c/0xb0 */";
+ char line2[] = "__cond_resched(); /* <-load_elf_phdrs+0x6c/0xb0 ret=0x0 */";
+ char line3[] = "do_filp_open();";
+
+ ftrace_filter_retaddr(line1);
+ TEST_ASSERT_EQUAL("filter standard retaddr", strcmp(line1, "kernel_read() {"), 0);
+
+ ftrace_filter_retaddr(line2);
+ TEST_ASSERT_EQUAL("filter retaddr with ret", strcmp(line2, "__cond_resched();"), 0);
+
+ ftrace_filter_retaddr(line3);
+ TEST_ASSERT_EQUAL("filter line without retaddr unchanged",
+ strcmp(line3, "do_filp_open();"), 0);
+
+ return TEST_OK;
+}
+
+static int test__ftrace(struct test_suite *test __maybe_unused, int subtest __maybe_unused)
+{
+ int ret;
+
+ ret = test_parse_retaddr();
+ if (ret != TEST_OK)
+ return ret;
+
+ ret = test_filter_retaddr();
+ if (ret != TEST_OK)
+ return ret;
+
+ return TEST_OK;
+}
+
+DEFINE_SUITE("Ftrace return address processing", ftrace);
diff --git a/tools/perf/tests/tests.h b/tools/perf/tests/tests.h
index 9c96f33483d1..1aec32e094c8 100644
--- a/tools/perf/tests/tests.h
+++ b/tools/perf/tests/tests.h
@@ -163,6 +163,7 @@ DECLARE_SUITE(perf_hooks);
DECLARE_SUITE(unit_number__scnprint);
DECLARE_SUITE(mem2node);
DECLARE_SUITE(maps);
+DECLARE_SUITE(ftrace);
DECLARE_SUITE(time_utils);
DECLARE_SUITE(jit_write_elf);
DECLARE_SUITE(api_io);
diff --git a/tools/perf/util/Build b/tools/perf/util/Build
index 2c1f880c4c47..5b082d09006d 100644
--- a/tools/perf/util/Build
+++ b/tools/perf/util/Build
@@ -169,6 +169,7 @@ perf-util-y += list_sort.o
perf-util-y += mutex.o
perf-util-y += sharded_mutex.o
perf-util-y += intel-tpebs.o
+perf-util-y += ftrace.o
perf-util-$(CONFIG_PERF_BPF_SKEL) += bpf_counter.o
perf-util-$(CONFIG_PERF_BPF_SKEL) += bpf_counter_cgroup.o
diff --git a/tools/perf/util/ftrace.c b/tools/perf/util/ftrace.c
new file mode 100644
index 000000000000..5f582cf8531f
--- /dev/null
+++ b/tools/perf/util/ftrace.c
@@ -0,0 +1,68 @@
+// SPDX-License-Identifier: GPL-2.0
+#include <linux/compiler.h>
+#include <linux/kernel.h>
+#include <linux/string.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <ctype.h>
+#include "util/ftrace.h"
+
+bool ftrace_parse_retaddr(const char *str, char *sym_name, size_t sym_len, u64 *offset)
+{
+ const char *p = strstr(str, "<-");
+ const char *plus;
+ const char *bracket;
+ char *endptr;
+ size_t name_len;
+
+ if (!p)
+ return false;
+
+ p += 2;
+ while (*p == ' ')
+ p++;
+
+ if (*p == '[') {
+ bracket = strchr(p, ']');
+ if (bracket)
+ p = bracket + 1;
+ while (*p == ' ')
+ p++;
+ }
+
+ plus = strchr(p, '+');
+ if (!plus)
+ return false;
+
+ name_len = plus - p;
+ if (name_len == 0 || name_len >= sym_len)
+ return false;
+
+ memcpy(sym_name, p, name_len);
+ sym_name[name_len] = '\0';
+
+ *offset = strtoull(plus + 1, &endptr, 16);
+ return true;
+}
+
+void ftrace_filter_retaddr(char *str)
+{
+ char *start = strstr(str, "/* <-");
+ char *end;
+
+ if (!start)
+ return;
+
+ end = strstr(start, "*/");
+ if (!end)
+ return;
+
+ end += 2; /* skip end-of-comment delimiter */
+
+ /* Also backtrack any spaces before comment */
+ while (start > str && *(start - 1) == ' ')
+ start--;
+
+ memmove(start, end, strlen(end) + 1);
+}
diff --git a/tools/perf/util/ftrace.h b/tools/perf/util/ftrace.h
index 950f2efafad2..50216d3fef64 100644
--- a/tools/perf/util/ftrace.h
+++ b/tools/perf/util/ftrace.h
@@ -39,6 +39,8 @@ struct perf_ftrace {
int graph_verbose;
int graph_thresh;
int graph_tail;
+ bool filter_retaddr;
+ bool filter_retaddr_set;
};
struct filter_entry {
@@ -93,4 +95,7 @@ perf_ftrace__latency_cleanup_bpf(struct perf_ftrace *ftrace __maybe_unused)
#endif /* HAVE_BPF_SKEL */
+bool ftrace_parse_retaddr(const char *str, char *sym_name, size_t sym_len, u64 *offset);
+void ftrace_filter_retaddr(char *str);
+
#endif /* __PERF_FTRACE_H__ */
--
2.55.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH perf-tools-next v2 3/4] perf ftrace: Support display of inlined functions in function graph tracer
2026-10-06 23:27 [PATCH perf-tools-next v2 0/4] perf ftrace: Support inlined functions and display enhancements in function graph tracer Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 1/4] perf ftrace: Optimise __cmd_ftrace() stack frame with strbuf streaming Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 2/4] perf ftrace: Support filtering return address comments Aaron Tomlin
@ 2026-10-06 23:27 ` Aaron Tomlin
2026-10-06 23:27 ` [PATCH perf-tools-next v2 4/4] perf ftrace: Support omitting execution duration " Aaron Tomlin
3 siblings, 0 replies; 5+ messages in thread
From: Aaron Tomlin @ 2026-10-06 23:27 UTC (permalink / raw)
To: peterz, mingo, acme, namhyung
Cc: mark.rutland, alexander.shishkin, jolsa, irogers, adrian.hunter,
atomlin, linux-perf-users, linux-kernel
The Linux kernel's function graph tracer operates at the machine
instruction level via compiler instrumentation
(-fpatchable-function-entry), recording entry and return events for
physical function calls. Consequently, functions inlined by the compiler
are invisible in the resulting call graph, as no discrete call or return
instructions are emitted for them. While technically expected, this
omission frequently obscures the logical execution flow when analysing
kernel subsystems heavily reliant upon inlining.
Address this by introducing the --inline option to perf ftrace for the
function_graph tracer.
When the kernel is configured with CONFIG_FUNCTION_GRAPH_RETADDR=y, the
funcgraph-retaddr trace option records the caller's return address on
each function entry, manifested in the trace stream as a comment
(e.g. /* <-irq_work_queue_on+0x81/0xa0 */). By interrogating DWARF debug
information from the kernel image (vmlinux) utilising libdw, perf ftrace
resolves this return address to its inlined callchain and synthesises the
intermediate inlined frames directly into the streamed call graph.
Key aspects of this implementation:
1. Inlined frame presentation
Synthesised inlined functions are rendered with an explicit
/* (inline) */ annotation at both entry and exit:
# CPU DURATION FUNCTION CALLS
# | | | | | | |
1) 0.856 us | mutex_unlock(lock=0xffffffffbc28bfa0);
0) | wake_up_new_task(p=0xffff8b1f183bac80) {
0) | rb_irq_work_queue() { /* (inline) */
0) | rb_wakeups() { /* (inline) */
0) 1.374 us | housekeeping_any_cpu(type=3 [HK_TYPE_KERNEL_NOISE]);
0) | arch_irq_work_raise() {
0) | apic_wait_icr_idle() { /* (inline) */
0) | x2apic_send_IPI_self(vector=246) {
0) | instr_sysvec_irq_work() { /* (inline) */
0) | irq_enter_rcu() {
0) | instr_sysvec_irq_work() { /* (inline) */
0) 0.673 us | irqtime_account_irq(curr=0xffff8b1f086a2c80, offset=0x1000000);
0) | } /* instr_sysvec_irq_work (inline) */
0) 1.697 us | }
This affords immediate visual clarity to the reader upon
entering an inlined scope, without necessitating a search for
the closing delimiter.
2. Call graph reconstruction and dynamic stack tracking
Maintains a per-CPU stack tracking real and inlined frames,
synthesising entry and exit delimiters to accurately reconstruct
the execution hierarchy. Bare closing braces without trailing
function names (the default when funcgraph-tail is disabled) are
correctly matched against preceding frames on the stack.
Furthermore, stack depth saturation guards ensure that inlined
frame depth tracking remains clamped and indentation cannot
drift even under deep call trees exceeding FTRACE_MAX_STACK.
3. Option handling and diagnostics
Adds -k or --vmlinux to designate an explicit debug image, and
provides clear diagnostic guidance should --inline be invoked on
a kernel lacking CONFIG_FUNCTION_GRAPH_RETADDR. Automatically
enables return address comment filtering unless explicitly
requested.
4. Automated verification
Expands the unit test suite in tools/perf/tests/ftrace.c to
validate DWARF inline symbol resolution, comment filtering
toggles, bare closing braces handling, and stack saturation
recovery.
Assisted-by: Antigravity:gemini-3.1-pro
Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
tools/perf/Documentation/perf-ftrace.txt | 19 +-
tools/perf/builtin-ftrace.c | 56 ++-
tools/perf/tests/ftrace.c | 162 +++++++-
tools/perf/util/ftrace.c | 475 ++++++++++++++++++++++-
tools/perf/util/ftrace.h | 11 +
5 files changed, 704 insertions(+), 19 deletions(-)
diff --git a/tools/perf/Documentation/perf-ftrace.txt b/tools/perf/Documentation/perf-ftrace.txt
index ee126604a8ca..ca0f1d523dbc 100644
--- a/tools/perf/Documentation/perf-ftrace.txt
+++ b/tools/perf/Documentation/perf-ftrace.txt
@@ -127,7 +127,7 @@ OPTIONS for 'perf ftrace trace'
- retval - Show function return value.
- retval-hex - Show function return value in hexadecimal format.
- retaddr - Show function return address.
- - filter-retaddr - Filter out function return address comments from output.
+ - filter-retaddr - Filter out function return address comments (enabled by default with --inline).
- nosleep-time - Measure on-CPU time only for function_graph tracer.
- noirqs - Ignore functions that happen inside interrupt.
- verbose - Show process names, PIDs, timestamps, etc.
@@ -135,9 +135,24 @@ OPTIONS for 'perf ftrace trace'
- depth=<n> - Set max depth for function graph tracer to follow.
- tail - Print function name at the end.
+--inline::
+ Show inlined functions in function_graph tracer. This requires
+ kernel support for the funcgraph-retaddr option and a vmlinux with
+ debug symbols. Inlined functions are displayed in the call graph
+ hierarchy, and their inlined status is indicated with `/* (inline) */`
+ on entry and exit, e.g. `func() { /* (inline) */` and `} /* func (inline) */`.
+ By default, raw return address comments (`/* <-caller+offset */`)
+ are hidden in the output; specify `--graph-opts retaddr` to show them.
+
--filter-retaddr::
Filter out function return address comments (`/* <-caller+offset */`)
- from function_graph output.
+ from function_graph output. This is enabled by default when `--inline`
+ is used; use `--no-filter-retaddr` to disable it.
+
+-k::
+--vmlinux=::
+ Path to the vmlinux file containing debug symbols for resolving
+ inlined functions.
OPTIONS for 'perf ftrace latency'
diff --git a/tools/perf/builtin-ftrace.c b/tools/perf/builtin-ftrace.c
index ddb567781125..e11418e61937 100644
--- a/tools/perf/builtin-ftrace.c
+++ b/tools/perf/builtin-ftrace.c
@@ -574,9 +574,14 @@ static int set_tracing_funcgraph_retval(struct perf_ftrace *ftrace)
static int set_tracing_funcgraph_retaddr(struct perf_ftrace *ftrace)
{
- if (ftrace->graph_retaddr) {
- if (write_tracing_option_file("funcgraph-retaddr", "1") < 0)
+ if (ftrace->graph_retaddr || ftrace->use_inline) {
+ if (write_tracing_option_file("funcgraph-retaddr", "1") < 0) {
+ if (ftrace->use_inline) {
+ pr_err("failed to set funcgraph-retaddr option in tracing\n");
+ pr_err("Hint: ensure kernel is compiled with CONFIG_FUNCTION_GRAPH_RETADDR=y\n");
+ }
return -1;
+ }
}
return 0;
@@ -769,6 +774,9 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
goto out_reset;
}
+ if (perf_ftrace__setup_inlines(ftrace) < 0)
+ goto out_reset;
+
setup_pager();
buf = malloc(TRACE_BUF_SIZE);
@@ -827,9 +835,10 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
break;
for (int i = 0; i < n; i++) {
if (buf[i] == '\n') {
- if (ftrace->filter_retaddr)
- ftrace_filter_retaddr(linebuf.buf);
- fprintf(stdout, "%s\n", linebuf.buf);
+ if (ftrace_process_fgraph_line(ftrace,
+ linebuf.buf,
+ stdout) < 0)
+ goto out_close_fd;
strbuf_setlen(&linebuf, 0);
} else {
if (strbuf_addch(&linebuf, buf[i]) < 0)
@@ -859,9 +868,10 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
break;
for (int i = 0; i < n; i++) {
if (buf[i] == '\n') {
- if (ftrace->filter_retaddr)
- ftrace_filter_retaddr(linebuf.buf);
- fprintf(stdout, "%s\n", linebuf.buf);
+ if (ftrace_process_fgraph_line(ftrace,
+ linebuf.buf,
+ stdout) < 0)
+ goto out_close_fd;
strbuf_setlen(&linebuf, 0);
} else {
if (strbuf_addch(&linebuf, buf[i]) < 0)
@@ -871,9 +881,9 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
}
if (linebuf.len > 0) {
- if (ftrace->filter_retaddr)
- ftrace_filter_retaddr(linebuf.buf);
- fprintf(stdout, "%s\n", linebuf.buf);
+ if (ftrace_process_fgraph_line(ftrace, linebuf.buf,
+ stdout) < 0)
+ goto out_close_fd;
strbuf_setlen(&linebuf, 0);
}
@@ -882,6 +892,7 @@ static int __cmd_ftrace(struct perf_ftrace *ftrace)
strbuf_release(&linebuf);
close(trace_fd);
out_reset:
+ perf_ftrace__cleanup_inlines(ftrace);
exit_tracing_instance();
out:
return (done && !workload_exec_errno) ? 0 : -1;
@@ -1844,8 +1855,13 @@ int cmd_ftrace(int argc, const char **argv)
"Size of per cpu buffer, needs to use a B, K, M or G suffix.", parse_buffer_size),
OPT_BOOLEAN(0, "inherit", &ftrace.inherit,
"Trace children processes"),
- OPT_BOOLEAN_SET(0, "filter-retaddr", &ftrace.filter_retaddr, &ftrace.filter_retaddr_set,
- "Filter out funcgraph return address comments"),
+ OPT_BOOLEAN(0, "inline", &ftrace.use_inline,
+ "Show inlined functions in function_graph tracer"),
+ OPT_BOOLEAN_SET(0, "filter-retaddr", &ftrace.filter_retaddr,
+ &ftrace.filter_retaddr_set,
+ "Filter out funcgraph return address comments (default with --inline)"),
+ OPT_STRING('k', "vmlinux", &symbol_conf.vmlinux_name, "file",
+ "vmlinux pathname"),
OPT_INTEGER('D', "delay", &ftrace.target.initial_delay,
"Number of milliseconds to wait before starting tracing after program start"),
OPT_PARENT(common_options),
@@ -1961,6 +1977,20 @@ int cmd_ftrace(int argc, const char **argv)
switch (subcmd) {
case PERF_FTRACE_TRACE:
+ if (ftrace.use_inline) {
+ if (ftrace.tracer && strcmp(ftrace.tracer, "function_graph") != 0) {
+ pr_err("Error: --inline option is only supported with function_graph tracer\n");
+ ret = -EINVAL;
+ goto out_delete_filters;
+ }
+ ftrace.tracer = "function_graph";
+ }
+ if (!ftrace.filter_retaddr_set) {
+ if (ftrace.use_inline && !ftrace.graph_retaddr)
+ ftrace.filter_retaddr = true;
+ else
+ ftrace.filter_retaddr = false;
+ }
cmd_func = __cmd_ftrace;
break;
case PERF_FTRACE_LATENCY:
diff --git a/tools/perf/tests/ftrace.c b/tools/perf/tests/ftrace.c
index 596cecee844e..599c55132910 100644
--- a/tools/perf/tests/ftrace.c
+++ b/tools/perf/tests/ftrace.c
@@ -7,6 +7,10 @@
#include <unistd.h>
#include "tests.h"
#include "debug.h"
+#include "dso.h"
+#include "machine.h"
+#include "map.h"
+#include "symbol.h"
#include "util/ftrace.h"
static int test_parse_retaddr(void)
@@ -62,6 +66,154 @@ static int test_filter_retaddr(void)
return TEST_OK;
}
+static int test_inline_processing(void)
+{
+ static const char line_entry[] = " 0) | wake_up_new_task() {";
+ static const char line_enqueue[] =
+ " 0) | enqueue_task() { /* <-wake_up_new_task+0x1d1/0x3e0 */";
+ static const char line_exit_enqueue[] = " 0) + 10.000 us | } /* enqueue_task */";
+ static const char line_exit_task[] = " 0) + 20.000 us | } /* wake_up_new_task */";
+ static const char line_bare_enqueue[] = " 0) + 10.000 us | }";
+ static const char line_bare_task[] = " 0) + 20.000 us | }";
+ struct perf_ftrace ftrace;
+ char *out_buf = NULL;
+ size_t out_len = 0;
+ FILE *out;
+ int ret;
+
+ memset(&ftrace, 0, sizeof(ftrace));
+ ftrace.use_inline = true;
+ ftrace.filter_retaddr = 1; /* Default when --graph-opts retaddr is not specified */
+ ftrace.tracer = "function_graph";
+
+ /* If vmlinux is present, test resolving symbols */
+ if (access("vmlinux", R_OK) == 0) {
+ symbol_conf.vmlinux_name = "vmlinux";
+ symbol_conf.ignore_vmlinux_buildid = true;
+ ret = perf_ftrace__setup_inlines(&ftrace);
+ if (ret < 0 || !ftrace.machine)
+ return TEST_SKIP;
+
+ out = open_memstream(&out_buf, &out_len);
+ if (!out) {
+ perf_ftrace__cleanup_inlines(&ftrace);
+ return TEST_FAIL;
+ }
+
+ /* Feed simulated function graph trace lines */
+ ftrace_process_fgraph_line(&ftrace, line_entry, out);
+ ftrace_process_fgraph_line(&ftrace, line_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_exit_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_exit_task, out);
+
+ fclose(out);
+ ftrace_cpu_states__clear();
+
+ pr_debug("Processed output:\n%s\n", out_buf);
+
+ /* Verify activate_task entry with inline hint */
+ TEST_ASSERT_VAL("contains activate_task entry with inline hint",
+ strstr(out_buf, "activate_task() { /* (inline) */") != NULL);
+
+ /* Verify activate_task exit with inline hint */
+ TEST_ASSERT_VAL("contains activate_task exit with inline hint",
+ strstr(out_buf, "} /* activate_task (inline) */") != NULL);
+
+ /* Verify return address comment was filtered out by default */
+ TEST_ASSERT_VAL("retaddr comment filtered out",
+ strstr(out_buf, "/* <-wake_up_new_task") == NULL);
+
+ free(out_buf);
+ out_buf = NULL;
+
+ /* Now test with filter_retaddr = 0 (user specified --graph-opts retaddr) */
+ ftrace.filter_retaddr = 0;
+ out = open_memstream(&out_buf, &out_len);
+ if (!out) {
+ perf_ftrace__cleanup_inlines(&ftrace);
+ return TEST_FAIL;
+ }
+
+ ftrace_process_fgraph_line(&ftrace, line_entry, out);
+ ftrace_process_fgraph_line(&ftrace, line_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_exit_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_exit_task, out);
+
+ fclose(out);
+ ftrace_cpu_states__clear();
+
+ /* Verify return address comment IS preserved when filter_retaddr = 0 */
+ TEST_ASSERT_VAL("retaddr comment preserved when requested",
+ strstr(out_buf, "/* <-wake_up_new_task") != NULL);
+
+ free(out_buf);
+ out_buf = NULL;
+
+ /* Test with bare closing braces (funcgraph-tail disabled by default) */
+ ftrace.filter_retaddr = 1;
+ out = open_memstream(&out_buf, &out_len);
+ if (!out) {
+ perf_ftrace__cleanup_inlines(&ftrace);
+ return TEST_FAIL;
+ }
+
+ ftrace_process_fgraph_line(&ftrace, line_entry, out);
+ ftrace_process_fgraph_line(&ftrace, line_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_bare_enqueue, out);
+ ftrace_process_fgraph_line(&ftrace, line_bare_task, out);
+
+ fclose(out);
+ perf_ftrace__cleanup_inlines(&ftrace);
+
+ TEST_ASSERT_VAL("bare braces inline entry",
+ strstr(out_buf, "activate_task() { /* (inline) */") != NULL);
+ TEST_ASSERT_VAL("bare braces inline exit",
+ strstr(out_buf, "} /* activate_task (inline) */") != NULL);
+
+ free(out_buf);
+ }
+
+ return TEST_OK;
+}
+
+static int test_stack_saturation(void)
+{
+ struct perf_ftrace ftrace;
+ char *out_buf = NULL;
+ size_t out_len = 0;
+ FILE *out;
+ int i;
+
+ memset(&ftrace, 0, sizeof(ftrace));
+ ftrace.use_inline = true;
+ ftrace_cpu_states__clear();
+
+ out = open_memstream(&out_buf, &out_len);
+ if (!out)
+ return TEST_FAIL;
+
+ /* Push more than FTRACE_MAX_STACK (256) frames */
+ for (i = 0; i < 300; i++)
+ ftrace_process_fgraph_line(&ftrace, " 0) | sub_func() {", out);
+
+ /* Pop all 300 frames */
+ for (i = 0; i < 300; i++)
+ ftrace_process_fgraph_line(&ftrace, " 0) | }", out);
+
+ /* Feed one more function entry - it must have base indentation (no indentation drift) */
+ ftrace_process_fgraph_line(&ftrace, " 0) | base_func() {", out);
+
+ fclose(out);
+ ftrace_cpu_states__clear();
+
+ /* Verify base_func starts at standard 2 spaces indentation without drift */
+ TEST_ASSERT_VAL("no indentation drift after stack saturation",
+ strstr(out_buf, "| base_func() {") != NULL);
+
+ free(out_buf);
+ return TEST_OK;
+}
+
static int test__ftrace(struct test_suite *test __maybe_unused, int subtest __maybe_unused)
{
int ret;
@@ -74,7 +226,15 @@ static int test__ftrace(struct test_suite *test __maybe_unused, int subtest __ma
if (ret != TEST_OK)
return ret;
+ ret = test_stack_saturation();
+ if (ret != TEST_OK)
+ return ret;
+
+ ret = test_inline_processing();
+ if (ret != TEST_OK)
+ return ret;
+
return TEST_OK;
}
-DEFINE_SUITE("Ftrace return address processing", ftrace);
+DEFINE_SUITE("Ftrace inline processing", ftrace);
diff --git a/tools/perf/util/ftrace.c b/tools/perf/util/ftrace.c
index 5f582cf8531f..a0ceaae0a978 100644
--- a/tools/perf/util/ftrace.c
+++ b/tools/perf/util/ftrace.c
@@ -1,13 +1,68 @@
// SPDX-License-Identifier: GPL-2.0
-#include <linux/compiler.h>
-#include <linux/kernel.h>
-#include <linux/string.h>
+/*
+ * ftrace.c - Ftrace utilities and inlined function processing
+ *
+ * Copyright (C) 2026 Aaron Tomlin <atomlin@atomlin.com>
+ */
#include <stdio.h>
#include <stdlib.h>
+#include <errno.h>
#include <string.h>
#include <ctype.h>
+#include <stdbool.h>
+#include <linux/kernel.h>
+#include <linux/string.h>
+#include "debug.h"
+#include "dso.h"
+#include "machine.h"
+#include "map.h"
+#include "srcline.h"
+#include "symbol.h"
#include "util/ftrace.h"
+#define FTRACE_MAX_CPUS 4096
+#define FTRACE_MAX_STACK 256
+#define FTRACE_MAX_INLINES 32
+
+struct ftrace_stack_frame {
+ char *name;
+ bool inlined;
+ int indent;
+};
+
+struct ftrace_cpu_state {
+ struct ftrace_stack_frame stack[FTRACE_MAX_STACK];
+ int depth;
+ int inlined_depth;
+};
+
+static struct ftrace_cpu_state *ftrace_cpu_states[FTRACE_MAX_CPUS];
+
+void ftrace_cpu_states__clear(void)
+{
+ int i, j;
+
+ for (i = 0; i < FTRACE_MAX_CPUS; i++) {
+ if (ftrace_cpu_states[i]) {
+ for (j = 0; j < ftrace_cpu_states[i]->depth; j++)
+ free(ftrace_cpu_states[i]->stack[j].name);
+ free(ftrace_cpu_states[i]);
+ ftrace_cpu_states[i] = NULL;
+ }
+ }
+}
+
+static struct ftrace_cpu_state *get_cpu_state(int cpu)
+{
+ if (cpu < 0 || cpu >= FTRACE_MAX_CPUS)
+ return NULL;
+
+ if (!ftrace_cpu_states[cpu])
+ ftrace_cpu_states[cpu] = zalloc(sizeof(struct ftrace_cpu_state));
+
+ return ftrace_cpu_states[cpu];
+}
+
bool ftrace_parse_retaddr(const char *str, char *sym_name, size_t sym_len, u64 *offset)
{
const char *p = strstr(str, "<-");
@@ -66,3 +121,417 @@ void ftrace_filter_retaddr(char *str)
memmove(start, end, strlen(end) + 1);
}
+
+int ftrace_resolve_inlines(struct machine *machine,
+ const char *sym_name, u64 offset,
+ const char **inlined_names, int max_inlines)
+{
+ struct map *map = NULL;
+ struct symbol *sym;
+ struct dso *dso;
+ struct inline_node *node;
+ struct inline_list *ilist;
+ u64 ip, addr;
+ int count = 0;
+
+ if (!machine)
+ return 0;
+
+ sym = machine__find_kernel_symbol_by_name(machine, sym_name, &map);
+ if (!sym || !map)
+ return 0;
+
+ dso = map__dso(map);
+ if (!dso) {
+ map__put(map);
+ return 0;
+ }
+
+ ip = sym->start + offset;
+ addr = map__rip_2objdump(map, ip);
+
+ node = inlines__tree_find(dso__inlined_nodes(dso), addr);
+ if (!node) {
+ node = dso__parse_addr_inlines(dso, addr, sym);
+ if (node)
+ inlines__tree_insert(dso__inlined_nodes(dso), node);
+ }
+
+ if (!node) {
+ map__put(map);
+ return 0;
+ }
+
+ list_for_each_entry(ilist, &node->val, list) {
+ if (ilist->symbol && symbol__inlined(ilist->symbol)) {
+ if (count < max_inlines) {
+ char *name = strdup(ilist->symbol->name);
+
+ if (name)
+ inlined_names[count++] = name;
+ }
+ }
+ }
+
+ map__put(map);
+ return count;
+}
+
+static void extract_entry_func_name(const char *str, char *buf, size_t size)
+{
+ size_t i = 0;
+
+ while (*str && *str != '(' && *str != '{' && !isspace(*str) && i < size - 1)
+ buf[i++] = *str++;
+ buf[i] = '\0';
+}
+
+static void extract_exit_func_name(const char *str, char *buf, size_t size)
+{
+ const char *p = strchr(str, '}');
+ size_t i = 0;
+
+ buf[0] = '\0';
+ if (!p)
+ return;
+
+ p = strstr(p, "/*");
+ if (!p)
+ return;
+
+ p = skip_spaces(p + 2);
+ while (*p && !isspace(*p) && *p != '*' && *p != '/' && i < size - 1)
+ buf[i++] = *p++;
+ buf[i] = '\0';
+}
+
+static void make_blank_prefix(const char *line, const char *pipe_char, char *buf, size_t size)
+{
+ size_t len = pipe_char - line + 1;
+ const char *paren = strchr(line, ')');
+ char *c;
+
+ if (len >= size)
+ len = size - 1;
+
+ memcpy(buf, line, len);
+ buf[len] = '\0';
+
+ /* Replace duration between ')' and '|' with spaces */
+ if (paren && paren < line + len) {
+ char *end = buf + (pipe_char - line);
+
+ if (end > buf + len)
+ end = buf + len;
+ for (c = buf + (paren - line) + 1; c < end; c++)
+ *c = ' ';
+ }
+}
+
+int ftrace_process_fgraph_line(struct perf_ftrace *ftrace, const char *line, FILE *out)
+{
+ char caller_sym[128];
+ char blank_prefix[128];
+ const char *inlined_names[FTRACE_MAX_INLINES];
+ const char *pipe_char;
+ const char *paren;
+ const char *after_pipe;
+ const char *func_start;
+ struct ftrace_cpu_state *cs;
+ u64 caller_offset = 0;
+ int num_inlines = 0;
+ int cpu = -1;
+ int leading_spaces;
+ int extra_indent;
+ int caller_idx;
+ int active_inlines;
+ int common;
+ int match_idx;
+ int i;
+ bool is_exit;
+ bool has_open_brace;
+ bool has_semicolon;
+
+ if (!ftrace->use_inline) {
+ if (ftrace->filter_retaddr) {
+ char *filtered = strdup(line);
+
+ if (filtered) {
+ ftrace_filter_retaddr(filtered);
+ fprintf(out, "%s\n", filtered);
+ free(filtered);
+ } else {
+ pr_err("Not enough memory\n");
+ return -ENOMEM;
+ }
+ return 0;
+ }
+ fprintf(out, "%s\n", line);
+ return 0;
+ }
+
+ pipe_char = strchr(line, '|');
+ if (!pipe_char) {
+ fprintf(out, "%s\n", line);
+ return 0;
+ }
+
+ /* Extract CPU number */
+ paren = strchr(line, ')');
+ if (paren && paren > line) {
+ char *endptr;
+ long val = strtol(skip_spaces(line), &endptr, 10);
+
+ if (endptr == paren)
+ cpu = (int)val;
+ }
+
+ cs = get_cpu_state(cpu);
+ if (!cs) {
+ fprintf(out, "%s\n", line);
+ return 0;
+ }
+
+ make_blank_prefix(line, pipe_char, blank_prefix, sizeof(blank_prefix));
+
+ after_pipe = pipe_char + 1;
+ func_start = skip_spaces(after_pipe);
+ leading_spaces = func_start - after_pipe;
+
+ if (*func_start == '\0') {
+ fprintf(out, "%s\n", line);
+ return 0;
+ }
+
+ is_exit = (*func_start == '}');
+
+ if (is_exit) {
+ char exit_name[128];
+
+ extract_exit_func_name(func_start, exit_name, sizeof(exit_name));
+
+ /* Find matching frame on stack */
+ match_idx = -1;
+ for (i = cs->depth - 1; i >= 0; i--) {
+ if (!cs->stack[i].inlined) {
+ if (exit_name[0] == '\0' ||
+ (cs->stack[i].name &&
+ strcmp(cs->stack[i].name, exit_name) == 0)) {
+ match_idx = i;
+ break;
+ }
+ }
+ }
+
+ /* Close all inlined frames above match_idx */
+ while (cs->depth > 0 && (match_idx < 0 || cs->depth - 1 > match_idx)) {
+ if (cs->stack[cs->depth - 1].inlined) {
+ struct ftrace_stack_frame *top = &cs->stack[cs->depth - 1];
+ int exit_indent = top->indent;
+
+ if (cs->inlined_depth > 0)
+ cs->inlined_depth--;
+ fprintf(out, "%s%*s} /* %s (inline) */\n",
+ blank_prefix, exit_indent, "", top->name);
+ free(top->name);
+ cs->depth--;
+ } else {
+ break;
+ }
+ }
+
+ /* Pop the matched real frame */
+ if (cs->depth > 0 && match_idx == cs->depth - 1 &&
+ !cs->stack[cs->depth - 1].inlined) {
+ free(cs->stack[cs->depth - 1].name);
+ cs->depth--;
+ }
+
+ /* Emit the exit line with extra indentation if within inlined functions */
+ extra_indent = cs->inlined_depth * 2;
+
+ fprintf(out, "%.*s%*s%s\n",
+ (int)(pipe_char - line + 1), line,
+ leading_spaces + extra_indent, "", func_start);
+ return 0;
+ }
+
+ /* It's an entry or a leaf call */
+ has_open_brace = (strchr(func_start, '{') != NULL);
+ has_semicolon = (strchr(func_start, ';') != NULL);
+
+ if (ftrace_parse_retaddr(func_start, caller_sym, sizeof(caller_sym), &caller_offset)) {
+ num_inlines = ftrace_resolve_inlines(ftrace->machine, caller_sym,
+ caller_offset, inlined_names,
+ FTRACE_MAX_INLINES);
+ }
+
+ /* Determine which inlined frames are currently open above the caller */
+ caller_idx = -1;
+ if (num_inlines > 0) {
+ for (i = cs->depth - 1; i >= 0; i--) {
+ if (!cs->stack[i].inlined && cs->stack[i].name &&
+ strcmp(cs->stack[i].name, caller_sym) == 0) {
+ caller_idx = i;
+ break;
+ }
+ }
+ }
+
+ active_inlines = 0;
+ if (caller_idx >= 0) {
+ for (i = caller_idx + 1; i < cs->depth; i++) {
+ if (cs->stack[i].inlined)
+ active_inlines++;
+ else
+ break;
+ }
+ }
+
+ /* Find common prefix of inlined functions */
+ common = 0;
+ if (caller_idx >= 0) {
+ while (common < active_inlines && common < num_inlines &&
+ cs->stack[caller_idx + 1 + common].name &&
+ !strcmp(cs->stack[caller_idx + 1 + common].name,
+ inlined_names[common])) {
+ common++;
+ }
+ }
+
+ /* Close inlined frames that are no longer active */
+ while (active_inlines > common && cs->depth > 0) {
+ if (cs->stack[cs->depth - 1].inlined) {
+ struct ftrace_stack_frame *top = &cs->stack[cs->depth - 1];
+ int exit_indent = top->indent;
+
+ if (cs->inlined_depth > 0)
+ cs->inlined_depth--;
+ fprintf(out, "%s%*s} /* %s (inline) */\n",
+ blank_prefix, exit_indent, "", top->name);
+ free(top->name);
+ cs->depth--;
+ active_inlines--;
+ } else {
+ break;
+ }
+ }
+
+ /* Open new inlined frames */
+ for (i = common; i < num_inlines; i++) {
+ int entry_indent = leading_spaces + cs->inlined_depth * 2;
+
+ if (!inlined_names[i]) {
+ pr_err("Not enough memory\n");
+ for (int j = 0; j < num_inlines; j++)
+ free((void *)inlined_names[j]);
+ return -ENOMEM;
+ }
+ if (cs->depth >= FTRACE_MAX_STACK)
+ break;
+
+ fprintf(out, "%s%*s%s() { /* (inline) */\n",
+ blank_prefix, entry_indent, "", inlined_names[i]);
+ cs->stack[cs->depth].name = strdup(inlined_names[i]);
+ if (!cs->stack[cs->depth].name) {
+ pr_err("Not enough memory\n");
+ for (int j = 0; j < num_inlines; j++)
+ free((void *)inlined_names[j]);
+ return -ENOMEM;
+ }
+ cs->stack[cs->depth].inlined = true;
+ cs->stack[cs->depth].indent = entry_indent;
+ cs->depth++;
+ cs->inlined_depth++;
+ }
+
+ /* Print current line with updated indentation */
+ extra_indent = cs->inlined_depth * 2;
+
+ if (ftrace->filter_retaddr) {
+ char *filtered = strdup(func_start);
+
+ if (filtered) {
+ ftrace_filter_retaddr(filtered);
+ fprintf(out, "%.*s%*s%s\n",
+ (int)(pipe_char - line + 1), line,
+ leading_spaces + extra_indent, "", filtered);
+ free(filtered);
+ } else {
+ pr_err("Not enough memory\n");
+ for (i = 0; i < num_inlines; i++)
+ free((void *)inlined_names[i]);
+ return -ENOMEM;
+ }
+ } else {
+ fprintf(out, "%.*s%*s%s\n",
+ (int)(pipe_char - line + 1), line,
+ leading_spaces + extra_indent, "", func_start);
+ }
+
+ /* If non-leaf entry, push onto stack */
+ if (has_open_brace && !has_semicolon) {
+ char entry_name[128];
+
+ extract_entry_func_name(func_start, entry_name, sizeof(entry_name));
+ if (cs->depth < FTRACE_MAX_STACK) {
+ cs->stack[cs->depth].name = strdup(entry_name);
+ if (!cs->stack[cs->depth].name) {
+ pr_err("Not enough memory\n");
+ for (i = 0; i < num_inlines; i++)
+ free((void *)inlined_names[i]);
+ return -ENOMEM;
+ }
+ cs->stack[cs->depth].inlined = false;
+ cs->stack[cs->depth].indent = leading_spaces + extra_indent;
+ cs->depth++;
+ }
+ }
+
+ for (i = 0; i < num_inlines; i++)
+ free((void *)inlined_names[i]);
+ return 0;
+}
+
+int perf_ftrace__setup_inlines(struct perf_ftrace *ftrace)
+{
+ struct map *kmap;
+
+ if (!ftrace->use_inline)
+ return 0;
+
+ symbol_conf.try_vmlinux_path = (symbol_conf.vmlinux_name == NULL);
+ symbol_conf.inline_name = true;
+ if (symbol_conf.vmlinux_name)
+ symbol_conf.ignore_vmlinux_buildid = true;
+
+ if (symbol__init(NULL) < 0) {
+ pr_warning("Failed to initialize symbols for inlined functions\n");
+ return -1;
+ }
+
+ ftrace->machine = machine__new_host(NULL);
+ if (!ftrace->machine) {
+ pr_warning("Failed to create host machine for symbols\n");
+ return -1;
+ }
+
+ if (symbol_conf.vmlinux_name) {
+ kmap = machine__kernel_map(ftrace->machine);
+ if (kmap)
+ dso__load_vmlinux(map__dso(kmap), kmap, symbol_conf.vmlinux_name, false);
+ } else {
+ machine__load_vmlinux_path(ftrace->machine);
+ }
+
+ return 0;
+}
+
+void perf_ftrace__cleanup_inlines(struct perf_ftrace *ftrace)
+{
+ if (ftrace->machine) {
+ machine__delete(ftrace->machine);
+ ftrace->machine = NULL;
+ }
+ ftrace_cpu_states__clear();
+}
diff --git a/tools/perf/util/ftrace.h b/tools/perf/util/ftrace.h
index 50216d3fef64..4140902d9de6 100644
--- a/tools/perf/util/ftrace.h
+++ b/tools/perf/util/ftrace.h
@@ -41,6 +41,8 @@ struct perf_ftrace {
int graph_tail;
bool filter_retaddr;
bool filter_retaddr_set;
+ bool use_inline;
+ struct machine *machine;
};
struct filter_entry {
@@ -95,7 +97,16 @@ perf_ftrace__latency_cleanup_bpf(struct perf_ftrace *ftrace __maybe_unused)
#endif /* HAVE_BPF_SKEL */
+struct machine;
+
bool ftrace_parse_retaddr(const char *str, char *sym_name, size_t sym_len, u64 *offset);
void ftrace_filter_retaddr(char *str);
+int ftrace_resolve_inlines(struct machine *machine,
+ const char *sym_name, u64 offset,
+ const char **inlined_names, int max_inlines);
+int ftrace_process_fgraph_line(struct perf_ftrace *ftrace, const char *line, FILE *out);
+void ftrace_cpu_states__clear(void);
+int perf_ftrace__setup_inlines(struct perf_ftrace *ftrace);
+void perf_ftrace__cleanup_inlines(struct perf_ftrace *ftrace);
#endif /* __PERF_FTRACE_H__ */
--
2.55.0
^ permalink raw reply [flat|nested] 5+ messages in thread* [PATCH perf-tools-next v2 4/4] perf ftrace: Support omitting execution duration in function graph tracer
2026-10-06 23:27 [PATCH perf-tools-next v2 0/4] perf ftrace: Support inlined functions and display enhancements in function graph tracer Aaron Tomlin
` (2 preceding siblings ...)
2026-10-06 23:27 ` [PATCH perf-tools-next v2 3/4] perf ftrace: Support display of inlined functions in function graph tracer Aaron Tomlin
@ 2026-10-06 23:27 ` Aaron Tomlin
3 siblings, 0 replies; 5+ messages in thread
From: Aaron Tomlin @ 2026-10-06 23:27 UTC (permalink / raw)
To: peterz, mingo, acme, namhyung
Cc: mark.rutland, alexander.shishkin, jolsa, irogers, adrian.hunter,
atomlin, linux-perf-users, linux-kernel
The Linux kernel's function graph tracer displays the execution duration
of completed functions by default via the funcgraph-duration trace
option. Whilst informative, in scenarios where the user is solely
interested in the call graph topology, call flow sequence, or inlined
hierarchy, the duration column can add visual clutter and widen each
trace line.
Introduce support for omitting function duration by exposing the
noduration sub-option under --graph-opts (and conversely
duration=[0|1]). When specified, perf ftrace configures the kernel's
tracing instance option funcgraph-duration to 0, and restores it to 1
upon resetting tracing options.
Signed-off-by: Aaron Tomlin <atomlin@atomlin.com>
---
tools/perf/Documentation/perf-ftrace.txt | 1 +
tools/perf/builtin-ftrace.c | 25 +++++++++++++++++++++++-
tools/perf/util/ftrace.h | 1 +
3 files changed, 26 insertions(+), 1 deletion(-)
diff --git a/tools/perf/Documentation/perf-ftrace.txt b/tools/perf/Documentation/perf-ftrace.txt
index ca0f1d523dbc..2e0b69a6e5ed 100644
--- a/tools/perf/Documentation/perf-ftrace.txt
+++ b/tools/perf/Documentation/perf-ftrace.txt
@@ -130,6 +130,7 @@ OPTIONS for 'perf ftrace trace'
- filter-retaddr - Filter out function return address comments (enabled by default with --inline).
- nosleep-time - Measure on-CPU time only for function_graph tracer.
- noirqs - Ignore functions that happen inside interrupt.
+ - noduration - Do not print function duration.
- verbose - Show process names, PIDs, timestamps, etc.
- thresh=<n> - Setup trace duration threshold in microseconds.
- depth=<n> - Set max depth for function graph tracer to follow.
diff --git a/tools/perf/builtin-ftrace.c b/tools/perf/builtin-ftrace.c
index e11418e61937..88df6b09d9b0 100644
--- a/tools/perf/builtin-ftrace.c
+++ b/tools/perf/builtin-ftrace.c
@@ -299,6 +299,7 @@ static void reset_tracing_options(struct perf_ftrace *ftrace __maybe_unused)
write_tracing_option_file("func_stack_trace", "0");
write_tracing_option_file("sleep-time", "1");
write_tracing_option_file("funcgraph-irqs", "1");
+ write_tracing_option_file("funcgraph-duration", "1");
write_tracing_option_file("funcgraph-proc", "0");
write_tracing_option_file("funcgraph-abstime", "0");
write_tracing_option_file("funcgraph-tail", "0");
@@ -598,6 +599,17 @@ static int set_tracing_funcgraph_irqs(struct perf_ftrace *ftrace)
return 0;
}
+static int set_tracing_funcgraph_duration(struct perf_ftrace *ftrace)
+{
+ if (!ftrace->graph_noduration)
+ return 0;
+
+ if (write_tracing_option_file("funcgraph-duration", "0") < 0)
+ return -1;
+
+ return 0;
+}
+
static int set_tracing_funcgraph_verbose(struct perf_ftrace *ftrace)
{
if (!ftrace->graph_verbose)
@@ -707,6 +719,11 @@ static int set_tracing_options(struct perf_ftrace *ftrace)
return -1;
}
+ if (set_tracing_funcgraph_duration(ftrace) < 0) {
+ pr_err("failed to set tracing option funcgraph-duration\n");
+ return -1;
+ }
+
if (set_tracing_funcgraph_verbose(ftrace) < 0) {
pr_err("failed to set tracing option funcgraph-proc/funcgraph-abstime\n");
return -1;
@@ -1747,6 +1764,7 @@ static int parse_graph_tracer_opts(const struct option *opt,
int ret;
struct perf_ftrace *ftrace = (struct perf_ftrace *) opt->value;
int filter_retaddr = -1;
+ int duration = -1;
struct sublevel_option graph_tracer_opts[] = {
{ .name = "args", .value_ptr = &ftrace->graph_args },
{ .name = "retval", .value_ptr = &ftrace->graph_retval },
@@ -1754,6 +1772,8 @@ static int parse_graph_tracer_opts(const struct option *opt,
{ .name = "retaddr", .value_ptr = &ftrace->graph_retaddr },
{ .name = "nosleep-time", .value_ptr = &ftrace->graph_nosleep_time },
{ .name = "noirqs", .value_ptr = &ftrace->graph_noirqs },
+ { .name = "noduration", .value_ptr = &ftrace->graph_noduration },
+ { .name = "duration", .value_ptr = &duration },
{ .name = "verbose", .value_ptr = &ftrace->graph_verbose },
{ .name = "thresh", .value_ptr = &ftrace->graph_thresh },
{ .name = "depth", .value_ptr = &ftrace->graph_depth },
@@ -1769,6 +1789,9 @@ static int parse_graph_tracer_opts(const struct option *opt,
if (ret)
return ret;
+ if (duration != -1)
+ ftrace->graph_noduration = !duration;
+
if (filter_retaddr != -1) {
ftrace->filter_retaddr = (filter_retaddr != 0);
ftrace->filter_retaddr_set = true;
@@ -1849,7 +1872,7 @@ int cmd_ftrace(int argc, const char **argv)
OPT_CALLBACK('g', "nograph-funcs", &ftrace.nograph_funcs, "func",
"Set nograph filter on given functions", parse_filter_func),
OPT_CALLBACK(0, "graph-opts", &ftrace, "options",
- "Graph tracer options, available options: args,retval,retval-hex,retaddr,filter-retaddr,nosleep-time,noirqs,verbose,thresh=<n>,depth=<n>",
+ "Graph tracer options, available options: args,retval,retval-hex,retaddr,filter-retaddr,nosleep-time,noirqs,noduration,verbose,thresh=<n>,depth=<n>,tail",
parse_graph_tracer_opts),
OPT_CALLBACK('m', "buffer-size", &ftrace.percpu_buffer_size, "size",
"Size of per cpu buffer, needs to use a B, K, M or G suffix.", parse_buffer_size),
diff --git a/tools/perf/util/ftrace.h b/tools/perf/util/ftrace.h
index 4140902d9de6..d202a09d112f 100644
--- a/tools/perf/util/ftrace.h
+++ b/tools/perf/util/ftrace.h
@@ -36,6 +36,7 @@ struct perf_ftrace {
int graph_retaddr;
int graph_nosleep_time;
int graph_noirqs;
+ int graph_noduration;
int graph_verbose;
int graph_thresh;
int graph_tail;
--
2.55.0
^ permalink raw reply [flat|nested] 5+ messages in thread