mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH v2] tracing/filters: Check perf permissions before accessing kernel data
@ 2026-10-07  5:03 Kyle Zeng
  2026-10-07  8:01 ` Masami Hiramatsu
  0 siblings, 1 reply; 2+ messages in thread
From: Kyle Zeng @ 2026-10-07  5:03 UTC (permalink / raw)
  To: linux-trace-kernel
  Cc: linux-kernel, rostedt, mhiramat, outbounddisclosures, Kyle Zeng, stable

perf_trace_event_perm() allows tracepoint counters that do not request
PERF_SAMPLE_RAW without raw tracepoint permissions. A self-targeted
event with exclude_kernel=1 can therefore reach the filter compiler even
at perf_event_paranoid=2.

The .function suffix resolves its operand through kallsyms. The result
of a numeric filter discloses whether an address belongs to a known
kernel symbol range, allowing the randomized kernel image base to be
recovered. A named filter also exposes the resolved range through the
counter when the tracepoint field is controlled by the caller.

Pointer-string filters expose kernel memory in the same way. A caller
can supply a kernel address as the filename argument to openat() and
install a string filter on sys_enter_openat. Without .ustring, the
filter uses strncpy_from_kernel_nofault() on that address. Whether the
counter increments reveals whether the kernel bytes match the pattern.
The nofault copy prevents faults but does not authorize disclosure.

Pass the filter's perf origin to the predicate parser and require
perf_allow_tracepoint() before resolving a .function operand or creating
a kernel-pointer string predicate. This uses the existing sysctl,
initial-namespace capability and LSM policy for raw tracepoint access.
Keep ordinary counting filters, explicit user-string predicates and
filters created through the separately controlled tracefs interfaces
unchanged.

Fixes: e6745a4da964 ("tracing: Add a way to filter function addresses to function names")
Fixes: 5967bd5c4239 ("tracing: Let filter_assign_type() detect FILTER_PTR_STRING")
Cc: stable@vger.kernel.org
Assisted-by: LLM
Signed-off-by: Kyle Zeng <kylebot@openai.com>
---
Changes in v2:
- Require raw-tracepoint permission for kernel-pointer string filters
  as well as .function predicates, preserving explicit .ustring filters.
- Add the pointer-string Fixes tag and describe the confirmed
  sys_enter_openat counter oracle.

 kernel/trace/trace_events_filter.c | 41 ++++++++++++++++++++++--------
 1 file changed, 31 insertions(+), 10 deletions(-)

diff --git a/kernel/trace/trace_events_filter.c b/kernel/trace/trace_events_filter.c
index 2b46ca536045..84d4a4fe56ad 100644
--- a/kernel/trace/trace_events_filter.c
+++ b/kernel/trace/trace_events_filter.c
@@ -1625,12 +1625,18 @@ static int filter_pred_fn_call(struct filter_pred *pred, void *event)
 	}
 }
 
+struct filter_parse_data {
+	struct trace_event_call *call;
+	bool is_perf;
+};
+
 /* Called when a predicate is encountered by predicate_parse() */
 static int parse_pred(const char *str, void *data,
 		      int pos, struct filter_parse_error *pe,
 		      struct filter_pred **pred_ptr)
 {
-	struct trace_event_call *call = data;
+	struct filter_parse_data *pdata = data;
+	struct trace_event_call *call = pdata->call;
 	struct ftrace_event_field *field;
 	struct filter_pred *pred = NULL;
 	unsigned long offset;
@@ -1687,6 +1693,16 @@ static int parse_pred(const char *str, void *data,
 		i += len;
 	}
 
+	/* Even counting filters can disclose kernel addresses and data. */
+	if (pdata->is_perf &&
+	    (function || (!ustring && field->filter_type == FILTER_PTR_STRING))) {
+		ret = perf_allow_tracepoint();
+		if (ret) {
+			parse_error(pe, ret, pos + i);
+			return ret;
+		}
+	}
+
 	while (isspace(str[i]))
 		i++;
 
@@ -2205,8 +2221,12 @@ static int calc_stack(const char *str, int *parens, int *preds, int *err)
 static int process_preds(struct trace_event_call *call,
 			 const char *filter_string,
 			 struct event_filter *filter,
-			 struct filter_parse_error *pe)
+			 struct filter_parse_error *pe, bool is_perf)
 {
+	struct filter_parse_data data = {
+		.call = call,
+		.is_perf = is_perf,
+	};
 	struct prog_entry *prog;
 	int nr_parens;
 	int nr_preds;
@@ -2232,7 +2252,7 @@ static int process_preds(struct trace_event_call *call,
 		return -EINVAL;
 
 	prog = predicate_parse(filter_string, nr_parens, nr_preds,
-			       parse_pred, call, pe);
+			       parse_pred, &data, pe);
 	if (IS_ERR(prog))
 		return PTR_ERR(prog);
 
@@ -2281,7 +2301,7 @@ static int process_system_preds(struct trace_subsystem_dir *dir,
 		if (!filter->filter_string)
 			goto fail_mem;
 
-		err = process_preds(file->event_call, filter_string, filter, pe);
+		err = process_preds(file->event_call, filter_string, filter, pe, false);
 		if (err) {
 			filter_disable(file);
 			parse_error(pe, FILT_ERR_BAD_SUBSYS_FILTER, 0);
@@ -2377,6 +2397,7 @@ static void create_filter_finish(struct filter_parse_error *pe)
  * @call: trace_event_call to create a filter for
  * @filter_string: filter string
  * @set_str: remember @filter_str and enable detailed error in filter
+ * @is_perf: filter is being created for a perf event
  * @filterp: out param for created filter (always updated on return)
  *           Must be a pointer that references a NULL pointer.
  *
@@ -2391,7 +2412,7 @@ static void create_filter_finish(struct filter_parse_error *pe)
  */
 static int create_filter(struct trace_array *tr,
 			 struct trace_event_call *call,
-			 char *filter_string, bool set_str,
+			 char *filter_string, bool set_str, bool is_perf,
 			 struct event_filter **filterp)
 {
 	struct filter_parse_error *pe = NULL;
@@ -2405,7 +2426,7 @@ static int create_filter(struct trace_array *tr,
 	if (err)
 		return err;
 
-	err = process_preds(call, filter_string, *filterp, pe);
+	err = process_preds(call, filter_string, *filterp, pe, is_perf);
 	if (err && set_str)
 		append_filter_err(tr, pe, *filterp);
 	create_filter_finish(pe);
@@ -2418,7 +2439,7 @@ int create_event_filter(struct trace_array *tr,
 			char *filter_str, bool set_str,
 			struct event_filter **filterp)
 {
-	return create_filter(tr, call, filter_str, set_str, filterp);
+	return create_filter(tr, call, filter_str, set_str, false, filterp);
 }
 
 /**
@@ -2476,7 +2497,7 @@ int apply_event_filter(struct trace_event_file *file, char *filter_string)
 		return 0;
 	}
 
-	err = create_filter(file->tr, call, filter_string, true, &filter);
+	err = create_filter(file->tr, call, filter_string, true, false, &filter);
 
 	/*
 	 * Always swap the call filter with the new filter
@@ -2721,7 +2742,7 @@ int ftrace_profile_set_filter(struct perf_event *event, int event_id,
 	if (event->filter)
 		return -EEXIST;
 
-	err = create_filter(NULL, call, filter_str, false, &filter);
+	err = create_filter(NULL, call, filter_str, false, true, &filter);
 	if (err)
 		goto free_filter;
 
@@ -2868,7 +2889,7 @@ static __init int ftrace_test_event_filter(void)
 		int err;
 
 		err = create_filter(NULL, &event_ftrace_test_filter,
-				    d->filter, false, &filter);
+				    d->filter, false, false, &filter);
 		if (err) {
 			printk(KERN_INFO
 			       "Failed to get filter for '%s', err %d\n",

base-commit: 602042bf29f6efde39cfb5fdd9289bf4854bc0c5

^ permalink raw reply	[flat|nested] 2+ messages in thread

* Re: [PATCH v2] tracing/filters: Check perf permissions before accessing kernel data
  2026-10-07  5:03 [PATCH v2] tracing/filters: Check perf permissions before accessing kernel data Kyle Zeng
@ 2026-10-07  8:01 ` Masami Hiramatsu
  0 siblings, 0 replies; 2+ messages in thread
From: Masami Hiramatsu @ 2026-10-07  8:01 UTC (permalink / raw)
  To: kylebot
  Cc: linux-trace-kernel, linux-kernel, rostedt, mhiramat,
	outbounddisclosures, Kyle Zeng, stable

On Tue, 06 Oct 2026 22:03:15 -0700, Kyle Zeng <kylebot@openai.com> wrote:
> perf_trace_event_perm() allows tracepoint counters that do not request
> PERF_SAMPLE_RAW without raw tracepoint permissions. A self-targeted
> event with exclude_kernel=1 can therefore reach the filter compiler even
> at perf_event_paranoid=2.
>
> The .function suffix resolves its operand through kallsyms. The result
> of a numeric filter discloses whether an address belongs to a known
> kernel symbol range, allowing the randomized kernel image base to be
> recovered. A named filter also exposes the resolved range through the
> counter when the tracepoint field is controlled by the caller.
>
> Pointer-string filters expose kernel memory in the same way. A caller
> can supply a kernel address as the filename argument to openat() and
> install a string filter on sys_enter_openat. Without .ustring, the
> filter uses strncpy_from_kernel_nofault() on that address. Whether the
> counter increments reveals whether the kernel bytes match the pattern.
> The nofault copy prevents faults but does not authorize disclosure.
>
> Pass the filter's perf origin to the predicate parser and require
> perf_allow_tracepoint() before resolving a .function operand or creating
> a kernel-pointer string predicate. This uses the existing sysctl,
> initial-namespace capability and LSM policy for raw tracepoint access.
> Keep ordinary counting filters, explicit user-string predicates and
> filters created through the separately controlled tracefs interfaces
> unchanged.
>
> Fixes: e6745a4da964 ("tracing: Add a way to filter function addresses to function names")
> Fixes: 5967bd5c4239 ("tracing: Let filter_assign_type() detect FILTER_PTR_STRING")
> Cc: stable@vger.kernel.org
> Assisted-by: LLM
> Signed-off-by: Kyle Zeng <kylebot@openai.com>

Looks good to me.

Reviewed-by: Masami Hiramatsu (Google) <mhiramat@kernel.org>

Thanks!

-- 
Masami Hiramatsu (Google) <mhiramat@kernel.org>

^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-10-07  8:01 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-07  5:03 [PATCH v2] tracing/filters: Check perf permissions before accessing kernel data Kyle Zeng
2026-10-07  8:01 ` Masami Hiramatsu

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®