mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* [PATCH bpf-next] libbpf: Resolve typeless ksyms in the kernel
@ 2026-10-09 13:07 Samuel Wu
  2026-10-10 10:09 ` Alan Maguire
  0 siblings, 1 reply; 2+ messages in thread
From: Samuel Wu @ 2026-10-09 13:07 UTC (permalink / raw)
  To: Andrii Nakryiko, Eduard Zingerman, Ihor Solodrai,
	Alexei Starovoitov, Daniel Borkmann, Kumar Kartikeya Dwivedi,
	Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
	Emil Tsalapatis, Nathan Chancellor, Nick Desaulniers,
	Bill Wendling, Justin Stitt
  Cc: Samuel Wu, kernel-team, bpf, linux-kernel, llvm

libbpf resolves typeless symbols by parsing /proc/kallsyms, which
dominates load time for BPF programs with typeless symbols. This mostly
comes from needing to do thousands of syscalls and a linear scan of
kallsyms.

Call bpf_kallsyms_lookup_name() from BPF_PROG_TYPE_SYSCALL, such that
the resolution can be done with a couple of syscalls and a binary
search. This is a similar approach to what the light skeleton does. The
fast path passes the name and result through ctx, which requires commit
ae5ef001aa98 ("bpf: Support variable offsets for syscall PTR_TO_CTX").

Any fast path error (e.g. kernels without ae5ef001aa98) falls back to
the legacy /proc/kallsyms scan, including strong misses, since the
kernel doesn't match LTO "<name>.llvm.<hash>" symbols. Weak misses
default to zero.

Depending on the device, I see ~9-11x loading speedup when resolving a
BPF program with 11 typeless lock externs.
  - x86_64 VM, 6.18 + ae5ef001aa98: 147ms -> 17ms
  - ARM64 phone, 7.1: 626ms -> 55ms

Signed-off-by: Samuel Wu <wusamuel@google.com>
---
 tools/lib/bpf/libbpf.c | 59 +++++++++++++++++++++++++++++++++++++++++-
 1 file changed, 58 insertions(+), 1 deletion(-)

diff --git a/tools/lib/bpf/libbpf.c b/tools/lib/bpf/libbpf.c
index cb09ded90773..7564ec6bde64 100644
--- a/tools/lib/bpf/libbpf.c
+++ b/tools/lib/bpf/libbpf.c
@@ -9590,6 +9590,58 @@ static int bpf_object__read_kallsyms_file(struct bpf_object *obj)
 	return libbpf_kallsyms_parse(kallsyms_cb, obj);
 }
 
+static int bpf_object__resolve_ksyms_in_kernel(struct bpf_object *obj)
+{
+	struct extern_desc *ext;
+	int i, prog_fd, err = 0;
+	struct {
+		__u64 addr;
+		char name[512];
+	} ctx = {};
+	LIBBPF_OPTS(bpf_prog_load_opts, opts, .prog_flags = BPF_F_SLEEPABLE);
+	LIBBPF_OPTS(bpf_test_run_opts, topts, .ctx_in = &ctx, .ctx_size_in = sizeof(ctx));
+	/* bpf_kallsyms_lookup_name(ctx.name, sizeof(ctx.name), 0, &ctx.addr) */
+	static const struct bpf_insn insns[] = {
+		BPF_MOV64_REG(BPF_REG_4, BPF_REG_1),
+		BPF_ALU64_IMM(BPF_ADD, BPF_REG_1, 8),
+		BPF_MOV64_IMM(BPF_REG_2, sizeof(ctx.name)),
+		BPF_MOV64_IMM(BPF_REG_3, 0),
+		BPF_EMIT_CALL(BPF_FUNC_kallsyms_lookup_name),
+		BPF_EXIT_INSN(),
+	};
+
+	if (obj->gen_loader || obj->token_fd)
+		return -EOPNOTSUPP;
+
+	prog_fd = bpf_prog_load(BPF_PROG_TYPE_SYSCALL, NULL, "GPL",
+				insns, ARRAY_SIZE(insns), &opts);
+	if (prog_fd < 0)
+		return prog_fd;
+
+	for (i = 0; i < obj->nr_extern; i++) {
+		ext = &obj->externs[i];
+		if (ext->type != EXT_KSYM || ext->ksym.type_id)
+			continue;
+		if (snprintf(ctx.name, sizeof(ctx.name), "%s", ext->name) >= sizeof(ctx.name)) {
+			err = -ENAMETOOLONG;
+			break;
+		}
+		err = bpf_prog_test_run_opts(prog_fd, &topts) ?: (int)topts.retval;
+		if (err == -ENOENT && ext->is_weak) {
+			err = 0;
+			continue;
+		}
+		if (err)
+			break;
+		ext->is_set = true;
+		ext->ksym.addr = ctx.addr;
+		pr_debug("extern (ksym) '%s': resolved in kernel to 0x%llx\n",
+			 ext->name, ext->ksym.addr);
+	}
+	close(prog_fd);
+	return err;
+}
+
 static int find_ksym_btf_id(struct bpf_object *obj, const char *ksym_name,
 			    __u16 kind, struct btf **res_btf,
 			    struct module_btf **res_mod_btf)
@@ -9863,7 +9915,12 @@ static int bpf_object__resolve_externs(struct bpf_object *obj,
 			return -EINVAL;
 	}
 	if (need_kallsyms) {
-		err = bpf_object__read_kallsyms_file(obj);
+		err = bpf_object__resolve_ksyms_in_kernel(obj);
+		if (err) {
+			pr_debug("failed to resolve ksyms in kernel: %s, falling back to /proc/kallsyms\n",
+				 errstr(err));
+			err = bpf_object__read_kallsyms_file(obj);
+		}
 		if (err)
 			return -EINVAL;
 	}
-- 
2.56.0.385.gd3acb90ef8-goog


^ permalink raw reply	[flat|nested] 2+ messages in thread

* Re: [PATCH bpf-next] libbpf: Resolve typeless ksyms in the kernel
  2026-10-09 13:07 [PATCH bpf-next] libbpf: Resolve typeless ksyms in the kernel Samuel Wu
@ 2026-10-10 10:09 ` Alan Maguire
  0 siblings, 0 replies; 2+ messages in thread
From: Alan Maguire @ 2026-10-10 10:09 UTC (permalink / raw)
  To: Samuel Wu, Andrii Nakryiko, Eduard Zingerman, Ihor Solodrai,
	Alexei Starovoitov, Daniel Borkmann, Kumar Kartikeya Dwivedi,
	Martin KaFai Lau, Song Liu, Yonghong Song, Jiri Olsa,
	Emil Tsalapatis, Nathan Chancellor, Nick Desaulniers,
	Bill Wendling, Justin Stitt
  Cc: kernel-team, bpf, linux-kernel, llvm

On 09/10/2026 14:07, Samuel Wu wrote:
> libbpf resolves typeless symbols by parsing /proc/kallsyms, which
> dominates load time for BPF programs with typeless symbols. This mostly
> comes from needing to do thousands of syscalls and a linear scan of
> kallsyms.
> 
> Call bpf_kallsyms_lookup_name() from BPF_PROG_TYPE_SYSCALL, such that
> the resolution can be done with a couple of syscalls and a binary
> search. This is a similar approach to what the light skeleton does. The
> fast path passes the name and result through ctx, which requires commit
> ae5ef001aa98 ("bpf: Support variable offsets for syscall PTR_TO_CTX").
> 
> Any fast path error (e.g. kernels without ae5ef001aa98) falls back to
> the legacy /proc/kallsyms scan, including strong misses, since the
> kernel doesn't match LTO "<name>.llvm.<hash>" symbols. Weak misses
> default to zero.
> 
> Depending on the device, I see ~9-11x loading speedup when resolving a
> BPF program with 11 typeless lock externs.
>   - x86_64 VM, 6.18 + ae5ef001aa98: 147ms -> 17ms
>   - ARM64 phone, 7.1: 626ms -> 55ms
> 
> Signed-off-by: Samuel Wu <wusamuel@google.com>
> ---
>  tools/lib/bpf/libbpf.c | 59 +++++++++++++++++++++++++++++++++++++++++-
>  1 file changed, 58 insertions(+), 1 deletion(-)
> 
> diff --git a/tools/lib/bpf/libbpf.c b/tools/lib/bpf/libbpf.c
> index cb09ded90773..7564ec6bde64 100644
> --- a/tools/lib/bpf/libbpf.c
> +++ b/tools/lib/bpf/libbpf.c
> @@ -9590,6 +9590,58 @@ static int bpf_object__read_kallsyms_file(struct bpf_object *obj)
>  	return libbpf_kallsyms_parse(kallsyms_cb, obj);
>  }
>  
> +static int bpf_object__resolve_ksyms_in_kernel(struct bpf_object *obj)
> +{
> +	struct extern_desc *ext;
> +	int i, prog_fd, err = 0;
> +	struct {
> +		__u64 addr;
> +		char name[512];
> +	} ctx = {};
> +	LIBBPF_OPTS(bpf_prog_load_opts, opts, .prog_flags = BPF_F_SLEEPABLE);
> +	LIBBPF_OPTS(bpf_test_run_opts, topts, .ctx_in = &ctx, .ctx_size_in = sizeof(ctx));
> +	/* bpf_kallsyms_lookup_name(ctx.name, sizeof(ctx.name), 0, &ctx.addr) */
> +	static const struct bpf_insn insns[] = {
> +		BPF_MOV64_REG(BPF_REG_4, BPF_REG_1),
> +		BPF_ALU64_IMM(BPF_ADD, BPF_REG_1, 8),
> +		BPF_MOV64_IMM(BPF_REG_2, sizeof(ctx.name)),
> +		BPF_MOV64_IMM(BPF_REG_3, 0),
> +		BPF_EMIT_CALL(BPF_FUNC_kallsyms_lookup_name),
> +		BPF_EXIT_INSN(),
> +	};
> +
> +	if (obj->gen_loader || obj->token_fd)
> +		return -EOPNOTSUPP;
> +
> +	prog_fd = bpf_prog_load(BPF_PROG_TYPE_SYSCALL, NULL, "GPL",
> +				insns, ARRAY_SIZE(insns), &opts);
> +	if (prog_fd < 0)
> +		return prog_fd;
> +
> +	for (i = 0; i < obj->nr_extern; i++) {
> +		ext = &obj->externs[i];
> +		if (ext->type != EXT_KSYM || ext->ksym.type_id)
> +			continue;
> +		if (snprintf(ctx.name, sizeof(ctx.name), "%s", ext->name) >= sizeof(ctx.name)) {
> +			err = -ENAMETOOLONG;
> +			break;
> +		}
> +		err = bpf_prog_test_run_opts(prog_fd, &topts) ?: (int)topts.retval;
> +		if (err == -ENOENT && ext->is_weak) {
> +			err = 0;
> +			continue;
> +		}
> +		if (err)
> +			break;
> +		ext->is_set = true;
> +		ext->ksym.addr = ctx.addr;
> +		pr_debug("extern (ksym) '%s': resolved in kernel to 0x%llx\n",
> +			 ext->name, ext->ksym.addr);
> +	}

Great to see the win in terms of resolution time, but I wonder if we could
speed things up further by resolving multiple symbols at once, since we have
all the externs available (either by dynamically creating your program that 
does symbol resolution as a side-effect, or adding a more direct resolution method)?

A single syscall could provide an even faster speedup, so might be worth a look.

Another concern would be a lot of production kernels won't have BPF test run
support baked in, so relying on that without fallback would be a problem.


> +	close(prog_fd);
> +	return err;
> +}
> +
>  static int find_ksym_btf_id(struct bpf_object *obj, const char *ksym_name,
>  			    __u16 kind, struct btf **res_btf,
>  			    struct module_btf **res_mod_btf)
> @@ -9863,7 +9915,12 @@ static int bpf_object__resolve_externs(struct bpf_object *obj,
>  			return -EINVAL;
>  	}
>  	if (need_kallsyms) {
> -		err = bpf_object__read_kallsyms_file(obj);
> +		err = bpf_object__resolve_ksyms_in_kernel(obj);
> +		if (err) {
> +			pr_debug("failed to resolve ksyms in kernel: %s, falling back to /proc/kallsyms\n",
> +				 errstr(err));
> +			err = bpf_object__read_kallsyms_file(obj);
> +		}
>  		if (err)
>  			return -EINVAL;
>  	}


^ permalink raw reply	[flat|nested] 2+ messages in thread

end of thread, other threads:[~2026-10-10 10:09 UTC | newest]

Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-10-09 13:07 [PATCH bpf-next] libbpf: Resolve typeless ksyms in the kernel Samuel Wu
2026-10-10 10:09 ` Alan Maguire

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®