* [PATCH] bpf: clear stale IPv4 options after LWT encapsulation
@ 2026-09-15 17:01 Weiming Shi
0 siblings, 0 replies; only message in thread
From: Weiming Shi @ 2026-09-15 17:01 UTC (permalink / raw)
To: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko,
Eduard Zingerman, Kumar Kartikeya Dwivedi, Martin KaFai Lau,
Song Liu, Yonghong Song, Jiri Olsa, Emil Tsalapatis,
Ihor Solodrai, John Fastabend, David S . Miller, Eric Dumazet,
Jakub Kicinski, Paolo Abeni, Simon Horman
Cc: bpf, linux-kernel, netdev, Peter Oskolkov, Xiang Mei, Weiming Shi
From: Xiang Mei <xmei5@asu.edu>
bpf_lwt_push_ip_encap() rebases the network header after prepending an IP
header, but leaves IPCB(skb)->opt describing the inner IPv4 header. An
ingress LWT route can consequently make an ICMP error interpret an
inner-header byte as an option length and copy 255 bytes into 40 bytes of
stack storage. The trace decoded with scripts/decode_stacktrace.sh is:
BUG: KASAN: stack-out-of-bounds in __ip_options_echo
Write of size 255
Call Trace:
<IRQ>
__asan_memcpy (mm/kasan/shadow.c:106)
__ip_options_echo (net/ipv4/ip_options.c:96)
__icmp_send (net/ipv4/icmp.c:949)
ip_forward (net/ipv4/ip_forward.c:176)
lwtunnel_input (net/core/lwtunnel.c:465)
ip_rcv (net/ipv4/ip_input.c:612)
__netif_receive_skb_one_core (net/core/dev.c:6264)
process_backlog (net/core/dev.c:6728)
__napi_poll (net/core/dev.c:7787)
net_rx_action (net/core/dev.c:8007)
handle_softirqs (kernel/softirq.c:645)
do_softirq.part.0 (kernel/softirq.c:546)
</IRQ>
<TASK>
__local_bh_enable_ip (kernel/softirq.c:473)
__dev_queue_xmit (net/core/dev.c:4961)
packet_sendmsg (net/packet/af_packet.c:3143)
__sys_sendto (net/socket.c:2281)
__x64_sys_sendto (net/socket.c:2288)
do_syscall_64 (arch/x86/entry/syscall_64.c:84)
entry_SYSCALL_64_after_hwframe (arch/x86/entry/entry_64.S:121)
</TASK>
A helper-only reset can be restored by bpf_prog_run_save_cb(), while a
program without ctx->cb[] access can clone-redirect the skb before a
return-only reset. Track active LWT runs and their control-block family
in the BPF network context. Reset it immediately when the control block
is not BPF scratch space, and after bpf_prog_run_save_cb() restores it.
Preserve the state across nested runs and retain the ingress interface and
L3-slave state when the protocol family changes.
Fixes: 52f278774e79 ("bpf: implement BPF_LWT_ENCAP_IP mode in bpf_lwt_push_encap")
Reported-by: Weiming Shi <bestswngs@gmail.com>
Assisted-by: LLM
Signed-off-by: Xiang Mei <xmei5@asu.edu>
---
include/linux/filter.h | 10 ++++++
net/core/lwt_bpf.c | 81 ++++++++++++++++++++++++++++++++++++++++++
2 files changed, 91 insertions(+)
diff --git a/include/linux/filter.h b/include/linux/filter.h
index 39decde7fc730..0edd3e6ce563f 100644
--- a/include/linux/filter.h
+++ b/include/linux/filter.h
@@ -848,6 +848,15 @@ struct bpf_nh_params {
#define BPF_RI_F_CPU_MAP_INIT BIT(2)
#define BPF_RI_F_DEV_MAP_INIT BIT(3)
#define BPF_RI_F_XSK_MAP_INIT BIT(4)
+#define BPF_RI_F_LWT_IP_ENCAP BIT(5)
+#define BPF_RI_F_LWT_RUN BIT(6)
+
+struct bpf_lwt_ip_encap_state {
+ int iif;
+ __be16 cb_proto;
+ bool l3slave;
+ bool cb_access;
+};
struct bpf_redirect_info {
u64 tgt_index;
@@ -858,6 +867,7 @@ struct bpf_redirect_info {
enum bpf_map_type map_type;
struct bpf_nh_params nh;
u32 kern_flags;
+ struct bpf_lwt_ip_encap_state lwt_ip_encap;
};
struct bpf_net_context {
diff --git a/net/core/lwt_bpf.c b/net/core/lwt_bpf.c
index da49364ec63de..762d62c959f9f 100644
--- a/net/core/lwt_bpf.c
+++ b/net/core/lwt_bpf.c
@@ -36,19 +36,88 @@ static inline struct bpf_lwt *bpf_lwt_lwtunnel(struct lwtunnel_state *lwt)
#define NO_REDIRECT false
#define CAN_REDIRECT true
+static void bpf_lwt_reset_ip_cb(struct sk_buff *skb, __be16 orig_proto,
+ int iif, bool l3slave, bool use_new_proto)
+{
+ __be16 cb_proto = orig_proto;
+
+ if (use_new_proto)
+ cb_proto = skb->protocol;
+
+ if (cb_proto == htons(ETH_P_IP)) {
+ if (orig_proto == htons(ETH_P_IP)) {
+ memset(&IPCB(skb)->opt, 0, sizeof(IPCB(skb)->opt));
+ } else {
+ memset(IPCB(skb), 0, sizeof(*IPCB(skb)));
+ IPCB(skb)->iif = iif;
+ if (l3slave)
+ IPCB(skb)->flags |= IPSKB_L3SLAVE;
+ }
+ } else if (cb_proto == htons(ETH_P_IPV6)) {
+ memset(IP6CB(skb), 0, sizeof(*IP6CB(skb)));
+ IP6CB(skb)->iif = iif;
+ IP6CB(skb)->nhoff = offsetof(struct ipv6hdr, nexthdr);
+ if (l3slave)
+ IP6CB(skb)->flags |= IP6SKB_L3SLAVE;
+ } else if (orig_proto == htons(ETH_P_IP)) {
+ memset(&IPCB(skb)->opt, 0, sizeof(IPCB(skb)->opt));
+ }
+}
+
static int run_lwt_bpf(struct sk_buff *skb, struct bpf_lwt_prog *lwt,
struct dst_entry *dst, bool can_redirect)
{
struct bpf_net_context __bpf_net_ctx, *bpf_net_ctx;
+ struct bpf_lwt_ip_encap_state nested_lwt_ip_encap_state;
+ struct bpf_redirect_info *ri;
+ bool lwt_ip_encap, nested_lwt_ip_encap, nested_lwt_run;
+ __be16 orig_proto = skb->protocol;
+ bool l3slave = false;
+ int iif = 0;
int ret;
+ if (orig_proto == htons(ETH_P_IP)) {
+ iif = IPCB(skb)->iif;
+ l3slave = ipv4_l3mdev_skb(IPCB(skb)->flags);
+ } else if (orig_proto == htons(ETH_P_IPV6)) {
+ iif = IP6CB(skb)->iif;
+ l3slave = ipv6_l3mdev_skb(IP6CB(skb)->flags);
+ }
+
/* Disabling BH is needed to protect per-CPU bpf_redirect_info between
* BPF prog and skb_do_redirect().
*/
local_bh_disable();
bpf_net_ctx = bpf_net_ctx_set(&__bpf_net_ctx);
+ ri = bpf_net_ctx_get_ri();
+ nested_lwt_run = ri->kern_flags & BPF_RI_F_LWT_RUN;
+ nested_lwt_ip_encap = ri->kern_flags & BPF_RI_F_LWT_IP_ENCAP;
+ if (nested_lwt_run)
+ nested_lwt_ip_encap_state = ri->lwt_ip_encap;
+ ri->kern_flags &= ~BPF_RI_F_LWT_IP_ENCAP;
+ ri->kern_flags |= BPF_RI_F_LWT_RUN;
+ ri->lwt_ip_encap.iif = iif;
+ ri->lwt_ip_encap.cb_proto = orig_proto;
+ ri->lwt_ip_encap.l3slave = l3slave;
+ ri->lwt_ip_encap.cb_access = lwt->prog->cb_access;
bpf_compute_data_pointers(skb);
ret = bpf_prog_run_save_cb(lwt->prog, skb);
+ lwt_ip_encap = ri->kern_flags & BPF_RI_F_LWT_IP_ENCAP;
+ ri->kern_flags &= ~BPF_RI_F_LWT_IP_ENCAP;
+
+ if (lwt_ip_encap && ri->lwt_ip_encap.cb_access)
+ bpf_lwt_reset_ip_cb(skb, ri->lwt_ip_encap.cb_proto,
+ ri->lwt_ip_encap.iif,
+ ri->lwt_ip_encap.l3slave,
+ (ret == BPF_LWT_REROUTE &&
+ lwt->prog->type != BPF_PROG_TYPE_LWT_OUT) ||
+ (ret == BPF_REDIRECT && can_redirect));
+ if (nested_lwt_run)
+ ri->lwt_ip_encap = nested_lwt_ip_encap_state;
+ else
+ ri->kern_flags &= ~BPF_RI_F_LWT_RUN;
+ if (nested_lwt_ip_encap)
+ ri->kern_flags |= BPF_RI_F_LWT_IP_ENCAP;
switch (ret) {
case BPF_OK:
@@ -604,6 +673,7 @@ static int handle_gso_encap(struct sk_buff *skb, bool ipv4, int encap_len)
int bpf_lwt_push_ip_encap(struct sk_buff *skb, void *hdr, u32 len, bool ingress)
{
+ struct bpf_redirect_info *ri;
bool is_udp_tunnel;
struct iphdr *iph;
bool ipv4;
@@ -657,6 +727,7 @@ int bpf_lwt_push_ip_encap(struct sk_buff *skb, void *hdr, u32 len, bool ingress)
memcpy(skb_network_header(skb), hdr, len);
bpf_compute_data_pointers(skb);
skb_clear_hash(skb);
+ ri = bpf_net_ctx_get_ri();
if (ipv4) {
skb->protocol = htons(ETH_P_IP);
@@ -669,6 +740,16 @@ int bpf_lwt_push_ip_encap(struct sk_buff *skb, void *hdr, u32 len, bool ingress)
skb->protocol = htons(ETH_P_IPV6);
}
+ if (ri->kern_flags & BPF_RI_F_LWT_RUN) {
+ ri->kern_flags |= BPF_RI_F_LWT_IP_ENCAP;
+ if (!ri->lwt_ip_encap.cb_access) {
+ bpf_lwt_reset_ip_cb(skb, ri->lwt_ip_encap.cb_proto,
+ ri->lwt_ip_encap.iif,
+ ri->lwt_ip_encap.l3slave, true);
+ ri->lwt_ip_encap.cb_proto = skb->protocol;
+ }
+ }
+
if (skb_is_gso(skb))
return handle_gso_encap(skb, ipv4, len);
^ permalink raw reply [flat|nested] only message in thread
only message in thread, other threads:[~2026-09-15 17:02 UTC | newest]
Thread overview: (only message) (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-15 17:01 [PATCH] bpf: clear stale IPv4 options after LWT encapsulation Weiming Shi
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®