* [net-next, v4 01/10] bnge: restructure VNIC and filter code
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
@ 2026-09-28 6:12 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:12 ` [net-next, v4 02/10] bnge: add NTUPLE/ARFS VNIC Vikas Gupta
` (8 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:12 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Move VNIC and L2 filter code out of bnge_netdev.c into dedicated
bnge_vnic.c/h and bnge_filter.c/h files in preparation for multi-VNIC,
RSS context, and NTUPLE filter support.
This is a code reorganization with no functional change, which helps
centralize the functions into their respective modules.
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
---
drivers/net/ethernet/broadcom/bnge/Makefile | 4 +-
drivers/net/ethernet/broadcom/bnge/bnge.h | 7 -
.../net/ethernet/broadcom/bnge/bnge_filter.c | 113 ++++++++++++
.../net/ethernet/broadcom/bnge/bnge_filter.h | 58 ++++++
.../ethernet/broadcom/bnge/bnge_hwrm_lib.c | 2 +
.../ethernet/broadcom/bnge/bnge_hwrm_lib.h | 2 +
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 170 +-----------------
.../net/ethernet/broadcom/bnge/bnge_netdev.h | 72 --------
.../net/ethernet/broadcom/bnge/bnge_resc.c | 31 +---
.../net/ethernet/broadcom/bnge/bnge_vnic.c | 100 +++++++++++
.../net/ethernet/broadcom/bnge/bnge_vnic.h | 63 +++++++
11 files changed, 350 insertions(+), 272 deletions(-)
create mode 100644 drivers/net/ethernet/broadcom/bnge/bnge_filter.c
create mode 100644 drivers/net/ethernet/broadcom/bnge/bnge_filter.h
create mode 100644 drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
create mode 100644 drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
diff --git a/drivers/net/ethernet/broadcom/bnge/Makefile b/drivers/net/ethernet/broadcom/bnge/Makefile
index 8e07cb307d21..d0a3cb24dc1d 100644
--- a/drivers/net/ethernet/broadcom/bnge/Makefile
+++ b/drivers/net/ethernet/broadcom/bnge/Makefile
@@ -12,4 +12,6 @@ bng_en-y := bnge_core.o \
bnge_ethtool.o \
bnge_auxr.o \
bnge_txrx.o \
- bnge_link.o
+ bnge_link.o \
+ bnge_vnic.o \
+ bnge_filter.o
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge.h b/drivers/net/ethernet/broadcom/bnge/bnge.h
index 4479ccd071f5..bde54ba5d50f 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge.h
@@ -163,13 +163,6 @@ struct bnge_dev {
#define BNGE_L2_FLTR_MAX_FLTR 1024
u32 *rss_indir_tbl;
-#define BNGE_RSS_TABLE_ENTRIES 64
-#define BNGE_RSS_TABLE_SIZE (BNGE_RSS_TABLE_ENTRIES * 4)
-#define BNGE_RSS_TABLE_MAX_TBL 8
-#define BNGE_MAX_RSS_TABLE_SIZE \
- (BNGE_RSS_TABLE_SIZE * BNGE_RSS_TABLE_MAX_TBL)
-#define BNGE_MAX_RSS_TABLE_ENTRIES \
- (BNGE_RSS_TABLE_ENTRIES * BNGE_RSS_TABLE_MAX_TBL)
u16 rss_indir_tbl_entries;
u32 rss_cap;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
new file mode 100644
index 000000000000..a8bb441ebef9
--- /dev/null
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
@@ -0,0 +1,113 @@
+// SPDX-License-Identifier: GPL-2.0
+// Copyright (c) 2026 Broadcom.
+
+#include <linux/kernel.h>
+#include <linux/dma-mapping.h>
+#include <linux/jhash.h>
+
+#include "bnge.h"
+#include "bnge_netdev.h"
+#include "bnge_vnic.h"
+#include "bnge_hwrm_lib.h"
+#include "bnge_filter.h"
+
+void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr)
+{
+ if (!refcount_dec_and_test(&fltr->refcnt))
+ return;
+ hlist_del_rcu(&fltr->base.hlist);
+ kfree_rcu(fltr, base.rcu);
+}
+
+static void bnge_init_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_filter *fltr,
+ struct bnge_l2_key *key, u32 idx)
+{
+ struct hlist_head *head;
+
+ ether_addr_copy(fltr->l2_key.dst_mac_addr, key->dst_mac_addr);
+ fltr->l2_key.vlan = key->vlan;
+ fltr->base.type = BNGE_FLTR_TYPE_L2;
+
+ head = &bn->l2_fltr_hash_tbl[idx];
+ hlist_add_head_rcu(&fltr->base.hlist, head);
+ refcount_set(&fltr->refcnt, 1);
+}
+
+static struct bnge_l2_filter *__bnge_lookup_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u32 idx)
+{
+ struct bnge_l2_filter *fltr;
+ struct hlist_head *head;
+
+ head = &bn->l2_fltr_hash_tbl[idx];
+ hlist_for_each_entry_rcu(fltr, head, base.hlist) {
+ struct bnge_l2_key *l2_key = &fltr->l2_key;
+
+ if (ether_addr_equal(l2_key->dst_mac_addr, key->dst_mac_addr) &&
+ l2_key->vlan == key->vlan)
+ return fltr;
+ }
+ return NULL;
+}
+
+static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u32 idx)
+{
+ struct bnge_l2_filter *fltr;
+
+ rcu_read_lock();
+ fltr = __bnge_lookup_l2_filter(bn, key, idx);
+ if (fltr)
+ refcount_inc(&fltr->refcnt);
+ rcu_read_unlock();
+ return fltr;
+}
+
+static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ gfp_t gfp)
+{
+ struct bnge_l2_filter *fltr;
+ u32 idx;
+
+ idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
+ BNGE_L2_FLTR_HASH_MASK;
+ fltr = bnge_lookup_l2_filter(bn, key, idx);
+ if (fltr)
+ return fltr;
+
+ fltr = kzalloc_obj(*fltr, gfp);
+ if (!fltr)
+ return ERR_PTR(-ENOMEM);
+
+ bnge_init_l2_filter(bn, fltr, key, idx);
+ return fltr;
+}
+
+int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
+ const u8 *mac_addr)
+{
+ struct bnge_l2_filter *fltr;
+ struct bnge_l2_key key;
+ int rc;
+
+ ether_addr_copy(key.dst_mac_addr, mac_addr);
+ key.vlan = 0;
+ fltr = bnge_alloc_l2_filter(bn, &key, GFP_KERNEL);
+ if (IS_ERR(fltr))
+ return PTR_ERR(fltr);
+
+ fltr->base.fw_vnic_id = bn->vnic_info[vnic_id].fw_vnic_id;
+ rc = bnge_hwrm_l2_filter_alloc(bn->bd, fltr);
+ if (rc)
+ goto err_del_l2_filter;
+ bn->vnic_info[vnic_id].l2_filters[idx] = fltr;
+ return rc;
+
+err_del_l2_filter:
+ bnge_del_l2_filter(bn, fltr);
+ return rc;
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
new file mode 100644
index 000000000000..ea1acefd70b4
--- /dev/null
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
@@ -0,0 +1,58 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Copyright (c) 2026 Broadcom */
+
+#ifndef _BNGE_FILTER_H_
+#define _BNGE_FILTER_H_
+
+#include <linux/types.h>
+#include <linux/if_ether.h>
+#include <linux/refcount.h>
+#include <linux/jhash.h>
+
+struct bnge_net;
+
+enum {
+ BNGE_FLTR_TYPE_L2 = 1
+};
+
+enum {
+ BNGE_FLTR_VALID,
+ BNGE_FLTR_FW_DELETED
+};
+
+struct bnge_filter_base {
+ struct hlist_node hlist;
+ struct list_head list_node;
+ __le64 filter_id;
+ u8 type;
+ u8 flags;
+ u16 rxq;
+ u16 fw_vnic_id;
+ u16 vf_idx;
+ unsigned long state;
+
+ struct rcu_head rcu;
+};
+
+struct bnge_l2_key {
+ union {
+ struct {
+ u8 dst_mac_addr[ETH_ALEN];
+ u16 vlan;
+ };
+ u32 filter_key;
+ };
+};
+
+#define BNGE_L2_KEY_SIZE (sizeof(struct bnge_l2_key) / 4)
+struct bnge_l2_filter {
+ /* base filter must be the first member */
+ struct bnge_filter_base base;
+ struct bnge_l2_key l2_key;
+ refcount_t refcnt;
+};
+
+void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr);
+int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
+ const u8 *mac_addr);
+#endif /* _BNGE_FILTER_H_ */
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
index 651c5e783516..91d246c8dcf2 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
@@ -15,6 +15,8 @@
#include "bnge_rmem.h"
#include "bnge_resc.h"
#include "bnge_netdev.h"
+#include "bnge_vnic.h"
+#include "bnge_filter.h"
static const u16 bnge_async_events_arr[] = {
ASYNC_EVENT_CMPL_EVENT_ID_LINK_STATUS_CHANGE,
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
index bf452e390d5b..ae03041c36ac 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
@@ -4,6 +4,8 @@
#ifndef _BNGE_HWRM_LIB_H_
#define _BNGE_HWRM_LIB_H_
+struct bnge_l2_filter;
+
#define BNGE_PLC_EN_JUMBO_THRES_VALID \
VNIC_PLCMODES_CFG_REQ_ENABLES_JUMBO_THRESH_VALID
#define BNGE_PLC_EN_HDS_THRES_VALID \
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index a4288f0258f8..9e64b1933c02 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -25,6 +25,8 @@
#include "bnge_ethtool.h"
#include "bnge_rmem.h"
#include "bnge_txrx.h"
+#include "bnge_vnic.h"
+#include "bnge_filter.h"
#define BNGE_RING_TO_TC_OFF(bd, tx) \
((tx) % (bd)->tx_nr_rings_per_tc)
@@ -1977,174 +1979,6 @@ static int bnge_hwrm_ring_alloc(struct bnge_net *bn)
return rc;
}
-void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic)
-{
- __le16 *ring_tbl = vnic->rss_table;
- struct bnge_rx_ring_info *rxr;
- struct bnge_dev *bd = bn->bd;
- u16 tbl_size, i;
-
- tbl_size = bnge_get_rxfh_indir_size(bd);
-
- for (i = 0; i < tbl_size; i++) {
- u32 j;
-
- j = bd->rss_indir_tbl[i];
- rxr = &bn->rx_ring[j];
-
- *ring_tbl++ = cpu_to_le16(rxr->rx_ring_struct.fw_ring_id);
- *ring_tbl++ = cpu_to_le16(bnge_cp_ring_for_rx(rxr));
- }
-}
-
-static int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
- struct bnge_vnic_info *vnic)
-{
- int rc;
-
- rc = bnge_hwrm_vnic_set_rss(bn, vnic, true);
- if (rc) {
- netdev_err(bn->netdev, "hwrm vnic %d set rss failure rc: %d\n",
- vnic->vnic_id, rc);
- return rc;
- }
- rc = bnge_hwrm_vnic_cfg(bn, vnic);
- if (rc)
- netdev_err(bn->netdev, "hwrm vnic %d cfg failure rc: %d\n",
- vnic->vnic_id, rc);
- return rc;
-}
-
-static int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic)
-{
- struct bnge_dev *bd = bn->bd;
- int rc, i, nr_ctxs;
-
- nr_ctxs = bnge_cal_nr_rss_ctxs(bd->rx_nr_rings);
- for (i = 0; i < nr_ctxs; i++) {
- rc = bnge_hwrm_vnic_ctx_alloc(bd, vnic, i);
- if (rc) {
- netdev_err(bn->netdev, "hwrm vnic %d ctx %d alloc failure rc: %d\n",
- vnic->vnic_id, i, rc);
- return -ENOMEM;
- }
- bn->rsscos_nr_ctxs++;
- }
-
- rc = bnge_hwrm_vnic_rss_cfg(bn, vnic);
- if (rc)
- return rc;
-
- if (bnge_is_agg_reqd(bd)) {
- rc = bnge_hwrm_vnic_set_hds(bn, vnic);
- if (rc)
- netdev_err(bn->netdev, "hwrm vnic %d set hds failure rc: %d\n",
- vnic->vnic_id, rc);
- }
- return rc;
-}
-
-static void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr)
-{
- if (!refcount_dec_and_test(&fltr->refcnt))
- return;
- hlist_del_rcu(&fltr->base.hash);
- kfree_rcu(fltr, base.rcu);
-}
-
-static void bnge_init_l2_filter(struct bnge_net *bn,
- struct bnge_l2_filter *fltr,
- struct bnge_l2_key *key, u32 idx)
-{
- struct hlist_head *head;
-
- ether_addr_copy(fltr->l2_key.dst_mac_addr, key->dst_mac_addr);
- fltr->l2_key.vlan = key->vlan;
- fltr->base.type = BNGE_FLTR_TYPE_L2;
-
- head = &bn->l2_fltr_hash_tbl[idx];
- hlist_add_head_rcu(&fltr->base.hash, head);
- refcount_set(&fltr->refcnt, 1);
-}
-
-static struct bnge_l2_filter *__bnge_lookup_l2_filter(struct bnge_net *bn,
- struct bnge_l2_key *key,
- u32 idx)
-{
- struct bnge_l2_filter *fltr;
- struct hlist_head *head;
-
- head = &bn->l2_fltr_hash_tbl[idx];
- hlist_for_each_entry_rcu(fltr, head, base.hash) {
- struct bnge_l2_key *l2_key = &fltr->l2_key;
-
- if (ether_addr_equal(l2_key->dst_mac_addr, key->dst_mac_addr) &&
- l2_key->vlan == key->vlan)
- return fltr;
- }
- return NULL;
-}
-
-static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
- struct bnge_l2_key *key,
- u32 idx)
-{
- struct bnge_l2_filter *fltr;
-
- rcu_read_lock();
- fltr = __bnge_lookup_l2_filter(bn, key, idx);
- if (fltr)
- refcount_inc(&fltr->refcnt);
- rcu_read_unlock();
- return fltr;
-}
-
-static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
- struct bnge_l2_key *key,
- gfp_t gfp)
-{
- struct bnge_l2_filter *fltr;
- u32 idx;
-
- idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
- BNGE_L2_FLTR_HASH_MASK;
- fltr = bnge_lookup_l2_filter(bn, key, idx);
- if (fltr)
- return fltr;
-
- fltr = kzalloc_obj(*fltr, gfp);
- if (!fltr)
- return ERR_PTR(-ENOMEM);
-
- bnge_init_l2_filter(bn, fltr, key, idx);
- return fltr;
-}
-
-static int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
- const u8 *mac_addr)
-{
- struct bnge_l2_filter *fltr;
- struct bnge_l2_key key;
- int rc;
-
- ether_addr_copy(key.dst_mac_addr, mac_addr);
- key.vlan = 0;
- fltr = bnge_alloc_l2_filter(bn, &key, GFP_KERNEL);
- if (IS_ERR(fltr))
- return PTR_ERR(fltr);
-
- fltr->base.fw_vnic_id = bn->vnic_info[vnic_id].fw_vnic_id;
- rc = bnge_hwrm_l2_filter_alloc(bn->bd, fltr);
- if (rc)
- goto err_del_l2_filter;
- bn->vnic_info[vnic_id].l2_filters[idx] = fltr;
- return rc;
-
-err_del_l2_filter:
- bnge_del_l2_filter(bn, fltr);
- return rc;
-}
-
static bool bnge_mc_list_updated(struct bnge_net *bn, u32 *rx_mask,
const struct netdev_hw_addr_list *mc)
{
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index 476b5bab96fe..ee649cc644db 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -558,8 +558,6 @@ struct bnge_napi {
};
#define INVALID_STATS_CTX_ID -1
-#define BNGE_VNIC_DEFAULT 0
-#define BNGE_MAX_UC_ADDRS 4
#define BNGE_RX_MASK_CFG_FLAGS \
(CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS | \
@@ -567,78 +565,8 @@ struct bnge_napi {
CFA_L2_SET_RX_MASK_REQ_MASK_ALL_MCAST | \
CFA_L2_SET_RX_MASK_REQ_MASK_BCAST)
-struct bnge_vnic_info {
- u16 fw_vnic_id;
-#define BNGE_MAX_CTX_PER_VNIC 8
- u16 fw_rss_cos_lb_ctx[BNGE_MAX_CTX_PER_VNIC];
- u16 mru;
- /* index 0 always dev_addr */
- struct bnge_l2_filter *l2_filters[BNGE_MAX_UC_ADDRS];
- u16 uc_filter_count;
- u8 *uc_list;
- dma_addr_t rss_table_dma_addr;
- __le16 *rss_table;
- dma_addr_t rss_hash_key_dma_addr;
- u64 *rss_hash_key;
- int rss_table_size;
-#define BNGE_RSS_TABLE_ENTRIES 64
-#define BNGE_RSS_TABLE_SIZE (BNGE_RSS_TABLE_ENTRIES * 4)
-#define BNGE_RSS_TABLE_MAX_TBL 8
-#define BNGE_MAX_RSS_TABLE_SIZE \
- (BNGE_RSS_TABLE_SIZE * BNGE_RSS_TABLE_MAX_TBL)
- u32 rx_mask;
-
- u8 *mc_list;
- int mc_list_size;
- int mc_list_count;
- dma_addr_t mc_list_mapping;
-#define BNGE_MAX_MC_ADDRS 16
-
- u32 flags;
-#define BNGE_VNIC_RSS_FLAG 1
-#define BNGE_VNIC_MCAST_FLAG 4
-#define BNGE_VNIC_UCAST_FLAG 8
- u32 vnic_id;
-};
-
-struct bnge_filter_base {
- struct hlist_node hash;
- struct list_head list;
- __le64 filter_id;
- u8 type;
-#define BNGE_FLTR_TYPE_L2 2
- u8 flags;
- u16 rxq;
- u16 fw_vnic_id;
- u16 vf_idx;
- unsigned long state;
-#define BNGE_FLTR_VALID 0
-#define BNGE_FLTR_FW_DELETED 2
-
- struct rcu_head rcu;
-};
-
-struct bnge_l2_key {
- union {
- struct {
- u8 dst_mac_addr[ETH_ALEN];
- u16 vlan;
- };
- u32 filter_key;
- };
-};
-
-#define BNGE_L2_KEY_SIZE (sizeof(struct bnge_l2_key) / 4)
-struct bnge_l2_filter {
- /* base filter must be the first member */
- struct bnge_filter_base base;
- struct bnge_l2_key l2_key;
- refcount_t refcnt;
-};
-
u32 bnge_cp_ring_for_rx(struct bnge_rx_ring_info *rxr);
u32 bnge_cp_ring_for_tx(struct bnge_tx_ring_info *txr);
-void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic);
int bnge_alloc_rx_data(struct bnge_net *bn, struct bnge_rx_ring_info *rxr,
u16 prod, gfp_t gfp);
u16 bnge_find_next_agg_idx(struct bnge_rx_ring_info *rxr, u16 idx);
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
index 4711dd4945ff..69a894b52485 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
@@ -11,6 +11,7 @@
#include "bnge_hwrm.h"
#include "bnge_hwrm_lib.h"
#include "bnge_resc.h"
+#include "bnge_vnic.h"
static u16 bnge_num_tx_to_cp(struct bnge_dev *bd, u16 tx)
{
@@ -186,13 +187,13 @@ int bnge_cal_nr_rss_ctxs(u16 rx_rings)
BNGE_RSS_TABLE_ENTRIES);
}
-static u16 bnge_rss_ctxs_in_use(struct bnge_dev *bd,
- struct bnge_hw_rings *hwr)
+static u16 bnge_get_total_rss_ctxs(struct bnge_dev *bd,
+ struct bnge_hw_rings *hwr)
{
return bnge_cal_nr_rss_ctxs(hwr->grp);
}
-static u16 bnge_get_total_vnics(struct bnge_dev *bd, u16 rx_rings)
+static u16 bnge_get_total_vnics(struct bnge_dev *bd)
{
return 1;
}
@@ -203,24 +204,6 @@ u32 bnge_get_rxfh_indir_size(struct bnge_dev *bd)
BNGE_RSS_TABLE_ENTRIES;
}
-static void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd)
-{
- u16 max_entries, pad;
- u32 *rss_indir_tbl;
- int i;
-
- max_entries = bnge_get_rxfh_indir_size(bd);
- rss_indir_tbl = &bd->rss_indir_tbl[0];
-
- for (i = 0; i < max_entries; i++)
- rss_indir_tbl[i] = ethtool_rxfh_indir_default(i,
- bd->rx_nr_rings);
-
- pad = bd->rss_indir_tbl_entries - max_entries;
- if (pad)
- memset(&rss_indir_tbl[i], 0, pad * sizeof(*rss_indir_tbl));
-}
-
static void bnge_copy_reserved_rings(struct bnge_dev *bd,
struct bnge_hw_rings *hwr)
{
@@ -253,7 +236,7 @@ static bool bnge_need_reserve_rings(struct bnge_dev *bd)
if (hw_resc->resv_tx_rings != bd->tx_nr_rings)
return true;
- vnic = bnge_get_total_vnics(bd, rx);
+ vnic = bnge_get_total_vnics(bd);
if (bnge_is_agg_reqd(bd))
rx <<= 1;
@@ -299,12 +282,12 @@ int bnge_reserve_rings(struct bnge_dev *bd)
sh = true;
hwr.cmpl = hwr.rx + hwr.tx;
- hwr.vnic = bnge_get_total_vnics(bd, hwr.rx);
+ hwr.vnic = bnge_get_total_vnics(bd);
if (bnge_is_agg_reqd(bd))
hwr.rx <<= 1;
hwr.grp = bd->rx_nr_rings;
- hwr.rss_ctx = bnge_rss_ctxs_in_use(bd, &hwr);
+ hwr.rss_ctx = bnge_get_total_rss_ctxs(bd, &hwr);
hwr.stat = bnge_func_stat_ctxs_demand(bd);
old_rx_rings = bd->hw_resc.resv_rx_rings;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
new file mode 100644
index 000000000000..3a6c8f0a5954
--- /dev/null
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
@@ -0,0 +1,100 @@
+// SPDX-License-Identifier: GPL-2.0
+// Copyright (c) 2026 Broadcom.
+
+#include <linux/kernel.h>
+#include <linux/dma-mapping.h>
+#include <linux/ethtool.h>
+
+#include "bnge.h"
+#include "bnge_netdev.h"
+#include "bnge_vnic.h"
+#include "bnge_hwrm_lib.h"
+#include "bnge_filter.h"
+#include "bnge_resc.h"
+
+void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd)
+{
+ u16 max_entries, pad;
+ u32 *rss_indir_tbl;
+ u16 i;
+
+ max_entries = bnge_get_rxfh_indir_size(bd);
+ rss_indir_tbl = &bd->rss_indir_tbl[0];
+
+ for (i = 0; i < max_entries; i++)
+ rss_indir_tbl[i] = ethtool_rxfh_indir_default(i,
+ bd->rx_nr_rings);
+
+ pad = bd->rss_indir_tbl_entries - max_entries;
+ if (pad)
+ memset(&rss_indir_tbl[i], 0, pad * sizeof(*rss_indir_tbl));
+}
+
+void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic)
+{
+ __le16 *ring_tbl = vnic->rss_table;
+ struct bnge_rx_ring_info *rxr;
+ struct bnge_dev *bd = bn->bd;
+ u16 tbl_size, i;
+
+ tbl_size = bnge_get_rxfh_indir_size(bd);
+
+ for (i = 0; i < tbl_size; i++) {
+ u16 ring_id, j;
+
+ j = bd->rss_indir_tbl[i];
+ rxr = &bn->rx_ring[j];
+
+ ring_id = rxr->rx_ring_struct.fw_ring_id;
+ *ring_tbl++ = cpu_to_le16(ring_id);
+ ring_id = bnge_cp_ring_for_rx(rxr);
+ *ring_tbl++ = cpu_to_le16(ring_id);
+ }
+}
+
+int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
+ struct bnge_vnic_info *vnic)
+{
+ int rc;
+
+ rc = bnge_hwrm_vnic_set_rss(bn, vnic, true);
+ if (rc) {
+ netdev_err(bn->netdev, "hwrm vnic %d set rss failure rc: %d\n",
+ vnic->vnic_id, rc);
+ return rc;
+ }
+ rc = bnge_hwrm_vnic_cfg(bn, vnic);
+ if (rc)
+ netdev_err(bn->netdev, "hwrm vnic %d cfg failure rc: %d\n",
+ vnic->vnic_id, rc);
+ return rc;
+}
+
+int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic)
+{
+ struct bnge_dev *bd = bn->bd;
+ int rc, i, nr_ctxs;
+
+ nr_ctxs = bnge_cal_nr_rss_ctxs(bd->rx_nr_rings);
+ for (i = 0; i < nr_ctxs; i++) {
+ rc = bnge_hwrm_vnic_ctx_alloc(bd, vnic, i);
+ if (rc) {
+ netdev_err(bn->netdev, "hwrm vnic %d ctx %d alloc failure rc: %d\n",
+ vnic->vnic_id, i, rc);
+ return -ENOMEM;
+ }
+ bn->rsscos_nr_ctxs++;
+ }
+
+ rc = bnge_hwrm_vnic_rss_cfg(bn, vnic);
+ if (rc)
+ return rc;
+
+ if (bnge_is_agg_reqd(bd)) {
+ rc = bnge_hwrm_vnic_set_hds(bn, vnic);
+ if (rc)
+ netdev_err(bn->netdev, "hwrm vnic %d set hds failure rc: %d\n",
+ vnic->vnic_id, rc);
+ }
+ return rc;
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
new file mode 100644
index 000000000000..c81d3a64561e
--- /dev/null
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
@@ -0,0 +1,63 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/* Copyright (c) 2026 Broadcom */
+
+#ifndef _BNGE_VNIC_H_
+#define _BNGE_VNIC_H_
+
+struct bnge_net;
+struct bnge_dev;
+struct bnge_l2_filter;
+
+#define BNGE_RSS_TABLE_ENTRIES 64
+#define BNGE_RSS_TABLE_SIZE (BNGE_RSS_TABLE_ENTRIES * 4)
+#define BNGE_RSS_TABLE_MAX_TBL 8
+#define BNGE_MAX_RSS_TABLE_SIZE \
+ (BNGE_RSS_TABLE_SIZE * BNGE_RSS_TABLE_MAX_TBL)
+#define BNGE_MAX_RSS_TABLE_ENTRIES \
+ (BNGE_RSS_TABLE_ENTRIES * BNGE_RSS_TABLE_MAX_TBL)
+
+#define BNGE_MAX_CTX_PER_VNIC 8
+
+#define BNGE_MAX_MC_ADDRS 16
+#define BNGE_MAX_UC_ADDRS 4
+
+enum {
+ BNGE_VNIC_DEFAULT = 0
+};
+
+enum {
+ BNGE_VNIC_RSS_FLAG = BIT(0),
+ BNGE_VNIC_MCAST_FLAG = BIT(1),
+ BNGE_VNIC_UCAST_FLAG = BIT(2)
+};
+
+struct bnge_vnic_info {
+ u16 fw_vnic_id;
+ u16 fw_rss_cos_lb_ctx[BNGE_MAX_CTX_PER_VNIC];
+ u16 mru;
+ /* index 0 always dev_addr */
+ struct bnge_l2_filter *l2_filters[BNGE_MAX_UC_ADDRS];
+ u16 uc_filter_count;
+ u8 *uc_list;
+ dma_addr_t rss_table_dma_addr;
+ __le16 *rss_table;
+ dma_addr_t rss_hash_key_dma_addr;
+ u64 *rss_hash_key;
+ int rss_table_size;
+ u32 rx_mask;
+
+ u8 *mc_list;
+ int mc_list_size;
+ int mc_list_count;
+ dma_addr_t mc_list_mapping;
+
+ u32 flags;
+ u32 vnic_id;
+};
+
+void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic);
+int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
+ struct bnge_vnic_info *vnic);
+int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic);
+void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd);
+#endif /* _BNGE_VNIC_H_ */
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 01/10] bnge: restructure VNIC and filter code
2026-09-28 6:12 ` [net-next, v4 01/10] bnge: restructure VNIC and filter code Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found more severe
issue(s) in other patches of this series. This mail is only an FYI, in case
you want to address the nit picks below when you respin.
Low priority / nit picks for your consideration:
- [Low] The commit message does not describe everything the patch
changes.
Pre-existing issues:
- [Medium] The RSS arrays are sized for at most 8 contexts, but the
context count that indexes them has no upper bound.
- [Medium] bnge_cal_nr_rss_ctxs() returns too few RSS contexts for some
ring counts, so some receive queues never appear in the default…
- [Medium] At this commit, a duplicate unicast MAC shares one refcounted
filter, which causes a firmware filter leak and a double firmware free.
--
bnge: restructure VNIC and filter code
This patch moves the VNIC and L2 filter code out of bnge_netdev.c, and the
default RSS indirection table helper out of bnge_resc.c, into new
bnge_vnic.c/h and bnge_filter.c/h files. The commit message describes it
as a reorganization with no functional change, in preparation for
multi-VNIC, RSS context and NTUPLE filter support.
> Move VNIC and L2 filter code out of bnge_netdev.c into dedicated
> bnge_vnic.c/h and bnge_filter.c/h files in preparation for multi-VNIC,
> RSS context, and NTUPLE filter support.
>
> This is a code reorganization with no functional change, which helps
> centralize the functions into their respective modules.
[Severity: Low]
This isn't a bug, but does this description cover everything the patch
changes? Besides moving code out of bnge_netdev.c, the patch also:
- renumbers constants while turning them into enums: BNGE_FLTR_TYPE_L2
goes from 2 to 1, BNGE_FLTR_FW_DELETED from 2 to 1,
BNGE_VNIC_MCAST_FLAG from 4 to BIT(1) and BNGE_VNIC_UCAST_FLAG from 8
to BIT(2)
- renames the struct bnge_filter_base members hash and list to hlist and
list_node
- renames bnge_rss_ctxs_in_use() to bnge_get_total_rss_ctxs() and drops
the rx_rings argument of bnge_get_total_vnics()
- narrows types: j in bnge_fill_hw_rss_tbl() goes from u32 to u16, the
ring ids now go through a u16 temporary, and the loop counter in
bnge_set_dflt_rss_indir_tbl() goes from int to u16
- makes bnge_set_dflt_rss_indir_tbl() and bnge_hwrm_vnic_rss_cfg()
non-static, and moves bnge_set_dflt_rss_indir_tbl() out of
bnge_resc.c, not bnge_netdev.c
All of these values seem to stay inside the driver, so runtime behaviour
looks unchanged. Even so, the removed and added lines can't be checked
against each other as a pure move.
Could the commit message list these changes? Another option is to split
the patch into a pure move followed by a cleanup.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> new file mode 100644
> index 0000000000000..a8bb441ebef9a
> --- /dev/null
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> @@ -0,0 +1,113 @@
[ ... ]
> +static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
> + struct bnge_l2_key *key,
> + u32 idx)
> +{
> + struct bnge_l2_filter *fltr;
> +
> + rcu_read_lock();
> + fltr = __bnge_lookup_l2_filter(bn, key, idx);
> + if (fltr)
> + refcount_inc(&fltr->refcnt);
> + rcu_read_unlock();
> + return fltr;
> +}
[ ... ]
> +int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
> + const u8 *mac_addr)
> +{
> + struct bnge_l2_filter *fltr;
> + struct bnge_l2_key key;
> + int rc;
> +
> + ether_addr_copy(key.dst_mac_addr, mac_addr);
> + key.vlan = 0;
> + fltr = bnge_alloc_l2_filter(bn, &key, GFP_KERNEL);
> + if (IS_ERR(fltr))
> + return PTR_ERR(fltr);
> +
> + fltr->base.fw_vnic_id = bn->vnic_info[vnic_id].fw_vnic_id;
> + rc = bnge_hwrm_l2_filter_alloc(bn->bd, fltr);
[Severity: Medium]
This is a pre-existing issue that this patch only moves. What happens when
a unicast address matches a filter already in bn->l2_fltr_hash_tbl?
For example, "bridge fdb add <own-mac> dev X self" ends up in
dev_uc_add_excl(). At this commit, bnge_cfg_rx_mode() copies every uc
entry and does not skip dev_addr.
In that case bnge_alloc_l2_filter() returns the existing object, with an
extra reference taken in bnge_lookup_l2_filter().
bnge_hwrm_set_vnic_filter() still calls bnge_hwrm_l2_filter_alloc() on
that object, and this overwrites its only firmware handle:
bnge_hwrm_l2_filter_alloc() {
...
fltr->base.filter_id = resp->l2_filter_id;
...
}
l2_filters[0] and l2_filters[i] then point to the same object. On
teardown, is the same filter_id freed twice in firmware while the first
firmware filter leaks?
This appears to be fixed later in the series by "bnge: remove refcount
from L2 filter". That commit makes bnge_alloc_l2_filter() return
ERR_PTR(-EEXIST) on a lookup hit. The final bnge_cfg_rx_mode() also
skips entries equal to dev->dev_addr.
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> new file mode 100644
> index 0000000000000..3a6c8f0a59541
> --- /dev/null
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> @@ -0,0 +1,100 @@
[ ... ]
> +void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd)
> +{
> + u16 max_entries, pad;
> + u32 *rss_indir_tbl;
> + u16 i;
> +
> + max_entries = bnge_get_rxfh_indir_size(bd);
> + rss_indir_tbl = &bd->rss_indir_tbl[0];
> +
> + for (i = 0; i < max_entries; i++)
> + rss_indir_tbl[i] = ethtool_rxfh_indir_default(i,
> + bd->rx_nr_rings);
[Severity: Medium]
This isn't a bug introduced by this patch, since the logic is unchanged.
Still, can max_entries be too small for some ring counts?
bnge_get_rxfh_indir_size() returns bnge_cal_nr_rss_ctxs(rx_nr_rings) * 64.
bnge_cal_nr_rss_ctxs() calls bnge_adjust_pow_two(rx_rings - 1, 64):
bnge_resc.h:bnge_adjust_pow_two() {
u32 blks = total_ent / ent_per_blk;
if (blks == 0 || blks == 1)
return ++blks;
if (!is_power_of_2(blks))
blks = roundup_pow_of_two(blks);
return blks;
}
The extra block is added only when blks is 0 or 1. With 129..192 rings,
blks is 2 and comes back unchanged, which gives a 128-entry table. The
same thing happens for 257..320 rings (256 entries) and 513..576 rings
(512 entries).
With 160 RX rings, would queues 128..159 never get default RSS traffic,
both here and in bnge_fill_hw_rss_tbl()? For comparison, the bnxt helper
bnxt_calc_nr_ring_pages() always adds one before rounding up.
bnge_net_init_dflt_rings() sets rx_nr_rings to the minimum of
netif_get_num_default_rss_queues() and the firmware limits, so it does
not have to be a power of two.
> +
> + pad = bd->rss_indir_tbl_entries - max_entries;
> + if (pad)
> + memset(&rss_indir_tbl[i], 0, pad * sizeof(*rss_indir_tbl));
> +}
[ ... ]
> +int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic)
> +{
> + struct bnge_dev *bd = bn->bd;
> + int rc, i, nr_ctxs;
> +
> + nr_ctxs = bnge_cal_nr_rss_ctxs(bd->rx_nr_rings);
> + for (i = 0; i < nr_ctxs; i++) {
> + rc = bnge_hwrm_vnic_ctx_alloc(bd, vnic, i);
[Severity: Medium]
This is a pre-existing issue, but is nr_ctxs checked against
BNGE_MAX_CTX_PER_VNIC anywhere?
bnge_hwrm_vnic_ctx_alloc() writes vnic->fw_rss_cos_lb_ctx[ctx_idx], and
that array has BNGE_MAX_CTX_PER_VNIC (8) entries. bd->rss_indir_tbl is
always allocated with BNGE_MAX_RSS_TABLE_ENTRIES (512) entries, and
vnic->rss_table is sized by BNGE_MAX_RSS_TABLE_SIZE.
With 577 or more RX rings, bnge_cal_nr_rss_ctxs() returns 16. Then:
- this loop writes past fw_rss_cos_lb_ctx[] for i >= 8
- bnge_set_dflt_rss_indir_tbl() writes 1024 entries into the 512-entry
rss_indir_tbl, and the u16 pad subtraction wraps
- bnge_fill_hw_rss_tbl() overruns vnic->rss_table
The only limit on rx_nr_rings seems to be in bnge_net_init_dflt_rings().
It comes from netif_get_num_default_rss_queues() and the firmware ring
limits, not from BNGE_MAX_RSS_TABLE_ENTRIES. Triggering this needs a very
large host and matching firmware limits.
Should rx_nr_rings be capped to what these arrays can hold?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 02/10] bnge: add NTUPLE/ARFS VNIC
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
2026-09-28 6:12 ` [net-next, v4 01/10] bnge: restructure VNIC and filter code Vikas Gupta
@ 2026-09-28 6:12 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 03/10] bnge: add helper functions for multi RSS contexts Vikas Gupta
` (7 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:12 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Initialize the second VNIC, which is specifically used for the
NTUPLE/ARFS feature. This VNIC helps support the aRFS/NTUPLE
functionality. With this change, the total number of VNICs becomes 2.
Add bnge_set_dflt_rfs() to configure the default NTUPLE feature state
at netdev allocation time.
Also, allocate and configure the ARFS VNIC.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
drivers/net/ethernet/broadcom/bnge/bnge.h | 10 ++-
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 76 +++++++++++++++++--
.../net/ethernet/broadcom/bnge/bnge_netdev.h | 1 +
.../net/ethernet/broadcom/bnge/bnge_resc.c | 60 ++++++++++++++-
.../net/ethernet/broadcom/bnge/bnge_resc.h | 1 +
.../net/ethernet/broadcom/bnge/bnge_vnic.c | 33 +++++++-
.../net/ethernet/broadcom/bnge/bnge_vnic.h | 7 +-
7 files changed, 176 insertions(+), 12 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge.h b/drivers/net/ethernet/broadcom/bnge/bnge.h
index bde54ba5d50f..a88aac140c7d 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge.h
@@ -63,9 +63,7 @@ enum {
BNGE_FW_CAP_THRESHOLD_TEMP_SUPPORTED = BIT_ULL(21),
BNGE_FW_CAP_DFLT_VLAN_TPID_PCP = BIT_ULL(22),
BNGE_FW_CAP_VNIC_TUNNEL_TPA = BIT_ULL(23),
- BNGE_FW_CAP_CFA_NTUPLE_RX_EXT_IP_PROTO = BIT_ULL(24),
- BNGE_FW_CAP_CFA_RFS_RING_TBL_IDX_V3 = BIT_ULL(25),
- BNGE_FW_CAP_VNIC_RE_FLUSH = BIT_ULL(26),
+ BNGE_FW_CAP_VNIC_RE_FLUSH = BIT_ULL(24),
};
enum {
@@ -74,6 +72,7 @@ enum {
BNGE_EN_STRIP_VLAN = BIT_ULL(2),
BNGE_EN_SHARED_CHNL = BIT_ULL(3),
BNGE_EN_UDP_GSO_SUPP = BIT_ULL(4),
+ BNGE_EN_ARFS_CAP = BIT_ULL(5),
};
#define BNGE_EN_ROCE (BNGE_EN_ROCE_V1 | BNGE_EN_ROCE_V2)
@@ -218,6 +217,11 @@ static inline bool bnge_is_roce_en(struct bnge_dev *bd)
return bd->flags & BNGE_EN_ROCE;
}
+static inline bool bnge_is_arfs_cap(struct bnge_dev *bd)
+{
+ return bd->flags & BNGE_EN_ARFS_CAP;
+}
+
static inline bool bnge_is_agg_reqd(struct bnge_dev *bd)
{
if (bd->netdev) {
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index 9e64b1933c02..c6b9048586b3 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -1149,12 +1149,10 @@ static int bnge_alloc_vnic_attributes(struct bnge_net *bn)
static int bnge_alloc_vnics(struct bnge_net *bn)
{
- int num_vnics;
+ int num_vnics = 1;
- /* Allocate only 1 VNIC for now
- * Additional VNICs will be added based on RFS/NTUPLE in future patches
- */
- num_vnics = 1;
+ if (bn->priv_flags & BNGE_NET_EN_NTUPLE)
+ num_vnics++;
bn->vnic_info = kzalloc_objs(struct bnge_vnic_info, num_vnics);
if (!bn->vnic_info)
@@ -1320,6 +1318,10 @@ static int bnge_alloc_core(struct bnge_net *bn)
bn->vnic_info[BNGE_VNIC_DEFAULT].flags |= BNGE_VNIC_RSS_FLAG |
BNGE_VNIC_MCAST_FLAG |
BNGE_VNIC_UCAST_FLAG;
+ if (bn->priv_flags & BNGE_NET_EN_NTUPLE)
+ bn->vnic_info[BNGE_VNIC_NTUPLE].flags |= BNGE_VNIC_RSS_FLAG |
+ BNGE_VNIC_NTUPLE_FLAG;
+
rc = bnge_alloc_vnic_attributes(bn);
if (rc)
goto err_free_core;
@@ -2546,6 +2548,18 @@ static int bnge_init_chip(struct bnge_net *bn)
if (rc)
goto err_out;
+ if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) && bnge_is_arfs_cap(bd)) {
+ rc = bnge_alloc_rfs_vnic(bn);
+ if (rc) {
+ netdev_warn(bn->netdev,
+ "Failed to allocate aRFS VNIC (%d), disabling ARFS\n",
+ rc);
+ bd->flags &= ~BNGE_EN_ARFS_CAP;
+ bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
+ bn->netdev->features &= ~NETIF_F_NTUPLE;
+ }
+ }
+
if (bd->rss_cap & BNGE_RSS_CAP_RSS_HASH_TYPE_DELTA)
bnge_hwrm_update_rss_hash_cfg(bn);
@@ -3019,6 +3033,8 @@ static int bnge_close(struct net_device *dev)
bnge_hwrm_if_change(bn->bd, false);
bn->sp_event = 0;
+ netdev_update_features(dev);
+
return 0;
}
@@ -3101,6 +3117,40 @@ static const struct netdev_stat_ops bnge_stat_ops = {
.get_base_stats = bnge_get_base_stats,
};
+static netdev_features_t bnge_fix_features(struct net_device *dev,
+ netdev_features_t features)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+
+ /* NTUPLE can only be changed while the interface is down. */
+ if (netif_running(dev)) {
+ if (dev->features & NETIF_F_NTUPLE)
+ features |= NETIF_F_NTUPLE;
+ else
+ features &= ~NETIF_F_NTUPLE;
+ } else if ((features & NETIF_F_NTUPLE) && !bnge_is_arfs_cap(bn->bd)) {
+ features &= ~NETIF_F_NTUPLE;
+ }
+ return features;
+}
+
+static int bnge_set_features(struct net_device *dev, netdev_features_t features)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ u32 flags = bn->priv_flags;
+
+ flags &= ~BNGE_NET_EN_NTUPLE;
+ if (features & NETIF_F_NTUPLE)
+ flags |= BNGE_NET_EN_NTUPLE;
+
+ if (flags == bn->priv_flags)
+ return 0;
+
+ bn->priv_flags = flags;
+
+ return 0;
+}
+
static const struct net_device_ops bnge_netdev_ops = {
.ndo_open = bnge_open,
.ndo_stop = bnge_close,
@@ -3108,6 +3158,8 @@ static const struct net_device_ops bnge_netdev_ops = {
.ndo_get_stats64 = bnge_get_stats64,
.ndo_set_rx_mode_async = bnge_set_rx_mode,
.ndo_features_check = bnge_features_check,
+ .ndo_fix_features = bnge_fix_features,
+ .ndo_set_features = bnge_set_features,
};
static void bnge_init_mac_addr(struct bnge_dev *bd)
@@ -3233,6 +3285,18 @@ static void bnge_init_ring_params(struct bnge_net *bn)
bn->netdev->cfg->hds_thresh = max(BNGE_DEFAULT_RX_COPYBREAK, rx_size);
}
+static void bnge_set_dflt_rfs(struct bnge_net *bn)
+{
+ bn->netdev->hw_features &= ~NETIF_F_NTUPLE;
+ bn->netdev->features &= ~NETIF_F_NTUPLE;
+ bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
+ if (bnge_is_arfs_cap(bn->bd)) {
+ bn->netdev->hw_features |= NETIF_F_NTUPLE;
+ bn->netdev->features |= NETIF_F_NTUPLE;
+ bn->priv_flags |= BNGE_NET_EN_NTUPLE;
+ }
+}
+
int bnge_netdev_alloc(struct bnge_dev *bd, int max_irqs)
{
struct net_device *netdev;
@@ -3339,6 +3403,8 @@ int bnge_netdev_alloc(struct bnge_dev *bd, int max_irqs)
bnge_set_ring_params(bd);
bnge_init_l2_fltr_tbl(bn);
+ bnge_set_dflt_rfs(bn);
+
bnge_init_mac_addr(bd);
rc = bnge_probe_phy(bn, true);
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index ee649cc644db..c47a874df4ba 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -152,6 +152,7 @@ enum {
BNGE_NET_EN_GRO = BIT(0),
BNGE_NET_EN_LRO = BIT(1),
BNGE_NET_EN_JUMBO = BIT(2),
+ BNGE_NET_EN_NTUPLE = BIT(3),
};
#define BNGE_NET_EN_TPA (BNGE_NET_EN_GRO | BNGE_NET_EN_LRO)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
index 69a894b52485..c7024e75f925 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
@@ -178,6 +178,16 @@ static int bnge_adjust_rings(struct bnge_dev *bd, u16 *rx,
return bnge_fix_rings_count(rx, tx, max_nq, sh);
}
+static unsigned int bnge_get_max_func_rss_ctxs(struct bnge_dev *bd)
+{
+ return bd->hw_resc.max_rsscos_ctxs;
+}
+
+static unsigned int bnge_get_max_func_vnics(struct bnge_dev *bd)
+{
+ return bd->hw_resc.max_vnics;
+}
+
int bnge_cal_nr_rss_ctxs(u16 rx_rings)
{
if (!rx_rings)
@@ -190,11 +200,18 @@ int bnge_cal_nr_rss_ctxs(u16 rx_rings)
static u16 bnge_get_total_rss_ctxs(struct bnge_dev *bd,
struct bnge_hw_rings *hwr)
{
- return bnge_cal_nr_rss_ctxs(hwr->grp);
+ u16 rss_ctx = bnge_cal_nr_rss_ctxs(hwr->grp);
+
+ rss_ctx *= hwr->vnic;
+
+ return rss_ctx;
}
static u16 bnge_get_total_vnics(struct bnge_dev *bd)
{
+ if (bnge_is_arfs_cap(bd))
+ return 2;
+
return 1;
}
@@ -563,6 +580,44 @@ static int bnge_alloc_rss_indir_tbl(struct bnge_dev *bd)
return 0;
}
+/* If runtime conditions support RFS */
+bool bnge_arfs_capable(struct bnge_dev *bd, bool new_rss_ctx)
+{
+ struct bnge_hw_rings hwr = {};
+ int max_vnics, max_rss_ctxs;
+
+ hwr.grp = bd->rx_nr_rings;
+ hwr.vnic = bnge_get_total_vnics(bd);
+
+ if (!bnge_is_arfs_cap(bd))
+ hwr.vnic++;
+
+ if (new_rss_ctx)
+ hwr.vnic++;
+ hwr.rss_ctx = bnge_get_total_rss_ctxs(bd, &hwr);
+ max_vnics = bnge_get_max_func_vnics(bd);
+ max_rss_ctxs = bnge_get_max_func_rss_ctxs(bd);
+
+ if (hwr.vnic > max_vnics || hwr.rss_ctx > max_rss_ctxs) {
+ if (bd->rx_nr_rings > 1)
+ dev_warn(bd->dev,
+ "Not enough resources to support NTUPLE filters\n");
+ return false;
+ }
+
+ if (hwr.vnic <= bd->hw_resc.resv_vnics &&
+ hwr.rss_ctx <= bd->hw_resc.resv_rsscos_ctxs)
+ return true;
+
+ bnge_hwrm_reserve_rings(bd, &hwr);
+ if (hwr.vnic <= bd->hw_resc.resv_vnics &&
+ hwr.rss_ctx <= bd->hw_resc.resv_rsscos_ctxs)
+ return true;
+
+ dev_warn(bd->dev, "Unable to reserve resources to support NTUPLE filters\n");
+ return false;
+}
+
int bnge_net_init_dflt_config(struct bnge_dev *bd)
{
struct bnge_hw_resc *hw_resc;
@@ -576,6 +631,9 @@ int bnge_net_init_dflt_config(struct bnge_dev *bd)
if (rc)
goto err_free_tbl;
+ if (bnge_arfs_capable(bd, false))
+ bd->flags |= BNGE_EN_ARFS_CAP;
+
hw_resc = &bd->hw_resc;
bd->max_fltr = hw_resc->max_rx_em_flows + hw_resc->max_rx_wm_flows +
BNGE_L2_FLTR_MAX_FLTR;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.h b/drivers/net/ethernet/broadcom/bnge/bnge_resc.h
index b62a634669f6..1e55fbe6985b 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.h
@@ -75,6 +75,7 @@ void bnge_aux_init_dflt_config(struct bnge_dev *bd);
u32 bnge_get_rxfh_indir_size(struct bnge_dev *bd);
int bnge_cal_nr_rss_ctxs(u16 rx_rings);
bool bnge_aux_has_enough_resources(struct bnge_dev *bd);
+bool bnge_arfs_capable(struct bnge_dev *bd, bool new_rss_ctx);
static inline u32
bnge_adjust_pow_two(u32 total_ent, u16 ent_per_blk)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
index 3a6c8f0a5954..d98a6196c859 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
@@ -42,7 +42,11 @@ void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic)
for (i = 0; i < tbl_size; i++) {
u16 ring_id, j;
- j = bd->rss_indir_tbl[i];
+ if (vnic->flags & BNGE_VNIC_NTUPLE_FLAG)
+ j = ethtool_rxfh_indir_default(i, bd->rx_nr_rings);
+ else
+ j = bd->rss_indir_tbl[i];
+
rxr = &bn->rx_ring[j];
ring_id = rxr->rx_ring_struct.fw_ring_id;
@@ -98,3 +102,30 @@ int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic)
}
return rc;
}
+
+static int bnge_alloc_and_setup_vnic(struct bnge_net *bn,
+ struct bnge_vnic_info *vnic,
+ u16 rx_rings)
+{
+ int rc;
+
+ rc = bnge_hwrm_vnic_alloc(bn->bd, vnic, rx_rings);
+ if (rc) {
+ netdev_err(bn->netdev, "hwrm vnic %u alloc failure rc: %d\n",
+ vnic->vnic_id, rc);
+ return rc;
+ }
+
+ /* If bnge_setup_vnic() fails, the VNIC allocated above is not freed
+ * here; the caller (bnge_init_chip) unwinds via its err_out path.
+ */
+ return bnge_setup_vnic(bn, vnic);
+}
+
+int bnge_alloc_rfs_vnic(struct bnge_net *bn)
+{
+ struct bnge_vnic_info *vnic;
+
+ vnic = &bn->vnic_info[BNGE_VNIC_NTUPLE];
+ return bnge_alloc_and_setup_vnic(bn, vnic, bn->bd->rx_nr_rings);
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
index c81d3a64561e..4fa1c7570dc4 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
@@ -22,13 +22,15 @@ struct bnge_l2_filter;
#define BNGE_MAX_UC_ADDRS 4
enum {
- BNGE_VNIC_DEFAULT = 0
+ BNGE_VNIC_DEFAULT = 0,
+ BNGE_VNIC_NTUPLE = 1
};
enum {
BNGE_VNIC_RSS_FLAG = BIT(0),
BNGE_VNIC_MCAST_FLAG = BIT(1),
- BNGE_VNIC_UCAST_FLAG = BIT(2)
+ BNGE_VNIC_UCAST_FLAG = BIT(2),
+ BNGE_VNIC_NTUPLE_FLAG = BIT(3)
};
struct bnge_vnic_info {
@@ -60,4 +62,5 @@ int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
struct bnge_vnic_info *vnic);
int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic);
void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd);
+int bnge_alloc_rfs_vnic(struct bnge_net *bn);
#endif /* _BNGE_VNIC_H_ */
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 02/10] bnge: add NTUPLE/ARFS VNIC
2026-09-28 6:12 ` [net-next, v4 02/10] bnge: add NTUPLE/ARFS VNIC Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 7 potential
issue(s) to consider.
Critical: 0 · High: 0 · Medium: 2 · Low: 5
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [Medium] bnge_alloc_and_setup_vnic() (bnge_vnic.c) allocates the NTUPLE
VNIC in firmware with bnge_hwrm_vnic_alloc() and then calls…
- [Medium] Any bnge_alloc_rfs_vnic() failure during open, including a
transient HWRM or firmware error, makes bnge_init_chip() clear…
- [Low] At this commit, bnge_set_dflt_rfs() adds NETIF_F_NTUPLE to
hw_features and features, and ndo_fix_features/ndo_set_features let…
- [Low] The reservation side and the allocation side use different
conditions.
- [Low] On aRFS VNIC allocation failure, bnge_init_chip() writes directly
to bn->netdev->features (&= ~NETIF_F_NTUPLE) from the ndo_open path.
- [Low] The patch adds the first runtime read-modify-write of bd->flags
on the ndo_open path: 'bd->flags &= ~BNGE_EN_ARFS_CAP' in…
- [Low] bnge_arfs_capable() (bnge_resc.c) sends HWRM_FUNC_CFG through
bnge_hwrm_reserve_rings() with only vnic and rss_ctx set, so…
--
bnge: add NTUPLE/ARFS VNIC
This commit adds a second VNIC for NTUPLE/aRFS. On aRFS-capable devices
it reserves the extra VNIC and RSS contexts and turns NETIF_F_NTUPLE on
by default. It also adds ndo_fix_features and ndo_set_features, so
NTUPLE can be changed only while the interface is down.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index 9e64b1933c023..c6b9048586b36 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
[ ... ]
> @@ -2546,6 +2548,18 @@ static int bnge_init_chip(struct bnge_net *bn)
> if (rc)
> goto err_out;
>
> + if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) && bnge_is_arfs_cap(bd)) {
> + rc = bnge_alloc_rfs_vnic(bn);
> + if (rc) {
> + netdev_warn(bn->netdev,
> + "Failed to allocate aRFS VNIC (%d), disabling ARFS\n",
> + rc);
> + bd->flags &= ~BNGE_EN_ARFS_CAP;
[Severity: Medium]
Does clearing BNGE_EN_ARFS_CAP here turn a transient firmware error into
a permanent loss of the feature?
The only code that sets this bit is in bnge_net_init_dflt_config(), and
that function is only reached from bnge_probe_one():
if (bnge_arfs_capable(bd, false))
bd->flags |= BNGE_EN_ARFS_CAP;
After one failed open, bnge_fix_features() drops NTUPLE from every later
request:
} else if ((features & NETIF_F_NTUPLE) && !bnge_is_arfs_cap(bn->bd)) {
features &= ~NETIF_F_NTUPLE;
hw_features still advertises NETIF_F_NTUPLE. So "ethtool -K <dev> ntuple
on" silently does nothing until the driver is reloaded.
This fallback also leaves bn->nr_vnics at 2, and vnic_info[1] may be only
half configured. Later in bnge_init_chip(), bnge_set_tpa() still walks
that VNIC. If HWRM_VNIC_TPA_CFG fails there, the code takes goto err_out.
Could that fail the same open this fallback is meant to save? That part
depends on how the firmware treats the half-configured VNIC.
[Severity: Low]
Can this read-modify-write of bd->flags race with
bnge_rdma_aux_device_add()?
bnge_probe_one() registers the netdev in bnge_netdev_alloc() before it
calls bnge_rdma_aux_device_add(). If auxiliary_device_add() fails, that
path does this with no lock held:
auxiliary_device_uninit(aux_dev);
bd->flags &= ~BNGE_EN_ROCE;
The update here runs under rtnl and the netdev instance lock, so the two
writers share no lock. Userspace can open the new netdev at the same
time. If both paths fail together, one of the two plain u64 updates can
be lost. For example, BNGE_EN_ARFS_CAP could reappear after open decided
to disable aRFS.
Before this patch, bd->flags was only written during probe or on that aux
path, so this race is new. The window is small and needs two failures at
the same time.
> + bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
> + bn->netdev->features &= ~NETIF_F_NTUPLE;
[Severity: Low]
Is it fine to write netdev->features directly from the ndo_open path?
Documentation/networking/netdev-features.rst says netdev->features
"should be changed only by network core or in error paths of
ndo_set_features callback".
The direct write skips the feature recalculation and the
NETDEV_FEAT_CHANGE notification. The netdev_update_features() call added
to bnge_close() then computes the same set, which is already cleared. So
no notification is ever sent for this on-to-off change.
Would calling netdev_update_features() here work instead? ndo_open
already holds rtnl and the instance lock.
> + }
> + }
> +
[ ... ]
> @@ -3233,6 +3285,18 @@ static void bnge_init_ring_params(struct bnge_net *bn)
> bn->netdev->cfg->hds_thresh = max(BNGE_DEFAULT_RX_COPYBREAK, rx_size);
> }
>
> +static void bnge_set_dflt_rfs(struct bnge_net *bn)
> +{
> + bn->netdev->hw_features &= ~NETIF_F_NTUPLE;
> + bn->netdev->features &= ~NETIF_F_NTUPLE;
> + bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
> + if (bnge_is_arfs_cap(bn->bd)) {
> + bn->netdev->hw_features |= NETIF_F_NTUPLE;
> + bn->netdev->features |= NETIF_F_NTUPLE;
> + bn->priv_flags |= BNGE_NET_EN_NTUPLE;
> + }
> +}
[Severity: Low]
With this patch, NETIF_F_NTUPLE is advertised and on by default, and it
can be toggled through ndo_fix_features and ndo_set_features. Is the code
behind the feature there yet?
At this commit:
- bnge_netdev_ops has no ndo_rx_flow_steer.
- rx_cpu_rmap is never allocated.
- bnge_ethtool_ops has no get_rxnfc or set_rxnfc.
So "ntuple-filters: on" is reported and "ethtool -N" returns -EOPNOTSUPP.
An extra firmware VNIC and extra RSS contexts are also reserved for a
feature that does nothing yet.
Later patches in the series fill this in:
- "bnge: add NTUPLE filter support in ethtool" adds get_rxnfc and
set_rxnfc.
- "bnge: add aRFS flow steering ndo support" adds ndo_rx_flow_steer.
- "bnge: add cpu_rmap support for IRQ affinity" allocates rx_cpu_rmap
after that.
Since rx_cpu_rmap only shows up after ndo_rx_flow_steer exists, nothing
can crash in the meantime. Would it make sense to advertise
NETIF_F_NTUPLE only once that code is in place?
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> index 69a894b52485e..c7024e75f9257 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
[ ... ]
> @@ -190,11 +200,18 @@ int bnge_cal_nr_rss_ctxs(u16 rx_rings)
> static u16 bnge_get_total_rss_ctxs(struct bnge_dev *bd,
> struct bnge_hw_rings *hwr)
> {
> - return bnge_cal_nr_rss_ctxs(hwr->grp);
> + u16 rss_ctx = bnge_cal_nr_rss_ctxs(hwr->grp);
> +
> + rss_ctx *= hwr->vnic;
> +
> + return rss_ctx;
> }
>
> static u16 bnge_get_total_vnics(struct bnge_dev *bd)
> {
> + if (bnge_is_arfs_cap(bd))
> + return 2;
> +
> return 1;
> }
[Severity: Low]
Should this check BNGE_NET_EN_NTUPLE rather than BNGE_EN_ARFS_CAP?
The reservation side keys on the capability. On capable devices,
bnge_reserve_rings() and bnge_need_reserve_rings() always ask for 2
VNICs, and bnge_get_total_rss_ctxs() doubles the RSS context count.
The allocation side keys on the flag the user controls:
bnge_alloc_vnics():
if (bn->priv_flags & BNGE_NET_EN_NTUPLE)
num_vnics++;
After "ethtool -K <dev> ntuple off" with the interface down, the next
open still reserves 2 VNICs and twice the RSS contexts in firmware, but
uses only one VNIC. For comparison, bnxt_get_total_vnics() keys on the
enabled flag BNXT_FLAG_RFS.
[ ... ]
> @@ -563,6 +580,44 @@ static int bnge_alloc_rss_indir_tbl(struct bnge_dev *bd)
> return 0;
> }
>
> +/* If runtime conditions support RFS */
> +bool bnge_arfs_capable(struct bnge_dev *bd, bool new_rss_ctx)
> +{
[ ... ]
> + if (hwr.vnic <= bd->hw_resc.resv_vnics &&
> + hwr.rss_ctx <= bd->hw_resc.resv_rsscos_ctxs)
> + return true;
> +
> + bnge_hwrm_reserve_rings(bd, &hwr);
> + if (hwr.vnic <= bd->hw_resc.resv_vnics &&
> + hwr.rss_ctx <= bd->hw_resc.resv_rsscos_ctxs)
> + return true;
> +
> + dev_warn(bd->dev, "Unable to reserve resources to support NTUPLE filters\n");
> + return false;
> +}
[Severity: Low]
If firmware grants only part of this request, does anything release the
extra resources?
bnge_hwrm_reserve_rings() is called with only vnic and rss_ctx set, and
its return value is ignored. Suppose firmware grants fewer than 2 VNICs
but all the RSS contexts. This function then warns and returns false,
and the partial grant stays in place.
bnxt_rfs_capable() re-reserves with vnic = 1 on this path. Here,
bnge_need_reserve_rings() checks resv_vnics but never resv_rsscos_ctxs:
if (hw_resc->resv_rx_rings != rx || hw_resc->resv_cp_rings != cprs ||
hw_resc->resv_vnics != vnic || hw_resc->resv_stat_ctxs != stat)
So the extra RSS contexts seem to stay reserved to this function
indefinitely.
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> index 3a6c8f0a59541..d98a6196c859c 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
[ ... ]
> +static int bnge_alloc_and_setup_vnic(struct bnge_net *bn,
> + struct bnge_vnic_info *vnic,
> + u16 rx_rings)
> +{
> + int rc;
> +
> + rc = bnge_hwrm_vnic_alloc(bn->bd, vnic, rx_rings);
> + if (rc) {
> + netdev_err(bn->netdev, "hwrm vnic %u alloc failure rc: %d\n",
> + vnic->vnic_id, rc);
> + return rc;
> + }
> +
> + /* If bnge_setup_vnic() fails, the VNIC allocated above is not freed
> + * here; the caller (bnge_init_chip) unwinds via its err_out path.
> + */
> + return bnge_setup_vnic(bn, vnic);
> +}
[Severity: Medium]
Is this comment accurate? In the bnge_init_chip() hunk above, a failure
from bnge_alloc_rfs_vnic() only logs a warning and clears
BNGE_EN_ARFS_CAP, BNGE_NET_EN_NTUPLE and NETIF_F_NTUPLE. Then it carries
on without jumping to err_out.
Suppose bnge_hwrm_vnic_alloc() succeeds and bnge_setup_vnic() fails
partway. The firmware VNIC and any RSS contexts already allocated by
bnge_hwrm_vnic_ctx_alloc() then stay in firmware while the interface is
up.
bn->nr_vnics stays 2, so bnge_set_tpa() still sends HWRM_VNIC_TPA_CFG to
the half-configured vnic_info[1]:
for (i = 0; i < bn->nr_vnics; i++) {
rc = bnge_hwrm_vnic_set_tpa(bn->bd, &bn->vnic_info[i], tpa_flags);
bnge_hwrm_vnic_set_tpa() skips a VNIC only when its fw_vnic_id is
INVALID_HW_RING_ID, and this VNIC has a valid ID.
Meanwhile, bnge_get_total_vnics() now returns 1, while firmware holds 2
VNICs and their contexts. bnge_clear_vnic() frees them at close, so they
are not held forever.
Should this function free the VNIC when setup fails, or should
bnge_init_chip() unwind it?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 03/10] bnge: add helper functions for multi RSS contexts
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
2026-09-28 6:12 ` [net-next, v4 01/10] bnge: restructure VNIC and filter code Vikas Gupta
2026-09-28 6:12 ` [net-next, v4 02/10] bnge: add NTUPLE/ARFS VNIC Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 04/10] bnge: add RXFH ethtool support Vikas Gupta
` (6 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Add infrastructure for creating, modifying, and deleting user-defined
RSS contexts. This helps the next patch implement commands for ethtool.
For each user RSS context, a VNIC is allocated.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_netdev.h | 2 +
.../net/ethernet/broadcom/bnge/bnge_resc.c | 12 +-
.../net/ethernet/broadcom/bnge/bnge_vnic.c | 113 ++++++++++++++++++
.../net/ethernet/broadcom/bnge/bnge_vnic.h | 15 +++
4 files changed, 140 insertions(+), 2 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index c47a874df4ba..3c0fea0a7608 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -371,6 +371,8 @@ struct bnge_net {
u8 pri2cos_idx[8];
bool pri2cos_valid;
+
+ u16 num_rss_ctx;
};
#define BNGE_DEFAULT_RX_RING_SIZE 511
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
index c7024e75f925..e0744e59c6ae 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
@@ -209,8 +209,16 @@ static u16 bnge_get_total_rss_ctxs(struct bnge_dev *bd,
static u16 bnge_get_total_vnics(struct bnge_dev *bd)
{
- if (bnge_is_arfs_cap(bd))
- return 2;
+ if (bnge_is_arfs_cap(bd)) {
+ u16 vnics = 2;
+
+ if (bd->netdev) {
+ struct bnge_net *bn = netdev_priv(bd->netdev);
+
+ vnics += bn->num_rss_ctx;
+ }
+ return vnics;
+ }
return 1;
}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
index d98a6196c859..a44971a7c74d 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
@@ -129,3 +129,116 @@ int bnge_alloc_rfs_vnic(struct bnge_net *bn)
vnic = &bn->vnic_info[BNGE_VNIC_NTUPLE];
return bnge_alloc_and_setup_vnic(bn, vnic, bn->bd->rx_nr_rings);
}
+
+void bnge_modify_rss(struct bnge_net *bn, struct ethtool_rxfh_context *ctx,
+ struct bnge_rss_ctx *rss_ctx,
+ const struct ethtool_rxfh_param *rxfh)
+{
+ struct bnge_dev *bd = bn->bd;
+
+ if (rxfh->key) {
+ if (rss_ctx) {
+ memcpy(rss_ctx->vnic.rss_hash_key, rxfh->key,
+ HW_HASH_KEY_SIZE);
+ } else {
+ memcpy(bn->rss_hash_key, rxfh->key, HW_HASH_KEY_SIZE);
+ bn->rss_hash_key_updated = true;
+ }
+ }
+
+ if (rxfh->indir) {
+ u32 i, pad, tbl_size = bnge_get_rxfh_indir_size(bd);
+ u32 *indir_tbl = bd->rss_indir_tbl;
+
+ if (rss_ctx)
+ indir_tbl = ethtool_rxfh_context_indir(ctx);
+ for (i = 0; i < tbl_size; i++)
+ indir_tbl[i] = rxfh->indir[i];
+ pad = bd->rss_indir_tbl_entries - tbl_size;
+ if (pad)
+ memset(&indir_tbl[i], 0, pad * sizeof(*indir_tbl));
+ }
+}
+
+void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
+ bool all)
+{
+ struct bnge_vnic_info *vnic = &rss_ctx->vnic;
+ int i;
+
+ bnge_hwrm_vnic_free_one(bn->bd, &rss_ctx->vnic);
+ for (i = 0; i < BNGE_MAX_CTX_PER_VNIC; i++) {
+ if (vnic->fw_rss_cos_lb_ctx[i] != INVALID_HW_RING_ID)
+ bnge_hwrm_vnic_ctx_free_one(bn->bd, vnic, i);
+ }
+
+ if (!all)
+ return;
+
+ if (vnic->rss_table)
+ dma_free_coherent(bn->bd->dev, vnic->rss_table_size,
+ vnic->rss_table,
+ vnic->rss_table_dma_addr);
+ bn->num_rss_ctx--;
+}
+
+void bnge_hwrm_realloc_rss_ctx_vnic(struct bnge_net *bn)
+{
+ u32 tpa_flags = bn->priv_flags & BNGE_NET_EN_TPA;
+ struct ethtool_rxfh_context *ctx;
+ unsigned long context;
+
+ xa_for_each(&bn->netdev->ethtool->rss_ctx, context, ctx) {
+ struct bnge_rss_ctx *rss_ctx = ethtool_rxfh_context_priv(ctx);
+ struct bnge_vnic_info *vnic = &rss_ctx->vnic;
+
+ if (bnge_hwrm_vnic_alloc(bn->bd, vnic, bn->bd->rx_nr_rings) ||
+ bnge_hwrm_vnic_set_tpa(bn->bd, vnic, tpa_flags) ||
+ bnge_setup_vnic(bn, vnic)) {
+ netdev_err(bn->netdev, "Failed to restore RSS ctx %d\n",
+ rss_ctx->index);
+ bnge_del_one_rss_ctx(bn, rss_ctx, true);
+ ethtool_rxfh_context_lost(bn->netdev, rss_ctx->index);
+ }
+ }
+}
+
+void bnge_clear_rss_ctxs(struct bnge_net *bn)
+{
+ struct ethtool_rxfh_context *ctx;
+ unsigned long context;
+
+ xa_for_each(&bn->netdev->ethtool->rss_ctx, context, ctx) {
+ struct bnge_rss_ctx *rss_ctx = ethtool_rxfh_context_priv(ctx);
+
+ bnge_del_one_rss_ctx(bn, rss_ctx, false);
+ }
+}
+
+struct bnge_rss_ctx *bnge_get_rss_ctx_from_index(struct bnge_net *bn, u32 idx)
+{
+ struct ethtool_rxfh_context *ctx;
+
+ ctx = xa_load(&bn->netdev->ethtool->rss_ctx, idx);
+ if (!ctx)
+ return NULL;
+ return ethtool_rxfh_context_priv(ctx);
+}
+
+int bnge_alloc_vnic_rss_table(struct bnge_net *bn,
+ struct bnge_vnic_info *vnic)
+{
+ unsigned int size = L1_CACHE_ALIGN(BNGE_MAX_RSS_TABLE_SIZE);
+
+ vnic->rss_table_size = size + HW_HASH_KEY_SIZE;
+ vnic->rss_table = dma_alloc_coherent(bn->bd->dev,
+ vnic->rss_table_size,
+ &vnic->rss_table_dma_addr,
+ GFP_KERNEL);
+ if (!vnic->rss_table)
+ return -ENOMEM;
+
+ vnic->rss_hash_key = ((void *)vnic->rss_table) + size;
+ vnic->rss_hash_key_dma_addr = vnic->rss_table_dma_addr + size;
+ return 0;
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
index 4fa1c7570dc4..ff7244cdf1a8 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
@@ -57,10 +57,25 @@ struct bnge_vnic_info {
u32 vnic_id;
};
+struct bnge_rss_ctx {
+ struct bnge_vnic_info vnic;
+ u32 index;
+};
+
void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic);
int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
struct bnge_vnic_info *vnic);
int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic);
void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd);
int bnge_alloc_rfs_vnic(struct bnge_net *bn);
+int bnge_alloc_vnic_rss_table(struct bnge_net *bn,
+ struct bnge_vnic_info *vnic);
+struct bnge_rss_ctx *bnge_get_rss_ctx_from_index(struct bnge_net *bn, u32 idx);
+void bnge_modify_rss(struct bnge_net *bn, struct ethtool_rxfh_context *ctx,
+ struct bnge_rss_ctx *rss_ctx,
+ const struct ethtool_rxfh_param *rxfh);
+void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
+ bool all);
+void bnge_hwrm_realloc_rss_ctx_vnic(struct bnge_net *bn);
+void bnge_clear_rss_ctxs(struct bnge_net *bn);
#endif /* _BNGE_VNIC_H_ */
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 03/10] bnge: add helper functions for multi RSS contexts
2026-09-28 6:13 ` [net-next, v4 03/10] bnge: add helper functions for multi RSS contexts Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 3 potential
issue(s) to consider.
Critical: 0 · High: 0 · Medium: 3 · Low: 0
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [Medium] bnge_get_total_vnics()
(drivers/net/ethernet/broadcom/bnge/bnge_resc.c) adds bn->num_rss_ctx…
- [Medium] At this commit, bnge_modify_rss() writes a user context's
indirection table into ethtool_rxfh_context_indir(ctx), but…
- [Medium] For a user context, bnge_modify_rss() copies tbl_size entries
and then zero-pads up to bd->rss_indir_tbl_entries (512) into…
--
bnge: add helper functions for multi RSS contexts
This commit adds helpers to create, modify, restore and delete
user-defined RSS contexts in the bnge driver, with one VNIC per context.
It also adds a num_rss_ctx counter to struct bnge_net and includes it in
the VNIC count that bnge_get_total_vnics() returns.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> index c7024e75f9257..e0744e59c6aea 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> @@ -209,8 +209,16 @@ static u16 bnge_get_total_rss_ctxs(struct bnge_dev *bd,
>
> static u16 bnge_get_total_vnics(struct bnge_dev *bd)
> {
> - if (bnge_is_arfs_cap(bd))
> - return 2;
> + if (bnge_is_arfs_cap(bd)) {
> + u16 vnics = 2;
> +
> + if (bd->netdev) {
> + struct bnge_net *bn = netdev_priv(bd->netdev);
> +
> + vnics += bn->num_rss_ctx;
> + }
> + return vnics;
> + }
>
> return 1;
> }
[Severity: Medium]
bn->num_rss_ctx is only counted while bnge_is_arfs_cap(bd) is true. Can
that flag be cleared while user RSS contexts still exist?
bnge_init_chip() clears it when the aRFS VNIC allocation fails, and it
does not check netdev->ethtool->rss_ctx first:
drivers/net/ethernet/broadcom/bnge/bnge_netdev.c:bnge_init_chip() {
...
if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) && bnge_is_arfs_cap(bd)) {
rc = bnge_alloc_rfs_vnic(bn);
if (rc) {
...
bd->flags &= ~BNGE_EN_ARFS_CAP;
bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
bn->netdev->features &= ~NETIF_F_NTUPLE;
}
}
...
}
After this, bnge_get_total_vnics() returns 1. bnge_need_reserve_rings()
and bnge_reserve_rings() then ask firmware for one VNIC and one VNIC's
worth of RSS contexts. Each remaining user context still needs its own
VNIC.
On the next bnge_open_core(), bnge_reserve_rings() runs before
bnge_hwrm_realloc_rss_ctx_vnic(). Could the context VNIC allocations
fail at that point? Each context would then be torn down through
bnge_del_one_rss_ctx() and ethtool_rxfh_context_lost(). I haven't
confirmed that firmware enforces the lower reservation on later VNIC
allocs, so this step is partly speculative.
Also, bnge_init_chip() clears BNGE_NET_EN_NTUPLE directly. That bypasses
the -EBUSY check that later patches in the series add to
bnge_set_features(). That check stops NTUPLE from being disabled while
bn->num_rss_ctx is non-zero.
The capability is only set again in bnge_net_init_dflt_config(). While
it is clear, bnge_fix_features() strips NETIF_F_NTUPLE. So the remaining
contexts can't be modified, because bnge_rxfh_context_check() requires
BNGE_NET_EN_NTUPLE. Their VNICs also drop out of resource accounting.
This is still the case at the end of the series.
Should num_rss_ctx be counted whether or not aRFS is available? Or
should the aRFS fallback in bnge_init_chip() take existing contexts into
account?
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> index d98a6196c859c..a44971a7c74d3 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
> @@ -129,3 +129,116 @@ int bnge_alloc_rfs_vnic(struct bnge_net *bn)
> vnic = &bn->vnic_info[BNGE_VNIC_NTUPLE];
> return bnge_alloc_and_setup_vnic(bn, vnic, bn->bd->rx_nr_rings);
> }
> +
> +void bnge_modify_rss(struct bnge_net *bn, struct ethtool_rxfh_context *ctx,
> + struct bnge_rss_ctx *rss_ctx,
> + const struct ethtool_rxfh_param *rxfh)
> +{
[ ... ]
> + if (rxfh->indir) {
> + u32 i, pad, tbl_size = bnge_get_rxfh_indir_size(bd);
> + u32 *indir_tbl = bd->rss_indir_tbl;
> +
> + if (rss_ctx)
> + indir_tbl = ethtool_rxfh_context_indir(ctx);
[Severity: Medium]
At this commit, is anything reading the per-context table written here?
For non-NTUPLE VNICs, bnge_fill_hw_rss_tbl() only reads the default
table:
j = bd->rss_indir_tbl[i];
struct bnge_vnic_info also has no link back to the ctx. Used as they are
in this patch, wouldn't bnge_hwrm_realloc_rss_ctx_vnic()->bnge_setup_vnic()
program the default table into a context VNIC?
The helpers have no callers yet. The next patch, "bnge: add ethtool
support to manage RSS contexts", adds BNGE_VNIC_RSSCTX_FLAG and a
vnic->rss_ctx back-pointer. It also changes bnge_fill_hw_rss_tbl() to
read ethtool_rxfh_context_indir(vnic->rss_ctx). Could the reader side
move into this patch so these helpers are consistent on their own?
> + for (i = 0; i < tbl_size; i++)
> + indir_tbl[i] = rxfh->indir[i];
> + pad = bd->rss_indir_tbl_entries - tbl_size;
> + if (pad)
> + memset(&indir_tbl[i], 0, pad * sizeof(*indir_tbl));
[Severity: Medium]
Can this memset() overflow the ctx indirection table when rss_ctx is
non-NULL?
The pad runs up to bd->rss_indir_tbl_entries, which
bnge_alloc_rss_indir_tbl() sets to BNGE_MAX_RSS_TABLE_ENTRIES. That is
the device-wide capacity, not the size of the ctx allocation.
ethtool_rxfh_ctx_alloc() sizes the ctx indirection area as
max(indir_size, ops->rxfh_indir_space). bnge doesn't set rxfh_indir_space
at this commit. The pad would then write past the indirection table,
into the key area and beyond the end of the kzalloc'd ctx.
This can't be reached yet because bnge_modify_rss() has no caller. Later
in the series the pad is guarded with "if (pad && !rss_ctx)", and
bnge_ethtool_ops sets .rxfh_indir_space = BNGE_MAX_RSS_TABLE_ENTRIES.
Could those changes be folded into this patch?
> + }
> +}
[ ... ]
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 04/10] bnge: add RXFH ethtool support
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (2 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 03/10] bnge: add helper functions for multi RSS contexts Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 05/10] bnge: add ethtool support to manage RSS contexts Vikas Gupta
` (5 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Introduce ethtool RXFH operations including get/set_rxfh for
reading and programming the RSS hash key and indirection table.
Currently set_rxfh and set_rxfh_fields are supported only while the
device is down. A future patch series will rework
bnge_close_core()/bnge_open_core() so they can be driven safely
from the ethtool path, adding full functionality by applying the
configuration on a live device.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Bhargava Chenna Marreddy <bhargava.marreddy@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
drivers/net/ethernet/broadcom/bnge/bnge.h | 1 +
.../net/ethernet/broadcom/bnge/bnge_ethtool.c | 281 ++++++++++++++++++
.../ethernet/broadcom/bnge/bnge_hwrm_lib.c | 2 +
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 10 +-
.../net/ethernet/broadcom/bnge/bnge_vnic.h | 1 +
5 files changed, 288 insertions(+), 7 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge.h b/drivers/net/ethernet/broadcom/bnge/bnge.h
index a88aac140c7d..560b9d193dea 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge.h
@@ -86,6 +86,7 @@ enum {
BNGE_RSS_CAP_AH_V6_RSS_CAP = BIT(5),
BNGE_RSS_CAP_ESP_V4_RSS_CAP = BIT(6),
BNGE_RSS_CAP_ESP_V6_RSS_CAP = BIT(7),
+ BNGE_RSS_CAP_IPV6_FLOW_LABEL_RSS_CAP = BIT(8),
};
#define BNGE_MAX_QUEUE 8
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
index 2467e44de291..6fbda4fc1a0c 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
@@ -10,6 +10,9 @@
#include <linux/ethtool_netlink.h>
#include "bnge.h"
+#include "bnge_netdev.h"
+#include "bnge_vnic.h"
+#include "bnge_resc.h"
#include "bnge_ethtool.h"
#include "bnge_hwrm_lib.h"
@@ -740,6 +743,272 @@ static int bnge_set_pauseparam(struct net_device *dev,
return rc;
}
+static u64 bnge_get_ethtool_ipv4_rss(struct bnge_dev *bd)
+{
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4)
+ return RXH_IP_SRC | RXH_IP_DST;
+ return 0;
+}
+
+static u64 bnge_get_ethtool_ipv6_rss(struct bnge_dev *bd)
+{
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6)
+ return RXH_IP_SRC | RXH_IP_DST;
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL)
+ return RXH_IP_SRC | RXH_IP_DST | RXH_IP6_FL;
+ return 0;
+}
+
+static int bnge_get_rxfh_fields(struct net_device *dev,
+ struct ethtool_rxfh_fields *cmd)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+
+ cmd->data = 0;
+ switch (cmd->flow_type) {
+ case TCP_V4_FLOW:
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV4)
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
+ break;
+ case UDP_V4_FLOW:
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV4)
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
+ break;
+ case AH_ESP_V4_FLOW:
+ if (bd->rss_hash_cfg &
+ (VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV4))
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
+ break;
+ case SCTP_V4_FLOW:
+ case AH_V4_FLOW:
+ case ESP_V4_FLOW:
+ case IPV4_FLOW:
+ cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
+ break;
+ case TCP_V6_FLOW:
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV6)
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv6_rss(bd);
+ break;
+ case UDP_V6_FLOW:
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV6)
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv6_rss(bd);
+ break;
+ case AH_ESP_V6_FLOW:
+ if (bd->rss_hash_cfg &
+ (VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV6 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV6))
+ cmd->data |= RXH_IP_SRC | RXH_IP_DST |
+ RXH_L4_B_0_1 | RXH_L4_B_2_3;
+ cmd->data |= bnge_get_ethtool_ipv6_rss(bd);
+ break;
+ case SCTP_V6_FLOW:
+ case AH_V6_FLOW:
+ case ESP_V6_FLOW:
+ case IPV6_FLOW:
+ cmd->data |= bnge_get_ethtool_ipv6_rss(bd);
+ break;
+ default:
+ return -EOPNOTSUPP;
+ }
+
+ return 0;
+}
+
+#define RXH_4TUPLE (RXH_IP_SRC | RXH_IP_DST | RXH_L4_B_0_1 | RXH_L4_B_2_3)
+#define RXH_2TUPLE (RXH_IP_SRC | RXH_IP_DST)
+
+static int bnge_set_rxfh_fields(struct net_device *dev,
+ const struct ethtool_rxfh_fields *cmd,
+ struct netlink_ext_ack *extack)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+ u32 rss_hash_cfg;
+ int tuple;
+
+ rss_hash_cfg = bd->rss_hash_cfg;
+
+ if (cmd->data == RXH_4TUPLE ||
+ cmd->data == (RXH_4TUPLE | RXH_IP6_FL))
+ tuple = 4;
+ else if (cmd->data == RXH_2TUPLE ||
+ cmd->data == (RXH_2TUPLE | RXH_IP6_FL))
+ tuple = 2;
+ else if (!cmd->data)
+ tuple = 0;
+ else
+ return -EINVAL;
+
+ if (cmd->data & RXH_IP6_FL &&
+ !(bd->rss_cap & BNGE_RSS_CAP_IPV6_FLOW_LABEL_RSS_CAP))
+ return -EINVAL;
+
+ if (cmd->flow_type == TCP_V4_FLOW) {
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV4;
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV4;
+ } else if (cmd->flow_type == UDP_V4_FLOW) {
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV4;
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV4;
+ } else if (cmd->flow_type == TCP_V6_FLOW) {
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV6;
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV6;
+ } else if (cmd->flow_type == UDP_V6_FLOW) {
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV6;
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV6;
+ } else if (cmd->flow_type == AH_ESP_V4_FLOW) {
+ if (tuple == 4 &&
+ (!(bd->rss_cap & BNGE_RSS_CAP_AH_V4_RSS_CAP) ||
+ !(bd->rss_cap & BNGE_RSS_CAP_ESP_V4_RSS_CAP)))
+ return -EINVAL;
+ rss_hash_cfg &= ~(VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV4);
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV4;
+ } else if (cmd->flow_type == AH_ESP_V6_FLOW) {
+ if (tuple == 4 &&
+ (!(bd->rss_cap & BNGE_RSS_CAP_AH_V6_RSS_CAP) ||
+ !(bd->rss_cap & BNGE_RSS_CAP_ESP_V6_RSS_CAP)))
+ return -EINVAL;
+ rss_hash_cfg &= ~(VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV6 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV6);
+ if (tuple == 4)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV6 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV6;
+ } else if (tuple == 4) {
+ return -EINVAL;
+ }
+
+ switch (cmd->flow_type) {
+ case TCP_V4_FLOW:
+ case UDP_V4_FLOW:
+ case SCTP_V4_FLOW:
+ case AH_ESP_V4_FLOW:
+ case AH_V4_FLOW:
+ case ESP_V4_FLOW:
+ case IPV4_FLOW:
+ if (tuple == 2)
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4;
+ else if (!tuple && cmd->flow_type == IPV4_FLOW)
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4;
+ break;
+
+ case TCP_V6_FLOW:
+ case UDP_V6_FLOW:
+ case SCTP_V6_FLOW:
+ case AH_ESP_V6_FLOW:
+ case AH_V6_FLOW:
+ case ESP_V6_FLOW:
+ case IPV6_FLOW:
+ if (cmd->data & RXH_IP6_FL) {
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL;
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6;
+ } else if (tuple == 2 || tuple == 4) {
+ rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6;
+ rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL;
+ } else if (!tuple && cmd->flow_type == IPV6_FLOW) {
+ rss_hash_cfg &= ~(VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6 |
+ VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL);
+ }
+ break;
+ default:
+ return -EOPNOTSUPP;
+ }
+
+ if (bd->rss_hash_cfg == rss_hash_cfg)
+ return 0;
+
+ if (netif_running(dev)) {
+ NL_SET_ERR_MSG_MOD(extack,
+ "RSS configuration can only be changed while the interface is down");
+ return -EBUSY;
+ }
+
+ bd->rss_hash_cfg = rss_hash_cfg;
+
+ return 0;
+}
+
+static u32 bnge_get_rxfh_indir_size_eth(struct net_device *dev)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+
+ return bnge_get_rxfh_indir_size(bd);
+}
+
+static u32 bnge_get_rxfh_key_size(struct net_device *dev)
+{
+ return HW_HASH_KEY_SIZE;
+}
+
+static int bnge_get_rxfh(struct net_device *dev,
+ struct ethtool_rxfh_param *rxfh)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+ u32 i, tbl_size;
+ u32 *indir_tbl;
+
+ indir_tbl = bd->rss_indir_tbl;
+ rxfh->hfunc = ETH_RSS_HASH_TOP;
+
+ if (rxfh->indir && indir_tbl) {
+ tbl_size = bnge_get_rxfh_indir_size(bd);
+ for (i = 0; i < tbl_size; i++)
+ rxfh->indir[i] = indir_tbl[i];
+ }
+
+ if (rxfh->key)
+ memcpy(rxfh->key, bn->rss_hash_key, HW_HASH_KEY_SIZE);
+
+ return 0;
+}
+
+static int bnge_set_rxfh(struct net_device *dev,
+ struct ethtool_rxfh_param *rxfh,
+ struct netlink_ext_ack *extack)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+
+ if (rxfh->hfunc && rxfh->hfunc != ETH_RSS_HASH_TOP)
+ return -EOPNOTSUPP;
+
+ if (netif_running(dev)) {
+ NL_SET_ERR_MSG_MOD(extack,
+ "RSS configuration can only be changed while the interface is down");
+ return -EBUSY;
+ }
+
+ bnge_modify_rss(bn, NULL, NULL, rxfh);
+
+ return 0;
+}
+
+static u32 bnge_get_rx_ring_count(struct net_device *dev)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+
+ return bd->rx_nr_rings;
+}
+
static const struct ethtool_ops bnge_ethtool_ops = {
.cap_link_lanes_supported = 1,
.get_link_ksettings = bnge_get_link_ksettings,
@@ -757,6 +1026,18 @@ static const struct ethtool_ops bnge_ethtool_ops = {
.get_eth_ctrl_stats = bnge_get_eth_ctrl_stats,
.get_pause_stats = bnge_get_pause_stats,
.get_rmon_stats = bnge_get_rmon_stats,
+ /* RXFH */
+ .rxfh_per_ctx_key = 1,
+ .rxfh_max_num_contexts = BNGE_MAX_ETH_RSS_CTX + 1,
+ .rxfh_indir_space = BNGE_MAX_RSS_TABLE_ENTRIES,
+ .rxfh_priv_size = sizeof(struct bnge_rss_ctx),
+ .get_rx_ring_count = bnge_get_rx_ring_count,
+ .get_rxfh_indir_size = bnge_get_rxfh_indir_size_eth,
+ .get_rxfh_key_size = bnge_get_rxfh_key_size,
+ .get_rxfh = bnge_get_rxfh,
+ .set_rxfh = bnge_set_rxfh,
+ .get_rxfh_fields = bnge_get_rxfh_fields,
+ .set_rxfh_fields = bnge_set_rxfh_fields,
};
void bnge_set_ethtool_ops(struct net_device *dev)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
index 91d246c8dcf2..86edb70d323e 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
@@ -676,6 +676,8 @@ int bnge_hwrm_vnic_qcaps(struct bnge_dev *bd)
bd->rss_cap |= BNGE_RSS_CAP_ESP_V4_RSS_CAP;
if (flags & VNIC_QCAPS_RESP_FLAGS_RSS_IPSEC_ESP_SPI_IPV6_CAP)
bd->rss_cap |= BNGE_RSS_CAP_ESP_V6_RSS_CAP;
+ if (flags & VNIC_QCAPS_RESP_FLAGS_RSS_IPV6_FLOW_LABEL_CAP)
+ bd->rss_cap |= BNGE_RSS_CAP_IPV6_FLOW_LABEL_RSS_CAP;
}
bnge_hwrm_req_drop(bd, req);
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index c6b9048586b3..ef5d13337e60 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -1739,13 +1739,6 @@ static void bnge_init_vnics(struct bnge_net *bn)
u8 *key = (void *)vnic->rss_hash_key;
int k;
- if (!bn->rss_hash_key_valid &&
- !bn->rss_hash_key_updated) {
- get_random_bytes(bn->rss_hash_key,
- HW_HASH_KEY_SIZE);
- bn->rss_hash_key_updated = true;
- }
-
memcpy(vnic->rss_hash_key, bn->rss_hash_key,
HW_HASH_KEY_SIZE);
@@ -3398,6 +3391,9 @@ int bnge_netdev_alloc(struct bnge_dev *bd, int max_irqs)
bn->tx_ring_size = BNGE_DEFAULT_TX_RING_SIZE;
bn->rx_dir = DMA_FROM_DEVICE;
+ netdev_rss_key_fill(bn->rss_hash_key, sizeof(bn->rss_hash_key));
+ bn->rss_hash_key_updated = true;
+
bnge_set_tpa_flags(bd);
bnge_init_ring_params(bn);
bnge_set_ring_params(bd);
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
index ff7244cdf1a8..bc2ed911ced3 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
@@ -16,6 +16,7 @@ struct bnge_l2_filter;
#define BNGE_MAX_RSS_TABLE_ENTRIES \
(BNGE_RSS_TABLE_ENTRIES * BNGE_RSS_TABLE_MAX_TBL)
+#define BNGE_MAX_ETH_RSS_CTX 32
#define BNGE_MAX_CTX_PER_VNIC 8
#define BNGE_MAX_MC_ADDRS 16
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 04/10] bnge: add RXFH ethtool support
2026-09-28 6:13 ` [net-next, v4 04/10] bnge: add RXFH ethtool support Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 7 potential
issue(s) to consider.
Critical: 0 · High: 1 · Medium: 3 · Low: 3
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [High] buffer overflow and memset integer underflow when rx_nr_rings >
512
- [Medium] bnge_set_rxfh_fields() reports success for data == 0 (disable
hashing) on every flow type except IPV4_FLOW/IPV6_FLOW, but does not…
- [Medium] In bnge_set_rxfh_fields(), the IPv6 branch of the switch
handles tuple == 4 differently from the IPv4 branch.
- [Medium] The individual AH_V4_FLOW/ESP_V4_FLOW (and
AH_V6_FLOW/ESP_V6_FLOW) flow types ignore the SPI-hash bits that…
- [Low] In this commit, bnge_set_rxfh() (bnge_ethtool.c) lets userspace
store a default-context indirection table in bd->rss_indir_tbl while…
- [Low] bnge_ethtool_ops sets .rxfh_per_ctx_key = 1,
.rxfh_max_num_contexts = BNGE_MAX_ETH_RSS_CTX + 1 and .rxfh_priv_size…
- [Low] The patch changes how the default RSS key is generated, and the
commit message does not say so.
--
bnge: add RXFH ethtool support
This commit adds ethtool RXFH operations to bnge for reading and
programming the RSS hash key, indirection table and hash fields. The set
operations are only allowed while the interface is down. It also moves
default RSS key generation to netdev_rss_key_fill() in bnge_netdev_alloc()
and adds an IPv6 flow label RSS capability bit.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> index 2467e44de291a..6fbda4fc1a0c4 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
[ ... ]
> @@ -740,6 +743,272 @@ static int bnge_set_pauseparam(struct net_device *dev,
[ ... ]
> +static int bnge_get_rxfh_fields(struct net_device *dev,
> + struct ethtool_rxfh_fields *cmd)
> +{
[ ... ]
> + case AH_ESP_V4_FLOW:
> + if (bd->rss_hash_cfg &
> + (VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 |
> + VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV4))
> + cmd->data |= RXH_IP_SRC | RXH_IP_DST |
> + RXH_L4_B_0_1 | RXH_L4_B_2_3;
> + cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
> + break;
> + case SCTP_V4_FLOW:
> + case AH_V4_FLOW:
> + case ESP_V4_FLOW:
> + case IPV4_FLOW:
> + cmd->data |= bnge_get_ethtool_ipv4_rss(bd);
> + break;
[Severity: Medium]
Should AH_V4_FLOW and ESP_V4_FLOW report the SPI bits here too?
When VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 and ESP_SPI_IPV4 are set, the
hardware hashes AH/ESP packets on the SPI, and AH_ESP_V4_FLOW reports that
as a 4-tuple. AH_V4_FLOW and ESP_V4_FLOW fall into the generic IPv4 case
instead, and only report RXH_IP_SRC | RXH_IP_DST for the same packets.
The set side splits the same way. In bnge_set_rxfh_fields(), only
AH_ESP_V4_FLOW reaches the branch that clears the SPI bits:
} else if (cmd->flow_type == AH_ESP_V4_FLOW) {
...
rss_hash_cfg &= ~(VNIC_RSS_CFG_REQ_HASH_TYPE_AH_SPI_IPV4 |
VNIC_RSS_CFG_REQ_HASH_TYPE_ESP_SPI_IPV4);
As a result, a 2-tuple or zero request for AH_V4_FLOW or ESP_V4_FLOW
returns success while SPI hashing stays on.
The ethtool netlink RSS flow-hash attributes expose ah4/esp4 separately
from ah-esp4. An RSS GET can therefore show ah-esp4 and ah4 disagreeing
about the same packets.
The IPv6 cases (AH_V6_FLOW and ESP_V6_FLOW versus AH_ESP_V6_FLOW) look the
same. bnxt_get_rxfh_fields() and bnxt_set_rxfh_fields() follow the same
pattern.
[ ... ]
> +static int bnge_set_rxfh_fields(struct net_device *dev,
> + const struct ethtool_rxfh_fields *cmd,
> + struct netlink_ext_ack *extack)
> +{
[ ... ]
> + switch (cmd->flow_type) {
> + case TCP_V4_FLOW:
> + case UDP_V4_FLOW:
> + case SCTP_V4_FLOW:
> + case AH_ESP_V4_FLOW:
> + case AH_V4_FLOW:
> + case ESP_V4_FLOW:
> + case IPV4_FLOW:
> + if (tuple == 2)
> + rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4;
> + else if (!tuple && cmd->flow_type == IPV4_FLOW)
> + rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4;
> + break;
[Severity: Medium]
What happens when data == 0 is requested for a flow type other than
IPV4_FLOW?
With tuple == 0, VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4 is only cleared for
IPV4_FLOW. For SCTP_V4_FLOW, AH_V4_FLOW and ESP_V4_FLOW nothing changes,
so this check further down returns success, even while the interface is
running:
if (bd->rss_hash_cfg == rss_hash_cfg)
return 0;
For TCP/UDP/AH_ESP, only the L4 bit is cleared and the IPv4 2-tuple bit
stays set.
bnge_get_rxfh_fields() then reports RXH_IP_SRC | RXH_IP_DST for a flow type
that was just set to 0, and the hardware keeps hashing those packets. The
default rss_hash_cfg includes VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4, so something
like this takes the no-op path:
ethtool -N ethX rx-flow-hash sctp4 ""
The IPv6 branch has the same restriction on IPV6_FLOW.
bnxt_set_rxfh_fields() clears the L3 bit for any !tuple request, which is
where the two drivers differ. Would it be better to return an error for
zero requests that can't be applied, instead of returning 0?
> +
> + case TCP_V6_FLOW:
> + case UDP_V6_FLOW:
> + case SCTP_V6_FLOW:
> + case AH_ESP_V6_FLOW:
> + case AH_V6_FLOW:
> + case ESP_V6_FLOW:
> + case IPV6_FLOW:
> + if (cmd->data & RXH_IP6_FL) {
> + rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL;
> + rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6;
> + } else if (tuple == 2 || tuple == 4) {
> + rss_hash_cfg |= VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6;
> + rss_hash_cfg &= ~VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6_FLOW_LABEL;
[Severity: Medium]
Is tuple == 4 meant to change the global IPv6 bits here?
On the IPv4 side, a 4-tuple request leaves VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4
alone; only tuple == 2 sets it. Here, a 4-tuple tcp6/udp6/ah-esp6 request
sets VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6 and clears IPV6_FLOW_LABEL.
This sequence turns 2-tuple hashing back on for ip6/sctp6/ah6/esp6:
ethtool -N ethX rx-flow-hash ip6 ""
ethtool -N ethX rx-flow-hash tcp6 sdfn
In the same way, a tcp6 sdfn request after ip6 sdl silently turns off flow
label hashing.
RXH_4TUPLE already contains RXH_IP_SRC | RXH_IP_DST, so the TCP_V6_FLOW
readback does not need the IPV6 bit set.
> + } else if (!tuple && cmd->flow_type == IPV6_FLOW) {
[ ... ]
> +static int bnge_set_rxfh(struct net_device *dev,
> + struct ethtool_rxfh_param *rxfh,
> + struct netlink_ext_ack *extack)
> +{
> + struct bnge_net *bn = netdev_priv(dev);
> +
> + if (rxfh->hfunc && rxfh->hfunc != ETH_RSS_HASH_TOP)
> + return -EOPNOTSUPP;
> +
> + if (netif_running(dev)) {
> + NL_SET_ERR_MSG_MOD(extack,
> + "RSS configuration can only be changed while the interface is down");
> + return -EBUSY;
> + }
> +
> + bnge_modify_rss(bn, NULL, NULL, rxfh);
[Severity: High]
What limits tbl_size here when bd->rx_nr_rings is large?
bnge_modify_rss() copies bnge_get_rxfh_indir_size(bd) entries into
bd->rss_indir_tbl. That size grows with bd->rx_nr_rings through
bnge_cal_nr_rss_ctxs(). The table itself only holds
bd->rss_indir_tbl_entries entries (BNGE_MAX_RSS_TABLE_ENTRIES, 512):
drivers/net/ethernet/broadcom/bnge/bnge_vnic.c:bnge_modify_rss() {
...
for (i = 0; i < tbl_size; i++)
indir_tbl[i] = rxfh->indir[i];
pad = bd->rss_indir_tbl_entries - tbl_size;
if (pad && !rss_ctx)
memset(&indir_tbl[i], 0, pad * sizeof(*indir_tbl));
...
}
With more than 512 RX rings, tbl_size is larger than the table. The
copy loop then writes past the end of bd->rss_indir_tbl. pad is a u32,
so it wraps around to a value near 4G, and the memset() clears
gigabytes of kernel memory after the table.
The ethtool core sizes rxfh->indir from get_rxfh_indir_size, so the
user buffer is fine. It is the driver table that overflows.
bnge_get_rxfh() has the matching out-of-bounds read, because it
copies tbl_size entries out of bd->rss_indir_tbl.
bnge_fill_hw_rss_tbl() uses the same tbl_size. It writes two __le16
values per entry into vnic->rss_table, which would overrun that DMA
buffer on open. That code predates this patch, but the root cause is
the same.
Does anything in the ring reservation path keep rx_nr_rings at or
below BNGE_MAX_RSS_TABLE_ENTRIES? If not, should the RX ring count be
capped there, or should bnge_get_rxfh_indir_size() be clamped to
bd->rss_indir_tbl_entries?
[Severity: Low]
Does the indirection table stored here survive the next open?
bnge_open_core() calls bnge_reserve_rings(), which still resets the table
without checking netif_is_rxfh_configured():
drivers/net/ethernet/broadcom/bnge/bnge_resc.c:bnge_reserve_rings() {
...
if (old_rx_rings != bd->hw_resc.resv_rx_rings)
bnge_set_dflt_rss_indir_tbl(bd);
...
}
The reserved RX ring count can change while the interface is down. For
example, bnge_is_agg_reqd() can flip after an MTU or LRO/GRO change, or
firmware can grant a different number of RX rings. In that case the user
table is silently overwritten while IFF_RXFH_CONFIGURED stays set.
The next patch in the series, "bnge: add ethtool support to manage RSS
contexts", appears to add a netif_is_rxfh_configured() guard and
ethtool_rxfh_indir_lost() handling here. If so, only this intermediate
commit is affected.
[ ... ]
> @@ -757,6 +1026,18 @@ static const struct ethtool_ops bnge_ethtool_ops = {
> .get_eth_ctrl_stats = bnge_get_eth_ctrl_stats,
> .get_pause_stats = bnge_get_pause_stats,
> .get_rmon_stats = bnge_get_rmon_stats,
> + /* RXFH */
> + .rxfh_per_ctx_key = 1,
> + .rxfh_max_num_contexts = BNGE_MAX_ETH_RSS_CTX + 1,
> + .rxfh_indir_space = BNGE_MAX_RSS_TABLE_ENTRIES,
> + .rxfh_priv_size = sizeof(struct bnge_rss_ctx),
[Severity: Low]
This isn't a bug, but rxfh_per_ctx_key, rxfh_max_num_contexts and
rxfh_priv_size are set here without create_rxfh_context,
modify_rxfh_context or remove_rxfh_context. At this commit the ethtool core
rejects every context request with -EOPNOTSUPP, so these fields and
BNGE_MAX_ETH_RSS_CTX have no effect.
The next patch, "bnge: add ethtool support to manage RSS contexts", adds
the callbacks. Would these fields fit better in that patch?
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index c6b9048586b36..ef5d13337e60b 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> @@ -1739,13 +1739,6 @@ static void bnge_init_vnics(struct bnge_net *bn)
> u8 *key = (void *)vnic->rss_hash_key;
> int k;
>
> - if (!bn->rss_hash_key_valid &&
> - !bn->rss_hash_key_updated) {
> - get_random_bytes(bn->rss_hash_key,
> - HW_HASH_KEY_SIZE);
> - bn->rss_hash_key_updated = true;
> - }
> -
> memcpy(vnic->rss_hash_key, bn->rss_hash_key,
> HW_HASH_KEY_SIZE);
>
> @@ -3398,6 +3391,9 @@ int bnge_netdev_alloc(struct bnge_dev *bd, int max_irqs)
> bn->tx_ring_size = BNGE_DEFAULT_TX_RING_SIZE;
> bn->rx_dir = DMA_FROM_DEVICE;
>
> + netdev_rss_key_fill(bn->rss_hash_key, sizeof(bn->rss_hash_key));
> + bn->rss_hash_key_updated = true;
> +
[Severity: Low]
This isn't a bug, but the commit message doesn't mention this change to
the default RSS key.
netdev_rss_key_fill() copies the global netdev_rss_key, so all bnge ports
now share one key. Before, each port got its own get_random_bytes() key on
first open.
Also, the removed check in bnge_init_vnics() was the only reader of
bn->rss_hash_key_valid. bnge_init_vnics() still sets it:
bn->rss_hash_key_valid = true;
but nothing tests it any more. Can that field be removed?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 05/10] bnge: add ethtool support to manage RSS contexts
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (3 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 04/10] bnge: add RXFH ethtool support Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 06/10] bnge: remove refcount from L2 filter Vikas Gupta
` (4 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Introduce ethtool callbacks to manage RSS contexts.
Each context allocates a dedicated VNIC with its own RSS indirection
table and hash key, configured with TPA settings matching those of the
default VNIC.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Bhargava Chenna Marreddy <bhargava.marreddy@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_ethtool.c | 163 ++++++++++++++++++
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 9 +
.../net/ethernet/broadcom/bnge/bnge_resc.c | 26 ++-
.../net/ethernet/broadcom/bnge/bnge_vnic.c | 39 ++++-
.../net/ethernet/broadcom/bnge/bnge_vnic.h | 14 +-
5 files changed, 241 insertions(+), 10 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
index 6fbda4fc1a0c..85dbe64d4c12 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
@@ -1009,6 +1009,166 @@ static u32 bnge_get_rx_ring_count(struct net_device *dev)
return bd->rx_nr_rings;
}
+static int bnge_rxfh_context_check(struct bnge_net *bn,
+ const struct ethtool_rxfh_param *rxfh,
+ struct netlink_ext_ack *extack)
+{
+ if (rxfh->hfunc && rxfh->hfunc != ETH_RSS_HASH_TOP) {
+ NL_SET_ERR_MSG_MOD(extack, "RSS hash function not supported");
+ return -EOPNOTSUPP;
+ }
+
+ if (!(bn->priv_flags & BNGE_NET_EN_NTUPLE)) {
+ NL_SET_ERR_MSG_MOD(extack,
+ "Enable ntuple filtering before adding RSS contexts");
+ return -EOPNOTSUPP;
+ }
+
+ if (!netif_running(bn->netdev)) {
+ NL_SET_ERR_MSG_MOD(extack, "Unable to set RSS contexts when interface is down");
+ return -EAGAIN;
+ }
+
+ return 0;
+}
+
+static int bnge_create_rxfh_context(struct net_device *dev,
+ struct ethtool_rxfh_context *ctx,
+ const struct ethtool_rxfh_param *rxfh,
+ struct netlink_ext_ack *extack)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_rss_ctx *rss_ctx;
+ struct bnge_vnic_info *vnic;
+ int rc;
+
+ rc = bnge_rxfh_context_check(bn, rxfh, extack);
+ if (rc)
+ return rc;
+
+ if (bn->num_rss_ctx >= BNGE_MAX_ETH_RSS_CTX) {
+ NL_SET_ERR_MSG_FMT_MOD(extack, "Out of RSS contexts, maximum %u",
+ BNGE_MAX_ETH_RSS_CTX);
+ return -EINVAL;
+ }
+
+ if (!bnge_arfs_capable(bn->bd, true)) {
+ NL_SET_ERR_MSG_MOD(extack, "Out of hardware resources");
+ return -ENOMEM;
+ }
+
+ rss_ctx = ethtool_rxfh_context_priv(ctx);
+
+ bn->num_rss_ctx++;
+
+ vnic = &rss_ctx->vnic;
+
+ bnge_init_vnic_mem(vnic);
+
+ vnic->rss_ctx = ctx;
+ vnic->flags |= BNGE_VNIC_RSSCTX_FLAG;
+ rc = bnge_alloc_vnic_rss_table(bn, vnic);
+ if (rc)
+ goto err_del_rss_ctx;
+
+ /* Populate defaults in the context */
+ bnge_set_dflt_rss_indir_tbl(bn->bd, ctx);
+ ctx->hfunc = ETH_RSS_HASH_TOP;
+ memcpy(vnic->rss_hash_key, bn->rss_hash_key, HW_HASH_KEY_SIZE);
+ memcpy(ethtool_rxfh_context_key(ctx),
+ bn->rss_hash_key, HW_HASH_KEY_SIZE);
+
+ rc = bnge_hwrm_vnic_alloc(bn->bd, vnic, bn->bd->rx_nr_rings);
+ if (rc) {
+ NL_SET_ERR_MSG_MOD(extack, "Unable to allocate VNIC");
+ goto err_del_rss_ctx;
+ }
+
+ rc = bnge_hwrm_vnic_set_tpa(bn->bd, vnic,
+ bn->priv_flags & BNGE_NET_EN_TPA);
+ if (rc) {
+ NL_SET_ERR_MSG_MOD(extack,
+ "Unable to set TPA settings to vnic");
+ goto err_del_rss_ctx;
+ }
+ bnge_modify_rss(bn, ctx, rss_ctx, rxfh);
+
+ rc = bnge_setup_vnic(bn, vnic);
+ if (rc) {
+ NL_SET_ERR_MSG_MOD(extack, "Unable to setup vnic");
+ goto err_del_rss_ctx;
+ }
+
+ rss_ctx->index = rxfh->rss_context;
+ return 0;
+
+err_del_rss_ctx:
+ bnge_del_one_rss_ctx(bn, rss_ctx, true);
+ return rc;
+}
+
+static int bnge_modify_rxfh_context(struct net_device *dev,
+ struct ethtool_rxfh_context *ctx,
+ const struct ethtool_rxfh_param *rxfh,
+ struct netlink_ext_ack *extack)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ u8 old_key[HW_HASH_KEY_SIZE];
+ struct bnge_rss_ctx *rss_ctx;
+ u32 *old_indir = NULL;
+ u32 tbl_size;
+ int rc;
+
+ rc = bnge_rxfh_context_check(bn, rxfh, extack);
+ if (rc)
+ return rc;
+
+ rss_ctx = ethtool_rxfh_context_priv(ctx);
+ tbl_size = bnge_get_rxfh_indir_size(bn->bd);
+
+ /* Snapshot the software state so it can be restored if the hardware
+ * update fails, keeping the reported config consistent with the
+ * hardware.
+ */
+ if (rxfh->key)
+ memcpy(old_key, rss_ctx->vnic.rss_hash_key, HW_HASH_KEY_SIZE);
+ if (rxfh->indir) {
+ old_indir = kmemdup(ethtool_rxfh_context_indir(ctx),
+ tbl_size * sizeof(*old_indir), GFP_KERNEL);
+ if (!old_indir)
+ return -ENOMEM;
+ }
+
+ bnge_modify_rss(bn, ctx, rss_ctx, rxfh);
+
+ rc = bnge_hwrm_vnic_rss_cfg(bn, &rss_ctx->vnic);
+ if (rc) {
+ if (rxfh->key)
+ memcpy(rss_ctx->vnic.rss_hash_key, old_key,
+ HW_HASH_KEY_SIZE);
+ if (rxfh->indir)
+ memcpy(ethtool_rxfh_context_indir(ctx), old_indir,
+ tbl_size * sizeof(*old_indir));
+ }
+
+ kfree(old_indir);
+ return rc;
+}
+
+static int bnge_remove_rxfh_context(struct net_device *dev,
+ struct ethtool_rxfh_context *ctx,
+ u32 rss_context,
+ struct netlink_ext_ack *extack)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_rss_ctx *rss_ctx;
+
+ rss_ctx = ethtool_rxfh_context_priv(ctx);
+
+ bnge_del_one_rss_ctx(bn, rss_ctx, true);
+ return 0;
+}
+
static const struct ethtool_ops bnge_ethtool_ops = {
.cap_link_lanes_supported = 1,
.get_link_ksettings = bnge_get_link_ksettings,
@@ -1038,6 +1198,9 @@ static const struct ethtool_ops bnge_ethtool_ops = {
.set_rxfh = bnge_set_rxfh,
.get_rxfh_fields = bnge_get_rxfh_fields,
.set_rxfh_fields = bnge_set_rxfh_fields,
+ .create_rxfh_context = bnge_create_rxfh_context,
+ .modify_rxfh_context = bnge_modify_rxfh_context,
+ .remove_rxfh_context = bnge_remove_rxfh_context,
};
void bnge_set_ethtool_ops(struct net_device *dev)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index ef5d13337e60..ac77ba813f01 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -2784,6 +2784,8 @@ static int bnge_open_core(struct bnge_net *bn)
/* Poll link status and check for SFP+ module status */
bnge_get_port_module_status(bn);
+ bnge_hwrm_realloc_rss_ctx_vnic(bn);
+
return 0;
err_free_irq:
@@ -3001,7 +3003,10 @@ static void bnge_close_core(struct bnge_net *bn)
clear_bit(BNGE_STATE_OPEN, &bd->state);
timer_delete_sync(&bn->timer);
+
+ bnge_clear_rss_ctxs(bn);
bnge_shutdown_nic(bn);
+
bnge_disable_napi(bn);
/* Save ring stats before shutdown */
@@ -3139,6 +3144,10 @@ static int bnge_set_features(struct net_device *dev, netdev_features_t features)
if (flags == bn->priv_flags)
return 0;
+ if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) &&
+ !(flags & BNGE_NET_EN_NTUPLE) && bn->num_rss_ctx)
+ return -EBUSY;
+
bn->priv_flags = flags;
return 0;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
index e0744e59c6ae..8f1b0f42773a 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
@@ -229,6 +229,19 @@ u32 bnge_get_rxfh_indir_size(struct bnge_dev *bd)
BNGE_RSS_TABLE_ENTRIES;
}
+static u16 bnge_get_max_rss_ring(struct bnge_dev *bd)
+{
+ u32 i, tbl_size, max_ring = 0;
+
+ if (!bd->rss_indir_tbl)
+ return 0;
+
+ tbl_size = bnge_get_rxfh_indir_size(bd);
+ for (i = 0; i < tbl_size; i++)
+ max_ring = max_t(u32, max_ring, bd->rss_indir_tbl[i]);
+ return max_ring;
+}
+
static void bnge_copy_reserved_rings(struct bnge_dev *bd,
struct bnge_hw_rings *hwr)
{
@@ -344,9 +357,15 @@ int bnge_reserve_rings(struct bnge_dev *bd)
hwr.nq = sh ? max_t(u16, tx_cp, rx_rings) : tx_cp + rx_rings;
bd->tx_nr_rings = hwr.tx;
- if (rx_rings != bd->rx_nr_rings)
+ if (rx_rings != bd->rx_nr_rings) {
dev_warn(bd->dev, "RX rings resv reduced to %d than earlier %d requested\n",
rx_rings, bd->rx_nr_rings);
+ if (bd->netdev && netif_is_rxfh_configured(bd->netdev) &&
+ (bnge_cal_nr_rss_ctxs(bd->rx_nr_rings) !=
+ bnge_cal_nr_rss_ctxs(rx_rings) ||
+ bnge_get_max_rss_ring(bd) >= rx_rings))
+ ethtool_rxfh_indir_lost(bd->netdev);
+ }
bd->rx_nr_rings = rx_rings;
bd->nq_nr_rings = hwr.nq;
@@ -354,8 +373,9 @@ int bnge_reserve_rings(struct bnge_dev *bd)
if (!bnge_rings_ok(&hwr))
return -ENOMEM;
- if (old_rx_rings != bd->hw_resc.resv_rx_rings)
- bnge_set_dflt_rss_indir_tbl(bd);
+ if (old_rx_rings != bd->hw_resc.resv_rx_rings &&
+ (!bd->netdev || !netif_is_rxfh_configured(bd->netdev)))
+ bnge_set_dflt_rss_indir_tbl(bd, NULL);
if (!bnge_aux_registered(bd)) {
u16 resv_msix, resv_ctx, aux_ctxs;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
index a44971a7c74d..4ad676e983f2 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
@@ -12,21 +12,26 @@
#include "bnge_filter.h"
#include "bnge_resc.h"
-void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd)
+void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd,
+ struct ethtool_rxfh_context *rss_ctx)
{
u16 max_entries, pad;
u32 *rss_indir_tbl;
u16 i;
max_entries = bnge_get_rxfh_indir_size(bd);
- rss_indir_tbl = &bd->rss_indir_tbl[0];
+
+ if (rss_ctx)
+ rss_indir_tbl = ethtool_rxfh_context_indir(rss_ctx);
+ else
+ rss_indir_tbl = &bd->rss_indir_tbl[0];
for (i = 0; i < max_entries; i++)
rss_indir_tbl[i] = ethtool_rxfh_indir_default(i,
bd->rx_nr_rings);
pad = bd->rss_indir_tbl_entries - max_entries;
- if (pad)
+ if (pad && !rss_ctx)
memset(&rss_indir_tbl[i], 0, pad * sizeof(*rss_indir_tbl));
}
@@ -44,9 +49,14 @@ void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic)
if (vnic->flags & BNGE_VNIC_NTUPLE_FLAG)
j = ethtool_rxfh_indir_default(i, bd->rx_nr_rings);
+ else if (vnic->flags & BNGE_VNIC_RSSCTX_FLAG)
+ j = ethtool_rxfh_context_indir(vnic->rss_ctx)[i];
else
j = bd->rss_indir_tbl[i];
+ if (j >= bd->rx_nr_rings)
+ j = ethtool_rxfh_indir_default(i, bd->rx_nr_rings);
+
rxr = &bn->rx_ring[j];
ring_id = rxr->rx_ring_struct.fw_ring_id;
@@ -155,7 +165,7 @@ void bnge_modify_rss(struct bnge_net *bn, struct ethtool_rxfh_context *ctx,
for (i = 0; i < tbl_size; i++)
indir_tbl[i] = rxfh->indir[i];
pad = bd->rss_indir_tbl_entries - tbl_size;
- if (pad)
+ if (pad && !rss_ctx)
memset(&indir_tbl[i], 0, pad * sizeof(*indir_tbl));
}
}
@@ -168,8 +178,11 @@ void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
bnge_hwrm_vnic_free_one(bn->bd, &rss_ctx->vnic);
for (i = 0; i < BNGE_MAX_CTX_PER_VNIC; i++) {
- if (vnic->fw_rss_cos_lb_ctx[i] != INVALID_HW_RING_ID)
+ if (vnic->fw_rss_cos_lb_ctx[i] != INVALID_HW_RING_ID) {
bnge_hwrm_vnic_ctx_free_one(bn->bd, vnic, i);
+ vnic->fw_rss_cos_lb_ctx[i] = INVALID_HW_RING_ID;
+ bn->rsscos_nr_ctxs--;
+ }
}
if (!all)
@@ -188,6 +201,10 @@ void bnge_hwrm_realloc_rss_ctx_vnic(struct bnge_net *bn)
struct ethtool_rxfh_context *ctx;
unsigned long context;
+ /* Erasing lost contexts touches netdev->ethtool->rss_ctx and calls
+ * ethtool_rxfh_context_lost(), which requires the ethtool rss_lock.
+ */
+ mutex_lock(&bn->netdev->ethtool->rss_lock);
xa_for_each(&bn->netdev->ethtool->rss_ctx, context, ctx) {
struct bnge_rss_ctx *rss_ctx = ethtool_rxfh_context_priv(ctx);
struct bnge_vnic_info *vnic = &rss_ctx->vnic;
@@ -201,6 +218,7 @@ void bnge_hwrm_realloc_rss_ctx_vnic(struct bnge_net *bn)
ethtool_rxfh_context_lost(bn->netdev, rss_ctx->index);
}
}
+ mutex_unlock(&bn->netdev->ethtool->rss_lock);
}
void bnge_clear_rss_ctxs(struct bnge_net *bn)
@@ -242,3 +260,14 @@ int bnge_alloc_vnic_rss_table(struct bnge_net *bn,
vnic->rss_hash_key_dma_addr = vnic->rss_table_dma_addr + size;
return 0;
}
+
+void bnge_init_vnic_mem(struct bnge_vnic_info *vnic)
+{
+ int i;
+
+ vnic->fw_vnic_id = INVALID_HW_RING_ID;
+ vnic->vnic_id = BNGE_VNIC_ID_INVALID;
+
+ for (i = 0; i < BNGE_MAX_CTX_PER_VNIC; i++)
+ vnic->fw_rss_cos_lb_ctx[i] = INVALID_HW_RING_ID;
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
index bc2ed911ced3..1cc2481ba217 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.h
@@ -22,6 +22,11 @@ struct bnge_l2_filter;
#define BNGE_MAX_MC_ADDRS 16
#define BNGE_MAX_UC_ADDRS 4
+#define BNGE_VNIC_ID_INVALID 0xffffffff
+
+struct ethtool_rxfh_context;
+struct ethtool_rxfh_param;
+
enum {
BNGE_VNIC_DEFAULT = 0,
BNGE_VNIC_NTUPLE = 1
@@ -31,7 +36,8 @@ enum {
BNGE_VNIC_RSS_FLAG = BIT(0),
BNGE_VNIC_MCAST_FLAG = BIT(1),
BNGE_VNIC_UCAST_FLAG = BIT(2),
- BNGE_VNIC_NTUPLE_FLAG = BIT(3)
+ BNGE_VNIC_NTUPLE_FLAG = BIT(3),
+ BNGE_VNIC_RSSCTX_FLAG = BIT(4)
};
struct bnge_vnic_info {
@@ -56,6 +62,8 @@ struct bnge_vnic_info {
u32 flags;
u32 vnic_id;
+
+ struct ethtool_rxfh_context *rss_ctx;
};
struct bnge_rss_ctx {
@@ -67,7 +75,8 @@ void bnge_fill_hw_rss_tbl(struct bnge_net *bn, struct bnge_vnic_info *vnic);
int bnge_hwrm_vnic_rss_cfg(struct bnge_net *bn,
struct bnge_vnic_info *vnic);
int bnge_setup_vnic(struct bnge_net *bn, struct bnge_vnic_info *vnic);
-void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd);
+void bnge_set_dflt_rss_indir_tbl(struct bnge_dev *bd,
+ struct ethtool_rxfh_context *ctx);
int bnge_alloc_rfs_vnic(struct bnge_net *bn);
int bnge_alloc_vnic_rss_table(struct bnge_net *bn,
struct bnge_vnic_info *vnic);
@@ -79,4 +88,5 @@ void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
bool all);
void bnge_hwrm_realloc_rss_ctx_vnic(struct bnge_net *bn);
void bnge_clear_rss_ctxs(struct bnge_net *bn);
+void bnge_init_vnic_mem(struct bnge_vnic_info *vnic);
#endif /* _BNGE_VNIC_H_ */
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 05/10] bnge: add ethtool support to manage RSS contexts
2026-09-28 6:13 ` [net-next, v4 05/10] bnge: add ethtool support to manage RSS contexts Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 5 potential
issue(s) to consider.
Critical: 0 · High: 0 · Medium: 4 · Low: 1
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [Medium] When bnge_reserve_rings() gets fewer usable RX rings at open
(it logs 'RX rings resv reduced to %d than earlier %d requested'), it…
- [Medium] RSS contexts survive a forced NTUPLE/aRFS disable in
bnge_init_chip(), which breaks the rule this patch adds in…
- [Medium] In bnge_reserve_rings(), marking the default table lost and
resetting it use different conditions.
- [Medium] When bnge_hwrm_vnic_rss_cfg() fails,
bnge_modify_rxfh_context() restores only the software state: the old…
- [Low] The commit message says only that ethtool callbacks for RSS
contexts are added.
--
bnge: add ethtool support to manage RSS contexts
This commit adds the create/modify/remove_rxfh_context ethtool callbacks
to bnge. Each RSS context gets its own VNIC with its own indirection
table and hash key. Contexts are torn down on close and rebuilt on open.
> Introduce ethtool callbacks to manage RSS contexts.
[Severity: Low]
The patch also makes several behaviour changes that the commit message
does not mention. Some of them affect the default (context 0) table:
- When resv_rx_rings changes, bnge_reserve_rings() used to always reset
the default indirection table. It now keeps a user-configured table,
or calls ethtool_rxfh_indir_lost(). That call logs "device error, RSS
indirection table lost" and sends ETHTOOL_MSG_RSS_NTF.
- bnge_fill_hw_rss_tbl() now silently replaces any out-of-range ring
index with ethtool_rxfh_indir_default(). This applies to every VNIC,
including the default one.
- bnge_set_features() now returns -EBUSY if NTUPLE is cleared while RSS
contexts exist.
- Contexts are torn down on close and rebuilt on open. If the rebuild
fails, the context is destroyed.
Could these changes be described in the commit message, or split into
separate patches?
The changes to bnge_del_one_rss_ctx() (rsscos_nr_ctxs accounting) and
to bnge_hwrm_realloc_rss_ctx_vnic() (taking rss_lock) only touch helpers
that had no callers before this patch. So they don't fix reachable bugs
in the earlier patches.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> index 6fbda4fc1a0c4..85dbe64d4c129 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
[ ... ]
> +static int bnge_modify_rxfh_context(struct net_device *dev,
> + struct ethtool_rxfh_context *ctx,
> + const struct ethtool_rxfh_param *rxfh,
> + struct netlink_ext_ack *extack)
> +{
[ ... ]
> + rss_ctx = ethtool_rxfh_context_priv(ctx);
> + tbl_size = bnge_get_rxfh_indir_size(bn->bd);
> +
> + /* Snapshot the software state so it can be restored if the hardware
> + * update fails, keeping the reported config consistent with the
> + * hardware.
> + */
[ ... ]
> + bnge_modify_rss(bn, ctx, rss_ctx, rxfh);
> +
> + rc = bnge_hwrm_vnic_rss_cfg(bn, &rss_ctx->vnic);
> + if (rc) {
> + if (rxfh->key)
> + memcpy(rss_ctx->vnic.rss_hash_key, old_key,
> + HW_HASH_KEY_SIZE);
> + if (rxfh->indir)
> + memcpy(ethtool_rxfh_context_indir(ctx), old_indir,
> + tbl_size * sizeof(*old_indir));
> + }
[Severity: Medium]
Does this rollback really keep the reported config consistent with the
hardware, as the comment above says?
bnge_hwrm_vnic_rss_cfg() is not atomic. First, bnge_hwrm_vnic_set_rss()
uses bnge_fill_hw_rss_tbl() to write the new ring IDs into
vnic->rss_table. It then sends one HWRM_VNIC_RSS_CFG per 64-entry
segment and stops at the first failure:
bnge_hwrm_vnic_set_rss() {
...
for (i = 0; i < nr_ctxs; ring_tbl_map += BNGE_RSS_TABLE_SIZE, i++) {
...
rc = bnge_hwrm_req_send(bd, req);
if (rc)
goto exit;
}
...
}
After that, bnge_hwrm_vnic_rss_cfg() sends a separate HWRM_VNIC_CFG.
A later segment can fail, or HWRM_VNIC_CFG can fail after all the
RSS_CFG commands succeeded. In either case the device is already using
the new (or partly new) key and table.
Only the software copies are restored here. So "ethtool -x ... context
N" would report the old config, while the DMA buffer vnic->rss_table
still holds the new ring IDs.
Should the error path program the restored state back into the
hardware?
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index ef5d13337e60b..ac77ba813f01c 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> @@ -2784,6 +2784,8 @@ static int bnge_open_core(struct bnge_net *bn)
> /* Poll link status and check for SFP+ module status */
> bnge_get_port_module_status(bn);
>
> + bnge_hwrm_realloc_rss_ctx_vnic(bn);
> +
> return 0;
[Severity: Medium]
What happens to existing RSS contexts if bnge_init_chip() force-disables
NTUPLE on reopen?
Contexts survive close, because bnge_clear_rss_ctxs() calls
bnge_del_one_rss_ctx(..., false), so num_rss_ctx stays non-zero.
On reopen, bnge_open_core()->bnge_init_nic()->bnge_init_chip() calls
bnge_alloc_rfs_vnic(). If that fails, the fallback clears the flags
without checking num_rss_ctx:
bnge_init_chip() {
...
if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) && bnge_is_arfs_cap(bd)) {
rc = bnge_alloc_rfs_vnic(bn);
if (rc) {
...
bd->flags &= ~BNGE_EN_ARFS_CAP;
bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
bn->netdev->features &= ~NETIF_F_NTUPLE;
}
}
...
}
bnge_hwrm_realloc_rss_ctx_vnic() is then called unconditionally here and
re-creates every context VNIC. This breaks the rule this patch adds to
bnge_set_features(): NTUPLE can't be turned off while num_rss_ctx is
non-zero.
After that:
- bnge_modify_rxfh_context() always fails in bnge_rxfh_context_check()
with "Enable ntuple filtering before adding RSS contexts".
- bnge_fix_features() strips NETIF_F_NTUPLE whenever
!bnge_is_arfs_cap(), so NTUPLE can't be turned back on.
- bnge_get_total_vnics() returns 1 and no longer counts num_rss_ctx.
Later ring reservations therefore don't cover the VNIC and RSS COS
resources of the live contexts. A later realloc can then fail and
destroy the contexts via ethtool_rxfh_context_lost().
By the end of the series, bnge_srxclsrlins() also requires NTUPLE
(-EPERM), so no filter can target these contexts. Only
bnge_remove_rxfh_context() still works on them.
Should the contexts be dropped when this fallback runs, or should the
fallback be skipped when contexts exist?
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> index e0744e59c6aea..8f1b0f42773a2 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
> @@ -229,6 +229,19 @@ u32 bnge_get_rxfh_indir_size(struct bnge_dev *bd)
> BNGE_RSS_TABLE_ENTRIES;
> }
>
> +static u16 bnge_get_max_rss_ring(struct bnge_dev *bd)
> +{
> + u32 i, tbl_size, max_ring = 0;
> +
> + if (!bd->rss_indir_tbl)
> + return 0;
> +
> + tbl_size = bnge_get_rxfh_indir_size(bd);
> + for (i = 0; i < tbl_size; i++)
> + max_ring = max_t(u32, max_ring, bd->rss_indir_tbl[i]);
> + return max_ring;
> +}
[ ... ]
> @@ -344,9 +357,15 @@ int bnge_reserve_rings(struct bnge_dev *bd)
> hwr.nq = sh ? max_t(u16, tx_cp, rx_rings) : tx_cp + rx_rings;
> bd->tx_nr_rings = hwr.tx;
>
> - if (rx_rings != bd->rx_nr_rings)
> + if (rx_rings != bd->rx_nr_rings) {
> dev_warn(bd->dev, "RX rings resv reduced to %d than earlier %d requested\n",
> rx_rings, bd->rx_nr_rings);
> + if (bd->netdev && netif_is_rxfh_configured(bd->netdev) &&
> + (bnge_cal_nr_rss_ctxs(bd->rx_nr_rings) !=
> + bnge_cal_nr_rss_ctxs(rx_rings) ||
> + bnge_get_max_rss_ring(bd) >= rx_rings))
> + ethtool_rxfh_indir_lost(bd->netdev);
> + }
[Severity: Medium]
This only checks the default table. What about the non-default RSS
contexts?
bnge_get_max_rss_ring() scans only bd->rss_indir_tbl, and
ethtool_rxfh_indir_lost() only affects context 0.
The other contexts survive close/open and are programmed again through
this chain:
bnge_open_core()->bnge_hwrm_realloc_rss_ctx_vnic()->bnge_setup_vnic()->
bnge_hwrm_vnic_rss_cfg()->bnge_hwrm_vnic_set_rss()->bnge_fill_hw_rss_tbl()
At the end of that chain, bnge_fill_hw_rss_tbl() reads
ethtool_rxfh_context_indir(vnic->rss_ctx)[i].
The new clamp in bnge_fill_hw_rss_tbl() fixes stale entries only in the
hardware DMA table. Also, tbl_size there follows the new ring count, not
ctx->indir_size.
If the number of segments drops (for example 72 -> 60 rings, so 128 ->
64 entries), only the first part of the stored context table reaches the
hardware.
bnge_modify_rxfh_context() has the same issue: it snapshots and restores
bnge_get_rxfh_indir_size() entries, not ctx->indir_size entries.
Programming succeeds, so ethtool_rxfh_context_lost() is never called.
The driver also never uses ethtool_rxfh_ctxs_can_resize() or
ethtool_rxfh_ctxs_resize().
Wouldn't the hardware then send the context's flows to queues the user
never configured? Meanwhile ETHTOOL_MSG_RSS_GET would keep reporting the
old table and indir_size, and userspace would get no notification.
> bd->rx_nr_rings = rx_rings;
> bd->nq_nr_rings = hwr.nq;
> @@ -354,8 +373,9 @@ int bnge_reserve_rings(struct bnge_dev *bd)
> if (!bnge_rings_ok(&hwr))
> return -ENOMEM;
>
> - if (old_rx_rings != bd->hw_resc.resv_rx_rings)
> - bnge_set_dflt_rss_indir_tbl(bd);
> + if (old_rx_rings != bd->hw_resc.resv_rx_rings &&
> + (!bd->netdev || !netif_is_rxfh_configured(bd->netdev)))
> + bnge_set_dflt_rss_indir_tbl(bd, NULL);
[Severity: Medium]
Can the default table end up marked lost but never reset?
The ethtool_rxfh_indir_lost() call above depends on the usable ring
count (rx_rings != bd->rx_nr_rings). This reset depends on the reserved
count (old_rx_rings != bd->hw_resc.resv_rx_rings).
The usable count can shrink while resv_rx_rings stays the same. One way
is these two steps earlier in bnge_reserve_rings():
rx_rings = min_t(u16, rx_rings, hwr.grp);
...
rc = bnge_adjust_rings(bd, &rx_rings, &hwr.tx, hwr.nq, sh);
Another way is when aggregation doubles the requested hwr.rx and
firmware grants the same number of rings as before. For example: the
old reservation is 8 without agg, the new request is 16 with agg, and
firmware grants 8, leaving 4 usable.
In that case ethtool_rxfh_indir_lost() clears rss_indir_user_size. But
resv_rx_rings didn't change, so bd->rss_indir_tbl keeps the stale user
entries, which may point at queues >= the new rx_nr_rings.
get_rxfh would then report a non-default table with nonexistent queues,
while the core treats the table as unconfigured. At the same time,
bnge_fill_hw_rss_tbl() programs different, clamped values.
Should the default table also be reset whenever
ethtool_rxfh_indir_lost() is called here?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 06/10] bnge: remove refcount from L2 filter
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (4 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 05/10] bnge: add ethtool support to manage RSS contexts Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 07/10] bnge: add NTUPLE filter infrastructure Vikas Gupta
` (3 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
A reference count was previously used under the assumption that any
context holding the last reference could delete the L2 filter.
However, going forward, the filter will only be removed by the same
context that created it, making refcount unnecessary.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_filter.c | 28 ++++---------------
.../net/ethernet/broadcom/bnge/bnge_filter.h | 1 -
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 23 +++++++++++----
3 files changed, 23 insertions(+), 29 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
index a8bb441ebef9..3b26939b0457 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
@@ -13,8 +13,6 @@
void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr)
{
- if (!refcount_dec_and_test(&fltr->refcnt))
- return;
hlist_del_rcu(&fltr->base.hlist);
kfree_rcu(fltr, base.rcu);
}
@@ -31,18 +29,17 @@ static void bnge_init_l2_filter(struct bnge_net *bn,
head = &bn->l2_fltr_hash_tbl[idx];
hlist_add_head_rcu(&fltr->base.hlist, head);
- refcount_set(&fltr->refcnt, 1);
}
-static struct bnge_l2_filter *__bnge_lookup_l2_filter(struct bnge_net *bn,
- struct bnge_l2_key *key,
- u32 idx)
+static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u32 idx)
{
struct bnge_l2_filter *fltr;
struct hlist_head *head;
head = &bn->l2_fltr_hash_tbl[idx];
- hlist_for_each_entry_rcu(fltr, head, base.hlist) {
+ hlist_for_each_entry(fltr, head, base.hlist) {
struct bnge_l2_key *l2_key = &fltr->l2_key;
if (ether_addr_equal(l2_key->dst_mac_addr, key->dst_mac_addr) &&
@@ -52,20 +49,6 @@ static struct bnge_l2_filter *__bnge_lookup_l2_filter(struct bnge_net *bn,
return NULL;
}
-static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
- struct bnge_l2_key *key,
- u32 idx)
-{
- struct bnge_l2_filter *fltr;
-
- rcu_read_lock();
- fltr = __bnge_lookup_l2_filter(bn, key, idx);
- if (fltr)
- refcount_inc(&fltr->refcnt);
- rcu_read_unlock();
- return fltr;
-}
-
static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
struct bnge_l2_key *key,
gfp_t gfp)
@@ -75,9 +58,10 @@ static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
BNGE_L2_FLTR_HASH_MASK;
+
fltr = bnge_lookup_l2_filter(bn, key, idx);
if (fltr)
- return fltr;
+ return ERR_PTR(-EEXIST);
fltr = kzalloc_obj(*fltr, gfp);
if (!fltr)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
index ea1acefd70b4..a4a386202ac3 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
@@ -49,7 +49,6 @@ struct bnge_l2_filter {
/* base filter must be the first member */
struct bnge_filter_base base;
struct bnge_l2_key l2_key;
- refcount_t refcnt;
};
void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr);
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index ac77ba813f01..29a271f50e04 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -2011,19 +2011,24 @@ static bool bnge_uc_list_updated(struct bnge_net *bn,
const struct netdev_hw_addr_list *uc)
{
struct bnge_vnic_info *vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
+ struct net_device *dev = bn->netdev;
struct netdev_hw_addr *ha;
- int off = 0;
-
- if (netdev_hw_addr_list_count(uc) != (vnic->uc_filter_count - 1))
- return true;
+ int off = 0, count = 0;
+ /* dev_addr is installed as L2 filter 0 and skipped when building
+ * uc_list, so compare only the unicast entries that differ from it.
+ */
netdev_hw_addr_list_for_each(ha, uc) {
+ if (ether_addr_equal(ha->addr, dev->dev_addr))
+ continue;
+ if (count >= vnic->uc_filter_count - 1)
+ return true;
if (!ether_addr_equal(ha->addr, vnic->uc_list + off))
return true;
-
off += ETH_ALEN;
+ count++;
}
- return false;
+ return count != vnic->uc_filter_count - 1;
}
static bool bnge_promisc_ok(struct bnge_net *bn)
@@ -2058,6 +2063,12 @@ static int bnge_cfg_rx_mode(struct bnge_net *bn, struct netdev_hw_addr_list *uc,
vnic->rx_mask |= CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS;
} else {
netdev_hw_addr_list_for_each(ha, uc) {
+ /* dev_addr is already installed as L2 filter 0; a uc
+ * entry equal to it hashes to the same key and would be
+ * rejected as a duplicate, so skip it here.
+ */
+ if (ether_addr_equal(ha->addr, dev->dev_addr))
+ continue;
memcpy(vnic->uc_list + off, ha->addr, ETH_ALEN);
off += ETH_ALEN;
vnic->uc_filter_count++;
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 06/10] bnge: remove refcount from L2 filter
2026-09-28 6:13 ` [net-next, v4 06/10] bnge: remove refcount from L2 filter Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 2 potential
issue(s) to consider.
Critical: 0 · High: 0 · Medium: 1 · Low: 1
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [Medium] The commit message does not describe what the patch actually
changes.
- [Low] In bnge_cfg_rx_mode()
(drivers/net/ethernet/broadcom/bnge/bnge_netdev.c), the patch now…
--
bnge: remove refcount from L2 filter
This commit removes the refcount from struct bnge_l2_filter, because a
filter will now only be removed by the context that created it.
bnge_alloc_l2_filter() now returns -EEXIST for a duplicate key. The unicast
rx-mode code now skips uc entries that match dev->dev_addr.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> index a8bb441ebef9a..3b26939b04578 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
[ ... ]
> @@ -75,9 +58,10 @@ static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
>
> idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
> BNGE_L2_FLTR_HASH_MASK;
> +
> fltr = bnge_lookup_l2_filter(bn, key, idx);
> if (fltr)
> - return fltr;
> + return ERR_PTR(-EEXIST);
[Severity: Medium]
Should the commit message describe this change? It only says:
However, going forward, the filter will only be removed by the same
context that created it, making refcount unnecessary.
The patch also makes two behaviour changes that depend on each other:
- bnge_alloc_l2_filter() now fails a duplicate key with -EEXIST. Before,
it shared the existing filter.
- bnge_cfg_rx_mode() and bnge_uc_list_updated() now skip uc entries equal
to dev->dev_addr.
Without the skip, opening the device with dev_addr in dev->uc would now
fail with -EEXIST.
Together these two changes seem to fix a bug that dates back to commit
9afad4a17174 ("bng_en: Configure default VNIC"). Before this patch, dev->uc
could hold the device's own MAC, for example after "bridge fdb add
<own-mac> dev X self". In that case the open path did:
bnge_init_chip()
bnge_hwrm_set_vnic_filter(bn, 0, 0, dev_addr)
bnge_cfg_rx_mode()
bnge_hwrm_set_vnic_filter(bn, 0, 1, dev_addr)
bnge_alloc_l2_filter()
returns l2_filters[0] with refcount bumped
bnge_hwrm_l2_filter_alloc()
fltr->base.filter_id = resp->l2_filter_id;
After this, l2_filters[0] and l2_filters[1] pointed at the same object.
The original firmware ID for filter 0 was overwritten and leaked.
The next rx-mode update freed the new ID through l2_filters[1]. Later,
bnge_hwrm_clear_vnic_filter() freed that same ID a second time through
l2_filters[0]. If the firmware rejected the duplicate alloc instead,
bnge_init_chip() failed the open.
Could the commit message describe the -EEXIST and dev_addr skip changes?
Could it also carry a Fixes: tag, so this can be considered for stable?
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index ac77ba813f01c..29a271f50e044 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
[ ... ]
> @@ -2058,6 +2063,12 @@ static int bnge_cfg_rx_mode(struct bnge_net *bn, struct netdev_hw_addr_list *uc,
> vnic->rx_mask |= CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS;
> } else {
> netdev_hw_addr_list_for_each(ha, uc) {
> + /* dev_addr is already installed as L2 filter 0; a uc
> + * entry equal to it hashes to the same key and would be
> + * rejected as a duplicate, so skip it here.
> + */
> + if (ether_addr_equal(ha->addr, dev->dev_addr))
> + continue;
[Severity: Low]
Should the overflow check just above this loop also leave out the dev_addr
entry? It still counts every entry in uc:
if (netdev_hw_addr_list_count(uc) > (BNGE_MAX_UC_ADDRS - 1)) {
vnic->rx_mask |= CFA_L2_SET_RX_MASK_REQ_MASK_PROMISCUOUS;
uc_list has room for BNGE_MAX_UC_ADDRS - 1 entries, which is 3. A skipped
dev_addr entry now takes neither a uc_list slot nor an l2_filters[] slot.
Take uc = {dev_addr, A, B, C}. The check sees 4 > 3 and turns on
promiscuous mode, although A, B and C would fit in exact-match filters.
In that state uc_filter_count stays at 1. So in bnge_uc_list_updated(),
the first entry that is not dev_addr hits:
if (count >= vnic->uc_filter_count - 1)
return true;
As a result, every ndo_set_rx_mode call redoes the uc rebuild and the
promiscuous rx mask setup over HWRM.
The same input also went promiscuous before this patch. Back then,
though, dev_addr did take a slot, so the count and the fill loop agreed.
The fill and compare loops now count only entries that are not dev_addr.
Could the threshold count the same way?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 07/10] bnge: add NTUPLE filter infrastructure
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (5 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 06/10] bnge: remove refcount from L2 filter Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 08/10] bnge: add NTUPLE filter support in ethtool Vikas Gupta
` (2 subsequent siblings)
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Add the core NTUPLE filter data structures and firmware command
wrappers needed for flow rule management.
This helps further patches implement user-created NTUPLE rules
as well as rules via kernel flow steering.
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
---
drivers/net/ethernet/broadcom/bnge/bnge.h | 1 -
.../net/ethernet/broadcom/bnge/bnge_filter.c | 494 +++++++++++++++++-
.../net/ethernet/broadcom/bnge/bnge_filter.h | 67 ++-
.../ethernet/broadcom/bnge/bnge_hwrm_lib.c | 144 ++++-
.../ethernet/broadcom/bnge/bnge_hwrm_lib.h | 6 +
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 18 +-
.../net/ethernet/broadcom/bnge/bnge_netdev.h | 13 +
.../net/ethernet/broadcom/bnge/bnge_resc.c | 6 +-
.../net/ethernet/broadcom/bnge/bnge_vnic.c | 18 +
9 files changed, 752 insertions(+), 15 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge.h b/drivers/net/ethernet/broadcom/bnge/bnge.h
index 560b9d193dea..bda51fb58237 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge.h
@@ -160,7 +160,6 @@ struct bnge_dev {
u16 tso_max_segs;
int max_fltr;
-#define BNGE_L2_FLTR_MAX_FLTR 1024
u32 *rss_indir_tbl;
u16 rss_indir_tbl_entries;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
index 3b26939b0457..7fbb67021b5d 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
@@ -4,6 +4,8 @@
#include <linux/kernel.h>
#include <linux/dma-mapping.h>
#include <linux/jhash.h>
+#include <net/netdev_lock.h>
+#include <net/ipv6.h>
#include "bnge.h"
#include "bnge_netdev.h"
@@ -11,24 +13,150 @@
#include "bnge_hwrm_lib.h"
#include "bnge_filter.h"
-void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr)
+#define BNGE_IPV6_MASK_ALL {{{ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, \
+ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff }}}
+#define BNGE_IPV6_MASK_NONE {{{ 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 }}}
+
+const struct bnge_flow_masks BNGE_FLOW_MASK_NONE = {
+ .ports = {
+ .src = 0,
+ .dst = 0,
+ },
+ .addrs = {
+ .v6addrs = {
+ .src = BNGE_IPV6_MASK_NONE,
+ .dst = BNGE_IPV6_MASK_NONE,
+ },
+ },
+};
+
+const struct bnge_flow_masks BNGE_FLOW_IPV6_MASK_ALL = {
+ .ports = {
+ .src = cpu_to_be16(0xffff),
+ .dst = cpu_to_be16(0xffff),
+ },
+ .addrs = {
+ .v6addrs = {
+ .src = BNGE_IPV6_MASK_ALL,
+ .dst = BNGE_IPV6_MASK_ALL,
+ },
+ },
+};
+
+const struct bnge_flow_masks BNGE_FLOW_IPV4_MASK_ALL = {
+ .ports = {
+ .src = cpu_to_be16(0xffff),
+ .dst = cpu_to_be16(0xffff),
+ },
+ .addrs = {
+ .v4addrs = {
+ .src = cpu_to_be32(0xffffffff),
+ .dst = cpu_to_be32(0xffffffff),
+ },
+ },
+};
+
+static bool bnge_is_usr_fltr(struct bnge_filter_base *fltr)
+{
+ return (fltr->type == BNGE_FLTR_TYPE_L2 &&
+ fltr->flags & BNGE_ACT_RING_DST) ||
+ (fltr->type == BNGE_FLTR_TYPE_NTUPLE &&
+ fltr->flags & BNGE_ACT_NO_AGING);
+}
+
+static void bnge_insert_usr_fltr(struct bnge_net *bn,
+ struct bnge_filter_base *fltr)
+{
+ if (!bnge_is_usr_fltr(fltr))
+ return;
+
+ INIT_LIST_HEAD(&fltr->list_node);
+ list_add_tail(&fltr->list_node, &bn->usr_fltr_list);
+ bn->user_fltr_count++;
+}
+
+static void bnge_del_usr_fltr_node(struct bnge_net *bn,
+ struct bnge_filter_base *fltr)
{
+ if (!bnge_is_usr_fltr(fltr))
+ return;
+
+ if (!list_empty(&fltr->list_node)) {
+ list_del_init(&fltr->list_node);
+ bn->user_fltr_count--;
+ }
+}
+
+void bnge_del_l2_filter_rcu(struct bnge_net *bn, struct bnge_l2_filter *lfltr)
+{
+ struct bnge_filter_base *fltr = &lfltr->base;
+
+ hlist_del_rcu(&fltr->hlist);
+ bnge_del_usr_fltr_node(bn, fltr);
+ clear_bit(fltr->sw_id, bn->l2_fltr_bmap);
+ bn->l2_fltr_count--;
+ kfree_rcu(lfltr, base.rcu);
+}
+
+void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *lfltr)
+{
+ struct bnge_filter_base *fltr = &lfltr->base;
+
+ hlist_del(&fltr->hlist);
+ bnge_del_usr_fltr_node(bn, fltr);
+ clear_bit(fltr->sw_id, bn->l2_fltr_bmap);
+ bn->l2_fltr_count--;
+ kfree(fltr);
+}
+
+void bnge_del_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *nfltr)
+{
+ struct bnge_filter_base *fltr = &nfltr->base;
+
+ hlist_del(&fltr->hlist);
+ bnge_del_usr_fltr_node(bn, fltr);
+ clear_bit(fltr->sw_id, bn->ntp_fltr_bmap);
+ bn->ntp_fltr_count--;
+ kfree(fltr);
+}
+
+void bnge_del_ntp_filter_rcu(struct bnge_net *bn,
+ struct bnge_ntuple_filter *fltr)
+{
+ spin_lock_bh(&bn->ntp_fltr_lock);
hlist_del_rcu(&fltr->base.hlist);
+ bnge_del_usr_fltr_node(bn, &fltr->base);
+ clear_bit(fltr->base.sw_id, bn->ntp_fltr_bmap);
+ bn->ntp_fltr_count--;
+ spin_unlock_bh(&bn->ntp_fltr_lock);
+
kfree_rcu(fltr, base.rcu);
}
-static void bnge_init_l2_filter(struct bnge_net *bn,
- struct bnge_l2_filter *fltr,
- struct bnge_l2_key *key, u32 idx)
+static int bnge_init_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_filter *fltr,
+ struct bnge_l2_key *key, u32 idx)
{
struct hlist_head *head;
+ int bit_id;
ether_addr_copy(fltr->l2_key.dst_mac_addr, key->dst_mac_addr);
fltr->l2_key.vlan = key->vlan;
fltr->base.type = BNGE_FLTR_TYPE_L2;
+ fltr->base.filter_id = BNGE_FLTR_ID_INVALID;
+
+ bit_id = bitmap_find_free_region(bn->l2_fltr_bmap,
+ BNGE_MAX_L2_FLTRS, 0);
+ if (bit_id < 0)
+ return -ENOMEM;
+ fltr->base.sw_id = (u16)bit_id;
+
+ bn->l2_fltr_count++;
head = &bn->l2_fltr_hash_tbl[idx];
hlist_add_head_rcu(&fltr->base.hlist, head);
+ bnge_insert_usr_fltr(bn, &fltr->base);
+ return 0;
}
static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
@@ -49,12 +177,38 @@ static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
return NULL;
}
+__le64 bnge_lookup_l2_filter_rcu(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u32 idx)
+{
+ __le64 id = BNGE_FLTR_ID_INVALID;
+ struct bnge_l2_filter *fltr;
+ struct hlist_head *head;
+
+ rcu_read_lock();
+
+ head = &bn->l2_fltr_hash_tbl[idx];
+ hlist_for_each_entry_rcu(fltr, head, base.hlist) {
+ struct bnge_l2_key *l2_key = &fltr->l2_key;
+
+ if (ether_addr_equal(l2_key->dst_mac_addr, key->dst_mac_addr) &&
+ l2_key->vlan == key->vlan) {
+ id = fltr->base.filter_id;
+ break;
+ }
+ }
+
+ rcu_read_unlock();
+ return id;
+}
+
static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
struct bnge_l2_key *key,
gfp_t gfp)
{
struct bnge_l2_filter *fltr;
u32 idx;
+ int rc;
idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
BNGE_L2_FLTR_HASH_MASK;
@@ -67,10 +221,338 @@ static struct bnge_l2_filter *bnge_alloc_l2_filter(struct bnge_net *bn,
if (!fltr)
return ERR_PTR(-ENOMEM);
- bnge_init_l2_filter(bn, fltr, key, idx);
+ rc = bnge_init_l2_filter(bn, fltr, key, idx);
+ if (rc) {
+ kfree(fltr);
+ fltr = ERR_PTR(rc);
+ }
+
+ return fltr;
+}
+
+struct bnge_l2_filter *bnge_alloc_user_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u8 flags)
+{
+ struct bnge_l2_filter *fltr;
+ u32 idx;
+ int rc;
+
+ idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
+ BNGE_L2_FLTR_HASH_MASK;
+ fltr = bnge_lookup_l2_filter(bn, key, idx);
+ if (fltr) {
+ fltr = ERR_PTR(-EEXIST);
+ goto l2_filter_exit;
+ }
+ fltr = kzalloc_obj(*fltr, GFP_KERNEL);
+ if (!fltr) {
+ fltr = ERR_PTR(-ENOMEM);
+ goto l2_filter_exit;
+ }
+ fltr->base.flags = flags;
+ rc = bnge_init_l2_filter(bn, fltr, key, idx);
+ if (rc) {
+ kfree(fltr);
+ return ERR_PTR(rc);
+ }
+
+l2_filter_exit:
return fltr;
}
+void bnge_free_l2_filters(struct bnge_net *bn)
+{
+ int i;
+
+ netdev_assert_locked_or_invisible(bn->netdev);
+
+ for (i = 0; i < BNGE_L2_FLTR_HASH_SIZE; i++) {
+ struct bnge_l2_filter *fltr;
+ struct hlist_head *head;
+ struct hlist_node *tmp;
+
+ head = &bn->l2_fltr_hash_tbl[i];
+ hlist_for_each_entry_safe(fltr, tmp, head, base.hlist)
+ bnge_del_l2_filter(bn, fltr);
+ }
+
+ bitmap_free(bn->l2_fltr_bmap);
+ bn->l2_fltr_bmap = NULL;
+ bn->l2_fltr_count = 0;
+}
+
+void bnge_free_ntp_fltrs(struct bnge_net *bn, bool skip_user)
+{
+ int i;
+
+ netdev_assert_locked_or_invisible(bn->netdev);
+
+ /* Under netdev instance lock and all our NAPIs have been disabled.
+ * It's safe to delete the hash table.
+ */
+ for (i = 0; i < BNGE_NTP_FLTR_HASH_SIZE; i++) {
+ struct bnge_ntuple_filter *fltr;
+ struct hlist_head *head;
+ struct hlist_node *tmp;
+
+ head = &bn->ntp_fltr_hash_tbl[i];
+ hlist_for_each_entry_safe(fltr, tmp, head, base.hlist) {
+ if (skip_user && bnge_is_usr_fltr(&fltr->base))
+ continue;
+ bnge_del_ntp_filter(bn, fltr);
+ }
+ }
+
+ if (skip_user)
+ return;
+
+ bitmap_free(bn->ntp_fltr_bmap);
+ bn->ntp_fltr_bmap = NULL;
+ bn->ntp_fltr_count = 0;
+}
+
+static int bnge_alloc_l2_fltrs_mem(struct bnge_net *bn)
+{
+ int i, rc = 0;
+
+ if (bn->l2_fltr_bmap)
+ return 0;
+
+ for (i = 0; i < BNGE_L2_FLTR_HASH_SIZE; i++)
+ INIT_HLIST_HEAD(&bn->l2_fltr_hash_tbl[i]);
+
+ bn->l2_fltr_count = 0;
+ bn->l2_fltr_bmap = bitmap_zalloc(BNGE_MAX_L2_FLTRS, GFP_KERNEL);
+
+ if (!bn->l2_fltr_bmap)
+ rc = -ENOMEM;
+
+ return rc;
+}
+
+static int bnge_alloc_ntp_fltrs_mem(struct bnge_net *bn)
+{
+ struct bnge_dev *bd = bn->bd;
+ int i, rc = 0;
+
+ if (!(bn->priv_flags & BNGE_NET_EN_NTUPLE) || bn->ntp_fltr_bmap)
+ return 0;
+
+ for (i = 0; i < BNGE_NTP_FLTR_HASH_SIZE; i++)
+ INIT_HLIST_HEAD(&bn->ntp_fltr_hash_tbl[i]);
+
+ bn->ntp_fltr_count = 0;
+ bn->ntp_fltr_bmap = bitmap_zalloc(bd->max_fltr, GFP_KERNEL);
+
+ if (!bn->ntp_fltr_bmap)
+ rc = -ENOMEM;
+
+ return rc;
+}
+
+int bnge_fltrs_mem(struct bnge_net *bn)
+{
+ bool ntp_alloced = !bn->ntp_fltr_bmap;
+ int rc;
+
+ rc = bnge_alloc_ntp_fltrs_mem(bn);
+ if (rc)
+ return rc;
+
+ rc = bnge_alloc_l2_fltrs_mem(bn);
+ if (rc)
+ goto err_free_ntp_mem;
+
+ return 0;
+
+err_free_ntp_mem:
+ if (ntp_alloced) {
+ bitmap_free(bn->ntp_fltr_bmap);
+ bn->ntp_fltr_bmap = NULL;
+ }
+ return rc;
+}
+
+#define BNGE_IPV4_4TUPLE(bd, fkeys) \
+ (((fkeys)->basic.ip_proto == IPPROTO_TCP && \
+ (bd)->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV4) || \
+ ((fkeys)->basic.ip_proto == IPPROTO_UDP && \
+ (bd)->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV4))
+
+#define BNGE_IPV6_4TUPLE(bd, fkeys) \
+ (((fkeys)->basic.ip_proto == IPPROTO_TCP && \
+ (bd)->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_TCP_IPV6) || \
+ ((fkeys)->basic.ip_proto == IPPROTO_UDP && \
+ (bd)->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_UDP_IPV6))
+
+static u32 bnge_get_rss_flow_tuple_len(struct bnge_dev *bd,
+ struct flow_keys *fkeys)
+{
+ if (fkeys->basic.n_proto == htons(ETH_P_IP)) {
+ if (BNGE_IPV4_4TUPLE(bd, fkeys))
+ return sizeof(fkeys->addrs.v4addrs) +
+ sizeof(fkeys->ports);
+
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_IPV4)
+ return sizeof(fkeys->addrs.v4addrs);
+ }
+
+ if (fkeys->basic.n_proto == htons(ETH_P_IPV6)) {
+ if (BNGE_IPV6_4TUPLE(bd, fkeys))
+ return sizeof(fkeys->addrs.v6addrs) +
+ sizeof(fkeys->ports);
+
+ if (bd->rss_hash_cfg & VNIC_RSS_CFG_REQ_HASH_TYPE_IPV6)
+ return sizeof(fkeys->addrs.v6addrs);
+ }
+
+ return 0;
+}
+
+static u32 bnge_toeplitz(struct bnge_net *bn, struct flow_keys *fkeys,
+ const unsigned char *key)
+{
+ u64 prefix = bn->toeplitz_prefix, hash = 0;
+ struct bnge_ipv4_tuple tuple4;
+ struct bnge_ipv6_tuple tuple6;
+ struct bnge_dev *bd = bn->bd;
+ u8 *four_tuple;
+ int i, j, len;
+
+ len = bnge_get_rss_flow_tuple_len(bd, fkeys);
+ if (!len)
+ return 0;
+
+ if (fkeys->basic.n_proto == htons(ETH_P_IP)) {
+ tuple4.v4addrs = fkeys->addrs.v4addrs;
+ tuple4.ports = fkeys->ports;
+ four_tuple = (u8 *)&tuple4;
+ } else {
+ tuple6.v6addrs = fkeys->addrs.v6addrs;
+ tuple6.ports = fkeys->ports;
+ four_tuple = (u8 *)&tuple6;
+ }
+
+ for (i = 0, j = 8; i < len; i++, j++) {
+ u8 byte = four_tuple[i];
+ int bit;
+
+ for (bit = 0; bit < 8; bit++, prefix <<= 1, byte <<= 1) {
+ if (byte & 0x80)
+ hash ^= prefix;
+ }
+ prefix |= (j < HW_HASH_KEY_SIZE) ? key[j] : 0;
+ }
+
+ /* The valid part of the hash is in the upper 32 bits. */
+ return (hash >> 32) & BNGE_NTP_FLTR_HASH_MASK;
+}
+
+u32 bnge_get_ntp_filter_idx(struct bnge_net *bn, struct flow_keys *fkeys,
+ const struct sk_buff *skb)
+{
+ struct bnge_vnic_info *vnic;
+
+ if (skb)
+ return skb_get_hash_raw(skb) & BNGE_NTP_FLTR_HASH_MASK;
+
+ vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
+ return bnge_toeplitz(bn, fkeys, (void *)vnic->rss_hash_key);
+}
+
+static bool bnge_fltr_match(struct bnge_ntuple_filter *f1,
+ struct bnge_ntuple_filter *f2)
+{
+ struct bnge_flow_masks *masks1 = &f1->fmasks;
+ struct bnge_flow_masks *masks2 = &f2->fmasks;
+ struct flow_keys *keys1 = &f1->fkeys;
+ struct flow_keys *keys2 = &f2->fkeys;
+
+ if (keys1->basic.n_proto != keys2->basic.n_proto ||
+ keys1->basic.ip_proto != keys2->basic.ip_proto)
+ return false;
+
+ if (keys1->basic.n_proto == htons(ETH_P_IP)) {
+ if (keys1->addrs.v4addrs.src != keys2->addrs.v4addrs.src ||
+ masks1->addrs.v4addrs.src != masks2->addrs.v4addrs.src ||
+ keys1->addrs.v4addrs.dst != keys2->addrs.v4addrs.dst ||
+ masks1->addrs.v4addrs.dst != masks2->addrs.v4addrs.dst)
+ return false;
+ } else {
+ if (!ipv6_addr_equal(&keys1->addrs.v6addrs.src,
+ &keys2->addrs.v6addrs.src) ||
+ !ipv6_addr_equal(&masks1->addrs.v6addrs.src,
+ &masks2->addrs.v6addrs.src) ||
+ !ipv6_addr_equal(&keys1->addrs.v6addrs.dst,
+ &keys2->addrs.v6addrs.dst) ||
+ !ipv6_addr_equal(&masks1->addrs.v6addrs.dst,
+ &masks2->addrs.v6addrs.dst))
+ return false;
+ }
+
+ return keys1->ports.src == keys2->ports.src &&
+ masks1->ports.src == masks2->ports.src &&
+ keys1->ports.dst == keys2->ports.dst &&
+ masks1->ports.dst == masks2->ports.dst &&
+ keys1->control.flags == keys2->control.flags &&
+ f1->l2_filter_id == f2->l2_filter_id;
+}
+
+int bnge_insert_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *fltr,
+ u32 idx)
+{
+ struct bnge_ntuple_filter *f;
+ struct hlist_head *head;
+ int bit_id;
+
+ spin_lock_bh(&bn->ntp_fltr_lock);
+
+ head = &bn->ntp_fltr_hash_tbl[idx];
+ hlist_for_each_entry(f, head, base.hlist) {
+ if (bnge_fltr_match(f, fltr)) {
+ spin_unlock_bh(&bn->ntp_fltr_lock);
+ return -EEXIST;
+ }
+ }
+
+ bit_id = bitmap_find_free_region(bn->ntp_fltr_bmap,
+ bn->bd->max_fltr, 0);
+ if (bit_id < 0) {
+ spin_unlock_bh(&bn->ntp_fltr_lock);
+ return -ENOMEM;
+ }
+
+ fltr->base.sw_id = (u16)bit_id;
+ fltr->base.type = BNGE_FLTR_TYPE_NTUPLE;
+ fltr->base.flags |= BNGE_ACT_RING_DST;
+
+ head = &bn->ntp_fltr_hash_tbl[idx];
+ hlist_add_head_rcu(&fltr->base.hlist, head);
+
+ bnge_insert_usr_fltr(bn, &fltr->base);
+ bn->ntp_fltr_count++;
+
+ spin_unlock_bh(&bn->ntp_fltr_lock);
+
+ return 0;
+}
+
+struct bnge_ntuple_filter *
+bnge_lookup_ntp_filter_from_idx(struct bnge_net *bn,
+ struct bnge_ntuple_filter *fltr, u32 idx)
+{
+ struct bnge_ntuple_filter *f;
+ struct hlist_head *head;
+
+ head = &bn->ntp_fltr_hash_tbl[idx];
+ hlist_for_each_entry_rcu(f, head, base.hlist) {
+ if (bnge_fltr_match(f, fltr))
+ return f;
+ }
+ return NULL;
+}
+
int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
const u8 *mac_addr)
{
@@ -92,6 +574,6 @@ int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
return rc;
err_del_l2_filter:
- bnge_del_l2_filter(bn, fltr);
+ bnge_del_l2_filter_rcu(bn, fltr);
return rc;
}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
index a4a386202ac3..6cf3febe3372 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
@@ -12,7 +12,8 @@
struct bnge_net;
enum {
- BNGE_FLTR_TYPE_L2 = 1
+ BNGE_FLTR_TYPE_L2 = 1,
+ BNGE_FLTR_TYPE_NTUPLE = 2
};
enum {
@@ -20,12 +21,23 @@ enum {
BNGE_FLTR_FW_DELETED
};
+enum {
+ BNGE_ACT_DROP = 0x1,
+ BNGE_ACT_RING_DST = 0x2,
+ BNGE_ACT_NO_AGING = 0x4,
+ BNGE_ACT_RSS_CTX = 0x8
+};
+
+#define BNGE_MAX_L2_FLTRS 128
+#define BNGE_MAX_NTUPLE_FLTRS (8 << 10)
+
struct bnge_filter_base {
struct hlist_node hlist;
struct list_head list_node;
__le64 filter_id;
u8 type;
u8 flags;
+ u16 sw_id;
u16 rxq;
u16 fw_vnic_id;
u16 vf_idx;
@@ -51,7 +63,60 @@ struct bnge_l2_filter {
struct bnge_l2_key l2_key;
};
+struct bnge_ipv4_tuple {
+ struct flow_dissector_key_ipv4_addrs v4addrs;
+ struct flow_dissector_key_ports ports;
+};
+
+struct bnge_ipv6_tuple {
+ struct flow_dissector_key_ipv6_addrs v6addrs;
+ struct flow_dissector_key_ports ports;
+};
+
+struct bnge_flow_masks {
+ struct flow_dissector_key_ports ports;
+ struct flow_dissector_key_addrs addrs;
+};
+
+extern const struct bnge_flow_masks BNGE_FLOW_MASK_NONE;
+extern const struct bnge_flow_masks BNGE_FLOW_IPV6_MASK_ALL;
+extern const struct bnge_flow_masks BNGE_FLOW_IPV4_MASK_ALL;
+
+struct bnge_ntuple_filter {
+ /* base filter must be the first member */
+ struct bnge_filter_base base;
+ struct flow_keys fkeys;
+ struct bnge_flow_masks fmasks;
+ __le64 l2_filter_id;
+ u32 flow_id;
+};
+
+#define BNGE_L2_FLTR_IS_NTP_MAC(fltr) ((fltr)->base.flags & BNGE_ACT_RING_DST)
+
+#define BNGE_FLTR_ID_INVALID cpu_to_le64(0xffffffffffffffffULL)
+
+void bnge_free_ntp_fltrs(struct bnge_net *bn, bool skip_user_filters);
+u32 bnge_get_ntp_filter_idx(struct bnge_net *bn, struct flow_keys *fkeys,
+ const struct sk_buff *skb);
+int bnge_insert_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *fltr,
+ u32 idx);
+struct bnge_ntuple_filter *
+bnge_lookup_ntp_filter_from_idx(struct bnge_net *bn,
+ struct bnge_ntuple_filter *fltr, u32 idx);
void bnge_del_l2_filter(struct bnge_net *bn, struct bnge_l2_filter *fltr);
+void bnge_del_l2_filter_rcu(struct bnge_net *bn, struct bnge_l2_filter *fltr);
+__le64 bnge_lookup_l2_filter_rcu(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u32 idx);
+void bnge_free_l2_filters(struct bnge_net *bn);
int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
const u8 *mac_addr);
+struct bnge_l2_filter *bnge_alloc_user_l2_filter(struct bnge_net *bn,
+ struct bnge_l2_key *key,
+ u8 flags);
+int bnge_fltrs_mem(struct bnge_net *bn);
+void bnge_del_ntp_filter(struct bnge_net *bn,
+ struct bnge_ntuple_filter *nfltr);
+void bnge_del_ntp_filter_rcu(struct bnge_net *bn,
+ struct bnge_ntuple_filter *fltr);
#endif /* _BNGE_FILTER_H_ */
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
index 86edb70d323e..3eb67379c1a5 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
@@ -8,6 +8,7 @@
#include <linux/bnge/hsi.h>
#include <linux/if_vlan.h>
#include <net/netdev_queues.h>
+#include <linux/ethtool.h>
#include "bnge.h"
#include "bnge_hwrm.h"
@@ -932,11 +933,152 @@ int bnge_hwrm_l2_filter_alloc(struct bnge_dev *bd, struct bnge_l2_filter *fltr)
req->l2_ivlan_mask = cpu_to_le16(0xfff);
}
+ if (BNGE_L2_FLTR_IS_NTP_MAC(fltr)) {
+ req->enables |= cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX);
+ req->rfs_ring_tbl_idx = cpu_to_le16(fltr->base.rxq);
+ }
+
resp = bnge_hwrm_req_hold(bd, req);
rc = bnge_hwrm_req_send(bd, req);
- if (!rc)
+ if (!rc) {
fltr->base.filter_id = resp->l2_filter_id;
+ set_bit(BNGE_FLTR_VALID, &fltr->base.state);
+ }
+
+ bnge_hwrm_req_drop(bd, req);
+ return rc;
+}
+
+int bnge_hwrm_cfa_ntuple_filter_free(struct bnge_dev *bd,
+ struct bnge_ntuple_filter *fltr)
+{
+ struct hwrm_cfa_ntuple_filter_free_input *req;
+ int rc;
+
+ set_bit(BNGE_FLTR_FW_DELETED, &fltr->base.state);
+ if (!test_bit(BNGE_STATE_OPEN, &bd->state))
+ return 0;
+
+ rc = bnge_hwrm_req_init(bd, req, HWRM_CFA_NTUPLE_FILTER_FREE);
+ if (rc)
+ return rc;
+
+ req->ntuple_filter_id = fltr->base.filter_id;
+ return bnge_hwrm_req_send(bd, req);
+}
+
+#define BNGE_NTP_FLTR_FLAGS \
+ (CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_L2_FILTER_ID | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_ETHERTYPE | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IPADDR_TYPE | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_IPADDR_MASK | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_IPADDR_MASK | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_IP_PROTOCOL | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_SRC_PORT_MASK | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_PORT_MASK | \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_DST_ID)
+
+#define BNGE_NTP_TUNNEL_FLTR_FLAG \
+ CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_TUNNEL_TYPE
+
+static void
+bnge_cfg_rfs_ring_tbl_idx(struct bnge_dev *bd,
+ struct hwrm_cfa_ntuple_filter_alloc_input *req,
+ struct bnge_ntuple_filter *fltr)
+{
+ struct bnge_net *bn = netdev_priv(bd->netdev);
+ struct bnge_vnic_info *def_vnic;
+ u32 rxq = fltr->base.rxq;
+ u32 enables;
+
+ if (fltr->base.flags & BNGE_ACT_RSS_CTX) {
+ struct ethtool_rxfh_context *ctx;
+ struct bnge_rss_ctx *rss_ctx;
+ struct bnge_vnic_info *vnic;
+
+ /* The RSS context was already validated via xa_load() when the
+ * filter was added (bnge_ethtool.c) under netdev instance lock,
+ * so it is always present here.
+ */
+ ctx = xa_load(&bd->netdev->ethtool->rss_ctx,
+ fltr->base.fw_vnic_id);
+ rss_ctx = ethtool_rxfh_context_priv(ctx);
+ vnic = &rss_ctx->vnic;
+ req->dst_id = cpu_to_le16(vnic->fw_vnic_id);
+ return;
+ }
+
+ def_vnic = &bn->vnic_info[BNGE_VNIC_NTUPLE];
+ req->dst_id = cpu_to_le16(def_vnic->fw_vnic_id);
+ enables = CFA_NTUPLE_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX;
+ req->enables |= cpu_to_le32(enables);
+ req->rfs_ring_tbl_idx = cpu_to_le16(rxq);
+}
+
+int bnge_hwrm_cfa_ntuple_filter_alloc(struct bnge_dev *bd,
+ struct bnge_ntuple_filter *fltr)
+{
+ struct hwrm_cfa_ntuple_filter_alloc_output *resp;
+ struct hwrm_cfa_ntuple_filter_alloc_input *req;
+ struct bnge_flow_masks *masks = &fltr->fmasks;
+ struct flow_keys *keys = &fltr->fkeys;
+ int rc;
+ rc = bnge_hwrm_req_init(bd, req, HWRM_CFA_NTUPLE_FILTER_ALLOC);
+ if (rc)
+ return rc;
+
+ req->l2_filter_id = fltr->l2_filter_id;
+
+ if (fltr->base.flags & BNGE_ACT_DROP)
+ req->flags =
+ cpu_to_le32(CFA_NTUPLE_FILTER_ALLOC_REQ_FLAGS_DROP);
+ else
+ bnge_cfg_rfs_ring_tbl_idx(bd, req, fltr);
+
+ req->enables |= cpu_to_le32(BNGE_NTP_FLTR_FLAGS);
+
+ req->ethertype = htons(ETH_P_IP);
+ req->ip_addr_type = CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV4;
+ req->ip_protocol = keys->basic.ip_proto;
+
+ if (keys->basic.n_proto == htons(ETH_P_IPV6)) {
+ req->ethertype = htons(ETH_P_IPV6);
+ req->ip_addr_type =
+ CFA_NTUPLE_FILTER_ALLOC_REQ_IP_ADDR_TYPE_IPV6;
+ memcpy(&req->src_ipaddr, &keys->addrs.v6addrs.src,
+ sizeof(struct in6_addr));
+ memcpy(&req->src_ipaddr_mask, &masks->addrs.v6addrs.src,
+ sizeof(struct in6_addr));
+ memcpy(&req->dst_ipaddr, &keys->addrs.v6addrs.dst,
+ sizeof(struct in6_addr));
+ memcpy(&req->dst_ipaddr_mask, &masks->addrs.v6addrs.dst,
+ sizeof(struct in6_addr));
+ } else {
+ req->src_ipaddr[0] = keys->addrs.v4addrs.src;
+ req->src_ipaddr_mask[0] = masks->addrs.v4addrs.src;
+ req->dst_ipaddr[0] = keys->addrs.v4addrs.dst;
+ req->dst_ipaddr_mask[0] = masks->addrs.v4addrs.dst;
+ }
+ if (keys->control.flags & FLOW_DIS_ENCAPSULATION) {
+ req->enables |= cpu_to_le32(BNGE_NTP_TUNNEL_FLTR_FLAG);
+ req->tunnel_type =
+ CFA_NTUPLE_FILTER_ALLOC_REQ_TUNNEL_TYPE_ANYTUNNEL;
+ }
+
+ req->src_port = keys->ports.src;
+ req->src_port_mask = masks->ports.src;
+ req->dst_port = keys->ports.dst;
+ req->dst_port_mask = masks->ports.dst;
+
+ resp = bnge_hwrm_req_hold(bd, req);
+ rc = bnge_hwrm_req_send(bd, req);
+ if (!rc)
+ fltr->base.filter_id = resp->ntuple_filter_id;
bnge_hwrm_req_drop(bd, req);
return rc;
}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
index ae03041c36ac..081233924449 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.h
@@ -13,6 +13,8 @@ struct bnge_l2_filter;
#define BNGE_VNIC_CFG_ROCE_DUAL_MODE \
VNIC_CFG_REQ_FLAGS_ROCE_DUAL_VNIC_MODE
+struct bnge_ntuple_filter;
+
int bnge_hwrm_ver_get(struct bnge_dev *bd);
int bnge_hwrm_func_reset(struct bnge_dev *bd);
int bnge_hwrm_fw_set_time(struct bnge_dev *bd);
@@ -67,4 +69,8 @@ int bnge_hwrm_shutdown_link(struct bnge_dev *bd);
int bnge_hwrm_port_qstats(struct bnge_dev *bd, u8 flags);
int bnge_hwrm_port_qstats_ext(struct bnge_dev *bd, u8 flags);
int bnge_hwrm_func_qstat_ext(struct bnge_dev *bd, struct bnge_stats_mem *stats);
+int bnge_hwrm_cfa_ntuple_filter_alloc(struct bnge_dev *bd,
+ struct bnge_ntuple_filter *fltr);
+int bnge_hwrm_cfa_ntuple_filter_free(struct bnge_dev *bd,
+ struct bnge_ntuple_filter *fltr);
#endif /* _BNGE_HWRM_LIB_H_ */
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index 29a271f50e04..e59db7d7be16 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -1205,6 +1205,10 @@ static void bnge_free_core(struct bnge_net *bn)
bnge_free_ring_stats(bn);
bnge_free_ring_grps(bn);
bnge_free_vnics(bn);
+
+ /* Free non-user filters only */
+ bnge_free_ntp_fltrs(bn, true);
+
kfree(bn->tx_ring_map);
bn->tx_ring_map = NULL;
kfree(bn->tx_ring);
@@ -1293,6 +1297,10 @@ static int bnge_alloc_core(struct bnge_net *bn)
bnge_init_stats(bn);
+ rc = bnge_fltrs_mem(bn);
+ if (rc)
+ goto err_free_core;
+
rc = bnge_alloc_vnics(bn);
if (rc)
goto err_free_core;
@@ -2052,7 +2060,7 @@ static int bnge_cfg_rx_mode(struct bnge_net *bn, struct netdev_hw_addr_list *uc,
struct bnge_l2_filter *fltr = vnic->l2_filters[i];
bnge_hwrm_l2_filter_free(bd, fltr);
- bnge_del_l2_filter(bn, fltr);
+ bnge_del_l2_filter_rcu(bn, fltr);
}
vnic->uc_filter_count = 1;
@@ -2266,7 +2274,7 @@ static void bnge_hwrm_clear_vnic_filter(struct bnge_net *bn)
struct bnge_l2_filter *fltr = vnic->l2_filters[i];
bnge_hwrm_l2_filter_free(bn->bd, fltr);
- bnge_del_l2_filter(bn, fltr);
+ bnge_del_l2_filter_rcu(bn, fltr);
}
vnic->uc_filter_count = 0;
@@ -3432,6 +3440,9 @@ int bnge_netdev_alloc(struct bnge_dev *bd, int max_irqs)
goto err_free_workq;
spin_lock_init(&bn->stats_lock);
+ spin_lock_init(&bn->ntp_fltr_lock);
+ INIT_LIST_HEAD(&bn->usr_fltr_list);
+
netdev->request_ops_lock = true;
rc = register_netdev(netdev);
if (rc) {
@@ -3459,6 +3470,9 @@ void bnge_netdev_free(struct bnge_dev *bd)
unregister_netdev(netdev);
+ bnge_free_ntp_fltrs(bn, false);
+ bnge_free_l2_filters(bn);
+
timer_shutdown_sync(&bn->timer);
cancel_work_sync(&bn->sp_task);
bn->sp_event = 0;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index 3c0fea0a7608..221dfcd81e78 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -284,6 +284,9 @@ enum bnge_sp_event {
BNGE_PERIODIC_STATS_SP_EVENT,
};
+#define BNGE_NTP_FLTR_HASH_SIZE 512
+#define BNGE_NTP_FLTR_HASH_MASK (BNGE_NTP_FLTR_HASH_SIZE - 1)
+
struct bnge_net {
struct bnge_dev *bd;
struct net_device *netdev;
@@ -373,6 +376,16 @@ struct bnge_net {
bool pri2cos_valid;
u16 num_rss_ctx;
+
+ struct list_head usr_fltr_list;
+ unsigned long *ntp_fltr_bmap;
+ unsigned long *l2_fltr_bmap;
+ int ntp_fltr_count;
+ int l2_fltr_count;
+ int user_fltr_count;
+
+ struct hlist_head ntp_fltr_hash_tbl[BNGE_NTP_FLTR_HASH_SIZE];
+ spinlock_t ntp_fltr_lock; /* for hash table add, del */
};
#define BNGE_DEFAULT_RX_RING_SIZE 511
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
index 8f1b0f42773a..164f91bf470c 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_resc.c
@@ -12,6 +12,7 @@
#include "bnge_hwrm_lib.h"
#include "bnge_resc.h"
#include "bnge_vnic.h"
+#include "bnge_filter.h"
static u16 bnge_num_tx_to_cp(struct bnge_dev *bd, u16 tx)
{
@@ -648,7 +649,6 @@ bool bnge_arfs_capable(struct bnge_dev *bd, bool new_rss_ctx)
int bnge_net_init_dflt_config(struct bnge_dev *bd)
{
- struct bnge_hw_resc *hw_resc;
int rc;
rc = bnge_alloc_rss_indir_tbl(bd);
@@ -662,9 +662,7 @@ int bnge_net_init_dflt_config(struct bnge_dev *bd)
if (bnge_arfs_capable(bd, false))
bd->flags |= BNGE_EN_ARFS_CAP;
- hw_resc = &bd->hw_resc;
- bd->max_fltr = hw_resc->max_rx_em_flows + hw_resc->max_rx_wm_flows +
- BNGE_L2_FLTR_MAX_FLTR;
+ bd->max_fltr = BNGE_MAX_NTUPLE_FLTRS;
return 0;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
index 4ad676e983f2..7d89abf806e2 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_vnic.c
@@ -174,8 +174,13 @@ void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
bool all)
{
struct bnge_vnic_info *vnic = &rss_ctx->vnic;
+ struct bnge_filter_base *usr_fltr, *tmp;
+ struct bnge_ntuple_filter *ntp_fltr;
int i;
+ /* freeing the VNIC first is fine, the other HW resources
+ * can be freed independently.
+ */
bnge_hwrm_vnic_free_one(bn->bd, &rss_ctx->vnic);
for (i = 0; i < BNGE_MAX_CTX_PER_VNIC; i++) {
if (vnic->fw_rss_cos_lb_ctx[i] != INVALID_HW_RING_ID) {
@@ -188,6 +193,19 @@ void bnge_del_one_rss_ctx(struct bnge_net *bn, struct bnge_rss_ctx *rss_ctx,
if (!all)
return;
+ /* Delete all filters associated with this user RSS context */
+ list_for_each_entry_safe(usr_fltr, tmp, &bn->usr_fltr_list, list_node) {
+ if (usr_fltr->type == BNGE_FLTR_TYPE_NTUPLE &&
+ (usr_fltr->flags & BNGE_ACT_RSS_CTX) &&
+ usr_fltr->fw_vnic_id == rss_ctx->index) {
+ ntp_fltr = container_of(usr_fltr,
+ struct bnge_ntuple_filter,
+ base);
+ bnge_hwrm_cfa_ntuple_filter_free(bn->bd, ntp_fltr);
+ bnge_del_ntp_filter_rcu(bn, ntp_fltr);
+ }
+ }
+
if (vnic->rss_table)
dma_free_coherent(bn->bd->dev, vnic->rss_table_size,
vnic->rss_table,
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 07/10] bnge: add NTUPLE filter infrastructure
2026-09-28 6:13 ` [net-next, v4 07/10] bnge: add NTUPLE filter infrastructure Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found more severe
issue(s) in other patches of this series. This mail is only an FYI, in case
you want to address the nit picks below when you respin.
Low priority / nit picks for your consideration:
- [Low] Readers and writers access the __le64 filter_id of an RCU-
published L2 filter concurrently, with plain loads and stores and no…
--
bnge: add NTUPLE filter infrastructure
This commit adds the NTUPLE filter data structures, the hash table and
bitmap management, and HWRM wrappers that allocate and free ntuple filters.
It also adds an RCU lookup helper for L2 filter IDs, which later patches
use for kernel flow steering.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> index 3b26939b04578..7fbb67021b5d9 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
[ ... ]
> @@ -49,12 +177,38 @@ static struct bnge_l2_filter *bnge_lookup_l2_filter(struct bnge_net *bn,
> return NULL;
> }
>
> +__le64 bnge_lookup_l2_filter_rcu(struct bnge_net *bn,
> + struct bnge_l2_key *key,
> + u32 idx)
> +{
> + __le64 id = BNGE_FLTR_ID_INVALID;
> + struct bnge_l2_filter *fltr;
> + struct hlist_head *head;
> +
> + rcu_read_lock();
> +
> + head = &bn->l2_fltr_hash_tbl[idx];
> + hlist_for_each_entry_rcu(fltr, head, base.hlist) {
> + struct bnge_l2_key *l2_key = &fltr->l2_key;
> +
> + if (ether_addr_equal(l2_key->dst_mac_addr, key->dst_mac_addr) &&
> + l2_key->vlan == key->vlan) {
> + id = fltr->base.filter_id;
^^^^^^^^^
[Severity: Low]
Can this plain load race with the store in bnge_hwrm_l2_filter_alloc()?
bnge_init_l2_filter() publishes the filter with hlist_add_head_rcu() while
filter_id is still BNGE_FLTR_ID_INVALID. The real ID is written later,
after the HWRM round trip (which can sleep). That write is a plain store,
and only the netdev instance lock is held:
bnge_hwrm_l2_filter_alloc() {
...
if (!rc) {
fltr->base.filter_id = resp->l2_filter_id;
set_bit(BNGE_FLTR_VALID, &fltr->base.state);
}
...
}
In the later patch "bnge: add aRFS flow steering ndo support", this helper
is called from bnge_rx_flow_steer(). That function runs in RX softirq and
holds neither the netdev lock nor ntp_fltr_lock.
The writer can run at the same time. bnge_cfg_rx_mode() calls
bnge_hwrm_set_vnic_filter() for secondary unicast MACs while the device
is open.
KCSAN should report this as a data race. BNGE can also be built for
32-bit, where the 64-bit load can tear. A value that is half INVALID and
half the real ID would pass this check in bnge_rx_flow_steer():
filter_id = bnge_lookup_l2_filter_from_key(bn, &key);
if (filter_id == BNGE_FLTR_ID_INVALID) {
The torn value would then be sent to firmware as req->l2_filter_id.
The aRFS commit message already says that a stale ID is either rejected
by firmware or aged out through rps_may_expire_flow(). A torn value would
end up in one of those two outcomes, so the effect is limited.
Would it make sense to use READ_ONCE() here, paired with WRITE_ONCE() in
bnge_hwrm_l2_filter_alloc()? The series already pairs them this way for
ntp_fltr_count.
> + break;
> + }
> + }
[ ... ]
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 08/10] bnge: add NTUPLE filter support in ethtool
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (6 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 07/10] bnge: add NTUPLE filter infrastructure Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 09/10] bnge: add aRFS flow steering ndo support Vikas Gupta
2026-09-28 6:13 ` [net-next, v4 10/10] bnge: add cpu_rmap support for IRQ affinity Vikas Gupta
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Introduce ethtool rx flow classification (rxnfc) operations for
NTUPLE filter management.
Since with ethtool support user filters creation is supported, when
interface is down clearing of filters which are in software remains
intact.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Bhargava Chenna Marreddy <bhargava.marreddy@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_ethtool.c | 591 ++++++++++++++++++
.../net/ethernet/broadcom/bnge/bnge_filter.c | 60 ++
.../net/ethernet/broadcom/bnge/bnge_filter.h | 2 +
.../ethernet/broadcom/bnge/bnge_hwrm_lib.c | 8 +-
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 4 +
5 files changed, 662 insertions(+), 3 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
index 85dbe64d4c12..8e9cfbad98e0 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
@@ -15,6 +15,7 @@
#include "bnge_resc.h"
#include "bnge_ethtool.h"
#include "bnge_hwrm_lib.h"
+#include "bnge_filter.h"
static int bnge_nway_reset(struct net_device *dev)
{
@@ -1169,6 +1170,594 @@ static int bnge_remove_rxfh_context(struct net_device *dev,
return 0;
}
+#define BNGE_IP_PROTO_FULL_MASK 0xFF
+#define BNGE_IP_PROTO_WILDCARD 0x0
+
+static u32 bnge_get_all_fltr_ids_rcu(struct bnge_net *bn,
+ struct hlist_head tbl[],
+ u32 tbl_size, u32 *ids, u32 start,
+ u32 id_cnt, u32 offset)
+{
+ u32 i, j = start;
+
+ if (j >= id_cnt)
+ return j;
+
+ for (i = 0; i < tbl_size; i++) {
+ struct bnge_filter_base *fltr;
+ struct hlist_head *head;
+
+ head = &tbl[i];
+ hlist_for_each_entry_rcu(fltr, head, hlist) {
+ if (!fltr->flags ||
+ test_bit(BNGE_FLTR_FW_DELETED, &fltr->state))
+ continue;
+ ids[j++] = fltr->sw_id + offset;
+ if (j == id_cnt)
+ return j;
+ }
+ }
+ return j;
+}
+
+static struct bnge_filter_base *bnge_get_one_fltr_rcu(struct bnge_net *bn,
+ struct hlist_head tbl[],
+ u32 tbl_size, u32 id,
+ u32 offset)
+{
+ u32 i;
+
+ for (i = 0; i < tbl_size; i++) {
+ struct bnge_filter_base *fltr;
+ struct hlist_head *head;
+
+ head = &tbl[i];
+ hlist_for_each_entry_rcu(fltr, head, hlist) {
+ if (fltr->flags && fltr->sw_id + offset == id)
+ return fltr;
+ }
+ }
+ return NULL;
+}
+
+static int bnge_grxclsrlall(struct bnge_net *bn, struct ethtool_rxnfc *cmd,
+ u32 *rule_locs)
+{
+ u32 count;
+
+ cmd->data = bn->user_fltr_count;
+ rcu_read_lock();
+ count = bnge_get_all_fltr_ids_rcu(bn, bn->l2_fltr_hash_tbl,
+ BNGE_L2_FLTR_HASH_SIZE, rule_locs, 0,
+ cmd->rule_cnt, 0);
+ cmd->rule_cnt = bnge_get_all_fltr_ids_rcu(bn, bn->ntp_fltr_hash_tbl,
+ BNGE_NTP_FLTR_HASH_SIZE,
+ rule_locs, count,
+ cmd->rule_cnt,
+ BNGE_MAX_L2_FLTRS);
+ rcu_read_unlock();
+
+ return 0;
+}
+
+static int bnge_grxclsrule(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
+{
+ struct ethtool_rx_flow_spec *fs =
+ (struct ethtool_rx_flow_spec *)&cmd->fs;
+ struct bnge_filter_base *fltr_base;
+ struct bnge_ntuple_filter *fltr;
+ struct bnge_flow_masks *fmasks;
+ struct bnge_dev *bd = bn->bd;
+ struct flow_keys *fkeys;
+ int rc = -EINVAL;
+
+ if (fs->location >= BNGE_MAX_L2_FLTRS + bd->max_fltr)
+ return rc;
+
+ rcu_read_lock();
+ fltr_base = bnge_get_one_fltr_rcu(bn, bn->l2_fltr_hash_tbl,
+ BNGE_L2_FLTR_HASH_SIZE,
+ fs->location, 0);
+ if (fltr_base) {
+ struct ethhdr *h_ether = &fs->h_u.ether_spec;
+ struct ethhdr *m_ether = &fs->m_u.ether_spec;
+ struct bnge_l2_filter *l2_fltr;
+ struct bnge_l2_key *l2_key;
+
+ l2_fltr = container_of(fltr_base, struct bnge_l2_filter, base);
+ l2_key = &l2_fltr->l2_key;
+ fs->flow_type = ETHER_FLOW;
+ ether_addr_copy(h_ether->h_dest, l2_key->dst_mac_addr);
+ eth_broadcast_addr(m_ether->h_dest);
+ if (l2_key->vlan) {
+ struct ethtool_flow_ext *m_ext = &fs->m_ext;
+ struct ethtool_flow_ext *h_ext = &fs->h_ext;
+
+ fs->flow_type |= FLOW_EXT;
+ m_ext->vlan_tci = htons(0xfff);
+ h_ext->vlan_tci = htons(l2_key->vlan);
+ }
+ if (fltr_base->flags & BNGE_ACT_RING_DST)
+ fs->ring_cookie = fltr_base->rxq;
+ rcu_read_unlock();
+ return 0;
+ }
+ fltr_base = bnge_get_one_fltr_rcu(bn, bn->ntp_fltr_hash_tbl,
+ BNGE_NTP_FLTR_HASH_SIZE,
+ fs->location, BNGE_MAX_L2_FLTRS);
+ if (!fltr_base) {
+ rcu_read_unlock();
+ return rc;
+ }
+ fltr = container_of(fltr_base, struct bnge_ntuple_filter, base);
+
+ fkeys = &fltr->fkeys;
+ fmasks = &fltr->fmasks;
+ if (fkeys->basic.n_proto == htons(ETH_P_IP)) {
+ if (fkeys->basic.ip_proto == BNGE_IP_PROTO_WILDCARD) {
+ fs->flow_type = IP_USER_FLOW;
+ fs->h_u.usr_ip4_spec.ip_ver = ETH_RX_NFC_IP4;
+ fs->h_u.usr_ip4_spec.proto = BNGE_IP_PROTO_WILDCARD;
+ fs->m_u.usr_ip4_spec.proto = 0;
+ } else if (fkeys->basic.ip_proto == IPPROTO_ICMP) {
+ fs->flow_type = IP_USER_FLOW;
+ fs->h_u.usr_ip4_spec.ip_ver = ETH_RX_NFC_IP4;
+ fs->h_u.usr_ip4_spec.proto = IPPROTO_ICMP;
+ fs->m_u.usr_ip4_spec.proto = BNGE_IP_PROTO_FULL_MASK;
+ } else if (fkeys->basic.ip_proto == IPPROTO_TCP) {
+ fs->flow_type = TCP_V4_FLOW;
+ } else if (fkeys->basic.ip_proto == IPPROTO_UDP) {
+ fs->flow_type = UDP_V4_FLOW;
+ } else {
+ goto fltr_err;
+ }
+
+ fs->h_u.tcp_ip4_spec.ip4src = fkeys->addrs.v4addrs.src;
+ fs->m_u.tcp_ip4_spec.ip4src = fmasks->addrs.v4addrs.src;
+ fs->h_u.tcp_ip4_spec.ip4dst = fkeys->addrs.v4addrs.dst;
+ fs->m_u.tcp_ip4_spec.ip4dst = fmasks->addrs.v4addrs.dst;
+ if (fs->flow_type == TCP_V4_FLOW ||
+ fs->flow_type == UDP_V4_FLOW) {
+ fs->h_u.tcp_ip4_spec.psrc = fkeys->ports.src;
+ fs->m_u.tcp_ip4_spec.psrc = fmasks->ports.src;
+ fs->h_u.tcp_ip4_spec.pdst = fkeys->ports.dst;
+ fs->m_u.tcp_ip4_spec.pdst = fmasks->ports.dst;
+ }
+ } else {
+ if (fkeys->basic.ip_proto == BNGE_IP_PROTO_WILDCARD) {
+ fs->flow_type = IPV6_USER_FLOW;
+ fs->h_u.usr_ip6_spec.l4_proto =
+ BNGE_IP_PROTO_WILDCARD;
+ fs->m_u.usr_ip6_spec.l4_proto = 0;
+ } else if (fkeys->basic.ip_proto == IPPROTO_ICMPV6) {
+ fs->flow_type = IPV6_USER_FLOW;
+ fs->h_u.usr_ip6_spec.l4_proto = IPPROTO_ICMPV6;
+ fs->m_u.usr_ip6_spec.l4_proto =
+ BNGE_IP_PROTO_FULL_MASK;
+ } else if (fkeys->basic.ip_proto == IPPROTO_TCP) {
+ fs->flow_type = TCP_V6_FLOW;
+ } else if (fkeys->basic.ip_proto == IPPROTO_UDP) {
+ fs->flow_type = UDP_V6_FLOW;
+ } else {
+ goto fltr_err;
+ }
+
+ memcpy(fs->h_u.tcp_ip6_spec.ip6src,
+ &fkeys->addrs.v6addrs.src,
+ sizeof(struct in6_addr));
+ memcpy(fs->m_u.tcp_ip6_spec.ip6src,
+ &fmasks->addrs.v6addrs.src,
+ sizeof(struct in6_addr));
+ memcpy(fs->h_u.tcp_ip6_spec.ip6dst,
+ &fkeys->addrs.v6addrs.dst,
+ sizeof(struct in6_addr));
+ memcpy(fs->m_u.tcp_ip6_spec.ip6dst,
+ &fmasks->addrs.v6addrs.dst,
+ sizeof(struct in6_addr));
+
+ if (fs->flow_type == TCP_V6_FLOW ||
+ fs->flow_type == UDP_V6_FLOW) {
+ fs->h_u.tcp_ip6_spec.psrc = fkeys->ports.src;
+ fs->m_u.tcp_ip6_spec.psrc = fmasks->ports.src;
+ fs->h_u.tcp_ip6_spec.pdst = fkeys->ports.dst;
+ fs->m_u.tcp_ip6_spec.pdst = fmasks->ports.dst;
+ }
+ }
+
+ if (fltr->base.flags & BNGE_ACT_DROP) {
+ fs->ring_cookie = RX_CLS_FLOW_DISC;
+ } else if (fltr->base.flags & BNGE_ACT_RSS_CTX) {
+ fs->flow_type |= FLOW_RSS;
+ cmd->rss_context = fltr->base.fw_vnic_id;
+ } else {
+ fs->ring_cookie = fltr->base.rxq;
+ }
+ rc = 0;
+
+fltr_err:
+ rcu_read_unlock();
+
+ return rc;
+}
+
+static bool bnge_verify_ntuple_ip4_flow(struct ethtool_usrip4_spec *ip_spec,
+ struct ethtool_usrip4_spec *ip_mask)
+{
+ u8 mproto = ip_mask->proto;
+ u8 sproto = ip_spec->proto;
+
+ if (ip_mask->l4_4_bytes || ip_mask->tos ||
+ ip_spec->ip_ver != ETH_RX_NFC_IP4 ||
+ (mproto && (mproto != BNGE_IP_PROTO_FULL_MASK ||
+ sproto != IPPROTO_ICMP)))
+ return false;
+ return true;
+}
+
+static bool bnge_verify_ntuple_ip6_flow(struct ethtool_usrip6_spec *ip_spec,
+ struct ethtool_usrip6_spec *ip_mask)
+{
+ u8 mproto = ip_mask->l4_proto;
+ u8 sproto = ip_spec->l4_proto;
+
+ if (ip_mask->l4_4_bytes || ip_mask->tclass ||
+ (mproto && (mproto != BNGE_IP_PROTO_FULL_MASK ||
+ sproto != IPPROTO_ICMPV6)))
+ return false;
+ return true;
+}
+
+static int bnge_add_ntuple_cls_rule(struct bnge_net *bn,
+ struct ethtool_rxnfc *cmd)
+{
+ struct ethtool_rx_flow_spec *fs = &cmd->fs;
+ struct bnge_ntuple_filter *new_fltr, *fltr;
+ u32 flow_type = fs->flow_type & 0xff;
+ struct bnge_l2_filter *l2_fltr;
+ struct bnge_flow_masks *fmasks;
+ struct flow_keys *fkeys;
+ u32 idx;
+ int rc;
+
+ if (!bn->vnic_info)
+ return -EAGAIN;
+
+ if (fs->flow_type & (FLOW_MAC_EXT | FLOW_EXT))
+ return -EOPNOTSUPP;
+
+ if (fs->ring_cookie != RX_CLS_FLOW_DISC &&
+ ethtool_get_flow_spec_ring_vf(fs->ring_cookie))
+ return -EOPNOTSUPP;
+
+ if (flow_type == IP_USER_FLOW) {
+ if (!bnge_verify_ntuple_ip4_flow(&fs->h_u.usr_ip4_spec,
+ &fs->m_u.usr_ip4_spec))
+ return -EOPNOTSUPP;
+ }
+
+ if (flow_type == IPV6_USER_FLOW) {
+ if (!bnge_verify_ntuple_ip6_flow(&fs->h_u.usr_ip6_spec,
+ &fs->m_u.usr_ip6_spec))
+ return -EOPNOTSUPP;
+ }
+
+ new_fltr = kzalloc_obj(*new_fltr, GFP_KERNEL);
+ if (!new_fltr)
+ return -ENOMEM;
+
+ l2_fltr = bn->vnic_info[BNGE_VNIC_DEFAULT].l2_filters[0];
+ new_fltr->l2_filter_id = l2_fltr->base.filter_id;
+ fmasks = &new_fltr->fmasks;
+ fkeys = &new_fltr->fkeys;
+
+ rc = -EOPNOTSUPP;
+ switch (flow_type) {
+ case IP_USER_FLOW: {
+ struct ethtool_usrip4_spec *ip_spec = &fs->h_u.usr_ip4_spec;
+ struct ethtool_usrip4_spec *ip_mask = &fs->m_u.usr_ip4_spec;
+
+ fkeys->basic.ip_proto = ip_mask->proto ? ip_spec->proto
+ : BNGE_IP_PROTO_WILDCARD;
+ fkeys->basic.n_proto = htons(ETH_P_IP);
+ fkeys->addrs.v4addrs.src = ip_spec->ip4src;
+ fmasks->addrs.v4addrs.src = ip_mask->ip4src;
+ fkeys->addrs.v4addrs.dst = ip_spec->ip4dst;
+ fmasks->addrs.v4addrs.dst = ip_mask->ip4dst;
+ break;
+ }
+ case TCP_V4_FLOW:
+ case UDP_V4_FLOW: {
+ struct ethtool_tcpip4_spec *ip_spec = &fs->h_u.tcp_ip4_spec;
+ struct ethtool_tcpip4_spec *ip_mask = &fs->m_u.tcp_ip4_spec;
+
+ if (ip_mask->tos)
+ goto err_free_fltr;
+
+ fkeys->basic.ip_proto = IPPROTO_TCP;
+ if (flow_type == UDP_V4_FLOW)
+ fkeys->basic.ip_proto = IPPROTO_UDP;
+ fkeys->basic.n_proto = htons(ETH_P_IP);
+ fkeys->addrs.v4addrs.src = ip_spec->ip4src;
+ fmasks->addrs.v4addrs.src = ip_mask->ip4src;
+ fkeys->addrs.v4addrs.dst = ip_spec->ip4dst;
+ fmasks->addrs.v4addrs.dst = ip_mask->ip4dst;
+ fkeys->ports.src = ip_spec->psrc;
+ fmasks->ports.src = ip_mask->psrc;
+ fkeys->ports.dst = ip_spec->pdst;
+ fmasks->ports.dst = ip_mask->pdst;
+ break;
+ }
+ case IPV6_USER_FLOW: {
+ struct ethtool_usrip6_spec *ip_spec = &fs->h_u.usr_ip6_spec;
+ struct ethtool_usrip6_spec *ip_mask = &fs->m_u.usr_ip6_spec;
+
+ fkeys->basic.ip_proto = ip_mask->l4_proto ? ip_spec->l4_proto
+ : BNGE_IP_PROTO_WILDCARD;
+ fkeys->basic.n_proto = htons(ETH_P_IPV6);
+
+ memcpy(&fkeys->addrs.v6addrs.src, ip_spec->ip6src,
+ sizeof(struct in6_addr));
+ memcpy(&fmasks->addrs.v6addrs.src, ip_mask->ip6src,
+ sizeof(struct in6_addr));
+ memcpy(&fkeys->addrs.v6addrs.dst, ip_spec->ip6dst,
+ sizeof(struct in6_addr));
+ memcpy(&fmasks->addrs.v6addrs.dst, ip_mask->ip6dst,
+ sizeof(struct in6_addr));
+ break;
+ }
+ case TCP_V6_FLOW:
+ case UDP_V6_FLOW: {
+ struct ethtool_tcpip6_spec *ip_spec = &fs->h_u.tcp_ip6_spec;
+ struct ethtool_tcpip6_spec *ip_mask = &fs->m_u.tcp_ip6_spec;
+
+ if (ip_mask->tclass)
+ goto err_free_fltr;
+
+ fkeys->basic.ip_proto = IPPROTO_TCP;
+ if (flow_type == UDP_V6_FLOW)
+ fkeys->basic.ip_proto = IPPROTO_UDP;
+ fkeys->basic.n_proto = htons(ETH_P_IPV6);
+
+ memcpy(&fkeys->addrs.v6addrs.src, ip_spec->ip6src,
+ sizeof(struct in6_addr));
+ memcpy(&fmasks->addrs.v6addrs.src, ip_mask->ip6src,
+ sizeof(struct in6_addr));
+ memcpy(&fkeys->addrs.v6addrs.dst, ip_spec->ip6dst,
+ sizeof(struct in6_addr));
+ memcpy(&fmasks->addrs.v6addrs.dst, ip_mask->ip6dst,
+ sizeof(struct in6_addr));
+
+ fkeys->ports.src = ip_spec->psrc;
+ fmasks->ports.src = ip_mask->psrc;
+ fkeys->ports.dst = ip_spec->pdst;
+ fmasks->ports.dst = ip_mask->pdst;
+ break;
+ }
+ default:
+ rc = -EOPNOTSUPP;
+ goto err_free_fltr;
+ }
+ if (!memcmp(&BNGE_FLOW_MASK_NONE, fmasks, sizeof(*fmasks)))
+ goto err_free_fltr;
+
+ idx = bnge_get_ntp_filter_idx(bn, fkeys, NULL);
+ rcu_read_lock();
+ fltr = bnge_lookup_ntp_filter_from_idx(bn, new_fltr, idx);
+ if (fltr) {
+ rcu_read_unlock();
+ rc = -EEXIST;
+ goto err_free_fltr;
+ }
+ rcu_read_unlock();
+
+ new_fltr->base.flags = BNGE_ACT_NO_AGING;
+ if (fs->flow_type & FLOW_RSS) {
+ struct bnge_rss_ctx *rss_ctx;
+
+ new_fltr->base.fw_vnic_id = 0;
+ new_fltr->base.flags |= BNGE_ACT_RSS_CTX;
+ rss_ctx = bnge_get_rss_ctx_from_index(bn, cmd->rss_context);
+ if (rss_ctx) {
+ new_fltr->base.fw_vnic_id = rss_ctx->index;
+ } else {
+ rc = -EINVAL;
+ goto err_free_fltr;
+ }
+ }
+ if (fs->ring_cookie == RX_CLS_FLOW_DISC)
+ new_fltr->base.flags |= BNGE_ACT_DROP;
+ else
+ new_fltr->base.rxq = ethtool_get_flow_spec_ring(fs->ring_cookie);
+ __set_bit(BNGE_FLTR_VALID, &new_fltr->base.state);
+ rc = bnge_insert_ntp_filter(bn, new_fltr, idx);
+ if (!rc) {
+ rc = bnge_hwrm_cfa_ntuple_filter_alloc(bn->bd, new_fltr);
+ if (rc) {
+ bnge_del_ntp_filter_rcu(bn, new_fltr);
+ return rc;
+ }
+ fs->location = BNGE_MAX_L2_FLTRS + new_fltr->base.sw_id;
+ return 0;
+ }
+
+err_free_fltr:
+ kfree(new_fltr);
+ return rc;
+}
+
+static int bnge_add_l2_cls_rule(struct bnge_net *bn,
+ struct ethtool_rx_flow_spec *fs)
+{
+ u32 ring = ethtool_get_flow_spec_ring(fs->ring_cookie);
+ struct ethhdr *h_ether = &fs->h_u.ether_spec;
+ struct ethhdr *m_ether = &fs->m_u.ether_spec;
+ struct bnge_l2_filter *fltr;
+ struct bnge_l2_key key;
+ u16 vnic_id;
+ u8 flags;
+ int rc;
+
+ if (ethtool_get_flow_spec_ring_vf(fs->ring_cookie))
+ return -EOPNOTSUPP;
+
+ if (!is_broadcast_ether_addr(m_ether->h_dest))
+ return -EINVAL;
+
+ if (is_broadcast_ether_addr(h_ether->h_dest) ||
+ is_multicast_ether_addr(h_ether->h_dest))
+ return -EINVAL;
+
+ ether_addr_copy(key.dst_mac_addr, h_ether->h_dest);
+ key.vlan = 0;
+ if (fs->flow_type & FLOW_EXT) {
+ struct ethtool_flow_ext *m_ext = &fs->m_ext;
+ struct ethtool_flow_ext *h_ext = &fs->h_ext;
+
+ if (m_ext->vlan_tci != htons(0xfff) || !h_ext->vlan_tci)
+ return -EINVAL;
+ key.vlan = ntohs(h_ext->vlan_tci);
+ }
+
+ flags = BNGE_ACT_RING_DST;
+ vnic_id = bn->vnic_info[BNGE_VNIC_DEFAULT].fw_vnic_id;
+
+ fltr = bnge_alloc_user_l2_filter(bn, &key, flags);
+ if (IS_ERR(fltr))
+ return PTR_ERR(fltr);
+
+ fltr->base.fw_vnic_id = vnic_id;
+ fltr->base.rxq = ring;
+ rc = bnge_hwrm_l2_filter_alloc(bn->bd, fltr);
+ if (rc)
+ bnge_del_l2_filter_rcu(bn, fltr);
+ else
+ fs->location = fltr->base.sw_id;
+ return rc;
+}
+
+static int bnge_srxclsrlins(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
+{
+ struct ethtool_rx_flow_spec *fs = &cmd->fs;
+ struct bnge_dev *bd = bn->bd;
+ u32 ring, flow_type;
+ int rc;
+
+ if (!netif_running(bn->netdev))
+ return -EAGAIN;
+ if (!(bn->priv_flags & BNGE_NET_EN_NTUPLE))
+ return -EPERM;
+ if (fs->location != RX_CLS_LOC_ANY)
+ return -EINVAL;
+
+ flow_type = fs->flow_type;
+
+ if (flow_type & FLOW_MAC_EXT)
+ return -EINVAL;
+
+ flow_type &= ~FLOW_EXT;
+
+ if (fs->ring_cookie == RX_CLS_FLOW_DISC && flow_type != ETHER_FLOW)
+ return bnge_add_ntuple_cls_rule(bn, cmd);
+
+ ring = ethtool_get_flow_spec_ring(fs->ring_cookie);
+ if (ring >= bd->rx_nr_rings)
+ return -EINVAL;
+
+ if (flow_type == ETHER_FLOW)
+ rc = bnge_add_l2_cls_rule(bn, fs);
+ else
+ rc = bnge_add_ntuple_cls_rule(bn, cmd);
+ return rc;
+}
+
+static int bnge_srxclsrldel(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
+{
+ struct ethtool_rx_flow_spec *fs = &cmd->fs;
+ struct bnge_filter_base *fltr_base;
+ struct bnge_ntuple_filter *fltr;
+ u32 id = fs->location;
+
+ rcu_read_lock();
+ fltr_base = bnge_get_one_fltr_rcu(bn, bn->l2_fltr_hash_tbl,
+ BNGE_L2_FLTR_HASH_SIZE, id, 0);
+ if (fltr_base) {
+ struct bnge_l2_filter *l2_fltr;
+
+ l2_fltr = container_of(fltr_base, struct bnge_l2_filter, base);
+ rcu_read_unlock();
+ bnge_hwrm_l2_filter_free(bn->bd, l2_fltr);
+ bnge_del_l2_filter_rcu(bn, l2_fltr);
+ return 0;
+ }
+ fltr_base = bnge_get_one_fltr_rcu(bn, bn->ntp_fltr_hash_tbl,
+ BNGE_NTP_FLTR_HASH_SIZE, id,
+ BNGE_MAX_L2_FLTRS);
+ if (!fltr_base) {
+ rcu_read_unlock();
+ return -ENOENT;
+ }
+
+ fltr = container_of(fltr_base, struct bnge_ntuple_filter, base);
+ if (!(fltr->base.flags & BNGE_ACT_NO_AGING)) {
+ rcu_read_unlock();
+ return -EINVAL;
+ }
+ rcu_read_unlock();
+ bnge_hwrm_cfa_ntuple_filter_free(bn->bd, fltr);
+ bnge_del_ntp_filter_rcu(bn, fltr);
+ return 0;
+}
+
+static int bnge_get_rxnfc(struct net_device *dev, struct ethtool_rxnfc *cmd,
+ u32 *rule_locs)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ struct bnge_dev *bd = bn->bd;
+ int rc = 0;
+
+ switch (cmd->cmd) {
+ case ETHTOOL_GRXCLSRLCNT:
+ cmd->rule_cnt = bn->user_fltr_count;
+ cmd->data = bd->max_fltr | RX_CLS_LOC_SPECIAL;
+ break;
+
+ case ETHTOOL_GRXCLSRLALL:
+ rc = bnge_grxclsrlall(bn, cmd, (u32 *)rule_locs);
+ break;
+
+ case ETHTOOL_GRXCLSRULE:
+ rc = bnge_grxclsrule(bn, cmd);
+ break;
+
+ default:
+ rc = -EOPNOTSUPP;
+ break;
+ }
+
+ return rc;
+}
+
+static int bnge_set_rxnfc(struct net_device *dev, struct ethtool_rxnfc *cmd)
+{
+ struct bnge_net *bn = netdev_priv(dev);
+ int rc;
+
+ switch (cmd->cmd) {
+ case ETHTOOL_SRXCLSRLINS:
+ rc = bnge_srxclsrlins(bn, cmd);
+ break;
+
+ case ETHTOOL_SRXCLSRLDEL:
+ rc = bnge_srxclsrldel(bn, cmd);
+ break;
+
+ default:
+ rc = -EOPNOTSUPP;
+ break;
+ }
+ return rc;
+}
+
static const struct ethtool_ops bnge_ethtool_ops = {
.cap_link_lanes_supported = 1,
.get_link_ksettings = bnge_get_link_ksettings,
@@ -1201,6 +1790,8 @@ static const struct ethtool_ops bnge_ethtool_ops = {
.create_rxfh_context = bnge_create_rxfh_context,
.modify_rxfh_context = bnge_modify_rxfh_context,
.remove_rxfh_context = bnge_remove_rxfh_context,
+ .get_rxnfc = bnge_get_rxnfc,
+ .set_rxnfc = bnge_set_rxnfc,
};
void bnge_set_ethtool_ops(struct net_device *dev)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
index 7fbb67021b5d..764c967c6457 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
@@ -577,3 +577,63 @@ int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
bnge_del_l2_filter_rcu(bn, fltr);
return rc;
}
+
+static void bnge_cfg_one_usr_fltr(struct bnge_net *bn,
+ struct bnge_filter_base *fltr)
+{
+ struct bnge_ntuple_filter *ntp_fltr;
+ struct bnge_l2_filter *l2_fltr;
+
+ if (list_empty(&fltr->list_node))
+ return;
+
+ if (fltr->type == BNGE_FLTR_TYPE_NTUPLE) {
+ ntp_fltr = container_of(fltr, struct bnge_ntuple_filter, base);
+ l2_fltr = bn->vnic_info[BNGE_VNIC_DEFAULT].l2_filters[0];
+ ntp_fltr->l2_filter_id = l2_fltr->base.filter_id;
+ if (bnge_hwrm_cfa_ntuple_filter_alloc(bn->bd, ntp_fltr)) {
+ netdev_err(bn->netdev,
+ "restoring previously configured ntuple filter id %d failed\n",
+ fltr->sw_id);
+ bnge_del_ntp_filter_rcu(bn, ntp_fltr);
+ }
+ } else if (fltr->type == BNGE_FLTR_TYPE_L2) {
+ l2_fltr = container_of(fltr, struct bnge_l2_filter, base);
+ if (bnge_hwrm_l2_filter_alloc(bn->bd, l2_fltr)) {
+ netdev_err(bn->netdev,
+ "restoring previously configured l2 filter id %d failed\n",
+ fltr->sw_id);
+ bnge_del_l2_filter_rcu(bn, l2_fltr);
+ }
+ }
+}
+
+void bnge_cfg_usr_fltrs(struct bnge_net *bn)
+{
+ struct bnge_filter_base *usr_fltr, *tmp;
+
+ list_for_each_entry_safe(usr_fltr, tmp, &bn->usr_fltr_list, list_node)
+ bnge_cfg_one_usr_fltr(bn, usr_fltr);
+}
+
+void bnge_clear_usr_fltrs(struct bnge_net *bn)
+{
+ struct bnge_filter_base *usr_fltr, *tmp;
+ struct bnge_ntuple_filter *ntp_fltr;
+ struct bnge_l2_filter *l2_fltr;
+
+ netdev_assert_locked(bn->netdev);
+
+ list_for_each_entry_safe(usr_fltr, tmp, &bn->usr_fltr_list, list_node) {
+ if (usr_fltr->type == BNGE_FLTR_TYPE_NTUPLE) {
+ ntp_fltr = container_of(usr_fltr,
+ struct bnge_ntuple_filter,
+ base);
+ bnge_del_ntp_filter(bn, ntp_fltr);
+ } else if (usr_fltr->type == BNGE_FLTR_TYPE_L2) {
+ l2_fltr = container_of(usr_fltr, struct bnge_l2_filter,
+ base);
+ bnge_del_l2_filter(bn, l2_fltr);
+ }
+ }
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
index 6cf3febe3372..a0d961a01847 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
@@ -119,4 +119,6 @@ void bnge_del_ntp_filter(struct bnge_net *bn,
struct bnge_ntuple_filter *nfltr);
void bnge_del_ntp_filter_rcu(struct bnge_net *bn,
struct bnge_ntuple_filter *fltr);
+void bnge_cfg_usr_fltrs(struct bnge_net *bn);
+void bnge_clear_usr_fltrs(struct bnge_net *bn);
#endif /* _BNGE_FILTER_H_ */
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
index 3eb67379c1a5..473d1aec80f3 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
@@ -906,15 +906,17 @@ int bnge_hwrm_l2_filter_alloc(struct bnge_dev *bd, struct bnge_l2_filter *fltr)
{
struct hwrm_cfa_l2_filter_alloc_output *resp;
struct hwrm_cfa_l2_filter_alloc_input *req;
+ u32 flags;
int rc;
rc = bnge_hwrm_req_init(bd, req, HWRM_CFA_L2_FILTER_ALLOC);
if (rc)
return rc;
- req->flags = cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_FLAGS_PATH_RX);
-
- req->flags |= cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_FLAGS_OUTERMOST);
+ flags = CFA_L2_FILTER_ALLOC_REQ_FLAGS_PATH_RX |
+ CFA_L2_FILTER_ALLOC_REQ_FLAGS_TRAFFIC_L2 |
+ CFA_L2_FILTER_ALLOC_REQ_FLAGS_OUTERMOST;
+ req->flags = cpu_to_le32(flags);
req->dst_id = cpu_to_le16(fltr->base.fw_vnic_id);
req->enables =
cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_ENABLES_L2_ADDR |
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index e59db7d7be16..19ec36b08765 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -2804,6 +2804,7 @@ static int bnge_open_core(struct bnge_net *bn)
bnge_get_port_module_status(bn);
bnge_hwrm_realloc_rss_ctx_vnic(bn);
+ bnge_cfg_usr_fltrs(bn);
return 0;
@@ -3167,6 +3168,9 @@ static int bnge_set_features(struct net_device *dev, netdev_features_t features)
!(flags & BNGE_NET_EN_NTUPLE) && bn->num_rss_ctx)
return -EBUSY;
+ if (!(flags & BNGE_NET_EN_NTUPLE))
+ bnge_clear_usr_fltrs(bn);
+
bn->priv_flags = flags;
return 0;
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 08/10] bnge: add NTUPLE filter support in ethtool
2026-09-28 6:13 ` [net-next, v4 08/10] bnge: add NTUPLE filter support in ethtool Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 12 potential
issue(s) to consider.
Critical: 0 · High: 1 · Medium: 8 · Low: 3
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [High] Once a user ETHER_FLOW rule exists for MAC M, adding M as a
secondary unicast address (for example a macvlan on top of the port)…
- [Medium] The patch adds CFA_L2_FILTER_ALLOC_REQ_FLAGS_TRAFFIC_L2
unconditionally in bnge_hwrm_l2_filter_alloc().
- [Medium] This patch makes user filters exist only while NTUPLE is
enabled (bnge_set_features calls bnge_clear_usr_fltrs when NTUPLE goes…
- [Medium] When a user L2 (ETHER_FLOW) filter is restored on open, it
reuses a firmware vNIC id that may no longer be valid.
- [Medium] The rule table size reported to userspace is inconsistent and
too small for the locations the driver hands out.
- [Medium] bnge_add_l2_cls_rule() accepts ETHER_FLOW rules that ask to
match on fields it does not implement, then drops those fields without…
- [Medium] User filters are kept in software across close, but the close
path never frees their firmware entries, and the new open path…
- [Medium] ETHER_FLOW ring rules may land on the wrong queue once the RSS
indirection table has been changed.
- [Medium] If firmware grants fewer RX rings when the interface is
reopened, restored filters can steer to a different ring than the one…
- [Low] If the caller's rule_locs buffer is too small,
ETHTOOL_GRXCLSRLALL silently returns a truncated list as a success.
- [Low] bnge_add_l2_cls_rule() requires m_ext->vlan_tci == htons(0xfff)
but sets key.vlan = ntohs(h_ext->vlan_tci) without masking to…
- [Low] net: bnge: lost destination ring information for RSS-enabled
NTUPLE filters
--
bnge: add NTUPLE filter support in ethtool
This commit adds get_rxnfc/set_rxnfc ethtool operations to bnge for
listing, inserting and deleting ETHER_FLOW and ntuple rules. User filters
are restored in bnge_open_core() and cleared when NTUPLE is disabled in
bnge_set_features(). It also adds the TRAFFIC_L2 flag to L2 filter
allocation requests.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> index 85dbe64d4c129..8e9cfbad98e04 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
[ ... ]
> @@ -1169,6 +1170,594 @@ static int bnge_remove_rxfh_context(struct net_device *dev,
[ ... ]
> +static int bnge_grxclsrlall(struct bnge_net *bn, struct ethtool_rxnfc *cmd,
> + u32 *rule_locs)
> +{
> + u32 count;
> +
> + cmd->data = bn->user_fltr_count;
> + rcu_read_lock();
> + count = bnge_get_all_fltr_ids_rcu(bn, bn->l2_fltr_hash_tbl,
> + BNGE_L2_FLTR_HASH_SIZE, rule_locs, 0,
> + cmd->rule_cnt, 0);
> + cmd->rule_cnt = bnge_get_all_fltr_ids_rcu(bn, bn->ntp_fltr_hash_tbl,
> + BNGE_NTP_FLTR_HASH_SIZE,
> + rule_locs, count,
> + cmd->rule_cnt,
> + BNGE_MAX_L2_FLTRS);
> + rcu_read_unlock();
> +
> + return 0;
> +}
[Severity: Low]
What happens when rule_cnt is smaller than the number of installed rules?
bnge_get_all_fltr_ids_rcu() stops once id_cnt entries have been copied and
doesn't check whether more remain. bnge_grxclsrlall() then always returns 0.
The get_rxnfc documentation in include/linux/ethtool.h says drivers must
return -EMSGSIZE when rule_cnt is too small. With two rules installed and
rule_cnt = 1, userspace would get one rule and no error. The same thing can
happen if rules are added between ETHTOOL_GRXCLSRLCNT and
ETHTOOL_GRXCLSRLALL.
[ ... ]
> +static int bnge_add_l2_cls_rule(struct bnge_net *bn,
> + struct ethtool_rx_flow_spec *fs)
> +{
[ ... ]
> + if (!is_broadcast_ether_addr(m_ether->h_dest))
> + return -EINVAL;
> +
> + if (is_broadcast_ether_addr(h_ether->h_dest) ||
> + is_multicast_ether_addr(h_ether->h_dest))
> + return -EINVAL;
> +
> + ether_addr_copy(key.dst_mac_addr, h_ether->h_dest);
> + key.vlan = 0;
[Severity: Medium]
Should this also reject masks for fields the filter doesn't implement?
Only the dst MAC mask and the VLAN TCI mask are checked, and only dst MAC
and VLAN are copied into key.
Non-zero masks in m_u.ether_spec.h_source and m_u.ether_spec.h_proto are
silently dropped. The same goes for m_ext.vlan_etype and m_ext.data when
FLOW_EXT is set. ethtool_set_rxnfc() in net/ethtool/ioctl.c doesn't check
these masks either.
For example, 'flow-type ether dst M src S proto 0x0800 action 3' would be
installed as a dst-only rule. It would then steer traffic that doesn't
match the requested source or EtherType.
> + if (fs->flow_type & FLOW_EXT) {
> + struct ethtool_flow_ext *m_ext = &fs->m_ext;
> + struct ethtool_flow_ext *h_ext = &fs->h_ext;
> +
> + if (m_ext->vlan_tci != htons(0xfff) || !h_ext->vlan_tci)
> + return -EINVAL;
> + key.vlan = ntohs(h_ext->vlan_tci);
> + }
[Severity: Low]
The mask has to be htons(0xfff), but key.vlan isn't masked with
VLAN_VID_MASK. Can PCP/DEI bits end up in the software key and in
req->l2_ivlan, while l2_ivlan_mask is 0xfff?
bnge_lookup_l2_filter() compares the full vlan. So two rules with the same
VID but different PCP would pass the duplicate check and create duplicate
hardware filters.
A priority-tagged TCI such as 0x2000 would also pass the !h_ext->vlan_tci
check and install a num_vlans = 1 filter on VID 0.
> +
> + flags = BNGE_ACT_RING_DST;
> + vnic_id = bn->vnic_info[BNGE_VNIC_DEFAULT].fw_vnic_id;
[Severity: Medium]
Do ETHER_FLOW ring rules still reach the requested ring after the RSS
indirection table has been changed?
With BNGE_ACT_RING_DST set, bnge_hwrm_l2_filter_alloc() programs dst_id as
the default vNIC and uses rxq as the ring table index:
if (BNGE_L2_FLTR_IS_NTP_MAC(fltr)) {
req->enables |= cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_ENABLES_RFS_RING_TBL_IDX);
req->rfs_ring_tbl_idx = cpu_to_le16(fltr->base.rxq);
}
The ntuple path in bnge_cfg_rfs_ring_tbl_idx() uses BNGE_VNIC_NTUPLE
instead. bnge_fill_hw_rss_tbl() gives only that vNIC an identity table, via
ethtool_rxfh_indir_default(). The default vNIC's table comes from
bd->rss_indir_tbl.
After 'ethtool -X', wouldn't matching packets go to rss_indir_tbl[ring]
while bnge_grxclsrule() still reports the requested ring?
For comparison, bnxt doesn't use RFS_RING_TBL_IDX on L2 filters and refuses
ETHER_FLOW ring rules on P5+ chips.
> +
> + fltr = bnge_alloc_user_l2_filter(bn, &key, flags);
> + if (IS_ERR(fltr))
> + return PTR_ERR(fltr);
[Severity: High]
Can a user ETHER_FLOW rule for MAC M break RX filtering once M is also
added as a secondary unicast address, for example by a macvlan on top of
the port?
User L2 rules now go into the same l2_fltr_hash_tbl as the vNIC unicast
filters. bnge_alloc_l2_filter() returns ERR_PTR(-EEXIST) when the
{mac, vlan} key is already there. bnge_cfg_rx_mode() only skips uc entries
equal to dev_addr, so M reaches bnge_hwrm_set_vnic_filter():
bnge_set_rx_mode()
bnge_cfg_rx_mode()
bnge_hwrm_set_vnic_filter()
bnge_alloc_l2_filter() -> -EEXIST
vnic->uc_filter_count = i;
return rc;
This returns before the skip_uc label, so bnge_hwrm_cfa_l2_set_rx_mask()
never runs and the bcast/mcast/promisc mask updates are lost.
User L2 filters also survive close. bnge_hwrm_clear_vnic_filter() only
frees vnic->l2_filters[], and bnge_free_l2_filters() is only called from
bnge_netdev_free(). On the next open:
bnge_init_chip()
bnge_cfg_rx_mode() -> -EEXIST
else if (rc)
goto err_out;
Wouldn't the interface then fail to open until the ethtool rule is deleted?
In that case bnge_cfg_usr_fltrs() is never reached either.
Before this patch only vNIC unicast filters could be in the L2 hash, so
this collision couldn't happen.
[ ... ]
> +static int bnge_srxclsrldel(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
> +{
[ ... ]
> + if (fltr_base) {
> + struct bnge_l2_filter *l2_fltr;
> +
> + l2_fltr = container_of(fltr_base, struct bnge_l2_filter, base);
> + rcu_read_unlock();
> + bnge_hwrm_l2_filter_free(bn->bd, l2_fltr);
> + bnge_del_l2_filter_rcu(bn, l2_fltr);
> + return 0;
> + }
[Severity: Medium]
Are the firmware entries for user filters ever freed across close/open?
User filters stay in software over close, but the close path doesn't free
them in firmware:
- bnge_hwrm_clear_vnic_filter() only frees vnic->l2_filters[]
- bnge_free_core() -> bnge_free_ntp_fltrs(bn, true) skips user filters and
issues no HWRM call
- on open, bnge_cfg_usr_fltrs() allocates every user filter in firmware
again and overwrites base.filter_id
If firmware doesn't free these implicitly, each close/open cycle leaks the
previous entries. bnge_clear_usr_fltrs() has the same problem when NTUPLE
is turned off. That can only happen while the interface is down, because
fix_features blocks the change while it is running.
If firmware does free them with the vNIC, then deleting an L2 rule here
while the interface is down sends HWRM_CFA_L2_FILTER_FREE with the
filter_id from the previous up cycle. Unlike the ntuple free helper,
bnge_hwrm_l2_filter_free() has no BNGE_STATE_OPEN check.
Could that command fail, or free an unrelated filter if the id has been
reused?
[ ... ]
> + switch (cmd->cmd) {
> + case ETHTOOL_GRXCLSRLCNT:
> + cmd->rule_cnt = bn->user_fltr_count;
> + cmd->data = bd->max_fltr | RX_CLS_LOC_SPECIAL;
> + break;
[Severity: Medium]
Is this the right rule table size? bd->max_fltr is BNGE_MAX_NTUPLE_FLTRS
and doesn't include the L2 range. Ntuple rules, however, are returned at:
fs->location = BNGE_MAX_L2_FLTRS + new_fltr->base.sw_id;
bnge_grxclsrule() also accepts locations up to
BNGE_MAX_L2_FLTRS + bd->max_fltr. So valid locations can go past the
advertised size.
bnge_grxclsrlall() reports a rule count instead of the table size:
cmd->data = bn->user_fltr_count;
So a single ntuple rule at location 128 gives @data = 1.
include/uapi/linux/ethtool.h defines @data as the size of the rule table
for both ETHTOOL_GRXCLSRLCNT and ETHTOOL_GRXCLSRLALL. bnxt avoids the first
mismatch because its max_fltr includes BNXT_L2_FLTR_MAX_FLTR.
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> index 7fbb67021b5d9..764c967c64572 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> @@ -577,3 +577,63 @@ int bnge_hwrm_set_vnic_filter(struct bnge_net *bn, u16 vnic_id, u16 idx,
[ ... ]
> + if (fltr->type == BNGE_FLTR_TYPE_NTUPLE) {
> + ntp_fltr = container_of(fltr, struct bnge_ntuple_filter, base);
> + l2_fltr = bn->vnic_info[BNGE_VNIC_DEFAULT].l2_filters[0];
> + ntp_fltr->l2_filter_id = l2_fltr->base.filter_id;
> + if (bnge_hwrm_cfa_ntuple_filter_alloc(bn->bd, ntp_fltr)) {
[Severity: Medium]
Can the saved base.rxq be out of range here? On open,
bnge_reserve_rings() can lower bd->rx_nr_rings and log "RX rings resv
reduced". It handles the RSS indirection table through
ethtool_rxfh_indir_lost(), but it doesn't touch ring-steering filters.
This path reprograms each filter with its old rxq and doesn't check it
against the new rx_nr_rings. The NTUPLE vNIC's ring table is
ethtool_rxfh_indir_default(i, rx_nr_rings).
Wouldn't an rxq at or above the new count then resolve to
rxq % rx_nr_rings, while get_rxnfc still reports the old ring?
> + netdev_err(bn->netdev,
> + "restoring previously configured ntuple filter id %d failed\n",
> + fltr->sw_id);
> + bnge_del_ntp_filter_rcu(bn, ntp_fltr);
> + }
> + } else if (fltr->type == BNGE_FLTR_TYPE_L2) {
> + l2_fltr = container_of(fltr, struct bnge_l2_filter, base);
> + if (bnge_hwrm_l2_filter_alloc(bn->bd, l2_fltr)) {
[Severity: Medium]
Is fltr->base.fw_vnic_id still valid at this point? bnge_add_l2_cls_rule()
saved the default vNIC's firmware id when the rule was added:
vnic_id = bn->vnic_info[BNGE_VNIC_DEFAULT].fw_vnic_id;
...
fltr->base.fw_vnic_id = vnic_id;
bnge_clear_vnic() frees that vNIC on close, and on open
bnge_hwrm_vnic_alloc() gets a new id from firmware (resp->vnic_id). This
branch doesn't refresh fw_vnic_id, so req->dst_id still carries the old id.
The ntuple branch above does refresh its per-open handle (l2_filter_id).
If firmware hands out a different vNIC id, wouldn't the restored rule point
at a stale vNIC? Or, if firmware rejects the id, the rule would be removed
by bnge_del_l2_filter_rcu() with only a log message.
[ ... ]
> +void bnge_cfg_usr_fltrs(struct bnge_net *bn)
> +{
> + struct bnge_filter_base *usr_fltr, *tmp;
> +
> + list_for_each_entry_safe(usr_fltr, tmp, &bn->usr_fltr_list, list_node)
> + bnge_cfg_one_usr_fltr(bn, usr_fltr);
> +}
[Severity: Medium]
Should this check BNGE_NET_EN_NTUPLE before reprogramming filters?
User filters are cleared when NTUPLE is turned off through
bnge_set_features(). However, bnge_init_chip() can also turn it off when
bnge_alloc_rfs_vnic() fails during open, and it doesn't clear user filters:
bd->flags &= ~BNGE_EN_ARFS_CAP;
bn->priv_flags &= ~BNGE_NET_EN_NTUPLE;
bn->netdev->features &= ~NETIF_F_NTUPLE;
bnge_open_core() then calls bnge_cfg_usr_fltrs(), which restores every
saved filter. DROP and L2 filters stay active. Non-RSS ntuple filters get
dst_id = vnic_info[BNGE_VNIC_NTUPLE].fw_vnic_id from
bnge_cfg_rfs_ring_tbl_idx(), and that is the vNIC that just failed.
A later 'ethtool -K ntuple off' would also do nothing. bnge_set_features()
returns early when flags == priv_flags, so bnge_clear_usr_fltrs() never
runs.
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
> index 3eb67379c1a57..473d1aec80f38 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_hwrm_lib.c
> @@ -906,15 +906,17 @@ int bnge_hwrm_l2_filter_alloc(struct bnge_dev *bd, struct bnge_l2_filter *fltr)
[ ... ]
> - req->flags = cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_FLAGS_PATH_RX);
> -
> - req->flags |= cpu_to_le32(CFA_L2_FILTER_ALLOC_REQ_FLAGS_OUTERMOST);
> + flags = CFA_L2_FILTER_ALLOC_REQ_FLAGS_PATH_RX |
> + CFA_L2_FILTER_ALLOC_REQ_FLAGS_TRAFFIC_L2 |
> + CFA_L2_FILTER_ALLOC_REQ_FLAGS_OUTERMOST;
> + req->flags = cpu_to_le32(flags);
[Severity: Medium]
CFA_L2_FILTER_ALLOC_REQ_FLAGS_TRAFFIC_L2 is now set for every L2 filter.
That includes the default vNIC MAC filters that bnge_hwrm_set_vnic_filter()
installs on every open.
The traffic field used to be 0 (TRAFFIC_NO_ROCE_L2), and bnxt doesn't set
this flag. Is this change intended, and could it affect RoCE/L2 steering
for the aux RoCE driver?
The commit message doesn't mention this change. It also doesn't mention
that user filters are now restored on open (bnge_cfg_usr_fltrs()) and
flushed when NTUPLE is disabled (bnge_clear_usr_fltrs()).
The commit message says:
Since with ethtool support user filters creation is supported, when
interface is down clearing of filters which are in software remains
intact.
Isn't this already the existing behaviour? bnge_free_core() already calls
bnge_free_ntp_fltrs(bn, true) under the comment "Free non-user filters
only", and this patch doesn't change that.
[ ... ]
> @@ -1169,6 +1170,594 @@ static int bnge_remove_rxfh_context(struct net_device *dev,
> return 0;
> }
>
> +#define BNGE_IP_PROTO_FULL_MASK 0xFF
> +#define BNGE_IP_PROTO_WILDCARD 0x0
> +
> +static u32 bnge_get_all_fltr_ids_rcu(struct bnge_net *bn,
> + struct hlist_head tbl[],
> + u32 tbl_size, u32 *ids, u32 start,
> + u32 id_cnt, u32 offset)
> +{
> + u32 i, j = start;
> +
> + if (j >= id_cnt)
> + return j;
> +
> + for (i = 0; i < tbl_size; i++) {
> + struct bnge_filter_base *fltr;
> + struct hlist_head *head;
> +
> + head = &tbl[i];
> + hlist_for_each_entry_rcu(fltr, head, hlist) {
> + if (!fltr->flags ||
> + test_bit(BNGE_FLTR_FW_DELETED, &fltr->state))
> + continue;
> + ids[j++] = fltr->sw_id + offset;
> + if (j == id_cnt)
> + return j;
> + }
> + }
> + return j;
> +}
> +
> +static struct bnge_filter_base *bnge_get_one_fltr_rcu(struct bnge_net *bn,
> + struct hlist_head tbl[],
> + u32 tbl_size, u32 id,
> + u32 offset)
> +{
> + u32 i;
> +
> + for (i = 0; i < tbl_size; i++) {
> + struct bnge_filter_base *fltr;
> + struct hlist_head *head;
> +
> + head = &tbl[i];
> + hlist_for_each_entry_rcu(fltr, head, hlist) {
> + if (fltr->flags && fltr->sw_id + offset == id)
> + return fltr;
> + }
> + }
> + return NULL;
> +}
> +
> +static int bnge_grxclsrlall(struct bnge_net *bn, struct ethtool_rxnfc *cmd,
> + u32 *rule_locs)
> +{
> + u32 count;
> +
> + cmd->data = bn->user_fltr_count;
> + rcu_read_lock();
> + count = bnge_get_all_fltr_ids_rcu(bn, bn->l2_fltr_hash_tbl,
> + BNGE_L2_FLTR_HASH_SIZE, rule_locs, 0,
> + cmd->rule_cnt, 0);
> + cmd->rule_cnt = bnge_get_all_fltr_ids_rcu(bn, bn->ntp_fltr_hash_tbl,
> + BNGE_NTP_FLTR_HASH_SIZE,
> + rule_locs, count,
> + cmd->rule_cnt,
> + BNGE_MAX_L2_FLTRS);
> + rcu_read_unlock();
> +
> + return 0;
> +}
> +
> +static int bnge_grxclsrule(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
> +{
> + struct ethtool_rx_flow_spec *fs =
> + (struct ethtool_rx_flow_spec *)&cmd->fs;
> + struct bnge_filter_base *fltr_base;
> + struct bnge_ntuple_filter *fltr;
> + struct bnge_flow_masks *fmasks;
> + struct bnge_dev *bd = bn->bd;
> + struct flow_keys *fkeys;
> + int rc = -EINVAL;
> +
> + if (fs->location >= BNGE_MAX_L2_FLTRS + bd->max_fltr)
> + return rc;
> +
> + rcu_read_lock();
> + fltr_base = bnge_get_one_fltr_rcu(bn, bn->l2_fltr_hash_tbl,
> + BNGE_L2_FLTR_HASH_SIZE,
> + fs->location, 0);
> + if (fltr_base) {
> + struct ethhdr *h_ether = &fs->h_u.ether_spec;
> + struct ethhdr *m_ether = &fs->m_u.ether_spec;
> + struct bnge_l2_filter *l2_fltr;
> + struct bnge_l2_key *l2_key;
> +
> + l2_fltr = container_of(fltr_base, struct bnge_l2_filter, base);
> + l2_key = &l2_fltr->l2_key;
> + fs->flow_type = ETHER_FLOW;
> + ether_addr_copy(h_ether->h_dest, l2_key->dst_mac_addr);
> + eth_broadcast_addr(m_ether->h_dest);
> + if (l2_key->vlan) {
> + struct ethtool_flow_ext *m_ext = &fs->m_ext;
> + struct ethtool_flow_ext *h_ext = &fs->h_ext;
> +
> + fs->flow_type |= FLOW_EXT;
> + m_ext->vlan_tci = htons(0xfff);
> + h_ext->vlan_tci = htons(l2_key->vlan);
> + }
> + if (fltr_base->flags & BNGE_ACT_RING_DST)
> + fs->ring_cookie = fltr_base->rxq;
> + rcu_read_unlock();
> + return 0;
> + }
> + fltr_base = bnge_get_one_fltr_rcu(bn, bn->ntp_fltr_hash_tbl,
> + BNGE_NTP_FLTR_HASH_SIZE,
> + fs->location, BNGE_MAX_L2_FLTRS);
> + if (!fltr_base) {
> + rcu_read_unlock();
> + return rc;
> + }
> + fltr = container_of(fltr_base, struct bnge_ntuple_filter, base);
> +
> + fkeys = &fltr->fkeys;
> + fmasks = &fltr->fmasks;
> + if (fkeys->basic.n_proto == htons(ETH_P_IP)) {
> + if (fkeys->basic.ip_proto == BNGE_IP_PROTO_WILDCARD) {
> + fs->flow_type = IP_USER_FLOW;
> + fs->h_u.usr_ip4_spec.ip_ver = ETH_RX_NFC_IP4;
> + fs->h_u.usr_ip4_spec.proto = BNGE_IP_PROTO_WILDCARD;
> + fs->m_u.usr_ip4_spec.proto = 0;
> + } else if (fkeys->basic.ip_proto == IPPROTO_ICMP) {
> + fs->flow_type = IP_USER_FLOW;
> + fs->h_u.usr_ip4_spec.ip_ver = ETH_RX_NFC_IP4;
> + fs->h_u.usr_ip4_spec.proto = IPPROTO_ICMP;
> + fs->m_u.usr_ip4_spec.proto = BNGE_IP_PROTO_FULL_MASK;
> + } else if (fkeys->basic.ip_proto == IPPROTO_TCP) {
> + fs->flow_type = TCP_V4_FLOW;
> + } else if (fkeys->basic.ip_proto == IPPROTO_UDP) {
> + fs->flow_type = UDP_V4_FLOW;
> + } else {
> + goto fltr_err;
> + }
> +
> + fs->h_u.tcp_ip4_spec.ip4src = fkeys->addrs.v4addrs.src;
> + fs->m_u.tcp_ip4_spec.ip4src = fmasks->addrs.v4addrs.src;
> + fs->h_u.tcp_ip4_spec.ip4dst = fkeys->addrs.v4addrs.dst;
> + fs->m_u.tcp_ip4_spec.ip4dst = fmasks->addrs.v4addrs.dst;
> + if (fs->flow_type == TCP_V4_FLOW ||
> + fs->flow_type == UDP_V4_FLOW) {
> + fs->h_u.tcp_ip4_spec.psrc = fkeys->ports.src;
> + fs->m_u.tcp_ip4_spec.psrc = fmasks->ports.src;
> + fs->h_u.tcp_ip4_spec.pdst = fkeys->ports.dst;
> + fs->m_u.tcp_ip4_spec.pdst = fmasks->ports.dst;
> + }
> + } else {
> + if (fkeys->basic.ip_proto == BNGE_IP_PROTO_WILDCARD) {
> + fs->flow_type = IPV6_USER_FLOW;
> + fs->h_u.usr_ip6_spec.l4_proto =
> + BNGE_IP_PROTO_WILDCARD;
> + fs->m_u.usr_ip6_spec.l4_proto = 0;
> + } else if (fkeys->basic.ip_proto == IPPROTO_ICMPV6) {
> + fs->flow_type = IPV6_USER_FLOW;
> + fs->h_u.usr_ip6_spec.l4_proto = IPPROTO_ICMPV6;
> + fs->m_u.usr_ip6_spec.l4_proto =
> + BNGE_IP_PROTO_FULL_MASK;
> + } else if (fkeys->basic.ip_proto == IPPROTO_TCP) {
> + fs->flow_type = TCP_V6_FLOW;
> + } else if (fkeys->basic.ip_proto == IPPROTO_UDP) {
> + fs->flow_type = UDP_V6_FLOW;
> + } else {
> + goto fltr_err;
> + }
> +
> + memcpy(fs->h_u.tcp_ip6_spec.ip6src,
> + &fkeys->addrs.v6addrs.src,
> + sizeof(struct in6_addr));
> + memcpy(fs->m_u.tcp_ip6_spec.ip6src,
> + &fmasks->addrs.v6addrs.src,
> + sizeof(struct in6_addr));
> + memcpy(fs->h_u.tcp_ip6_spec.ip6dst,
> + &fkeys->addrs.v6addrs.dst,
> + sizeof(struct in6_addr));
> + memcpy(fs->m_u.tcp_ip6_spec.ip6dst,
> + &fmasks->addrs.v6addrs.dst,
> + sizeof(struct in6_addr));
> +
> + if (fs->flow_type == TCP_V6_FLOW ||
> + fs->flow_type == UDP_V6_FLOW) {
> + fs->h_u.tcp_ip6_spec.psrc = fkeys->ports.src;
> + fs->m_u.tcp_ip6_spec.psrc = fmasks->ports.src;
> + fs->h_u.tcp_ip6_spec.pdst = fkeys->ports.dst;
> + fs->m_u.tcp_ip6_spec.pdst = fmasks->ports.dst;
> + }
> + }
> +
> + if (fltr->base.flags & BNGE_ACT_DROP) {
> + fs->ring_cookie = RX_CLS_FLOW_DISC;
> + } else if (fltr->base.flags & BNGE_ACT_RSS_CTX) {
> + fs->flow_type |= FLOW_RSS;
> + cmd->rss_context = fltr->base.fw_vnic_id;
> + } else {
> + fs->ring_cookie = fltr->base.rxq;
> + }
> + rc = 0;
> +
> +fltr_err:
> + rcu_read_unlock();
> +
> + return rc;
> +}
> +
> +static bool bnge_verify_ntuple_ip4_flow(struct ethtool_usrip4_spec *ip_spec,
> + struct ethtool_usrip4_spec *ip_mask)
> +{
> + u8 mproto = ip_mask->proto;
> + u8 sproto = ip_spec->proto;
> +
> + if (ip_mask->l4_4_bytes || ip_mask->tos ||
> + ip_spec->ip_ver != ETH_RX_NFC_IP4 ||
> + (mproto && (mproto != BNGE_IP_PROTO_FULL_MASK ||
> + sproto != IPPROTO_ICMP)))
> + return false;
> + return true;
> +}
> +
> +static bool bnge_verify_ntuple_ip6_flow(struct ethtool_usrip6_spec *ip_spec,
> + struct ethtool_usrip6_spec *ip_mask)
> +{
> + u8 mproto = ip_mask->l4_proto;
> + u8 sproto = ip_spec->l4_proto;
> +
> + if (ip_mask->l4_4_bytes || ip_mask->tclass ||
> + (mproto && (mproto != BNGE_IP_PROTO_FULL_MASK ||
> + sproto != IPPROTO_ICMPV6)))
> + return false;
> + return true;
> +}
> +
> +static int bnge_add_ntuple_cls_rule(struct bnge_net *bn,
> + struct ethtool_rxnfc *cmd)
> +{
> + struct ethtool_rx_flow_spec *fs = &cmd->fs;
> + struct bnge_ntuple_filter *new_fltr, *fltr;
> + u32 flow_type = fs->flow_type & 0xff;
> + struct bnge_l2_filter *l2_fltr;
> + struct bnge_flow_masks *fmasks;
> + struct flow_keys *fkeys;
> + u32 idx;
> + int rc;
> +
> + if (!bn->vnic_info)
> + return -EAGAIN;
> +
> + if (fs->flow_type & (FLOW_MAC_EXT | FLOW_EXT))
> + return -EOPNOTSUPP;
> +
> + if (fs->ring_cookie != RX_CLS_FLOW_DISC &&
> + ethtool_get_flow_spec_ring_vf(fs->ring_cookie))
> + return -EOPNOTSUPP;
> +
> + if (flow_type == IP_USER_FLOW) {
> + if (!bnge_verify_ntuple_ip4_flow(&fs->h_u.usr_ip4_spec,
> + &fs->m_u.usr_ip4_spec))
> + return -EOPNOTSUPP;
> + }
> +
> + if (flow_type == IPV6_USER_FLOW) {
> + if (!bnge_verify_ntuple_ip6_flow(&fs->h_u.usr_ip6_spec,
> + &fs->m_u.usr_ip6_spec))
> + return -EOPNOTSUPP;
> + }
> +
> + new_fltr = kzalloc_obj(*new_fltr, GFP_KERNEL);
> + if (!new_fltr)
> + return -ENOMEM;
> +
> + l2_fltr = bn->vnic_info[BNGE_VNIC_DEFAULT].l2_filters[0];
> + new_fltr->l2_filter_id = l2_fltr->base.filter_id;
> + fmasks = &new_fltr->fmasks;
> + fkeys = &new_fltr->fkeys;
> +
> + rc = -EOPNOTSUPP;
> + switch (flow_type) {
> + case IP_USER_FLOW: {
> + struct ethtool_usrip4_spec *ip_spec = &fs->h_u.usr_ip4_spec;
> + struct ethtool_usrip4_spec *ip_mask = &fs->m_u.usr_ip4_spec;
> +
> + fkeys->basic.ip_proto = ip_mask->proto ? ip_spec->proto
> + : BNGE_IP_PROTO_WILDCARD;
> + fkeys->basic.n_proto = htons(ETH_P_IP);
> + fkeys->addrs.v4addrs.src = ip_spec->ip4src;
> + fmasks->addrs.v4addrs.src = ip_mask->ip4src;
> + fkeys->addrs.v4addrs.dst = ip_spec->ip4dst;
> + fmasks->addrs.v4addrs.dst = ip_mask->ip4dst;
> + break;
> + }
> + case TCP_V4_FLOW:
> + case UDP_V4_FLOW: {
> + struct ethtool_tcpip4_spec *ip_spec = &fs->h_u.tcp_ip4_spec;
> + struct ethtool_tcpip4_spec *ip_mask = &fs->m_u.tcp_ip4_spec;
> +
> + if (ip_mask->tos)
> + goto err_free_fltr;
> +
> + fkeys->basic.ip_proto = IPPROTO_TCP;
> + if (flow_type == UDP_V4_FLOW)
> + fkeys->basic.ip_proto = IPPROTO_UDP;
> + fkeys->basic.n_proto = htons(ETH_P_IP);
> + fkeys->addrs.v4addrs.src = ip_spec->ip4src;
> + fmasks->addrs.v4addrs.src = ip_mask->ip4src;
> + fkeys->addrs.v4addrs.dst = ip_spec->ip4dst;
> + fmasks->addrs.v4addrs.dst = ip_mask->ip4dst;
> + fkeys->ports.src = ip_spec->psrc;
> + fmasks->ports.src = ip_mask->psrc;
> + fkeys->ports.dst = ip_spec->pdst;
> + fmasks->ports.dst = ip_mask->pdst;
> + break;
> + }
> + case IPV6_USER_FLOW: {
> + struct ethtool_usrip6_spec *ip_spec = &fs->h_u.usr_ip6_spec;
> + struct ethtool_usrip6_spec *ip_mask = &fs->m_u.usr_ip6_spec;
> +
> + fkeys->basic.ip_proto = ip_mask->l4_proto ? ip_spec->l4_proto
> + : BNGE_IP_PROTO_WILDCARD;
> + fkeys->basic.n_proto = htons(ETH_P_IPV6);
> +
> + memcpy(&fkeys->addrs.v6addrs.src, ip_spec->ip6src,
> + sizeof(struct in6_addr));
> + memcpy(&fmasks->addrs.v6addrs.src, ip_mask->ip6src,
> + sizeof(struct in6_addr));
> + memcpy(&fkeys->addrs.v6addrs.dst, ip_spec->ip6dst,
> + sizeof(struct in6_addr));
> + memcpy(&fmasks->addrs.v6addrs.dst, ip_mask->ip6dst,
> + sizeof(struct in6_addr));
> + break;
> + }
> + case TCP_V6_FLOW:
> + case UDP_V6_FLOW: {
> + struct ethtool_tcpip6_spec *ip_spec = &fs->h_u.tcp_ip6_spec;
> + struct ethtool_tcpip6_spec *ip_mask = &fs->m_u.tcp_ip6_spec;
> +
> + if (ip_mask->tclass)
> + goto err_free_fltr;
> +
> + fkeys->basic.ip_proto = IPPROTO_TCP;
> + if (flow_type == UDP_V6_FLOW)
> + fkeys->basic.ip_proto = IPPROTO_UDP;
> + fkeys->basic.n_proto = htons(ETH_P_IPV6);
> +
> + memcpy(&fkeys->addrs.v6addrs.src, ip_spec->ip6src,
> + sizeof(struct in6_addr));
> + memcpy(&fmasks->addrs.v6addrs.src, ip_mask->ip6src,
> + sizeof(struct in6_addr));
> + memcpy(&fkeys->addrs.v6addrs.dst, ip_spec->ip6dst,
> + sizeof(struct in6_addr));
> + memcpy(&fmasks->addrs.v6addrs.dst, ip_mask->ip6dst,
> + sizeof(struct in6_addr));
> +
> + fkeys->ports.src = ip_spec->psrc;
> + fmasks->ports.src = ip_mask->psrc;
> + fkeys->ports.dst = ip_spec->pdst;
> + fmasks->ports.dst = ip_mask->pdst;
> + break;
> + }
> + default:
> + rc = -EOPNOTSUPP;
> + goto err_free_fltr;
> + }
> + if (!memcmp(&BNGE_FLOW_MASK_NONE, fmasks, sizeof(*fmasks)))
> + goto err_free_fltr;
> +
> + idx = bnge_get_ntp_filter_idx(bn, fkeys, NULL);
> + rcu_read_lock();
> + fltr = bnge_lookup_ntp_filter_from_idx(bn, new_fltr, idx);
> + if (fltr) {
> + rcu_read_unlock();
> + rc = -EEXIST;
> + goto err_free_fltr;
> + }
> + rcu_read_unlock();
> +
> + new_fltr->base.flags = BNGE_ACT_NO_AGING;
> + if (fs->flow_type & FLOW_RSS) {
> + struct bnge_rss_ctx *rss_ctx;
> +
> + new_fltr->base.fw_vnic_id = 0;
> + new_fltr->base.flags |= BNGE_ACT_RSS_CTX;
> + rss_ctx = bnge_get_rss_ctx_from_index(bn, cmd->rss_context);
> + if (rss_ctx) {
> + new_fltr->base.fw_vnic_id = rss_ctx->index;
> + } else {
> + rc = -EINVAL;
> + goto err_free_fltr;
> + }
> + }
> + if (fs->ring_cookie == RX_CLS_FLOW_DISC)
> + new_fltr->base.flags |= BNGE_ACT_DROP;
> + else
> + new_fltr->base.rxq = ethtool_get_flow_spec_ring(fs->ring_cookie);
> + __set_bit(BNGE_FLTR_VALID, &new_fltr->base.state);
> + rc = bnge_insert_ntp_filter(bn, new_fltr, idx);
> + if (!rc) {
> + rc = bnge_hwrm_cfa_ntuple_filter_alloc(bn->bd, new_fltr);
> + if (rc) {
> + bnge_del_ntp_filter_rcu(bn, new_fltr);
> + return rc;
> + }
> + fs->location = BNGE_MAX_L2_FLTRS + new_fltr->base.sw_id;
> + return 0;
> + }
> +
> +err_free_fltr:
> + kfree(new_fltr);
> + return rc;
> +}
> +
> +static int bnge_add_l2_cls_rule(struct bnge_net *bn,
> + struct ethtool_rx_flow_spec *fs)
> +{
> + u32 ring = ethtool_get_flow_spec_ring(fs->ring_cookie);
> + struct ethhdr *h_ether = &fs->h_u.ether_spec;
> + struct ethhdr *m_ether = &fs->m_u.ether_spec;
> + struct bnge_l2_filter *fltr;
> + struct bnge_l2_key key;
> + u16 vnic_id;
> + u8 flags;
> + int rc;
> +
> + if (ethtool_get_flow_spec_ring_vf(fs->ring_cookie))
> + return -EOPNOTSUPP;
> +
> + if (!is_broadcast_ether_addr(m_ether->h_dest))
> + return -EINVAL;
> +
> + if (is_broadcast_ether_addr(h_ether->h_dest) ||
> + is_multicast_ether_addr(h_ether->h_dest))
> + return -EINVAL;
> +
> + ether_addr_copy(key.dst_mac_addr, h_ether->h_dest);
> + key.vlan = 0;
> + if (fs->flow_type & FLOW_EXT) {
> + struct ethtool_flow_ext *m_ext = &fs->m_ext;
> + struct ethtool_flow_ext *h_ext = &fs->h_ext;
> +
> + if (m_ext->vlan_tci != htons(0xfff) || !h_ext->vlan_tci)
> + return -EINVAL;
> + key.vlan = ntohs(h_ext->vlan_tci);
> + }
> +
> + flags = BNGE_ACT_RING_DST;
> + vnic_id = bn->vnic_info[BNGE_VNIC_DEFAULT].fw_vnic_id;
> +
> + fltr = bnge_alloc_user_l2_filter(bn, &key, flags);
> + if (IS_ERR(fltr))
> + return PTR_ERR(fltr);
> +
> + fltr->base.fw_vnic_id = vnic_id;
> + fltr->base.rxq = ring;
> + rc = bnge_hwrm_l2_filter_alloc(bn->bd, fltr);
> + if (rc)
> + bnge_del_l2_filter_rcu(bn, fltr);
> + else
> + fs->location = fltr->base.sw_id;
> + return rc;
> +}
> +
> +static int bnge_srxclsrlins(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
> +{
> + struct ethtool_rx_flow_spec *fs = &cmd->fs;
> + struct bnge_dev *bd = bn->bd;
> + u32 ring, flow_type;
> + int rc;
> +
> + if (!netif_running(bn->netdev))
> + return -EAGAIN;
> + if (!(bn->priv_flags & BNGE_NET_EN_NTUPLE))
> + return -EPERM;
> + if (fs->location != RX_CLS_LOC_ANY)
> + return -EINVAL;
> +
> + flow_type = fs->flow_type;
> +
> + if (flow_type & FLOW_MAC_EXT)
> + return -EINVAL;
> +
> + flow_type &= ~FLOW_EXT;
> +
> + if (fs->ring_cookie == RX_CLS_FLOW_DISC && flow_type != ETHER_FLOW)
> + return bnge_add_ntuple_cls_rule(bn, cmd);
> +
> + ring = ethtool_get_flow_spec_ring(fs->ring_cookie);
> + if (ring >= bd->rx_nr_rings)
> + return -EINVAL;
> +
> + if (flow_type == ETHER_FLOW)
> + rc = bnge_add_l2_cls_rule(bn, fs);
> + else
> + rc = bnge_add_ntuple_cls_rule(bn, cmd);
> + return rc;
> +}
> +
> +static int bnge_srxclsrldel(struct bnge_net *bn, struct ethtool_rxnfc *cmd)
> +{
> + struct ethtool_rx_flow_spec *fs = &cmd->fs;
> + struct bnge_filter_base *fltr_base;
> + struct bnge_ntuple_filter *fltr;
> + u32 id = fs->location;
> +
> + rcu_read_lock();
> + fltr_base = bnge_get_one_fltr_rcu(bn, bn->l2_fltr_hash_tbl,
> + BNGE_L2_FLTR_HASH_SIZE, id, 0);
> + if (fltr_base) {
> + struct bnge_l2_filter *l2_fltr;
> +
> + l2_fltr = container_of(fltr_base, struct bnge_l2_filter, base);
> + rcu_read_unlock();
> + bnge_hwrm_l2_filter_free(bn->bd, l2_fltr);
> + bnge_del_l2_filter_rcu(bn, l2_fltr);
> + return 0;
> + }
> + fltr_base = bnge_get_one_fltr_rcu(bn, bn->ntp_fltr_hash_tbl,
> + BNGE_NTP_FLTR_HASH_SIZE, id,
> + BNGE_MAX_L2_FLTRS);
> + if (!fltr_base) {
> + rcu_read_unlock();
> + return -ENOENT;
> + }
> +
> + fltr = container_of(fltr_base, struct bnge_ntuple_filter, base);
> + if (!(fltr->base.flags & BNGE_ACT_NO_AGING)) {
> + rcu_read_unlock();
> + return -EINVAL;
> + }
> + rcu_read_unlock();
> + bnge_hwrm_cfa_ntuple_filter_free(bn->bd, fltr);
> + bnge_del_ntp_filter_rcu(bn, fltr);
> + return 0;
> +}
> +
> +static int bnge_get_rxnfc(struct net_device *dev, struct ethtool_rxnfc *cmd,
> + u32 *rule_locs)
> +{
> + struct bnge_net *bn = netdev_priv(dev);
> + struct bnge_dev *bd = bn->bd;
> + int rc = 0;
> +
> + switch (cmd->cmd) {
> + case ETHTOOL_GRXCLSRLCNT:
> + cmd->rule_cnt = bn->user_fltr_count;
> + cmd->data = bd->max_fltr | RX_CLS_LOC_SPECIAL;
> + break;
> +
> + case ETHTOOL_GRXCLSRLALL:
> + rc = bnge_grxclsrlall(bn, cmd, (u32 *)rule_locs);
> + break;
> +
> + case ETHTOOL_GRXCLSRULE:
> + rc = bnge_grxclsrule(bn, cmd);
> + break;
> +
> + default:
> + rc = -EOPNOTSUPP;
> + break;
> + }
> +
> + return rc;
> +}
> +
> +static int bnge_set_rxnfc(struct net_device *dev, struct ethtool_rxnfc *cmd)
> +{
> + struct bnge_net *bn = netdev_priv(dev);
> + int rc;
> +
> + switch (cmd->cmd) {
> + case ETHTOOL_SRXCLSRLINS:
> + rc = bnge_srxclsrlins(bn, cmd);
> + break;
> +
> + case ETHTOOL_SRXCLSRLDEL:
> + rc = bnge_srxclsrldel(bn, cmd);
> + break;
> +
> + default:
> + rc = -EOPNOTSUPP;
> + break;
> + }
> + return rc;
> +}
> +
> static const struct ethtool_ops bnge_ethtool_ops = {
> .cap_link_lanes_supported = 1,
> .get_link_ksettings = bnge_get_link_ksettings,
[Severity: Low]
When BNGE_ACT_RSS_CTX is set, this branch sets FLOW_RSS and
cmd->rss_context, but it never writes fs->ring_cookie.
bnge_add_ntuple_cls_rule() still stores the requested ring in base.rxq
for these filters, because the RX_CLS_FLOW_DISC check there doesn't
depend on FLOW_RSS.
ethtool_get_rxnfc() copies the whole struct ethtool_rxnfc from
userspace. So for ETHTOOL_GRXCLSRULE, the ring_cookie reported for an
RSS filter is whatever the caller happened to pass in, not the ring
offset stored with the rule. The drop and plain-ring branches do fill
it in.
The ethtool core rejects a non-zero ring with FLOW_RSS unless
cap_rss_rxnfc_adds is set, and bnge_ethtool_ops doesn't set it. So
base.rxq should always be 0 here, and the practical effect is that the
caller's input gets echoed back. Should the RSS branch also set
fs->ring_cookie = fltr->base.rxq, so the rule reads back the way it
was inserted?
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 09/10] bnge: add aRFS flow steering ndo support
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (7 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 08/10] bnge: add NTUPLE filter support in ethtool Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
2026-09-28 6:13 ` [net-next, v4 10/10] bnge: add cpu_rmap support for IRQ affinity Vikas Gupta
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
Introduce the ndo_rx_flow_steer callback (bnge_rx_flow_steer) to
support aRFS.
This callback creates an NTUPLE filter that steers matching packets
to the appropriate RX queue. The implementation dissects flow keys
from the SKB, creates an NTUPLE filter with the associated L2 filter,
inserts it into the NTUPLE hash table, and schedules the sp_task
workqueue via BNGE_RX_NTP_FLTR_SP_EVENT to program the filter
in the firmware.
Add bnge_cfg_ntp_filters() called from the periodic timer via sp_task
to handle deferred filter programming and expiration via
rps_may_expire_flow().
Note that, If the L2 filter is freed before sp_task programs the rule,
the captured firmware id is stale, the NTUPLE alloc either
fails (id gone) or binds to a recycled id for a different filter.
Either way the rule is dropped or aged out by rps_may_expire_flow()
and recreated -- best-effort aRFS self-heals, and no filter object
is dereferenced (only the id is copied).
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_ethtool.c | 2 +-
.../net/ethernet/broadcom/bnge/bnge_filter.c | 53 +++++++-
.../net/ethernet/broadcom/bnge/bnge_filter.h | 2 +
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 127 ++++++++++++++++++
.../net/ethernet/broadcom/bnge/bnge_netdev.h | 1 +
5 files changed, 180 insertions(+), 5 deletions(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
index 8e9cfbad98e0..08db761e4ca6 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
@@ -1189,7 +1189,7 @@ static u32 bnge_get_all_fltr_ids_rcu(struct bnge_net *bn,
head = &tbl[i];
hlist_for_each_entry_rcu(fltr, head, hlist) {
- if (!fltr->flags ||
+ if (!bnge_is_usr_fltr(fltr) ||
test_bit(BNGE_FLTR_FW_DELETED, &fltr->state))
continue;
ids[j++] = fltr->sw_id + offset;
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
index 764c967c6457..17a837fea79d 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
@@ -56,7 +56,7 @@ const struct bnge_flow_masks BNGE_FLOW_IPV4_MASK_ALL = {
},
};
-static bool bnge_is_usr_fltr(struct bnge_filter_base *fltr)
+bool bnge_is_usr_fltr(struct bnge_filter_base *fltr)
{
return (fltr->type == BNGE_FLTR_TYPE_L2 &&
fltr->flags & BNGE_ACT_RING_DST) ||
@@ -116,7 +116,7 @@ void bnge_del_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *nfltr)
hlist_del(&fltr->hlist);
bnge_del_usr_fltr_node(bn, fltr);
clear_bit(fltr->sw_id, bn->ntp_fltr_bmap);
- bn->ntp_fltr_count--;
+ WRITE_ONCE(bn->ntp_fltr_count, bn->ntp_fltr_count - 1);
kfree(fltr);
}
@@ -127,7 +127,7 @@ void bnge_del_ntp_filter_rcu(struct bnge_net *bn,
hlist_del_rcu(&fltr->base.hlist);
bnge_del_usr_fltr_node(bn, &fltr->base);
clear_bit(fltr->base.sw_id, bn->ntp_fltr_bmap);
- bn->ntp_fltr_count--;
+ WRITE_ONCE(bn->ntp_fltr_count, bn->ntp_fltr_count - 1);
spin_unlock_bh(&bn->ntp_fltr_lock);
kfree_rcu(fltr, base.rcu);
@@ -531,7 +531,7 @@ int bnge_insert_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *fltr,
hlist_add_head_rcu(&fltr->base.hlist, head);
bnge_insert_usr_fltr(bn, &fltr->base);
- bn->ntp_fltr_count++;
+ WRITE_ONCE(bn->ntp_fltr_count, bn->ntp_fltr_count + 1);
spin_unlock_bh(&bn->ntp_fltr_lock);
@@ -637,3 +637,48 @@ void bnge_clear_usr_fltrs(struct bnge_net *bn)
}
}
}
+
+void bnge_cfg_ntp_filters(struct bnge_net *bn)
+{
+#ifdef CONFIG_RFS_ACCEL
+ struct bnge_ntuple_filter *fltr, *tmp;
+ LIST_HEAD(install_list);
+ LIST_HEAD(expire_list);
+ int i, rc;
+
+ rcu_read_lock();
+ for (i = 0; i < BNGE_NTP_FLTR_HASH_SIZE; i++) {
+ hlist_for_each_entry_rcu(fltr, &bn->ntp_fltr_hash_tbl[i],
+ base.hlist) {
+ if (test_bit(BNGE_FLTR_VALID, &fltr->base.state)) {
+ if (fltr->base.flags & BNGE_ACT_NO_AGING)
+ continue;
+ if (rps_may_expire_flow(bn->netdev,
+ fltr->base.rxq,
+ fltr->flow_id,
+ fltr->base.sw_id))
+ list_add(&fltr->base.list_node,
+ &expire_list);
+ } else {
+ list_add(&fltr->base.list_node, &install_list);
+ }
+ }
+ }
+ rcu_read_unlock();
+
+ list_for_each_entry_safe(fltr, tmp, &install_list, base.list_node) {
+ list_del_init(&fltr->base.list_node);
+ rc = bnge_hwrm_cfa_ntuple_filter_alloc(bn->bd, fltr);
+ if (rc)
+ bnge_del_ntp_filter_rcu(bn, fltr);
+ else
+ set_bit(BNGE_FLTR_VALID, &fltr->base.state);
+ }
+
+ list_for_each_entry_safe(fltr, tmp, &expire_list, base.list_node) {
+ list_del_init(&fltr->base.list_node);
+ bnge_hwrm_cfa_ntuple_filter_free(bn->bd, fltr);
+ bnge_del_ntp_filter_rcu(bn, fltr);
+ }
+#endif
+}
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
index a0d961a01847..4b302c02dd19 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.h
@@ -96,6 +96,7 @@ struct bnge_ntuple_filter {
#define BNGE_FLTR_ID_INVALID cpu_to_le64(0xffffffffffffffffULL)
void bnge_free_ntp_fltrs(struct bnge_net *bn, bool skip_user_filters);
+bool bnge_is_usr_fltr(struct bnge_filter_base *fltr);
u32 bnge_get_ntp_filter_idx(struct bnge_net *bn, struct flow_keys *fkeys,
const struct sk_buff *skb);
int bnge_insert_ntp_filter(struct bnge_net *bn, struct bnge_ntuple_filter *fltr,
@@ -121,4 +122,5 @@ void bnge_del_ntp_filter_rcu(struct bnge_net *bn,
struct bnge_ntuple_filter *fltr);
void bnge_cfg_usr_fltrs(struct bnge_net *bn);
void bnge_clear_usr_fltrs(struct bnge_net *bn);
+void bnge_cfg_ntp_filters(struct bnge_net *bn);
#endif /* _BNGE_FILTER_H_ */
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index 19ec36b08765..de7818d2c2af 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -303,6 +303,12 @@ static void bnge_timer(struct timer_list *t)
if (BNGE_LINK_IS_UP(bd) && bn->stats_coal_ticks)
bnge_queue_sp_work(bn, BNGE_PERIODIC_STATS_SP_EVENT);
+#ifdef CONFIG_RFS_ACCEL
+ if ((bn->priv_flags & BNGE_NET_EN_NTUPLE) &&
+ READ_ONCE(bn->ntp_fltr_count))
+ bnge_queue_sp_work(bn, BNGE_RX_NTP_FLTR_SP_EVENT);
+#endif
+
mod_timer(&bn->timer, jiffies + bn->current_interval);
}
@@ -444,6 +450,9 @@ static void bnge_sp_task(struct work_struct *work)
bnge_init_ethtool_link_settings(bn);
}
+ if (test_and_clear_bit(BNGE_RX_NTP_FLTR_SP_EVENT, &bn->sp_event))
+ bnge_cfg_ntp_filters(bn);
+
netdev_unlock(bn->netdev);
}
@@ -3176,6 +3185,121 @@ static int bnge_set_features(struct net_device *dev, netdev_features_t features)
return 0;
}
+#ifdef CONFIG_RFS_ACCEL
+static __le64 bnge_lookup_l2_filter_from_key(struct bnge_net *bn,
+ struct bnge_l2_key *key)
+{
+ u32 idx;
+
+ idx = jhash2(&key->filter_key, BNGE_L2_KEY_SIZE, bn->hash_seed) &
+ BNGE_L2_FLTR_HASH_MASK;
+ return bnge_lookup_l2_filter_rcu(bn, key, idx);
+}
+
+static int bnge_rx_flow_steer(struct net_device *dev, const struct sk_buff *skb,
+ u16 rxq_index, u32 flow_id)
+{
+ struct ethhdr *eth = (struct ethhdr *)skb_mac_header(skb);
+ struct bnge_ntuple_filter *fltr, *new_fltr;
+ struct bnge_net *bn = netdev_priv(dev);
+ struct flow_keys *fkeys;
+ __le64 filter_id;
+ u32 flags, idx;
+ int rc = 0;
+
+ /* Hold rcu_read_lock() across the L2 filter lookup and check
+ * BNGE_STATE_OPEN inside it. bnge_close_core() clears the
+ * flag before bnge_shutdown_nic() frees the L2 filters via
+ * kfree_rcu(), so if we observe the device open here the
+ * filter cannot be freed under us for the duration of
+ * this critical section.
+ */
+ rcu_read_lock();
+ if (!test_bit(BNGE_STATE_OPEN, &bn->bd->state)) {
+ rcu_read_unlock();
+ return -EINVAL;
+ }
+
+ if (ether_addr_equal(dev->dev_addr, eth->h_dest)) {
+ struct bnge_vnic_info *vnic = &bn->vnic_info[BNGE_VNIC_DEFAULT];
+ struct bnge_l2_filter *l2_filter;
+
+ if (!vnic->uc_filter_count) {
+ rcu_read_unlock();
+ return -EINVAL;
+ }
+ l2_filter = vnic->l2_filters[0];
+ filter_id = l2_filter->base.filter_id;
+ } else {
+ struct bnge_l2_key key;
+
+ ether_addr_copy(key.dst_mac_addr, eth->h_dest);
+ key.vlan = 0;
+
+ filter_id = bnge_lookup_l2_filter_from_key(bn, &key);
+ if (filter_id == BNGE_FLTR_ID_INVALID) {
+ rcu_read_unlock();
+ return -EINVAL;
+ }
+ }
+ rcu_read_unlock();
+
+ new_fltr = kzalloc_obj(*new_fltr, GFP_ATOMIC);
+ if (!new_fltr)
+ return -ENOMEM;
+
+ fkeys = &new_fltr->fkeys;
+ if (!skb_flow_dissect_flow_keys(skb, fkeys, 0)) {
+ rc = -EPROTONOSUPPORT;
+ goto err_free;
+ }
+
+ if ((fkeys->basic.n_proto != htons(ETH_P_IP) &&
+ fkeys->basic.n_proto != htons(ETH_P_IPV6)) ||
+ (fkeys->basic.ip_proto != IPPROTO_TCP &&
+ fkeys->basic.ip_proto != IPPROTO_UDP)) {
+ rc = -EPROTONOSUPPORT;
+ goto err_free;
+ }
+ new_fltr->fmasks = BNGE_FLOW_IPV4_MASK_ALL;
+ if (fkeys->basic.n_proto == htons(ETH_P_IPV6))
+ new_fltr->fmasks = BNGE_FLOW_IPV6_MASK_ALL;
+
+ flags = fkeys->control.flags;
+ if (flags & FLOW_DIS_IS_FRAGMENT) {
+ rc = -EPROTONOSUPPORT;
+ goto err_free;
+ }
+
+ new_fltr->l2_filter_id = filter_id;
+
+ idx = bnge_get_ntp_filter_idx(bn, fkeys, skb);
+ rcu_read_lock();
+ fltr = bnge_lookup_ntp_filter_from_idx(bn, new_fltr, idx);
+ /* Filter already exists; return its id. A stale filter (queue
+ * changed) is freed later via rps_may_expire_flow() and recreated.
+ */
+ if (fltr) {
+ rc = fltr->base.sw_id;
+ rcu_read_unlock();
+ goto err_free;
+ }
+ rcu_read_unlock();
+
+ new_fltr->flow_id = flow_id;
+ new_fltr->base.rxq = rxq_index;
+ rc = bnge_insert_ntp_filter(bn, new_fltr, idx);
+ if (!rc) {
+ bnge_queue_sp_work(bn, BNGE_RX_NTP_FLTR_SP_EVENT);
+ return new_fltr->base.sw_id;
+ }
+
+err_free:
+ kfree(new_fltr);
+ return rc;
+}
+#endif
+
static const struct net_device_ops bnge_netdev_ops = {
.ndo_open = bnge_open,
.ndo_stop = bnge_close,
@@ -3185,6 +3309,9 @@ static const struct net_device_ops bnge_netdev_ops = {
.ndo_features_check = bnge_features_check,
.ndo_fix_features = bnge_fix_features,
.ndo_set_features = bnge_set_features,
+#ifdef CONFIG_RFS_ACCEL
+ .ndo_rx_flow_steer = bnge_rx_flow_steer,
+#endif
};
static void bnge_init_mac_addr(struct bnge_dev *bd)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
index 221dfcd81e78..bb68762f62a8 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.h
@@ -282,6 +282,7 @@ enum bnge_sp_event {
BNGE_LINK_CFG_CHANGE_SP_EVENT,
BNGE_UPDATE_PHY_SP_EVENT,
BNGE_PERIODIC_STATS_SP_EVENT,
+ BNGE_RX_NTP_FLTR_SP_EVENT,
};
#define BNGE_NTP_FLTR_HASH_SIZE 512
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 09/10] bnge: add aRFS flow steering ndo support
2026-09-28 6:13 ` [net-next, v4 09/10] bnge: add aRFS flow steering ndo support Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found 4 potential
issue(s) to consider.
Critical: 0 · High: 0 · Medium: 1 · Low: 3
If you believe none of the issues are legitimate, please reply to each
issue of Medium severity or higher and explain why the AI is wrong.
If any of them are legitimate and you plan to respin, please let patchwork
know by including "pw-bot: cr" as a separate line at the end of your reply
(one such reply per series is enough).
- [Medium] The expire loop in bnge_cfg_ntp_filters()
(bnge_filter.c:680-681) ignores the return value of…
- [Low] Baseline decision: false.
- [Low] The ethtool rxnfc GET paths disagree about aRFS filters.
- [Low] In bnge_rx_flow_steer(), when an NTUPLE filter for the same tuple
already exists, the callback returns its sw_id without updating…
--
bnge: add aRFS flow steering ndo support
This adds bnge_rx_flow_steer() as the ndo_rx_flow_steer callback. It
builds an NTUPLE filter from the dissected flow keys and inserts it into
the NTUPLE hash table. Firmware programming and aging are deferred to
bnge_cfg_ntp_filters(), which runs from sp_task.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> index 8e9cfbad98e04..08db761e4ca60 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_ethtool.c
> @@ -1189,7 +1189,7 @@ static u32 bnge_get_all_fltr_ids_rcu(struct bnge_net *bn,
>
> head = &tbl[i];
> hlist_for_each_entry_rcu(fltr, head, hlist) {
> - if (!fltr->flags ||
> + if (!bnge_is_usr_fltr(fltr) ||
> test_bit(BNGE_FLTR_FW_DELETED, &fltr->state))
> continue;
> ids[j++] = fltr->sw_id + offset;
[Severity: Low]
With this change, GRXCLSRLALL uses bnge_is_usr_fltr() to skip aRFS
filters. Does bnge_get_one_fltr_rcu() need the same change? It still uses
the old test:
bnge_get_one_fltr_rcu() {
...
hlist_for_each_entry_rcu(fltr, head, hlist) {
if (fltr->flags && fltr->sw_id + offset == id)
return fltr;
}
...
}
bnge_insert_ntp_filter() always sets BNGE_ACT_RING_DST on aRFS filters,
so their flags are never zero. A GRXCLSRULE query for location
BNGE_MAX_L2_FLTRS + sw_id (for example ethtool -n <dev> rule <loc>) would
then report an aRFS filter as a configured classification rule, with its
5-tuple and ring_cookie.
That same filter is missing from GRXCLSRLALL and is not counted in
GRXCLSRLCNT (bn->user_fltr_count). bnge_srxclsrldel() also refuses to
delete it:
if (!(fltr->base.flags & BNGE_ACT_NO_AGING)) {
rcu_read_unlock();
return -EINVAL;
}
This still seems to be present at the end of the series.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> index 764c967c64572..17a837fea79d7 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_filter.c
[ ... ]
> @@ -637,3 +637,48 @@ void bnge_clear_usr_fltrs(struct bnge_net *bn)
> }
> }
> }
> +
> +void bnge_cfg_ntp_filters(struct bnge_net *bn)
> +{
[ ... ]
> + list_for_each_entry_safe(fltr, tmp, &expire_list, base.list_node) {
> + list_del_init(&fltr->base.list_node);
> + bnge_hwrm_cfa_ntuple_filter_free(bn->bd, fltr);
> + bnge_del_ntp_filter_rcu(bn, fltr);
[Severity: Medium]
Should the return value of bnge_hwrm_cfa_ntuple_filter_free() be
checked here?
It can fail before anything reaches firmware. For example,
bnge_hwrm_req_init()->bnge_hwrm_req_create() returns -ENOMEM when
dma_pool_alloc() fails. bnge_hwrm_req_send() can also return a firmware
error or a timeout.
In either case the firmware keeps the CFA ntuple rule and keeps steering
the tuple to the old ring. Meanwhile bnge_del_ntp_filter_rcu() frees the
only copy of base.filter_id and releases the sw_id bit.
Does this leak the firmware CFA entry until the owning L2 filter or VNIC
is torn down? If the flow is steered again, a second firmware rule for
the same tuple would be allocated. bnxt has the same pattern in its
aging path.
> + }
> +#endif
> +}
[ ... ]
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index 19ec36b08765c..de7818d2c2af2 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
[ ... ]
> @@ -3176,6 +3185,121 @@ static int bnge_set_features(struct net_device *dev, netdev_features_t features)
[ ... ]
> +static int bnge_rx_flow_steer(struct net_device *dev, const struct sk_buff *skb,
> + u16 rxq_index, u32 flow_id)
> +{
[ ... ]
> + idx = bnge_get_ntp_filter_idx(bn, fkeys, skb);
> + rcu_read_lock();
> + fltr = bnge_lookup_ntp_filter_from_idx(bn, new_fltr, idx);
> + /* Filter already exists; return its id. A stale filter (queue
> + * changed) is freed later via rps_may_expire_flow() and recreated.
> + */
> + if (fltr) {
> + rc = fltr->base.sw_id;
> + rcu_read_unlock();
> + goto err_free;
> + }
[Severity: Low]
Is the "freed later via rps_may_expire_flow() and recreated" part of
this comment always true?
bnge_fltr_match() does not compare rxq. Suppose flow F is steered to
queue B and then moves to a CPU that maps to queue C. set_rps_cpu() calls
this function with rxq_index C and gets F's sw_id back. F's base.rxq and
flow_id are left unchanged, and nothing is reprogrammed.
The core then records F in C's rps_dev_flow entry and clears it from
B's entry, while the hardware still delivers the flow to B.
On a later pass, bnge_cfg_ntp_filters() calls rps_may_expire_flow() with
the stale rxq B. It sees that B's entry no longer holds F and frees the
rule. Packets then arrive on some RSS queue X.
When set_rps_cpu() runs for X, C's entry still holds F and can still
look active:
net/core/dev.c:set_rps_cpu() {
...
if (rps_flow_is_active(tmp_rflow, rps_tag_to_log(q_tag_ptr), tmp_cpu)) {
if (hash != READ_ONCE(tmp_rflow->hash) ||
next_cpu == tmp_cpu)
goto out;
}
...
}
In that case ndo_rx_flow_steer is not called and X's entry gets
cpu = next_cpu. After that, get_rps_cpu() stops calling set_rps_cpu()
because tcpu == next_cpu.
Can the flow then stay without a hardware aRFS rule until the application
moves CPU again? RPS still steers it in software, so the impact is on
performance. It also depends on timing, namely how far the target CPU's
input_queue_head has advanced. bnxt_rx_flow_steer() has the same
existing-filter branch.
Should this branch handle a changed rxq_index instead of returning the
old sw_id unchanged?
[ ... ]
> @@ -3185,6 +3309,9 @@ static const struct net_device_ops bnge_netdev_ops = {
> .ndo_features_check = bnge_features_check,
> .ndo_fix_features = bnge_fix_features,
> .ndo_set_features = bnge_set_features,
> +#ifdef CONFIG_RFS_ACCEL
> + .ndo_rx_flow_steer = bnge_rx_flow_steer,
> +#endif
> };
[Severity: Low]
Can bnge_rx_flow_steer() actually be called at this point in the
series? set_rps_cpu() in net/core/dev.c skips the callback when there is
no rx_cpu_rmap:
if (!skb_rx_queue_recorded(skb) || !dev->rx_cpu_rmap ||
!(dev->features & NETIF_F_NTUPLE))
goto out;
bnge does not set dev->rx_cpu_rmap yet, so the new aRFS code does not
run at this commit.
A later patch in this series, "bnge: add cpu_rmap support for IRQ
affinity", fixes this. It adds alloc_irq_cpu_rmap(bd->rx_nr_rings), calls
irq_cpu_rmap_add() in bnge_request_irq() for NAPIs that have an rx_ring,
and frees the map on teardown. So the callback is reachable by the end of
the series, and nothing breaks at this commit.
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread
* [net-next, v4 10/10] bnge: add cpu_rmap support for IRQ affinity
2026-09-28 6:12 [net-next, v4 00/10] add features to bnge Vikas Gupta
` (8 preceding siblings ...)
2026-09-28 6:13 ` [net-next, v4 09/10] bnge: add aRFS flow steering ndo support Vikas Gupta
@ 2026-09-28 6:13 ` Vikas Gupta
2026-10-01 9:13 ` netdev-bot+sashiko
9 siblings, 1 reply; 21+ messages in thread
From: Vikas Gupta @ 2026-09-28 6:13 UTC (permalink / raw)
To: davem, edumazet, kuba, pabeni, andrew+netdev, horms
Cc: netdev, linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde, Vikas Gupta
When CONFIG_RFS_ACCEL is enabled, allocate and populate a cpu_rmap
for the RX IRQ vectors so that the kernel's aRFS infrastructure can
correctly map flows to the RX queues serviced by the nearest CPU.
This ensures that NTUPLE rules created by the kernel steer packets
to the optimal queue based on CPU/IRQ affinity.
Signed-off-by: Vikas Gupta <vikas.gupta@broadcom.com>
Reviewed-by: Bhargava Chenna Marreddy <bhargava.marreddy@broadcom.com>
Reviewed-by: Dharmender Garg <dharmender.garg@broadcom.com>
---
.../net/ethernet/broadcom/bnge/bnge_netdev.c | 42 ++++++++++++++++++-
1 file changed, 41 insertions(+), 1 deletion(-)
diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
index de7818d2c2af..a4156a12c871 100644
--- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
+++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
@@ -18,6 +18,7 @@
#include <net/ip.h>
#include <linux/skbuff.h>
#include <net/page_pool/helpers.h>
+#include <linux/cpu_rmap.h>
#include "bnge.h"
#include "bnge_hwrm.h"
@@ -2437,10 +2438,23 @@ static int bnge_setup_interrupts(struct bnge_net *bn)
{
struct net_device *dev = bn->netdev;
struct bnge_dev *bd = bn->bd;
+ int rc;
bnge_setup_msix(bn);
- return netif_set_real_num_queues(dev, bd->tx_nr_rings, bd->rx_nr_rings);
+ rc = netif_set_real_num_queues(dev, bd->tx_nr_rings, bd->rx_nr_rings);
+ if (rc)
+ return rc;
+
+#ifdef CONFIG_RFS_ACCEL
+ if (bn->priv_flags & BNGE_NET_EN_NTUPLE) {
+ dev->rx_cpu_rmap = alloc_irq_cpu_rmap(bd->rx_nr_rings);
+ if (!dev->rx_cpu_rmap)
+ return -ENOMEM;
+ }
+#endif
+
+ return rc;
}
static void bnge_hwrm_resource_free(struct bnge_net *bn, bool close_path)
@@ -2456,6 +2470,11 @@ static void bnge_free_irq(struct bnge_net *bn)
struct bnge_irq *irq;
int i;
+#ifdef CONFIG_RFS_ACCEL
+ free_irq_cpu_rmap(bn->netdev->rx_cpu_rmap);
+ bn->netdev->rx_cpu_rmap = NULL;
+#endif
+
for (i = 0; i < bd->nq_nr_rings; i++) {
int map_idx = bnge_cp_num_to_irq_num(bn, i);
@@ -2475,6 +2494,7 @@ static void bnge_free_irq(struct bnge_net *bn)
static int bnge_request_irq(struct bnge_net *bn)
{
+ struct cpu_rmap *rmap = NULL;
struct bnge_dev *bd = bn->bd;
int i, rc;
@@ -2483,10 +2503,30 @@ static int bnge_request_irq(struct bnge_net *bn)
netdev_err(bn->netdev, "bnge_setup_interrupts err: %d\n", rc);
return rc;
}
+
+#ifdef CONFIG_RFS_ACCEL
+ rmap = bn->netdev->rx_cpu_rmap;
+#endif
+
for (i = 0; i < bd->nq_nr_rings; i++) {
int map_idx = bnge_cp_num_to_irq_num(bn, i);
struct bnge_irq *irq = &bd->irq_tbl[map_idx];
+ if (IS_ENABLED(CONFIG_RFS_ACCEL) &&
+ rmap && bn->bnapi[i]->rx_ring) {
+ rc = irq_cpu_rmap_add(rmap, irq->vector);
+ if (rc) {
+ netdev_warn(bn->netdev,
+ "failed adding irq rmap for ring %d, disabling aRFS\n",
+ i);
+#ifdef CONFIG_RFS_ACCEL
+ free_irq_cpu_rmap(bn->netdev->rx_cpu_rmap);
+ bn->netdev->rx_cpu_rmap = NULL;
+#endif
+ rmap = NULL;
+ }
+ }
+
rc = request_irq(irq->vector, irq->handler, 0, irq->name,
bn->bnapi[i]);
if (rc)
--
2.52.0
^ permalink raw reply [flat|nested] 21+ messages in thread* Re: [net-next, v4 10/10] bnge: add cpu_rmap support for IRQ affinity
2026-09-28 6:13 ` [net-next, v4 10/10] bnge: add cpu_rmap support for IRQ affinity Vikas Gupta
@ 2026-10-01 9:13 ` netdev-bot+sashiko
0 siblings, 0 replies; 21+ messages in thread
From: netdev-bot+sashiko @ 2026-10-01 9:13 UTC (permalink / raw)
To: vikas.gupta
Cc: davem, edumazet, kuba, pabeni, andrew+netdev, horms, netdev,
linux-kernel, bhargava.marreddy, rahul-rg.gupta,
vsrama-krishna.nemani, rajashekar.hudumula, dharmender.garg,
ajit.khaparde
Thank you for your contribution! Sashiko AI review found more severe
issue(s) in other patches of this series. This mail is only an FYI, in case
you want to address the nit picks below when you respin.
Low priority / nit picks for your consideration:
- [Low] Now that this patch publishes rx_cpu_rmap, the core starts
passing target queues to bnge_rx_flow_steer().
--
bnge: add cpu_rmap support for IRQ affinity
When CONFIG_RFS_ACCEL is enabled, this patch allocates dev->rx_cpu_rmap
and fills it with the RX IRQ vectors. The kernel's aRFS code can then map
flows to the RX queue serviced by the nearest CPU and call
ndo_rx_flow_steer() for that queue.
> diff --git a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> index de7818d2c2af2..a4156a12c8714 100644
> --- a/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
> +++ b/drivers/net/ethernet/broadcom/bnge/bnge_netdev.c
[ ... ]
> @@ -2437,10 +2438,23 @@ static int bnge_setup_interrupts(struct bnge_net *bn)
> {
> struct net_device *dev = bn->netdev;
> struct bnge_dev *bd = bn->bd;
> + int rc;
>
> bnge_setup_msix(bn);
>
> - return netif_set_real_num_queues(dev, bd->tx_nr_rings, bd->rx_nr_rings);
> + rc = netif_set_real_num_queues(dev, bd->tx_nr_rings, bd->rx_nr_rings);
> + if (rc)
> + return rc;
> +
> +#ifdef CONFIG_RFS_ACCEL
> + if (bn->priv_flags & BNGE_NET_EN_NTUPLE) {
> + dev->rx_cpu_rmap = alloc_irq_cpu_rmap(bd->rx_nr_rings);
[Severity: Low]
Once dev->rx_cpu_rmap is set here, the core starts calling
bnge_rx_flow_steer() with CPU-derived target queues. That function came
from an earlier patch in this series ("bnge: add aRFS flow steering ndo
support"), but it can only be reached after this patch.
When bnge_rx_flow_steer() finds an existing filter for the flow, it
returns that filter's id. It never compares fltr->base.rxq with
rxq_index:
drivers/net/ethernet/broadcom/bnge/bnge_netdev.c:bnge_rx_flow_steer() {
...
fltr = bnge_lookup_ntp_filter_from_idx(bn, new_fltr, idx);
/* Filter already exists; return its id. A stale filter (queue
* changed) is freed later via rps_may_expire_flow() and recreated.
*/
if (fltr) {
rc = fltr->base.sw_id;
rcu_read_unlock();
goto err_free;
}
...
}
set_rps_cpu() counts any non-negative return as success. It stores the id
in the target queue's flow table and clears the old queue's entry:
net/core/dev.c:set_rps_cpu() {
...
rc = dev->netdev_ops->ndo_rx_flow_steer(dev, skb,
rxq_index, flow_id);
if (rc < 0)
goto out;
old_rflow = rflow;
rflow = tmp_rflow;
WRITE_ONCE(rflow->filter, rc);
WRITE_ONCE(rflow->hash, hash);
if (old_rflow->filter == rc)
WRITE_ONCE(old_rflow->filter, RPS_NO_FILTER);
...
}
The hardware filter keeps steering to the old queue anyway.
Next, bnge_cfg_ntp_filters() calls rps_may_expire_flow() with
fltr->base.rxq, which is still the old queue. The core has already cleared
that slot, so the filter is expired and freed while the flow is still
active.
Is the "recreated" part of that comment actually guaranteed? The target
queue's entry still holds the sw_id, with cpu equal to next_cpu. After the
flow falls back to RSS, set_rps_cpu() can take this branch without calling
ndo_rx_flow_steer():
if (READ_ONCE(tmp_rflow->filter) != RPS_NO_FILTER) {
if (rps_flow_is_active(tmp_rflow,
rps_tag_to_log(q_tag_ptr),
tmp_cpu)) {
if (hash != READ_ONCE(tmp_rflow->hash) ||
next_cpu == tmp_cpu)
goto out;
}
}
That writes next_cpu into the RSS queue's entry. From then on,
get_rps_cpu() sees tcpu == next_cpu and never calls set_rps_cpu() again.
Could the flow then stay on software RFS, with no hardware filter, until
the application moves to another CPU?
mlx5 and sfc update the existing rule's queue in this case. bnxt uses the
same pattern as bnge.
Would it make sense for the existing-filter path to either reprogram the
filter to rxq_index, or return an error so the core doesn't move the
filter id to the new queue?
> + if (!dev->rx_cpu_rmap)
> + return -ENOMEM;
> + }
> +#endif
> +
> + return rc;
> }
[ ... ]
--
Sashiko AI review · https://netdev-ai.bots.linux.dev/sashiko/#/patchset/20260928061307.1172344-1-vikas.gupta%40broadcom.com
^ permalink raw reply [flat|nested] 21+ messages in thread