mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Tariq Toukan <tariqt@nvidia.com>
To: Andrew Lunn <andrew+netdev@lunn.ch>,
	"David S. Miller" <davem@davemloft.net>,
	Eric Dumazet <edumazet@google.com>,
	Jakub Kicinski <kuba@kernel.org>, <netdev@vger.kernel.org>,
	Paolo Abeni <pabeni@redhat.com>
Cc: Cosmin Ratiu <cratiu@nvidia.com>, Gal Pressman <gal@nvidia.com>,
	"Leon Romanovsky" <leon@kernel.org>,
	<linux-kernel@vger.kernel.org>, <linux-rdma@vger.kernel.org>,
	Mark Bloch <mbloch@nvidia.com>,
	Saeed Mahameed <saeedm@nvidia.com>,
	Tariq Toukan <tariqt@nvidia.com>,
	Yael Chemla <ychemla@nvidia.com>
Subject: [PATCH net-next] net/mlx5: E-Switch, defer fwd2vport egress ACL allocation
Date: Thu, 23 Jul 2026 10:04:27 +0300	[thread overview]
Message-ID: <20260723070427.1861502-1-tariqt@nvidia.com> (raw)

From: Yael Chemla <ychemla@nvidia.com>

On every VF/SF vport enable, esw_acl_egress_ofld_setup() allocates an
egress ACL flow table and a fwd_grp whenever the device supports
egress_acl_forward_to_vport. The only consumer of that group is the
active/passive fwd2vport rule installed when two representor netdevs
are bonded - a path that almost never fires. As a result, hosts with
many VFs/SFs pay a per-vport flow table and flow group cost for a
feature most ports never use.

Defer the flow table and fwd_grp creation to the moment they are
actually needed, when mlx5e_rep_esw_bond_netevent() drives
mlx5_esw_acl_egress_vport_bond() for the passive vport:

- esw_acl_egress_ofld_setup() now returns early unless
  prio_tag_required is set. When prio_tag_required is set the
  flow table is still allocated eagerly for the VLAN pop rule, and
  its size is grown by one when fwd2vport is supported so the lazy
  fwd_grp can later be added without re-creating the table. Only
  the VLAN group is built up-front.

- A new helper, esw_acl_egress_ofld_fwd2vport_setup(), allocates
  the egress ACL flow table (size 1) and the fwd_grp on demand,
  and rolls back the flow table if group creation fails and the
  helper had just allocated it. Existing cleanup paths
  (esw_acl_egress_ofld_cleanup() -> *_groups_destroy() /
  *_table_destroy()) already tolerate NULL fields, so vport
  disable continues to free everything that was actually
  allocated.

- mlx5_esw_acl_egress_vport_bond() calls the helper for the
  passive vport before installing the fwd2vport rule. The active
  vport does not need the flow table on its own: with a NULL
  fwd_dest, esw_acl_egress_ofld_rules_create() is a no-op unless
  prio_tag_required is set, in which case the eager path already
  built the table.

mlx5_esw_acl_egress_vport_bond() and mlx5_esw_acl_egress_vport_unbond()
now take esw->state_lock for the duration of the operation, because
they may mutate vport->egress.acl, which is also written by the vport
enable/disable path under the same lock.

Signed-off-by: Yael Chemla <ychemla@nvidia.com>
Reviewed-by: Cosmin Ratiu <cratiu@nvidia.com>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
 .../mellanox/mlx5/core/esw/acl/egress_ofld.c  | 103 ++++++++++++------
 1 file changed, 68 insertions(+), 35 deletions(-)

diff --git a/drivers/net/ethernet/mellanox/mlx5/core/esw/acl/egress_ofld.c b/drivers/net/ethernet/mellanox/mlx5/core/esw/acl/egress_ofld.c
index 24b1ca4e4ff8..981ce4463582 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/esw/acl/egress_ofld.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/esw/acl/egress_ofld.c
@@ -113,8 +113,8 @@ static void esw_acl_egress_ofld_rules_destroy(struct mlx5_vport *vport)
 	esw_acl_egress_ofld_bounce_rules_destroy(vport);
 }
 
-static int esw_acl_egress_ofld_groups_create(struct mlx5_eswitch *esw,
-					     struct mlx5_vport *vport)
+static int esw_acl_egress_ofld_fwd_grp_create(struct mlx5_eswitch *esw,
+					      struct mlx5_vport *vport)
 {
 	int inlen = MLX5_ST_SZ_BYTES(create_flow_group_in);
 	struct mlx5_flow_group *fwd_grp;
@@ -122,22 +122,12 @@ static int esw_acl_egress_ofld_groups_create(struct mlx5_eswitch *esw,
 	u32 flow_index = 0;
 	int ret = 0;
 
-	if (MLX5_CAP_GEN(esw->dev, prio_tag_required)) {
-		ret = esw_acl_egress_vlan_grp_create(esw, vport);
-		if (ret)
-			return ret;
-
+	if (MLX5_CAP_GEN(esw->dev, prio_tag_required))
 		flow_index++;
-	}
-
-	if (!mlx5_esw_acl_egress_fwd2vport_supported(esw))
-		goto out;
 
 	flow_group_in = kvzalloc(inlen, GFP_KERNEL);
-	if (!flow_group_in) {
-		ret = -ENOMEM;
-		goto fwd_grp_err;
-	}
+	if (!flow_group_in)
+		return -ENOMEM;
 
 	/* This group holds 1 FTE to forward all packets to other vport
 	 * when bond vports is supported.
@@ -150,16 +140,15 @@ static int esw_acl_egress_ofld_groups_create(struct mlx5_eswitch *esw,
 		esw_warn(esw->dev,
 			 "Failed to create vport[%d] egress fwd2vport flow group, err(%d)\n",
 			 vport->vport, ret);
-		kvfree(flow_group_in);
-		goto fwd_grp_err;
+		goto out;
 	}
 	vport->egress.offloads.fwd_grp = fwd_grp;
-	kvfree(flow_group_in);
-	return 0;
+	esw_debug(esw->dev,
+		  "lazy-created fwd_grp for vport %d (flow_index=%u)\n",
+		  vport->vport, flow_index);
 
-fwd_grp_err:
-	esw_acl_egress_vlan_grp_destroy(vport);
 out:
+	kvfree(flow_group_in);
 	return ret;
 }
 
@@ -185,22 +174,23 @@ static bool esw_acl_egress_needed(struct mlx5_eswitch *esw, u16 vport_num)
 
 int esw_acl_egress_ofld_setup(struct mlx5_eswitch *esw, struct mlx5_vport *vport)
 {
-	int table_size = 0;
+	int table_size = 1;
 	int err;
 
-	if (!mlx5_esw_acl_egress_fwd2vport_supported(esw) &&
-	    !MLX5_CAP_GEN(esw->dev, prio_tag_required))
+	if (!esw_acl_egress_needed(esw, vport->vport))
 		return 0;
 
-	if (!esw_acl_egress_needed(esw, vport->vport))
+	/* The fwd2vport FT/group is created lazily on bond events; if prio_tag
+	 * is not required, skip eager FT allocation here entirely.
+	 */
+	if (!MLX5_CAP_GEN(esw->dev, prio_tag_required))
 		return 0;
 
 	esw_acl_egress_ofld_rules_destroy(vport);
 
+	/* Reserve an extra FTE so the fwd_grp can be added lazily later. */
 	if (mlx5_esw_acl_egress_fwd2vport_supported(esw))
 		table_size++;
-	if (MLX5_CAP_GEN(esw->dev, prio_tag_required))
-		table_size++;
 	vport->egress.acl = esw_acl_table_create(esw, vport,
 						 MLX5_FLOW_NAMESPACE_ESW_EGRESS, table_size);
 	if (IS_ERR(vport->egress.acl)) {
@@ -210,21 +200,21 @@ int esw_acl_egress_ofld_setup(struct mlx5_eswitch *esw, struct mlx5_vport *vport
 	}
 	vport->egress.type = VPORT_EGRESS_ACL_TYPE_DEFAULT;
 
-	err = esw_acl_egress_ofld_groups_create(esw, vport);
+	err = esw_acl_egress_vlan_grp_create(esw, vport);
 	if (err)
-		goto group_err;
+		goto table_err;
 
 	esw_debug(esw->dev, "vport[%d] configure egress rules\n", vport->vport);
 
 	err = esw_acl_egress_ofld_rules_create(esw, vport, NULL);
 	if (err)
-		goto rules_err;
+		goto vlan_grp_err;
 
 	return 0;
 
-rules_err:
-	esw_acl_egress_ofld_groups_destroy(vport);
-group_err:
+vlan_grp_err:
+	esw_acl_egress_vlan_grp_destroy(vport);
+table_err:
 	esw_acl_egress_table_destroy(vport);
 	return err;
 }
@@ -236,18 +226,53 @@ void esw_acl_egress_ofld_cleanup(struct mlx5_vport *vport)
 	esw_acl_egress_table_destroy(vport);
 }
 
+/* Lazily allocate the egress ACL table and fwd_grp for a vport that is about
+ * to receive a fwd2vport rule. The table is otherwise created eagerly only
+ * when prio_tag is required.
+ */
+static int esw_acl_egress_ofld_fwd2vport_setup(struct mlx5_eswitch *esw,
+					       struct mlx5_vport *vport)
+{
+	struct mlx5_flow_table *acl = NULL;
+	int err;
+
+	if (!vport->egress.acl) {
+		acl = esw_acl_table_create(esw, vport,
+					   MLX5_FLOW_NAMESPACE_ESW_EGRESS, 1);
+		if (IS_ERR(acl))
+			return PTR_ERR(acl);
+		vport->egress.acl = acl;
+		vport->egress.type = VPORT_EGRESS_ACL_TYPE_DEFAULT;
+	}
+
+	if (vport->egress.offloads.fwd_grp)
+		return 0;
+
+	err = esw_acl_egress_ofld_fwd_grp_create(esw, vport);
+	if (err && acl)
+		esw_acl_egress_table_destroy(vport);
+	return err;
+}
+
 int mlx5_esw_acl_egress_vport_bond(struct mlx5_eswitch *esw, u16 active_vport_num,
 				   u16 passive_vport_num)
 {
 	struct mlx5_vport *passive_vport = mlx5_eswitch_get_vport(esw, passive_vport_num);
 	struct mlx5_vport *active_vport = mlx5_eswitch_get_vport(esw, active_vport_num);
 	struct mlx5_flow_destination fwd_dest = {};
+	int err;
 
 	if (IS_ERR(active_vport))
 		return PTR_ERR(active_vport);
 	if (IS_ERR(passive_vport))
 		return PTR_ERR(passive_vport);
 
+	mutex_lock(&esw->state_lock);
+
+	err = esw_acl_egress_ofld_fwd2vport_setup(esw, passive_vport);
+	if (err)
+		goto unlock;
+
 	/* Cleanup and recreate rules WITHOUT fwd2vport of active vport */
 	esw_acl_egress_ofld_rules_destroy(active_vport);
 	esw_acl_egress_ofld_rules_create(esw, active_vport, NULL);
@@ -259,16 +284,24 @@ int mlx5_esw_acl_egress_vport_bond(struct mlx5_eswitch *esw, u16 active_vport_nu
 	fwd_dest.vport.vhca_id = MLX5_CAP_GEN(esw->dev, vhca_id);
 	fwd_dest.vport.flags = MLX5_FLOW_DEST_VPORT_VHCA_ID;
 
-	return esw_acl_egress_ofld_rules_create(esw, passive_vport, &fwd_dest);
+	err = esw_acl_egress_ofld_rules_create(esw, passive_vport, &fwd_dest);
+
+unlock:
+	mutex_unlock(&esw->state_lock);
+	return err;
 }
 
 int mlx5_esw_acl_egress_vport_unbond(struct mlx5_eswitch *esw, u16 vport_num)
 {
 	struct mlx5_vport *vport = mlx5_eswitch_get_vport(esw, vport_num);
+	int err;
 
 	if (IS_ERR(vport))
 		return PTR_ERR(vport);
 
+	mutex_lock(&esw->state_lock);
 	esw_acl_egress_ofld_rules_destroy(vport);
-	return esw_acl_egress_ofld_rules_create(esw, vport, NULL);
+	err = esw_acl_egress_ofld_rules_create(esw, vport, NULL);
+	mutex_unlock(&esw->state_lock);
+	return err;
 }

base-commit: 1df10cef2d1e7f9f2fb7eddb67fc70d3abf101f9
-- 
2.44.0


             reply	other threads:[~2026-07-23  7:05 UTC|newest]

Thread overview: 2+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-07-23  7:04 Tariq Toukan [this message]
2026-07-28  1:30 ` patchwork-bot+netdevbpf

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260723070427.1861502-1-tariqt@nvidia.com \
    --to=tariqt@nvidia.com \
    --cc=andrew+netdev@lunn.ch \
    --cc=cratiu@nvidia.com \
    --cc=davem@davemloft.net \
    --cc=edumazet@google.com \
    --cc=gal@nvidia.com \
    --cc=kuba@kernel.org \
    --cc=leon@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-rdma@vger.kernel.org \
    --cc=mbloch@nvidia.com \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    --cc=saeedm@nvidia.com \
    --cc=ychemla@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®