mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Tariq Toukan <tariqt@nvidia.com>
To: Andrew Lunn <andrew+netdev@lunn.ch>,
	"David S. Miller" <davem@davemloft.net>,
	Eric Dumazet <edumazet@kernel.org>,
	Jakub Kicinski <kuba@kernel.org>, <netdev@vger.kernel.org>,
	Paolo Abeni <pabeni@redhat.com>
Cc: Andrea Parri <parri.andrea@gmail.com>,
	Boris Pismenny <borisp@nvidia.com>,
	Carolina Jubran <cjubran@nvidia.com>,
	Cosmin Ratiu <cratiu@nvidia.com>,
	Dragos Tatulea <dtatulea@nvidia.com>,
	Fernando Fernandez Mancera <fmancera@suse.de>,
	Gal Pressman <gal@nvidia.com>, Jianbo Liu <jianbol@nvidia.com>,
	Kees Cook <kees@kernel.org>, Leon Romanovsky <leon@kernel.org>,
	open list <linux-kernel@vger.kernel.org>,
	<linux-rdma@vger.kernel.org>, Mark Bloch <mbloch@nvidia.com>,
	Parav Pandit <parav@nvidia.com>,
	Patrisious Haddad <phaddad@nvidia.com>,
	Raed Salem <raeds@nvidia.com>, Roi Dayan <roid@nvidia.com>,
	Saeed Mahameed <saeedm@nvidia.com>,
	Steffen Klassert <steffen.klassert@secunet.com>,
	"Tariq Toukan" <tariqt@nvidia.com>
Subject: [PATCH net V2 2/4] net/mlx5e: ipsec: Block eswitch mode changes before accessing priv->ipsec
Date: Wed, 30 Sep 2026 15:11:17 +0300	[thread overview]
Message-ID: <20260930121119.141953-3-tariqt@nvidia.com> (raw)
In-Reply-To: <20260930121119.141953-1-tariqt@nvidia.com>

From: Cosmin Ratiu <cratiu@nvidia.com>

mlx5e_xfrm_add_state() reads priv->ipsec and validates mode-dependent
capabilities before blocking eswitch mode changes. A concurrent profile
change can free the saved IPsec context and cause use-after-free.

Move the mode block before saving the IPsec context and validating the
state, and release it on all error paths. Retain an early availability
check to preserve software fallback when IPsec is unavailable, and check
the context again after taking the mode block. Keep the atomic
acquire-placeholder path exempt, since it creates no hardware state and
cannot take sleeping locks.

As in policy creation, do not check eswitch users when taking this
temporary mode block. This allows states to reuse existing IPsec tables
when TC rules exist on VF representors, instead of rejecting them
unconditionally. New RX/TX tables still check eswitch users, and the
TC/IPsec exclusion counters still reject conflicting packet offloads.

Fixes: 22239eb258bc ("net/mlx5e: Prevent tunnel reformat when tunnel mode not allowed")
Signed-off-by: Cosmin Ratiu <cratiu@nvidia.com>
Reviewed-by: Dragos Tatulea <dtatulea@nvidia.com>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
 .../mellanox/mlx5/core/en_accel/ipsec.c       | 50 +++++++++++++------
 1 file changed, 34 insertions(+), 16 deletions(-)

diff --git a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec.c b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec.c
index 841ecdc2c4d9..cf721ef83d59 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/en_accel/ipsec.c
@@ -771,28 +771,44 @@ static int mlx5e_xfrm_add_state(struct net_device *dev,
 				struct xfrm_state *x,
 				struct netlink_ext_ack *extack)
 {
+	bool is_acq = x->xso.flags & XFRM_DEV_OFFLOAD_FLAG_ACQ;
 	struct mlx5e_ipsec_sa_entry *sa_entry = NULL;
 	bool allow_tunnel_mode = false;
+	struct mlx5_core_dev *mdev;
 	struct mlx5e_ipsec *ipsec;
 	struct mlx5e_priv *priv;
 	gfp_t gfp;
 	int err;
 
 	priv = netdev_priv(dev);
-	if (!priv->ipsec)
+	mdev = priv->mdev;
+	if (!mdev || !priv->ipsec)
 		return -EOPNOTSUPP;
 
+	if (!is_acq) {
+		err = mlx5_eswitch_block_mode(mdev, false);
+		if (err)
+			return err;
+	}
+
 	ipsec = priv->ipsec;
-	gfp = (x->xso.flags & XFRM_DEV_OFFLOAD_FLAG_ACQ) ? GFP_ATOMIC : GFP_KERNEL;
+	if (!ipsec) {
+		err = -EOPNOTSUPP;
+		goto unblock_mode;
+	}
+
+	gfp = is_acq ? GFP_ATOMIC : GFP_KERNEL;
 	sa_entry = kzalloc_obj(*sa_entry, gfp);
-	if (!sa_entry)
-		return -ENOMEM;
+	if (!sa_entry) {
+		err = -ENOMEM;
+		goto unblock_mode;
+	}
 
 	sa_entry->x = x;
 	sa_entry->dev = dev;
 	sa_entry->ipsec = ipsec;
 	/* Check if this SA is originated from acquire flow temporary SA */
-	if (x->xso.flags & XFRM_DEV_OFFLOAD_FLAG_ACQ) {
+	if (is_acq) {
 		x->xso.offload_handle = (unsigned long)sa_entry;
 		return 0;
 	}
@@ -806,10 +822,6 @@ static int mlx5e_xfrm_add_state(struct net_device *dev,
 		goto err_xfrm;
 	}
 
-	err = mlx5_eswitch_block_mode(priv->mdev, true);
-	if (err)
-		goto unblock_ipsec;
-
 	if (x->props.mode == XFRM_MODE_TUNNEL &&
 	    x->xso.type == XFRM_DEV_OFFLOAD_PACKET) {
 		allow_tunnel_mode = mlx5e_ipsec_fs_tunnel_allowed(sa_entry);
@@ -817,7 +829,7 @@ static int mlx5e_xfrm_add_state(struct net_device *dev,
 			NL_SET_ERR_MSG_MOD(extack,
 					   "Packet offload tunnel mode is disabled due to encap settings");
 			err = -EINVAL;
-			goto unblock_mode;
+			goto unblock_ipsec;
 		}
 	}
 
@@ -876,7 +888,7 @@ static int mlx5e_xfrm_add_state(struct net_device *dev,
 	if (allow_tunnel_mode)
 		mlx5_eswitch_unblock_encap(priv->mdev);
 
-	mlx5_eswitch_unblock_mode(priv->mdev);
+	mlx5_eswitch_unblock_mode(mdev);
 
 	return 0;
 
@@ -893,13 +905,14 @@ static int mlx5e_xfrm_add_state(struct net_device *dev,
 unblock_encap:
 	if (allow_tunnel_mode)
 		mlx5_eswitch_unblock_encap(priv->mdev);
-unblock_mode:
-	mlx5_eswitch_unblock_mode(priv->mdev);
 unblock_ipsec:
 	mlx5_eswitch_unblock_ipsec(priv->mdev);
 err_xfrm:
 	kfree(sa_entry);
 	NL_SET_ERR_MSG_WEAK_MOD(extack, "Device failed to offload this state");
+unblock_mode:
+	if (!is_acq)
+		mlx5_eswitch_unblock_mode(mdev);
 	return err;
 }
 
@@ -1262,12 +1275,17 @@ static int mlx5e_xfrm_add_policy(struct xfrm_policy *x,
 {
 	struct net_device *netdev = x->xdo.dev;
 	struct mlx5e_ipsec_pol_entry *pol_entry;
+	struct mlx5_core_dev *mdev;
 	struct mlx5e_priv *priv;
 	int err;
 
 	priv = netdev_priv(netdev);
+	mdev = priv->mdev;
+	if (!mdev)
+		return -EOPNOTSUPP;
+
 	/* Block esw mode changes until the policy holds its own block. */
-	err = mlx5_eswitch_block_mode(priv->mdev, false);
+	err = mlx5_eswitch_block_mode(mdev, false);
 	if (err) {
 		NL_SET_ERR_MSG_MOD(extack, "Eswitch busy, can't add policy");
 		return err;
@@ -1303,7 +1321,7 @@ static int mlx5e_xfrm_add_policy(struct xfrm_policy *x,
 		goto err_fs;
 
 	x->xdo.offload_handle = (unsigned long)pol_entry;
-	mlx5_eswitch_unblock_mode(priv->mdev);
+	mlx5_eswitch_unblock_mode(mdev);
 	return 0;
 
 err_fs:
@@ -1312,7 +1330,7 @@ static int mlx5e_xfrm_add_policy(struct xfrm_policy *x,
 	kfree(pol_entry);
 	NL_SET_ERR_MSG_MOD(extack, "Device failed to offload this policy");
 unblock_mode:
-	mlx5_eswitch_unblock_mode(priv->mdev);
+	mlx5_eswitch_unblock_mode(mdev);
 	return err;
 }
 
-- 
2.44.0


  parent reply	other threads:[~2026-09-30 12:12 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-30 12:11 [PATCH net V2 0/4] net/mlx5e: Fix offload lifetime and exclusion bugs Tariq Toukan
2026-09-30 12:11 ` [PATCH net V2 1/4] net/mlx5e: ipsec: Block eswitch mode changes during policy creation Tariq Toukan
2026-09-30 12:11 ` Tariq Toukan [this message]
2026-09-30 12:11 ` [PATCH net V2 3/4] net/mlx5e: Serialize TC and IPsec offload exclusion counters Tariq Toukan
2026-09-30 12:11 ` [PATCH net V2 4/4] net/mlx5e: tc: Tie esw & accel blocking refs to the flow's lifetime Tariq Toukan
2026-09-30 12:18 ` [PATCH net V2 0/4] net/mlx5e: Fix offload lifetime and exclusion bugs netdev-bot+sinfo

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260930121119.141953-3-tariqt@nvidia.com \
    --to=tariqt@nvidia.com \
    --cc=andrew+netdev@lunn.ch \
    --cc=borisp@nvidia.com \
    --cc=cjubran@nvidia.com \
    --cc=cratiu@nvidia.com \
    --cc=davem@davemloft.net \
    --cc=dtatulea@nvidia.com \
    --cc=edumazet@kernel.org \
    --cc=fmancera@suse.de \
    --cc=gal@nvidia.com \
    --cc=jianbol@nvidia.com \
    --cc=kees@kernel.org \
    --cc=kuba@kernel.org \
    --cc=leon@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-rdma@vger.kernel.org \
    --cc=mbloch@nvidia.com \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    --cc=parav@nvidia.com \
    --cc=parri.andrea@gmail.com \
    --cc=phaddad@nvidia.com \
    --cc=raeds@nvidia.com \
    --cc=roid@nvidia.com \
    --cc=saeedm@nvidia.com \
    --cc=steffen.klassert@secunet.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®