From: Tariq Toukan <tariqt@nvidia.com>
To: Andrew Lunn <andrew+netdev@lunn.ch>,
"David S. Miller" <davem@davemloft.net>,
Eric Dumazet <edumazet@kernel.org>,
Jakub Kicinski <kuba@kernel.org>, <netdev@vger.kernel.org>,
Paolo Abeni <pabeni@redhat.com>
Cc: Edward Srouji <edwards@nvidia.com>, Gal Pressman <gal@nvidia.com>,
"Leon Romanovsky" <leon@kernel.org>,
open list <linux-kernel@vger.kernel.org>,
<linux-rdma@vger.kernel.org>, Maher Sanalla <msanalla@nvidia.com>,
Mark Bloch <mbloch@nvidia.com>, Or Har-Toov <ohartoov@nvidia.com>,
Saeed Mahameed <saeedm@nvidia.com>, Shay Drori <shayd@nvidia.com>,
Tariq Toukan <tariqt@nvidia.com>
Subject: [PATCH net V2] net/mlx5: Lag, split aggregate speed into oper and max helpers
Date: Wed, 30 Sep 2026 20:04:06 +0300 [thread overview]
Message-ID: <20260930170406.148548-1-tariqt@nvidia.com> (raw)
From: Or Har-Toov <ohartoov@nvidia.com>
mlx5_lag_sum_devices_speed computes the LAG aggregate by summing oper
speeds across all ports. This has two bugs.
First, it relies on the assumption that a port whose carrier is down
will report an oper speed of zero and therefore not contribute to the
sum. This assumption does not always hold: when the link partner
disconnects the port transitions to DOWN state but firmware may still
report a non-zero oper speed, causing the aggregate to include a port
that is not actively carrying traffic.
Second, in active-backup mode only one port transmits at a time, so
the aggregate should reflect a single port speed rather than the sum
of all ports.
Fix this by splitting mlx5_lag_sum_devices_speed into two helpers.
mlx5_lag_get_devices_oper_speed reflects the speed currently available:
it queries the vport state of each port and skips any port that is not
UP, rather than relying on oper speed being zero.
mlx5_lag_get_devices_max_speed is state-independent and returns the
maximum achievable speed used as a fallback when speed is 0; for
active-backup it takes the maximum single-port speed instead of the sum.
Fixes: 28ea6036dad2 ("net/mlx5: Handle port and vport speed change events in MPESW")
Signed-off-by: Or Har-Toov <ohartoov@nvidia.com>
Reviewed-by: Shay Drori <shayd@nvidia.com>
Reviewed-by: Mark Bloch <mbloch@nvidia.com>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
.../net/ethernet/mellanox/mlx5/core/lag/lag.c | 79 +++++++++++++------
1 file changed, 57 insertions(+), 22 deletions(-)
V2:
- MPESW state check now goes through mlx5_query_vport_max_tx_speed(),
propagating a failed FW query as an error instead of silently
treating it as VPORT_STATE_DOWN.
- Take maximum single-port speed for max in active-backup instead of a
sum.
V1:
https://lore.kernel.org/all/20260910102432.3845360-2-tariqt@nvidia.com/
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
index 3b34bec559e0..4e173b08cb37 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
@@ -1432,30 +1432,40 @@ static bool mlx5_lag_should_disable_lag(struct mlx5_lag *ldev, bool do_bond)
}
#ifdef CONFIG_MLX5_ESWITCH
-static int
-mlx5_lag_sum_devices_speed(struct mlx5_lag *ldev, u32 *sum_speed,
- int (*get_speed)(struct mlx5_core_dev *, u32 *))
+static int mlx5_lag_get_devices_oper_speed(struct mlx5_lag *ldev,
+ u32 *sum_speed)
{
- struct mlx5_core_dev *pf_mdev;
- struct lag_func *pf;
int pf_idx;
- u32 speed;
- int ret;
*sum_speed = 0;
mlx5_ldev_for_each(pf_idx, 0, ldev) {
+ u8 opmod = MLX5_VPORT_STATE_OP_MOD_VNIC_VPORT;
+ struct mlx5_core_dev *pf_mdev;
+ struct lag_func *pf;
+ u32 speed;
+ u8 state;
+ int ret;
+
pf = mlx5_lag_pf(ldev, pf_idx);
if (!pf)
continue;
pf_mdev = pf->dev;
if (!pf_mdev)
continue;
+ ret = mlx5_query_vport_max_tx_speed(pf_mdev, opmod, 0, 0,
+ &speed, &state);
+ if (ret) {
+ mlx5_core_dbg(pf_mdev, "State query failed (err=%d)\n",
+ ret);
+ return ret;
+ }
+ if (state != VPORT_STATE_UP)
+ continue;
- ret = get_speed(pf_mdev, &speed);
+ ret = mlx5_port_oper_linkspeed(pf_mdev, &speed);
if (ret) {
mlx5_core_dbg(pf_mdev,
- "Failed to get device speed using %ps. Device %s speed is not available (err=%d)\n",
- get_speed, dev_name(pf_mdev->device),
+ "Failed to get oper speed (err=%d)\n",
ret);
return ret;
}
@@ -1466,17 +1476,42 @@ mlx5_lag_sum_devices_speed(struct mlx5_lag *ldev, u32 *sum_speed,
return 0;
}
-static int mlx5_lag_sum_devices_max_speed(struct mlx5_lag *ldev, u32 *max_speed)
+static int mlx5_lag_get_devices_max_speed(struct mlx5_lag *ldev, u32 *max_speed)
{
- return mlx5_lag_sum_devices_speed(ldev, max_speed,
- mlx5_port_max_linkspeed);
-}
+ bool take_max;
+ int pf_idx;
-static int mlx5_lag_sum_devices_oper_speed(struct mlx5_lag *ldev,
- u32 *oper_speed)
-{
- return mlx5_lag_sum_devices_speed(ldev, oper_speed,
- mlx5_port_oper_linkspeed);
+ take_max = ldev->tracker.tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP;
+ if (ldev->mode == MLX5_LAG_MODE_MPESW)
+ take_max = false;
+
+ *max_speed = 0;
+ mlx5_ldev_for_each(pf_idx, 0, ldev) {
+ struct mlx5_core_dev *pf_mdev;
+ struct lag_func *pf;
+ u32 speed;
+ int ret;
+
+ pf = mlx5_lag_pf(ldev, pf_idx);
+ if (!pf)
+ continue;
+ pf_mdev = pf->dev;
+ if (!pf_mdev)
+ continue;
+
+ ret = mlx5_port_max_linkspeed(pf_mdev, &speed);
+ if (ret) {
+ mlx5_core_dbg(pf_mdev,
+ "Failed to get max speed (err=%d)\n",
+ ret);
+ return ret;
+ }
+
+ *max_speed = take_max ?
+ max(*max_speed, speed) : *max_speed + speed;
+ }
+
+ return 0;
}
static void mlx5_lag_modify_device_vports_speed(struct mlx5_core_dev *mdev,
@@ -1525,7 +1560,7 @@ void mlx5_lag_set_vports_agg_speed(struct mlx5_lag *ldev)
int pf_idx;
if (ldev->mode == MLX5_LAG_MODE_MPESW) {
- if (mlx5_lag_sum_devices_oper_speed(ldev, &speed))
+ if (mlx5_lag_get_devices_oper_speed(ldev, &speed))
return;
} else {
speed = ldev->tracker.bond_speed_mbps;
@@ -1533,8 +1568,8 @@ void mlx5_lag_set_vports_agg_speed(struct mlx5_lag *ldev)
return;
}
- /* If speed is not set, use the sum of max speeds of all PFs */
- if (!speed && mlx5_lag_sum_devices_max_speed(ldev, &speed))
+ /* If speed is not set, fall back to the max achievable speed */
+ if (!speed && mlx5_lag_get_devices_max_speed(ldev, &speed))
return;
speed = speed / MLX5_MAX_TX_SPEED_UNIT;
base-commit: 99b43ede9e355ba35244cc9470bf1819774ce39d
--
2.44.0
next reply other threads:[~2026-09-30 17:05 UTC|newest]
Thread overview: 2+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-30 17:04 Tariq Toukan [this message]
2026-09-30 17:08 ` netdev-bot+sinfo
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260930170406.148548-1-tariqt@nvidia.com \
--to=tariqt@nvidia.com \
--cc=andrew+netdev@lunn.ch \
--cc=davem@davemloft.net \
--cc=edumazet@kernel.org \
--cc=edwards@nvidia.com \
--cc=gal@nvidia.com \
--cc=kuba@kernel.org \
--cc=leon@kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-rdma@vger.kernel.org \
--cc=mbloch@nvidia.com \
--cc=msanalla@nvidia.com \
--cc=netdev@vger.kernel.org \
--cc=ohartoov@nvidia.com \
--cc=pabeni@redhat.com \
--cc=saeedm@nvidia.com \
--cc=shayd@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®