From: Tariq Toukan <tariqt@nvidia.com>
To: Andrew Lunn <andrew+netdev@lunn.ch>,
"David S. Miller" <davem@davemloft.net>,
Eric Dumazet <edumazet@kernel.org>,
Jakub Kicinski <kuba@kernel.org>, <netdev@vger.kernel.org>,
Paolo Abeni <pabeni@redhat.com>
Cc: Edward Srouji <edwards@nvidia.com>, Gal Pressman <gal@nvidia.com>,
"Leon Romanovsky" <leon@kernel.org>,
open list <linux-kernel@vger.kernel.org>,
<linux-rdma@vger.kernel.org>, Maher Sanalla <msanalla@nvidia.com>,
Mark Bloch <mbloch@nvidia.com>, Or Har-Toov <ohartoov@nvidia.com>,
Saeed Mahameed <saeedm@nvidia.com>, Shay Drori <shayd@nvidia.com>,
Tariq Toukan <tariqt@nvidia.com>
Subject: [PATCH net V2] net/mlx5: Lag, split aggregate speed into oper and max helpers
Date: Wed, 30 Sep 2026 20:04:06 +0300 [thread overview]
Message-ID: <20260930170406.148548-1-tariqt@nvidia.com> (raw)
From: Or Har-Toov <ohartoov@nvidia.com>
mlx5_lag_sum_devices_speed computes the LAG aggregate by summing oper
speeds across all ports. This has two bugs.
First, it relies on the assumption that a port whose carrier is down
will report an oper speed of zero and therefore not contribute to the
sum. This assumption does not always hold: when the link partner
disconnects the port transitions to DOWN state but firmware may still
report a non-zero oper speed, causing the aggregate to include a port
that is not actively carrying traffic.
Second, in active-backup mode only one port transmits at a time, so
the aggregate should reflect a single port speed rather than the sum
of all ports.
Fix this by splitting mlx5_lag_sum_devices_speed into two helpers.
mlx5_lag_get_devices_oper_speed reflects the speed currently available:
it queries the vport state of each port and skips any port that is not
UP, rather than relying on oper speed being zero.
mlx5_lag_get_devices_max_speed is state-independent and returns the
maximum achievable speed used as a fallback when speed is 0; for
active-backup it takes the maximum single-port speed instead of the sum.
Fixes: 28ea6036dad2 ("net/mlx5: Handle port and vport speed change events in MPESW")
Signed-off-by: Or Har-Toov <ohartoov@nvidia.com>
Reviewed-by: Shay Drori <shayd@nvidia.com>
Reviewed-by: Mark Bloch <mbloch@nvidia.com>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
.../net/ethernet/mellanox/mlx5/core/lag/lag.c | 79 +++++++++++++------
1 file changed, 57 insertions(+), 22 deletions(-)
V2:
- MPESW state check now goes through mlx5_query_vport_max_tx_speed(),
propagating a failed FW query as an error instead of silently
treating it as VPORT_STATE_DOWN.
- Take maximum single-port speed for max in active-backup instead of a
sum.
V1:
https://lore.kernel.org/all/20260910102432.3845360-2-tariqt@nvidia.com/
diff --git a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
index 3b34bec559e0..4e173b08cb37 100644
--- a/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
+++ b/drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c
@@ -1432,30 +1432,40 @@ static bool mlx5_lag_should_disable_lag(struct mlx5_lag *ldev, bool do_bond)
}
#ifdef CONFIG_MLX5_ESWITCH
-static int
-mlx5_lag_sum_devices_speed(struct mlx5_lag *ldev, u32 *sum_speed,
- int (*get_speed)(struct mlx5_core_dev *, u32 *))
+static int mlx5_lag_get_devices_oper_speed(struct mlx5_lag *ldev,
+ u32 *sum_speed)
{
- struct mlx5_core_dev *pf_mdev;
- struct lag_func *pf;
int pf_idx;
- u32 speed;
- int ret;
*sum_speed = 0;
mlx5_ldev_for_each(pf_idx, 0, ldev) {
+ u8 opmod = MLX5_VPORT_STATE_OP_MOD_VNIC_VPORT;
+ struct mlx5_core_dev *pf_mdev;
+ struct lag_func *pf;
+ u32 speed;
+ u8 state;
+ int ret;
+
pf = mlx5_lag_pf(ldev, pf_idx);
if (!pf)
continue;
pf_mdev = pf->dev;
if (!pf_mdev)
continue;
+ ret = mlx5_query_vport_max_tx_speed(pf_mdev, opmod, 0, 0,
+ &speed, &state);
+ if (ret) {
+ mlx5_core_dbg(pf_mdev, "State query failed (err=%d)\n",
+ ret);
+ return ret;
+ }
+ if (state != VPORT_STATE_UP)
+ continue;
- ret = get_speed(pf_mdev, &speed);
+ ret = mlx5_port_oper_linkspeed(pf_mdev, &speed);
if (ret) {
mlx5_core_dbg(pf_mdev,
- "Failed to get device speed using %ps. Device %s speed is not available (err=%d)\n",
- get_speed, dev_name(pf_mdev->device),
+ "Failed to get oper speed (err=%d)\n",
ret);
return ret;
}
@@ -1466,17 +1476,42 @@ mlx5_lag_sum_devices_speed(struct mlx5_lag *ldev, u32 *sum_speed,
return 0;
}
-static int mlx5_lag_sum_devices_max_speed(struct mlx5_lag *ldev, u32 *max_speed)
+static int mlx5_lag_get_devices_max_speed(struct mlx5_lag *ldev, u32 *max_speed)
{
- return mlx5_lag_sum_devices_speed(ldev, max_speed,
- mlx5_port_max_linkspeed);
-}
+ bool take_max;
+ int pf_idx;
-static int mlx5_lag_sum_devices_oper_speed(struct mlx5_lag *ldev,
- u32 *oper_speed)
-{
- return mlx5_lag_sum_devices_speed(ldev, oper_speed,
- mlx5_port_oper_linkspeed);
+ take_max = ldev->tracker.tx_type == NETDEV_LAG_TX_TYPE_ACTIVEBACKUP;
+ if (ldev->mode == MLX5_LAG_MODE_MPESW)
+ take_max = false;
+
+ *max_speed = 0;
+ mlx5_ldev_for_each(pf_idx, 0, ldev) {
+ struct mlx5_core_dev *pf_mdev;
+ struct lag_func *pf;
+ u32 speed;
+ int ret;
+
+ pf = mlx5_lag_pf(ldev, pf_idx);
+ if (!pf)
+ continue;
+ pf_mdev = pf->dev;
+ if (!pf_mdev)
+ continue;
+
+ ret = mlx5_port_max_linkspeed(pf_mdev, &speed);
+ if (ret) {
+ mlx5_core_dbg(pf_mdev,
+ "Failed to get max speed (err=%d)\n",
+ ret);
+ return ret;
+ }
+
+ *max_speed = take_max ?
+ max(*max_speed, speed) : *max_speed + speed;
+ }
+
+ return 0;
}
static void mlx5_lag_modify_device_vports_speed(struct mlx5_core_dev *mdev,
@@ -1525,7 +1560,7 @@ void mlx5_lag_set_vports_agg_speed(struct mlx5_lag *ldev)
int pf_idx;
if (ldev->mode == MLX5_LAG_MODE_MPESW) {
- if (mlx5_lag_sum_devices_oper_speed(ldev, &speed))
+ if (mlx5_lag_get_devices_oper_speed(ldev, &speed))
return;
} else {
speed = ldev->tracker.bond_speed_mbps;
@@ -1533,8 +1568,8 @@ void mlx5_lag_set_vports_agg_speed(struct mlx5_lag *ldev)
return;
}
- /* If speed is not set, use the sum of max speeds of all PFs */
- if (!speed && mlx5_lag_sum_devices_max_speed(ldev, &speed))
+ /* If speed is not set, fall back to the max achievable speed */
+ if (!speed && mlx5_lag_get_devices_max_speed(ldev, &speed))
return;
speed = speed / MLX5_MAX_TX_SPEED_UNIT;
base-commit: 99b43ede9e355ba35244cc9470bf1819774ce39d
--
2.44.0
next reply other threads:[~2026-09-30 17:05 UTC|newest]
Thread overview: 5+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-30 17:04 Tariq Toukan [this message]
2026-09-30 17:08 ` [PATCH net V2] net/mlx5: Lag, split aggregate speed into oper and max helpers netdev-bot+sinfo
2026-10-05 15:44 ` Or Har-Toov
2026-09-30 17:12 ` sashiko-bot
2026-10-05 23:00 ` patchwork-bot+netdevbpf
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260930170406.148548-1-tariqt@nvidia.com \
--to=tariqt@nvidia.com \
--cc=andrew+netdev@lunn.ch \
--cc=davem@davemloft.net \
--cc=edumazet@kernel.org \
--cc=edwards@nvidia.com \
--cc=gal@nvidia.com \
--cc=kuba@kernel.org \
--cc=leon@kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=linux-rdma@vger.kernel.org \
--cc=mbloch@nvidia.com \
--cc=msanalla@nvidia.com \
--cc=netdev@vger.kernel.org \
--cc=ohartoov@nvidia.com \
--cc=pabeni@redhat.com \
--cc=saeedm@nvidia.com \
--cc=shayd@nvidia.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.