]> git.ipfire.org Git - thirdparty/kernel/linux.git/commitdiff
net/mlx5e: Fix use-after-free in mlx5e_tx_reporter_timeout_recover
authorMatt Fleming <mfleming@cloudflare.com>
Wed, 13 May 2026 11:22:26 +0000 (12:22 +0100)
committerJakub Kicinski <kuba@kernel.org>
Fri, 15 May 2026 00:17:11 +0000 (17:17 -0700)
mlx5e_tx_reporter_timeout_recover() accesses sq->netdev after
mlx5e_safe_reopen_channels() has torn down and freed the channel (and
its embedded SQs). Replace the three sq->netdev references with
priv->netdev which is safe because priv outlives channel teardown.

The netdev_err() call already used priv->netdev for this reason; make
the trylock/unlock and health_channel_eq_recover calls consistent.

This fixes the following KASAN splat:

  BUG: KASAN: use-after-free in mlx5e_tx_reporter_timeout_recover+0x1dd/0x360 [mlx5_core]
  Read of size 8 at addr ffff889860ed0b28 by task kworker/u113:2/5277

  Call Trace:
   mlx5e_tx_reporter_timeout_recover+0x1dd/0x360 [mlx5_core]
   devlink_health_reporter_recover+0xa2/0x150
   devlink_health_report+0x254/0x7c0
   mlx5e_reporter_tx_timeout+0x297/0x380 [mlx5_core]
   mlx5e_tx_timeout_work+0x109/0x170 [mlx5_core]
   process_one_work+0x677/0xf20
   worker_thread+0x51f/0xd90
   kthread+0x3a5/0x810
   ret_from_fork+0x208/0x400
   ret_from_fork_asm+0x1a/0x30

Fixes: 83ac0304a2d7 ("net/mlx5e: Fix deadlocks between devlink and netdev instance locks")
Cc: stable@vger.kernel.org
Reviewed-by: Cosmin Ratiu <cratiu@nvidia.com>
Reviewed-by: Tariq Toukan <tariqt@nvidia.com>
Signed-off-by: Matt Fleming <mfleming@cloudflare.com>
Link: https://patch.msgid.link/20260513112226.140512-1-matt@readmodwrite.com
Signed-off-by: Jakub Kicinski <kuba@kernel.org>
drivers/net/ethernet/mellanox/mlx5/core/en/reporter_tx.c

index afdeb1b3d4256ba25c5270f61b757d39fe30fd8e..8409ae73768fffcfe81f187830530babbe77e351 100644 (file)
@@ -160,13 +160,13 @@ static int mlx5e_tx_reporter_timeout_recover(void *ctx)
         * channels are being closed for other reason and this work is not
         * relevant anymore.
         */
-       while (!netdev_trylock(sq->netdev)) {
+       while (!netdev_trylock(priv->netdev)) {
                if (!test_bit(MLX5E_STATE_CHANNELS_ACTIVE, &priv->state))
                        return 0;
                msleep(20);
        }
 
-       err = mlx5e_health_channel_eq_recover(sq->netdev, eq, sq->cq.ch_stats);
+       err = mlx5e_health_channel_eq_recover(priv->netdev, eq, sq->cq.ch_stats);
        if (!err) {
                to_ctx->status = 0; /* this sq recovered */
                goto out;
@@ -186,7 +186,7 @@ static int mlx5e_tx_reporter_timeout_recover(void *ctx)
                   "mlx5e_safe_reopen_channels failed recovering from a tx_timeout, err(%d).\n",
                   err);
 out:
-       netdev_unlock(sq->netdev);
+       netdev_unlock(priv->netdev);
        return err;
 }