]> git.ipfire.org Git - thirdparty/linux.git/commitdiff
scsi: core: pair EH runtime PM get and put
authorHongjie Fang <hongjiefang@asrmicro.com>
Wed, 29 Jul 2026 11:16:14 +0000 (19:16 +0800)
committerMartin K. Petersen (Oracle) <mkp@kernel.org>
Fri, 7 Aug 2026 15:00:49 +0000 (11:00 -0400)
shost->eh_noresume is currently consulted twice in one error handling
iteration: once before scsi_autopm_get_host() and once again before
scsi_autopm_put_host().

That is racy when a PM-triggered error path flips shost->eh_noresume
while the SCSI EH thread is still running.

The problem flow looks like this:
PM path
  ufshcd_set_dev_pwr_mode()
    shost->eh_noresume = 1
    ufshcd_execute_start_stop  <-- trigger EH
    ...
    shost->eh_noresume = 0

EH path
  scsi_error_handler()
    if (!shost->eh_noresume)
      scsi_autopm_get_host()  <-- skipped
    ...
    if (!shost->eh_noresume)
       scsi_autopm_put_host()  <-- executed later

In that case one EH iteration can skip autoresume on entry and still
drop a runtime PM reference on exit. That leaves an unmatched runtime PM
put and can trigger a runtime PM usage count underflow.

Fix this by making eh_noresume a regular bool so it can be accessed with
READ_ONCE() and WRITE_ONCE(). Snapshot it once per EH iteration and use
that snapshot for both runtime PM get and put decisions.

Fixes: ae0751ffc77e ("[SCSI] add flag to skip the runtime PM calls on the host")
Signed-off-by: Hongjie Fang <hongjiefang@asrmicro.com>
Reviewed-by: Bart Van Assche <bvanassche@acm.org>
Reviewed-by: Peter Wang <peter.wang@mediatek.com>
Link: https://patch.msgid.link/20260729111614.2407559-1-hongjiefang@asrmicro.com
Signed-off-by: Martin K. Petersen (Oracle) <mkp@kernel.org>
drivers/scsi/scsi_error.c
drivers/ufs/core/ufshcd.c
include/scsi/scsi_host.h

index 453a2232452dba2f0486b8af4f4a5a24222d90b6..74b70801269bd9bd68cf0d46d9ecc5ed2d9ba8bf 100644 (file)
@@ -2362,6 +2362,7 @@ static void scsi_unjam_host(struct Scsi_Host *shost)
 int scsi_error_handler(void *data)
 {
        struct Scsi_Host *shost = data;
+       bool eh_noresume;
 
        /*
         * We use TASK_INTERRUPTIBLE so that the thread is not
@@ -2403,7 +2404,8 @@ int scsi_error_handler(void *data)
                 * what we need to do to get it up and online again (if we can).
                 * If we fail, we end up taking the thing offline.
                 */
-               if (!shost->eh_noresume && scsi_autopm_get_host(shost) != 0) {
+               eh_noresume = READ_ONCE(shost->eh_noresume);
+               if (!eh_noresume && scsi_autopm_get_host(shost) != 0) {
                        SCSI_LOG_ERROR_RECOVERY(1,
                                shost_printk(KERN_ERR, shost,
                                             "scsi_eh_%d: unable to autoresume\n",
@@ -2427,7 +2429,7 @@ int scsi_error_handler(void *data)
                 * which are still online.
                 */
                scsi_restart_operations(shost);
-               if (!shost->eh_noresume)
+               if (!eh_noresume)
                        scsi_autopm_put_host(shost);
        }
        __set_current_state(TASK_RUNNING);
index 13e6357a8bbb5dcfe825861ea42ea6427bb76463..a41b56dbf3b7fa405532627b33c39be440c78528 100644 (file)
@@ -10016,7 +10016,7 @@ static int ufshcd_set_dev_pwr_mode(struct ufs_hba *hba,
         * we are functional while we are here, skip host resume in error
         * handling context.
         */
-       hba->host->eh_noresume = 1;
+       WRITE_ONCE(hba->host->eh_noresume, 1);
 
        /*
         * Current function would be generally called from the power management
@@ -10038,7 +10038,7 @@ static int ufshcd_set_dev_pwr_mode(struct ufs_hba *hba,
        }
 
        scsi_device_put(sdp);
-       hba->host->eh_noresume = 0;
+       WRITE_ONCE(hba->host->eh_noresume, 0);
        return ret;
 }
 
index f6b286fa59f214b98ea705b30d05ab08ce47e459..98b0ccf0813ea63b04f53ed387adfa996e0d4372 100644 (file)
@@ -664,6 +664,9 @@ struct Scsi_Host {
        /* Asynchronous scan in progress */
        bool async_scan __guarded_by(&scan_mutex);
 
+       /* Don't resume host in EH */
+       bool eh_noresume;
+
        unsigned active_mode:2;
 
        /*
@@ -682,9 +685,6 @@ struct Scsi_Host {
        /* Task mgmt function in progress */
        unsigned tmf_in_progress:1;
 
-       /* Don't resume host in EH */
-       unsigned eh_noresume:1;
-
        /* The controller does not support WRITE SAME */
        unsigned no_write_same:1;