scsi: core: pair EH runtime PM get and put

shost->eh_noresume is currently consulted twice in one error handling
iteration: once before scsi_autopm_get_host() and once again before
scsi_autopm_put_host().

That is racy when a PM-triggered error path flips shost->eh_noresume
while the SCSI EH thread is still running.

The problem flow looks like this:
PM path
  ufshcd_set_dev_pwr_mode()
    shost->eh_noresume = 1
    ufshcd_execute_start_stop  <-- trigger EH
    ...
    shost->eh_noresume = 0

EH path
  scsi_error_handler()
    if (!shost->eh_noresume)
      scsi_autopm_get_host()  <-- skipped
    ...
    if (!shost->eh_noresume)
       scsi_autopm_put_host()  <-- executed later

In that case one EH iteration can skip autoresume on entry and still
drop a runtime PM reference on exit. That leaves an unmatched runtime PM
put and can trigger a runtime PM usage count underflow.

Fix this by making eh_noresume a regular bool so it can be accessed with
READ_ONCE() and WRITE_ONCE(). Snapshot it once per EH iteration and use
that snapshot for both runtime PM get and put decisions.

Fixes: ae0751ffc7 ("[SCSI] add flag to skip the runtime PM calls on the host")
Signed-off-by: Hongjie Fang <hongjiefang@asrmicro.com>
Reviewed-by: Bart Van Assche <bvanassche@acm.org>
Reviewed-by: Peter Wang <peter.wang@mediatek.com>
Link: https://patch.msgid.link/20260729111614.2407559-1-hongjiefang@asrmicro.com
Signed-off-by: Martin K. Petersen (Oracle) <mkp@kernel.org>
This commit is contained in:
Hongjie Fang 2026-07-29 19:16:14 +08:00 committed by Martin K. Petersen (Oracle)
parent 0279fd451a
commit 872f486259
3 changed files with 9 additions and 7 deletions

View File

@ -2362,6 +2362,7 @@ static void scsi_unjam_host(struct Scsi_Host *shost)
int scsi_error_handler(void *data)
{
struct Scsi_Host *shost = data;
bool eh_noresume;
/*
* We use TASK_INTERRUPTIBLE so that the thread is not
@ -2403,7 +2404,8 @@ int scsi_error_handler(void *data)
* what we need to do to get it up and online again (if we can).
* If we fail, we end up taking the thing offline.
*/
if (!shost->eh_noresume && scsi_autopm_get_host(shost) != 0) {
eh_noresume = READ_ONCE(shost->eh_noresume);
if (!eh_noresume && scsi_autopm_get_host(shost) != 0) {
SCSI_LOG_ERROR_RECOVERY(1,
shost_printk(KERN_ERR, shost,
"scsi_eh_%d: unable to autoresume\n",
@ -2427,7 +2429,7 @@ int scsi_error_handler(void *data)
* which are still online.
*/
scsi_restart_operations(shost);
if (!shost->eh_noresume)
if (!eh_noresume)
scsi_autopm_put_host(shost);
}
__set_current_state(TASK_RUNNING);

View File

@ -10016,7 +10016,7 @@ static int ufshcd_set_dev_pwr_mode(struct ufs_hba *hba,
* we are functional while we are here, skip host resume in error
* handling context.
*/
hba->host->eh_noresume = 1;
WRITE_ONCE(hba->host->eh_noresume, 1);
/*
* Current function would be generally called from the power management
@ -10038,7 +10038,7 @@ static int ufshcd_set_dev_pwr_mode(struct ufs_hba *hba,
}
scsi_device_put(sdp);
hba->host->eh_noresume = 0;
WRITE_ONCE(hba->host->eh_noresume, 0);
return ret;
}

View File

@ -664,6 +664,9 @@ struct Scsi_Host {
/* Asynchronous scan in progress */
bool async_scan __guarded_by(&scan_mutex);
/* Don't resume host in EH */
bool eh_noresume;
unsigned active_mode:2;
/*
@ -682,9 +685,6 @@ struct Scsi_Host {
/* Task mgmt function in progress */
unsigned tmf_in_progress:1;
/* Don't resume host in EH */
unsigned eh_noresume:1;
/* The controller does not support WRITE SAME */
unsigned no_write_same:1;