From 90c8ff461abd5af537d986bd6c5845d21a907865 Mon Sep 17 00:00:00 2001 From: Nitin Rawat Date: Wed, 24 Nov 2021 01:29:48 +0530 Subject: [PATCH] scsi: ufs: fix deadlock between resume and eh_work A deadlock condition can occur as per below sequence of events: 1. SSU command timed out in context of ufshcd_resume. 2. ufshcd_abort invoked after ssu command timeout on device WLUN and scheduled eh_work and waiting for flush on eh_work. 3. pm_runtime_get_sync invoked from ufshcd_err_handling_prepare in eh_work remained pending as runtime_status is RPM_RESUMING due to ufshcd_resume(step1). Fix this by : 1. skipping the pm_runtime_get_sync call for WLUN(ssu command) in ufshcd_err_handling_prepare invoked as part of eh_work and continue with the err_handler. 2. Later check the device state and link state after ssu returned failure in ufshcd_resume as per below conditions. a. If current dev and link state is active, dont return error as dev and link state is in correct state and hence proceed with the resume. b. If current dev and link state is not active but err_handler is in progress, wait on err handler to get finished and then proceed with the resume. c. If current dev and link state is not active and err_handler is not in progress as well, then let resume process abort. Thread1(ufshcd_resume): wait_for_completion_io() blk_execute_rq() __scsi_execute() ufshcd_set_dev_pwr_mode() ufshcd_resume() ufshcd_runtime_resume Thread2(ufshcd_abort): flush_work() >> eh_work ufshcd_eh_host_reset_handler() ufshcd_abort() scsi_try_to_abort_cmd(inline) scmd_eh_abort_handler() Thread3(err_handler): rpm_resume() __pm_runtime_resume() ufshcd_err_handling_prepare() ufshcd_err_handler(). Change-Id: I04a3cddecad4beda957d4d4f2fa3d7096f111c6d Signed-off-by: Nitin Rawat --- drivers/scsi/ufs/ufshcd.c | 34 ++++++++++++++++++++++++++++++++++ drivers/scsi/ufs/ufshcd.h | 5 +++++ 2 files changed, 39 insertions(+) diff --git a/drivers/scsi/ufs/ufshcd.c b/drivers/scsi/ufs/ufshcd.c index 4cf0348b3b82..dcb8dd2b6362 100644 --- a/drivers/scsi/ufs/ufshcd.c +++ b/drivers/scsi/ufs/ufshcd.c @@ -5853,7 +5853,12 @@ static inline void ufshcd_schedule_eh_work(struct ufs_hba *hba) static void ufshcd_err_handling_prepare(struct ufs_hba *hba) { +#if defined(CONFIG_SCSI_UFSHCD_QTI) + if (!hba->abort_triggered_wlun) + pm_runtime_get_sync(hba->dev); +#else pm_runtime_get_sync(hba->dev); +#endif if (pm_runtime_suspended(hba->dev)) { /* * Don't assume anything of pm_runtime_get_sync(), if @@ -5885,7 +5890,13 @@ static void ufshcd_err_handling_unprepare(struct ufs_hba *hba) ufshcd_release(hba); if (hba->clk_scaling.is_allowed) ufshcd_resume_clkscaling(hba); +#if defined(CONFIG_SCSI_UFSHCD_QTI) + if (!hba->abort_triggered_wlun) + pm_runtime_put(hba->dev); + hba->abort_triggered_wlun = false; +#else pm_runtime_put(hba->dev); +#endif } static inline bool ufshcd_err_handling_should_stop(struct ufs_hba *hba) @@ -6777,8 +6788,15 @@ static int ufshcd_abort(struct scsi_cmnd *cmd) * To avoid these unnecessary/illegal step we skip to the last error * handling stage: reset and restore. */ +#if defined(CONFIG_SCSI_UFSHCD_QTI) + if (lrbp->lun == UFS_UPIU_UFS_DEVICE_WLUN) { + hba->abort_triggered_wlun = true; + return ufshcd_eh_host_reset_handler(cmd); + } +#else if (lrbp->lun == UFS_UPIU_UFS_DEVICE_WLUN) return ufshcd_eh_host_reset_handler(cmd); +#endif ufshcd_hold(hba, false); reg = ufshcd_readl(hba, REG_UTP_TRANSFER_REQ_DOOR_BELL); @@ -8990,6 +9008,22 @@ static int ufshcd_resume(struct ufs_hba *hba, enum ufs_pm_op pm_op) if (!ufshcd_is_ufs_dev_active(hba)) { ret = ufshcd_set_dev_pwr_mode(hba, UFS_ACTIVE_PWR_MODE); +#if defined(CONFIG_SCSI_UFSHCD_QTI) + if (ret) { + if (ufshcd_is_ufs_dev_active(hba) && + ufshcd_is_link_active(hba)) { + ret = 0; + dev_err(hba->dev, "UFS device and link are Active\n"); + } else if ((work_pending(&hba->eh_work)) || + ufshcd_eh_in_progress(hba)) { + flush_work(&hba->eh_work); + ret = 0; + dev_err(hba->dev, "dev pwr mode=%d, UIC link state=%d\n", + hba->curr_dev_pwr_mode, + hba->uic_link_state); + } + } +#endif if (ret) goto set_old_link_state; } diff --git a/drivers/scsi/ufs/ufshcd.h b/drivers/scsi/ufs/ufshcd.h index db44c7ca098e..d9003a0b12a4 100644 --- a/drivers/scsi/ufs/ufshcd.h +++ b/drivers/scsi/ufs/ufshcd.h @@ -1032,6 +1032,11 @@ struct ufs_hba { ANDROID_KABI_RESERVE(2); ANDROID_KABI_RESERVE(3); ANDROID_KABI_RESERVE(4); +#ifdef CONFIG_SCSI_UFSHCD_QTI + /* distinguish between resume and restore */ + bool restore; + bool abort_triggered_wlun; +#endif }; /* Returns true if clocks can be gated. Otherwise false */