From b35f0ac1f73bd0155e368154b231fae67a01756d Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Thu, 8 Aug 2019 20:58:55 -0600 Subject: [PATCH 1/8] msm: kgsl: Handle the very first gmu boot failure The very first boot starts from the INIT state. If the gmu fails to boot here, set the state to KGSL_STATE_RESET to effect a RESET -> INIT transition where we reset the GMU. Change-Id: I0fddf745d299b855aab0c1989997dcab5e77954a Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/kgsl_pwrctrl.c | 41 ++++++++++++++++++++-------------- 1 file changed, 24 insertions(+), 17 deletions(-) diff --git a/drivers/gpu/msm/kgsl_pwrctrl.c b/drivers/gpu/msm/kgsl_pwrctrl.c index fe2788e41c43..d611a042816f 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.c +++ b/drivers/gpu/msm/kgsl_pwrctrl.c @@ -1695,13 +1695,23 @@ static int _init(struct kgsl_device *device) int status = 0; switch (device->state) { + case KGSL_STATE_RESET: + if (gmu_core_isenabled(device)) { + /* + * If we fail a INIT -> AWARE transition, we will + * transition back to INIT. However, we must hard reset + * the GMU as we go back to INIT. This is done by + * forcing a RESET -> INIT transition. + */ + gmu_core_suspend(device); + kgsl_pwrctrl_set_state(device, KGSL_STATE_INIT); + } + break; case KGSL_STATE_NAP: /* Force power on to do the stop */ status = kgsl_pwrctrl_enable(device); /* fall through */ case KGSL_STATE_ACTIVE: - /* fall through */ - case KGSL_STATE_RESET: kgsl_pwrctrl_irq(device, KGSL_PWRFLAGS_OFF); del_timer_sync(&device->idle_timer); kgsl_pwrscale_midframe_timer_cancel(device); @@ -1813,12 +1823,6 @@ _aware(struct kgsl_device *device) status = gmu_core_start(device); break; case KGSL_STATE_INIT: - /* if GMU already in FAULT */ - if (gmu_core_isenabled(device) && - test_bit(GMU_FAULT, &device->gmu_core.flags)) { - status = -EINVAL; - break; - } status = kgsl_pwrctrl_enable(device); break; /* The following 3 cases shouldn't occur, but don't panic. */ @@ -1832,20 +1836,23 @@ _aware(struct kgsl_device *device) break; case KGSL_STATE_SLUMBER: status = kgsl_pwrctrl_enable(device); - if (status && gmu_core_isenabled(device)) - /* - * SLUMBER -> AWARE failed which means GMU boot failed. - * Make sure we reset the GMU while transitioning back - * to SLUMBER. - */ - kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET); - break; default: status = -EINVAL; } - if (!status) + if (status && gmu_core_isenabled(device)) + /* + * If a SLUMBER/INIT -> AWARE fails, we transition back to + * SLUMBER/INIT state. We must hard reset the GMU while + * transitioning back to SLUMBER/INIT. A RESET -> AWARE + * transition is different. It happens when dispatcher is + * attempting reset/recovery as part of fault handling. If it + * fails, we should still transition back to RESET in case + * we want to attempt another reset/recovery. + */ + kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET); + else kgsl_pwrctrl_set_state(device, KGSL_STATE_AWARE); return status; From 7e53b80b0b975db21fae21925a71a168a279407a Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Thu, 8 Aug 2019 21:00:42 -0600 Subject: [PATCH 2/8] msm: kgsl: Set gmu fault inside gmu_snapshot The two always happen together. So set the gmu_fault inside the gmu_snapshot function. Also, if we have already recorded a gmu fault, then do not send nmi or try to snapshot a gmu which is already in nmi. Change-Id: I403a9c2c3cb7a1330a7931c41a23b4b4a2b66998 Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/kgsl_gmu.c | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index a5d5a86cd5df..c9386b59da9f 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -1530,6 +1530,10 @@ static void gmu_snapshot(struct kgsl_device *device) struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); struct gmu_device *gmu = KGSL_GMU_DEVICE(device); + /* Abstain from sending another nmi or over-writing snapshot */ + if (test_and_set_bit(GMU_FAULT, &device->gmu_core.flags)) + return; + adreno_gmu_send_nmi(adreno_dev); /* Wait for the NMI to be handled */ udelay(100); @@ -1695,12 +1699,6 @@ static void gmu_stop(struct kgsl_device *device) return; error: - /* - * The power controller will change state to SLUMBER anyway - * Set GMU_FAULT flag to indicate to power contrller - * that hang recovery is needed to power on GPU - */ - set_bit(GMU_FAULT, &device->gmu_core.flags); dev_err(&gmu->pdev->dev, "Failed to stop GMU\n"); gmu_core_snapshot(device); } From 0d44684cabc075a671060b70b28a4f29da4f973c Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Thu, 8 Aug 2019 21:01:28 -0600 Subject: [PATCH 3/8] msm: kgsl: Take GMU snapshot on GMU failures GMU can fail to turn on GX or it can fail the START HFI. So take snapshot and put GMU in NMI. When this happens, kgsl will call gmu_stop() and as GMU is already in fault aka NMI, reset the GMU and GPU. If oob during adreno_stop fails, take gmu snapshot and only do things that don't need gx to be on. Change-Id: Iafc9b34063a7ff2415d3462dd289b52e425fbf3b Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/adreno.c | 44 ++++++++++++++++---------------------- drivers/gpu/msm/kgsl_gmu.c | 11 ++++++++++ 2 files changed, 30 insertions(+), 25 deletions(-) diff --git a/drivers/gpu/msm/adreno.c b/drivers/gpu/msm/adreno.c index 203521d9ce7c..d02abf18ad78 100644 --- a/drivers/gpu/msm/adreno.c +++ b/drivers/gpu/msm/adreno.c @@ -2057,12 +2057,16 @@ static int _adreno_start(struct adreno_device *adreno_dev) /* Send OOB request to turn on the GX */ status = gmu_core_dev_oob_set(device, oob_gpu); - if (status) - goto error_boot_oob_clear; + if (status) { + gmu_core_snapshot(device); + goto error_oob_clear; + } status = gmu_core_dev_hfi_start_msg(device); - if (status) + if (status) { + gmu_core_snapshot(device); goto error_oob_clear; + } if (device->pwrctrl.bus_control) { /* VBIF waiting for RAM */ @@ -2274,29 +2278,15 @@ static int adreno_stop(struct kgsl_device *device) error = gmu_core_dev_oob_set(device, oob_gpu); if (error) { gmu_core_dev_oob_clear(device, oob_gpu); - - if (gmu_core_regulator_isenabled(device)) { - /* GPU is on. Try recovery */ - set_bit(GMU_FAULT, &device->gmu_core.flags); gmu_core_snapshot(device); error = -EINVAL; - } + goto no_gx_power; } - adreno_dispatcher_stop(adreno_dev); - - adreno_ringbuffer_stop(adreno_dev); - kgsl_pwrscale_update_stats(device); adreno_irqctrl(adreno_dev, 0); - if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice)) - llcc_slice_deactivate(adreno_dev->gpu_llc_slice); - - if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice)) - llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice); - /* Save active coresight registers if applicable */ adreno_coresight_stop(adreno_dev); @@ -2314,19 +2304,23 @@ static int adreno_stop(struct kgsl_device *device) */ if (!error && gmu_core_dev_wait_for_lowest_idle(device)) { - set_bit(GMU_FAULT, &device->gmu_core.flags); gmu_core_snapshot(device); - /* - * Assume GMU hang after 10ms without responding. - * It shall be relative safe to clear vbif and stop - * MMU later. Early return in adreno_stop function - * will result in kernel panic in adreno_start - */ error = -EINVAL; } adreno_clear_pending_transactions(device); +no_gx_power: + adreno_dispatcher_stop(adreno_dev); + + adreno_ringbuffer_stop(adreno_dev); + + if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice)) + llcc_slice_deactivate(adreno_dev->gpu_llc_slice); + + if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice)) + llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice); + /* * The halt is not cleared in the above function if we have GBIF. * Clear it here if GMU is enabled as GMU stop needs access to diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index c9386b59da9f..0e8e06b369d3 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -1674,6 +1674,12 @@ static void gmu_stop(struct kgsl_device *device) if (!test_bit(GMU_CLK_ON, &device->gmu_core.flags)) return; + /* Force suspend if gmu is already in fault */ + if (test_bit(GMU_FAULT, &device->gmu_core.flags)) { + gmu_core_suspend(device); + return; + } + /* Wait for the lowest idle level we requested */ if (gmu_core_dev_wait_for_lowest_idle(device)) goto error; @@ -1701,6 +1707,11 @@ static void gmu_stop(struct kgsl_device *device) error: dev_err(&gmu->pdev->dev, "Failed to stop GMU\n"); gmu_core_snapshot(device); + /* + * We failed to stop the gmu successfully. Force a suspend + * to set things up for a fresh start. + */ + gmu_core_suspend(device); } static void gmu_remove(struct kgsl_device *device) From ab947c5e89861db932b34181568f199929e60129 Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Thu, 8 Aug 2019 21:03:19 -0600 Subject: [PATCH 4/8] msm: kgsl: Correctly handle CP_INIT failure A stuck CP means that we will not be able to enter slumber because gmu to cp interaction is impacted. Therefore, take a gmu snapshot which also sets gmu fault. The gmu fault will indicate the clean up code to force a gmu_suspend() so as to set the stage for the next submission to start afresh. Change-Id: Ia90e6c447e9c1c87e04cf9ca3ed87eed5c17b07c Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/adreno.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/msm/adreno.c b/drivers/gpu/msm/adreno.c index d02abf18ad78..4724b76afa0a 100644 --- a/drivers/gpu/msm/adreno.c +++ b/drivers/gpu/msm/adreno.c @@ -2990,7 +2990,15 @@ void adreno_spin_idle_debug(struct adreno_device *adreno_dev, dev_err(device->dev, " hwfault=%8.8X\n", hwfault); - kgsl_device_snapshot(device, NULL, adreno_gmu_gpu_fault(adreno_dev)); + /* + * If CP is stuck, gmu may not perform as expected. So force a gmu + * snapshot which captures entire state as well as sets the gmu fault + * because things need to be reset anyway. + */ + if (gmu_core_isenabled(device)) + gmu_core_snapshot(device); + else + kgsl_device_snapshot(device, NULL, false); } /** From d1e042852a1db464a2be6e9c9a08ef12c484a9e7 Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Tue, 4 Dec 2018 13:55:44 -0700 Subject: [PATCH 5/8] msm: kgsl: Correctly handle gmu fault interrupts Do not trigger dispatcher based snapshot or recovery. Instead, wait for the next kgsl -> GMU interaction to timeout thus triggering GMU snapshot and appropriate recovery steps based on whether gpu was active or not. Change-Id: I17b4245f4e0113bfc902d7dae46bb24d0bc2b65d Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/kgsl_gmu.c | 2 -- drivers/gpu/msm/kgsl_hfi.c | 8 ++------ 2 files changed, 2 insertions(+), 8 deletions(-) diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index 0e8e06b369d3..98b0b0688e72 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -1023,8 +1023,6 @@ static irqreturn_t gmu_irq_handler(int irq, void *data) dev_err_ratelimited(&gmu->pdev->dev, "GMU watchdog expired interrupt received\n"); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(device); } if (status & GMU_INT_HOST_AHB_BUS_ERR) dev_err_ratelimited(&gmu->pdev->dev, diff --git a/drivers/gpu/msm/kgsl_hfi.c b/drivers/gpu/msm/kgsl_hfi.c index 0ef1ec54e6ee..04f1212b9c27 100644 --- a/drivers/gpu/msm/kgsl_hfi.c +++ b/drivers/gpu/msm/kgsl_hfi.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2018-2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2018-2020, The Linux Foundation. All rights reserved. */ #include @@ -843,7 +843,6 @@ irqreturn_t hfi_irq_handler(int irq, void *data) struct kgsl_device *device = data; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); struct kgsl_hfi *hfi = &gmu->hfi; - struct adreno_device *adreno_dev = ADRENO_DEVICE(device); unsigned int status = 0; adreno_read_gmureg(ADRENO_DEVICE(device), @@ -853,12 +852,9 @@ irqreturn_t hfi_irq_handler(int irq, void *data) if (status & HFI_IRQ_DBGQ_MASK) tasklet_hi_schedule(&hfi->tasklet); - if (status & HFI_IRQ_CM3_FAULT_MASK) { + if (status & HFI_IRQ_CM3_FAULT_MASK) dev_err_ratelimited(&gmu->pdev->dev, "GMU CM3 fault interrupt received\n"); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(device); - } if (status & ~HFI_IRQ_MASK) dev_err_ratelimited(&gmu->pdev->dev, "Unhandled HFI interrupts 0x%lx\n", From adf19f375b9b5aeb53024a036a95072032b98706 Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Tue, 4 Dec 2018 13:59:44 -0700 Subject: [PATCH 6/8] msm: kgsl: Correctly handle oob and fenced write failures These both happen when gx is collapsed so its safe to assume that nothing is inflight. Take an inline snapshot and then request for a dispatcher based recovery. Change-Id: I4a3438660953d38406dd8e70868fb38172f9e967 Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/adreno.h | 3 ++- drivers/gpu/msm/adreno_a6xx_preempt.c | 7 ++++--- drivers/gpu/msm/adreno_ringbuffer.c | 14 ++++++++++---- 3 files changed, 16 insertions(+), 8 deletions(-) diff --git a/drivers/gpu/msm/adreno.h b/drivers/gpu/msm/adreno.h index c9018adad2a0..d13c9a1570a1 100644 --- a/drivers/gpu/msm/adreno.h +++ b/drivers/gpu/msm/adreno.h @@ -1609,8 +1609,9 @@ static inline int adreno_perfcntr_active_oob_get(struct kgsl_device *device) if (!ret) { ret = gmu_core_dev_oob_set(device, oob_perfcntr); if (ret) { + gmu_core_snapshot(device); adreno_set_gpu_fault(ADRENO_DEVICE(device), - ADRENO_GMU_FAULT); + ADRENO_GMU_FAULT_SKIP_SNAPSHOT); adreno_dispatcher_schedule(device); kgsl_active_count_put(device); } diff --git a/drivers/gpu/msm/adreno_a6xx_preempt.c b/drivers/gpu/msm/adreno_a6xx_preempt.c index b6326148a74b..827b551d4f02 100644 --- a/drivers/gpu/msm/adreno_a6xx_preempt.c +++ b/drivers/gpu/msm/adreno_a6xx_preempt.c @@ -373,10 +373,11 @@ void a6xx_preemption_trigger(struct adreno_device *adreno_dev) return; err: - - /* If fenced write fails, set the fault and trigger recovery */ + /* If fenced write fails, take inline snapshot and trigger recovery */ + if (!in_interrupt()) + gmu_core_snapshot(device); adreno_set_preempt_state(adreno_dev, ADRENO_PREEMPT_NONE); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); + adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT_SKIP_SNAPSHOT); adreno_dispatcher_schedule(device); /* Clear the keep alive */ if (gmu_core_isenabled(device)) diff --git a/drivers/gpu/msm/adreno_ringbuffer.c b/drivers/gpu/msm/adreno_ringbuffer.c index efe608fab0c4..175bf756ad0f 100644 --- a/drivers/gpu/msm/adreno_ringbuffer.c +++ b/drivers/gpu/msm/adreno_ringbuffer.c @@ -63,6 +63,7 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev, static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, struct adreno_ringbuffer *rb) { + struct kgsl_device *device = KGSL_DEVICE(adreno_dev); unsigned long flags; int ret = 0; @@ -74,7 +75,7 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, * Let the pwrscale policy know that new commands have * been submitted. */ - kgsl_pwrscale_busy(KGSL_DEVICE(adreno_dev)); + kgsl_pwrscale_busy(device); /* * Ensure the write posted after a possible @@ -99,9 +100,14 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, spin_unlock_irqrestore(&rb->preempt_lock, flags); if (ret) { - /* If WPTR update fails, set the fault and trigger recovery */ - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(KGSL_DEVICE(adreno_dev)); + /* + * If WPTR update fails, take inline snapshot and trigger + * recovery. + */ + gmu_core_snapshot(device); + adreno_set_gpu_fault(adreno_dev, + ADRENO_GMU_FAULT_SKIP_SNAPSHOT); + adreno_dispatcher_schedule(device); } } From 7636538fac91e6493244bd69ad3654168d28900e Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Thu, 8 Aug 2019 21:15:30 -0600 Subject: [PATCH 7/8] msm: kgsl: Restart a6xx gpu only once MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, as part of fault recovery, we try to restart the a6xx gpu five times. This loop is for legacy reasons for a hardware condition that existed on a4xx. That doesn’t exist for a6xx so remove the loop. Also, if there are no need to keep GPU active, we should transition back to SLUMBER. Change-Id: Ib6f7437bab2b424c4f632efca459ff8c3fd064e4 Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/adreno_a6xx.c | 30 ++++++++---------------------- 1 file changed, 8 insertions(+), 22 deletions(-) diff --git a/drivers/gpu/msm/adreno_a6xx.c b/drivers/gpu/msm/adreno_a6xx.c index 6042e4451527..52e709797d5e 100644 --- a/drivers/gpu/msm/adreno_a6xx.c +++ b/drivers/gpu/msm/adreno_a6xx.c @@ -1073,8 +1073,7 @@ static int64_t a6xx_read_throttling_counters(struct adreno_device *adreno_dev) static int a6xx_reset(struct kgsl_device *device, int fault) { struct adreno_device *adreno_dev = ADRENO_DEVICE(device); - int ret = -EINVAL; - int i = 0; + int ret; /* Use the regular reset sequence for No GMU */ if (!gmu_core_isenabled(device)) @@ -1086,33 +1085,20 @@ static int a6xx_reset(struct kgsl_device *device, int fault) /* since device is officially off now clear start bit */ clear_bit(ADRENO_DEVICE_STARTED, &adreno_dev->priv); - /* Keep trying to start the device until it works */ - for (i = 0; i < NUM_TIMES_RESET_RETRY; i++) { - ret = adreno_start(device, 0); - if (!ret) - break; - - msleep(20); - } - + ret = adreno_start(device, 0); if (ret) return ret; - if (i != 0) - dev_warn(device->dev, - "Device hard reset tried %d tries\n", i); + kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE); /* - * If active_cnt is non-zero then the system was active before - * going into a reset - put it back in that state + * If active_cnt is zero, there is no need to keep the GPU active. So, + * we should transition to SLUMBER. */ + if (!atomic_read(&device->active_cnt)) + kgsl_pwrctrl_change_state(device, KGSL_STATE_SLUMBER); - if (atomic_read(&device->active_cnt)) - kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE); - else - kgsl_pwrctrl_change_state(device, KGSL_STATE_NAP); - - return ret; + return 0; } static void a6xx_cp_hw_err_callback(struct adreno_device *adreno_dev, int bit) From 20a71ac90934b6cf1e702ba471d680b37e87583e Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Fri, 14 Feb 2020 17:14:41 -0700 Subject: [PATCH 8/8] msm: kgsl: Do not send NMI to GMU on CM3 fault Say GMU hits a hard fault and saves critical debug state and sends CM3 fault interrupt. It is possible that subsequent handshake between kgsl and GMU will timeout and kgsl will trigger GMU snapshot. This will send an NMI to GMU, which will over-write the critical debug state. To avoid this, check if there was a CM3 fault before sending NMI to the GMU. Change-Id: I0d4d4af3efe31110123f614c8dd8fe8d4a04c812 Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/adreno_a6xx_gmu.c | 3 +++ drivers/gpu/msm/kgsl_gmu.c | 30 ++++++++++++++++++++++++++---- drivers/gpu/msm/kgsl_gmu.h | 2 ++ drivers/gpu/msm/kgsl_hfi.c | 7 ++++++- 4 files changed, 37 insertions(+), 5 deletions(-) diff --git a/drivers/gpu/msm/adreno_a6xx_gmu.c b/drivers/gpu/msm/adreno_a6xx_gmu.c index e5dde885f890..5dae8152f4c3 100644 --- a/drivers/gpu/msm/adreno_a6xx_gmu.c +++ b/drivers/gpu/msm/adreno_a6xx_gmu.c @@ -1104,6 +1104,9 @@ static int a6xx_gmu_fw_start(struct kgsl_device *device, /* Populate the GMU version info before GMU boots */ load_gmu_version_info(device); + /* Clear any previously set cm3 fault */ + atomic_set(&gmu->cm3_fault, 0); + ret = a6xx_gmu_start(device); if (ret) return ret; diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index 98b0b0688e72..f7b28d9a70a3 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -1014,7 +1014,17 @@ static irqreturn_t gmu_irq_handler(int irq, void *data) ADRENO_REG_GMU_AO_HOST_INTERRUPT_MASK, (mask | GMU_INT_WDOG_BITE)); - adreno_gmu_send_nmi(adreno_dev); + /* make sure we're reading the latest cm3_fault */ + smp_rmb(); + + /* + * We should not send NMI if there was a CM3 fault reported + * because we don't want to overwrite the critical CM3 state + * captured by gmu before it sent the CM3 fault interrupt. + */ + if (!atomic_read(&gmu->cm3_fault)) + adreno_gmu_send_nmi(adreno_dev); + /* * There is sufficient delay for the GMU to have finished * handling the NMI before snapshot is taken, as the fault @@ -1532,9 +1542,21 @@ static void gmu_snapshot(struct kgsl_device *device) if (test_and_set_bit(GMU_FAULT, &device->gmu_core.flags)) return; - adreno_gmu_send_nmi(adreno_dev); - /* Wait for the NMI to be handled */ - udelay(100); + /* make sure we're reading the latest cm3_fault */ + smp_rmb(); + + /* + * We should not send NMI if there was a CM3 fault reported because we + * don't want to overwrite the critical CM3 state captured by gmu before + * it sent the CM3 fault interrupt. + */ + if (!atomic_read(&gmu->cm3_fault)) { + adreno_gmu_send_nmi(adreno_dev); + + /* Wait for the NMI to be handled */ + udelay(100); + } + kgsl_device_snapshot(device, NULL, true); adreno_write_gmureg(adreno_dev, diff --git a/drivers/gpu/msm/kgsl_gmu.h b/drivers/gpu/msm/kgsl_gmu.h index 312f7d5256d5..7c6ca7e03b5c 100644 --- a/drivers/gpu/msm/kgsl_gmu.h +++ b/drivers/gpu/msm/kgsl_gmu.h @@ -210,6 +210,8 @@ struct gmu_device { unsigned long kmem_bitmap; const struct gmu_vma_entry *vma; unsigned int log_wptr_retention; + /** @cm3_fault: whether gmu received a cm3 fault interrupt */ + atomic_t cm3_fault; }; struct gmu_memdesc *gmu_get_memdesc(struct gmu_device *gmu, diff --git a/drivers/gpu/msm/kgsl_hfi.c b/drivers/gpu/msm/kgsl_hfi.c index 04f1212b9c27..a7c5f3a307a5 100644 --- a/drivers/gpu/msm/kgsl_hfi.c +++ b/drivers/gpu/msm/kgsl_hfi.c @@ -852,9 +852,14 @@ irqreturn_t hfi_irq_handler(int irq, void *data) if (status & HFI_IRQ_DBGQ_MASK) tasklet_hi_schedule(&hfi->tasklet); - if (status & HFI_IRQ_CM3_FAULT_MASK) + if (status & HFI_IRQ_CM3_FAULT_MASK) { dev_err_ratelimited(&gmu->pdev->dev, "GMU CM3 fault interrupt received\n"); + atomic_set(&gmu->cm3_fault, 1); + + /* make sure other CPUs see the update */ + smp_wmb(); + } if (status & ~HFI_IRQ_MASK) dev_err_ratelimited(&gmu->pdev->dev, "Unhandled HFI interrupts 0x%lx\n",