diff --git a/drivers/gpu/msm/adreno.c b/drivers/gpu/msm/adreno.c index 203521d9ce7c..4724b76afa0a 100644 --- a/drivers/gpu/msm/adreno.c +++ b/drivers/gpu/msm/adreno.c @@ -2057,12 +2057,16 @@ static int _adreno_start(struct adreno_device *adreno_dev) /* Send OOB request to turn on the GX */ status = gmu_core_dev_oob_set(device, oob_gpu); - if (status) - goto error_boot_oob_clear; + if (status) { + gmu_core_snapshot(device); + goto error_oob_clear; + } status = gmu_core_dev_hfi_start_msg(device); - if (status) + if (status) { + gmu_core_snapshot(device); goto error_oob_clear; + } if (device->pwrctrl.bus_control) { /* VBIF waiting for RAM */ @@ -2274,29 +2278,15 @@ static int adreno_stop(struct kgsl_device *device) error = gmu_core_dev_oob_set(device, oob_gpu); if (error) { gmu_core_dev_oob_clear(device, oob_gpu); - - if (gmu_core_regulator_isenabled(device)) { - /* GPU is on. Try recovery */ - set_bit(GMU_FAULT, &device->gmu_core.flags); gmu_core_snapshot(device); error = -EINVAL; - } + goto no_gx_power; } - adreno_dispatcher_stop(adreno_dev); - - adreno_ringbuffer_stop(adreno_dev); - kgsl_pwrscale_update_stats(device); adreno_irqctrl(adreno_dev, 0); - if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice)) - llcc_slice_deactivate(adreno_dev->gpu_llc_slice); - - if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice)) - llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice); - /* Save active coresight registers if applicable */ adreno_coresight_stop(adreno_dev); @@ -2314,19 +2304,23 @@ static int adreno_stop(struct kgsl_device *device) */ if (!error && gmu_core_dev_wait_for_lowest_idle(device)) { - set_bit(GMU_FAULT, &device->gmu_core.flags); gmu_core_snapshot(device); - /* - * Assume GMU hang after 10ms without responding. - * It shall be relative safe to clear vbif and stop - * MMU later. Early return in adreno_stop function - * will result in kernel panic in adreno_start - */ error = -EINVAL; } adreno_clear_pending_transactions(device); +no_gx_power: + adreno_dispatcher_stop(adreno_dev); + + adreno_ringbuffer_stop(adreno_dev); + + if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice)) + llcc_slice_deactivate(adreno_dev->gpu_llc_slice); + + if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice)) + llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice); + /* * The halt is not cleared in the above function if we have GBIF. * Clear it here if GMU is enabled as GMU stop needs access to @@ -2996,7 +2990,15 @@ void adreno_spin_idle_debug(struct adreno_device *adreno_dev, dev_err(device->dev, " hwfault=%8.8X\n", hwfault); - kgsl_device_snapshot(device, NULL, adreno_gmu_gpu_fault(adreno_dev)); + /* + * If CP is stuck, gmu may not perform as expected. So force a gmu + * snapshot which captures entire state as well as sets the gmu fault + * because things need to be reset anyway. + */ + if (gmu_core_isenabled(device)) + gmu_core_snapshot(device); + else + kgsl_device_snapshot(device, NULL, false); } /** diff --git a/drivers/gpu/msm/adreno.h b/drivers/gpu/msm/adreno.h index c9018adad2a0..d13c9a1570a1 100644 --- a/drivers/gpu/msm/adreno.h +++ b/drivers/gpu/msm/adreno.h @@ -1609,8 +1609,9 @@ static inline int adreno_perfcntr_active_oob_get(struct kgsl_device *device) if (!ret) { ret = gmu_core_dev_oob_set(device, oob_perfcntr); if (ret) { + gmu_core_snapshot(device); adreno_set_gpu_fault(ADRENO_DEVICE(device), - ADRENO_GMU_FAULT); + ADRENO_GMU_FAULT_SKIP_SNAPSHOT); adreno_dispatcher_schedule(device); kgsl_active_count_put(device); } diff --git a/drivers/gpu/msm/adreno_a6xx.c b/drivers/gpu/msm/adreno_a6xx.c index 8a5c41cad415..7eb27fc41e18 100644 --- a/drivers/gpu/msm/adreno_a6xx.c +++ b/drivers/gpu/msm/adreno_a6xx.c @@ -1076,8 +1076,7 @@ static int64_t a6xx_read_throttling_counters(struct adreno_device *adreno_dev) static int a6xx_reset(struct kgsl_device *device, int fault) { struct adreno_device *adreno_dev = ADRENO_DEVICE(device); - int ret = -EINVAL; - int i = 0; + int ret; /* Use the regular reset sequence for No GMU */ if (!gmu_core_isenabled(device)) @@ -1089,33 +1088,20 @@ static int a6xx_reset(struct kgsl_device *device, int fault) /* since device is officially off now clear start bit */ clear_bit(ADRENO_DEVICE_STARTED, &adreno_dev->priv); - /* Keep trying to start the device until it works */ - for (i = 0; i < NUM_TIMES_RESET_RETRY; i++) { - ret = adreno_start(device, 0); - if (!ret) - break; - - msleep(20); - } - + ret = adreno_start(device, 0); if (ret) return ret; - if (i != 0) - dev_warn(device->dev, - "Device hard reset tried %d tries\n", i); + kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE); /* - * If active_cnt is non-zero then the system was active before - * going into a reset - put it back in that state + * If active_cnt is zero, there is no need to keep the GPU active. So, + * we should transition to SLUMBER. */ + if (!atomic_read(&device->active_cnt)) + kgsl_pwrctrl_change_state(device, KGSL_STATE_SLUMBER); - if (atomic_read(&device->active_cnt)) - kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE); - else - kgsl_pwrctrl_change_state(device, KGSL_STATE_NAP); - - return ret; + return 0; } static void a6xx_cp_hw_err_callback(struct adreno_device *adreno_dev, int bit) diff --git a/drivers/gpu/msm/adreno_a6xx_gmu.c b/drivers/gpu/msm/adreno_a6xx_gmu.c index e5dde885f890..5dae8152f4c3 100644 --- a/drivers/gpu/msm/adreno_a6xx_gmu.c +++ b/drivers/gpu/msm/adreno_a6xx_gmu.c @@ -1104,6 +1104,9 @@ static int a6xx_gmu_fw_start(struct kgsl_device *device, /* Populate the GMU version info before GMU boots */ load_gmu_version_info(device); + /* Clear any previously set cm3 fault */ + atomic_set(&gmu->cm3_fault, 0); + ret = a6xx_gmu_start(device); if (ret) return ret; diff --git a/drivers/gpu/msm/adreno_a6xx_preempt.c b/drivers/gpu/msm/adreno_a6xx_preempt.c index b6326148a74b..827b551d4f02 100644 --- a/drivers/gpu/msm/adreno_a6xx_preempt.c +++ b/drivers/gpu/msm/adreno_a6xx_preempt.c @@ -373,10 +373,11 @@ void a6xx_preemption_trigger(struct adreno_device *adreno_dev) return; err: - - /* If fenced write fails, set the fault and trigger recovery */ + /* If fenced write fails, take inline snapshot and trigger recovery */ + if (!in_interrupt()) + gmu_core_snapshot(device); adreno_set_preempt_state(adreno_dev, ADRENO_PREEMPT_NONE); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); + adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT_SKIP_SNAPSHOT); adreno_dispatcher_schedule(device); /* Clear the keep alive */ if (gmu_core_isenabled(device)) diff --git a/drivers/gpu/msm/adreno_ringbuffer.c b/drivers/gpu/msm/adreno_ringbuffer.c index efe608fab0c4..175bf756ad0f 100644 --- a/drivers/gpu/msm/adreno_ringbuffer.c +++ b/drivers/gpu/msm/adreno_ringbuffer.c @@ -63,6 +63,7 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev, static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, struct adreno_ringbuffer *rb) { + struct kgsl_device *device = KGSL_DEVICE(adreno_dev); unsigned long flags; int ret = 0; @@ -74,7 +75,7 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, * Let the pwrscale policy know that new commands have * been submitted. */ - kgsl_pwrscale_busy(KGSL_DEVICE(adreno_dev)); + kgsl_pwrscale_busy(device); /* * Ensure the write posted after a possible @@ -99,9 +100,14 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev, spin_unlock_irqrestore(&rb->preempt_lock, flags); if (ret) { - /* If WPTR update fails, set the fault and trigger recovery */ - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(KGSL_DEVICE(adreno_dev)); + /* + * If WPTR update fails, take inline snapshot and trigger + * recovery. + */ + gmu_core_snapshot(device); + adreno_set_gpu_fault(adreno_dev, + ADRENO_GMU_FAULT_SKIP_SNAPSHOT); + adreno_dispatcher_schedule(device); } } diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index a5d5a86cd5df..f7b28d9a70a3 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -1014,7 +1014,17 @@ static irqreturn_t gmu_irq_handler(int irq, void *data) ADRENO_REG_GMU_AO_HOST_INTERRUPT_MASK, (mask | GMU_INT_WDOG_BITE)); - adreno_gmu_send_nmi(adreno_dev); + /* make sure we're reading the latest cm3_fault */ + smp_rmb(); + + /* + * We should not send NMI if there was a CM3 fault reported + * because we don't want to overwrite the critical CM3 state + * captured by gmu before it sent the CM3 fault interrupt. + */ + if (!atomic_read(&gmu->cm3_fault)) + adreno_gmu_send_nmi(adreno_dev); + /* * There is sufficient delay for the GMU to have finished * handling the NMI before snapshot is taken, as the fault @@ -1023,8 +1033,6 @@ static irqreturn_t gmu_irq_handler(int irq, void *data) dev_err_ratelimited(&gmu->pdev->dev, "GMU watchdog expired interrupt received\n"); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(device); } if (status & GMU_INT_HOST_AHB_BUS_ERR) dev_err_ratelimited(&gmu->pdev->dev, @@ -1530,9 +1538,25 @@ static void gmu_snapshot(struct kgsl_device *device) struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); struct gmu_device *gmu = KGSL_GMU_DEVICE(device); - adreno_gmu_send_nmi(adreno_dev); - /* Wait for the NMI to be handled */ - udelay(100); + /* Abstain from sending another nmi or over-writing snapshot */ + if (test_and_set_bit(GMU_FAULT, &device->gmu_core.flags)) + return; + + /* make sure we're reading the latest cm3_fault */ + smp_rmb(); + + /* + * We should not send NMI if there was a CM3 fault reported because we + * don't want to overwrite the critical CM3 state captured by gmu before + * it sent the CM3 fault interrupt. + */ + if (!atomic_read(&gmu->cm3_fault)) { + adreno_gmu_send_nmi(adreno_dev); + + /* Wait for the NMI to be handled */ + udelay(100); + } + kgsl_device_snapshot(device, NULL, true); adreno_write_gmureg(adreno_dev, @@ -1670,6 +1694,12 @@ static void gmu_stop(struct kgsl_device *device) if (!test_bit(GMU_CLK_ON, &device->gmu_core.flags)) return; + /* Force suspend if gmu is already in fault */ + if (test_bit(GMU_FAULT, &device->gmu_core.flags)) { + gmu_core_suspend(device); + return; + } + /* Wait for the lowest idle level we requested */ if (gmu_core_dev_wait_for_lowest_idle(device)) goto error; @@ -1695,14 +1725,13 @@ static void gmu_stop(struct kgsl_device *device) return; error: - /* - * The power controller will change state to SLUMBER anyway - * Set GMU_FAULT flag to indicate to power contrller - * that hang recovery is needed to power on GPU - */ - set_bit(GMU_FAULT, &device->gmu_core.flags); dev_err(&gmu->pdev->dev, "Failed to stop GMU\n"); gmu_core_snapshot(device); + /* + * We failed to stop the gmu successfully. Force a suspend + * to set things up for a fresh start. + */ + gmu_core_suspend(device); } static void gmu_remove(struct kgsl_device *device) diff --git a/drivers/gpu/msm/kgsl_gmu.h b/drivers/gpu/msm/kgsl_gmu.h index 312f7d5256d5..7c6ca7e03b5c 100644 --- a/drivers/gpu/msm/kgsl_gmu.h +++ b/drivers/gpu/msm/kgsl_gmu.h @@ -210,6 +210,8 @@ struct gmu_device { unsigned long kmem_bitmap; const struct gmu_vma_entry *vma; unsigned int log_wptr_retention; + /** @cm3_fault: whether gmu received a cm3 fault interrupt */ + atomic_t cm3_fault; }; struct gmu_memdesc *gmu_get_memdesc(struct gmu_device *gmu, diff --git a/drivers/gpu/msm/kgsl_hfi.c b/drivers/gpu/msm/kgsl_hfi.c index 0ef1ec54e6ee..a7c5f3a307a5 100644 --- a/drivers/gpu/msm/kgsl_hfi.c +++ b/drivers/gpu/msm/kgsl_hfi.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2018-2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2018-2020, The Linux Foundation. All rights reserved. */ #include @@ -843,7 +843,6 @@ irqreturn_t hfi_irq_handler(int irq, void *data) struct kgsl_device *device = data; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); struct kgsl_hfi *hfi = &gmu->hfi; - struct adreno_device *adreno_dev = ADRENO_DEVICE(device); unsigned int status = 0; adreno_read_gmureg(ADRENO_DEVICE(device), @@ -856,8 +855,10 @@ irqreturn_t hfi_irq_handler(int irq, void *data) if (status & HFI_IRQ_CM3_FAULT_MASK) { dev_err_ratelimited(&gmu->pdev->dev, "GMU CM3 fault interrupt received\n"); - adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT); - adreno_dispatcher_schedule(device); + atomic_set(&gmu->cm3_fault, 1); + + /* make sure other CPUs see the update */ + smp_wmb(); } if (status & ~HFI_IRQ_MASK) dev_err_ratelimited(&gmu->pdev->dev, diff --git a/drivers/gpu/msm/kgsl_pwrctrl.c b/drivers/gpu/msm/kgsl_pwrctrl.c index fe2788e41c43..d611a042816f 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.c +++ b/drivers/gpu/msm/kgsl_pwrctrl.c @@ -1695,13 +1695,23 @@ static int _init(struct kgsl_device *device) int status = 0; switch (device->state) { + case KGSL_STATE_RESET: + if (gmu_core_isenabled(device)) { + /* + * If we fail a INIT -> AWARE transition, we will + * transition back to INIT. However, we must hard reset + * the GMU as we go back to INIT. This is done by + * forcing a RESET -> INIT transition. + */ + gmu_core_suspend(device); + kgsl_pwrctrl_set_state(device, KGSL_STATE_INIT); + } + break; case KGSL_STATE_NAP: /* Force power on to do the stop */ status = kgsl_pwrctrl_enable(device); /* fall through */ case KGSL_STATE_ACTIVE: - /* fall through */ - case KGSL_STATE_RESET: kgsl_pwrctrl_irq(device, KGSL_PWRFLAGS_OFF); del_timer_sync(&device->idle_timer); kgsl_pwrscale_midframe_timer_cancel(device); @@ -1813,12 +1823,6 @@ _aware(struct kgsl_device *device) status = gmu_core_start(device); break; case KGSL_STATE_INIT: - /* if GMU already in FAULT */ - if (gmu_core_isenabled(device) && - test_bit(GMU_FAULT, &device->gmu_core.flags)) { - status = -EINVAL; - break; - } status = kgsl_pwrctrl_enable(device); break; /* The following 3 cases shouldn't occur, but don't panic. */ @@ -1832,20 +1836,23 @@ _aware(struct kgsl_device *device) break; case KGSL_STATE_SLUMBER: status = kgsl_pwrctrl_enable(device); - if (status && gmu_core_isenabled(device)) - /* - * SLUMBER -> AWARE failed which means GMU boot failed. - * Make sure we reset the GMU while transitioning back - * to SLUMBER. - */ - kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET); - break; default: status = -EINVAL; } - if (!status) + if (status && gmu_core_isenabled(device)) + /* + * If a SLUMBER/INIT -> AWARE fails, we transition back to + * SLUMBER/INIT state. We must hard reset the GMU while + * transitioning back to SLUMBER/INIT. A RESET -> AWARE + * transition is different. It happens when dispatcher is + * attempting reset/recovery as part of fault handling. If it + * fails, we should still transition back to RESET in case + * we want to attempt another reset/recovery. + */ + kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET); + else kgsl_pwrctrl_set_state(device, KGSL_STATE_AWARE); return status;