Merge "msm: kgsl: Do not send NMI to GMU on CM3 fault"

This commit is contained in:
qctecmdr 2020-03-12 01:49:56 -07:00 • committed by Gerrit - the friendly Code Review server
commit 8a8b86fb81
10 changed files with 127 additions and 89 deletions

View file

@ -2057,12 +2057,16 @@ static int _adreno_start(struct adreno_device *adreno_dev)
/* Send OOB request to turn on the GX */
status = gmu_core_dev_oob_set(device, oob_gpu);
if (status)
goto error_boot_oob_clear;
if (status) {
gmu_core_snapshot(device);
goto error_oob_clear;
}
status = gmu_core_dev_hfi_start_msg(device);
if (status)
if (status) {
gmu_core_snapshot(device);
goto error_oob_clear;
}
if (device->pwrctrl.bus_control) {
/* VBIF waiting for RAM */
@ -2274,29 +2278,15 @@ static int adreno_stop(struct kgsl_device *device)
error = gmu_core_dev_oob_set(device, oob_gpu);
if (error) {
gmu_core_dev_oob_clear(device, oob_gpu);
if (gmu_core_regulator_isenabled(device)) {
/* GPU is on. Try recovery */
set_bit(GMU_FAULT, &device->gmu_core.flags);
gmu_core_snapshot(device);
error = -EINVAL;
}
goto no_gx_power;
}
adreno_dispatcher_stop(adreno_dev);
adreno_ringbuffer_stop(adreno_dev);
kgsl_pwrscale_update_stats(device);
adreno_irqctrl(adreno_dev, 0);
if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice))
llcc_slice_deactivate(adreno_dev->gpu_llc_slice);
if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice))
llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice);
/* Save active coresight registers if applicable */
adreno_coresight_stop(adreno_dev);
@ -2314,19 +2304,23 @@ static int adreno_stop(struct kgsl_device *device)
*/
if (!error && gmu_core_dev_wait_for_lowest_idle(device)) {
set_bit(GMU_FAULT, &device->gmu_core.flags);
gmu_core_snapshot(device);
/*
* Assume GMU hang after 10ms without responding.
* It shall be relative safe to clear vbif and stop
* MMU later. Early return in adreno_stop function
* will result in kernel panic in adreno_start
*/
error = -EINVAL;
}
adreno_clear_pending_transactions(device);
no_gx_power:
adreno_dispatcher_stop(adreno_dev);
adreno_ringbuffer_stop(adreno_dev);
if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice))
llcc_slice_deactivate(adreno_dev->gpu_llc_slice);
if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice))
llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice);
/*
* The halt is not cleared in the above function if we have GBIF.
* Clear it here if GMU is enabled as GMU stop needs access to
@ -2996,7 +2990,15 @@ void adreno_spin_idle_debug(struct adreno_device *adreno_dev,
dev_err(device->dev, " hwfault=%8.8X\n", hwfault);
kgsl_device_snapshot(device, NULL, adreno_gmu_gpu_fault(adreno_dev));
/*
* If CP is stuck, gmu may not perform as expected. So force a gmu
* snapshot which captures entire state as well as sets the gmu fault
* because things need to be reset anyway.
*/
if (gmu_core_isenabled(device))
gmu_core_snapshot(device);
else
kgsl_device_snapshot(device, NULL, false);
}
/**

View file

@ -1609,8 +1609,9 @@ static inline int adreno_perfcntr_active_oob_get(struct kgsl_device *device)
if (!ret) {
ret = gmu_core_dev_oob_set(device, oob_perfcntr);
if (ret) {
gmu_core_snapshot(device);
adreno_set_gpu_fault(ADRENO_DEVICE(device),
ADRENO_GMU_FAULT);
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
kgsl_active_count_put(device);
}

View file

@ -1076,8 +1076,7 @@ static int64_t a6xx_read_throttling_counters(struct adreno_device *adreno_dev)
static int a6xx_reset(struct kgsl_device *device, int fault)
{
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
int ret = -EINVAL;
int i = 0;
int ret;
/* Use the regular reset sequence for No GMU */
if (!gmu_core_isenabled(device))
@ -1089,33 +1088,20 @@ static int a6xx_reset(struct kgsl_device *device, int fault)
/* since device is officially off now clear start bit */
clear_bit(ADRENO_DEVICE_STARTED, &adreno_dev->priv);
/* Keep trying to start the device until it works */
for (i = 0; i < NUM_TIMES_RESET_RETRY; i++) {
ret = adreno_start(device, 0);
if (!ret)
break;
msleep(20);
}
ret = adreno_start(device, 0);
if (ret)
return ret;
if (i != 0)
dev_warn(device->dev,
"Device hard reset tried %d tries\n", i);
kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE);
/*
* If active_cnt is non-zero then the system was active before
* going into a reset - put it back in that state
* If active_cnt is zero, there is no need to keep the GPU active. So,
* we should transition to SLUMBER.
*/
if (!atomic_read(&device->active_cnt))
kgsl_pwrctrl_change_state(device, KGSL_STATE_SLUMBER);
if (atomic_read(&device->active_cnt))
kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE);
else
kgsl_pwrctrl_change_state(device, KGSL_STATE_NAP);
return ret;
return 0;
}
static void a6xx_cp_hw_err_callback(struct adreno_device *adreno_dev, int bit)

View file

@ -1104,6 +1104,9 @@ static int a6xx_gmu_fw_start(struct kgsl_device *device,
/* Populate the GMU version info before GMU boots */
load_gmu_version_info(device);
/* Clear any previously set cm3 fault */
atomic_set(&gmu->cm3_fault, 0);
ret = a6xx_gmu_start(device);
if (ret)
return ret;

View file

@ -373,10 +373,11 @@ void a6xx_preemption_trigger(struct adreno_device *adreno_dev)
return;
err:
/* If fenced write fails, set the fault and trigger recovery */
/* If fenced write fails, take inline snapshot and trigger recovery */
if (!in_interrupt())
gmu_core_snapshot(device);
adreno_set_preempt_state(adreno_dev, ADRENO_PREEMPT_NONE);
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
/* Clear the keep alive */
if (gmu_core_isenabled(device))

View file

@ -63,6 +63,7 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev,
static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
struct adreno_ringbuffer *rb)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
unsigned long flags;
int ret = 0;
@ -74,7 +75,7 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
* Let the pwrscale policy know that new commands have
* been submitted.
*/
kgsl_pwrscale_busy(KGSL_DEVICE(adreno_dev));
kgsl_pwrscale_busy(device);
/*
* Ensure the write posted after a possible
@ -99,9 +100,14 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
spin_unlock_irqrestore(&rb->preempt_lock, flags);
if (ret) {
/* If WPTR update fails, set the fault and trigger recovery */
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
adreno_dispatcher_schedule(KGSL_DEVICE(adreno_dev));
/*
* If WPTR update fails, take inline snapshot and trigger
* recovery.
*/
gmu_core_snapshot(device);
adreno_set_gpu_fault(adreno_dev,
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
}
}

View file

@ -1014,7 +1014,17 @@ static irqreturn_t gmu_irq_handler(int irq, void *data)
ADRENO_REG_GMU_AO_HOST_INTERRUPT_MASK,
(mask | GMU_INT_WDOG_BITE));
adreno_gmu_send_nmi(adreno_dev);
/* make sure we're reading the latest cm3_fault */
smp_rmb();
/*
* We should not send NMI if there was a CM3 fault reported
* because we don't want to overwrite the critical CM3 state
* captured by gmu before it sent the CM3 fault interrupt.
*/
if (!atomic_read(&gmu->cm3_fault))
adreno_gmu_send_nmi(adreno_dev);
/*
* There is sufficient delay for the GMU to have finished
* handling the NMI before snapshot is taken, as the fault
@ -1023,8 +1033,6 @@ static irqreturn_t gmu_irq_handler(int irq, void *data)
dev_err_ratelimited(&gmu->pdev->dev,
"GMU watchdog expired interrupt received\n");
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
adreno_dispatcher_schedule(device);
}
if (status & GMU_INT_HOST_AHB_BUS_ERR)
dev_err_ratelimited(&gmu->pdev->dev,
@ -1530,9 +1538,25 @@ static void gmu_snapshot(struct kgsl_device *device)
struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device);
struct gmu_device *gmu = KGSL_GMU_DEVICE(device);
adreno_gmu_send_nmi(adreno_dev);
/* Wait for the NMI to be handled */
udelay(100);
/* Abstain from sending another nmi or over-writing snapshot */
if (test_and_set_bit(GMU_FAULT, &device->gmu_core.flags))
return;
/* make sure we're reading the latest cm3_fault */
smp_rmb();
/*
* We should not send NMI if there was a CM3 fault reported because we
* don't want to overwrite the critical CM3 state captured by gmu before
* it sent the CM3 fault interrupt.
*/
if (!atomic_read(&gmu->cm3_fault)) {
adreno_gmu_send_nmi(adreno_dev);
/* Wait for the NMI to be handled */
udelay(100);
}
kgsl_device_snapshot(device, NULL, true);
adreno_write_gmureg(adreno_dev,
@ -1670,6 +1694,12 @@ static void gmu_stop(struct kgsl_device *device)
if (!test_bit(GMU_CLK_ON, &device->gmu_core.flags))
return;
/* Force suspend if gmu is already in fault */
if (test_bit(GMU_FAULT, &device->gmu_core.flags)) {
gmu_core_suspend(device);
return;
}
/* Wait for the lowest idle level we requested */
if (gmu_core_dev_wait_for_lowest_idle(device))
goto error;
@ -1695,14 +1725,13 @@ static void gmu_stop(struct kgsl_device *device)
return;
error:
/*
* The power controller will change state to SLUMBER anyway
* Set GMU_FAULT flag to indicate to power contrller
* that hang recovery is needed to power on GPU
*/
set_bit(GMU_FAULT, &device->gmu_core.flags);
dev_err(&gmu->pdev->dev, "Failed to stop GMU\n");
gmu_core_snapshot(device);
/*
* We failed to stop the gmu successfully. Force a suspend
* to set things up for a fresh start.
*/
gmu_core_suspend(device);
}
static void gmu_remove(struct kgsl_device *device)

View file

@ -210,6 +210,8 @@ struct gmu_device {
unsigned long kmem_bitmap;
const struct gmu_vma_entry *vma;
unsigned int log_wptr_retention;
/** @cm3_fault: whether gmu received a cm3 fault interrupt */
atomic_t cm3_fault;
};
struct gmu_memdesc *gmu_get_memdesc(struct gmu_device *gmu,

View file

@ -1,6 +1,6 @@
// SPDX-License-Identifier: GPL-2.0-only
/*
* Copyright (c) 2018-2019, The Linux Foundation. All rights reserved.
* Copyright (c) 2018-2020, The Linux Foundation. All rights reserved.
*/
#include <linux/delay.h>
@ -843,7 +843,6 @@ irqreturn_t hfi_irq_handler(int irq, void *data)
struct kgsl_device *device = data;
struct gmu_device *gmu = KGSL_GMU_DEVICE(device);
struct kgsl_hfi *hfi = &gmu->hfi;
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
unsigned int status = 0;
adreno_read_gmureg(ADRENO_DEVICE(device),
@ -856,8 +855,10 @@ irqreturn_t hfi_irq_handler(int irq, void *data)
if (status & HFI_IRQ_CM3_FAULT_MASK) {
dev_err_ratelimited(&gmu->pdev->dev,
"GMU CM3 fault interrupt received\n");
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
adreno_dispatcher_schedule(device);
atomic_set(&gmu->cm3_fault, 1);
/* make sure other CPUs see the update */
smp_wmb();
}
if (status & ~HFI_IRQ_MASK)
dev_err_ratelimited(&gmu->pdev->dev,

View file

@ -1695,13 +1695,23 @@ static int _init(struct kgsl_device *device)
int status = 0;
switch (device->state) {
case KGSL_STATE_RESET:
if (gmu_core_isenabled(device)) {
/*
* If we fail a INIT -> AWARE transition, we will
* transition back to INIT. However, we must hard reset
* the GMU as we go back to INIT. This is done by
* forcing a RESET -> INIT transition.
*/
gmu_core_suspend(device);
kgsl_pwrctrl_set_state(device, KGSL_STATE_INIT);
}
break;
case KGSL_STATE_NAP:
/* Force power on to do the stop */
status = kgsl_pwrctrl_enable(device);
/* fall through */
case KGSL_STATE_ACTIVE:
/* fall through */
case KGSL_STATE_RESET:
kgsl_pwrctrl_irq(device, KGSL_PWRFLAGS_OFF);
del_timer_sync(&device->idle_timer);
kgsl_pwrscale_midframe_timer_cancel(device);
@ -1813,12 +1823,6 @@ _aware(struct kgsl_device *device)
status = gmu_core_start(device);
break;
case KGSL_STATE_INIT:
/* if GMU already in FAULT */
if (gmu_core_isenabled(device) &&
test_bit(GMU_FAULT, &device->gmu_core.flags)) {
status = -EINVAL;
break;
}
status = kgsl_pwrctrl_enable(device);
break;
/* The following 3 cases shouldn't occur, but don't panic. */
@ -1832,20 +1836,23 @@ _aware(struct kgsl_device *device)
break;
case KGSL_STATE_SLUMBER:
status = kgsl_pwrctrl_enable(device);
if (status && gmu_core_isenabled(device))
/*
* SLUMBER -> AWARE failed which means GMU boot failed.
* Make sure we reset the GMU while transitioning back
* to SLUMBER.
*/
kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET);
break;
default:
status = -EINVAL;
}
if (!status)
if (status && gmu_core_isenabled(device))
/*
* If a SLUMBER/INIT -> AWARE fails, we transition back to
* SLUMBER/INIT state. We must hard reset the GMU while
* transitioning back to SLUMBER/INIT. A RESET -> AWARE
* transition is different. It happens when dispatcher is
* attempting reset/recovery as part of fault handling. If it
* fails, we should still transition back to RESET in case
* we want to attempt another reset/recovery.
*/
kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET);
else
kgsl_pwrctrl_set_state(device, KGSL_STATE_AWARE);
return status;