mirror of
https://github.com/BobTheBlinker/android_kernel_motorola_sm6375.git
synced 2026-10-10 06:09:23 -04:00
Merge "msm: kgsl: Do not send NMI to GMU on CM3 fault"
This commit is contained in:
commit
8a8b86fb81
10 changed files with 127 additions and 89 deletions
|
|
@ -2057,12 +2057,16 @@ static int _adreno_start(struct adreno_device *adreno_dev)
|
|||
|
||||
/* Send OOB request to turn on the GX */
|
||||
status = gmu_core_dev_oob_set(device, oob_gpu);
|
||||
if (status)
|
||||
goto error_boot_oob_clear;
|
||||
if (status) {
|
||||
gmu_core_snapshot(device);
|
||||
goto error_oob_clear;
|
||||
}
|
||||
|
||||
status = gmu_core_dev_hfi_start_msg(device);
|
||||
if (status)
|
||||
if (status) {
|
||||
gmu_core_snapshot(device);
|
||||
goto error_oob_clear;
|
||||
}
|
||||
|
||||
if (device->pwrctrl.bus_control) {
|
||||
/* VBIF waiting for RAM */
|
||||
|
|
@ -2274,29 +2278,15 @@ static int adreno_stop(struct kgsl_device *device)
|
|||
error = gmu_core_dev_oob_set(device, oob_gpu);
|
||||
if (error) {
|
||||
gmu_core_dev_oob_clear(device, oob_gpu);
|
||||
|
||||
if (gmu_core_regulator_isenabled(device)) {
|
||||
/* GPU is on. Try recovery */
|
||||
set_bit(GMU_FAULT, &device->gmu_core.flags);
|
||||
gmu_core_snapshot(device);
|
||||
error = -EINVAL;
|
||||
}
|
||||
goto no_gx_power;
|
||||
}
|
||||
|
||||
adreno_dispatcher_stop(adreno_dev);
|
||||
|
||||
adreno_ringbuffer_stop(adreno_dev);
|
||||
|
||||
kgsl_pwrscale_update_stats(device);
|
||||
|
||||
adreno_irqctrl(adreno_dev, 0);
|
||||
|
||||
if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice))
|
||||
llcc_slice_deactivate(adreno_dev->gpu_llc_slice);
|
||||
|
||||
if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice))
|
||||
llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice);
|
||||
|
||||
/* Save active coresight registers if applicable */
|
||||
adreno_coresight_stop(adreno_dev);
|
||||
|
||||
|
|
@ -2314,19 +2304,23 @@ static int adreno_stop(struct kgsl_device *device)
|
|||
*/
|
||||
|
||||
if (!error && gmu_core_dev_wait_for_lowest_idle(device)) {
|
||||
set_bit(GMU_FAULT, &device->gmu_core.flags);
|
||||
gmu_core_snapshot(device);
|
||||
/*
|
||||
* Assume GMU hang after 10ms without responding.
|
||||
* It shall be relative safe to clear vbif and stop
|
||||
* MMU later. Early return in adreno_stop function
|
||||
* will result in kernel panic in adreno_start
|
||||
*/
|
||||
error = -EINVAL;
|
||||
}
|
||||
|
||||
adreno_clear_pending_transactions(device);
|
||||
|
||||
no_gx_power:
|
||||
adreno_dispatcher_stop(adreno_dev);
|
||||
|
||||
adreno_ringbuffer_stop(adreno_dev);
|
||||
|
||||
if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice))
|
||||
llcc_slice_deactivate(adreno_dev->gpu_llc_slice);
|
||||
|
||||
if (!IS_ERR_OR_NULL(adreno_dev->gpuhtw_llc_slice))
|
||||
llcc_slice_deactivate(adreno_dev->gpuhtw_llc_slice);
|
||||
|
||||
/*
|
||||
* The halt is not cleared in the above function if we have GBIF.
|
||||
* Clear it here if GMU is enabled as GMU stop needs access to
|
||||
|
|
@ -2996,7 +2990,15 @@ void adreno_spin_idle_debug(struct adreno_device *adreno_dev,
|
|||
|
||||
dev_err(device->dev, " hwfault=%8.8X\n", hwfault);
|
||||
|
||||
kgsl_device_snapshot(device, NULL, adreno_gmu_gpu_fault(adreno_dev));
|
||||
/*
|
||||
* If CP is stuck, gmu may not perform as expected. So force a gmu
|
||||
* snapshot which captures entire state as well as sets the gmu fault
|
||||
* because things need to be reset anyway.
|
||||
*/
|
||||
if (gmu_core_isenabled(device))
|
||||
gmu_core_snapshot(device);
|
||||
else
|
||||
kgsl_device_snapshot(device, NULL, false);
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -1609,8 +1609,9 @@ static inline int adreno_perfcntr_active_oob_get(struct kgsl_device *device)
|
|||
if (!ret) {
|
||||
ret = gmu_core_dev_oob_set(device, oob_perfcntr);
|
||||
if (ret) {
|
||||
gmu_core_snapshot(device);
|
||||
adreno_set_gpu_fault(ADRENO_DEVICE(device),
|
||||
ADRENO_GMU_FAULT);
|
||||
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
|
||||
adreno_dispatcher_schedule(device);
|
||||
kgsl_active_count_put(device);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1076,8 +1076,7 @@ static int64_t a6xx_read_throttling_counters(struct adreno_device *adreno_dev)
|
|||
static int a6xx_reset(struct kgsl_device *device, int fault)
|
||||
{
|
||||
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
|
||||
int ret = -EINVAL;
|
||||
int i = 0;
|
||||
int ret;
|
||||
|
||||
/* Use the regular reset sequence for No GMU */
|
||||
if (!gmu_core_isenabled(device))
|
||||
|
|
@ -1089,33 +1088,20 @@ static int a6xx_reset(struct kgsl_device *device, int fault)
|
|||
/* since device is officially off now clear start bit */
|
||||
clear_bit(ADRENO_DEVICE_STARTED, &adreno_dev->priv);
|
||||
|
||||
/* Keep trying to start the device until it works */
|
||||
for (i = 0; i < NUM_TIMES_RESET_RETRY; i++) {
|
||||
ret = adreno_start(device, 0);
|
||||
if (!ret)
|
||||
break;
|
||||
|
||||
msleep(20);
|
||||
}
|
||||
|
||||
ret = adreno_start(device, 0);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
if (i != 0)
|
||||
dev_warn(device->dev,
|
||||
"Device hard reset tried %d tries\n", i);
|
||||
kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE);
|
||||
|
||||
/*
|
||||
* If active_cnt is non-zero then the system was active before
|
||||
* going into a reset - put it back in that state
|
||||
* If active_cnt is zero, there is no need to keep the GPU active. So,
|
||||
* we should transition to SLUMBER.
|
||||
*/
|
||||
if (!atomic_read(&device->active_cnt))
|
||||
kgsl_pwrctrl_change_state(device, KGSL_STATE_SLUMBER);
|
||||
|
||||
if (atomic_read(&device->active_cnt))
|
||||
kgsl_pwrctrl_change_state(device, KGSL_STATE_ACTIVE);
|
||||
else
|
||||
kgsl_pwrctrl_change_state(device, KGSL_STATE_NAP);
|
||||
|
||||
return ret;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void a6xx_cp_hw_err_callback(struct adreno_device *adreno_dev, int bit)
|
||||
|
|
|
|||
|
|
@ -1104,6 +1104,9 @@ static int a6xx_gmu_fw_start(struct kgsl_device *device,
|
|||
/* Populate the GMU version info before GMU boots */
|
||||
load_gmu_version_info(device);
|
||||
|
||||
/* Clear any previously set cm3 fault */
|
||||
atomic_set(&gmu->cm3_fault, 0);
|
||||
|
||||
ret = a6xx_gmu_start(device);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
|
|
|||
|
|
@ -373,10 +373,11 @@ void a6xx_preemption_trigger(struct adreno_device *adreno_dev)
|
|||
|
||||
return;
|
||||
err:
|
||||
|
||||
/* If fenced write fails, set the fault and trigger recovery */
|
||||
/* If fenced write fails, take inline snapshot and trigger recovery */
|
||||
if (!in_interrupt())
|
||||
gmu_core_snapshot(device);
|
||||
adreno_set_preempt_state(adreno_dev, ADRENO_PREEMPT_NONE);
|
||||
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
|
||||
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
|
||||
adreno_dispatcher_schedule(device);
|
||||
/* Clear the keep alive */
|
||||
if (gmu_core_isenabled(device))
|
||||
|
|
|
|||
|
|
@ -63,6 +63,7 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev,
|
|||
static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
|
||||
struct adreno_ringbuffer *rb)
|
||||
{
|
||||
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
|
||||
unsigned long flags;
|
||||
int ret = 0;
|
||||
|
||||
|
|
@ -74,7 +75,7 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
|
|||
* Let the pwrscale policy know that new commands have
|
||||
* been submitted.
|
||||
*/
|
||||
kgsl_pwrscale_busy(KGSL_DEVICE(adreno_dev));
|
||||
kgsl_pwrscale_busy(device);
|
||||
|
||||
/*
|
||||
* Ensure the write posted after a possible
|
||||
|
|
@ -99,9 +100,14 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
|
|||
spin_unlock_irqrestore(&rb->preempt_lock, flags);
|
||||
|
||||
if (ret) {
|
||||
/* If WPTR update fails, set the fault and trigger recovery */
|
||||
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
|
||||
adreno_dispatcher_schedule(KGSL_DEVICE(adreno_dev));
|
||||
/*
|
||||
* If WPTR update fails, take inline snapshot and trigger
|
||||
* recovery.
|
||||
*/
|
||||
gmu_core_snapshot(device);
|
||||
adreno_set_gpu_fault(adreno_dev,
|
||||
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
|
||||
adreno_dispatcher_schedule(device);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1014,7 +1014,17 @@ static irqreturn_t gmu_irq_handler(int irq, void *data)
|
|||
ADRENO_REG_GMU_AO_HOST_INTERRUPT_MASK,
|
||||
(mask | GMU_INT_WDOG_BITE));
|
||||
|
||||
adreno_gmu_send_nmi(adreno_dev);
|
||||
/* make sure we're reading the latest cm3_fault */
|
||||
smp_rmb();
|
||||
|
||||
/*
|
||||
* We should not send NMI if there was a CM3 fault reported
|
||||
* because we don't want to overwrite the critical CM3 state
|
||||
* captured by gmu before it sent the CM3 fault interrupt.
|
||||
*/
|
||||
if (!atomic_read(&gmu->cm3_fault))
|
||||
adreno_gmu_send_nmi(adreno_dev);
|
||||
|
||||
/*
|
||||
* There is sufficient delay for the GMU to have finished
|
||||
* handling the NMI before snapshot is taken, as the fault
|
||||
|
|
@ -1023,8 +1033,6 @@ static irqreturn_t gmu_irq_handler(int irq, void *data)
|
|||
|
||||
dev_err_ratelimited(&gmu->pdev->dev,
|
||||
"GMU watchdog expired interrupt received\n");
|
||||
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
|
||||
adreno_dispatcher_schedule(device);
|
||||
}
|
||||
if (status & GMU_INT_HOST_AHB_BUS_ERR)
|
||||
dev_err_ratelimited(&gmu->pdev->dev,
|
||||
|
|
@ -1530,9 +1538,25 @@ static void gmu_snapshot(struct kgsl_device *device)
|
|||
struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device);
|
||||
struct gmu_device *gmu = KGSL_GMU_DEVICE(device);
|
||||
|
||||
adreno_gmu_send_nmi(adreno_dev);
|
||||
/* Wait for the NMI to be handled */
|
||||
udelay(100);
|
||||
/* Abstain from sending another nmi or over-writing snapshot */
|
||||
if (test_and_set_bit(GMU_FAULT, &device->gmu_core.flags))
|
||||
return;
|
||||
|
||||
/* make sure we're reading the latest cm3_fault */
|
||||
smp_rmb();
|
||||
|
||||
/*
|
||||
* We should not send NMI if there was a CM3 fault reported because we
|
||||
* don't want to overwrite the critical CM3 state captured by gmu before
|
||||
* it sent the CM3 fault interrupt.
|
||||
*/
|
||||
if (!atomic_read(&gmu->cm3_fault)) {
|
||||
adreno_gmu_send_nmi(adreno_dev);
|
||||
|
||||
/* Wait for the NMI to be handled */
|
||||
udelay(100);
|
||||
}
|
||||
|
||||
kgsl_device_snapshot(device, NULL, true);
|
||||
|
||||
adreno_write_gmureg(adreno_dev,
|
||||
|
|
@ -1670,6 +1694,12 @@ static void gmu_stop(struct kgsl_device *device)
|
|||
if (!test_bit(GMU_CLK_ON, &device->gmu_core.flags))
|
||||
return;
|
||||
|
||||
/* Force suspend if gmu is already in fault */
|
||||
if (test_bit(GMU_FAULT, &device->gmu_core.flags)) {
|
||||
gmu_core_suspend(device);
|
||||
return;
|
||||
}
|
||||
|
||||
/* Wait for the lowest idle level we requested */
|
||||
if (gmu_core_dev_wait_for_lowest_idle(device))
|
||||
goto error;
|
||||
|
|
@ -1695,14 +1725,13 @@ static void gmu_stop(struct kgsl_device *device)
|
|||
return;
|
||||
|
||||
error:
|
||||
/*
|
||||
* The power controller will change state to SLUMBER anyway
|
||||
* Set GMU_FAULT flag to indicate to power contrller
|
||||
* that hang recovery is needed to power on GPU
|
||||
*/
|
||||
set_bit(GMU_FAULT, &device->gmu_core.flags);
|
||||
dev_err(&gmu->pdev->dev, "Failed to stop GMU\n");
|
||||
gmu_core_snapshot(device);
|
||||
/*
|
||||
* We failed to stop the gmu successfully. Force a suspend
|
||||
* to set things up for a fresh start.
|
||||
*/
|
||||
gmu_core_suspend(device);
|
||||
}
|
||||
|
||||
static void gmu_remove(struct kgsl_device *device)
|
||||
|
|
|
|||
|
|
@ -210,6 +210,8 @@ struct gmu_device {
|
|||
unsigned long kmem_bitmap;
|
||||
const struct gmu_vma_entry *vma;
|
||||
unsigned int log_wptr_retention;
|
||||
/** @cm3_fault: whether gmu received a cm3 fault interrupt */
|
||||
atomic_t cm3_fault;
|
||||
};
|
||||
|
||||
struct gmu_memdesc *gmu_get_memdesc(struct gmu_device *gmu,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
// SPDX-License-Identifier: GPL-2.0-only
|
||||
/*
|
||||
* Copyright (c) 2018-2019, The Linux Foundation. All rights reserved.
|
||||
* Copyright (c) 2018-2020, The Linux Foundation. All rights reserved.
|
||||
*/
|
||||
|
||||
#include <linux/delay.h>
|
||||
|
|
@ -843,7 +843,6 @@ irqreturn_t hfi_irq_handler(int irq, void *data)
|
|||
struct kgsl_device *device = data;
|
||||
struct gmu_device *gmu = KGSL_GMU_DEVICE(device);
|
||||
struct kgsl_hfi *hfi = &gmu->hfi;
|
||||
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
|
||||
unsigned int status = 0;
|
||||
|
||||
adreno_read_gmureg(ADRENO_DEVICE(device),
|
||||
|
|
@ -856,8 +855,10 @@ irqreturn_t hfi_irq_handler(int irq, void *data)
|
|||
if (status & HFI_IRQ_CM3_FAULT_MASK) {
|
||||
dev_err_ratelimited(&gmu->pdev->dev,
|
||||
"GMU CM3 fault interrupt received\n");
|
||||
adreno_set_gpu_fault(adreno_dev, ADRENO_GMU_FAULT);
|
||||
adreno_dispatcher_schedule(device);
|
||||
atomic_set(&gmu->cm3_fault, 1);
|
||||
|
||||
/* make sure other CPUs see the update */
|
||||
smp_wmb();
|
||||
}
|
||||
if (status & ~HFI_IRQ_MASK)
|
||||
dev_err_ratelimited(&gmu->pdev->dev,
|
||||
|
|
|
|||
|
|
@ -1695,13 +1695,23 @@ static int _init(struct kgsl_device *device)
|
|||
int status = 0;
|
||||
|
||||
switch (device->state) {
|
||||
case KGSL_STATE_RESET:
|
||||
if (gmu_core_isenabled(device)) {
|
||||
/*
|
||||
* If we fail a INIT -> AWARE transition, we will
|
||||
* transition back to INIT. However, we must hard reset
|
||||
* the GMU as we go back to INIT. This is done by
|
||||
* forcing a RESET -> INIT transition.
|
||||
*/
|
||||
gmu_core_suspend(device);
|
||||
kgsl_pwrctrl_set_state(device, KGSL_STATE_INIT);
|
||||
}
|
||||
break;
|
||||
case KGSL_STATE_NAP:
|
||||
/* Force power on to do the stop */
|
||||
status = kgsl_pwrctrl_enable(device);
|
||||
/* fall through */
|
||||
case KGSL_STATE_ACTIVE:
|
||||
/* fall through */
|
||||
case KGSL_STATE_RESET:
|
||||
kgsl_pwrctrl_irq(device, KGSL_PWRFLAGS_OFF);
|
||||
del_timer_sync(&device->idle_timer);
|
||||
kgsl_pwrscale_midframe_timer_cancel(device);
|
||||
|
|
@ -1813,12 +1823,6 @@ _aware(struct kgsl_device *device)
|
|||
status = gmu_core_start(device);
|
||||
break;
|
||||
case KGSL_STATE_INIT:
|
||||
/* if GMU already in FAULT */
|
||||
if (gmu_core_isenabled(device) &&
|
||||
test_bit(GMU_FAULT, &device->gmu_core.flags)) {
|
||||
status = -EINVAL;
|
||||
break;
|
||||
}
|
||||
status = kgsl_pwrctrl_enable(device);
|
||||
break;
|
||||
/* The following 3 cases shouldn't occur, but don't panic. */
|
||||
|
|
@ -1832,20 +1836,23 @@ _aware(struct kgsl_device *device)
|
|||
break;
|
||||
case KGSL_STATE_SLUMBER:
|
||||
status = kgsl_pwrctrl_enable(device);
|
||||
if (status && gmu_core_isenabled(device))
|
||||
/*
|
||||
* SLUMBER -> AWARE failed which means GMU boot failed.
|
||||
* Make sure we reset the GMU while transitioning back
|
||||
* to SLUMBER.
|
||||
*/
|
||||
kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET);
|
||||
|
||||
break;
|
||||
default:
|
||||
status = -EINVAL;
|
||||
}
|
||||
|
||||
if (!status)
|
||||
if (status && gmu_core_isenabled(device))
|
||||
/*
|
||||
* If a SLUMBER/INIT -> AWARE fails, we transition back to
|
||||
* SLUMBER/INIT state. We must hard reset the GMU while
|
||||
* transitioning back to SLUMBER/INIT. A RESET -> AWARE
|
||||
* transition is different. It happens when dispatcher is
|
||||
* attempting reset/recovery as part of fault handling. If it
|
||||
* fails, we should still transition back to RESET in case
|
||||
* we want to attempt another reset/recovery.
|
||||
*/
|
||||
kgsl_pwrctrl_set_state(device, KGSL_STATE_RESET);
|
||||
else
|
||||
kgsl_pwrctrl_set_state(device, KGSL_STATE_AWARE);
|
||||
|
||||
return status;
|
||||
|
|
|
|||
Loading…
Reference in a new issue