msm: kgsl: Add hwsched based reset and recovery

There can be various gmu faults or a context_bad hfi alongside
an actively rendering gpu, for example, dcvs timeout, context
register or unregister timeout or a perfcounter oob timeout.
It is important to request dispatcher to perform reset and
recovery so that un-finished inflight submissions can be
re-submitted post recovery.

Change-Id: I2e66c5b24c45f1af7514044caa74abe1bf84dd0c
Signed-off-by: Harshdeep Dhatt <hdhatt@codeaurora.org>
This commit is contained in:
Harshdeep Dhatt 2020-07-27 10:46:00 -06:00
commit 44ed382d7f
10 changed files with 276 additions and 27 deletions

View file

@ -1658,12 +1658,8 @@ static inline int adreno_perfcntr_active_oob_get(
if (!ret) {
ret = gmu_core_dev_oob_set(device, oob_perfcntr);
if (ret) {
adreno_set_gpu_fault(adreno_dev,
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
if (ret)
adreno_active_count_put(adreno_dev);
}
}
return ret;
@ -1935,4 +1931,14 @@ int adreno_suspend_context(struct kgsl_device *device);
* submission.
*/
void adreno_profile_submit_time(struct adreno_submit_time *time);
/**
* adreno_mark_guilty_context - Mark the given context as guilty
* (failed recovery)
* @device: Pointer to a KGSL device structure
* @id: Context ID of the guilty context (or 0 to mark all as guilty)
*
* Mark the given (or all) context(s) as guilty (failed recovery)
*/
void adreno_mark_guilty_context(struct kgsl_device *device, unsigned int id);
#endif /*__ADRENO_H */

View file

@ -18,6 +18,7 @@
#include "adreno.h"
#include "adreno_a6xx.h"
#include "adreno_hwsched.h"
#include "kgsl_bus.h"
#include "kgsl_device.h"
#include "kgsl_trace.h"
@ -728,6 +729,29 @@ static const char *oob_to_str(enum oob_request req)
return "unknown";
}
static void trigger_reset_recovery(struct adreno_device *adreno_dev,
enum oob_request req)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
/*
* Trigger recovery for perfcounter oob only since only
* perfcounter oob can happen alongside an actively rendering gpu.
*/
if (req != oob_perfcntr)
return;
if (test_bit(GMU_DISPATCH, &device->gmu_core.flags)) {
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
} else {
adreno_set_gpu_fault(adreno_dev,
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
}
}
int a6xx_gmu_oob_set(struct kgsl_device *device,
enum oob_request req)
{
@ -762,6 +786,7 @@ int a6xx_gmu_oob_set(struct kgsl_device *device,
gmu_fault_snapshot(device);
ret = -ETIMEDOUT;
WARN(1, "OOB request %s timed out\n", oob_to_str(req));
trigger_reset_recovery(adreno_dev, req);
}
gmu_core_regwrite(device, A6XX_GMU_GMU2HOST_INTR_CLR, check);

View file

@ -510,10 +510,11 @@ struct hfi_context_rule_cmd {
/* F2H */
struct hfi_context_bad_cmd {
uint32_t hdr;
uint32_t ctxt_id;
uint32_t status;
uint32_t error;
u32 hdr;
u32 ctxt_id;
u32 policy;
u32 ts;
u32 error;
} __packed;
/* H2F */

View file

@ -632,11 +632,24 @@ static int a6xx_hwsched_dcvs_set(struct adreno_device *adreno_dev,
ret = a6xx_hfi_send_cmd_async(adreno_dev, &req);
if (ret)
if (ret) {
dev_err_ratelimited(&gmu->pdev->dev,
"Failed to set GPU perf idx %d, bw idx %d\n",
req.freq, req.bw);
/*
* If this was a dcvs request along side an active gpu, request
* dispatcher based reset and recovery.
*/
if (test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags)) {
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
}
}
return ret;
}
@ -753,6 +766,56 @@ static void a6xx_hwsched_pm_resume(struct adreno_device *adreno_dev)
clear_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags);
}
static void a6xx_hwsched_drain_ctxt_unregister(struct adreno_device *adreno_dev)
{
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct pending_cmd *cmd = NULL;
read_lock(&hfi->msglock);
list_for_each_entry(cmd, &hfi->msglist, node) {
if (MSG_HDR_GET_ID(cmd->sent_hdr) == H2F_MSG_UNREGISTER_CONTEXT)
complete(&cmd->complete);
}
read_unlock(&hfi->msglock);
}
void a6xx_hwsched_restart(struct adreno_device *adreno_dev)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
int ret;
/*
* Any pending context unregister packets will be lost
* since we hard reset the GMU. This means any threads waiting
* for context unregister hfi ack will timeout. Wake them
* to avoid false positive ack timeout messages later.
*/
a6xx_hwsched_drain_ctxt_unregister(adreno_dev);
read_lock(&device->context_lock);
idr_for_each(&device->context_idr, unregister_context_hwsched, NULL);
read_unlock(&device->context_lock);
if (!test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags))
return;
a6xx_hwsched_hfi_stop(adreno_dev);
a6xx_disable_gpu_irq(adreno_dev);
a6xx_gmu_suspend(adreno_dev);
clear_bit(GMU_PRIV_GPU_STARTED, &gmu->flags);
ret = a6xx_hwsched_boot(adreno_dev);
BUG_ON(ret);
}
const struct adreno_power_ops a6xx_hwsched_power_ops = {
.first_open = a6xx_hwsched_first_open,
.last_close = a6xx_hwsched_power_off,

View file

@ -33,4 +33,10 @@ struct a6xx_hwsched_device {
*/
int a6xx_hwsched_probe(struct platform_device *pdev,
u32 chipid, const struct adreno_gpu_core *gpucore);
/**
* a6xx_hwsched_restart - Restart the gmu and gpu
* @adreno_dev: Pointer to the adreno device
*/
void a6xx_hwsched_restart(struct adreno_device *adreno_dev);
#endif

View file

@ -9,6 +9,7 @@
#include "adreno.h"
#include "adreno_a6xx.h"
#include "adreno_a6xx_hwsched.h"
#include "adreno_hwsched.h"
#include "adreno_pm4types.h"
#include "adreno_trace.h"
#include "kgsl_device.h"
@ -69,7 +70,7 @@ static const char * const memkind_strings[] = {
[MEMKIND_USER_PROFILE_IBS] = "GMU USER PROFILING",
};
static struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(
struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(
struct adreno_device *adreno_dev)
{
struct a6xx_device *a6xx_dev = container_of(adreno_dev,
@ -829,6 +830,14 @@ static void process_ts_retire(struct adreno_device *adreno_dev, u32 *rcvd)
adreno_hwsched_trigger(adreno_dev);
}
static void process_ctx_bad(struct adreno_device *adreno_dev, void *rcvd)
{
/* Block dispatcher to submit more commands */
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
}
static int hfi_f2h_main(void *arg)
{
struct adreno_device *adreno_dev = arg;
@ -852,6 +861,9 @@ static int hfi_f2h_main(void *arg)
if (MSG_HDR_GET_ID(pkt->rcvd[0]) == F2H_MSG_TS_RETIRE)
process_ts_retire(adreno_dev, pkt->rcvd);
if (MSG_HDR_GET_ID(pkt->rcvd[0]) == F2H_MSG_CONTEXT_BAD)
process_ctx_bad(adreno_dev, pkt->rcvd);
kmem_cache_free(f2h_cache, pkt);
}
}
@ -1149,7 +1161,7 @@ free:
}
static int send_context_unregister_hfi(struct adreno_device *adreno_dev,
u32 ctxt_id, u32 ts)
struct kgsl_context *context, u32 ts)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
@ -1160,7 +1172,7 @@ static int send_context_unregister_hfi(struct adreno_device *adreno_dev,
int rc;
CMD_MSG_HDR(cmd, H2F_MSG_UNREGISTER_CONTEXT);
cmd.ctxt_id = ctxt_id,
cmd.ctxt_id = context->id,
cmd.ts = ts,
seqnum = atomic_inc_return(&gmu->hfi.seqnum);
@ -1180,10 +1192,26 @@ static int send_context_unregister_hfi(struct adreno_device *adreno_dev,
dev_err(&gmu->pdev->dev,
"Ack timeout for context unregister seq: %d ctx: %d ts: %d\n",
MSG_HDR_GET_SEQNUM(pending_ack.sent_hdr),
ctxt_id, ts);
context->id, ts);
rc = -ETIMEDOUT;
mutex_lock(&device->mutex);
gmu_fault_snapshot(device);
/*
* Trigger dispatcher based reset and recovery. Invalidate the
* context so that any un-finished inflight submissions are not
* replayed after recovery.
*/
adreno_mark_guilty_context(device, context->id);
adreno_drawctxt_invalidate(device, context);
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
goto done;
}
@ -1209,7 +1237,7 @@ void a6xx_hwsched_context_detach(struct adreno_context *drawctxt)
/* Only send HFI if device is not in SLUMBER */
if (context->gmu_registered &&
test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags))
ret = send_context_unregister_hfi(adreno_dev, context->id,
ret = send_context_unregister_hfi(adreno_dev, context,
drawctxt->internal_timestamp);
if (!ret) {

View file

@ -204,4 +204,7 @@ int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev,
* to ensure all submissions from this context have retired
*/
void a6xx_hwsched_context_detach(struct adreno_context *drawctxt);
/* Helper function to get to a6xx hwsched hfi device from adreno device */
struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(struct adreno_device *adreno_dev);
#endif

View file

@ -1524,14 +1524,7 @@ static int _mark_context(int id, void *ptr, void *data)
return 0;
}
/**
* mark_guilty_context() - Mark the given context as guilty (failed recovery)
* @device: Pointer to a KGSL device structure
* @id: Context ID of the guilty context (or 0 to mark all as guilty)
*
* Mark the given (or all) context(s) as guilty (failed recovery)
*/
static void mark_guilty_context(struct kgsl_device *device, unsigned int id)
void adreno_mark_guilty_context(struct kgsl_device *device, unsigned int id)
{
/* Mark the status for all the contexts in the device */
@ -1953,7 +1946,7 @@ static void process_cmdobj_fault(struct kgsl_device *device,
state, drawobj->context->id, drawobj->timestamp);
/* Mark the context as failed */
mark_guilty_context(device, drawobj->context->id);
adreno_mark_guilty_context(device, drawobj->context->id);
/* Invalidate the context */
adreno_drawctxt_invalidate(device, drawobj->context);
@ -1991,7 +1984,8 @@ static void recover_dispatch_q(struct kgsl_device *device,
dispatch_q->cmd_q[ptr];
struct kgsl_drawobj *drawobj = DRAWOBJ(cmdobj);
mark_guilty_context(device, drawobj->context->id);
adreno_mark_guilty_context(device,
drawobj->context->id);
adreno_drawctxt_invalidate(device, drawobj->context);
kgsl_drawobj_destroy(drawobj);
@ -2064,7 +2058,7 @@ replay:
replay[i]->base.timestamp);
/* Mark this context as guilty (failed recovery) */
mark_guilty_context(device,
adreno_mark_guilty_context(device,
replay[i]->base.context->id);
adreno_drawctxt_invalidate(device,

View file

@ -272,6 +272,25 @@ static int hwsched_queue_context(struct adreno_device *adreno_dev,
return 0;
}
void adreno_hwsched_set_fault(struct adreno_device *adreno_dev)
{
struct adreno_hwsched *hwsched = to_hwsched(adreno_dev);
atomic_set(&hwsched->fault, ADRENO_HWSCHED_FAULT_RESTART);
/* make sure fault is written before triggering dispatcher */
smp_wmb();
adreno_hwsched_trigger(adreno_dev);
}
static bool hwsched_in_fault(struct adreno_hwsched *hwsched)
{
/* make sure we're reading the latest value */
smp_rmb();
return atomic_read(&hwsched->fault) != 0;
}
/**
* sendcmd() - Send a drawobj to the GPU hardware
* @dispatcher: Pointer to the adreno dispatcher struct
@ -332,6 +351,14 @@ static int hwsched_sendcmd(struct adreno_device *adreno_dev,
if (hwsched->inflight == 1) {
adreno_active_count_put(adreno_dev);
clear_bit(ADRENO_HWSCHED_POWER, &hwsched->flags);
} else if (device->gmu_fault) {
/*
* If we encountered a GMU fault, then trigger reset and
* recovery.
*/
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
}
hwsched->inflight--;
@ -556,7 +583,8 @@ static void adreno_hwsched_issuecmds(struct adreno_device *adreno_dev)
return;
}
hwsched_issuecmds(adreno_dev);
if (!hwsched_in_fault(hwsched))
hwsched_issuecmds(adreno_dev);
mutex_unlock(&hwsched->mutex);
}
@ -966,6 +994,85 @@ void adreno_hwsched_dispatcher_close(struct adreno_device *adreno_dev)
kmem_cache_destroy(obj_cache);
}
static void adreno_hwsched_init_replay(struct adreno_hwsched *hwsched)
{
struct cmd_list_obj *obj, *tmp;
list_for_each_entry_safe(obj, tmp, &hwsched->cmd_list, node) {
struct kgsl_drawobj_cmd *cmdobj = obj->cmdobj;
clear_bit(KGSL_FT_REPLAY, &cmdobj->fault_policy);
}
}
static void adreno_hwsched_complete_replay(struct adreno_device *adreno_dev)
{
struct adreno_hwsched *hwsched = to_hwsched(adreno_dev);
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct cmd_list_obj *obj, *tmp;
u32 retired = 0;
list_for_each_entry_safe(obj, tmp, &hwsched->cmd_list, node) {
struct kgsl_drawobj_cmd *cmdobj = obj->cmdobj;
struct kgsl_drawobj *drawobj = DRAWOBJ(cmdobj);
struct kgsl_context *context = drawobj->context;
/*
* Get rid of retired objects or objects that belong to detached
* or invalidated contexts
*/
if ((kgsl_check_timestamp(device, context, drawobj->timestamp))
|| kgsl_context_invalid(context)
|| kgsl_context_detached(context)) {
retire_cmdobj(cmdobj);
retired++;
list_del_init(&obj->node);
kmem_cache_free(obj_cache, obj);
continue;
}
if (!test_and_set_bit(KGSL_FT_REPLAY, &cmdobj->fault_policy))
a6xx_hwsched_submit_cmdobj(adreno_dev, cmdobj);
}
/* Signal fences */
if (retired)
kgsl_process_event_groups(device);
hwsched->inflight = 0;
/*
* We have nothing to replay. So clear the fault so that
* dispatcher starts accepting new submissions.
*/
if (list_empty(&hwsched->cmd_list)) {
atomic_set(&hwsched->fault, 0);
adreno_clear_gpu_halt(adreno_dev);
adreno_hwsched_trigger(adreno_dev);
}
}
static void adreno_hwsched_recovery(struct adreno_device *adreno_dev)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct adreno_hwsched *hwsched = to_hwsched(adreno_dev);
mutex_lock(&device->mutex);
if ((atomic_cmpxchg(&hwsched->fault, ADRENO_HWSCHED_FAULT_RESTART,
ADRENO_HWSCHED_FAULT_REPLAY) == ADRENO_HWSCHED_FAULT_RESTART)) {
a6xx_hwsched_restart(adreno_dev);
adreno_hwsched_init_replay(hwsched);
}
adreno_hwsched_complete_replay(adreno_dev);
mutex_unlock(&device->mutex);
}
static void adreno_hwsched_work(struct kthread_work *work)
{
struct adreno_hwsched *hwsched = container_of(work,
@ -976,6 +1083,12 @@ static void adreno_hwsched_work(struct kthread_work *work)
mutex_lock(&hwsched->mutex);
if (hwsched_in_fault(hwsched)) {
adreno_hwsched_recovery(adreno_dev);
mutex_unlock(&hwsched->mutex);
return;
}
/*
* As long as there are inflight commands, process retired comamnds from
* all drawqueues

View file

@ -24,10 +24,14 @@ struct adreno_hwsched {
struct kthread_work work;
/** @cmd_list: List of objects submitted to dispatch queues */
struct list_head cmd_list;
/** @fault: Atomic to record a fault */
atomic_t fault;
};
enum adreno_hwsched_flags {
ADRENO_HWSCHED_POWER = 0,
ADRENO_HWSCHED_FAULT_RESTART,
ADRENO_HWSCHED_FAULT_REPLAY,
};
/**
@ -87,4 +91,10 @@ void adreno_hwsched_init(struct adreno_device *adreno_dev);
* Free the dispatcher resources
*/
void adreno_hwsched_dispatcher_close(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_set_fault - Set hwsched fault to request recovery
* @adreno_dev: A handle to adreno device
*/
void adreno_hwsched_set_fault(struct adreno_device *adreno_dev);
#endif