Merge "msm: kgsl: Capture gpu globals in hwsched snapshot"

This commit is contained in:
qctecmdr 2020-09-03 22:49:08 -07:00 • committed by Gerrit - the friendly Code Review server
commit 4b9bef81d3
30 changed files with 2689 additions and 451 deletions

View file

@ -44,6 +44,7 @@ msm_kgsl-y += \
adreno_cp_parser.o \
adreno_dispatch.o \
adreno_drawctxt.o \
adreno_hwsched.o \
adreno_ioctl.o \
adreno_perfcounter.o \
adreno_ringbuffer.o \

View file

@ -22,6 +22,7 @@
#include "adreno_a5xx.h"
#include "adreno_a6xx.h"
#include "adreno_compat.h"
#include "adreno_hwsched.h"
#include "adreno_iommu.h"
#include "adreno_trace.h"
#include "kgsl_bus.h"
@ -1522,10 +1523,15 @@ static void adreno_unbind(struct device *dev)
kgsl_pwrscale_close(device);
adreno_dispatcher_close(adreno_dev);
adreno_ringbuffer_close(adreno_dev);
if (test_bit(GMU_DISPATCH, &device->gmu_core.flags))
adreno_hwsched_dispatcher_close(adreno_dev);
else {
adreno_dispatcher_close(adreno_dev);
adreno_fault_detect_stop(adreno_dev);
adreno_ringbuffer_close(adreno_dev);
adreno_fault_detect_stop(adreno_dev);
}
kfree(adreno_ft_regs);
adreno_ft_regs = NULL;
@ -1875,11 +1881,11 @@ static int adreno_last_close(struct kgsl_device *device)
* Wait up to 1 second for the active count to go low
* and then start complaining about it
*/
if (kgsl_active_count_wait(device, 0)) {
if (kgsl_active_count_wait(device, 0, HZ)) {
dev_err(device->dev,
"Waiting for the active count to become 0\n");
while (kgsl_active_count_wait(device, 0))
while (kgsl_active_count_wait(device, 0, HZ))
dev_err(device->dev,
"Still waiting for the active count\n");
}
@ -3195,6 +3201,45 @@ void adreno_cx_misc_regrmw(struct adreno_device *adreno_dev,
adreno_cx_misc_regwrite(adreno_dev, offsetwords, val | bits);
}
void adreno_profile_submit_time(struct adreno_submit_time *time)
{
struct kgsl_drawobj *drawobj;
struct kgsl_drawobj_cmd *cmdobj;
struct kgsl_mem_entry *entry;
struct kgsl_drawobj_profiling_buffer *profile_buffer;
drawobj = time->drawobj;
if (drawobj == NULL)
return;
cmdobj = CMDOBJ(drawobj);
entry = cmdobj->profiling_buf_entry;
if (!entry)
return;
profile_buffer = kgsl_gpuaddr_to_vaddr(&entry->memdesc,
cmdobj->profiling_buffer_gpuaddr);
if (profile_buffer == NULL)
return;
/* Return kernel clock time to the client if requested */
if (drawobj->flags & KGSL_DRAWOBJ_PROFILING_KTIME) {
u64 secs = time->ktime;
profile_buffer->wall_clock_ns =
do_div(secs, NSEC_PER_SEC);
profile_buffer->wall_clock_s = secs;
} else {
profile_buffer->wall_clock_s = time->utime.tv_sec;
profile_buffer->wall_clock_ns = time->utime.tv_nsec;
}
profile_buffer->gpu_ticks_queued = time->ticks;
kgsl_memdesc_unmap(&entry->memdesc);
}
/**
* adreno_waittimestamp - sleep while waiting for the specified timestamp
* @device - pointer to a KGSL device structure
@ -3607,6 +3652,12 @@ static int adreno_queue_cmds(struct kgsl_device_private *dev_priv,
struct kgsl_context *context, struct kgsl_drawobj *drawobj[],
u32 count, u32 *timestamp)
{
struct kgsl_device *device = dev_priv->device;
if (test_bit(GMU_DISPATCH, &device->gmu_core.flags))
return adreno_hwsched_queue_cmds(dev_priv, context, drawobj,
count, timestamp);
return adreno_dispatcher_queue_cmds(dev_priv, context, drawobj, count,
timestamp);
}
@ -3614,6 +3665,10 @@ static int adreno_queue_cmds(struct kgsl_device_private *dev_priv,
static void adreno_drawctxt_sched(struct kgsl_device *device,
struct kgsl_context *context)
{
if (test_bit(GMU_DISPATCH, &device->gmu_core.flags))
return adreno_hwsched_queue_context(device,
ADRENO_CONTEXT(context));
adreno_dispatcher_queue_context(device, ADRENO_CONTEXT(context));
}

View file

@ -806,7 +806,7 @@ struct adreno_gpudev {
int (*preemption_init)(struct adreno_device *adreno_dev);
void (*preemption_schedule)(struct adreno_device *adreno_dev);
int (*preemption_context_init)(struct kgsl_context *context);
void (*preemption_context_destroy)(struct kgsl_context *context);
void (*context_detach)(struct adreno_context *drawctxt);
void (*clk_set_options)(struct adreno_device *adreno_dev,
const char *name, struct clk *clk, bool on);
void (*pre_reset)(struct adreno_device *adreno_dev);
@ -1658,12 +1658,8 @@ static inline int adreno_perfcntr_active_oob_get(
if (!ret) {
ret = gmu_core_dev_oob_set(device, oob_perfcntr);
if (ret) {
adreno_set_gpu_fault(adreno_dev,
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
if (ret)
adreno_active_count_put(adreno_dev);
}
}
return ret;
@ -1777,6 +1773,13 @@ static inline void adreno_reg_offset_init(u32 *reg_offsets)
}
}
static inline u32 adreno_get_level(u32 priority)
{
u32 level = priority / KGSL_PRIORITY_MAX_RB_LEVELS;
return min_t(u32, level, KGSL_PRIORITY_MAX_RB_LEVELS - 1);
}
int adreno_gmu_fenced_write(struct adreno_device *adreno_dev,
enum adreno_regs offset, unsigned int val,
unsigned int fence_mask);
@ -1925,4 +1928,24 @@ void gmu_fault_snapshot(struct kgsl_device *device);
* Return: 0 on success or negative error on failure
*/
int adreno_suspend_context(struct kgsl_device *device);
/*
* adreno_profile_submit_time - Populate profiling buffer with timestamps
* @time: Container for the statistics
*
* Populate the draw object user profiling buffer with the timestamps
* recored in the adreno_submit_time structure at the time of draw object
* submission.
*/
void adreno_profile_submit_time(struct adreno_submit_time *time);
/**
* adreno_mark_guilty_context - Mark the given context as guilty
* (failed recovery)
* @device: Pointer to a KGSL device structure
* @id: Context ID of the guilty context (or 0 to mark all as guilty)
*
* Mark the given (or all) context(s) as guilty (failed recovery)
*/
void adreno_mark_guilty_context(struct kgsl_device *device, unsigned int id);
#endif /*__ADRENO_H */

View file

@ -17,22 +17,6 @@
#include "adreno_trace.h"
#include "kgsl_trace.h"
#define A6XX_INT_MASK \
((1 << A6XX_INT_CP_AHB_ERROR) | \
(1 << A6XX_INT_ATB_ASYNCFIFO_OVERFLOW) | \
(1 << A6XX_INT_RBBM_GPC_ERROR) | \
(1 << A6XX_INT_CP_SW) | \
(1 << A6XX_INT_CP_HW_ERROR) | \
(1 << A6XX_INT_CP_IB2) | \
(1 << A6XX_INT_CP_IB1) | \
(1 << A6XX_INT_CP_RB) | \
(1 << A6XX_INT_CP_CACHE_FLUSH_TS) | \
(1 << A6XX_INT_RBBM_ATB_BUS_OVERFLOW) | \
(1 << A6XX_INT_RBBM_HANG_DETECT) | \
(1 << A6XX_INT_UCHE_OOB_ACCESS) | \
(1 << A6XX_INT_UCHE_TRAP_INTR) | \
(1 << A6XX_INT_TSB_WRITE_ERROR))
/* IFPC & Preemption static powerup restore list */
static u32 a6xx_pwrup_reglist[] = {
A6XX_VSC_ADDR_MODE_CNTL,
@ -439,8 +423,6 @@ void a6xx_start(struct adreno_device *adreno_dev)
unsigned int rgb565_predicator = 0;
static bool patch_reglist;
adreno_dev->irq_mask = A6XX_INT_MASK;
/* enable hardware clockgating */
a6xx_hwcg_set(adreno_dev, true);
@ -2555,6 +2537,8 @@ static int a6xx_probe(struct platform_device *pdev,
INIT_WORK(&device->idle_check_ws, kgsl_idle_check);
adreno_dev->irq_mask = A6XX_INT_MASK;
return 0;
}
@ -2824,7 +2808,6 @@ struct adreno_gpudev adreno_a6xx_gpudev = {
.preemption_schedule = a6xx_preemption_schedule,
.set_marker = a6xx_set_marker,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.sptprac_is_on = a6xx_sptprac_is_on,
.ccu_invalidate = a6xx_ccu_invalidate,
.perfcounter_update = a6xx_perfcounter_update,
@ -2841,13 +2824,13 @@ struct adreno_gpudev adreno_a6xx_gpudev = {
struct adreno_gpudev adreno_a6xx_hwsched_gpudev = {
.reg_offsets = a6xx_register_offsets,
.probe = a6xx_hwsched_probe,
.snapshot = a6xx_gmu_snapshot,
.snapshot = a6xx_hwsched_snapshot,
.irq_handler = a6xx_irq_handler,
.perfcounters = &a6xx_perfcounters,
.read_throttling_counters = a6xx_read_throttling_counters,
.iommu_fault_block = a6xx_iommu_fault_block,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.context_detach = a6xx_hwsched_context_detach,
.perfcounter_update = a6xx_perfcounter_update,
#ifdef CONFIG_QCOM_KGSL_CORESIGHT
.coresight = {&a6xx_coresight, &a6xx_coresight_cx},
@ -2879,7 +2862,6 @@ struct adreno_gpudev adreno_a6xx_gmu_gpudev = {
.preemption_schedule = a6xx_preemption_schedule,
.set_marker = a6xx_set_marker,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.sptprac_is_on = a6xx_sptprac_is_on,
.ccu_invalidate = a6xx_ccu_invalidate,
.perfcounter_update = a6xx_perfcounter_update,
@ -2913,7 +2895,6 @@ struct adreno_gpudev adreno_a6xx_rgmu_gpudev = {
.preemption_schedule = a6xx_preemption_schedule,
.set_marker = a6xx_set_marker,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.sptprac_is_on = a6xx_sptprac_is_on,
.ccu_invalidate = a6xx_ccu_invalidate,
.perfcounter_update = a6xx_perfcounter_update,
@ -2947,7 +2928,6 @@ struct adreno_gpudev adreno_a619_holi_gpudev = {
.preemption_schedule = a6xx_preemption_schedule,
.set_marker = a6xx_set_marker,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.sptprac_is_on = a6xx_sptprac_is_on,
.ccu_invalidate = a6xx_ccu_invalidate,
.perfcounter_update = a6xx_perfcounter_update,
@ -2984,7 +2964,6 @@ struct adreno_gpudev adreno_a630_gpudev = {
.preemption_schedule = a6xx_preemption_schedule,
.set_marker = a6xx_set_marker,
.preemption_context_init = a6xx_preemption_context_init,
.preemption_context_destroy = a6xx_preemption_context_destroy,
.sptprac_is_on = a6xx_sptprac_is_on,
.ccu_invalidate = a6xx_ccu_invalidate,
.perfcounter_update = a6xx_perfcounter_update,

View file

@ -13,6 +13,9 @@
#include "adreno_a6xx_gmu.h"
#include "adreno_a6xx_rgmu.h"
/* Snapshot section size of each CP preemption record for A6XX */
#define A6XX_SNAPSHOT_CP_CTXRECORD_SIZE_IN_BYTES (64 * 1024)
extern const struct adreno_power_ops a6xx_gmu_power_ops;
extern const struct adreno_power_ops a6xx_rgmu_power_ops;
extern const struct adreno_power_ops a630_gmu_power_ops;
@ -186,6 +189,31 @@ struct cpu_gpu_lock {
/* Size of the CP_INIT pm4 stream in dwords */
#define A6XX_CP_INIT_DWORDS 12
#define A6XX_INT_MASK \
((1 << A6XX_INT_CP_AHB_ERROR) | \
(1 << A6XX_INT_ATB_ASYNCFIFO_OVERFLOW) | \
(1 << A6XX_INT_RBBM_GPC_ERROR) | \
(1 << A6XX_INT_CP_SW) | \
(1 << A6XX_INT_CP_HW_ERROR) | \
(1 << A6XX_INT_CP_IB2) | \
(1 << A6XX_INT_CP_IB1) | \
(1 << A6XX_INT_CP_RB) | \
(1 << A6XX_INT_CP_CACHE_FLUSH_TS) | \
(1 << A6XX_INT_RBBM_ATB_BUS_OVERFLOW) | \
(1 << A6XX_INT_RBBM_HANG_DETECT) | \
(1 << A6XX_INT_UCHE_OOB_ACCESS) | \
(1 << A6XX_INT_UCHE_TRAP_INTR) | \
(1 << A6XX_INT_TSB_WRITE_ERROR))
#define A6XX_HWSCHED_INT_MASK \
((1 << A6XX_INT_CP_AHB_ERROR) | \
(1 << A6XX_INT_ATB_ASYNCFIFO_OVERFLOW) | \
(1 << A6XX_INT_RBBM_GPC_ERROR) | \
(1 << A6XX_INT_RBBM_ATB_BUS_OVERFLOW) | \
(1 << A6XX_INT_UCHE_OOB_ACCESS) | \
(1 << A6XX_INT_UCHE_TRAP_INTR) | \
(1 << A6XX_INT_TSB_WRITE_ERROR))
/**
* to_a6xx_core - return the a6xx specific GPU core struct
* @adreno_dev: An Adreno GPU device handle

View file

@ -18,6 +18,7 @@
#include "adreno.h"
#include "adreno_a6xx.h"
#include "adreno_hwsched.h"
#include "kgsl_bus.h"
#include "kgsl_device.h"
#include "kgsl_trace.h"
@ -728,6 +729,29 @@ static const char *oob_to_str(enum oob_request req)
return "unknown";
}
static void trigger_reset_recovery(struct adreno_device *adreno_dev,
enum oob_request req)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
/*
* Trigger recovery for perfcounter oob only since only
* perfcounter oob can happen alongside an actively rendering gpu.
*/
if (req != oob_perfcntr)
return;
if (test_bit(GMU_DISPATCH, &device->gmu_core.flags)) {
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
} else {
adreno_set_gpu_fault(adreno_dev,
ADRENO_GMU_FAULT_SKIP_SNAPSHOT);
adreno_dispatcher_schedule(device);
}
}
int a6xx_gmu_oob_set(struct kgsl_device *device,
enum oob_request req)
{
@ -762,6 +786,7 @@ int a6xx_gmu_oob_set(struct kgsl_device *device,
gmu_fault_snapshot(device);
ret = -ETIMEDOUT;
WARN(1, "OOB request %s timed out\n", oob_to_str(req));
trigger_reset_recovery(adreno_dev, req);
}
gmu_core_regwrite(device, A6XX_GMU_GMU2HOST_INTR_CLR, check);
@ -841,9 +866,11 @@ static int a6xx_gmu_hfi_start_msg(struct adreno_device *adreno_dev)
* serves as a better means to identify targets that depend on
* legacy firmware.
*/
if (!ADRENO_QUIRK(adreno_dev, ADRENO_QUIRK_HFI_USE_REG))
return a6xx_hfi_send_req(adreno_dev,
H2F_MSG_START, &req);
if (!ADRENO_QUIRK(adreno_dev, ADRENO_QUIRK_HFI_USE_REG)) {
CMD_MSG_HDR(req, H2F_MSG_START);
return a6xx_hfi_send_generic_req(adreno_dev, &req);
}
return 0;
@ -1714,8 +1741,9 @@ static int a6xx_gmu_notify_slumber(struct adreno_device *adreno_dev)
.bw = bus_level,
};
ret = a6xx_hfi_send_req(adreno_dev,
H2F_MSG_PREPARE_SLUMBER, &req);
CMD_MSG_HDR(req, H2F_MSG_PREPARE_SLUMBER);
ret = a6xx_hfi_send_generic_req(adreno_dev, &req);
goto out;
}
@ -1803,11 +1831,12 @@ static int a6xx_gmu_dcvs_set(struct adreno_device *adreno_dev,
return 0;
}
CMD_MSG_HDR(req, H2F_MSG_GX_BW_PERF_VOTE);
if (ADRENO_QUIRK(adreno_dev, ADRENO_QUIRK_HFI_USE_REG))
ret = a6xx_gmu_dcvs_nohfi(device, req.freq, req.bw);
else
ret = a6xx_hfi_send_req(adreno_dev, H2F_MSG_GX_BW_PERF_VOTE,
&req);
ret = a6xx_hfi_send_generic_req(adreno_dev, &req);
if (ret) {
dev_err_ratelimited(&gmu->pdev->dev,
@ -2442,7 +2471,8 @@ static void a6xx_gmu_acd_probe(struct kgsl_device *device,
if (!ADRENO_FEATURE(adreno_dev, ADRENO_ACD))
return;
cmd->hdr = CMD_MSG_HDR(H2F_MSG_ACD_TBL, sizeof(*cmd));
cmd->hdr = CREATE_MSG_HDR(H2F_MSG_ACD_TBL, sizeof(*cmd), HFI_MSG_CMD);
cmd->version = 1;
cmd->stride = 1;
cmd->enable_by_level = 0;
@ -3199,7 +3229,7 @@ static int a6xx_gmu_pm_suspend(struct adreno_device *adreno_dev)
reinit_completion(&device->halt_gate);
/* wait for active count so device can be put in slumber */
ret = kgsl_active_count_wait(device, 0);
ret = kgsl_active_count_wait(device, 0, HZ);
if (ret) {
dev_err(device->dev,
"Timed out waiting for the active count\n");
@ -3230,10 +3260,9 @@ static void a6xx_gmu_pm_resume(struct adreno_device *adreno_dev)
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
if (!test_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags)) {
dev_err(device->dev, "resume invoked without a suspend\n");
if (WARN(!test_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags),
"resume invoked without a suspend\n"))
return;
}
adreno_dispatcher_unhalt(device);
@ -3336,6 +3365,8 @@ int a6xx_gmu_device_probe(struct platform_device *pdev,
timer_setup(&device->idle_timer, gmu_idle_timer, 0);
adreno_dev->irq_mask = A6XX_INT_MASK;
return 0;
}

View file

@ -103,13 +103,6 @@ int a6xx_hfi_queue_write(struct adreno_device *adreno_dev, uint32_t queue_idx,
if (hdr->status == HFI_QUEUE_STATUS_DISABLED)
return -EINVAL;
if (size > HFI_MAX_MSG_SIZE) {
dev_err(&gmu->pdev->dev,
"Message too big to send: sz=%d, id=%d\n",
size, id);
return -EINVAL;
}
queue = HOST_QUEUE_START_ADDR(gmu->hfi.hfi_mem, queue_idx);
trace_kgsl_hfi_send(id, size, MSG_HDR_GET_SEQNUM(*msg));
@ -359,13 +352,14 @@ static int a6xx_hfi_send_gmu_init(struct adreno_device *adreno_dev)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_gmu_init_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_INIT, sizeof(cmd)),
.seg_id = 0,
.dbg_buffer_addr = (unsigned int) gmu->dump_mem->gmuaddr,
.dbg_buffer_size = (unsigned int) gmu->dump_mem->size,
.boot_state = 0x1,
};
CMD_MSG_HDR(cmd, H2F_MSG_INIT);
return a6xx_hfi_send_generic_req(adreno_dev, &cmd);
}
@ -374,12 +368,13 @@ static int a6xx_hfi_get_fw_version(struct adreno_device *adreno_dev,
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_fw_version_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_FW_VER, sizeof(cmd)),
.supported_ver = expected_ver,
};
int rc;
struct pending_cmd ret_cmd;
CMD_MSG_HDR(cmd, H2F_MSG_FW_VER);
memset(&ret_cmd, 0, sizeof(ret_cmd));
rc = a6xx_hfi_send_cmd_wait_inline(adreno_dev, &cmd, &ret_cmd);
@ -399,10 +394,11 @@ static int a6xx_hfi_get_fw_version(struct adreno_device *adreno_dev,
int a6xx_hfi_send_core_fw_start(struct adreno_device *adreno_dev)
{
struct hfi_core_fw_start_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_CORE_FW_START, sizeof(cmd)),
.handle = 0x0,
};
CMD_MSG_HDR(cmd, H2F_MSG_CORE_FW_START);
return a6xx_hfi_send_generic_req(adreno_dev, &cmd);
}
@ -425,13 +421,14 @@ int a6xx_hfi_send_feature_ctrl(struct adreno_device *adreno_dev,
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_feature_ctrl_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_FEATURE_CTRL, sizeof(cmd)),
.feature = feature,
.enable = enable,
.data = data,
};
int ret;
CMD_MSG_HDR(cmd, H2F_MSG_FEATURE_CTRL);
ret = a6xx_hfi_send_generic_req(adreno_dev, &cmd);
if (ret)
dev_err(&gmu->pdev->dev,
@ -447,12 +444,13 @@ static int a6xx_hfi_send_dcvstbl_v1(struct adreno_device *adreno_dev)
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_dcvstable_cmd *table = &gmu->hfi.dcvs_table;
struct hfi_dcvstable_v1_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_PERF_TBL, sizeof(cmd)),
.gpu_level_num = table->gpu_level_num,
.gmu_level_num = table->gmu_level_num,
};
int i;
CMD_MSG_HDR(cmd, H2F_MSG_PERF_TBL);
for (i = 0; i < table->gpu_level_num; i++) {
cmd.gx_votes[i].vote = table->gx_votes[i].vote;
cmd.gx_votes[i].freq = table->gx_votes[i].freq;
@ -466,32 +464,11 @@ static int a6xx_hfi_send_dcvstbl_v1(struct adreno_device *adreno_dev)
return a6xx_hfi_send_generic_req(adreno_dev, &cmd);
}
static int a6xx_hfi_send_get_value(struct adreno_device *adreno_dev,
struct hfi_get_value_req *req)
{
struct hfi_get_value_cmd *cmd = &req->cmd;
struct pending_cmd ret_cmd;
struct hfi_get_value_reply_cmd *reply =
(struct hfi_get_value_reply_cmd *)ret_cmd.results;
int rc;
cmd->hdr = CMD_MSG_HDR(H2F_MSG_GET_VALUE, sizeof(*cmd));
rc = a6xx_hfi_send_cmd_wait_inline(adreno_dev, cmd, &ret_cmd);
if (rc)
return rc;
memset(&req->data, 0, sizeof(req->data));
memcpy(&req->data, &reply->data,
(MSG_HDR_GET_SIZE(reply->hdr) - 2) << 2);
return 0;
}
static int a6xx_hfi_send_test(struct adreno_device *adreno_dev)
{
struct hfi_test_cmd cmd = {
.hdr = CMD_MSG_HDR(H2F_MSG_TEST, sizeof(cmd)),
};
struct hfi_test_cmd cmd;
CMD_MSG_HDR(cmd, H2F_MSG_TEST);
return a6xx_hfi_send_generic_req(adreno_dev, &cmd);
}
@ -644,6 +621,8 @@ int a6xx_hfi_send_lm_feature_ctrl(struct adreno_device *adreno_dev)
nvmem_cell_read_u32(&device->pdev->dev, "isense_slope", &slope);
CMD_MSG_HDR(req, H2F_MSG_SET_VALUE);
req.type = HFI_VALUE_LM_CS0;
req.subtype = 0;
req.data = slope;
@ -652,7 +631,7 @@ int a6xx_hfi_send_lm_feature_ctrl(struct adreno_device *adreno_dev)
device->pwrctrl.throttle_mask);
if (!ret)
ret = a6xx_hfi_send_req(adreno_dev, H2F_MSG_SET_VALUE, &req);
ret = a6xx_hfi_send_generic_req(adreno_dev, &req);
return ret;
}
@ -793,51 +772,6 @@ void a6xx_hfi_stop(struct adreno_device *adreno_dev)
}
int a6xx_hfi_send_req(struct adreno_device *adreno_dev, unsigned int id,
void *data)
{
switch (id) {
case H2F_MSG_GX_BW_PERF_VOTE: {
struct hfi_gx_bw_perf_vote_cmd *cmd = data;
cmd->hdr = CMD_MSG_HDR(id, sizeof(*cmd));
return a6xx_hfi_send_generic_req(adreno_dev, cmd);
}
case H2F_MSG_PREPARE_SLUMBER: {
struct hfi_prep_slumber_cmd *cmd = data;
if (cmd->freq >= MAX_GX_LEVELS || cmd->bw >= MAX_GX_LEVELS)
return -EINVAL;
cmd->hdr = CMD_MSG_HDR(id, sizeof(*cmd));
return a6xx_hfi_send_generic_req(adreno_dev, cmd);
}
case H2F_MSG_START: {
struct hfi_start_cmd *cmd = data;
cmd->hdr = CMD_MSG_HDR(id, sizeof(*cmd));
return a6xx_hfi_send_generic_req(adreno_dev, cmd);
}
case H2F_MSG_GET_VALUE: {
return a6xx_hfi_send_get_value(adreno_dev, data);
}
case H2F_MSG_SET_VALUE: {
struct hfi_set_value_cmd *cmd = data;
cmd->hdr = CMD_MSG_HDR(id, sizeof(*cmd));
return a6xx_hfi_send_generic_req(adreno_dev, cmd);
}
default:
break;
}
return -EINVAL;
}
/* HFI interrupt handler */
irqreturn_t a6xx_hfi_irq_handler(int irq, void *data)
{

View file

@ -8,7 +8,7 @@
#define HFI_QUEUE_SIZE SZ_4K /* bytes, must be base 4dw */
#define MAX_RCVD_PAYLOAD_SIZE 16 /* dwords */
#define MAX_RCVD_SIZE (MAX_RCVD_PAYLOAD_SIZE + 3) /* dwords */
#define HFI_MAX_MSG_SIZE (SZ_1K>>2) /* dwords */
#define HFI_MAX_MSG_SIZE (SZ_1K)
#define HFI_CMD_ID 0
#define HFI_MSG_ID 1
@ -60,6 +60,7 @@
#define HFI_FEATURE_BCL 11
#define HFI_FEATURE_ACD 12
#define HFI_FEATURE_DIDT 13
#define HFI_FEATURE_KPROF 14
#define HFI_VALUE_FT_POLICY 100
#define HFI_VALUE_RB_MAX_CMDS 101
@ -132,7 +133,6 @@ struct hfi_queue_header {
/* Size is converted from Bytes to DWords */
#define CREATE_MSG_HDR(id, size, type) \
(((type) << 16) | ((((size) >> 2) & 0xFF) << 8) | ((id) & 0xFF))
#define CMD_MSG_HDR(id, size) CREATE_MSG_HDR(id, size, HFI_MSG_CMD)
#define ACK_MSG_HDR(id, size) CREATE_MSG_HDR(id, size, HFI_MSG_ACK)
#define HFI_QUEUE_DEFAULT_CNT 3
@ -481,10 +481,14 @@ struct hfi_ts_notify_cmd {
/* F2H */
struct hfi_ts_retire_cmd {
uint32_t hdr;
uint32_t ctxt_id;
uint32_t ts;
uint32_t ret;
u32 hdr;
u32 ctxt_id;
u32 ts;
u32 type;
u64 submitted_to_rb;
u64 sop;
u64 eop;
u64 retired_on_gmu;
} __packed;
/* H2F */
@ -506,10 +510,11 @@ struct hfi_context_rule_cmd {
/* F2H */
struct hfi_context_bad_cmd {
uint32_t hdr;
uint32_t ctxt_id;
uint32_t status;
uint32_t error;
u32 hdr;
u32 ctxt_id;
u32 policy;
u32 ts;
u32 error;
} __packed;
/* H2F */
@ -524,6 +529,8 @@ struct hfi_submit_cmd {
u32 ctxt_id;
u32 flags;
u32 ts;
u32 profile_gpuaddr_lo;
u32 profile_gpuaddr_hi;
u32 numibs;
} __packed;
@ -539,6 +546,8 @@ struct pending_cmd {
u32 results[MAX_RCVD_SIZE];
/** @complete: Completion to signal hfi ack has been received */
struct completion complete;
/** @node: to add it to the list of hfi packets waiting for ack */
struct list_head node;
};
/**
@ -560,6 +569,13 @@ struct a6xx_hfi {
struct hfi_dcvstable_cmd dcvs_table;
};
#define CMD_MSG_HDR(cmd, id) \
do { \
if (WARN_ON((sizeof(cmd)) > HFI_MAX_MSG_SIZE)) \
return -EMSGSIZE; \
cmd.hdr = CREATE_MSG_HDR((id), (sizeof(cmd)), HFI_MSG_CMD); \
} while (0)
struct a6xx_gmu_device;
/* a6xx_hfi_irq_handler - IRQ handler for HFI interripts */
@ -592,17 +608,6 @@ void a6xx_hfi_stop(struct adreno_device *adreno_dev);
*/
int a6xx_hfi_init(struct adreno_device *adreno_dev);
/**
* a6xx_hfi_send_req - Send an HFI packet to GMU
* @adreno_dev: Pointer to the adreno device
* @id: Packet id to be sent
* @data: Container for the data sent as part of this pcket
*
* Return: 0 on success or negative error on failure
*/
int a6xx_hfi_send_req(struct adreno_device *adreno_dev,
unsigned int id, void *data);
/* Helper function to get to a6xx hfi struct from adreno device */
struct a6xx_hfi *to_a6xx_hfi(struct adreno_device *adreno_dev);

View file

@ -11,10 +11,129 @@
#include "adreno.h"
#include "adreno_a6xx.h"
#include "adreno_a6xx_hwsched.h"
#include "adreno_snapshot.h"
#include "kgsl_device.h"
#include "kgsl_trace.h"
#include "kgsl_util.h"
static size_t adreno_hwsched_snapshot_rb(struct kgsl_device *device, u8 *buf,
size_t remain, void *priv)
{
struct kgsl_snapshot_rb_v2 *header = (struct kgsl_snapshot_rb_v2 *)buf;
u32 *data = (u32 *)(buf + sizeof(*header));
struct kgsl_memdesc *rb = (struct kgsl_memdesc *)priv;
if (remain < rb->size + sizeof(*header)) {
SNAPSHOT_ERR_NOMEM(device, "RB");
return 0;
}
header->start = 0;
header->end = rb->size >> 2;
header->rptr = 0;
header->rbsize = rb->size >> 2;
header->count = rb->size >> 2;
header->timestamp_queued = 0;
header->timestamp_retired = 0;
header->gpuaddr = rb->gpuaddr;
header->id = 0;
memcpy(data, rb->hostptr, rb->size);
return rb->size + sizeof(*header);
}
static void a6xx_hwsched_snapshot_preemption_record(struct kgsl_device *device,
struct kgsl_snapshot *snapshot, struct kgsl_memdesc *md, u64 offset)
{
struct kgsl_snapshot_section_header *section_header =
(struct kgsl_snapshot_section_header *)snapshot->ptr;
u8 *dest = snapshot->ptr + sizeof(*section_header);
struct kgsl_snapshot_gpu_object_v2 *header =
(struct kgsl_snapshot_gpu_object_v2 *)dest;
size_t section_size = sizeof(*section_header) + sizeof(*header) +
A6XX_SNAPSHOT_CP_CTXRECORD_SIZE_IN_BYTES;
if (snapshot->remain < section_size) {
SNAPSHOT_ERR_NOMEM(device, "PREEMPTION RECORD");
return;
}
section_header->magic = SNAPSHOT_SECTION_MAGIC;
section_header->id = KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2;
section_header->size = section_size;
header->size = A6XX_SNAPSHOT_CP_CTXRECORD_SIZE_IN_BYTES >> 2;
header->gpuaddr = md->gpuaddr + offset;
header->ptbase =
kgsl_mmu_pagetable_get_ttbr0(device->mmu.defaultpagetable);
header->type = SNAPSHOT_GPU_OBJECT_GLOBAL;
dest += sizeof(*header);
memcpy(dest, md->hostptr + offset,
A6XX_SNAPSHOT_CP_CTXRECORD_SIZE_IN_BYTES);
snapshot->ptr += section_header->size;
snapshot->remain -= section_header->size;
snapshot->size += section_header->size;
}
static void snapshot_preemption_records(struct kgsl_device *device,
struct kgsl_snapshot *snapshot, struct kgsl_memdesc *md)
{
const struct adreno_a6xx_core *a6xx_core =
to_a6xx_core(ADRENO_DEVICE(device));
u64 ctxt_record_size = A6XX_CP_CTXRECORD_SIZE_IN_BYTES;
u64 offset;
if (a6xx_core->ctxt_record_size)
ctxt_record_size = a6xx_core->ctxt_record_size;
/* All preemption records exist as a single mem alloc entry */
for (offset = 0; offset < md->size; offset += ctxt_record_size)
a6xx_hwsched_snapshot_preemption_record(device, snapshot, md,
offset);
}
void a6xx_hwsched_snapshot(struct adreno_device *adreno_dev,
struct kgsl_snapshot *snapshot)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_hwsched_hfi *hw_hfi = to_a6xx_hwsched_hfi(adreno_dev);
u32 i;
a6xx_gmu_snapshot(adreno_dev, snapshot);
for (i = 0; i < hw_hfi->mem_alloc_entries; i++) {
struct mem_alloc_entry *entry = &hw_hfi->mem_alloc_table[i];
if (entry->desc.mem_kind == MEMKIND_RB)
kgsl_snapshot_add_section(device,
KGSL_SNAPSHOT_SECTION_RB_V2,
snapshot, adreno_hwsched_snapshot_rb,
entry->gpu_md);
if (entry->desc.mem_kind == MEMKIND_SCRATCH)
kgsl_snapshot_add_section(device,
KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, adreno_snapshot_global,
entry->gpu_md);
if (entry->desc.mem_kind == MEMKIND_CSW_SMMU_INFO)
kgsl_snapshot_add_section(device,
KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, adreno_snapshot_global,
entry->gpu_md);
if (entry->desc.mem_kind == MEMKIND_CSW_PRIV_NON_SECURE)
snapshot_preemption_records(device, snapshot,
entry->gpu_md);
}
adreno_hwsched_parse_fault_cmdobj(adreno_dev, snapshot);
}
static int a6xx_hwsched_gmu_first_boot(struct adreno_device *adreno_dev)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
@ -177,6 +296,21 @@ static void a6xx_hwsched_active_count_put(struct adreno_device *adreno_dev)
wake_up(&device->active_cnt_wq);
}
static int unregister_context_hwsched(int id, void *ptr, void *data)
{
struct kgsl_context *context = ptr;
/*
* We don't need to send the unregister hfi packet because
* we are anyway going to lose the gmu state of registered
* contexts. So just reset the flag so that the context
* registers with gmu on its first submission post slumber.
*/
context->gmu_registered = false;
return 0;
}
static int a6xx_hwsched_notify_slumber(struct adreno_device *adreno_dev)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
@ -184,7 +318,8 @@ static int a6xx_hwsched_notify_slumber(struct adreno_device *adreno_dev)
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_prep_slumber_cmd req;
req.hdr = CMD_MSG_HDR(H2F_MSG_PREPARE_SLUMBER, sizeof(req));
CMD_MSG_HDR(req, H2F_MSG_PREPARE_SLUMBER);
req.freq = gmu->hfi.dcvs_table.gpu_level_num -
pwr->default_pwrlevel - 1;
req.bw = pwr->pwrlevels[pwr->default_pwrlevel].bus_freq;
@ -378,6 +513,8 @@ static int a6xx_hwsched_boot(struct adreno_device *adreno_dev)
if (ret)
return ret;
adreno_hwsched_start(adreno_dev);
mod_timer(&device->idle_timer, jiffies +
device->pwrctrl.interval_timeout);
@ -422,6 +559,10 @@ static int a6xx_hwsched_first_boot(struct adreno_device *adreno_dev)
if (ret)
return ret;
adreno_hwsched_init(adreno_dev);
adreno_hwsched_start(adreno_dev);
adreno_get_bus_counters(adreno_dev);
adreno_dev->cooperative_reset = ADRENO_FEATURE(adreno_dev,
@ -477,6 +618,10 @@ no_gx_power:
a6xx_hwsched_gmu_power_off(adreno_dev);
read_lock(&device->context_lock);
idr_for_each(&device->context_idr, unregister_context_hwsched, NULL);
read_unlock(&device->context_lock);
if (!IS_ERR_OR_NULL(adreno_dev->gpu_llc_slice))
llcc_slice_deactivate(adreno_dev->gpu_llc_slice);
@ -575,7 +720,6 @@ static int a6xx_hwsched_dcvs_set(struct adreno_device *adreno_dev,
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct hfi_dcvstable_cmd *table = &gmu->hfi.dcvs_table;
struct hfi_gx_bw_perf_vote_cmd req = {
.hdr = CMD_MSG_HDR(H2F_MSG_GX_BW_PERF_VOTE, sizeof(req)),
.ack_type = DCVS_ACK_BLOCK,
.freq = INVALID_DCVS_IDX,
.bw = INVALID_DCVS_IDX,
@ -603,13 +747,28 @@ static int a6xx_hwsched_dcvs_set(struct adreno_device *adreno_dev,
if ((req.freq == INVALID_DCVS_IDX) && (req.bw == INVALID_DCVS_IDX))
return 0;
CMD_MSG_HDR(req, H2F_MSG_GX_BW_PERF_VOTE);
ret = a6xx_hfi_send_cmd_async(adreno_dev, &req);
if (ret)
if (ret) {
dev_err_ratelimited(&gmu->pdev->dev,
"Failed to set GPU perf idx %d, bw idx %d\n",
req.freq, req.bw);
/*
* If this was a dcvs request along side an active gpu, request
* dispatcher based reset and recovery.
*/
if (test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags)) {
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
}
}
return ret;
}
@ -645,12 +804,145 @@ static int a6xx_hwsched_bus_set(struct adreno_device *adreno_dev, int buslevel,
return ret;
}
static int a6xx_hwsched_pm_suspend(struct adreno_device *adreno_dev)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
int ret;
if (test_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags))
return 0;
trace_kgsl_pwr_request_state(device, KGSL_STATE_SUSPEND);
/* Halt any new submissions */
reinit_completion(&device->halt_gate);
mutex_unlock(&device->mutex);
/* Flush any currently running instances of the dispatcher */
kthread_flush_worker(&kgsl_driver.worker);
mutex_lock(&device->mutex);
/* This ensures that dispatcher doesn't submit any new work */
adreno_dispatcher_halt(device);
/**
* Wait for the dispatcher to retire everything by waiting
* for the active count to go to zero.
*/
ret = kgsl_active_count_wait(device, 0, msecs_to_jiffies(100));
if (ret) {
dev_err(device->dev, "Timed out waiting for the active count\n");
goto err;
}
if (test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags)) {
unsigned long wait = jiffies +
msecs_to_jiffies(ADRENO_IDLE_TIMEOUT);
do {
if (a6xx_hw_isidle(adreno_dev))
break;
} while (time_before(jiffies, wait));
if (!a6xx_hw_isidle(adreno_dev)) {
dev_err(device->dev, "Timed out idling the gpu\n");
ret = -ETIMEDOUT;
goto err;
}
a6xx_hwsched_power_off(adreno_dev);
}
set_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags);
trace_kgsl_pwr_set_state(device, KGSL_STATE_SUSPEND);
return 0;
err:
adreno_dispatcher_unhalt(device);
adreno_hwsched_start(adreno_dev);
return ret;
}
static void a6xx_hwsched_pm_resume(struct adreno_device *adreno_dev)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
if (WARN(!test_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags),
"resume invoked without a suspend\n"))
return;
adreno_dispatcher_unhalt(device);
adreno_hwsched_start(adreno_dev);
clear_bit(GMU_PRIV_PM_SUSPEND, &gmu->flags);
}
static void a6xx_hwsched_drain_ctxt_unregister(struct adreno_device *adreno_dev)
{
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct pending_cmd *cmd = NULL;
read_lock(&hfi->msglock);
list_for_each_entry(cmd, &hfi->msglist, node) {
if (MSG_HDR_GET_ID(cmd->sent_hdr) == H2F_MSG_UNREGISTER_CONTEXT)
complete(&cmd->complete);
}
read_unlock(&hfi->msglock);
}
void a6xx_hwsched_restart(struct adreno_device *adreno_dev)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
int ret;
/*
* Any pending context unregister packets will be lost
* since we hard reset the GMU. This means any threads waiting
* for context unregister hfi ack will timeout. Wake them
* to avoid false positive ack timeout messages later.
*/
a6xx_hwsched_drain_ctxt_unregister(adreno_dev);
read_lock(&device->context_lock);
idr_for_each(&device->context_idr, unregister_context_hwsched, NULL);
read_unlock(&device->context_lock);
if (!test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags))
return;
a6xx_hwsched_hfi_stop(adreno_dev);
a6xx_disable_gpu_irq(adreno_dev);
a6xx_gmu_suspend(adreno_dev);
clear_bit(GMU_PRIV_GPU_STARTED, &gmu->flags);
ret = a6xx_hwsched_boot(adreno_dev);
BUG_ON(ret);
}
const struct adreno_power_ops a6xx_hwsched_power_ops = {
.first_open = a6xx_hwsched_first_open,
.last_close = a6xx_hwsched_power_off,
.active_count_get = a6xx_hwsched_active_count_get,
.active_count_put = a6xx_hwsched_active_count_put,
.touch_wakeup = a6xx_hwsched_touch_wakeup,
.pm_suspend = a6xx_hwsched_pm_suspend,
.pm_resume = a6xx_hwsched_pm_resume,
.gpu_clock_set = a6xx_hwsched_clock_set,
.gpu_bus_set = a6xx_hwsched_bus_set,
};
@ -680,6 +972,8 @@ int a6xx_hwsched_probe(struct platform_device *pdev,
timer_setup(&device->idle_timer, hwsched_idle_timer, 0);
adreno_dev->irq_mask = A6XX_HWSCHED_INT_MASK;
return 0;
}
@ -695,9 +989,13 @@ static int a6xx_hwsched_bind(struct device *dev, struct device *master,
ret = a6xx_hwsched_hfi_probe(ADRENO_DEVICE(device));
if (!ret) {
set_bit(GMU_DISPATCH, &device->gmu_core.flags);
return 0;
}
error:
if (ret)
a6xx_gmu_remove(device);
a6xx_gmu_remove(device);
return ret;
}

View file

@ -7,6 +7,7 @@
#define _ADRENO_A6XX_HWSCHED_H_
#include "adreno_a6xx_hwsched_hfi.h"
#include "adreno_hwsched.h"
/**
* struct a6xx_hwsched_device - Container for the a6xx hwscheduling device
@ -16,6 +17,8 @@ struct a6xx_hwsched_device {
struct a6xx_device a6xx_dev;
/** @hwsched_hfi: Container for hwscheduling specific hfi resources */
struct a6xx_hwsched_hfi hwsched_hfi;
/** @hwsched: Container for the hardware dispatcher */
struct adreno_hwsched hwsched;
};
/**
@ -30,4 +33,20 @@ struct a6xx_hwsched_device {
*/
int a6xx_hwsched_probe(struct platform_device *pdev,
u32 chipid, const struct adreno_gpu_core *gpucore);
/**
* a6xx_hwsched_restart - Restart the gmu and gpu
* @adreno_dev: Pointer to the adreno device
*/
void a6xx_hwsched_restart(struct adreno_device *adreno_dev);
/**
* a6xx_hwsched_snapshot - take a6xx hwsched snapshot
* @adreno_dev: Pointer to the adreno device
* @snapshot: Pointer to the snapshot instance
*
* Snapshot the faulty ib and then snapshot rest of a6xx gmu things
*/
void a6xx_hwsched_snapshot(struct adreno_device *adreno_dev,
struct kgsl_snapshot *snapshot);
#endif

View file

@ -4,17 +4,23 @@
*/
#include <linux/iommu.h>
#include <linux/sched/clock.h>
#include "adreno.h"
#include "adreno_a6xx.h"
#include "adreno_a6xx_hwsched.h"
#include "adreno_hwsched.h"
#include "adreno_pm4types.h"
#include "adreno_trace.h"
#include "kgsl_device.h"
#include "kgsl_pwrctrl.h"
#include "kgsl_trace.h"
#define HFI_QUEUE_MAX (HFI_QUEUE_DEFAULT_CNT + HFI_QUEUE_DISPATCH_MAX_CNT)
/* Use a kmem cache to speed up allocations for f2h packets */
static struct kmem_cache *f2h_cache;
#define DEFINE_QHDR(gmuaddr, id, prio) \
{\
.status = 1, \
@ -31,6 +37,20 @@
.write_index = 0, \
}
static struct dq_info {
/** @max_dq: Maximum number of dispatch queues per RB level */
u32 max_dq;
/** @base_dq_id: Base dqid for level */
u32 base_dq_id;
/** @offset: Next dqid to use for roundrobin context assignment */
u32 offset;
} a6xx_hfi_dqs[KGSL_PRIORITY_MAX_RB_LEVELS] = {
{ 4, 0, }, /* RB0 */
{ 4, 4, }, /* RB1 */
{ 3, 8, }, /* RB2 */
{ 3, 11, }, /* RB3 */
};
static const char * const memkind_strings[] = {
[MEMKIND_GENERIC] = "GMU GENERIC",
[MEMKIND_RB] = "GMU RB",
@ -50,7 +70,7 @@ static const char * const memkind_strings[] = {
[MEMKIND_USER_PROFILE_IBS] = "GMU USER PROFILING",
};
static struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(
struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(
struct adreno_device *adreno_dev)
{
struct a6xx_device *a6xx_dev = container_of(adreno_dev,
@ -61,10 +81,32 @@ static struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(
return &a6xx_hwsched->hwsched_hfi;
}
static void a6xx_receive_ack_async(struct adreno_device *adreno_dev, void *rcvd,
struct pending_cmd *ret_cmd)
static void add_waiter(struct a6xx_hwsched_hfi *hfi, u32 hdr,
struct pending_cmd *ack)
{
memset(ack, 0x0, sizeof(*ack));
init_completion(&ack->complete);
write_lock_irq(&hfi->msglock);
list_add_tail(&ack->node, &hfi->msglist);
write_unlock_irq(&hfi->msglock);
ack->sent_hdr = hdr;
}
static void del_waiter(struct a6xx_hwsched_hfi *hfi, struct pending_cmd *ack)
{
write_lock_irq(&hfi->msglock);
list_del(&ack->node);
write_unlock_irq(&hfi->msglock);
}
static void a6xx_receive_ack_async(struct adreno_device *adreno_dev, void *rcvd)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct pending_cmd *cmd = NULL;
u32 waiters[64], num_waiters = 0, i;
u32 *ack = rcvd;
u32 hdr = ack[0];
u32 req_hdr = ack[1];
@ -73,34 +115,96 @@ static void a6xx_receive_ack_async(struct adreno_device *adreno_dev, void *rcvd,
trace_kgsl_hfi_receive(MSG_HDR_GET_ID(req_hdr),
MSG_HDR_GET_SIZE(req_hdr), MSG_HDR_GET_SEQNUM(req_hdr));
if (size_bytes > sizeof(ret_cmd->results))
dev_err(&gmu->pdev->dev,
if (size_bytes > sizeof(cmd->results))
dev_err_ratelimited(&gmu->pdev->dev,
"Ack result too big: %d Truncating to: %d\n",
size_bytes, sizeof(ret_cmd->results));
size_bytes, sizeof(cmd->results));
if (HDR_CMP_SEQNUM(ret_cmd->sent_hdr, req_hdr)) {
memcpy(ret_cmd->results, ack,
min_t(u32, size_bytes, sizeof(ret_cmd->results)));
complete(&ret_cmd->complete);
return;
read_lock(&hfi->msglock);
list_for_each_entry(cmd, &hfi->msglist, node) {
if (HDR_CMP_SEQNUM(cmd->sent_hdr, req_hdr)) {
memcpy(cmd->results, ack,
min_t(u32, size_bytes,
sizeof(cmd->results)));
complete(&cmd->complete);
read_unlock(&hfi->msglock);
return;
}
if (num_waiters < ARRAY_SIZE(waiters))
waiters[num_waiters++] = cmd->sent_hdr;
}
read_unlock(&hfi->msglock);
/* Didn't find the sender, list the waiter */
dev_err_ratelimited(&gmu->pdev->dev,
"Unexpectedly got id %d seqnum %d while waiting for id %d seqnum %d\n",
"Unexpectedly got id %d seqnum %d. Total waiters: %d Top %d Waiters:\n",
MSG_HDR_GET_ID(req_hdr), MSG_HDR_GET_SEQNUM(req_hdr),
MSG_HDR_GET_ID(ret_cmd->sent_hdr),
MSG_HDR_GET_SEQNUM(ret_cmd->sent_hdr));
num_waiters, min_t(u32, num_waiters, 5));
for (i = 0; i < num_waiters && i < 5; i++)
dev_err_ratelimited(&gmu->pdev->dev,
" id %d seqnum %d\n",
MSG_HDR_GET_ID(waiters[i]),
MSG_HDR_GET_SEQNUM(waiters[i]));
}
static void log_profiling_info(struct adreno_device *adreno_dev, u32 *rcvd)
{
struct hfi_ts_retire_cmd *cmd = (struct hfi_ts_retire_cmd *)rcvd;
struct kgsl_context *context;
struct retire_info info = {0};
context = kgsl_context_get(KGSL_DEVICE(adreno_dev), cmd->ctxt_id);
if (context == NULL)
return;
info.timestamp = cmd->ts;
info.rb_id = adreno_get_level(context->priority);
info.gmu_dispatch_queue = context->gmu_dispatch_queue;
info.submitted_to_rb = cmd->submitted_to_rb;
info.sop = cmd->sop;
info.eop = cmd->eop;
info.retired_on_gmu = cmd->retired_on_gmu;
trace_adreno_cmdbatch_retired(context, &info, 0, 0, 0);
kgsl_context_put(context);
}
struct f2h_packet {
/** @rcvd: the contents of the fw to host packet */
u32 rcvd[MAX_RCVD_SIZE];
/** @node: To add to the fw to host msg list */
struct llist_node node;
};
static void add_f2h_packet(struct adreno_device *adreno_dev, u32 *msg)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct f2h_packet *pkt = kmem_cache_alloc(f2h_cache, GFP_ATOMIC);
u32 size = MSG_HDR_GET_SIZE(msg[0]) << 2;
if (!pkt)
return;
if (size > sizeof(pkt->rcvd))
dev_err_ratelimited(&gmu->pdev->dev,
"f2h packet too big: %d allowed: %d\n",
size, sizeof(pkt->rcvd));
memcpy(pkt->rcvd, msg, min_t(u32, size, sizeof(pkt->rcvd)));
llist_add(&pkt->node, &hfi->f2h_msglist);
}
static void process_msgq_irq(struct adreno_device *adreno_dev)
{
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
u32 rcvd[MAX_RCVD_SIZE];
if (a6xx_hfi_queue_read(gmu, HFI_MSG_ID, rcvd, sizeof(rcvd)) <= 0)
return;
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
while (a6xx_hfi_queue_read(gmu, HFI_MSG_ID, rcvd, sizeof(rcvd)) > 0) {
@ -109,9 +213,12 @@ static void process_msgq_irq(struct adreno_device *adreno_dev)
* because hfi sending thread waits for completion while
* holding the device mutex
*/
if (MSG_HDR_GET_TYPE(rcvd[0]) == HFI_MSG_ACK)
a6xx_receive_ack_async(adreno_dev, rcvd,
&hfi->pending_ack);
if (MSG_HDR_GET_TYPE(rcvd[0]) == HFI_MSG_ACK) {
a6xx_receive_ack_async(adreno_dev, rcvd);
} else {
add_f2h_packet(adreno_dev, rcvd);
wake_up_interruptible(&hfi->f2h_wq);
}
}
}
@ -155,19 +262,19 @@ static irqreturn_t a6xx_hwsched_hfi_handler(int irq, void *data)
#define HFI_IRQ_MSGQ_MASK BIT(0)
#define HFI_RSP_TIMEOUT 100 /* msec */
static int wait_ack_completion(struct adreno_device *adreno_dev, u32 *cmd)
static int wait_ack_completion(struct adreno_device *adreno_dev,
struct pending_cmd *ack)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
int rc;
rc = wait_for_completion_timeout(&hfi->pending_ack.complete,
rc = wait_for_completion_timeout(&ack->complete,
HFI_RSP_TIMEOUT);
if (!rc) {
dev_err(&gmu->pdev->dev,
"Ack timeout for id:%d sequence=%d\n",
MSG_HDR_GET_ID(*cmd),
MSG_HDR_GET_SEQNUM(*cmd));
MSG_HDR_GET_ID(ack->sent_hdr),
MSG_HDR_GET_SEQNUM(ack->sent_hdr));
gmu_fault_snapshot(KGSL_DEVICE(adreno_dev));
return -ETIMEDOUT;
}
@ -175,24 +282,20 @@ static int wait_ack_completion(struct adreno_device *adreno_dev, u32 *cmd)
return 0;
}
static int check_ack_failure(struct adreno_device *adreno_dev)
static int check_ack_failure(struct adreno_device *adreno_dev,
struct pending_cmd *ack)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct pending_cmd *cmd = &hfi->pending_ack;
int rc = cmd->results[2] ? -EINVAL : 0;
if (cmd->results[2] == 0xffffffff)
dev_err(&gmu->pdev->dev,
"HFI ACK failure: Req 0x%8.8x\n",
cmd->results[1]);
if (ack->results[2] != 0xffffffff)
return 0;
/* reset the ack */
memset(cmd->results, 0x0, sizeof(cmd->results));
reinit_completion(&hfi->pending_ack.complete);
cmd->sent_hdr = 0;
dev_err(&gmu->pdev->dev,
"ACK error: sender id %d seqnum %d\n",
MSG_HDR_GET_ID(ack->sent_hdr),
MSG_HDR_GET_SEQNUM(ack->sent_hdr));
return rc;
return -EINVAL;
}
int a6xx_hfi_send_cmd_async(struct adreno_device *adreno_dev, void *data)
@ -202,20 +305,26 @@ int a6xx_hfi_send_cmd_async(struct adreno_device *adreno_dev, void *data)
u32 *cmd = data;
u32 seqnum = atomic_inc_return(&gmu->hfi.seqnum);
int rc;
struct pending_cmd pending_ack;
*cmd = MSG_HDR_SET_SEQNUM(*cmd, seqnum);
hfi->pending_ack.sent_hdr = cmd[0];
add_waiter(hfi, *cmd, &pending_ack);
rc = a6xx_hfi_cmdq_write(adreno_dev, cmd);
if (rc)
return rc;
goto done;
rc = wait_ack_completion(adreno_dev, cmd);
rc = wait_ack_completion(adreno_dev, &pending_ack);
if (rc)
return rc;
goto done;
return check_ack_failure(adreno_dev);
rc = check_ack_failure(adreno_dev, &pending_ack);
done:
del_waiter(hfi, &pending_ack);
return rc;
}
static void init_queues(struct a6xx_hfi *hfi)
@ -462,16 +571,17 @@ static int send_start_msg(struct adreno_device *adreno_dev)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
unsigned int seqnum = atomic_inc_return(&gmu->hfi.seqnum);
int rc = 0;
struct hfi_start_cmd cmd;
u32 rcvd[MAX_RCVD_SIZE];
struct pending_cmd pending_ack = {0};
CMD_MSG_HDR(cmd, H2F_MSG_START);
cmd.hdr = CMD_MSG_HDR(H2F_MSG_START, sizeof(cmd));
cmd.hdr = MSG_HDR_SET_SEQNUM(cmd.hdr, seqnum);
hfi->pending_ack.sent_hdr = cmd.hdr;
pending_ack.sent_hdr = cmd.hdr;
rc = a6xx_hfi_cmdq_write(adreno_dev, (u32 *)&cmd);
if (rc)
@ -500,11 +610,11 @@ poll:
}
if (MSG_HDR_GET_TYPE(rcvd[0]) == HFI_MSG_ACK) {
rc = a6xx_receive_ack_cmd(gmu, rcvd, &hfi->pending_ack);
rc = a6xx_receive_ack_cmd(gmu, rcvd, &pending_ack);
if (rc)
return rc;
return check_ack_failure(adreno_dev);
return check_ack_failure(adreno_dev, &pending_ack);
}
if (MSG_HDR_GET_ID(rcvd[0]) == F2H_MSG_MEM_ALLOC) {
@ -572,8 +682,6 @@ static void enable_async_hfi(struct adreno_device *adreno_dev)
gmu_core_regwrite(KGSL_DEVICE(adreno_dev), A6XX_GMU_GMU2HOST_INTR_MASK,
(u32)~hfi->irq_mask);
init_completion(&hfi->pending_ack.complete);
}
int a6xx_hwsched_hfi_start(struct adreno_device *adreno_dev)
@ -610,6 +718,10 @@ int a6xx_hwsched_hfi_start(struct adreno_device *adreno_dev)
if (ret)
goto err;
ret = a6xx_hfi_send_feature_ctrl(adreno_dev, HFI_FEATURE_KPROF, 1, 0);
if (ret)
return ret;
ret = a6xx_hfi_send_core_fw_start(adreno_dev);
if (ret)
goto err;
@ -658,8 +770,9 @@ static int cp_init(struct adreno_device *adreno_dev)
{
u32 cmds[A6XX_CP_INIT_DWORDS + 1];
cmds[0] = CMD_MSG_HDR(H2F_MSG_ISSUE_CMD_RAW,
(A6XX_CP_INIT_DWORDS + 1) << 2);
cmds[0] = CREATE_MSG_HDR(H2F_MSG_ISSUE_CMD_RAW,
(A6XX_CP_INIT_DWORDS + 1) << 2, HFI_MSG_CMD);
memcpy(&cmds[1], adreno_dev->cp_init_cmds, A6XX_CP_INIT_DWORDS << 2);
return submit_raw_cmds(adreno_dev, cmds,
@ -670,7 +783,9 @@ static int send_switch_to_unsecure(struct adreno_device *adreno_dev)
{
u32 cmds[3];
cmds[0] = CMD_MSG_HDR(H2F_MSG_ISSUE_CMD_RAW, sizeof(cmds));
cmds[0] = CREATE_MSG_HDR(H2F_MSG_ISSUE_CMD_RAW, sizeof(cmds),
HFI_MSG_CMD);
cmds[1] = cp_type7_packet(CP_SET_SECURE_MODE, 1);
cmds[2] = 0;
@ -702,6 +817,55 @@ int a6xx_hwsched_cp_init(struct adreno_device *adreno_dev)
return ret;
}
static void process_ts_retire(struct adreno_device *adreno_dev, u32 *rcvd)
{
log_profiling_info(adreno_dev, rcvd);
adreno_hwsched_trigger(adreno_dev);
}
static void process_ctx_bad(struct adreno_device *adreno_dev, void *rcvd)
{
struct hfi_context_bad_cmd *cmd = rcvd;
/* Block dispatcher to submit more commands */
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_mark_drawobj(adreno_dev, cmd->ctxt_id, cmd->ts);
}
static int hfi_f2h_main(void *arg)
{
struct adreno_device *adreno_dev = arg;
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct llist_node *list;
struct f2h_packet *pkt, *tmp;
while (!kthread_should_stop()) {
wait_event_interruptible(hfi->f2h_wq,
(!llist_empty(&hfi->f2h_msglist) &&
!kthread_should_stop()));
if (kthread_should_stop())
break;
list = llist_del_all(&hfi->f2h_msglist);
list = llist_reverse_order(list);
llist_for_each_entry_safe(pkt, tmp, list, node) {
if (MSG_HDR_GET_ID(pkt->rcvd[0]) == F2H_MSG_TS_RETIRE)
process_ts_retire(adreno_dev, pkt->rcvd);
if (MSG_HDR_GET_ID(pkt->rcvd[0]) == F2H_MSG_CONTEXT_BAD)
process_ctx_bad(adreno_dev, pkt->rcvd);
kmem_cache_free(f2h_cache, pkt);
}
}
return 0;
}
int a6xx_hwsched_hfi_probe(struct adreno_device *adreno_dev)
{
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
@ -717,9 +881,69 @@ int a6xx_hwsched_hfi_probe(struct adreno_device *adreno_dev)
disable_irq(gmu->hfi.irq);
rwlock_init(&hw_hfi->msglock);
INIT_LIST_HEAD(&hw_hfi->msglist);
init_llist_head(&hw_hfi->f2h_msglist);
init_waitqueue_head(&hw_hfi->f2h_wq);
hw_hfi->f2h_task = kthread_run(hfi_f2h_main, adreno_dev, "gmu_f2h");
f2h_cache = KMEM_CACHE(f2h_packet, 0);
return 0;
}
static void add_profile_events(struct adreno_device *adreno_dev,
struct kgsl_drawobj *drawobj, struct adreno_submit_time *time)
{
unsigned long flags;
u64 time_in_s;
unsigned long time_in_ns;
struct kgsl_context *context = drawobj->context;
struct submission_info info = {0};
/*
* Here we are attempting to create a mapping between the
* GPU time domain (alwayson counter) and the CPU time domain
* (local_clock) by sampling both values as close together as
* possible. This is useful for many types of debugging and
* profiling. In order to make this mapping as accurate as
* possible, we must turn off interrupts to avoid running
* interrupt handlers between the two samples.
*/
local_irq_save(flags);
/* Read always on registers */
time->ticks = a6xx_read_alwayson(adreno_dev);
/* Trace the GPU time to create a mapping to ftrace time */
trace_adreno_cmdbatch_sync(context->id, context->priority,
drawobj->timestamp, time->ticks);
/* Get the kernel clock for time since boot */
time->ktime = local_clock();
/* Get the timeofday for the wall time (for the user) */
getnstimeofday(&time->utime);
local_irq_restore(flags);
/* Return kernel clock time to the client if requested */
time_in_s = time->ktime;
time_in_ns = do_div(time_in_s, 1000000000);
info.inflight = -1;
info.rb_id = adreno_get_level(context->priority);
info.gmu_dispatch_queue = context->gmu_dispatch_queue;
trace_adreno_cmdbatch_submitted(drawobj, &info, time->ticks,
(unsigned long) time_in_s, time_in_ns / 1000, 0);
}
#define CTXT_FLAG_PMODE 0x00000001
#define CTXT_FLAG_SWITCH_INTERNAL 0x00000002
#define CTXT_FLAG_SWITCH 0x00000008
@ -744,13 +968,31 @@ int a6xx_hwsched_hfi_probe(struct adreno_device *adreno_dev)
#define CTXT_FLAG_PREEMPT_STYLE_RB 1
#define CTXT_FLAG_PREEMPT_STYLE_FG 2
static u32 get_next_dq(u32 priority)
{
struct dq_info *info = &a6xx_hfi_dqs[priority];
u32 next = info->base_dq_id + info->offset;
info->offset = (info->offset + 1) % info->max_dq;
return next;
}
static u32 get_dq_id(u32 priority)
{
u32 level = adreno_get_level(priority);
return get_next_dq(level);
}
static int send_context_register(struct adreno_device *adreno_dev,
struct kgsl_context *context)
{
struct hfi_register_ctxt_cmd cmd;
struct kgsl_pagetable *pt = context->proc_priv->pagetable;
cmd.hdr = CMD_MSG_HDR(H2F_MSG_REGISTER_CONTEXT, sizeof(cmd));
CMD_MSG_HDR(cmd, H2F_MSG_REGISTER_CONTEXT);
cmd.ctxt_id = context->id;
cmd.flags = CTXT_FLAG_NOTIFY | context->flags;
cmd.pt_addr = kgsl_mmu_pagetable_get_ttbr0(pt);
@ -766,7 +1008,7 @@ static int send_context_pointers(struct adreno_device *adreno_dev,
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct hfi_context_pointers_cmd cmd;
cmd.hdr = CMD_MSG_HDR(H2F_MSG_CONTEXT_POINTERS, sizeof(cmd));
CMD_MSG_HDR(cmd, H2F_MSG_CONTEXT_POINTERS);
cmd.ctxt_id = context->id;
cmd.sop_addr = MEMSTORE_ID_GPU_ADDR(device, context->id, soptimestamp);
cmd.eop_addr = MEMSTORE_ID_GPU_ADDR(device, context->id, eoptimestamp);
@ -797,22 +1039,24 @@ static int hfi_context_register(struct adreno_device *adreno_dev,
}
ret = send_context_pointers(adreno_dev, context);
if (ret)
if (ret) {
dev_err(&gmu->pdev->dev,
"Unable to register context %d pointers: %d\n",
context->id, ret);
return ret;
}
if (!ret)
context->gmu_registered = true;
context->gmu_registered = true;
context->gmu_dispatch_queue = get_dq_id(context->priority);
return ret;
return 0;
}
#define HFI_DSP_IRQ_BASE 2
#define DISPQ_IRQ_BIT(_idx) BIT((_idx) + HFI_DSP_IRQ_BASE)
int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev,
struct kgsl_drawobj_cmd *cmdobj)
{
struct a6xx_hfi *hfi = to_a6xx_hfi(adreno_dev);
@ -822,6 +1066,7 @@ int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
struct kgsl_drawobj *drawobj = DRAWOBJ(cmdobj);
struct hfi_issue_ib *issue_ib;
struct hfi_submit_cmd *cmd;
struct adreno_submit_time time = {0};
ret = hfi_context_register(adreno_dev, drawobj->context);
if (ret)
@ -831,22 +1076,47 @@ int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
list_for_each_entry(ib, &cmdobj->cmdlist, node)
numibs++;
/* We need to dispatch a marker object but not execute it on the GPU */
if (test_bit(CMDOBJ_SKIP, &cmdobj->priv))
numibs = 0;
/* Add a *issue_ib struct for each IB */
cmd_sizebytes = sizeof(*cmd) + (sizeof(*issue_ib) * numibs);
if (WARN_ON(cmd_sizebytes > HFI_MAX_MSG_SIZE))
return -EMSGSIZE;
cmd = kvmalloc(cmd_sizebytes, GFP_KERNEL);
if (cmd == NULL)
return -ENOMEM;
cmd->hdr = CMD_MSG_HDR(H2F_MSG_ISSUE_CMD, cmd_sizebytes);
cmd->hdr = CREATE_MSG_HDR(H2F_MSG_ISSUE_CMD, cmd_sizebytes,
HFI_MSG_CMD);
cmd->hdr = MSG_HDR_SET_SEQNUM(cmd->hdr,
atomic_inc_return(&hfi->seqnum));
cmd->ctxt_id = drawobj->context->id;
cmd->flags = flags;
cmd->flags = CTXT_FLAG_NOTIFY;
cmd->ts = drawobj->timestamp;
cmd->numibs = numibs;
if (!numibs)
goto skipib;
if ((drawobj->flags & KGSL_DRAWOBJ_PROFILING) &&
!cmdobj->profiling_buf_entry) {
time.drawobj = drawobj;
cmd->profile_gpuaddr_lo =
lower_32_bits(cmdobj->profiling_buffer_gpuaddr);
cmd->profile_gpuaddr_hi =
upper_32_bits(cmdobj->profiling_buffer_gpuaddr);
/* Indicate to GMU to do user profiling for this submission */
cmd->flags |= BIT(4);
}
issue_ib = (struct hfi_issue_ib *)&cmd[1];
list_for_each_entry(ib, &cmdobj->cmdlist, node) {
@ -855,7 +1125,10 @@ int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
issue_ib++;
}
ret = a6xx_hfi_queue_write(adreno_dev, HFI_DSP_ID_0, (u32 *)cmd);
skipib:
ret = a6xx_hfi_queue_write(adreno_dev,
HFI_DSP_ID_0 + drawobj->context->gmu_dispatch_queue,
(u32 *)cmd);
if (ret)
goto free;
@ -865,12 +1138,116 @@ int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
*/
wmb();
add_profile_events(adreno_dev, drawobj, &time);
cmdobj->submit_ticks = time.ticks;
/* Send interrupt to GMU to receive the message */
gmu_core_regwrite(KGSL_DEVICE(adreno_dev), A6XX_GMU_HOST2GMU_INTR_SET,
DISPQ_IRQ_BIT(0));
DISPQ_IRQ_BIT(drawobj->context->gmu_dispatch_queue));
/* Put the profiling information in the user profiling buffer */
adreno_profile_submit_time(&time);
free:
kvfree(cmd);
return ret;
}
static int send_context_unregister_hfi(struct adreno_device *adreno_dev,
struct kgsl_context *context, u32 ts)
{
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
struct a6xx_hwsched_hfi *hfi = to_a6xx_hwsched_hfi(adreno_dev);
struct pending_cmd pending_ack;
struct hfi_unregister_ctxt_cmd cmd;
u32 seqnum;
int rc;
CMD_MSG_HDR(cmd, H2F_MSG_UNREGISTER_CONTEXT);
cmd.ctxt_id = context->id,
cmd.ts = ts,
seqnum = atomic_inc_return(&gmu->hfi.seqnum);
cmd.hdr = MSG_HDR_SET_SEQNUM(cmd.hdr, seqnum);
add_waiter(hfi, cmd.hdr, &pending_ack);
rc = a6xx_hfi_cmdq_write(adreno_dev, (u32 *)&cmd);
if (rc)
goto done;
mutex_unlock(&device->mutex);
rc = wait_for_completion_timeout(&pending_ack.complete,
msecs_to_jiffies(30 * 1000));
if (!rc) {
dev_err(&gmu->pdev->dev,
"Ack timeout for context unregister seq: %d ctx: %d ts: %d\n",
MSG_HDR_GET_SEQNUM(pending_ack.sent_hdr),
context->id, ts);
rc = -ETIMEDOUT;
mutex_lock(&device->mutex);
gmu_fault_snapshot(device);
/*
* Trigger dispatcher based reset and recovery. Invalidate the
* context so that any un-finished inflight submissions are not
* replayed after recovery.
*/
adreno_mark_guilty_context(device, context->id);
adreno_drawctxt_invalidate(device, context);
adreno_get_gpu_halt(adreno_dev);
adreno_hwsched_set_fault(adreno_dev);
goto done;
}
mutex_lock(&device->mutex);
rc = check_ack_failure(adreno_dev, &pending_ack);
done:
del_waiter(hfi, &pending_ack);
return rc;
}
void a6xx_hwsched_context_detach(struct adreno_context *drawctxt)
{
struct kgsl_context *context = &drawctxt->base;
struct kgsl_device *device = context->device;
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
struct a6xx_gmu_device *gmu = to_a6xx_gmu(adreno_dev);
int ret = 0;
mutex_lock(&device->mutex);
/* Only send HFI if device is not in SLUMBER */
if (context->gmu_registered &&
test_bit(GMU_PRIV_GPU_STARTED, &gmu->flags))
ret = send_context_unregister_hfi(adreno_dev, context,
drawctxt->internal_timestamp);
if (!ret) {
kgsl_sharedmem_writel(device->memstore,
KGSL_MEMSTORE_OFFSET(context->id, soptimestamp),
drawctxt->timestamp);
kgsl_sharedmem_writel(device->memstore,
KGSL_MEMSTORE_OFFSET(context->id, eoptimestamp),
drawctxt->timestamp);
adreno_profile_process_results(adreno_dev);
}
context->gmu_registered = false;
mutex_unlock(&device->mutex);
}

View file

@ -107,10 +107,18 @@ struct mem_alloc_entry {
struct a6xx_hwsched_hfi {
struct mem_alloc_entry mem_alloc_table[32];
u32 mem_alloc_entries;
/** @pending_ack: To track un-ack'd hfi packet */
struct pending_cmd pending_ack;
/** @irq_mask: Store the hfi interrupt mask */
u32 irq_mask;
/** @msglock: To protect the list of un-ACKed hfi packets */
rwlock_t msglock;
/** @msglist: List of un-ACKed hfi packets */
struct list_head msglist;
/** @f2h_task: Task for processing gmu fw to host packets */
struct task_struct *f2h_task;
/** @f2h_msglist: List of gmu fw to host packets */
struct llist_head f2h_msglist;
/** @f2h_wq: Waitqueue for the f2h_task */
wait_queue_head_t f2h_wq;
};
struct kgsl_drawobj_cmd;
@ -178,7 +186,6 @@ int a6xx_hfi_send_cmd_async(struct adreno_device *adreno_dev, void *data);
/**
* a6xx_hwsched_submit_cmdobj - Dispatch IBs to dispatch queues
* @adreno_dev: Pointer to adreno device structure
* @flags: Flags associated with the submission
* @cmdobj: The command object which needs to be submitted
*
* This function is used to register the context if needed and submit
@ -186,6 +193,18 @@ int a6xx_hfi_send_cmd_async(struct adreno_device *adreno_dev, void *data);
* Return: 0 on success and negative error on failure
*/
int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev, u32 flags,
int a6xx_hwsched_submit_cmdobj(struct adreno_device *adreno_dev,
struct kgsl_drawobj_cmd *cmdobj);
/**
* a6xx_hwsched_context_detach - Unregister a context with GMU
* @drawctxt: Pointer to the adreno context
*
* This function sends context unregister HFI and waits for the ack
* to ensure all submissions from this context have retired
*/
void a6xx_hwsched_context_detach(struct adreno_context *drawctxt);
/* Helper function to get to a6xx hwsched hfi device from adreno device */
struct a6xx_hwsched_hfi *to_a6xx_hwsched_hfi(struct adreno_device *adreno_dev);
#endif

View file

@ -714,20 +714,6 @@ int a6xx_preemption_init(struct adreno_device *adreno_dev)
return 0;
}
void a6xx_preemption_context_destroy(struct kgsl_context *context)
{
struct kgsl_device *device = context->device;
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
if (!adreno_is_preemption_enabled(adreno_dev))
return;
gpumem_free_entry(context->user_ctxt_record);
/* Put the extra ref from gpumem_alloc_entry() */
kgsl_mem_entry_put(context->user_ctxt_record);
}
int a6xx_preemption_context_init(struct kgsl_context *context)
{
struct kgsl_device *device = context->device;

View file

@ -1153,7 +1153,7 @@ static int a6xx_rgmu_pm_suspend(struct adreno_device *adreno_dev)
reinit_completion(&device->halt_gate);
/* wait for active count so device can be put in slumber */
ret = kgsl_active_count_wait(device, 0);
ret = kgsl_active_count_wait(device, 0, HZ);
if (ret) {
dev_err(device->dev,
"Timed out waiting for the active count\n");
@ -1183,10 +1183,9 @@ static void a6xx_rgmu_pm_resume(struct adreno_device *adreno_dev)
struct kgsl_device *device = KGSL_DEVICE(adreno_dev);
struct a6xx_rgmu_device *rgmu = to_a6xx_rgmu(adreno_dev);
if (!test_bit(RGMU_PRIV_PM_SUSPEND, &rgmu->flags)) {
dev_err(device->dev, "resume invoked without a suspend\n");
if (WARN(!test_bit(GMU_PRIV_PM_SUSPEND, &rgmu->flags),
"resume invoked without a suspend\n"))
return;
}
adreno_dispatcher_unhalt(device);
@ -1310,6 +1309,8 @@ int a6xx_rgmu_device_probe(struct platform_device *pdev,
timer_setup(&device->idle_timer, rgmu_idle_timer, 0);
adreno_dev->irq_mask = A6XX_INT_MASK;
return 0;
}

View file

@ -346,8 +346,7 @@ static int build_dcvs_table(struct adreno_device *adreno_dev)
struct rpmh_arc_vals gx_arc, cx_arc, mx_arc;
int ret;
hfi->dcvs_table.hdr = CMD_MSG_HDR(H2F_MSG_PERF_TBL,
sizeof(hfi->dcvs_table));
CMD_MSG_HDR(hfi->dcvs_table, H2F_MSG_PERF_TBL);
ret = rpmh_arc_cmds(&gx_arc, "gfx.lvl");
if (ret)
@ -388,7 +387,6 @@ static void build_bw_table_cmd(struct hfi_bwtable_cmd *cmd,
{
u32 i, j;
cmd->hdr = CMD_MSG_HDR(H2F_MSG_BW_VOTE_TBL, sizeof(*cmd));
cmd->bw_level_num = ddr->num_levels;
cmd->ddr_cmds_num = ddr->num_cmds;
cmd->ddr_wait_bitmask = ddr->wait_bitmask;
@ -442,6 +440,8 @@ static int build_bw_table(struct adreno_device *adreno_dev)
return PTR_ERR(cnoc);
}
CMD_MSG_HDR(gmu->hfi.bw_table, H2F_MSG_BW_VOTE_TBL);
build_bw_table_cmd(&gmu->hfi.bw_table, ddr, cnoc);
free_rpmh_bw_votes(ddr);

View file

@ -12,9 +12,6 @@
#define A6XX_NUM_XIN_AXI_BLOCKS 5
#define A6XX_NUM_XIN_CORE_BLOCKS 4
/* Snapshot section size of each CP preemption record for A6XX */
#define A6XX_SNAPSHOT_CP_CTXRECORD_SIZE_IN_BYTES (64 * 1024)
static const unsigned int a6xx_gras_cluster[] = {
0x8000, 0x8006, 0x8010, 0x8092, 0x8094, 0x809D, 0x80A0, 0x80A6,
0x80AF, 0x80F1, 0x8100, 0x8107, 0x8109, 0x8109, 0x8110, 0x8110,

View file

@ -271,6 +271,8 @@ static void _retire_timestamp(struct kgsl_drawobj *drawobj)
struct kgsl_context *context = drawobj->context;
struct adreno_context *drawctxt = ADRENO_CONTEXT(context);
struct kgsl_device *device = context->device;
struct adreno_ringbuffer *rb = drawctxt->rb;
struct retire_info info = {0};
/*
* Write the start and end timestamp to the memstore to keep the
@ -284,21 +286,29 @@ static void _retire_timestamp(struct kgsl_drawobj *drawobj)
KGSL_MEMSTORE_OFFSET(context->id, eoptimestamp),
drawobj->timestamp);
/* Retire pending GPU events for the object */
kgsl_process_event_group(device, &context->events);
info.inflight = -1;
info.rb_id = rb->id;
info.wptr = rb->wptr;
info.timestamp = drawobj->timestamp;
/*
* For A3xx we still get the rptr from the CP_RB_RPTR instead of
* rptr scratch out address. At this point GPU clocks turned off.
* So avoid reading GPU register directly for A3xx.
*/
if (adreno_is_a3xx(ADRENO_DEVICE(device)))
trace_adreno_cmdbatch_retired(drawobj, -1, 0, 0, drawctxt->rb,
0, 0);
else
trace_adreno_cmdbatch_retired(drawobj, -1, 0, 0, drawctxt->rb,
adreno_get_rptr(drawctxt->rb), 0);
if (adreno_is_a3xx(ADRENO_DEVICE(device))) {
trace_adreno_cmdbatch_retired(context, &info,
drawobj->flags, rb->dispatch_q.inflight, 0);
} else {
info.rptr = adreno_get_rptr(rb);
trace_adreno_cmdbatch_retired(context, &info,
drawobj->flags, rb->dispatch_q.inflight, 0);
}
kgsl_drawobj_destroy(drawobj);
}
@ -542,6 +552,7 @@ static int sendcmd(struct adreno_device *adreno_dev,
uint64_t secs = 0;
unsigned long nsecs = 0;
int ret;
struct submission_info info = {0};
mutex_lock(&device->mutex);
if (adreno_gpu_halt(adreno_dev) != 0) {
@ -650,9 +661,15 @@ static int sendcmd(struct adreno_device *adreno_dev,
dispatch_q->expires = jiffies +
msecs_to_jiffies(adreno_drawobj_timeout);
trace_adreno_cmdbatch_submitted(drawobj, (int) dispatcher->inflight,
time.ticks, (unsigned long) secs, nsecs / 1000, drawctxt->rb,
adreno_get_rptr(drawctxt->rb));
info.inflight = (int) dispatcher->inflight;
info.rb_id = drawctxt->rb->id;
info.rptr = adreno_get_rptr(drawctxt->rb);
info.wptr = drawctxt->rb->wptr;
info.gmu_dispatch_queue = -1;
trace_adreno_cmdbatch_submitted(drawobj, &info,
time.ticks, (unsigned long) secs, nsecs / 1000,
dispatch_q->inflight);
mutex_unlock(&device->mutex);
@ -1507,14 +1524,7 @@ static int _mark_context(int id, void *ptr, void *data)
return 0;
}
/**
* mark_guilty_context() - Mark the given context as guilty (failed recovery)
* @device: Pointer to a KGSL device structure
* @id: Context ID of the guilty context (or 0 to mark all as guilty)
*
* Mark the given (or all) context(s) as guilty (failed recovery)
*/
static void mark_guilty_context(struct kgsl_device *device, unsigned int id)
void adreno_mark_guilty_context(struct kgsl_device *device, unsigned int id)
{
/* Mark the status for all the contexts in the device */
@ -1936,7 +1946,7 @@ static void process_cmdobj_fault(struct kgsl_device *device,
state, drawobj->context->id, drawobj->timestamp);
/* Mark the context as failed */
mark_guilty_context(device, drawobj->context->id);
adreno_mark_guilty_context(device, drawobj->context->id);
/* Invalidate the context */
adreno_drawctxt_invalidate(device, drawobj->context);
@ -1974,7 +1984,8 @@ static void recover_dispatch_q(struct kgsl_device *device,
dispatch_q->cmd_q[ptr];
struct kgsl_drawobj *drawobj = DRAWOBJ(cmdobj);
mark_guilty_context(device, drawobj->context->id);
adreno_mark_guilty_context(device,
drawobj->context->id);
adreno_drawctxt_invalidate(device, drawobj->context);
kgsl_drawobj_destroy(drawobj);
@ -2047,7 +2058,7 @@ replay:
replay[i]->base.timestamp);
/* Mark this context as guilty (failed recovery) */
mark_guilty_context(device,
adreno_mark_guilty_context(device,
replay[i]->base.context->id);
adreno_drawctxt_invalidate(device,
@ -2335,7 +2346,9 @@ static void retire_cmdobj(struct adreno_device *adreno_dev,
struct adreno_dispatcher *dispatcher = &adreno_dev->dispatcher;
struct kgsl_drawobj *drawobj = DRAWOBJ(cmdobj);
struct adreno_context *drawctxt = ADRENO_CONTEXT(drawobj->context);
struct adreno_ringbuffer *rb = drawctxt->rb;
uint64_t start = 0, end = 0;
struct retire_info info = {0};
if (cmdobj->fault_recovery != 0) {
set_bit(ADRENO_CONTEXT_FAULT, &drawobj->context->priv);
@ -2345,20 +2358,28 @@ static void retire_cmdobj(struct adreno_device *adreno_dev,
if (test_bit(CMDOBJ_PROFILE, &cmdobj->priv))
cmdobj_profile_ticks(adreno_dev, cmdobj, &start, &end);
info.inflight = (int)dispatcher->inflight;
info.rb_id = rb->id;
info.wptr = rb->wptr;
info.timestamp = drawobj->timestamp;
info.sop = start;
info.eop = end;
/*
* For A3xx we still get the rptr from the CP_RB_RPTR instead of
* rptr scratch out address. At this point GPU clocks turned off.
* So avoid reading GPU register directly for A3xx.
*/
if (adreno_is_a3xx(adreno_dev))
trace_adreno_cmdbatch_retired(drawobj,
(int) dispatcher->inflight, start, end,
ADRENO_DRAWOBJ_RB(drawobj), 0, cmdobj->fault_recovery);
else
trace_adreno_cmdbatch_retired(drawobj,
(int) dispatcher->inflight, start, end,
ADRENO_DRAWOBJ_RB(drawobj),
adreno_get_rptr(drawctxt->rb), cmdobj->fault_recovery);
if (adreno_is_a3xx(adreno_dev)) {
trace_adreno_cmdbatch_retired(drawobj->context, &info,
drawobj->flags, rb->dispatch_q.inflight,
cmdobj->fault_recovery);
} else {
info.rptr = adreno_get_rptr(rb);
trace_adreno_cmdbatch_retired(drawobj->context, &info,
drawobj->flags, rb->dispatch_q.inflight,
cmdobj->fault_recovery);
}
drawctxt->submit_retire_ticks[drawctxt->ticks_index] =
end - cmdobj->submit_ticks;

View file

@ -381,9 +381,6 @@ adreno_drawctxt_create(struct kgsl_device_private *dev_priv,
(drawctxt->base.flags & KGSL_CONTEXT_PRIORITY_MASK) >>
KGSL_CONTEXT_PRIORITY_SHIFT;
/* set the context ringbuffer */
drawctxt->rb = adreno_ctx_get_rb(adreno_dev, drawctxt);
/*
* Now initialize the common part of the context. This allocates the
* context id, and then possibly another thread could look it up.
@ -405,8 +402,6 @@ adreno_drawctxt_create(struct kgsl_device_private *dev_priv,
adreno_context_debugfs_init(ADRENO_DEVICE(device), drawctxt);
INIT_LIST_HEAD(&drawctxt->active_node);
if (gpudev->preemption_context_init) {
ret = gpudev->preemption_context_init(&drawctxt->base);
if (ret != 0) {
@ -417,57 +412,23 @@ adreno_drawctxt_create(struct kgsl_device_private *dev_priv,
/* copy back whatever flags we dediced were valid */
*flags = drawctxt->base.flags;
if (!test_bit(GMU_DISPATCH, &device->gmu_core.flags)) {
/* set the context ringbuffer */
drawctxt->rb = adreno_ctx_get_rb(adreno_dev, drawctxt);
INIT_LIST_HEAD(&drawctxt->active_node);
}
return &drawctxt->base;
}
/**
* adreno_drawctxt_detach(): detach a context from the GPU
* @context: Generic KGSL context container for the context
*
*/
void adreno_drawctxt_detach(struct kgsl_context *context)
static void wait_for_timestamp_rb(struct kgsl_device *device,
struct adreno_context *drawctxt)
{
struct kgsl_device *device;
struct adreno_device *adreno_dev;
struct adreno_gpudev *gpudev;
struct adreno_context *drawctxt;
struct adreno_ringbuffer *rb;
int ret, count, i;
struct kgsl_drawobj *list[ADRENO_CONTEXT_DRAWQUEUE_SIZE];
if (context == NULL)
return;
device = context->device;
adreno_dev = ADRENO_DEVICE(device);
gpudev = ADRENO_GPU_DEVICE(adreno_dev);
drawctxt = ADRENO_CONTEXT(context);
rb = drawctxt->rb;
spin_lock(&drawctxt->lock);
spin_lock(&adreno_dev->active_list_lock);
list_del_init(&drawctxt->active_node);
spin_unlock(&adreno_dev->active_list_lock);
count = drawctxt_detach_drawobjs(drawctxt, list);
spin_unlock(&drawctxt->lock);
for (i = 0; i < count; i++) {
/*
* If the context is deteached while we are waiting for
* the next command in GFT SKIP CMD, print the context
* detached status here.
*/
adreno_fault_skipcmd_detached(adreno_dev, drawctxt, list[i]);
kgsl_drawobj_destroy(list[i]);
}
debugfs_remove_recursive(drawctxt->debug_root);
/* The debugfs file has a reference, release it */
if (drawctxt->debug_root)
kgsl_context_put(context);
struct adreno_device *adreno_dev = ADRENO_DEVICE(device);
struct kgsl_context *context = &drawctxt->base;
int ret;
/*
* internal_timestamp is set in adreno_ringbuffer_addcmds,
@ -482,7 +443,7 @@ void adreno_drawctxt_detach(struct kgsl_context *context)
* commands to retire will be greater than 10s. 30s should be sufficient
* time to wait for the commands even if a hang happens.
*/
ret = adreno_drawctxt_wait_rb(adreno_dev, context,
ret = adreno_drawctxt_wait_rb(adreno_dev, &drawctxt->base,
drawctxt->internal_timestamp, 30 * 1000);
/*
@ -525,9 +486,62 @@ void adreno_drawctxt_detach(struct kgsl_context *context)
adreno_profile_process_results(adreno_dev);
mutex_unlock(&device->mutex);
}
if (gpudev->preemption_context_destroy)
gpudev->preemption_context_destroy(context);
void adreno_drawctxt_detach(struct kgsl_context *context)
{
struct kgsl_device *device;
struct adreno_device *adreno_dev;
struct adreno_gpudev *gpudev;
struct adreno_context *drawctxt;
int count, i;
struct kgsl_drawobj *list[ADRENO_CONTEXT_DRAWQUEUE_SIZE];
if (context == NULL)
return;
device = context->device;
adreno_dev = ADRENO_DEVICE(device);
gpudev = ADRENO_GPU_DEVICE(adreno_dev);
drawctxt = ADRENO_CONTEXT(context);
spin_lock(&drawctxt->lock);
if (!test_bit(GMU_DISPATCH, &device->gmu_core.flags)) {
spin_lock(&adreno_dev->active_list_lock);
list_del_init(&drawctxt->active_node);
spin_unlock(&adreno_dev->active_list_lock);
}
count = drawctxt_detach_drawobjs(drawctxt, list);
spin_unlock(&drawctxt->lock);
for (i = 0; i < count; i++) {
/*
* If the context is detached while we are waiting for
* the next command in GFT SKIP CMD, print the context
* detached status here.
*/
adreno_fault_skipcmd_detached(adreno_dev, drawctxt, list[i]);
kgsl_drawobj_destroy(list[i]);
}
debugfs_remove_recursive(drawctxt->debug_root);
/* The debugfs file has a reference, release it */
if (drawctxt->debug_root)
kgsl_context_put(context);
if (gpudev->context_detach)
gpudev->context_detach(drawctxt);
else
wait_for_timestamp_rb(device, drawctxt);
if (context->user_ctxt_record) {
gpumem_free_entry(context->user_ctxt_record);
/* Put the extra ref from gpumem_alloc_entry() */
kgsl_mem_entry_put(context->user_ctxt_record);
}
/* wake threads waiting to submit commands from this context */
wake_up_all(&drawctxt->waiting);

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,123 @@
/* SPDX-License-Identifier: GPL-2.0-only */
/*
* Copyright (c) 2020, The Linux Foundation. All rights reserved.
*/
#ifndef _ADRENO_HWSCHED_H_
#define _ADRENO_HWSCHED_H_
/**
* struct adreno_hwsched - Container for the hardware scheduler
*/
struct adreno_hwsched {
/** @mutex: Mutex needed to run dispatcher function */
struct mutex mutex;
/** @flags: Container for the dispatcher internal flags */
unsigned long flags;
/** @inflight: Number of active submissions to the dispatch queues */
u32 inflight;
/** @jobs - Array of dispatch job lists for each priority level */
struct llist_head jobs[16];
/** @requeue - Array of lists for dispatch jobs that got requeued */
struct llist_head requeue[16];
/** @work: The work structure to execute dispatcher function */
struct kthread_work work;
/** @cmd_list: List of objects submitted to dispatch queues */
struct list_head cmd_list;
/** @fault: Atomic to record a fault */
atomic_t fault;
};
enum adreno_hwsched_flags {
ADRENO_HWSCHED_POWER = 0,
ADRENO_HWSCHED_FAULT_RESTART,
ADRENO_HWSCHED_FAULT_REPLAY,
};
/**
* adreno_hwsched_trigger - Function to schedule the hwsched thread
* @adreno_dev: A handle to adreno device
*
* Schedule the hw dispatcher for retiring and submitting command objects
*/
void adreno_hwsched_trigger(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_queue_cmds() - Queue a new draw object in the context
* @dev_priv: Pointer to the device private struct
* @context: Pointer to the kgsl draw context
* @drawobj: Pointer to the array of drawobj's being submitted
* @count: Number of drawobj's being submitted
* @timestamp: Pointer to the requested timestamp
*
* Queue a command in the context - if there isn't any room in the queue, then
* block until there is
*
* Return: 0 on success and negative error on failure to queue
*/
int adreno_hwsched_queue_cmds(struct kgsl_device_private *dev_priv,
struct kgsl_context *context, struct kgsl_drawobj *drawobj[],
u32 count, u32 *timestamp);
/**
* adreno_hwsched_queue_context() - schedule a drawctxt in the hw dispatcher
* @device: pointer to the KGSL device
* @drawctxt: pointer to the drawctxt to schedule
*
* Put a draw context on the dispatcher job listse and schedule the
* dispatcher. This is used to reschedule changes that might have been blocked
* for sync points or other concerns
*/
void adreno_hwsched_queue_context(struct kgsl_device *device,
struct adreno_context *drawctxt);
/**
* adreno_hwsched_start() - activate the hwsched dispatcher
* @adreno_dev: pointer to the adreno device
*
* Enable dispatcher thread to execute
*/
void adreno_hwsched_start(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_dispatcher_init() - Initialize the hwsched dispatcher
* @adreno_dev: pointer to the adreno device
*
* Set up the dispatcher resources
*/
void adreno_hwsched_init(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_dispatcher_close() - close the hwsched dispatcher
* @adreno_dev: pointer to the adreno device structure
*
* Free the dispatcher resources
*/
void adreno_hwsched_dispatcher_close(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_set_fault - Set hwsched fault to request recovery
* @adreno_dev: A handle to adreno device
*/
void adreno_hwsched_set_fault(struct adreno_device *adreno_dev);
/**
* adreno_hwsched_mark_drawobj() - Get the drawobj that faulted
* @adreno_dev: pointer to the adreno device
* @ctxt_id: context id of the faulty submission
* @ts: timestamp of the faulty submission
*
* When we get a context bad hfi, use this function to get to the
* faulty submission and mark the submission for snapshot purposes
*/
void adreno_hwsched_mark_drawobj(struct adreno_device *adreno_dev, u32 ctxt_id,
u32 ts);
/**
* adreno_hwsched_parse_fault_ib - Parse the faulty submission
* @adreno_dev: pointer to the adreno device
* @snapshot: Pointer to the snapshot structure
*
* Walk the list of active submissions to find the one that faulted and
* parse it so that relevant command buffers can be added to the snapshot
*/
void adreno_hwsched_parse_fault_cmdobj(struct adreno_device *adreno_dev,
struct kgsl_snapshot *snapshot);
#endif

View file

@ -34,6 +34,9 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev,
{
struct adreno_gpudev *gpudev = ADRENO_GPU_DEVICE(adreno_dev);
unsigned long flags;
struct adreno_context *drawctxt = rb->drawctxt_active;
struct kgsl_context *context = &drawctxt->base;
/*
* Here we are attempting to create a mapping between the
* GPU time domain (alwayson counter) and the CPU time domain
@ -49,7 +52,8 @@ static void adreno_get_submit_time(struct adreno_device *adreno_dev,
time->ticks = gpudev->read_alwayson(adreno_dev);
/* Trace the GPU time to create a mapping to ftrace time */
trace_adreno_cmdbatch_sync(rb->drawctxt_active, time->ticks);
trace_adreno_cmdbatch_sync(context->id, context->priority,
drawctxt->timestamp, time->ticks);
/* Get the kernel clock for time since boot */
time->ktime = local_clock();
@ -111,50 +115,6 @@ static void adreno_ringbuffer_wptr(struct adreno_device *adreno_dev,
}
}
static void adreno_profile_submit_time(struct adreno_submit_time *time)
{
struct kgsl_drawobj *drawobj;
struct kgsl_drawobj_cmd *cmdobj;
struct kgsl_mem_entry *entry;
if (time == NULL)
return;
drawobj = time->drawobj;
if (drawobj == NULL)
return;
cmdobj = CMDOBJ(drawobj);
entry = cmdobj->profiling_buf_entry;
if (entry) {
struct kgsl_drawobj_profiling_buffer *profile_buffer;
profile_buffer = kgsl_gpuaddr_to_vaddr(&entry->memdesc,
cmdobj->profiling_buffer_gpuaddr);
if (profile_buffer == NULL)
return;
/* Return kernel clock time to the the client if requested */
if (drawobj->flags & KGSL_DRAWOBJ_PROFILING_KTIME) {
uint64_t secs = time->ktime;
profile_buffer->wall_clock_ns =
do_div(secs, NSEC_PER_SEC);
profile_buffer->wall_clock_s = secs;
} else {
profile_buffer->wall_clock_s = time->utime.tv_sec;
profile_buffer->wall_clock_ns = time->utime.tv_nsec;
}
profile_buffer->gpu_ticks_queued = time->ticks;
kgsl_memdesc_unmap(&entry->memdesc);
}
}
void adreno_ringbuffer_submit(struct adreno_ringbuffer *rb,
struct adreno_submit_time *time)
{

View file

@ -171,13 +171,7 @@ static int snapshot_freeze_obj_list(struct kgsl_snapshot *snapshot,
return ret;
}
/*
* We want to store the last executed IB1 and IB2 in the static region to ensure
* that we get at least some information out of the snapshot even if we can't
* access the dynamic data from the sysfs file. Push all other IBs on the
* dynamic list
*/
static inline void parse_ib(struct kgsl_device *device,
void adreno_parse_ib(struct kgsl_device *device,
struct kgsl_snapshot *snapshot,
struct kgsl_process_private *process,
uint64_t gpuaddr, uint64_t dwords)
@ -245,8 +239,8 @@ static void dump_all_ibs(struct kgsl_device *device,
ibaddr, ibsize))
continue;
parse_ib(device, snapshot, snapshot->process, ibaddr,
ibsize);
adreno_parse_ib(device, snapshot, snapshot->process,
ibaddr, ibsize);
} else
index = index + 1;
}
@ -398,7 +392,7 @@ static void snapshot_rb_ibs(struct kgsl_device *device,
ibaddr, ibsize))
continue;
parse_ib(device, snapshot, snapshot->process,
adreno_parse_ib(device, snapshot, snapshot->process,
ibaddr, ibsize);
} else
index = (index + 1) % KGSL_RB_DWORDS;
@ -561,7 +555,7 @@ static void kgsl_snapshot_add_active_ib_obj_list(struct kgsl_device *device,
index = find_object(snapshot->ib2base, snapshot->process);
if (index != -ENOENT)
parse_ib(device, snapshot, snapshot->process,
adreno_parse_ib(device, snapshot, snapshot->process,
snapshot->ib2base, objbuf[index].size >> 2);
}
}
@ -742,7 +736,7 @@ done:
}
/* Snapshot a global memory buffer */
static size_t snapshot_global(struct kgsl_device *device, u8 *buf,
size_t adreno_snapshot_global(struct kgsl_device *device, u8 *buf,
size_t remain, void *priv)
{
struct kgsl_memdesc *memdesc = priv;
@ -786,12 +780,12 @@ static void adreno_snapshot_iommu(struct kgsl_device *device,
struct kgsl_iommu *iommu = KGSL_IOMMU_PRIV(device);
kgsl_snapshot_add_section(device, KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, snapshot_global, iommu->setstate);
snapshot, adreno_snapshot_global, iommu->setstate);
if (ADRENO_FEATURE(adreno_dev, ADRENO_PREEMPTION))
kgsl_snapshot_add_section(device,
KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, snapshot_global, iommu->smmu_info);
snapshot, adreno_snapshot_global, iommu->smmu_info);
}
static void adreno_snapshot_ringbuffer(struct kgsl_device *device,
@ -830,13 +824,13 @@ void adreno_snapshot(struct kgsl_device *device, struct kgsl_snapshot *snapshot,
snapshot_frozen_objsize = 0;
setup_fault_process(device, snapshot,
context ? context->proc_priv : NULL);
/* Add GPU specific sections - registers mainly, but other stuff too */
if (gpudev->snapshot)
gpudev->snapshot(adreno_dev, snapshot);
setup_fault_process(device, snapshot,
context ? context->proc_priv : NULL);
snapshot->ib1dumped = false;
snapshot->ib2dumped = false;
@ -854,10 +848,10 @@ void adreno_snapshot(struct kgsl_device *device, struct kgsl_snapshot *snapshot,
/* Dump selected global buffers */
kgsl_snapshot_add_section(device, KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, snapshot_global, device->memstore);
snapshot, adreno_snapshot_global, device->memstore);
kgsl_snapshot_add_section(device, KGSL_SNAPSHOT_SECTION_GPU_OBJECT_V2,
snapshot, snapshot_global,
snapshot, adreno_snapshot_global,
adreno_dev->pwron_fixup);
if (kgsl_mmu_get_mmutype(device) == KGSL_MMU_TYPE_IOMMU)

View file

@ -1,6 +1,6 @@
/* SPDX-License-Identifier: GPL-2.0-only */
/*
* Copyright (c) 2013-2015,2019, The Linux Foundation. All rights reserved.
* Copyright (c) 2013-2015,2020, The Linux Foundation. All rights reserved.
*/
#ifndef __ADRENO_SNAPSHOT_H
#define __ADRENO_SNAPSHOT_H
@ -39,4 +39,32 @@ void adreno_snapshot_vbif_registers(struct kgsl_device *device,
const struct adreno_vbif_snapshot_registers *list,
unsigned int count);
/**
* adreno_parse_ib - Parse the given IB
* @device: Pointer to the kgsl device
* @snapshot: Pointer to the snapshot structure
* @process: Process to which this IB belongs
* @gpuaddr: Gpu address of the IB
* @dwords: Size in dwords of the IB
*
* We want to store the last executed IB1 and IB2 in the static region to ensure
* that we get at least some information out of the snapshot even if we can't
* access the dynamic data from the sysfs file. Push all other IBs on the
* dynamic list
*/
void adreno_parse_ib(struct kgsl_device *device,
struct kgsl_snapshot *snapshot,
struct kgsl_process_private *process,
u64 gpuaddr, u64 dwords);
/**
* adreno_snapshot_global - Add global buffer to snapshot
* @device: Pointer to the kgsl device
* @buf: Where the global buffer section is to be written
* @remain: Remaining bytes in snapshot buffer
* @priv: Opaque data
*
* Return: Number of bytes written to the snapshot buffer
*/
size_t adreno_snapshot_global(struct kgsl_device *device, u8 *buf,
size_t remain, void *priv);
#endif /*__ADRENO_SNAPSHOT_H */

View file

@ -1,6 +1,6 @@
/* SPDX-License-Identifier: GPL-2.0-only */
/*
* Copyright (c) 2013-2019, The Linux Foundation. All rights reserved.
* Copyright (c) 2013-2020, The Linux Foundation. All rights reserved.
*/
#if !defined(_ADRENO_TRACE_H) || defined(TRACE_HEADER_MULTI_READ)
@ -54,10 +54,10 @@ TRACE_EVENT(adreno_cmdbatch_queued,
);
TRACE_EVENT(adreno_cmdbatch_submitted,
TP_PROTO(struct kgsl_drawobj *drawobj, int inflight, uint64_t ticks,
unsigned long secs, unsigned long usecs,
struct adreno_ringbuffer *rb, unsigned int rptr),
TP_ARGS(drawobj, inflight, ticks, secs, usecs, rb, rptr),
TP_PROTO(struct kgsl_drawobj *drawobj, struct submission_info *info,
uint64_t ticks, unsigned long secs, unsigned long usecs,
int q_inflight),
TP_ARGS(drawobj, info, ticks, secs, usecs, q_inflight),
TP_STRUCT__entry(
__field(unsigned int, id)
__field(unsigned int, timestamp)
@ -71,39 +71,40 @@ TRACE_EVENT(adreno_cmdbatch_submitted,
__field(unsigned int, rptr)
__field(unsigned int, wptr)
__field(int, q_inflight)
__field(int, dispatch_queue)
),
TP_fast_assign(
__entry->id = drawobj->context->id;
__entry->timestamp = drawobj->timestamp;
__entry->inflight = inflight;
__entry->inflight = info->inflight;
__entry->flags = drawobj->flags;
__entry->ticks = ticks;
__entry->secs = secs;
__entry->usecs = usecs;
__entry->prio = drawobj->context->priority;
__entry->rb_id = rb->id;
__entry->rptr = rptr;
__entry->wptr = rb->wptr;
__entry->q_inflight = rb->dispatch_q.inflight;
__entry->rb_id = info->rb_id;
__entry->rptr = info->rptr;
__entry->wptr = info->wptr;
__entry->q_inflight = q_inflight;
__entry->dispatch_queue = info->gmu_dispatch_queue;
),
TP_printk(
"ctx=%u ctx_prio=%d ts=%u inflight=%d flags=%s ticks=%lld time=%lu.%0lu rb_id=%d r/w=%x/%x, q_inflight=%d",
"ctx=%u ctx_prio=%d ts=%u inflight=%d flags=%s ticks=%lld time=%lu.%0lu rb_id=%d r/w=%x/%x, q_inflight=%d dq_id=%d",
__entry->id, __entry->prio, __entry->timestamp,
__entry->inflight,
__entry->flags ? __print_flags(__entry->flags, "|",
KGSL_DRAWOBJ_FLAGS) : "none",
__entry->ticks, __entry->secs, __entry->usecs,
__entry->rb_id, __entry->rptr, __entry->wptr,
__entry->q_inflight
__entry->q_inflight, __entry->dispatch_queue
)
);
TRACE_EVENT(adreno_cmdbatch_retired,
TP_PROTO(struct kgsl_drawobj *drawobj, int inflight,
uint64_t start, uint64_t retire,
struct adreno_ringbuffer *rb, unsigned int rptr,
unsigned long fault_recovery),
TP_ARGS(drawobj, inflight, start, retire, rb, rptr, fault_recovery),
TP_PROTO(struct kgsl_context *context, struct retire_info *info,
unsigned int flags, int q_inflight,
unsigned long fault_recovery),
TP_ARGS(context, info, flags, q_inflight, fault_recovery),
TP_STRUCT__entry(
__field(unsigned int, id)
__field(unsigned int, timestamp)
@ -118,41 +119,50 @@ TRACE_EVENT(adreno_cmdbatch_retired,
__field(unsigned int, wptr)
__field(int, q_inflight)
__field(unsigned long, fault_recovery)
),
__field(unsigned int, dispatch_queue)
__field(uint64_t, submitted_to_rb)
__field(uint64_t, retired_on_gmu)
),
TP_fast_assign(
__entry->id = drawobj->context->id;
__entry->timestamp = drawobj->timestamp;
__entry->inflight = inflight;
__entry->id = context->id;
__entry->timestamp = info->timestamp;
__entry->inflight = info->inflight;
__entry->recovery = fault_recovery;
__entry->flags = drawobj->flags;
__entry->start = start;
__entry->retire = retire;
__entry->prio = drawobj->context->priority;
__entry->rb_id = rb->id;
__entry->rptr = rptr;
__entry->wptr = rb->wptr;
__entry->q_inflight = rb->dispatch_q.inflight;
),
__entry->flags = flags;
__entry->start = info->sop;
__entry->retire = info->eop;
__entry->prio = context->priority;
__entry->rb_id = info->rb_id;
__entry->rptr = info->rptr;
__entry->wptr = info->wptr;
__entry->q_inflight = q_inflight;
__entry->dispatch_queue = info->gmu_dispatch_queue;
__entry->submitted_to_rb = info->submitted_to_rb;
__entry->retired_on_gmu = info->retired_on_gmu;
),
TP_printk(
"ctx=%u ctx_prio=%d ts=%u inflight=%d recovery=%s flags=%s start=%lld retire=%lld rb_id=%d, r/w=%x/%x, q_inflight=%d",
"ctx=%u prio=%d ts=%u inflight=%d recovery=%s flags=%s start=%llu retire=%llu rb_id=%d, r/w=%x/%x, q_inflight=%d, dq_id=%u, submitted_to_rb=%llu, retired_on_gmu=%llu",
__entry->id, __entry->prio, __entry->timestamp,
__entry->inflight,
__entry->recovery ?
__print_flags(__entry->recovery, "|",
__print_flags(__entry->fault_recovery, "|",
ADRENO_FT_TYPES) : "none",
__entry->flags ? __print_flags(__entry->flags, "|",
KGSL_DRAWOBJ_FLAGS) : "none",
__entry->start,
__entry->retire,
__entry->rb_id, __entry->rptr, __entry->wptr,
__entry->q_inflight
)
__entry->q_inflight,
__entry->dispatch_queue,
__entry->submitted_to_rb, __entry->retired_on_gmu
)
);
TRACE_EVENT(adreno_cmdbatch_sync,
TP_PROTO(struct adreno_context *drawctxt,
uint64_t ticks),
TP_ARGS(drawctxt, ticks),
TP_PROTO(unsigned int ctx_id, unsigned int ctx_prio,
unsigned int timestamp, uint64_t ticks),
TP_ARGS(ctx_id, ctx_prio, timestamp, ticks),
TP_STRUCT__entry(
__field(unsigned int, id)
__field(unsigned int, timestamp)
@ -160,10 +170,10 @@ TRACE_EVENT(adreno_cmdbatch_sync,
__field(int, prio)
),
TP_fast_assign(
__entry->id = drawctxt->base.id;
__entry->timestamp = drawctxt->timestamp;
__entry->id = ctx_id;
__entry->timestamp = timestamp;
__entry->ticks = ticks;
__entry->prio = drawctxt->base.priority;
__entry->prio = ctx_prio;
),
TP_printk(
"ctx=%u ctx_prio=%d ts=%u ticks=%lld",

View file

@ -349,6 +349,48 @@ struct sparse_bind_object {
uint64_t flags;
};
/**
* struct submission_info - Container for submission statistics
* @inflight: Number of commands that are inflight
* @rb_id: id of the ringbuffer to which this submission is made
* @rptr: Read pointer of the ringbuffer
* @wptr: Write pointer of the ringbuffer
* @gmu_dispatch_queue: GMU dispach queue to which this submission is made
*/
struct submission_info {
int inflight;
u32 rb_id;
u32 rptr;
u32 wptr;
u32 gmu_dispatch_queue;
};
/**
* struct retire_info - Container for retire statistics
* @inflight: NUmber of commands that are inflight
* @rb_id: id of the ringbuffer to which this submission is made
* @rptr: Read pointer of the ringbuffer
* @wptr: Write pointer of the ringbuffer
* @gmu_dispatch_queue: GMU dispach queue to which this submission is made
* @timestamp: Timestamp of submission that retired
* @submitted_to_rb: AO ticks when GMU put this submission on ringbuffer
* @sop: AO ticks when GPU started procssing this submission
* @eop: AO ticks when GPU finished this submission
* @retired_on_gmu: AO ticks when GMU retired this submission
*/
struct retire_info {
int inflight;
int rb_id;
u32 rptr;
u32 wptr;
u32 gmu_dispatch_queue;
u32 timestamp;
u64 submitted_to_rb;
u64 sop;
u64 eop;
u64 retired_on_gmu;
};
long kgsl_ioctl_device_getproperty(struct kgsl_device_private *dev_priv,
unsigned int cmd, void *data);
long kgsl_ioctl_device_setproperty(struct kgsl_device_private *dev_priv,

View file

@ -397,6 +397,11 @@ struct kgsl_context {
unsigned int total_fault_count;
unsigned int last_faulted_cmd_ts;
bool gmu_registered;
/**
* @gmu_dispatch_queue: dispatch queue id to which this context will be
* submitted
*/
u32 gmu_dispatch_queue;
};
#define _context_comm(_c) \

View file

@ -1,6 +1,6 @@
/* SPDX-License-Identifier: GPL-2.0-only */
/*
* Copyright (c) 2016-2019, The Linux Foundation. All rights reserved.
* Copyright (c) 2016-2020, The Linux Foundation. All rights reserved.
*/
#ifndef __KGSL_DRAWOBJ_H
@ -165,6 +165,7 @@ struct kgsl_drawobj_sparse {
* command obj
* @CMDOBJ_WFI - Force wait-for-idle for the submission
* @CMDOBJ_PROFILE - store the start / retire ticks for
* @CMDOBJ_FAULT - Mark the command object as faulted
* the command obj in the profiling buffer
*/
enum kgsl_drawobj_cmd_priv {
@ -172,6 +173,7 @@ enum kgsl_drawobj_cmd_priv {
CMDOBJ_FORCE_PREAMBLE,
CMDOBJ_WFI,
CMDOBJ_PROFILE,
CMDOBJ_FAULT,
};
struct kgsl_ibdesc;

View file

@ -40,6 +40,7 @@ enum gmu_core_flags {
GMU_ENABLED,
GMU_RSCC_SLEEP_SEQ_DONE,
GMU_DISABLE_SLUMBER,
GMU_DISPATCH,
};
/*

View file

@ -2030,7 +2030,7 @@ static int _suspend(struct kgsl_device *device)
/* drain to prevent from more commands being submitted */
device->ftbl->drain(device);
/* wait for active count so device can be put in slumber */
ret = kgsl_active_count_wait(device, 0);
ret = kgsl_active_count_wait(device, 0, HZ);
if (ret)
goto err;
@ -2159,17 +2159,10 @@ static int _check_active_count(struct kgsl_device *device, int count)
return atomic_read(&device->active_cnt) > count ? 0 : 1;
}
/**
* kgsl_active_count_wait() - Wait for activity to finish.
* @device: Pointer to a KGSL device
* @count: Active count value to wait for
*
* Block until the active_cnt value hits the desired value
*/
int kgsl_active_count_wait(struct kgsl_device *device, int count)
int kgsl_active_count_wait(struct kgsl_device *device, int count,
unsigned long wait_jiffies)
{
int result = 0;
long wait_jiffies = HZ;
if (WARN_ON(!mutex_is_locked(&device->mutex)))
return -EINVAL;

View file

@ -198,7 +198,16 @@ kgsl_pwrctrl_active_freq(struct kgsl_pwrctrl *pwr)
return pwr->pwrlevels[pwr->active_pwrlevel].gpu_freq;
}
int kgsl_active_count_wait(struct kgsl_device *device, int count);
/**
* kgsl_active_count_wait() - Wait for activity to finish.
* @device: Pointer to a KGSL device
* @count: Active count value to wait for
* @wait_jiffies: Jiffies to wait
*
* Block until the active_cnt value hits the desired value
*/
int kgsl_active_count_wait(struct kgsl_device *device, int count,
unsigned long wait_jiffies);
void kgsl_pwrctrl_busy_time(struct kgsl_device *device, u64 time, u64 busy);
void kgsl_pwrctrl_set_constraint(struct kgsl_device *device,
struct kgsl_pwr_constraint *pwrc, uint32_t id);