diff --git a/drivers/gpu/msm/Kconfig b/drivers/gpu/msm/Kconfig index 5f87cba2c9ea..e573fd8a0ba2 100644 --- a/drivers/gpu/msm/Kconfig +++ b/drivers/gpu/msm/Kconfig @@ -41,3 +41,13 @@ config QCOM_KGSL_CORESIGHT When enabled, the Adreno GPU is available as a source for Coresight data. On a6xx targets there are two sources available for the GX and CX domains respectively. Debug kernels should say 'Y' here. + +config QCOM_KGSL_IOCOHERENCY_DEFAULT + bool "Enable I/O coherency on cached GPU memory by default" + depends on QCOM_KGSL + default y if ARCH_LAHAINA + help + Say 'Y' here to enable I/O cache coherency by default on targets that + support hardware I/O coherency. If enabled all cached GPU memory + will use I/O coherency regardless of the user flags. If not enabled + the user can still selectively enable I/O coherency with a flag. diff --git a/drivers/gpu/msm/Makefile b/drivers/gpu/msm/Makefile index 808a9817c9a0..4c3f64d41c97 100644 --- a/drivers/gpu/msm/Makefile +++ b/drivers/gpu/msm/Makefile @@ -5,6 +5,7 @@ obj-$(CONFIG_QCOM_KGSL) += msm_kgsl.o msm_kgsl-y = \ kgsl.o \ + kgsl_bus.o \ kgsl_drawobj.o \ kgsl_events.o \ kgsl_ioctl.o \ diff --git a/drivers/gpu/msm/adreno.c b/drivers/gpu/msm/adreno.c index 4f866925e423..d5b56a712e1e 100644 --- a/drivers/gpu/msm/adreno.c +++ b/drivers/gpu/msm/adreno.c @@ -22,6 +22,7 @@ #include "adreno_compat.h" #include "adreno_iommu.h" #include "adreno_trace.h" +#include "kgsl_bus.h" #include "kgsl_trace.h" #include "kgsl_util.h" @@ -1450,6 +1451,12 @@ static int adreno_probe(struct platform_device *pdev) return status; } + status = kgsl_bus_init(device, pdev); + if (status) { + device->pdev = NULL; + return status; + } + /* * Probe/init GMU after initial gpu power probe * Another part of GPU power probe in platform_probe diff --git a/drivers/gpu/msm/governor_gpubw_mon.c b/drivers/gpu/msm/governor_gpubw_mon.c index 2f75049f1234..f133b37fb4ca 100644 --- a/drivers/gpu/msm/governor_gpubw_mon.c +++ b/drivers/gpu/msm/governor_gpubw_mon.c @@ -5,11 +5,11 @@ #include #include -#include #include #include "devfreq_trace.h" -#include "governor.h" +#include "../../devfreq/governor.h" +#include "msm_adreno_devfreq.h" #define MIN_BUSY 1000 #define LONG_FLOOR 50000 @@ -95,7 +95,7 @@ static int devfreq_gpubw_get_target(struct devfreq *df, } else { /* GPU votes for IB not AB so don't under vote the system */ norm_cycles = (100 * norm_cycles) / TARGET; - act_level = b.buslevel + b.mod; + act_level = b.buslevel; act_level = (act_level < 0) ? 0 : act_level; act_level = (act_level >= priv->bus.num) ? (priv->bus.num - 1) : act_level; @@ -157,16 +157,20 @@ static int gpubw_start(struct devfreq *devfreq) /* Set up the cut-over percentages for the bus calculation. */ for (i = 0; i < priv->bus.num; i++) { - t1 = (u32)(100 * priv->bus.ib[i]) / - (u32)priv->bus.ib[priv->bus.num - 1]; + t1 = (u32)(100 * priv->bus.ib_kbps[i]) / + (u32)priv->bus.ib_kbps[priv->bus.num - 1]; priv->bus.p_up[i] = t1 - HIST; priv->bus.p_down[i] = t2 - 2 * HIST; t2 = t1; } /* Set the upper-most and lower-most bounds correctly. */ priv->bus.p_down[0] = 0; - priv->bus.p_down[1] = (priv->bus.p_down[1] > (2 * HIST)) ? - priv->bus.p_down[1] : (2 * HIST); + + for (i = 0; i < priv->bus.num; i++) { + if (priv->bus.p_down[i] < 2 * HIST) + priv->bus.p_down[i] = 2 * HIST; + } + if (priv->bus.num >= 1) priv->bus.p_up[priv->bus.num - 1] = 100; _update_cutoff(priv, priv->bus.max); diff --git a/drivers/gpu/msm/governor_msm_adreno_tz.c b/drivers/gpu/msm/governor_msm_adreno_tz.c index 8ecd1efe9fc7..0a363218112f 100644 --- a/drivers/gpu/msm/governor_msm_adreno_tz.c +++ b/drivers/gpu/msm/governor_msm_adreno_tz.c @@ -5,17 +5,19 @@ #include #include #include +#include #include #include #include #include #include #include -#include #include #include -#include -#include "governor.h" +#include + +#include "../../devfreq/governor.h" +#include "msm_adreno_devfreq.h" static DEFINE_SPINLOCK(tz_lock); static DEFINE_SPINLOCK(sample_lock); @@ -197,7 +199,8 @@ static int __secure_tz_update_entry3(int level, s64 total_time, s64 busy_time, return ret; } -static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) +static int tz_init_ca(struct device *dev, + struct devfreq_msm_adreno_tz_data *priv) { unsigned int tz_ca_data[2]; phys_addr_t paddr; @@ -226,8 +229,8 @@ static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) memcpy(tz_buf, tz_ca_data, sizeof(tz_ca_data)); /* Ensure memcpy completes execution */ mb(); - dmac_flush_range(tz_buf, - tz_buf + PAGE_ALIGN(sizeof(tz_ca_data))); + dma_sync_single_for_device(dev, paddr, + PAGE_ALIGN(sizeof(tz_ca_data)), DMA_BIDIRECTIONAL); ret = qcom_scm_dcvs_init_ca_v2(paddr, sizeof(tz_ca_data)); @@ -239,7 +242,7 @@ static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) return ret; } -static int tz_init(struct devfreq_msm_adreno_tz_data *priv, +static int tz_init(struct device *dev, struct devfreq_msm_adreno_tz_data *priv, unsigned int *tz_pwrlevels, u32 size_pwrlevels, unsigned int *version, u32 size_version) { @@ -268,7 +271,8 @@ static int tz_init(struct devfreq_msm_adreno_tz_data *priv, memcpy(tz_buf, tz_pwrlevels, size_pwrlevels); /* Ensure memcpy completes execution */ mb(); - dmac_flush_range(tz_buf, tz_buf + PAGE_ALIGN(size_pwrlevels)); + dma_sync_single_for_device(dev, paddr, + PAGE_ALIGN(size_pwrlevels), DMA_BIDIRECTIONAL); ret = qcom_scm_dcvs_init_v2(paddr, size_pwrlevels, version); if (!ret) @@ -283,7 +287,7 @@ static int tz_init(struct devfreq_msm_adreno_tz_data *priv, /* Initialize context aware feature, if enabled. */ if (!ret && priv->ctxt_aware_enable) { if (priv->is_64 && qcom_scm_dcvs_ca_available()) { - ret = tz_init_ca(priv); + ret = tz_init_ca(dev, priv); /* * If context aware feature initialization fails, * just print an error message and return @@ -455,8 +459,8 @@ static int tz_start(struct devfreq *devfreq) INIT_WORK(&gpu_profile->partner_resume_event_ws, do_partner_resume_event); - ret = tz_init(priv, tz_pwrlevels, sizeof(tz_pwrlevels), &version, - sizeof(version)); + ret = tz_init(&devfreq->dev, priv, tz_pwrlevels, sizeof(tz_pwrlevels), + &version, sizeof(version)); if (ret != 0 || version > MAX_TZ_VERSION) { pr_err(TAG "tz_init failed\n"); return ret; diff --git a/drivers/gpu/msm/kgsl.c b/drivers/gpu/msm/kgsl.c index 9975e1cef2f4..befe80600e6a 100644 --- a/drivers/gpu/msm/kgsl.c +++ b/drivers/gpu/msm/kgsl.c @@ -2391,6 +2391,14 @@ static void _setup_cache_mode(struct kgsl_mem_entry *entry, entry->memdesc.flags |= (mode << KGSL_CACHEMODE_SHIFT); } +static bool is_cached(u64 flags) +{ + u32 mode = (flags & KGSL_CACHEMODE_MASK) >> KGSL_CACHEMODE_SHIFT; + + return (mode != KGSL_CACHEMODE_UNCACHED && + mode != KGSL_CACHEMODE_WRITECOMBINE); +} + static int kgsl_setup_dma_buf(struct kgsl_device *device, struct kgsl_pagetable *pagetable, struct kgsl_mem_entry *entry, @@ -2455,6 +2463,11 @@ static int kgsl_setup_dmabuf_useraddr(struct kgsl_device *device, /* Setup the user addr/cache mode for cache operations */ entry->memdesc.useraddr = hostptr; _setup_cache_mode(entry, vma); + + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(entry->memdesc.flags)) + entry->memdesc.flags |= KGSL_MEMFLAGS_IOCOHERENT; + up_read(¤t->mm->mmap_sem); return 0; } @@ -2979,7 +2992,6 @@ static int _kgsl_gpumem_sync_cache(struct kgsl_mem_entry *entry, { int ret = 0; int cacheop; - int mode; if (!entry) return 0; @@ -3010,9 +3022,7 @@ static int _kgsl_gpumem_sync_cache(struct kgsl_mem_entry *entry, length = entry->memdesc.size; } - mode = kgsl_memdesc_get_cachemode(&entry->memdesc); - if (mode != KGSL_CACHEMODE_UNCACHED - && mode != KGSL_CACHEMODE_WRITECOMBINE) { + if (is_cached(entry->memdesc.flags)) { trace_kgsl_mem_sync_cache(entry, offset, length, op); ret = kgsl_cache_range_op(&entry->memdesc, offset, length, cacheop); @@ -3319,6 +3329,10 @@ struct kgsl_mem_entry *gpumem_alloc_entry( if (entry == NULL) return ERR_PTR(-ENOMEM); + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(flags)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + ret = kgsl_allocate_user(dev_priv->device, &entry->memdesc, size, flags, 0); if (ret != 0) @@ -3540,6 +3554,11 @@ long kgsl_ioctl_sparse_phys_alloc(struct kgsl_device_private *dev_priv, ((ilog2(param->pagesize) << KGSL_MEMALIGN_SHIFT) & KGSL_MEMALIGN_MASK); + + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(flags)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + ret = kgsl_allocate_user(dev_priv->device, &entry->memdesc, param->size, flags, 0); if (ret) diff --git a/drivers/gpu/msm/kgsl_bus.c b/drivers/gpu/msm/kgsl_bus.c new file mode 100644 index 000000000000..106961e5604c --- /dev/null +++ b/drivers/gpu/msm/kgsl_bus.c @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019, The Linux Foundation. All rights reserved. + */ + +#include +#include + +#include "kgsl_bus.h" +#include "kgsl_device.h" +#include "kgsl_trace.h" + +static int gmu_bus_set(struct kgsl_device *device, int buslevel, + u32 ab) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int ret; + + ret = gmu_core_dcvs_set(device, INVALID_DCVS_IDX, buslevel); + + if (!ret) + icc_set_bw(pwr->icc_path, MBps_to_icc(ab), 0); + + return ret; +} + +static int interconnect_bus_set(struct kgsl_device *device, int level, + u32 ab) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + + icc_set_bw(pwr->icc_path, MBps_to_icc(ab), + kBps_to_icc(pwr->ddr_table[level])); + + return 0; +} + +static u32 _ab_buslevel_update(struct kgsl_pwrctrl *pwr, + u32 ib) +{ + if (!ib) + return 0; + + /* In the absence of any other settings, make ab 25% of ib */ + if ((!pwr->bus_percent_ab) && (!pwr->bus_ab_mbytes)) + return 25 * ib / 100; + + if (pwr->bus_width) + return pwr->bus_ab_mbytes; + + return (pwr->bus_percent_ab * pwr->bus_max) / 100; +} + + +void kgsl_bus_update(struct kgsl_device *device, bool on) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + /* FIXME: this might be wrong? */ + int cur = pwr->pwrlevels[pwr->active_pwrlevel].bus_freq; + int buslevel = 0; + u32 ab; + + /* the bus should be ON to update the active frequency */ + if (on && !(test_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags))) + return; + /* + * If the bus should remain on calculate our request and submit it, + * otherwise request bus level 0, off. + */ + if (on) { + buslevel = min_t(int, pwr->pwrlevels[0].bus_max, + cur + pwr->bus_mod); + buslevel = max_t(int, buslevel, 1); + } else { + /* If the bus is being turned off, reset to default level */ + pwr->bus_mod = 0; + pwr->bus_percent_ab = 0; + pwr->bus_ab_mbytes = 0; + } + trace_kgsl_buslevel(device, pwr->active_pwrlevel, buslevel); + pwr->cur_buslevel = buslevel; + + /* buslevel is the IB vote, update the AB */ + ab = _ab_buslevel_update(pwr, pwr->ddr_table[buslevel]); + + pwr->bus_set(device, buslevel, ab); +} + +static void validate_pwrlevels(struct kgsl_device *device, u32 *ibs, + int count) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int i; + + for (i = 0; i < pwr->num_pwrlevels - 1; i++) { + struct kgsl_pwrlevel *pwrlevel = &pwr->pwrlevels[i]; + + if (pwrlevel->bus_freq >= count) { + dev_err(device->dev, "Bus setting for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_freq = count - 1; + } + + if (pwrlevel->bus_max >= count) { + dev_err(device->dev, "Bus max for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_max = count - 1; + } + + if (pwrlevel->bus_min >= count) { + dev_err(device->dev, "Bus min for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_min = count - 1; + } + + if (pwrlevel->bus_min > pwrlevel->bus_max) { + dev_err(device->dev, "Bus min is bigger than bus max for GPU freq %d\n", + pwrlevel->gpu_freq); + pwrlevel->bus_min = pwrlevel->bus_max; + } + } +} + +u32 *kgsl_bus_get_table(struct platform_device *pdev, + const char *name, int *count) +{ + u32 *levels; + int i, num = of_property_count_elems_of_size(pdev->dev.of_node, + name, sizeof(u32)); + + /* If the bus wasn't specified, then build a static table */ + if (num <= 0) + return ERR_PTR(-EINVAL); + + levels = kcalloc(num, sizeof(*levels), GFP_KERNEL); + if (!levels) + return ERR_PTR(-ENOMEM); + + for (i = 0; i < num; i++) + of_property_read_u32_index(pdev->dev.of_node, + name, i, &levels[i]); + + *count = num; + return levels; +} + +int kgsl_bus_init(struct kgsl_device *device, struct platform_device *pdev) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int ret, count; + + pwr->ddr_table = kgsl_bus_get_table(pdev, "qcom,bus-table-ddr", &count); + if (IS_ERR(pwr->ddr_table)) { + ret = PTR_ERR(pwr->ddr_table); + pwr->ddr_table = NULL; + return ret; + } + + pwr->ddr_table_count = count; + + validate_pwrlevels(device, pwr->ddr_table, pwr->ddr_table_count); + + pwr->icc_path = of_icc_get(&pdev->dev, NULL); + if (IS_ERR(pwr->icc_path) && !gmu_core_scales_bandwidth(device)) { + WARN(1, "The CPU has no way to set the GPU bus levels\n"); + return PTR_ERR(pwr->icc_path); + } + + if (gmu_core_scales_bandwidth(device)) + pwr->bus_set = gmu_bus_set; + else + pwr->bus_set = interconnect_bus_set; + + return 0; +} + +void kgsl_bus_close(struct kgsl_device *device) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + + kfree(pwr->ddr_table); + + /* FIXME: Make sure icc put can handle NULL or IS_ERR */ + icc_put(pwr->icc_path); +} diff --git a/drivers/gpu/msm/kgsl_bus.h b/drivers/gpu/msm/kgsl_bus.h new file mode 100644 index 000000000000..2329cce8000b --- /dev/null +++ b/drivers/gpu/msm/kgsl_bus.h @@ -0,0 +1,19 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019, The Linux Foundation. All rights reserved. + */ + +#ifndef _KGSL_BUS_H +#define _KGSL_BUS_H + +struct kgsl_device; +struct platform_device; + +int kgsl_bus_init(struct kgsl_device *device, struct platform_device *pdev); +void kgsl_bus_close(struct kgsl_device *device); +void kgsl_bus_update(struct kgsl_device *device, bool on); + +u32 *kgsl_bus_get_table(struct platform_device *pdev, + const char *name, int *count); + +#endif diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index c6d9536b947c..6db4433fd584 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -7,15 +7,17 @@ #include #include #include +#include #include #include -#include #include #include #include #include +#include #include "adreno.h" +#include "kgsl_bus.h" #include "kgsl_device.h" #include "kgsl_gmu.h" #include "kgsl_util.h" @@ -490,6 +492,7 @@ static int gmu_dcvs_set(struct kgsl_device *device, int gpu_pwrlevel, int bus_level) { int ret = 0; + struct kgsl_pwrctrl *pwr = &device->pwrctrl; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); struct adreno_device *adreno_dev = ADRENO_DEVICE(device); struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); @@ -515,7 +518,7 @@ static int gmu_dcvs_set(struct kgsl_device *device, if (gpu_pwrlevel < gmu->num_gpupwrlevels - 1) req.freq = gmu->num_gpupwrlevels - gpu_pwrlevel - 1; - if (bus_level < gmu->num_bwlevels && bus_level > 0) + if (bus_level < pwr->ddr_table_count && bus_level > 0) req.bw = bus_level; /* GMU will vote for slumber levels through the sleep sequence */ @@ -698,7 +701,7 @@ static int rpmh_arc_votes_init(struct kgsl_device *device, num_freqs = gmu->num_gpupwrlevels; - if (num_freqs > pri_rail->num) { + if (num_freqs > pri_rail->num || num_freqs > ARRAY_SIZE(vlvl_tbl)) { dev_err(&gmu->pdev->dev, "Defined more GPU DCVS levels than RPMh can support\n"); return -EINVAL; @@ -712,129 +715,260 @@ static int rpmh_arc_votes_init(struct kgsl_device *device, sec_rail, vlvl_tbl, num_freqs); } -/* - * build_rpmh_bw_votes() - build TCS commands to vote for bandwidth. - * Each command sets frequency of a node along path to DDR or CNOC. - * @rpmh_vote: Pointer to RPMh vote needed by GMU to set BW via RPMh - * @num_usecases: Number of BW use cases (or BW levels) - * @handle: Provided by bus driver. It contains TCS command sets for - * all BW use cases of a bus client. - */ -static void build_rpmh_bw_votes(struct gmu_bw_votes *rpmh_vote, - unsigned int num_usecases, struct msm_bus_tcs_handle handle) -{ - struct msm_bus_tcs_usecase *tmp; - int i, j; +struct bcm { + const char *name; + u32 buswidth; + u32 channels; + u32 unit; + u16 width; + u8 vcd; + bool fixed; +}; - for (i = 0; i < num_usecases; i++) { - tmp = &handle.usecases[i]; - for (j = 0; j < tmp->num_cmds; j++) { - if (!i) { - /* - * Wait bitmask and TCS command addresses are - * same for all bw use cases. To save data volume - * exchanged between driver and GMU, only - * transfer bitmasks and TCS command addresses - * of first set of bw use case - */ - rpmh_vote->cmds_per_bw_vote = tmp->num_cmds; - rpmh_vote->cmds_wait_bitmask = - tmp->cmds[j].wait ? - rpmh_vote->cmds_wait_bitmask - | BIT(i) - : rpmh_vote->cmds_wait_bitmask - & (~BIT(i)); - rpmh_vote->cmd_addrs[j] = tmp->cmds[j].addr; - } - rpmh_vote->cmd_data[i][j] = tmp->cmds[j].data; +/* + * List of Bus Control Modules (BCMs) that need to be configured for the GPU + * to access DDR. For each bus level we will generate a vote each BC + */ +static struct bcm a660_ddr_bcms[] = { + { .name = "SH0", .buswidth = 16 }, + { .name = "MC0", .buswidth = 4 }, + { .name = "ACV", .fixed = true }, +}; + +/* Same as above, but for the CNOC BCMs */ +static struct bcm a660_cnoc_bcms[] = { + { .name = "CN0", .buswidth = 4 }, +}; + +/* Generate a set of bandwidth votes for the list of BCMs */ +static void tcs_cmd_data(struct bcm *bcms, int count, u32 ab, u32 ib, + u32 *data) +{ + int i; + + for (i = 0; i < count; i++) { + bool valid = true; + bool commit = false; + u64 avg, peak, x, y; + + if (i == count - 1 || bcms[i].vcd != bcms[i + 1].vcd) + commit = true; + + /* + * On a660, the "ACV" y vote should be 0x08 if there is a valid + * vote and 0x00 if not. This is kind of hacky and a660 specific + * but we can clean it up when we add a new target + */ + if (bcms[i].fixed) { + if (!ab && !ib) + data[i] = BCM_TCS_CMD(commit, false, 0x0, 0x0); + else + data[i] = BCM_TCS_CMD(commit, true, 0x0, 0x8); + continue; } + + /* Multiple the bandwidth by the width of the connection */ + avg = ((u64) ab) * bcms[i].width; + + /* And then divide by the total width across channels */ + do_div(avg, bcms[i].buswidth * bcms[i].channels); + + peak = ((u64) ib) * bcms[i].width; + do_div(peak, bcms[i].buswidth); + + /* Input bandwidth value is in KBps */ + x = avg * 1000ULL; + do_div(x, bcms[i].unit); + + /* Input bandwidth value is in KBps */ + y = peak * 1000ULL; + do_div(y, bcms[i].unit); + + /* + * If a bandwidth value was specified but the calculation ends + * rounding down to zero, set a minimum level + */ + if (ab && x == 0) + x = 1; + + if (ib && y == 0) + y = 1; + + x = min_t(u64, x, BCM_TCS_CMD_VOTE_MASK); + y = min_t(u64, y, BCM_TCS_CMD_VOTE_MASK); + + if (!x && !y) + valid = false; + + data[i] = BCM_TCS_CMD(commit, valid, x, y); } } -static void build_bwtable_cmd_cache(struct gmu_device *gmu) +struct bcm_data { + __le32 unit; + __le16 width; + u8 vcd; + u8 reserved; +}; + +struct rpmh_bw_votes { + u32 wait_bitmask; + u32 num_cmds; + u32 *addrs; + u32 num_levels; + u32 **cmds; +}; + +static void free_rpmh_bw_votes(struct rpmh_bw_votes *votes) +{ + int i; + + if (!votes) + return; + + for (i = 0; votes->cmds && i < votes->num_levels; i++) + kfree(votes->cmds[i]); + + kfree(votes->cmds); + kfree(votes->addrs); + kfree(votes); +} + +/* Build the votes table from the specified bandwidth levels */ +static struct rpmh_bw_votes *build_rpmh_bw_votes(struct bcm *bcms, + int bcm_count, u32 *levels, int levels_count) +{ + struct rpmh_bw_votes *votes; + int i; + + votes = kzalloc(sizeof(*votes), GFP_KERNEL); + if (!votes) + return ERR_PTR(-ENOMEM); + + votes->addrs = kcalloc(bcm_count, sizeof(*votes->cmds), GFP_KERNEL); + if (!votes->addrs) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + votes->cmds = kcalloc(levels_count, sizeof(*votes->cmds), GFP_KERNEL); + if (!votes->cmds) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + votes->num_cmds = bcm_count; + votes->num_levels = levels_count; + + /* Get the cmd-db information for each BCM */ + for (i = 0; i < bcm_count; i++) { + size_t l; + const struct bcm_data *data; + + data = cmd_db_read_aux_data(bcms[i].name, &l); + + votes->addrs[i] = cmd_db_read_addr(bcms[i].name); + + bcms[i].unit = le32_to_cpu(data->unit); + bcms[i].width = le16_to_cpu(data->width); + bcms[i].vcd = data->vcd; + } + + for (i = 0; i < bcm_count; i++) { + if (i == (bcm_count - 1) || bcms[i].vcd != bcms[i + 1].vcd) + votes->wait_bitmask |= (1 << i); + } + + for (i = 0; i < levels_count; i++) { + votes->cmds[i] = kcalloc(bcm_count, sizeof(u32), GFP_KERNEL); + if (!votes->cmds[i]) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + tcs_cmd_data(bcms, bcm_count, 0, levels[i], votes->cmds[i]); + } + + return votes; +} + +static void build_bwtable_cmd_cache(struct hfi_bwtable_cmd *cmd, + struct rpmh_bw_votes *ddr, struct rpmh_bw_votes *cnoc) { - struct hfi_bwtable_cmd *cmd = &gmu->hfi.bwtbl_cmd; - struct rpmh_votes_t *votes = &gmu->rpmh_votes; unsigned int i, j; cmd->hdr = 0xFFFFFFFF; - cmd->bw_level_num = gmu->num_bwlevels; - cmd->cnoc_cmds_num = votes->cnoc_votes.cmds_per_bw_vote; - cmd->cnoc_wait_bitmask = votes->cnoc_votes.cmds_wait_bitmask; - cmd->ddr_cmds_num = votes->ddr_votes.cmds_per_bw_vote; - cmd->ddr_wait_bitmask = votes->ddr_votes.cmds_wait_bitmask; + cmd->bw_level_num = ddr->num_levels; + cmd->ddr_cmds_num = ddr->num_cmds; + cmd->ddr_wait_bitmask = ddr->wait_bitmask; - for (i = 0; i < cmd->ddr_cmds_num; i++) - cmd->ddr_cmd_addrs[i] = votes->ddr_votes.cmd_addrs[i]; + for (i = 0; i < ddr->num_cmds; i++) + cmd->ddr_cmd_addrs[i] = ddr->addrs[i]; - for (i = 0; i < cmd->bw_level_num; i++) - for (j = 0; j < cmd->ddr_cmds_num; j++) - cmd->ddr_cmd_data[i][j] = - votes->ddr_votes.cmd_data[i][j]; + for (i = 0; i < ddr->num_levels; i++) + for (j = 0; j < ddr->num_cmds; j++) + cmd->ddr_cmd_data[i][j] = (u32) ddr->cmds[i][j]; - for (i = 0; i < cmd->cnoc_cmds_num; i++) - cmd->cnoc_cmd_addrs[i] = - votes->cnoc_votes.cmd_addrs[i]; + if (!cnoc) + return; - for (i = 0; i < MAX_CNOC_LEVELS; i++) - for (j = 0; j < cmd->cnoc_cmds_num; j++) - cmd->cnoc_cmd_data[i][j] = - votes->cnoc_votes.cmd_data[i][j]; + cmd->cnoc_cmds_num = cnoc->num_cmds; + cmd->cnoc_wait_bitmask = cnoc->wait_bitmask; + + for (i = 0; i < cnoc->num_cmds; i++) + cmd->cnoc_cmd_addrs[i] = cnoc->addrs[i]; + + for (i = 0; i < cnoc->num_levels; i++) + for (j = 0; j < cnoc->num_cmds; j++) + cmd->cnoc_cmd_data[i][j] = (u32) cnoc->cmds[i][j]; } -/* - * gmu_bus_vote_init - initialized RPMh votes needed for bw scaling by GMU. - * @gmu: Pointer to GMU device - * @pwr: Pointer to KGSL power controller - */ -static int gmu_bus_vote_init(struct gmu_device *gmu, struct kgsl_pwrctrl *pwr) +static int gmu_bus_vote_init(struct kgsl_device *device) { - struct msm_bus_tcs_usecase *usecases; - struct msm_bus_tcs_handle hdl; - struct rpmh_votes_t *votes = &gmu->rpmh_votes; - int ret; + struct gmu_device *gmu = KGSL_GMU_DEVICE(device); + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + struct rpmh_bw_votes *ddr, *cnoc = NULL; + u32 *cnoc_table; + u32 count; - usecases = kcalloc(gmu->num_bwlevels, sizeof(*usecases), GFP_KERNEL); - if (!usecases) - return -ENOMEM; + /* Build the DDR votes */ + ddr = build_rpmh_bw_votes(a660_ddr_bcms, ARRAY_SIZE(a660_ddr_bcms), + pwr->ddr_table, pwr->ddr_table_count); + if (IS_ERR(ddr)) + return PTR_ERR(ddr); - hdl.num_usecases = gmu->num_bwlevels; - hdl.usecases = usecases; + /* Get the CNOC table */ + cnoc_table = kgsl_bus_get_table(device->pdev, "qcom,bus-table-cnoc", + &count); - /* - * Query TCS command set for each use case defined in GPU b/w table - */ - ret = msm_bus_scale_query_tcs_cmd_all(&hdl, gmu->pcl); - if (ret) - goto out; + /* And build the votes for that, if it exists */ + if (count > 0) + cnoc = build_rpmh_bw_votes(a660_cnoc_bcms, + ARRAY_SIZE(a660_cnoc_bcms), cnoc_table, count); + kfree(cnoc_table); - build_rpmh_bw_votes(&votes->ddr_votes, gmu->num_bwlevels, hdl); + if (IS_ERR(cnoc)) { + free_rpmh_bw_votes(ddr); + return PTR_ERR(cnoc); + } - /* - *Query CNOC TCS command set for each use case defined in cnoc bw table - */ - ret = msm_bus_scale_query_tcs_cmd_all(&hdl, gmu->ccl); - if (ret) - goto out; + /* Build the HFI command once */ + build_bwtable_cmd_cache(&gmu->hfi.bwtbl_cmd, ddr, cnoc); - build_rpmh_bw_votes(&votes->cnoc_votes, gmu->num_cnocbwlevels, hdl); + free_rpmh_bw_votes(ddr); + free_rpmh_bw_votes(cnoc); - build_bwtable_cmd_cache(gmu); - -out: - kfree(usecases); - - return ret; + return 0; } static int gmu_rpmh_init(struct kgsl_device *device, - struct gmu_device *gmu, struct kgsl_pwrctrl *pwr) + struct gmu_device *gmu) { struct rpmh_arc_vals gfx_arc, cx_arc, mx_arc; int ret; /* Initialize BW tables */ - ret = gmu_bus_vote_init(gmu, pwr); + ret = gmu_bus_vote_init(device); if (ret) return ret; @@ -1255,7 +1389,7 @@ static int gmu_probe(struct kgsl_device *device, struct device_node *node) gmu->icc_path = of_icc_get(&gmu->pdev->dev, NULL); /* Populates RPMh configurations */ - ret = gmu_rpmh_init(device, gmu, pwr); + ret = gmu_rpmh_init(device, gmu); if (ret) goto error; @@ -1418,6 +1552,7 @@ static int gmu_start(struct kgsl_device *device) struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); struct kgsl_pwrctrl *pwr = &device->pwrctrl; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); + int level; switch (device->state) { case KGSL_STATE_INIT: @@ -1432,7 +1567,9 @@ static int gmu_start(struct kgsl_device *device) gmu_dev_ops->irq_enable(device); /* Vote for minimal DDR BW for GMU to init */ - icc_set_bw(gmu->icc_path, 0, MBps_to_icc(1171)); + level = pwr->pwrlevels[pwr->default_pwrlevel].bus_min; + icc_set_bw(gmu->icc_path, 0, + MBps_to_icc(pwr->ddr_table[level])); ret = gmu_dev_ops->rpmh_gpu_pwrctrl(device, GMU_FW_START, GMU_COLD_BOOT, 0); @@ -1447,8 +1584,6 @@ static int gmu_start(struct kgsl_device *device) ret = kgsl_pwrctrl_set_default_gpu_pwrlevel(device); if (ret) goto error_gmu; - - icc_set_bw(gmu->icc_path, 0, 0); break; case KGSL_STATE_SLUMBER: diff --git a/drivers/gpu/msm/kgsl_gmu.h b/drivers/gpu/msm/kgsl_gmu.h index 3e539c02039e..73b82e4aee08 100644 --- a/drivers/gpu/msm/kgsl_gmu.h +++ b/drivers/gpu/msm/kgsl_gmu.h @@ -118,18 +118,9 @@ struct gmu_memdesc { enum gmu_context_index ctx_idx; }; -struct gmu_bw_votes { - uint32_t cmds_wait_bitmask; - uint32_t cmds_per_bw_vote; - uint32_t cmd_addrs[MAX_BW_CMDS]; - uint32_t cmd_data[MAX_GX_LEVELS][MAX_BW_CMDS]; -}; - struct rpmh_votes_t { uint32_t gx_votes[MAX_GX_LEVELS]; uint32_t cx_votes[MAX_CX_LEVELS]; - struct gmu_bw_votes ddr_votes; - struct gmu_bw_votes cnoc_votes; }; enum gmu_load_mode { @@ -169,8 +160,6 @@ struct icc_path; * @load_mode: GMU FW load/boot mode * @wakeup_pwrlevel: GPU wake up power/DCVS level in case different * than default power level - * @pcl: GPU BW scaling client - * @ccl: CNOC BW scaling client * @idle_level: Minimal GPU idle power level * @fault_count: GMU fault count * @mailbox: Messages to AOP for ACD enable/disable go through this @@ -213,8 +202,6 @@ struct gmu_device { struct clk *gmu_clk; enum gmu_load_mode load_mode; unsigned int wakeup_pwrlevel; - unsigned int pcl; - unsigned int ccl; unsigned int idle_level; unsigned int fault_count; struct kgsl_mailbox mailbox; diff --git a/drivers/gpu/msm/kgsl_pwrctrl.c b/drivers/gpu/msm/kgsl_pwrctrl.c index 1468bbabc570..608397b50b9b 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.c +++ b/drivers/gpu/msm/kgsl_pwrctrl.c @@ -10,21 +10,14 @@ #include #include "kgsl_device.h" +#include "kgsl_bus.h" #include "kgsl_pwrscale.h" #include "kgsl_trace.h" -#define KGSL_PWRFLAGS_POWER_ON 0 -#define KGSL_PWRFLAGS_CLK_ON 1 -#define KGSL_PWRFLAGS_AXI_ON 2 -#define KGSL_PWRFLAGS_IRQ_ON 3 -#define KGSL_PWRFLAGS_NAP_OFF 5 - #define UPDATE_BUSY_VAL 1000000 #define KGSL_MAX_BUSLEVELS 20 -#define DEFAULT_BUS_P 25 - /* Order deeply matters here because reasons. New entries go on the end */ static const char * const clocks[] = { "src_clk", @@ -62,41 +55,6 @@ static void _gpu_clk_prepare_enable(struct kgsl_device *device, static void _bimc_clk_prepare_enable(struct kgsl_device *device, struct clk *clk, const char *name); -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -#include - -/** - * kgsl_get_bw() - Return latest msm bus IB vote - */ -static void kgsl_get_bw(unsigned long *ib, unsigned long *ab, void *data) -{ - struct kgsl_device *device = (struct kgsl_device *)data; - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - - if (gmu_core_scales_bandwidth(device)) - *ib = 0; - else - *ib = (unsigned long) - device->pwrctrl.bus_ibs[pwr->cur_buslevel]; - - *ab = last_ab; -} -#endif - -static u32 _ab_buslevel_update(struct kgsl_pwrctrl *pwr, u32 ib) -{ - if (!ib) - return 0; - - if ((!pwr->bus_percent_ab) && (!pwr->bus_ab_mbytes)) - return DEFAULT_BUS_P * ib / 100; - - if (pwr->bus_width) - return pwr->bus_ab_mbytes; - - return (pwr->bus_percent_ab * pwr->bus_max) / 100; -} - /** * _adjust_pwrlevel() - Given a requested power level do bounds checking on the * constraints and return the nearest possible level @@ -144,42 +102,6 @@ static unsigned int _adjust_pwrlevel(struct kgsl_pwrctrl *pwr, int level, return level; } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_vbif_update(u32 ib, u32 ab) -{ - /* ask a governor to vote on behalf of us */ - devfreq_vbif_update_bw((unsigned long) ib, (unsigned long) ab); -} -#else -static void kgsl_pwrctrl_vbif_update(u32 ib, u32 ab) -{ -} -#endif - -/** - * kgsl_bus_scale_request() - set GPU BW vote - * @device: Pointer to the kgsl_device struct - * @buslevel: index of bw vector[] table - */ -static int kgsl_bus_scale_request(struct kgsl_device *device, - unsigned int buslevel) -{ - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - int ret = 0; - - /* GMU scales BW */ - if (gmu_core_scales_bandwidth(device)) - ret = gmu_core_dcvs_set(device, INVALID_DCVS_IDX, buslevel); - else if (pwr->pcl) - /* Linux bus driver scales BW */ - icc_set_bw(pwr->icc_path, 0, MBps_to_icc(pwr->bus_ibs[i])); - - if (ret) - dev_err(device->dev, "GPU BW scaling failure: %d\n", ret); - - return ret; -} - /** * kgsl_clk_set_rate() - set GPU clock rate * @device: Pointer to the kgsl_device struct @@ -207,49 +129,6 @@ int kgsl_clk_set_rate(struct kgsl_device *device, return ret; } -/** - * kgsl_pwrctrl_buslevel_update() - Recalculate the bus vote and send it - * @device: Pointer to the kgsl_device struct - * @on: true for setting and active bus vote, false to turn off the vote - */ -void kgsl_pwrctrl_buslevel_update(struct kgsl_device *device, - bool on) -{ - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - int cur = pwr->pwrlevels[pwr->active_pwrlevel].bus_freq; - int buslevel = 0; - u32 ab; - - /* the bus should be ON to update the active frequency */ - if (on && !(test_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags))) - return; - /* - * If the bus should remain on calculate our request and submit it, - * otherwise request bus level 0, off. - */ - if (on) { - buslevel = min_t(int, pwr->pwrlevels[0].bus_max, - cur + pwr->bus_mod); - buslevel = max_t(int, buslevel, 1); - } else { - /* If the bus is being turned off, reset to default level */ - pwr->bus_mod = 0; - pwr->bus_percent_ab = 0; - pwr->bus_ab_mbytes = 0; - } - trace_kgsl_buslevel(device, pwr->active_pwrlevel, buslevel); - pwr->cur_buslevel = buslevel; - - /* buslevel is the IB vote, update the AB */ - ab = _ab_buslevel_update(pwr, pwr->bus_ibs[buslevel]); - - last_ab = ab; - - kgsl_bus_scale_request(device, buslevel); - - kgsl_pwrctrl_vbif_update(pwr->bus_ibs[buslevel], ab); -} - /** * kgsl_pwrctrl_pwrlevel_change_settings() - Program h/w during powerlevel * transitions @@ -350,7 +229,7 @@ void kgsl_pwrctrl_pwrlevel_change(struct kgsl_device *device, * Update the bus before the GPU clock to prevent underrun during * frequency increases. */ - kgsl_pwrctrl_buslevel_update(device, true); + kgsl_bus_update(device, true); pwrlevel = &pwr->pwrlevels[pwr->active_pwrlevel]; /* Change register settings if any BEFORE pwrlevel change*/ @@ -1340,28 +1219,6 @@ static void kgsl_pwrctrl_clk(struct kgsl_device *device, int state, } } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_suspend_devbw(struct kgsl_pwrctrl *pwr) -{ - if (pwr->devbw) - devfreq_suspend_devbw(pwr->devbw); -} - -static void kgsl_pwrctrl_resume_devbw(struct kgsl_pwrctrl *pwr) -{ - if (pwr->devbw) - devfreq_resume_devbw(pwr->devbw); -} -#else -static void kgsl_pwrctrl_suspend_devbw(struct kgsl_pwrctrl *pwr) -{ -} - -static void kgsl_pwrctrl_resume_devbw(struct kgsl_pwrctrl *pwr) -{ -} -#endif - static void kgsl_pwrctrl_axi(struct kgsl_device *device, int state) { struct kgsl_pwrctrl *pwr = &device->pwrctrl; @@ -1373,17 +1230,13 @@ static void kgsl_pwrctrl_axi(struct kgsl_device *device, int state) if (test_and_clear_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags)) { trace_kgsl_bus(device, state); - kgsl_pwrctrl_buslevel_update(device, false); - - kgsl_pwrctrl_suspend_devbw(pwr); + kgsl_bus_update(device, false); } } else if (state == KGSL_PWRFLAGS_ON) { if (!test_and_set_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags)) { trace_kgsl_bus(device, state); - kgsl_pwrctrl_buslevel_update(device, true); - - kgsl_pwrctrl_resume_devbw(pwr); + kgsl_bus_update(device, true); } } } @@ -1472,17 +1325,6 @@ static void kgsl_pwrctrl_irq(struct kgsl_device *device, int state) } } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_vbif_init(struct kgsl_device *device) -{ - devfreq_vbif_register_callback(kgsl_get_bw, device); -} -#else -static void kgsl_pwrctrl_vbif_init(struct kgsl_device *device) -{ -} -#endif - static int _get_clocks(struct kgsl_device *device) { struct device *dev = &device->pdev->dev; @@ -1584,41 +1426,13 @@ static int kgsl_pwrctrl_clk_set_rate(struct clk *grp_clk, unsigned int freq, return ret; } -static u32 *kgsl_bus_get_table(struct platform_device *pdev, int *count) -{ - u32 *levels; - int i, num = of_property_count_elems_of_size(pdev->dev.of_node, - "qcom,bus-table-ddr", sizeof(u32)); - - if (num <= 0) - return NULL; - - levels = kcalloc(num, sizeof(*levels), GFP_KERNEL); - if (!levels) - return NULL; - - for (i = 0; i < num; i++) - of_property_read_u32_index(pdev->dev.of_node, - "qcom,bus-table-ddr", i, &levels[i]); - - *count = num; - return levels; -} - static void kgsl_idle_check(struct work_struct *work); int kgsl_pwrctrl_init(struct kgsl_device *device) { - int i, result, freq, levels_count; + int i, result, freq; struct platform_device *pdev = device->pdev; struct kgsl_pwrctrl *pwr = &device->pwrctrl; - struct device_node *gpubw_dev_node = NULL; - struct platform_device *p2dev; - u32 *levels; - - levels = kgsl_bus_get_table(pdev, &levels_count); - if (!levels) - return -EINVAL; result = _get_clocks(device); if (result) @@ -1681,61 +1495,11 @@ int kgsl_pwrctrl_init(struct kgsl_device *device) pm_runtime_enable(&pdev->dev); - /* Check if gpu bandwidth vote device is defined in dts */ - if (pwr->bus_control) - /* Check if gpu bandwidth vote device is defined in dts */ - gpubw_dev_node = of_parse_phandle(pdev->dev.of_node, - "qcom,gpubw-dev", 0); - - /* - * Governor support enables the gpu bus scaling via governor - * and hence no need to register for bus scaling client - * if gpubw-dev is defined. - */ - if (gpubw_dev_node) { - p2dev = of_find_device_by_node(gpubw_dev_node); - if (p2dev) - pwr->devbw = &p2dev->dev; - } else { - /* Register for interconnect */ - pwr->icc_path = of_icc_get(&pdev->dev, NULL); - } - - pwr->bus_ibs = kzalloc(levels_count, sizeof(*pwr->bus_ib), GFP_KERNEL); - if (pwr->bus_ibs == NULL) { - result = -ENOMEM; - goto error_disable_pm; - } - - pwr->bus_ibs_count = levels_count; - - /* - * Pull the BW vote out of the bus table. They will be used to - * calculate the ratio between the votes. - */ - - for (i = 0; i < levels_count; i++) { - /* Convert the KBps value from the table to MBps */ - pwr->bus_ibs[i] = DIV_ROUND_UP_ULL(levels[i], 1000); - - pwr->bus_max = max(pwr->bus_max, pwr->bus_ibs[i]); - } - - kfree(levels); - - kgsl_pwrctrl_vbif_init(device); - /* temperature sensor name */ of_property_read_string(pdev->dev.of_node, "qcom,tzone-name", &pwr->tzone_name); - return result; - -error_disable_pm: - pm_runtime_disable(&pdev->dev); - kfree(levels); - - return result; + return 0; } void kgsl_pwrctrl_close(struct kgsl_device *device) @@ -1744,9 +1508,7 @@ void kgsl_pwrctrl_close(struct kgsl_device *device) pwr->power_flags = 0; - kfree(pwr->bus_ibs); - - icc_put(pwr->icc_path); + kgsl_bus_close(device); pm_runtime_disable(&device->pdev->dev); } diff --git a/drivers/gpu/msm/kgsl_pwrctrl.h b/drivers/gpu/msm/kgsl_pwrctrl.h index 7d2ca5d45195..ed146843d14c 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.h +++ b/drivers/gpu/msm/kgsl_pwrctrl.h @@ -21,6 +21,12 @@ #define KGSL_MAX_PWRLEVELS 10 +#define KGSL_PWRFLAGS_POWER_ON 0 +#define KGSL_PWRFLAGS_CLK_ON 1 +#define KGSL_PWRFLAGS_AXI_ON 2 +#define KGSL_PWRFLAGS_IRQ_ON 3 +#define KGSL_PWRFLAGS_NAP_OFF 5 + /* Only two supported levels, min & max */ #define KGSL_CONSTRAINT_PWR_MAXLEVELS 2 @@ -35,6 +41,7 @@ enum kgsl_pwrctrl_timer_type { }; struct platform_device; +struct icc_path; struct kgsl_clk_stats { unsigned int busy; @@ -91,7 +98,6 @@ struct kgsl_pwrlevel { * @throttle_mask - LM throttle mask * @interval_timeout - timeout in jiffies to be idle before a power event * @clock_times - Each GPU frequency's accumulated active time in us - * @pcl - bus scale identifier * @clk_stats - structure of clock statistics * @input_disable - To disable GPU wakeup on touch input event * @bus_control - true if the bus calculation is independent @@ -133,7 +139,6 @@ struct kgsl_pwrctrl { unsigned int throttle_mask; unsigned long interval_timeout; u64 clock_times[KGSL_MAX_PWRLEVELS]; - uint32_t pcl; struct kgsl_clk_stats clk_stats; bool input_disable; bool bus_control; @@ -141,15 +146,16 @@ struct kgsl_pwrctrl { unsigned int bus_percent_ab; unsigned int bus_width; unsigned long bus_ab_mbytes; - struct device *devbw; - /** @bus_ibs: List of the bus bandwidths in use by our target */ - u32 *bus_ibs; - /** @bus_ibs_count: Number of objects in @bus_ibs */ - int bus_ibs_count; + /** @ddr_table: List of the DDR bandwidths in KBps for the target */ + u32 *ddr_table; + /** @ddr_table_count: Number of objects in @ddr_table */ + int ddr_table_count; /** cur_buslevel: The last buslevel voted by the driver */ int cur_buslevel; /** @bus_max: The maximum bandwidth available to the device */ unsigned long bus_max; + /** @bus_set: Function for setting the bus constraints */ + int (*bus_set)(struct kgsl_device *device, int buslevel, u32 ab); struct kgsl_pwr_constraint constraint; bool superfast; unsigned int gpu_bimc_int_clk_freq; @@ -165,8 +171,6 @@ void kgsl_timer(struct timer_list *t); void kgsl_pre_hwaccess(struct kgsl_device *device); void kgsl_pwrctrl_pwrlevel_change(struct kgsl_device *device, unsigned int level); -void kgsl_pwrctrl_buslevel_update(struct kgsl_device *device, - bool on); int kgsl_pwrctrl_init_sysfs(struct kgsl_device *device); int kgsl_pwrctrl_change_state(struct kgsl_device *device, int state); int kgsl_clk_set_rate(struct kgsl_device *device, diff --git a/drivers/gpu/msm/kgsl_pwrscale.c b/drivers/gpu/msm/kgsl_pwrscale.c index 72f249678763..6820fe6a49a0 100644 --- a/drivers/gpu/msm/kgsl_pwrscale.c +++ b/drivers/gpu/msm/kgsl_pwrscale.c @@ -6,6 +6,7 @@ #include #include +#include "kgsl_bus.h" #include "kgsl_device.h" #include "kgsl_pwrscale.h" #include "kgsl_trace.h" @@ -389,7 +390,6 @@ int kgsl_devfreq_get_dev_status(struct device *dev, last_b->ram_time = device->pwrscale.accum_stats.ram_time; last_b->ram_wait = device->pwrscale.accum_stats.ram_wait; - last_b->mod = device->pwrctrl.bus_mod; last_b->buslevel = device->pwrctrl.cur_buslevel; } @@ -586,7 +586,7 @@ int kgsl_busmon_target(struct device *dev, unsigned long *freq, u32 flags) if ((pwr->bus_mod != b) || (pwr->bus_ab_mbytes != ab_mbytes)) { pwr->bus_percent_ab = device->pwrscale.bus_profile.percent_ab; pwr->bus_ab_mbytes = ab_mbytes; - kgsl_pwrctrl_buslevel_update(device, true); + kgsl_bus_update(device, true); } mutex_unlock(&device->mutex); @@ -786,8 +786,8 @@ int kgsl_pwrscale_init(struct kgsl_device *device, struct platform_device *pdev, * the bus bandwidth vote. */ if (pwr->bus_control) { - adreno_tz_data.bus.num = pwr->bus_ibs_count; - adreno_tz_data.bus.ib_mbps = pwr->bus_ibs; + adreno_tz_data.bus.num = pwr->ddr_table_count; + adreno_tz_data.bus.ib_kbps = pwr->ddr_table; adreno_tz_data.bus.width = pwr->bus_width; if (!kgsl_of_property_read_ddrtype(device->pdev->dev.of_node, diff --git a/drivers/gpu/msm/kgsl_sharedmem.c b/drivers/gpu/msm/kgsl_sharedmem.c index bebdda88ef3f..6fec82642dcc 100644 --- a/drivers/gpu/msm/kgsl_sharedmem.c +++ b/drivers/gpu/msm/kgsl_sharedmem.c @@ -13,6 +13,15 @@ #include "kgsl_pool.h" #include "kgsl_sharedmem.h" +/* + * For now, we either need the low level cache operations or + * QCOM_KGSL_IOCOHERENCY_DEFAULT enabled because we can't stop userspace + * from expecting to enable cached surfaces and have them work + */ +#if !defined(dmac_flush_range) && !IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) +#error "KGSL needs either dmac_flush_range or CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT enabled" +#endif + /* * The user can set this from debugfs to force failed memory allocations to * fail without trying OOM first. This is a debug setting useful for @@ -512,7 +521,8 @@ static inline unsigned int _fixup_cache_range_op(unsigned int op) } #endif -static inline void _cache_op(unsigned int op, +#ifdef dmac_flush_range +static void _cache_op(unsigned int op, const void *start, const void *end) { /* @@ -532,8 +542,8 @@ static inline void _cache_op(unsigned int op, } } -static int kgsl_do_cache_op(struct page *page, void *addr, - uint64_t offset, uint64_t size, unsigned int op) +static void kgsl_do_cache_op(struct page *page, void *addr, u64 offset, + u64 size, unsigned int op) { if (page != NULL) { unsigned long pfn = page_to_pfn(page) + offset / PAGE_SIZE; @@ -562,15 +572,21 @@ static int kgsl_do_cache_op(struct page *page, void *addr, offset = 0; } while (size); - return 0; + return; } addr = page_address(page); } _cache_op(op, addr + offset, addr + offset + (size_t) size); - return 0; } +#else + +static void kgsl_do_cache_op(struct page *page, void *addr, u64 offset, + u64 size, unsigned int op) +{ +} +#endif int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, uint64_t size, unsigned int op) @@ -579,7 +595,9 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, struct sg_table *sgt = NULL; struct scatterlist *sg; unsigned int i, pos = 0; - int ret = 0; + + if (memdesc->flags & KGSL_MEMFLAGS_IOCOHERENT) + return 0; if (size == 0 || size > UINT_MAX) return -EINVAL; @@ -598,8 +616,8 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, if (addr + ((size_t) offset + (size_t) size) < addr) return -ERANGE; - ret = kgsl_do_cache_op(NULL, addr, offset, size, op); - return ret; + kgsl_do_cache_op(NULL, addr, offset, size, op); + return 0; } /* @@ -610,7 +628,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, sgt = memdesc->sgt; else { if (memdesc->pages == NULL) - return ret; + return 0; sgt = kgsl_alloc_sgt_from_pages(memdesc); if (IS_ERR(sgt)) @@ -627,7 +645,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, sg_offset = offset > pos ? offset - pos : 0; sg_left = (sg->length - sg_offset > size) ? size : sg->length - sg_offset; - ret = kgsl_do_cache_op(sg_page(sg), NULL, sg_offset, + kgsl_do_cache_op(sg_page(sg), NULL, sg_offset, sg_left, op); size -= sg_left; if (size == 0) @@ -638,7 +656,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, if (memdesc->sgt == NULL) kgsl_free_sgt(sgt); - return ret; + return 0; } void kgsl_memdesc_init(struct kgsl_device *device, @@ -657,9 +675,14 @@ void kgsl_memdesc_init(struct kgsl_device *device, flags &= ~((uint64_t) KGSL_MEMFLAGS_USE_CPU_MAP); /* Disable IO coherence if it is not supported on the chip */ - if (!MMU_FEATURE(mmu, KGSL_MMU_IO_COHERENT)) + if (!MMU_FEATURE(mmu, KGSL_MMU_IO_COHERENT)) { flags &= ~((uint64_t) KGSL_MEMFLAGS_IOCOHERENT); + WARN_ONCE(IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT), + "I/O coherency is not supported on this target\n"); + } else if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + if (MMU_FEATURE(mmu, KGSL_MMU_NEED_GUARD_PAGE)) memdesc->priv |= KGSL_MEMDESC_GUARD_PAGE; diff --git a/drivers/gpu/msm/msm_adreno_devfreq.h b/drivers/gpu/msm/msm_adreno_devfreq.h index 9fec1d613442..0c06f3523b84 100644 --- a/drivers/gpu/msm/msm_adreno_devfreq.h +++ b/drivers/gpu/msm/msm_adreno_devfreq.h @@ -31,7 +31,6 @@ int kgsl_devfreq_del_notifier(struct device *device, struct xstats { u64 ram_time; u64 ram_wait; - int mod; int buslevel; }; @@ -55,7 +54,7 @@ struct devfreq_msm_adreno_tz_data { u32 *down; s32 *p_up; s32 *p_down; - u32 *ib_mbps; + u32 *ib_kbps; bool floating; } bus; unsigned int device_id;