From d50bb36c781e4b19bd68f785d0671310c9376413 Mon Sep 17 00:00:00 2001 From: Jordan Crouse Date: Mon, 18 Nov 2019 12:55:03 -0700 Subject: [PATCH 1/4] msm: kgsl: Modernize bus scaling Replace all of the legacy bus scaling code with the new interconnect API. This includes removing all msm_bus support as well as the devbw interface which is not needed since icc can set both the IB and the AB on our behalf. As part of this, move most of the bus specific code into its own file, add support for parsing a table of our own from the device tree and handle both the GPU and GMU bus setting paths properly. Change-Id: Ic0dedbadfbb3c48ec414ec60e3032b2204b4e724 Signed-off-by: Jordan Crouse --- drivers/gpu/msm/Makefile | 1 + drivers/gpu/msm/adreno.c | 7 + drivers/gpu/msm/kgsl_bus.c | 185 ++++++++++++++++++++ drivers/gpu/msm/kgsl_bus.h | 19 ++ drivers/gpu/msm/kgsl_gmu.c | 15 +- drivers/gpu/msm/kgsl_gmu.h | 4 - drivers/gpu/msm/kgsl_pwrctrl.c | 252 +-------------------------- drivers/gpu/msm/kgsl_pwrctrl.h | 22 ++- drivers/gpu/msm/kgsl_pwrscale.c | 7 +- drivers/gpu/msm/msm_adreno_devfreq.h | 2 +- 10 files changed, 247 insertions(+), 267 deletions(-) create mode 100644 drivers/gpu/msm/kgsl_bus.c create mode 100644 drivers/gpu/msm/kgsl_bus.h diff --git a/drivers/gpu/msm/Makefile b/drivers/gpu/msm/Makefile index 808a9817c9a0..4c3f64d41c97 100644 --- a/drivers/gpu/msm/Makefile +++ b/drivers/gpu/msm/Makefile @@ -5,6 +5,7 @@ obj-$(CONFIG_QCOM_KGSL) += msm_kgsl.o msm_kgsl-y = \ kgsl.o \ + kgsl_bus.o \ kgsl_drawobj.o \ kgsl_events.o \ kgsl_ioctl.o \ diff --git a/drivers/gpu/msm/adreno.c b/drivers/gpu/msm/adreno.c index 4f866925e423..d5b56a712e1e 100644 --- a/drivers/gpu/msm/adreno.c +++ b/drivers/gpu/msm/adreno.c @@ -22,6 +22,7 @@ #include "adreno_compat.h" #include "adreno_iommu.h" #include "adreno_trace.h" +#include "kgsl_bus.h" #include "kgsl_trace.h" #include "kgsl_util.h" @@ -1450,6 +1451,12 @@ static int adreno_probe(struct platform_device *pdev) return status; } + status = kgsl_bus_init(device, pdev); + if (status) { + device->pdev = NULL; + return status; + } + /* * Probe/init GMU after initial gpu power probe * Another part of GPU power probe in platform_probe diff --git a/drivers/gpu/msm/kgsl_bus.c b/drivers/gpu/msm/kgsl_bus.c new file mode 100644 index 000000000000..74c9e99b43d8 --- /dev/null +++ b/drivers/gpu/msm/kgsl_bus.c @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019, The Linux Foundation. All rights reserved. + */ + +#include +#include + +#include "kgsl_bus.h" +#include "kgsl_device.h" +#include "kgsl_trace.h" + +static int gmu_bus_set(struct kgsl_device *device, int buslevel, + u32 ab) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int ret; + + ret = gmu_core_dcvs_set(device, INVALID_DCVS_IDX, buslevel); + + if (!ret) + icc_set_bw(pwr->icc_path, MBps_to_icc(ab), 0); + + return ret; +} + +static int interconnect_bus_set(struct kgsl_device *device, int level, + u32 ab) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + + icc_set_bw(pwr->icc_path, MBps_to_icc(ab), + KBps_to_icc(pwr->ddr_table[level])); + + return 0; +} + +static u32 _ab_buslevel_update(struct kgsl_pwrctrl *pwr, + u32 ib) +{ + if (!ib) + return 0; + + /* In the absence of any other settings, make ab 25% of ib */ + if ((!pwr->bus_percent_ab) && (!pwr->bus_ab_mbytes)) + return 25 * ib / 100; + + if (pwr->bus_width) + return pwr->bus_ab_mbytes; + + return (pwr->bus_percent_ab * pwr->bus_max) / 100; +} + + +void kgsl_bus_update(struct kgsl_device *device, bool on) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + /* FIXME: this might be wrong? */ + int cur = pwr->pwrlevels[pwr->active_pwrlevel].bus_freq; + int buslevel = 0; + u32 ab; + + /* the bus should be ON to update the active frequency */ + if (on && !(test_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags))) + return; + /* + * If the bus should remain on calculate our request and submit it, + * otherwise request bus level 0, off. + */ + if (on) { + buslevel = min_t(int, pwr->pwrlevels[0].bus_max, + cur + pwr->bus_mod); + buslevel = max_t(int, buslevel, 1); + } else { + /* If the bus is being turned off, reset to default level */ + pwr->bus_mod = 0; + pwr->bus_percent_ab = 0; + pwr->bus_ab_mbytes = 0; + } + trace_kgsl_buslevel(device, pwr->active_pwrlevel, buslevel); + pwr->cur_buslevel = buslevel; + + /* buslevel is the IB vote, update the AB */ + ab = _ab_buslevel_update(pwr, pwr->ddr_table[buslevel]); + + pwr->bus_set(device, buslevel, ab); +} + +static void validate_pwrlevels(struct kgsl_device *device, u32 *ibs, + int count) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int i; + + for (i = 0; i < pwr->num_pwrlevels - 1; i++) { + struct kgsl_pwrlevel *pwrlevel = &pwr->pwrlevels[i]; + + if (pwrlevel->bus_freq >= count) { + dev_err(device->dev, "Bus setting for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_freq = count - 1; + } + + if (pwrlevel->bus_max >= count) { + dev_err(device->dev, "Bus max for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_max = count - 1; + } + + if (pwrlevel->bus_min >= count) { + dev_err(device->dev, "Bus min for GPU freq %d is out of bounds\n", + pwrlevel->gpu_freq); + pwrlevel->bus_min = count - 1; + } + + if (pwrlevel->bus_min > pwrlevel->bus_max) { + dev_err(device->dev, "Bus min is bigger than bus max for GPU freq %d\n", + pwrlevel->gpu_freq); + pwrlevel->bus_min = pwrlevel->bus_max; + } + } +} + +u32 *kgsl_bus_get_table(struct platform_device *pdev, + const char *name, int *count) +{ + u32 *levels; + int i, num = of_property_count_elems_of_size(pdev->dev.of_node, + name, sizeof(u32)); + + /* If the bus wasn't specified, then build a static table */ + if (num <= 0) + return ERR_PTR(-EINVAL); + + levels = kcalloc(num, sizeof(*levels), GFP_KERNEL); + if (!levels) + return ERR_PTR(-ENOMEM); + + for (i = 0; i < num; i++) + of_property_read_u32_index(pdev->dev.of_node, + name, i, &levels[i]); + + *count = num; + return levels; +} + +int kgsl_bus_init(struct kgsl_device *device, struct platform_device *pdev) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + int ret, count; + + pwr->ddr_table = kgsl_bus_get_table(pdev, "qcom,bus-table-ddr", &count); + if (IS_ERR(pwr->ddr_table)) { + ret = PTR_ERR(pwr->ddr_table); + pwr->ddr_table = NULL; + return ret; + } + + pwr->ddr_table_count = count; + + validate_pwrlevels(device, pwr->ddr_table, pwr->ddr_table_count); + + pwr->icc_path = of_icc_get(&pdev->dev, NULL); + if (IS_ERR(pwr->icc_path) && !gmu_core_scales_bandwidth(device)) { + WARN(1, "The CPU has no way to set the GPU bus levels\n"); + return PTR_ERR(pwr->icc_path); + } + + if (gmu_core_scales_bandwidth(device)) + pwr->bus_set = gmu_bus_set; + else + pwr->bus_set = interconnect_bus_set; + + return 0; +} + +void kgsl_bus_close(struct kgsl_device *device) +{ + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + + kfree(pwr->ddr_table); + + /* FIXME: Make sure icc put can handle NULL or IS_ERR */ + icc_put(pwr->icc_path); +} diff --git a/drivers/gpu/msm/kgsl_bus.h b/drivers/gpu/msm/kgsl_bus.h new file mode 100644 index 000000000000..2329cce8000b --- /dev/null +++ b/drivers/gpu/msm/kgsl_bus.h @@ -0,0 +1,19 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019, The Linux Foundation. All rights reserved. + */ + +#ifndef _KGSL_BUS_H +#define _KGSL_BUS_H + +struct kgsl_device; +struct platform_device; + +int kgsl_bus_init(struct kgsl_device *device, struct platform_device *pdev); +void kgsl_bus_close(struct kgsl_device *device); +void kgsl_bus_update(struct kgsl_device *device, bool on); + +u32 *kgsl_bus_get_table(struct platform_device *pdev, + const char *name, int *count); + +#endif diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index c6d9536b947c..551b767ade04 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -7,9 +7,9 @@ #include #include #include +#include #include #include -#include #include #include #include @@ -490,6 +490,7 @@ static int gmu_dcvs_set(struct kgsl_device *device, int gpu_pwrlevel, int bus_level) { int ret = 0; + struct kgsl_pwrctrl *pwr = &device->pwrctrl; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); struct adreno_device *adreno_dev = ADRENO_DEVICE(device); struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); @@ -515,7 +516,7 @@ static int gmu_dcvs_set(struct kgsl_device *device, if (gpu_pwrlevel < gmu->num_gpupwrlevels - 1) req.freq = gmu->num_gpupwrlevels - gpu_pwrlevel - 1; - if (bus_level < gmu->num_bwlevels && bus_level > 0) + if (bus_level < pwr->ddr_table_count && bus_level > 0) req.bw = bus_level; /* GMU will vote for slumber levels through the sleep sequence */ @@ -794,6 +795,9 @@ static int gmu_bus_vote_init(struct gmu_device *gmu, struct kgsl_pwrctrl *pwr) struct rpmh_votes_t *votes = &gmu->rpmh_votes; int ret; + if (!gmu->num_bwlevels) + return 0; + usecases = kcalloc(gmu->num_bwlevels, sizeof(*usecases), GFP_KERNEL); if (!usecases) return -ENOMEM; @@ -1418,6 +1422,7 @@ static int gmu_start(struct kgsl_device *device) struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); struct kgsl_pwrctrl *pwr = &device->pwrctrl; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); + unsigned long ib; switch (device->state) { case KGSL_STATE_INIT: @@ -1432,7 +1437,9 @@ static int gmu_start(struct kgsl_device *device) gmu_dev_ops->irq_enable(device); /* Vote for minimal DDR BW for GMU to init */ - icc_set_bw(gmu->icc_path, 0, MBps_to_icc(1171)); + level = pwr->pwrlevels[pwr->default_pwrlevel].bus_min; + icc_set_bw(gmu->icc_path, 0, + MBps_to_icc(pwr->ddr_table[level])); ret = gmu_dev_ops->rpmh_gpu_pwrctrl(device, GMU_FW_START, GMU_COLD_BOOT, 0); @@ -1447,8 +1454,6 @@ static int gmu_start(struct kgsl_device *device) ret = kgsl_pwrctrl_set_default_gpu_pwrlevel(device); if (ret) goto error_gmu; - - icc_set_bw(gmu->icc_path, 0, 0); break; case KGSL_STATE_SLUMBER: diff --git a/drivers/gpu/msm/kgsl_gmu.h b/drivers/gpu/msm/kgsl_gmu.h index 3e539c02039e..eff99adcef59 100644 --- a/drivers/gpu/msm/kgsl_gmu.h +++ b/drivers/gpu/msm/kgsl_gmu.h @@ -169,8 +169,6 @@ struct icc_path; * @load_mode: GMU FW load/boot mode * @wakeup_pwrlevel: GPU wake up power/DCVS level in case different * than default power level - * @pcl: GPU BW scaling client - * @ccl: CNOC BW scaling client * @idle_level: Minimal GPU idle power level * @fault_count: GMU fault count * @mailbox: Messages to AOP for ACD enable/disable go through this @@ -213,8 +211,6 @@ struct gmu_device { struct clk *gmu_clk; enum gmu_load_mode load_mode; unsigned int wakeup_pwrlevel; - unsigned int pcl; - unsigned int ccl; unsigned int idle_level; unsigned int fault_count; struct kgsl_mailbox mailbox; diff --git a/drivers/gpu/msm/kgsl_pwrctrl.c b/drivers/gpu/msm/kgsl_pwrctrl.c index 1468bbabc570..608397b50b9b 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.c +++ b/drivers/gpu/msm/kgsl_pwrctrl.c @@ -10,21 +10,14 @@ #include #include "kgsl_device.h" +#include "kgsl_bus.h" #include "kgsl_pwrscale.h" #include "kgsl_trace.h" -#define KGSL_PWRFLAGS_POWER_ON 0 -#define KGSL_PWRFLAGS_CLK_ON 1 -#define KGSL_PWRFLAGS_AXI_ON 2 -#define KGSL_PWRFLAGS_IRQ_ON 3 -#define KGSL_PWRFLAGS_NAP_OFF 5 - #define UPDATE_BUSY_VAL 1000000 #define KGSL_MAX_BUSLEVELS 20 -#define DEFAULT_BUS_P 25 - /* Order deeply matters here because reasons. New entries go on the end */ static const char * const clocks[] = { "src_clk", @@ -62,41 +55,6 @@ static void _gpu_clk_prepare_enable(struct kgsl_device *device, static void _bimc_clk_prepare_enable(struct kgsl_device *device, struct clk *clk, const char *name); -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -#include - -/** - * kgsl_get_bw() - Return latest msm bus IB vote - */ -static void kgsl_get_bw(unsigned long *ib, unsigned long *ab, void *data) -{ - struct kgsl_device *device = (struct kgsl_device *)data; - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - - if (gmu_core_scales_bandwidth(device)) - *ib = 0; - else - *ib = (unsigned long) - device->pwrctrl.bus_ibs[pwr->cur_buslevel]; - - *ab = last_ab; -} -#endif - -static u32 _ab_buslevel_update(struct kgsl_pwrctrl *pwr, u32 ib) -{ - if (!ib) - return 0; - - if ((!pwr->bus_percent_ab) && (!pwr->bus_ab_mbytes)) - return DEFAULT_BUS_P * ib / 100; - - if (pwr->bus_width) - return pwr->bus_ab_mbytes; - - return (pwr->bus_percent_ab * pwr->bus_max) / 100; -} - /** * _adjust_pwrlevel() - Given a requested power level do bounds checking on the * constraints and return the nearest possible level @@ -144,42 +102,6 @@ static unsigned int _adjust_pwrlevel(struct kgsl_pwrctrl *pwr, int level, return level; } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_vbif_update(u32 ib, u32 ab) -{ - /* ask a governor to vote on behalf of us */ - devfreq_vbif_update_bw((unsigned long) ib, (unsigned long) ab); -} -#else -static void kgsl_pwrctrl_vbif_update(u32 ib, u32 ab) -{ -} -#endif - -/** - * kgsl_bus_scale_request() - set GPU BW vote - * @device: Pointer to the kgsl_device struct - * @buslevel: index of bw vector[] table - */ -static int kgsl_bus_scale_request(struct kgsl_device *device, - unsigned int buslevel) -{ - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - int ret = 0; - - /* GMU scales BW */ - if (gmu_core_scales_bandwidth(device)) - ret = gmu_core_dcvs_set(device, INVALID_DCVS_IDX, buslevel); - else if (pwr->pcl) - /* Linux bus driver scales BW */ - icc_set_bw(pwr->icc_path, 0, MBps_to_icc(pwr->bus_ibs[i])); - - if (ret) - dev_err(device->dev, "GPU BW scaling failure: %d\n", ret); - - return ret; -} - /** * kgsl_clk_set_rate() - set GPU clock rate * @device: Pointer to the kgsl_device struct @@ -207,49 +129,6 @@ int kgsl_clk_set_rate(struct kgsl_device *device, return ret; } -/** - * kgsl_pwrctrl_buslevel_update() - Recalculate the bus vote and send it - * @device: Pointer to the kgsl_device struct - * @on: true for setting and active bus vote, false to turn off the vote - */ -void kgsl_pwrctrl_buslevel_update(struct kgsl_device *device, - bool on) -{ - struct kgsl_pwrctrl *pwr = &device->pwrctrl; - int cur = pwr->pwrlevels[pwr->active_pwrlevel].bus_freq; - int buslevel = 0; - u32 ab; - - /* the bus should be ON to update the active frequency */ - if (on && !(test_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags))) - return; - /* - * If the bus should remain on calculate our request and submit it, - * otherwise request bus level 0, off. - */ - if (on) { - buslevel = min_t(int, pwr->pwrlevels[0].bus_max, - cur + pwr->bus_mod); - buslevel = max_t(int, buslevel, 1); - } else { - /* If the bus is being turned off, reset to default level */ - pwr->bus_mod = 0; - pwr->bus_percent_ab = 0; - pwr->bus_ab_mbytes = 0; - } - trace_kgsl_buslevel(device, pwr->active_pwrlevel, buslevel); - pwr->cur_buslevel = buslevel; - - /* buslevel is the IB vote, update the AB */ - ab = _ab_buslevel_update(pwr, pwr->bus_ibs[buslevel]); - - last_ab = ab; - - kgsl_bus_scale_request(device, buslevel); - - kgsl_pwrctrl_vbif_update(pwr->bus_ibs[buslevel], ab); -} - /** * kgsl_pwrctrl_pwrlevel_change_settings() - Program h/w during powerlevel * transitions @@ -350,7 +229,7 @@ void kgsl_pwrctrl_pwrlevel_change(struct kgsl_device *device, * Update the bus before the GPU clock to prevent underrun during * frequency increases. */ - kgsl_pwrctrl_buslevel_update(device, true); + kgsl_bus_update(device, true); pwrlevel = &pwr->pwrlevels[pwr->active_pwrlevel]; /* Change register settings if any BEFORE pwrlevel change*/ @@ -1340,28 +1219,6 @@ static void kgsl_pwrctrl_clk(struct kgsl_device *device, int state, } } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_suspend_devbw(struct kgsl_pwrctrl *pwr) -{ - if (pwr->devbw) - devfreq_suspend_devbw(pwr->devbw); -} - -static void kgsl_pwrctrl_resume_devbw(struct kgsl_pwrctrl *pwr) -{ - if (pwr->devbw) - devfreq_resume_devbw(pwr->devbw); -} -#else -static void kgsl_pwrctrl_suspend_devbw(struct kgsl_pwrctrl *pwr) -{ -} - -static void kgsl_pwrctrl_resume_devbw(struct kgsl_pwrctrl *pwr) -{ -} -#endif - static void kgsl_pwrctrl_axi(struct kgsl_device *device, int state) { struct kgsl_pwrctrl *pwr = &device->pwrctrl; @@ -1373,17 +1230,13 @@ static void kgsl_pwrctrl_axi(struct kgsl_device *device, int state) if (test_and_clear_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags)) { trace_kgsl_bus(device, state); - kgsl_pwrctrl_buslevel_update(device, false); - - kgsl_pwrctrl_suspend_devbw(pwr); + kgsl_bus_update(device, false); } } else if (state == KGSL_PWRFLAGS_ON) { if (!test_and_set_bit(KGSL_PWRFLAGS_AXI_ON, &pwr->power_flags)) { trace_kgsl_bus(device, state); - kgsl_pwrctrl_buslevel_update(device, true); - - kgsl_pwrctrl_resume_devbw(pwr); + kgsl_bus_update(device, true); } } } @@ -1472,17 +1325,6 @@ static void kgsl_pwrctrl_irq(struct kgsl_device *device, int state) } } -#ifdef CONFIG_DEVFREQ_GOV_QCOM_GPUBW_MON -static void kgsl_pwrctrl_vbif_init(struct kgsl_device *device) -{ - devfreq_vbif_register_callback(kgsl_get_bw, device); -} -#else -static void kgsl_pwrctrl_vbif_init(struct kgsl_device *device) -{ -} -#endif - static int _get_clocks(struct kgsl_device *device) { struct device *dev = &device->pdev->dev; @@ -1584,41 +1426,13 @@ static int kgsl_pwrctrl_clk_set_rate(struct clk *grp_clk, unsigned int freq, return ret; } -static u32 *kgsl_bus_get_table(struct platform_device *pdev, int *count) -{ - u32 *levels; - int i, num = of_property_count_elems_of_size(pdev->dev.of_node, - "qcom,bus-table-ddr", sizeof(u32)); - - if (num <= 0) - return NULL; - - levels = kcalloc(num, sizeof(*levels), GFP_KERNEL); - if (!levels) - return NULL; - - for (i = 0; i < num; i++) - of_property_read_u32_index(pdev->dev.of_node, - "qcom,bus-table-ddr", i, &levels[i]); - - *count = num; - return levels; -} - static void kgsl_idle_check(struct work_struct *work); int kgsl_pwrctrl_init(struct kgsl_device *device) { - int i, result, freq, levels_count; + int i, result, freq; struct platform_device *pdev = device->pdev; struct kgsl_pwrctrl *pwr = &device->pwrctrl; - struct device_node *gpubw_dev_node = NULL; - struct platform_device *p2dev; - u32 *levels; - - levels = kgsl_bus_get_table(pdev, &levels_count); - if (!levels) - return -EINVAL; result = _get_clocks(device); if (result) @@ -1681,61 +1495,11 @@ int kgsl_pwrctrl_init(struct kgsl_device *device) pm_runtime_enable(&pdev->dev); - /* Check if gpu bandwidth vote device is defined in dts */ - if (pwr->bus_control) - /* Check if gpu bandwidth vote device is defined in dts */ - gpubw_dev_node = of_parse_phandle(pdev->dev.of_node, - "qcom,gpubw-dev", 0); - - /* - * Governor support enables the gpu bus scaling via governor - * and hence no need to register for bus scaling client - * if gpubw-dev is defined. - */ - if (gpubw_dev_node) { - p2dev = of_find_device_by_node(gpubw_dev_node); - if (p2dev) - pwr->devbw = &p2dev->dev; - } else { - /* Register for interconnect */ - pwr->icc_path = of_icc_get(&pdev->dev, NULL); - } - - pwr->bus_ibs = kzalloc(levels_count, sizeof(*pwr->bus_ib), GFP_KERNEL); - if (pwr->bus_ibs == NULL) { - result = -ENOMEM; - goto error_disable_pm; - } - - pwr->bus_ibs_count = levels_count; - - /* - * Pull the BW vote out of the bus table. They will be used to - * calculate the ratio between the votes. - */ - - for (i = 0; i < levels_count; i++) { - /* Convert the KBps value from the table to MBps */ - pwr->bus_ibs[i] = DIV_ROUND_UP_ULL(levels[i], 1000); - - pwr->bus_max = max(pwr->bus_max, pwr->bus_ibs[i]); - } - - kfree(levels); - - kgsl_pwrctrl_vbif_init(device); - /* temperature sensor name */ of_property_read_string(pdev->dev.of_node, "qcom,tzone-name", &pwr->tzone_name); - return result; - -error_disable_pm: - pm_runtime_disable(&pdev->dev); - kfree(levels); - - return result; + return 0; } void kgsl_pwrctrl_close(struct kgsl_device *device) @@ -1744,9 +1508,7 @@ void kgsl_pwrctrl_close(struct kgsl_device *device) pwr->power_flags = 0; - kfree(pwr->bus_ibs); - - icc_put(pwr->icc_path); + kgsl_bus_close(device); pm_runtime_disable(&device->pdev->dev); } diff --git a/drivers/gpu/msm/kgsl_pwrctrl.h b/drivers/gpu/msm/kgsl_pwrctrl.h index 7d2ca5d45195..ed146843d14c 100644 --- a/drivers/gpu/msm/kgsl_pwrctrl.h +++ b/drivers/gpu/msm/kgsl_pwrctrl.h @@ -21,6 +21,12 @@ #define KGSL_MAX_PWRLEVELS 10 +#define KGSL_PWRFLAGS_POWER_ON 0 +#define KGSL_PWRFLAGS_CLK_ON 1 +#define KGSL_PWRFLAGS_AXI_ON 2 +#define KGSL_PWRFLAGS_IRQ_ON 3 +#define KGSL_PWRFLAGS_NAP_OFF 5 + /* Only two supported levels, min & max */ #define KGSL_CONSTRAINT_PWR_MAXLEVELS 2 @@ -35,6 +41,7 @@ enum kgsl_pwrctrl_timer_type { }; struct platform_device; +struct icc_path; struct kgsl_clk_stats { unsigned int busy; @@ -91,7 +98,6 @@ struct kgsl_pwrlevel { * @throttle_mask - LM throttle mask * @interval_timeout - timeout in jiffies to be idle before a power event * @clock_times - Each GPU frequency's accumulated active time in us - * @pcl - bus scale identifier * @clk_stats - structure of clock statistics * @input_disable - To disable GPU wakeup on touch input event * @bus_control - true if the bus calculation is independent @@ -133,7 +139,6 @@ struct kgsl_pwrctrl { unsigned int throttle_mask; unsigned long interval_timeout; u64 clock_times[KGSL_MAX_PWRLEVELS]; - uint32_t pcl; struct kgsl_clk_stats clk_stats; bool input_disable; bool bus_control; @@ -141,15 +146,16 @@ struct kgsl_pwrctrl { unsigned int bus_percent_ab; unsigned int bus_width; unsigned long bus_ab_mbytes; - struct device *devbw; - /** @bus_ibs: List of the bus bandwidths in use by our target */ - u32 *bus_ibs; - /** @bus_ibs_count: Number of objects in @bus_ibs */ - int bus_ibs_count; + /** @ddr_table: List of the DDR bandwidths in KBps for the target */ + u32 *ddr_table; + /** @ddr_table_count: Number of objects in @ddr_table */ + int ddr_table_count; /** cur_buslevel: The last buslevel voted by the driver */ int cur_buslevel; /** @bus_max: The maximum bandwidth available to the device */ unsigned long bus_max; + /** @bus_set: Function for setting the bus constraints */ + int (*bus_set)(struct kgsl_device *device, int buslevel, u32 ab); struct kgsl_pwr_constraint constraint; bool superfast; unsigned int gpu_bimc_int_clk_freq; @@ -165,8 +171,6 @@ void kgsl_timer(struct timer_list *t); void kgsl_pre_hwaccess(struct kgsl_device *device); void kgsl_pwrctrl_pwrlevel_change(struct kgsl_device *device, unsigned int level); -void kgsl_pwrctrl_buslevel_update(struct kgsl_device *device, - bool on); int kgsl_pwrctrl_init_sysfs(struct kgsl_device *device); int kgsl_pwrctrl_change_state(struct kgsl_device *device, int state); int kgsl_clk_set_rate(struct kgsl_device *device, diff --git a/drivers/gpu/msm/kgsl_pwrscale.c b/drivers/gpu/msm/kgsl_pwrscale.c index 72f249678763..d907ea06b7e2 100644 --- a/drivers/gpu/msm/kgsl_pwrscale.c +++ b/drivers/gpu/msm/kgsl_pwrscale.c @@ -6,6 +6,7 @@ #include #include +#include "kgsl_bus.h" #include "kgsl_device.h" #include "kgsl_pwrscale.h" #include "kgsl_trace.h" @@ -586,7 +587,7 @@ int kgsl_busmon_target(struct device *dev, unsigned long *freq, u32 flags) if ((pwr->bus_mod != b) || (pwr->bus_ab_mbytes != ab_mbytes)) { pwr->bus_percent_ab = device->pwrscale.bus_profile.percent_ab; pwr->bus_ab_mbytes = ab_mbytes; - kgsl_pwrctrl_buslevel_update(device, true); + kgsl_bus_update(device, true); } mutex_unlock(&device->mutex); @@ -786,8 +787,8 @@ int kgsl_pwrscale_init(struct kgsl_device *device, struct platform_device *pdev, * the bus bandwidth vote. */ if (pwr->bus_control) { - adreno_tz_data.bus.num = pwr->bus_ibs_count; - adreno_tz_data.bus.ib_mbps = pwr->bus_ibs; + adreno_tz_data.bus.num = pwr->ddr_table_count; + adreno_tz_data.bus.ib_kbps = pwr->ddr_table; adreno_tz_data.bus.width = pwr->bus_width; if (!kgsl_of_property_read_ddrtype(device->pdev->dev.of_node, diff --git a/drivers/gpu/msm/msm_adreno_devfreq.h b/drivers/gpu/msm/msm_adreno_devfreq.h index 9fec1d613442..61264c170583 100644 --- a/drivers/gpu/msm/msm_adreno_devfreq.h +++ b/drivers/gpu/msm/msm_adreno_devfreq.h @@ -55,7 +55,7 @@ struct devfreq_msm_adreno_tz_data { u32 *down; s32 *p_up; s32 *p_down; - u32 *ib_mbps; + u32 *ib_kbps; bool floating; } bus; unsigned int device_id; From cac5edff49a2ceb1a2e64aca0c9b595c5adc86be Mon Sep 17 00:00:00 2001 From: Jordan Crouse Date: Mon, 14 Oct 2019 10:41:07 -0600 Subject: [PATCH 2/4] msm: kgsl: Generate TCS votes to send to the GMU With the demise of msm_bus_scale we lost the API to generate TCS commands so we need to do it ourselves. Change-Id: Ic0dedbadad9bead0508febcb942e5bb1bd6cb465 Signed-off-by: Jordan Crouse --- drivers/gpu/msm/kgsl_bus.c | 2 +- drivers/gpu/msm/kgsl_gmu.c | 326 ++++++++++++++++++++++++++----------- drivers/gpu/msm/kgsl_gmu.h | 9 - 3 files changed, 229 insertions(+), 108 deletions(-) diff --git a/drivers/gpu/msm/kgsl_bus.c b/drivers/gpu/msm/kgsl_bus.c index 74c9e99b43d8..106961e5604c 100644 --- a/drivers/gpu/msm/kgsl_bus.c +++ b/drivers/gpu/msm/kgsl_bus.c @@ -30,7 +30,7 @@ static int interconnect_bus_set(struct kgsl_device *device, int level, struct kgsl_pwrctrl *pwr = &device->pwrctrl; icc_set_bw(pwr->icc_path, MBps_to_icc(ab), - KBps_to_icc(pwr->ddr_table[level])); + kBps_to_icc(pwr->ddr_table[level])); return 0; } diff --git a/drivers/gpu/msm/kgsl_gmu.c b/drivers/gpu/msm/kgsl_gmu.c index 551b767ade04..6db4433fd584 100644 --- a/drivers/gpu/msm/kgsl_gmu.c +++ b/drivers/gpu/msm/kgsl_gmu.c @@ -14,8 +14,10 @@ #include #include #include +#include #include "adreno.h" +#include "kgsl_bus.h" #include "kgsl_device.h" #include "kgsl_gmu.h" #include "kgsl_util.h" @@ -699,7 +701,7 @@ static int rpmh_arc_votes_init(struct kgsl_device *device, num_freqs = gmu->num_gpupwrlevels; - if (num_freqs > pri_rail->num) { + if (num_freqs > pri_rail->num || num_freqs > ARRAY_SIZE(vlvl_tbl)) { dev_err(&gmu->pdev->dev, "Defined more GPU DCVS levels than RPMh can support\n"); return -EINVAL; @@ -713,132 +715,260 @@ static int rpmh_arc_votes_init(struct kgsl_device *device, sec_rail, vlvl_tbl, num_freqs); } -/* - * build_rpmh_bw_votes() - build TCS commands to vote for bandwidth. - * Each command sets frequency of a node along path to DDR or CNOC. - * @rpmh_vote: Pointer to RPMh vote needed by GMU to set BW via RPMh - * @num_usecases: Number of BW use cases (or BW levels) - * @handle: Provided by bus driver. It contains TCS command sets for - * all BW use cases of a bus client. - */ -static void build_rpmh_bw_votes(struct gmu_bw_votes *rpmh_vote, - unsigned int num_usecases, struct msm_bus_tcs_handle handle) -{ - struct msm_bus_tcs_usecase *tmp; - int i, j; +struct bcm { + const char *name; + u32 buswidth; + u32 channels; + u32 unit; + u16 width; + u8 vcd; + bool fixed; +}; - for (i = 0; i < num_usecases; i++) { - tmp = &handle.usecases[i]; - for (j = 0; j < tmp->num_cmds; j++) { - if (!i) { - /* - * Wait bitmask and TCS command addresses are - * same for all bw use cases. To save data volume - * exchanged between driver and GMU, only - * transfer bitmasks and TCS command addresses - * of first set of bw use case - */ - rpmh_vote->cmds_per_bw_vote = tmp->num_cmds; - rpmh_vote->cmds_wait_bitmask = - tmp->cmds[j].wait ? - rpmh_vote->cmds_wait_bitmask - | BIT(i) - : rpmh_vote->cmds_wait_bitmask - & (~BIT(i)); - rpmh_vote->cmd_addrs[j] = tmp->cmds[j].addr; - } - rpmh_vote->cmd_data[i][j] = tmp->cmds[j].data; +/* + * List of Bus Control Modules (BCMs) that need to be configured for the GPU + * to access DDR. For each bus level we will generate a vote each BC + */ +static struct bcm a660_ddr_bcms[] = { + { .name = "SH0", .buswidth = 16 }, + { .name = "MC0", .buswidth = 4 }, + { .name = "ACV", .fixed = true }, +}; + +/* Same as above, but for the CNOC BCMs */ +static struct bcm a660_cnoc_bcms[] = { + { .name = "CN0", .buswidth = 4 }, +}; + +/* Generate a set of bandwidth votes for the list of BCMs */ +static void tcs_cmd_data(struct bcm *bcms, int count, u32 ab, u32 ib, + u32 *data) +{ + int i; + + for (i = 0; i < count; i++) { + bool valid = true; + bool commit = false; + u64 avg, peak, x, y; + + if (i == count - 1 || bcms[i].vcd != bcms[i + 1].vcd) + commit = true; + + /* + * On a660, the "ACV" y vote should be 0x08 if there is a valid + * vote and 0x00 if not. This is kind of hacky and a660 specific + * but we can clean it up when we add a new target + */ + if (bcms[i].fixed) { + if (!ab && !ib) + data[i] = BCM_TCS_CMD(commit, false, 0x0, 0x0); + else + data[i] = BCM_TCS_CMD(commit, true, 0x0, 0x8); + continue; } + + /* Multiple the bandwidth by the width of the connection */ + avg = ((u64) ab) * bcms[i].width; + + /* And then divide by the total width across channels */ + do_div(avg, bcms[i].buswidth * bcms[i].channels); + + peak = ((u64) ib) * bcms[i].width; + do_div(peak, bcms[i].buswidth); + + /* Input bandwidth value is in KBps */ + x = avg * 1000ULL; + do_div(x, bcms[i].unit); + + /* Input bandwidth value is in KBps */ + y = peak * 1000ULL; + do_div(y, bcms[i].unit); + + /* + * If a bandwidth value was specified but the calculation ends + * rounding down to zero, set a minimum level + */ + if (ab && x == 0) + x = 1; + + if (ib && y == 0) + y = 1; + + x = min_t(u64, x, BCM_TCS_CMD_VOTE_MASK); + y = min_t(u64, y, BCM_TCS_CMD_VOTE_MASK); + + if (!x && !y) + valid = false; + + data[i] = BCM_TCS_CMD(commit, valid, x, y); } } -static void build_bwtable_cmd_cache(struct gmu_device *gmu) +struct bcm_data { + __le32 unit; + __le16 width; + u8 vcd; + u8 reserved; +}; + +struct rpmh_bw_votes { + u32 wait_bitmask; + u32 num_cmds; + u32 *addrs; + u32 num_levels; + u32 **cmds; +}; + +static void free_rpmh_bw_votes(struct rpmh_bw_votes *votes) +{ + int i; + + if (!votes) + return; + + for (i = 0; votes->cmds && i < votes->num_levels; i++) + kfree(votes->cmds[i]); + + kfree(votes->cmds); + kfree(votes->addrs); + kfree(votes); +} + +/* Build the votes table from the specified bandwidth levels */ +static struct rpmh_bw_votes *build_rpmh_bw_votes(struct bcm *bcms, + int bcm_count, u32 *levels, int levels_count) +{ + struct rpmh_bw_votes *votes; + int i; + + votes = kzalloc(sizeof(*votes), GFP_KERNEL); + if (!votes) + return ERR_PTR(-ENOMEM); + + votes->addrs = kcalloc(bcm_count, sizeof(*votes->cmds), GFP_KERNEL); + if (!votes->addrs) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + votes->cmds = kcalloc(levels_count, sizeof(*votes->cmds), GFP_KERNEL); + if (!votes->cmds) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + votes->num_cmds = bcm_count; + votes->num_levels = levels_count; + + /* Get the cmd-db information for each BCM */ + for (i = 0; i < bcm_count; i++) { + size_t l; + const struct bcm_data *data; + + data = cmd_db_read_aux_data(bcms[i].name, &l); + + votes->addrs[i] = cmd_db_read_addr(bcms[i].name); + + bcms[i].unit = le32_to_cpu(data->unit); + bcms[i].width = le16_to_cpu(data->width); + bcms[i].vcd = data->vcd; + } + + for (i = 0; i < bcm_count; i++) { + if (i == (bcm_count - 1) || bcms[i].vcd != bcms[i + 1].vcd) + votes->wait_bitmask |= (1 << i); + } + + for (i = 0; i < levels_count; i++) { + votes->cmds[i] = kcalloc(bcm_count, sizeof(u32), GFP_KERNEL); + if (!votes->cmds[i]) { + free_rpmh_bw_votes(votes); + return ERR_PTR(-ENOMEM); + } + + tcs_cmd_data(bcms, bcm_count, 0, levels[i], votes->cmds[i]); + } + + return votes; +} + +static void build_bwtable_cmd_cache(struct hfi_bwtable_cmd *cmd, + struct rpmh_bw_votes *ddr, struct rpmh_bw_votes *cnoc) { - struct hfi_bwtable_cmd *cmd = &gmu->hfi.bwtbl_cmd; - struct rpmh_votes_t *votes = &gmu->rpmh_votes; unsigned int i, j; cmd->hdr = 0xFFFFFFFF; - cmd->bw_level_num = gmu->num_bwlevels; - cmd->cnoc_cmds_num = votes->cnoc_votes.cmds_per_bw_vote; - cmd->cnoc_wait_bitmask = votes->cnoc_votes.cmds_wait_bitmask; - cmd->ddr_cmds_num = votes->ddr_votes.cmds_per_bw_vote; - cmd->ddr_wait_bitmask = votes->ddr_votes.cmds_wait_bitmask; + cmd->bw_level_num = ddr->num_levels; + cmd->ddr_cmds_num = ddr->num_cmds; + cmd->ddr_wait_bitmask = ddr->wait_bitmask; - for (i = 0; i < cmd->ddr_cmds_num; i++) - cmd->ddr_cmd_addrs[i] = votes->ddr_votes.cmd_addrs[i]; + for (i = 0; i < ddr->num_cmds; i++) + cmd->ddr_cmd_addrs[i] = ddr->addrs[i]; - for (i = 0; i < cmd->bw_level_num; i++) - for (j = 0; j < cmd->ddr_cmds_num; j++) - cmd->ddr_cmd_data[i][j] = - votes->ddr_votes.cmd_data[i][j]; + for (i = 0; i < ddr->num_levels; i++) + for (j = 0; j < ddr->num_cmds; j++) + cmd->ddr_cmd_data[i][j] = (u32) ddr->cmds[i][j]; - for (i = 0; i < cmd->cnoc_cmds_num; i++) - cmd->cnoc_cmd_addrs[i] = - votes->cnoc_votes.cmd_addrs[i]; + if (!cnoc) + return; - for (i = 0; i < MAX_CNOC_LEVELS; i++) - for (j = 0; j < cmd->cnoc_cmds_num; j++) - cmd->cnoc_cmd_data[i][j] = - votes->cnoc_votes.cmd_data[i][j]; + cmd->cnoc_cmds_num = cnoc->num_cmds; + cmd->cnoc_wait_bitmask = cnoc->wait_bitmask; + + for (i = 0; i < cnoc->num_cmds; i++) + cmd->cnoc_cmd_addrs[i] = cnoc->addrs[i]; + + for (i = 0; i < cnoc->num_levels; i++) + for (j = 0; j < cnoc->num_cmds; j++) + cmd->cnoc_cmd_data[i][j] = (u32) cnoc->cmds[i][j]; } -/* - * gmu_bus_vote_init - initialized RPMh votes needed for bw scaling by GMU. - * @gmu: Pointer to GMU device - * @pwr: Pointer to KGSL power controller - */ -static int gmu_bus_vote_init(struct gmu_device *gmu, struct kgsl_pwrctrl *pwr) +static int gmu_bus_vote_init(struct kgsl_device *device) { - struct msm_bus_tcs_usecase *usecases; - struct msm_bus_tcs_handle hdl; - struct rpmh_votes_t *votes = &gmu->rpmh_votes; - int ret; + struct gmu_device *gmu = KGSL_GMU_DEVICE(device); + struct kgsl_pwrctrl *pwr = &device->pwrctrl; + struct rpmh_bw_votes *ddr, *cnoc = NULL; + u32 *cnoc_table; + u32 count; - if (!gmu->num_bwlevels) - return 0; + /* Build the DDR votes */ + ddr = build_rpmh_bw_votes(a660_ddr_bcms, ARRAY_SIZE(a660_ddr_bcms), + pwr->ddr_table, pwr->ddr_table_count); + if (IS_ERR(ddr)) + return PTR_ERR(ddr); - usecases = kcalloc(gmu->num_bwlevels, sizeof(*usecases), GFP_KERNEL); - if (!usecases) - return -ENOMEM; + /* Get the CNOC table */ + cnoc_table = kgsl_bus_get_table(device->pdev, "qcom,bus-table-cnoc", + &count); - hdl.num_usecases = gmu->num_bwlevels; - hdl.usecases = usecases; + /* And build the votes for that, if it exists */ + if (count > 0) + cnoc = build_rpmh_bw_votes(a660_cnoc_bcms, + ARRAY_SIZE(a660_cnoc_bcms), cnoc_table, count); + kfree(cnoc_table); - /* - * Query TCS command set for each use case defined in GPU b/w table - */ - ret = msm_bus_scale_query_tcs_cmd_all(&hdl, gmu->pcl); - if (ret) - goto out; + if (IS_ERR(cnoc)) { + free_rpmh_bw_votes(ddr); + return PTR_ERR(cnoc); + } - build_rpmh_bw_votes(&votes->ddr_votes, gmu->num_bwlevels, hdl); + /* Build the HFI command once */ + build_bwtable_cmd_cache(&gmu->hfi.bwtbl_cmd, ddr, cnoc); - /* - *Query CNOC TCS command set for each use case defined in cnoc bw table - */ - ret = msm_bus_scale_query_tcs_cmd_all(&hdl, gmu->ccl); - if (ret) - goto out; + free_rpmh_bw_votes(ddr); + free_rpmh_bw_votes(cnoc); - build_rpmh_bw_votes(&votes->cnoc_votes, gmu->num_cnocbwlevels, hdl); - - build_bwtable_cmd_cache(gmu); - -out: - kfree(usecases); - - return ret; + return 0; } static int gmu_rpmh_init(struct kgsl_device *device, - struct gmu_device *gmu, struct kgsl_pwrctrl *pwr) + struct gmu_device *gmu) { struct rpmh_arc_vals gfx_arc, cx_arc, mx_arc; int ret; /* Initialize BW tables */ - ret = gmu_bus_vote_init(gmu, pwr); + ret = gmu_bus_vote_init(device); if (ret) return ret; @@ -1259,7 +1389,7 @@ static int gmu_probe(struct kgsl_device *device, struct device_node *node) gmu->icc_path = of_icc_get(&gmu->pdev->dev, NULL); /* Populates RPMh configurations */ - ret = gmu_rpmh_init(device, gmu, pwr); + ret = gmu_rpmh_init(device, gmu); if (ret) goto error; @@ -1422,7 +1552,7 @@ static int gmu_start(struct kgsl_device *device) struct gmu_dev_ops *gmu_dev_ops = GMU_DEVICE_OPS(device); struct kgsl_pwrctrl *pwr = &device->pwrctrl; struct gmu_device *gmu = KGSL_GMU_DEVICE(device); - unsigned long ib; + int level; switch (device->state) { case KGSL_STATE_INIT: diff --git a/drivers/gpu/msm/kgsl_gmu.h b/drivers/gpu/msm/kgsl_gmu.h index eff99adcef59..73b82e4aee08 100644 --- a/drivers/gpu/msm/kgsl_gmu.h +++ b/drivers/gpu/msm/kgsl_gmu.h @@ -118,18 +118,9 @@ struct gmu_memdesc { enum gmu_context_index ctx_idx; }; -struct gmu_bw_votes { - uint32_t cmds_wait_bitmask; - uint32_t cmds_per_bw_vote; - uint32_t cmd_addrs[MAX_BW_CMDS]; - uint32_t cmd_data[MAX_GX_LEVELS][MAX_BW_CMDS]; -}; - struct rpmh_votes_t { uint32_t gx_votes[MAX_GX_LEVELS]; uint32_t cx_votes[MAX_CX_LEVELS]; - struct gmu_bw_votes ddr_votes; - struct gmu_bw_votes cnoc_votes; }; enum gmu_load_mode { From 71d4d5cb94a5f74031dfd0c87cf87d6d2e42ab95 Mon Sep 17 00:00:00 2001 From: Harshdeep Dhatt Date: Mon, 18 Nov 2019 10:07:18 -0800 Subject: [PATCH 3/4] msm: kgsl: bus dcvs fixes Remove the unused bus.mod member and make the bus governor compile in the new directory structure. Change-Id: I8c9bf21bb3b260cc932e1a0e1835788905cd67b5 Signed-off-by: Harshdeep Dhatt --- drivers/gpu/msm/governor_gpubw_mon.c | 18 +++++++++------- drivers/gpu/msm/governor_msm_adreno_tz.c | 26 ++++++++++++++---------- drivers/gpu/msm/kgsl_pwrscale.c | 1 - drivers/gpu/msm/msm_adreno_devfreq.h | 1 - 4 files changed, 26 insertions(+), 20 deletions(-) diff --git a/drivers/gpu/msm/governor_gpubw_mon.c b/drivers/gpu/msm/governor_gpubw_mon.c index 2f75049f1234..f133b37fb4ca 100644 --- a/drivers/gpu/msm/governor_gpubw_mon.c +++ b/drivers/gpu/msm/governor_gpubw_mon.c @@ -5,11 +5,11 @@ #include #include -#include #include #include "devfreq_trace.h" -#include "governor.h" +#include "../../devfreq/governor.h" +#include "msm_adreno_devfreq.h" #define MIN_BUSY 1000 #define LONG_FLOOR 50000 @@ -95,7 +95,7 @@ static int devfreq_gpubw_get_target(struct devfreq *df, } else { /* GPU votes for IB not AB so don't under vote the system */ norm_cycles = (100 * norm_cycles) / TARGET; - act_level = b.buslevel + b.mod; + act_level = b.buslevel; act_level = (act_level < 0) ? 0 : act_level; act_level = (act_level >= priv->bus.num) ? (priv->bus.num - 1) : act_level; @@ -157,16 +157,20 @@ static int gpubw_start(struct devfreq *devfreq) /* Set up the cut-over percentages for the bus calculation. */ for (i = 0; i < priv->bus.num; i++) { - t1 = (u32)(100 * priv->bus.ib[i]) / - (u32)priv->bus.ib[priv->bus.num - 1]; + t1 = (u32)(100 * priv->bus.ib_kbps[i]) / + (u32)priv->bus.ib_kbps[priv->bus.num - 1]; priv->bus.p_up[i] = t1 - HIST; priv->bus.p_down[i] = t2 - 2 * HIST; t2 = t1; } /* Set the upper-most and lower-most bounds correctly. */ priv->bus.p_down[0] = 0; - priv->bus.p_down[1] = (priv->bus.p_down[1] > (2 * HIST)) ? - priv->bus.p_down[1] : (2 * HIST); + + for (i = 0; i < priv->bus.num; i++) { + if (priv->bus.p_down[i] < 2 * HIST) + priv->bus.p_down[i] = 2 * HIST; + } + if (priv->bus.num >= 1) priv->bus.p_up[priv->bus.num - 1] = 100; _update_cutoff(priv, priv->bus.max); diff --git a/drivers/gpu/msm/governor_msm_adreno_tz.c b/drivers/gpu/msm/governor_msm_adreno_tz.c index 8ecd1efe9fc7..0a363218112f 100644 --- a/drivers/gpu/msm/governor_msm_adreno_tz.c +++ b/drivers/gpu/msm/governor_msm_adreno_tz.c @@ -5,17 +5,19 @@ #include #include #include +#include #include #include #include #include #include #include -#include #include #include -#include -#include "governor.h" +#include + +#include "../../devfreq/governor.h" +#include "msm_adreno_devfreq.h" static DEFINE_SPINLOCK(tz_lock); static DEFINE_SPINLOCK(sample_lock); @@ -197,7 +199,8 @@ static int __secure_tz_update_entry3(int level, s64 total_time, s64 busy_time, return ret; } -static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) +static int tz_init_ca(struct device *dev, + struct devfreq_msm_adreno_tz_data *priv) { unsigned int tz_ca_data[2]; phys_addr_t paddr; @@ -226,8 +229,8 @@ static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) memcpy(tz_buf, tz_ca_data, sizeof(tz_ca_data)); /* Ensure memcpy completes execution */ mb(); - dmac_flush_range(tz_buf, - tz_buf + PAGE_ALIGN(sizeof(tz_ca_data))); + dma_sync_single_for_device(dev, paddr, + PAGE_ALIGN(sizeof(tz_ca_data)), DMA_BIDIRECTIONAL); ret = qcom_scm_dcvs_init_ca_v2(paddr, sizeof(tz_ca_data)); @@ -239,7 +242,7 @@ static int tz_init_ca(struct devfreq_msm_adreno_tz_data *priv) return ret; } -static int tz_init(struct devfreq_msm_adreno_tz_data *priv, +static int tz_init(struct device *dev, struct devfreq_msm_adreno_tz_data *priv, unsigned int *tz_pwrlevels, u32 size_pwrlevels, unsigned int *version, u32 size_version) { @@ -268,7 +271,8 @@ static int tz_init(struct devfreq_msm_adreno_tz_data *priv, memcpy(tz_buf, tz_pwrlevels, size_pwrlevels); /* Ensure memcpy completes execution */ mb(); - dmac_flush_range(tz_buf, tz_buf + PAGE_ALIGN(size_pwrlevels)); + dma_sync_single_for_device(dev, paddr, + PAGE_ALIGN(size_pwrlevels), DMA_BIDIRECTIONAL); ret = qcom_scm_dcvs_init_v2(paddr, size_pwrlevels, version); if (!ret) @@ -283,7 +287,7 @@ static int tz_init(struct devfreq_msm_adreno_tz_data *priv, /* Initialize context aware feature, if enabled. */ if (!ret && priv->ctxt_aware_enable) { if (priv->is_64 && qcom_scm_dcvs_ca_available()) { - ret = tz_init_ca(priv); + ret = tz_init_ca(dev, priv); /* * If context aware feature initialization fails, * just print an error message and return @@ -455,8 +459,8 @@ static int tz_start(struct devfreq *devfreq) INIT_WORK(&gpu_profile->partner_resume_event_ws, do_partner_resume_event); - ret = tz_init(priv, tz_pwrlevels, sizeof(tz_pwrlevels), &version, - sizeof(version)); + ret = tz_init(&devfreq->dev, priv, tz_pwrlevels, sizeof(tz_pwrlevels), + &version, sizeof(version)); if (ret != 0 || version > MAX_TZ_VERSION) { pr_err(TAG "tz_init failed\n"); return ret; diff --git a/drivers/gpu/msm/kgsl_pwrscale.c b/drivers/gpu/msm/kgsl_pwrscale.c index d907ea06b7e2..6820fe6a49a0 100644 --- a/drivers/gpu/msm/kgsl_pwrscale.c +++ b/drivers/gpu/msm/kgsl_pwrscale.c @@ -390,7 +390,6 @@ int kgsl_devfreq_get_dev_status(struct device *dev, last_b->ram_time = device->pwrscale.accum_stats.ram_time; last_b->ram_wait = device->pwrscale.accum_stats.ram_wait; - last_b->mod = device->pwrctrl.bus_mod; last_b->buslevel = device->pwrctrl.cur_buslevel; } diff --git a/drivers/gpu/msm/msm_adreno_devfreq.h b/drivers/gpu/msm/msm_adreno_devfreq.h index 61264c170583..0c06f3523b84 100644 --- a/drivers/gpu/msm/msm_adreno_devfreq.h +++ b/drivers/gpu/msm/msm_adreno_devfreq.h @@ -31,7 +31,6 @@ int kgsl_devfreq_del_notifier(struct device *device, struct xstats { u64 ram_time; u64 ram_wait; - int mod; int buslevel; }; From dc9f5d87f5a76265242a67824f80b5698065f794 Mon Sep 17 00:00:00 2001 From: Jordan Crouse Date: Mon, 14 Oct 2019 10:41:07 -0600 Subject: [PATCH 4/4] msm: kgsl: Add an option to always enable I/O coherency Add a Kconfig option to allow the driver to mark all memory buffers as I/O coherent by default if the target supports I/O coherency. This is a config option for now because I/O coherency has traditionally been a little fickle and it represents a paradigm shift to enable it universally. Eventually the goal is to leave it on by default and invert the polarity of this option. In conjunction we can use this option to manage kernel targets that do not not support fine grained cache operations. Because it is impossible to stop the user from creating and using cached surfaces the driver has to either support I/O coherency by default OR have access to fine grained cache operations. We can take advantage of the new Kconfig option to make some compile time decisions and compile out the cache code if it isn't supported. This isn't 100% foolproof though. If your target doesn't support I/O coherency and there are no cache operations available you are out of luck, so as it stands this precludes cached operations on a3xx and a5xx with this kernel. We will still allow cached surfaces but they will never be coherent. If this becomes a problem we'll need to figure out a way to disallow cached surfaces on those targets. Change-Id: Ic0dedbad7d923b54f64f34cd68a59c3522ca5ee9 Signed-off-by: Jordan Crouse --- drivers/gpu/msm/Kconfig | 10 +++++++ drivers/gpu/msm/kgsl.c | 27 +++++++++++++++--- drivers/gpu/msm/kgsl_sharedmem.c | 47 ++++++++++++++++++++++++-------- 3 files changed, 68 insertions(+), 16 deletions(-) diff --git a/drivers/gpu/msm/Kconfig b/drivers/gpu/msm/Kconfig index 5f87cba2c9ea..e573fd8a0ba2 100644 --- a/drivers/gpu/msm/Kconfig +++ b/drivers/gpu/msm/Kconfig @@ -41,3 +41,13 @@ config QCOM_KGSL_CORESIGHT When enabled, the Adreno GPU is available as a source for Coresight data. On a6xx targets there are two sources available for the GX and CX domains respectively. Debug kernels should say 'Y' here. + +config QCOM_KGSL_IOCOHERENCY_DEFAULT + bool "Enable I/O coherency on cached GPU memory by default" + depends on QCOM_KGSL + default y if ARCH_LAHAINA + help + Say 'Y' here to enable I/O cache coherency by default on targets that + support hardware I/O coherency. If enabled all cached GPU memory + will use I/O coherency regardless of the user flags. If not enabled + the user can still selectively enable I/O coherency with a flag. diff --git a/drivers/gpu/msm/kgsl.c b/drivers/gpu/msm/kgsl.c index 9975e1cef2f4..befe80600e6a 100644 --- a/drivers/gpu/msm/kgsl.c +++ b/drivers/gpu/msm/kgsl.c @@ -2391,6 +2391,14 @@ static void _setup_cache_mode(struct kgsl_mem_entry *entry, entry->memdesc.flags |= (mode << KGSL_CACHEMODE_SHIFT); } +static bool is_cached(u64 flags) +{ + u32 mode = (flags & KGSL_CACHEMODE_MASK) >> KGSL_CACHEMODE_SHIFT; + + return (mode != KGSL_CACHEMODE_UNCACHED && + mode != KGSL_CACHEMODE_WRITECOMBINE); +} + static int kgsl_setup_dma_buf(struct kgsl_device *device, struct kgsl_pagetable *pagetable, struct kgsl_mem_entry *entry, @@ -2455,6 +2463,11 @@ static int kgsl_setup_dmabuf_useraddr(struct kgsl_device *device, /* Setup the user addr/cache mode for cache operations */ entry->memdesc.useraddr = hostptr; _setup_cache_mode(entry, vma); + + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(entry->memdesc.flags)) + entry->memdesc.flags |= KGSL_MEMFLAGS_IOCOHERENT; + up_read(¤t->mm->mmap_sem); return 0; } @@ -2979,7 +2992,6 @@ static int _kgsl_gpumem_sync_cache(struct kgsl_mem_entry *entry, { int ret = 0; int cacheop; - int mode; if (!entry) return 0; @@ -3010,9 +3022,7 @@ static int _kgsl_gpumem_sync_cache(struct kgsl_mem_entry *entry, length = entry->memdesc.size; } - mode = kgsl_memdesc_get_cachemode(&entry->memdesc); - if (mode != KGSL_CACHEMODE_UNCACHED - && mode != KGSL_CACHEMODE_WRITECOMBINE) { + if (is_cached(entry->memdesc.flags)) { trace_kgsl_mem_sync_cache(entry, offset, length, op); ret = kgsl_cache_range_op(&entry->memdesc, offset, length, cacheop); @@ -3319,6 +3329,10 @@ struct kgsl_mem_entry *gpumem_alloc_entry( if (entry == NULL) return ERR_PTR(-ENOMEM); + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(flags)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + ret = kgsl_allocate_user(dev_priv->device, &entry->memdesc, size, flags, 0); if (ret != 0) @@ -3540,6 +3554,11 @@ long kgsl_ioctl_sparse_phys_alloc(struct kgsl_device_private *dev_priv, ((ilog2(param->pagesize) << KGSL_MEMALIGN_SHIFT) & KGSL_MEMALIGN_MASK); + + if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) && + is_cached(flags)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + ret = kgsl_allocate_user(dev_priv->device, &entry->memdesc, param->size, flags, 0); if (ret) diff --git a/drivers/gpu/msm/kgsl_sharedmem.c b/drivers/gpu/msm/kgsl_sharedmem.c index bebdda88ef3f..6fec82642dcc 100644 --- a/drivers/gpu/msm/kgsl_sharedmem.c +++ b/drivers/gpu/msm/kgsl_sharedmem.c @@ -13,6 +13,15 @@ #include "kgsl_pool.h" #include "kgsl_sharedmem.h" +/* + * For now, we either need the low level cache operations or + * QCOM_KGSL_IOCOHERENCY_DEFAULT enabled because we can't stop userspace + * from expecting to enable cached surfaces and have them work + */ +#if !defined(dmac_flush_range) && !IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT) +#error "KGSL needs either dmac_flush_range or CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT enabled" +#endif + /* * The user can set this from debugfs to force failed memory allocations to * fail without trying OOM first. This is a debug setting useful for @@ -512,7 +521,8 @@ static inline unsigned int _fixup_cache_range_op(unsigned int op) } #endif -static inline void _cache_op(unsigned int op, +#ifdef dmac_flush_range +static void _cache_op(unsigned int op, const void *start, const void *end) { /* @@ -532,8 +542,8 @@ static inline void _cache_op(unsigned int op, } } -static int kgsl_do_cache_op(struct page *page, void *addr, - uint64_t offset, uint64_t size, unsigned int op) +static void kgsl_do_cache_op(struct page *page, void *addr, u64 offset, + u64 size, unsigned int op) { if (page != NULL) { unsigned long pfn = page_to_pfn(page) + offset / PAGE_SIZE; @@ -562,15 +572,21 @@ static int kgsl_do_cache_op(struct page *page, void *addr, offset = 0; } while (size); - return 0; + return; } addr = page_address(page); } _cache_op(op, addr + offset, addr + offset + (size_t) size); - return 0; } +#else + +static void kgsl_do_cache_op(struct page *page, void *addr, u64 offset, + u64 size, unsigned int op) +{ +} +#endif int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, uint64_t size, unsigned int op) @@ -579,7 +595,9 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, struct sg_table *sgt = NULL; struct scatterlist *sg; unsigned int i, pos = 0; - int ret = 0; + + if (memdesc->flags & KGSL_MEMFLAGS_IOCOHERENT) + return 0; if (size == 0 || size > UINT_MAX) return -EINVAL; @@ -598,8 +616,8 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, if (addr + ((size_t) offset + (size_t) size) < addr) return -ERANGE; - ret = kgsl_do_cache_op(NULL, addr, offset, size, op); - return ret; + kgsl_do_cache_op(NULL, addr, offset, size, op); + return 0; } /* @@ -610,7 +628,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, sgt = memdesc->sgt; else { if (memdesc->pages == NULL) - return ret; + return 0; sgt = kgsl_alloc_sgt_from_pages(memdesc); if (IS_ERR(sgt)) @@ -627,7 +645,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, sg_offset = offset > pos ? offset - pos : 0; sg_left = (sg->length - sg_offset > size) ? size : sg->length - sg_offset; - ret = kgsl_do_cache_op(sg_page(sg), NULL, sg_offset, + kgsl_do_cache_op(sg_page(sg), NULL, sg_offset, sg_left, op); size -= sg_left; if (size == 0) @@ -638,7 +656,7 @@ int kgsl_cache_range_op(struct kgsl_memdesc *memdesc, uint64_t offset, if (memdesc->sgt == NULL) kgsl_free_sgt(sgt); - return ret; + return 0; } void kgsl_memdesc_init(struct kgsl_device *device, @@ -657,9 +675,14 @@ void kgsl_memdesc_init(struct kgsl_device *device, flags &= ~((uint64_t) KGSL_MEMFLAGS_USE_CPU_MAP); /* Disable IO coherence if it is not supported on the chip */ - if (!MMU_FEATURE(mmu, KGSL_MMU_IO_COHERENT)) + if (!MMU_FEATURE(mmu, KGSL_MMU_IO_COHERENT)) { flags &= ~((uint64_t) KGSL_MEMFLAGS_IOCOHERENT); + WARN_ONCE(IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT), + "I/O coherency is not supported on this target\n"); + } else if (IS_ENABLED(CONFIG_QCOM_KGSL_IOCOHERENCY_DEFAULT)) + flags |= KGSL_MEMFLAGS_IOCOHERENT; + if (MMU_FEATURE(mmu, KGSL_MMU_NEED_GUARD_PAGE)) memdesc->priv |= KGSL_MEMDESC_GUARD_PAGE;