From 1aace2908a1986bcc1b24a0590a588ac3f058b01 Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Wed, 16 Jan 2019 20:25:33 -0800 Subject: [PATCH 01/27] PM / devfreq: memlat: Add support for compute-bound logic Currently, an entirely separate governor is used to handle compute-bound cases. Roll this functionality into the existing memlat governor and allow devices to choose which logic to target by specifying different drivers in the 'compatible' field. Change-Id: I8dc41bf0474309c3564dec8e6c813511b8a7fc02 Signed-off-by: Jonathan Avila [avajid@codeaurora.org: renamed compatible and resolved trivial merge conflicts] Signed-off-by: Amir Vajid --- drivers/devfreq/arm-memlat-mon.c | 46 +++++++++++---- drivers/devfreq/governor_memlat.c | 95 +++++++++++++++++++++++++------ drivers/devfreq/governor_memlat.h | 8 ++- 3 files changed, 119 insertions(+), 30 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index cbc51baf2991..2c888327cb4a 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -23,6 +23,7 @@ #include "governor.h" #include "governor_memlat.h" #include +#include enum ev_index { INST_IDX, @@ -52,6 +53,10 @@ struct cpu_grp_info { struct memlat_hwmon hw; }; +struct memlat_mon_spec { + bool is_compute; +}; + #define to_cpustats(cpu_grp, cpu) \ (&cpu_grp->cpustats[cpu - cpumask_first(&cpu_grp->cpus)]) #define to_devstats(cpu_grp, cpu) \ @@ -83,6 +88,9 @@ static inline unsigned long read_event(struct event_data *event) unsigned long ev_count; u64 total, enabled, running; + if (!event->pevent) + return 0; + total = perf_event_read_value(event->pevent, &enabled, &running); ev_count = total - event->prev_count; event->prev_count = total; @@ -249,6 +257,7 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) struct device *dev = &pdev->dev; struct memlat_hwmon *hw; struct cpu_grp_info *cpu_grp; + const struct memlat_mon_spec *spec; int cpu, ret; u32 event_id; @@ -282,6 +291,22 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) cpu_grp->event_ids[CYC_IDX] = CYC_EV; + for_each_cpu(cpu, &cpu_grp->cpus) + to_devstats(cpu_grp, cpu)->id = cpu; + + hw->start_hwmon = &start_hwmon; + hw->stop_hwmon = &stop_hwmon; + hw->get_cnt = &get_cnt; + + spec = of_device_get_match_data(dev); + if (spec && spec->is_compute) { + ret = register_compute(dev, hw); + if (ret) + pr_err("Compute Gov registration failed\n"); + + return ret; + } + ret = of_property_read_u32(dev->of_node, "qcom,cachemiss-ev", &event_id); if (ret < 0) { @@ -306,24 +331,21 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) else cpu_grp->event_ids[STALL_CYC_IDX] = event_id; - for_each_cpu(cpu, &cpu_grp->cpus) - to_devstats(cpu_grp, cpu)->id = cpu; - - hw->start_hwmon = &start_hwmon; - hw->stop_hwmon = &stop_hwmon; - hw->get_cnt = &get_cnt; - ret = register_memlat(dev, hw); - if (ret < 0) { + if (ret < 0) pr_err("Mem Latency Gov registration failed: %d\n", ret); - return ret; - } - return 0; + return ret; } +static const struct memlat_mon_spec spec[] = { + [0] = { false }, + [1] = { true }, +}; + static const struct of_device_id memlat_match_table[] = { - { .compatible = "qcom,arm-memlat-mon" }, + { .compatible = "qcom,arm-memlat-mon", .data = &spec[0] }, + { .compatible = "qcom,arm-compute-mon", .data = &spec[1] }, {} }; diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index 9f1186542cea..f7d213a27983 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -40,7 +40,8 @@ struct memlat_node { static LIST_HEAD(memlat_list); static DEFINE_MUTEX(list_lock); -static int use_cnt; +static int memlat_use_cnt; +static int compute_use_cnt; static DEFINE_MUTEX(state_lock); #define show_attr(name) \ @@ -227,8 +228,7 @@ static int devfreq_memlat_get_freq(struct devfreq *df, if (hw->core_stats[i].mem_count) ratio /= hw->core_stats[i].mem_count; - if (!hw->core_stats[i].inst_count - || !hw->core_stats[i].freq) + if (!hw->core_stats[i].freq) continue; trace_memlat_dev_meas(dev_name(df->dev.parent), @@ -271,16 +271,26 @@ show_attr(stall_floor); store_attr(stall_floor, 0U, 100U); static DEVICE_ATTR_RW(stall_floor); -static struct attribute *dev_attr[] = { +static struct attribute *memlat_dev_attr[] = { &dev_attr_ratio_ceil.attr, &dev_attr_stall_floor.attr, &dev_attr_freq_map.attr, NULL, }; -static struct attribute_group dev_attr_group = { +static struct attribute *compute_dev_attr[] = { + &dev_attr_freq_map.attr, + NULL, +}; + +static struct attribute_group memlat_dev_attr_group = { .name = "mem_latency", - .attrs = dev_attr, + .attrs = memlat_dev_attr, +}; + +static struct attribute_group compute_dev_attr_group = { + .name = "compute", + .attrs = compute_dev_attr, }; #define MIN_MS 10U @@ -329,6 +339,12 @@ static struct devfreq_governor devfreq_gov_memlat = { .event_handler = devfreq_memlat_ev_handler, }; +static struct devfreq_governor devfreq_gov_compute = { + .name = "compute", + .get_target_freq = devfreq_memlat_get_freq, + .event_handler = devfreq_memlat_ev_handler, +}; + #define NUM_COLS 2 static struct core_dev_map *init_core_dev_map(struct device *dev, char *prop_name) @@ -371,20 +387,17 @@ static struct core_dev_map *init_core_dev_map(struct device *dev, return tbl; } -int register_memlat(struct device *dev, struct memlat_hwmon *hw) +static struct memlat_node *register_common(struct device *dev, + struct memlat_hwmon *hw) { - int ret = 0; struct memlat_node *node; if (!hw->dev && !hw->of_node) - return -EINVAL; + return ERR_PTR(-EINVAL); node = devm_kzalloc(dev, sizeof(*node), GFP_KERNEL); if (!node) - return -ENOMEM; - - node->gov = &devfreq_gov_memlat; - node->attr_grp = &dev_attr_group; + return ERR_PTR(-ENOMEM); node->ratio_ceil = 10; node->hw = hw; @@ -392,20 +405,68 @@ int register_memlat(struct device *dev, struct memlat_hwmon *hw) hw->freq_map = init_core_dev_map(dev, "qcom,core-dev-table"); if (!hw->freq_map) { dev_err(dev, "Couldn't find the core-dev freq table!\n"); - return -EINVAL; + return ERR_PTR(-EINVAL); } mutex_lock(&list_lock); list_add_tail(&node->list, &memlat_list); mutex_unlock(&list_lock); + return node; +} + +int register_compute(struct device *dev, struct memlat_hwmon *hw) +{ + struct memlat_node *node; + int ret = 0; + + node = register_common(dev, hw); + if (IS_ERR(node)) { + ret = PTR_ERR(node); + goto out; + } + mutex_lock(&state_lock); - if (!use_cnt) - ret = devfreq_add_governor(&devfreq_gov_memlat); + node->gov = &devfreq_gov_compute; + node->attr_grp = &compute_dev_attr_group; + + if (!compute_use_cnt) + ret = devfreq_add_governor(&devfreq_gov_compute); if (!ret) - use_cnt++; + compute_use_cnt++; mutex_unlock(&state_lock); +out: + if (!ret) + dev_info(dev, "Compute governor registered.\n"); + else + dev_err(dev, "Compute governor registration failed!\n"); + + return ret; +} + +int register_memlat(struct device *dev, struct memlat_hwmon *hw) +{ + struct memlat_node *node; + int ret = 0; + + node = register_common(dev, hw); + if (IS_ERR(node)) { + ret = PTR_ERR(node); + goto out; + } + + mutex_lock(&state_lock); + node->gov = &devfreq_gov_memlat; + node->attr_grp = &memlat_dev_attr_group; + + if (!memlat_use_cnt) + ret = devfreq_add_governor(&devfreq_gov_memlat); + if (!ret) + memlat_use_cnt++; + mutex_unlock(&state_lock); + +out: if (!ret) dev_info(dev, "Memory Latency governor registered.\n"); else diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h index 335ba7598b6b..c43edddb2695 100644 --- a/drivers/devfreq/governor_memlat.h +++ b/drivers/devfreq/governor_memlat.h @@ -66,10 +66,16 @@ struct memlat_hwmon { #ifdef CONFIG_DEVFREQ_GOV_MEMLAT int register_memlat(struct device *dev, struct memlat_hwmon *hw); +int register_compute(struct device *dev, struct memlat_hwmon *hw); int update_memlat(struct memlat_hwmon *hw); #else static inline int register_memlat(struct device *dev, - struct memlat_hwmon *hw) + struct memlat_hwmon *hw) +{ + return 0; +} +static inline int register_compute(struct device *dev, + struct memlat_hwmon *hw) { return 0; } From 532b26b84acd887d361f6af81cdbf147d4ec32e4 Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 20:26:45 -0800 Subject: [PATCH 02/27] PM / devfreq: bw_hwmon: Reset clear bits for some hardware versions Certain versions of the hardware modules have monitor/interrupt clear registers that are not self-clearing after being written to. Explicitly clear those register bits after writing to them. Change-Id: I0e6252be11d8503bcd4719480f503650e46a23ae Signed-off-by: Rohit Gupta --- drivers/devfreq/bimc-bwmon.c | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 0937cdf6c822..2f4b66c78e15 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bimc-bwmon: " fmt @@ -168,6 +168,14 @@ void mon_clear(struct bwmon *m, bool clear_all, enum mon_reg_type type) writel_relaxed(MON_CLEAR_ALL_BIT, MON3_CLEAR(m)); else writel_relaxed(MON_CLEAR_BIT, MON3_CLEAR(m)); + /* + * In some hardware versions since MON3_CLEAR(m) register does + * not have self-clearing capability it needs to be cleared + * explicitly. But we also need to ensure the writes to it + * are successful before clearing it. + */ + wmb(); + writel_relaxed(0, MON3_CLEAR(m)); break; } /* @@ -357,6 +365,14 @@ void mon_irq_clear(struct bwmon *m, enum mon_reg_type type) break; case MON3: writel_relaxed(MON3_INT_STATUS_MASK, MON3_INT_CLR(m)); + /* + * In some hardware versions since MON3_INT_CLEAR(m) register + * does not have self-clearing capability it needs to be + * cleared explicitly. But we also need to ensure the writes + * to it are successful before clearing it. + */ + wmb(); + writel_relaxed(0, MON3_INT_CLR(m)); break; } } From f53c359b505fe546e2bc48a22420f5fd11e55a94 Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 20:31:03 -0800 Subject: [PATCH 03/27] PM / devfreq: icc: Switch to OPP APIs OPP framework exposes a set of functions that could be used to fetch the list of supported bandwidth values from DT and to query the list for an appropriate value while honoring requests from governors/clients. Use OPP table APIs and get rid of frequency table logic to make the code more concise and upstream friendly. Change-Id: I120ae7230bdbfaaa9523263c65a8b84470583071 Signed-off-by: Rohit Gupta [avajid@codeaurora.org: renamed devbw to icc and resolved minor merge conflicts] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq_icc.c | 58 ++++++----------------------------- 1 file changed, 10 insertions(+), 48 deletions(-) diff --git a/drivers/devfreq/devfreq_icc.c b/drivers/devfreq/devfreq_icc.c index 28df27a35c31..f800094e7326 100644 --- a/drivers/devfreq/devfreq_icc.c +++ b/drivers/devfreq/devfreq_icc.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2013-2014, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2013-2014, 2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "devfreq-icc: " fmt @@ -57,33 +57,15 @@ static int set_bw(struct device *dev, u32 new_ib, u32 new_ab) return ret; } -static void find_freq(struct devfreq_dev_profile *p, unsigned long *freq, - u32 flags) -{ - int i; - unsigned long atmost, atleast, f; - - atmost = p->freq_table[0]; - atleast = p->freq_table[p->max_state-1]; - for (i = 0; i < p->max_state; i++) { - f = p->freq_table[i]; - if (f <= *freq) - atmost = max(f, atmost); - if (f >= *freq) - atleast = min(f, atleast); - } - - if (flags & DEVFREQ_FLAG_LEAST_UPPER_BOUND) - *freq = atmost; - else - *freq = atleast; -} - static int icc_target(struct device *dev, unsigned long *freq, u32 flags) { struct dev_data *d = dev_get_drvdata(dev); + struct dev_pm_opp *opp; + + opp = devfreq_recommended_opp(dev, freq, flags); + if (!IS_ERR(opp)) + dev_pm_opp_put(opp); - find_freq(&d->dp, freq, flags); return set_bw(dev, *freq, d->gov_ab); } @@ -96,7 +78,6 @@ static int icc_get_dev_status(struct device *dev, return 0; } -#define PROP_TBL "qcom,bw-tbl" #define PROP_ACTIVE "qcom,active-only" #define ACTIVE_ONLY_TAG 0x3 @@ -104,9 +85,8 @@ int devfreq_add_icc(struct device *dev) { struct dev_data *d; struct devfreq_dev_profile *p; - u32 *data; const char *gov_name; - int ret, len, i; + int ret; d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); if (!d) @@ -118,27 +98,9 @@ int devfreq_add_icc(struct device *dev) p->target = icc_target; p->get_dev_status = icc_get_dev_status; - if (of_find_property(dev->of_node, PROP_TBL, &len)) { - len /= sizeof(*data); - data = devm_kzalloc(dev, len * sizeof(*data), GFP_KERNEL); - if (!data) - return -ENOMEM; - - p->freq_table = devm_kzalloc(dev, - len * sizeof(*p->freq_table), - GFP_KERNEL); - if (!p->freq_table) - return -ENOMEM; - - ret = of_property_read_u32_array(dev->of_node, PROP_TBL, - data, len); - if (ret < 0) - return ret; - - for (i = 0; i < len; i++) - p->freq_table[i] = data[i]; - p->max_state = len; - } + ret = dev_pm_opp_of_add_table(dev); + if (ret < 0) + dev_err(dev, "Couldn't parse OPP table:%d\n", ret); d->icc_path = of_icc_get(dev, NULL); if (IS_ERR(d->icc_path)) { From 10e4379923136802c5e93f8ba95e50dd5a347f2b Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Wed, 16 Jan 2019 20:32:20 -0800 Subject: [PATCH 04/27] devfreq: memlat: Add suspend/resume for mem_latency Add suspend/resume support for the mem_latency governor. Uses hwmon driver- defined suspend/resume calls in addition to monitor start/stop calls within the devfreq framework. Change-Id: I900b322a05cd0312f082d70640e21851879cb18c Signed-off-by: Jonathan Avila [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_memlat.c | 72 ++++++++++++++++++++++++++++++- 1 file changed, 71 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index f7d213a27983..584a8d06239e 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2015-2017, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2015-2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "mem_lat: " fmt @@ -35,6 +35,7 @@ struct memlat_node { struct memlat_hwmon *hw; struct devfreq_governor *gov; struct attribute_group *attr_grp; + unsigned long resume_freq; }; static LIST_HEAD(memlat_list); @@ -199,6 +200,39 @@ err_start: return ret; } +static int gov_suspend(struct devfreq *df) +{ + struct memlat_node *node = df->data; + unsigned long prev_freq = df->previous_freq; + + node->mon_started = false; + devfreq_monitor_suspend(df); + + mutex_lock(&df->lock); + update_devfreq(df); + mutex_unlock(&df->lock); + + node->resume_freq = max(prev_freq, 1UL); + + return 0; +} + +static int gov_resume(struct devfreq *df) +{ + struct memlat_node *node = df->data; + + mutex_lock(&df->lock); + update_devfreq(df); + mutex_unlock(&df->lock); + + node->resume_freq = 0; + + devfreq_monitor_resume(df); + node->mon_started = true; + + return 0; +} + static void gov_stop(struct devfreq *df) { struct memlat_node *node = df->data; @@ -220,6 +254,18 @@ static int devfreq_memlat_get_freq(struct devfreq *df, unsigned long max_freq = 0; unsigned int ratio; + /* + * node->resume_freq is set to 0 at the end of resume (after the update) + * and is set to df->prev_freq at the end of suspend (after the update). + * This function will be called as part of the update_devfreq call in + * both scenarios. As a result, this block will cause a 0 vote during + * suspend and a vote for df->prev_freq during resume. + */ + if (!node->mon_started) { + *freq = node->resume_freq; + return 0; + } + hw->get_cnt(hw); for (i = 0; i < hw->num_cores; i++) { @@ -322,6 +368,30 @@ static int devfreq_memlat_ev_handler(struct devfreq *df, "Disabled Memory Latency governor\n"); break; + case DEVFREQ_GOV_SUSPEND: + ret = gov_suspend(df); + if (ret < 0) { + dev_err(df->dev.parent, + "Unable to suspend memlat governor (%d)\n", + ret); + return ret; + } + + dev_dbg(df->dev.parent, "Suspended memlat governor\n"); + break; + + case DEVFREQ_GOV_RESUME: + ret = gov_resume(df); + if (ret < 0) { + dev_err(df->dev.parent, + "Unable to resume memlat governor (%d)\n", + ret); + return ret; + } + + dev_dbg(df->dev.parent, "Resumed memlat governor\n"); + break; + case DEVFREQ_GOV_INTERVAL: sample_ms = *(unsigned int *)data; sample_ms = max(MIN_MS, sample_ms); From fd91ce15bd33e38e28b1c01da3de314e3d35b6df Mon Sep 17 00:00:00 2001 From: Santosh Mardi Date: Wed, 16 Jan 2019 20:33:40 -0800 Subject: [PATCH 05/27] devfreq: update freq variable in compute_freq function Update freq variable from unsigned long to uint64_t of compute_freq function to be in compliance with 32 bit environment. Change-Id: Ia9183b0e593daf3780135a8e1ae8ddb36db16f86 Signed-off-by: Santosh Mardi --- drivers/devfreq/arm-memlat-mon.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index 2c888327cb4a..25fecfb6d010 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "arm-memlat-mon: " fmt @@ -69,7 +69,7 @@ static unsigned long compute_freq(struct cpu_pmu_stats *cpustats, { ktime_t ts; unsigned int diff; - unsigned long freq = 0; + uint64_t freq = 0; ts = ktime_get(); diff = ktime_to_us(ktime_sub(ts, cpustats->prev_ts)); From 5f7ae5c5e5197406f94431c40e8b9fb7092756ea Mon Sep 17 00:00:00 2001 From: Santosh Mardi Date: Wed, 16 Jan 2019 20:35:16 -0800 Subject: [PATCH 06/27] devfreq: suppress platform driver bind / unbind feature For arm-memlat and bimc-hwmon platform driver does not support the manual bind / unbind feature through sysfs, when the governor is registered and started. Suppress the bind / unbind calls using driver attribute. Change-Id: I8287012e1e6931d80953382f3d625223315cec85 Signed-off-by: Santosh Mardi --- drivers/devfreq/arm-memlat-mon.c | 1 + drivers/devfreq/bimc-bwmon.c | 1 + 2 files changed, 2 insertions(+) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index 25fecfb6d010..4c79111cf89d 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -354,6 +354,7 @@ static struct platform_driver arm_memlat_mon_driver = { .driver = { .name = "arm-memlat-mon", .of_match_table = memlat_match_table, + .suppress_bind_attrs = true, }, }; diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 2f4b66c78e15..fc0800d9f1b5 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -1133,6 +1133,7 @@ static struct platform_driver bimc_bwmon_driver = { .driver = { .name = "bimc-bwmon", .of_match_table = bimc_bwmon_match_table, + .suppress_bind_attrs = true, }, }; From c5899ccfd377db270e14074909e8401f0064807c Mon Sep 17 00:00:00 2001 From: Santosh Mardi Date: Wed, 16 Jan 2019 20:36:59 -0800 Subject: [PATCH 07/27] devfreq: suppress platform driver bind / unbind feature For simple-dev and devbw platform driver does not support the manual bind / unbind feature through sysfs, when the governor is registered and started. Suppress the bind / unbind calls using driver attribute. Change-Id: Ide843476bdc3870c48f6557097ee549e7a794a22 Signed-off-by: Santosh Mardi [avajid@codeaurora.org: resolved minor merge conflict] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq_icc.c | 1 + drivers/devfreq/devfreq_simple_dev.c | 3 ++- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/devfreq_icc.c b/drivers/devfreq/devfreq_icc.c index f800094e7326..d1f1c396b432 100644 --- a/drivers/devfreq/devfreq_icc.c +++ b/drivers/devfreq/devfreq_icc.c @@ -169,6 +169,7 @@ static struct platform_driver devfreq_icc_driver = { .driver = { .name = "devfreq-icc", .of_match_table = devfreq_icc_match_table, + .suppress_bind_attrs = true, }, }; diff --git a/drivers/devfreq/devfreq_simple_dev.c b/drivers/devfreq/devfreq_simple_dev.c index c5431755a176..1d3c9e69f473 100644 --- a/drivers/devfreq/devfreq_simple_dev.c +++ b/drivers/devfreq/devfreq_simple_dev.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2015, 2017, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2017-2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "devfreq-simple-dev: " fmt @@ -204,6 +204,7 @@ static struct platform_driver devfreq_clock_driver = { .driver = { .name = "devfreq-simple-dev", .of_match_table = devfreq_simple_match_table, + .suppress_bind_attrs = true, }, }; module_platform_driver(devfreq_clock_driver); From 3f6460328db32ddb308e6cf240d264c9b4fde194 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Wed, 16 Jan 2019 20:42:03 -0800 Subject: [PATCH 08/27] PM / devfreq: bw_hwmon: Fix a race condition in hwmon stop Currently, the bw_hwmon governor handles DEVFREQ_GOV_STOP and DEVFREQ_GOV_SUSPEND events by calling devfreq_monitor_stop and devfreq_monitor_suspend respectively, prior to disabling the bw_hwmon irq. Doing so allows for the following race condition: T0: bw_hwmon irq thread: /* df->governor == bw_hwmon * changing to * powersave governor */ governor_store() /* Calls bw_hwmon governor event * handler with DEVFREQ_GOV_STOP * event */ df->governor->event_handler() gov_stop() stop_monitor() update_bw_hwmon() devfreq_monitor_stop() /* Cancels timer for future calls * to devfreq_monitor */ devfreq_monitor_stop() /* Calls free_irq(), which waits * for the bw_hwmon IRQ thread * to finish. */ hw->stop_hwmon() --finishes update_devfreq()-- /* Incorrectly starts * devfreq_monitoring */ devfreq_monitor_start() --thread finishes execution-- --finishes DEVFREQ_GOV_STOP handling-- --switches governor to powersave-- /* devfreq monitor for this * instance. This will call * queue_delayed_work() * which internally queues * the timer for this devfreq * structure. */ devfreq_monitor() /* df->governor == powersave * changing to bw_hwmon */ governor_store() /* powersave: DEVFREQ_GOV_STOP */ df->governor->event_handler() /* bw_hwmon: DEVFREQ_GOV_START */ df->governor->event_handler() gov_start() /* Will incorrectly adjust * the fields within the * timer and corrupt the * timer data structure */ devfreq_monitor_start() Since this corrupts the timer data structures, when the timer gets expired, it will be expired twice. Fix this race condition by introducing new lock to synchronize the access to the mon_started variable, so that the irq thread does not restart the devfreq monitor after it has been stopped. Change-Id: I2dced21d5343afd6ec2e13876e26aeb5c83a4d12 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: made minor styling change] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_bw_hwmon.c | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 77a91e8f65ca..fe0071b3db4f 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2013-2017, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2013-2018, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bw-hwmon: " fmt @@ -66,6 +66,7 @@ struct hwmon_node { struct bw_hwmon *hw; struct devfreq_governor *gov; struct attribute_group *attr_grp; + struct mutex mon_lock; }; #define UP_WAKE 1 @@ -493,9 +494,11 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) if (!node) return -ENODEV; - if (!node->mon_started) + mutex_lock(&node->mon_lock); + if (!node->mon_started) { + mutex_unlock(&node->mon_lock); return -EBUSY; - + } dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); @@ -507,6 +510,7 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) mutex_unlock(&df->lock); devfreq_monitor_start(df); + mutex_unlock(&node->mon_lock); return 0; } @@ -554,7 +558,9 @@ static void stop_monitor(struct devfreq *df, bool init) struct hwmon_node *node = df->data; struct bw_hwmon *hw = node->hw; + mutex_lock(&node->mon_lock); node->mon_started = false; + mutex_unlock(&node->mon_lock); if (init) { devfreq_monitor_stop(df); @@ -941,6 +947,7 @@ int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon) node->mbps_zones[0] = 0; node->hw = hwmon; + mutex_init(&node->mon_lock); mutex_lock(&list_lock); list_add_tail(&node->list, &hwmon_list); mutex_unlock(&list_lock); From 9170b2bc7b6cf1e73f80d0a745c09418cab71b7f Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Fri, 15 Mar 2019 16:25:07 -0700 Subject: [PATCH 09/27] devfreq: Do not round up bandwidth on BWMON4 devices Currently only BWMON5 devices increment the count for non-zero bandwidth while BWMON4 devices increment the count all the time. This results in a non-zero bandwidth report even when the count is zero for BWMON4 devices leading to unnecessarily higher bandwidth votes. Apply checks similar to BWMON5 devices for BWMON4 monitors as well. Change-Id: I0a2d1c0ed979973966668391a32f63bf2711f981 Signed-off-by: Rama Aparna Mallavarapu --- drivers/devfreq/bimc-bwmon.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index fc0800d9f1b5..1c8c47b56ecc 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -580,15 +580,16 @@ unsigned long get_zone_count(struct bwmon *m, unsigned int zone, WARN(1, "Invalid\n"); return 0; case MON2: - count = readl_relaxed(MON2_ZONE_MAX(m, zone)) + 1; + count = readl_relaxed(MON2_ZONE_MAX(m, zone)); break; case MON3: count = readl_relaxed(MON3_ZONE_MAX(m, zone)); - if (count) - count++; break; } + if (count) + count++; + return count; } From 2526ddbdbe46196d25026a25ae9fcb1a3e67ee3f Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Mon, 25 Mar 2019 11:46:23 -0700 Subject: [PATCH 10/27] devfreq: bw_mon: check for the return value of start_monitor The BWMON governor start is returning success on GOV_START event without checking for the return value of start_monitor. The return value of start_monitor is not being returned to ret variable. This would cause the governor to start successfully even when the monitor failed to start causing a NULL pointer derefence when accessing the device attributes. Fix it by checking the return value of start_monitor. Change-Id: I8c1f6933d44ae4533c6b81ccda8a5c4c0da3779c Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: minor change to check ret < 0] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_bw_hwmon.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index fe0071b3db4f..2a57dc847279 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -599,7 +599,8 @@ static int gov_start(struct devfreq *df) node->orig_data = df->data; df->data = node; - if (start_monitor(df, true)) + ret = start_monitor(df, true); + if (ret < 0) goto err_start; ret = sysfs_create_group(&df->dev.kobj, node->attr_grp); From 5310a380874b4b920a05b0eb9999839cd1202ee1 Mon Sep 17 00:00:00 2001 From: Santosh Mardi Date: Thu, 3 Jan 2019 10:52:14 +0530 Subject: [PATCH 11/27] devfreq: return error code when governor start fails If there is a failure in starting hw monitor required for memlat governor, return the error code to devfreq framework so that the framework will make sure the governor is switched back to previous governor. Change-Id: I76705725fc8b94d5341b1b8af7c1bd2b79e7642a Signed-off-by: Santosh Mardi [avajid@codeaurora.org: minor change to check ret < 0] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_memlat.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index 584a8d06239e..8997d0f11f53 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2015-2018, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2015-2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "mem_lat: " fmt @@ -182,7 +182,8 @@ static int gov_start(struct devfreq *df) node->orig_data = df->data; df->data = node; - if (start_monitor(df)) + ret = start_monitor(df); + if (ret < 0) goto err_start; ret = sysfs_create_group(&df->dev.kobj, node->attr_grp); From 4b5e72686977d7080829f88b2ded37bb51e27ef3 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Wed, 3 Apr 2019 12:12:30 -0700 Subject: [PATCH 12/27] devfreq: bimc_bwmon: Add support to enable BWMON clks Some BWMON devices requires clocks to be enabled, hence add necessary support in the bwmon driver to enable the required clocks if any. Change-Id: Ie06731f764ff24d602ae2a86dea4f39ce75df800 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 82 ++++++++++++++++++++++++++++++++++++ 1 file changed, 82 insertions(+) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 1c8c47b56ecc..dbb046630342 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -20,6 +20,7 @@ #include #include #include +#include #include "governor_bw_hwmon.h" #define GLB_INT_STATUS(m) ((m)->global_base + 0x100) @@ -91,6 +92,8 @@ struct bwmon { void __iomem *global_base; unsigned int mport; int irq; + int nr_clks; + struct clk **clks; const struct bwmon_spec *spec; struct device *dev; struct bw_hwmon hw; @@ -775,6 +778,27 @@ void mon_set_byte_count_filter(struct bwmon *m, enum mon_reg_type type) } } +static __always_inline int mon_clk_enable(struct bwmon *m) +{ + int ret; + int i; + + for (i = 0; i < m->nr_clks; i++) { + ret = clk_prepare_enable(m->clks[i]); + if (ret < 0) { + dev_err(m->dev, "BWMON clk not enabled: %d\n", ret); + goto err; + } + } + + return 0; +err: + for (i--; i >= 0; i--) + clk_disable_unprepare(m->clks[i]); + + return ret; +} + static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps, enum mon_reg_type type) { @@ -783,6 +807,12 @@ static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, int ret; irq_handler_t handler; + ret = mon_clk_enable(m); + if (ret < 0) { + dev_err(m->dev, "Unable to turn on bwmon clks! (%d)\n", ret); + return ret; + } + switch (type) { case MON1: handler = bwmon_intr_handler; @@ -849,6 +879,14 @@ static int start_bw_hwmon3(struct bw_hwmon *hw, unsigned long mbps) return __start_bw_hwmon(hw, mbps, MON3); } +static __always_inline void mon_clk_disable(struct bwmon *m) +{ + int i; + + for (i = m->nr_clks - 1; i >= 0; i--) + clk_disable_unprepare(m->clks[i]); +} + static __always_inline void __stop_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { @@ -859,6 +897,7 @@ void __stop_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) mon_disable(m, type); mon_clear(m, true, type); mon_irq_clear(m, type); + mon_clk_disable(m); } static void stop_bw_hwmon(struct bw_hwmon *hw) @@ -911,6 +950,12 @@ int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) int ret; irq_handler_t handler; + ret = mon_clk_enable(m); + if (ret < 0) { + dev_err(m->dev, "Unable to turn on bwmon clks! (%d)\n", ret); + return ret; + } + switch (type) { case MON1: handler = bwmon_intr_handler; @@ -1014,6 +1059,7 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) struct bwmon *m; int ret; u32 data, count_unit; + unsigned int len, i; m = devm_kzalloc(dev, sizeof(*m), GFP_KERNEL); if (!m) @@ -1059,6 +1105,42 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->mport = data; } + if (of_find_property(dev->of_node, "qcom,bwmon_clks", &len)) { + m->nr_clks = of_property_count_strings(dev->of_node, + "qcom,bwmon_clks"); + if (!m->nr_clks) { + dev_err(dev, "Failed to get clock names\n"); + return -EINVAL; + } + + m->clks = devm_kzalloc(dev, sizeof(struct clk *) * m->nr_clks, + GFP_KERNEL); + if (!m->clks) + return -ENOMEM; + + for (i = 0; i < m->nr_clks; i++) { + const char *clock_name; + + ret = of_property_read_string_index(dev->of_node, + "qcom,bwmon_clks", i, + &clock_name); + if (ret < 0) { + pr_err("failed to read clk index %d ret %d\n", + i, ret); + return ret; + } + m->clks[i] = devm_clk_get(dev, clock_name); + if (IS_ERR(m->clks[i])) { + ret = PTR_ERR(m->clks[i]); + if (ret != -EPROBE_DEFER) + dev_err(dev, "Error to get %s clk %d\n", + clock_name, ret); + return ret; + } + } + } else + m->nr_clks = 0; + m->irq = platform_get_irq(pdev, 0); if (m->irq < 0) { dev_err(dev, "Unable to get IRQ number\n"); From b04e29f99165bfe34f0fed9a9390daac5bb6d2ec Mon Sep 17 00:00:00 2001 From: Santosh Mardi Date: Tue, 13 Nov 2018 14:57:42 +0530 Subject: [PATCH 13/27] PM / devfreq: bw_hwmon: use unsigned parameter for bytes_to_mbps In bytes_to_mbps function, the parameter is all unsigned, so change the decleration of the function to include unsigned long long to avoid compilation errors in 32 bit environment. Also changed the return value as unsigned long to avoid any data loss possible in 64 bit environment. Change-Id: I90aebfe32e86ebc17c414d6f3c0c3cbd6dddb0bd Signed-off-by: Santosh Mardi --- drivers/devfreq/governor_bw_hwmon.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 2a57dc847279..fd95f4f145a8 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -154,7 +154,7 @@ out: \ #define MAX_MS 500U /* Returns MBps of read/writes for the sampling window. */ -static unsigned int bytes_to_mbps(long long bytes, unsigned int us) +static unsigned long bytes_to_mbps(unsigned long long bytes, unsigned int us) { bytes *= USEC_PER_SEC; do_div(bytes, us); From c727848bb4f0b841b6ade4eced73e03a7887af87 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Tue, 23 Apr 2019 11:51:53 +0530 Subject: [PATCH 14/27] devfreq: detect ddr type and add frequency table accordingly Some targets support different DDR types. Detect the DDR type and populate the frequency table accordingly. OPP framework supports opp-supported-hw bit map which allows the selected frequencies to be added to the opp table based on the hardware version check. Implement the same in the bandwidth monitor device so that the right frequencies would be added to the opp table. This patch also adds checks for the latency based device to detect the DDR type at runtime and add the corresponding frequency map. Change-Id: Ice5a0b14da67b3f2f07e98bc7349220da7d4efdb Signed-off-by: Santosh Mardi Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: resolved minor conflicts and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/arm-memlat-mon.c | 23 +++++++++++++++++++++++ drivers/devfreq/devfreq_icc.c | 15 +++++++++++++++ drivers/devfreq/governor_memlat.c | 23 ++++++++++++++++++----- drivers/devfreq/governor_memlat.h | 1 + 4 files changed, 57 insertions(+), 5 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index 4c79111cf89d..f1d2f3d38fa9 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -20,6 +20,7 @@ #include #include #include +#include #include "governor.h" #include "governor_memlat.h" #include @@ -252,6 +253,26 @@ static int get_mask_from_dev_handle(struct platform_device *pdev, return ret; } +static struct device_node *parse_child_nodes(struct device *dev) +{ + struct device_node *of_child; + int ddr_type_of = -1; + int ddr_type = of_fdt_get_ddrtype(); + int ret; + + for_each_child_of_node(dev->of_node, of_child) { + ret = of_property_read_u32(of_child, "qcom,ddr-type", + &ddr_type_of); + if (!ret && (ddr_type == ddr_type_of)) { + dev_dbg(dev, + "ddr-type = %d, is matching DT entry\n", + ddr_type_of); + return of_child; + } + } + return NULL; +} + static int arm_memlat_mon_driver_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; @@ -297,6 +318,8 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) hw->start_hwmon = &start_hwmon; hw->stop_hwmon = &stop_hwmon; hw->get_cnt = &get_cnt; + if (of_get_child_count(dev->of_node)) + hw->get_child_of_node = &parse_child_nodes; spec = of_device_get_match_data(dev); if (spec && spec->is_compute) { diff --git a/drivers/devfreq/devfreq_icc.c b/drivers/devfreq/devfreq_icc.c index d1f1c396b432..a9888faedacc 100644 --- a/drivers/devfreq/devfreq_icc.c +++ b/drivers/devfreq/devfreq_icc.c @@ -17,7 +17,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -87,6 +89,8 @@ int devfreq_add_icc(struct device *dev) struct devfreq_dev_profile *p; const char *gov_name; int ret; + struct opp_table *opp_table; + u32 version; d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); if (!d) @@ -98,6 +102,15 @@ int devfreq_add_icc(struct device *dev) p->target = icc_target; p->get_dev_status = icc_get_dev_status; + if (of_device_is_compatible(dev->of_node, "qcom,devfreq-icc-ddr")) { + version = (1 << of_fdt_get_ddrtype()); + opp_table = dev_pm_opp_set_supported_hw(dev, &version, 1); + if (IS_ERR(opp_table)) { + dev_err(dev, "Failed to set supported hardware\n"); + return PTR_ERR(opp_table); + } + } + ret = dev_pm_opp_of_add_table(dev); if (ret < 0) dev_err(dev, "Couldn't parse OPP table:%d\n", ret); @@ -159,6 +172,8 @@ static int devfreq_icc_remove(struct platform_device *pdev) } static const struct of_device_id devfreq_icc_match_table[] = { + { .compatible = "qcom,devfreq-icc-llcc" }, + { .compatible = "qcom,devfreq-icc-ddr" }, { .compatible = "qcom,devfreq-icc" }, {} }; diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index 8997d0f11f53..04ff9abdefb9 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -18,6 +18,7 @@ #include #include #include +#include #include #include #include "governor.h" @@ -418,14 +419,18 @@ static struct devfreq_governor devfreq_gov_compute = { #define NUM_COLS 2 static struct core_dev_map *init_core_dev_map(struct device *dev, - char *prop_name) + struct device_node *of_node, + char *prop_name) { int len, nf, i, j; u32 data; struct core_dev_map *tbl; int ret; - if (!of_find_property(dev->of_node, prop_name, &len)) + if (!of_node) + of_node = dev->of_node; + + if (!of_find_property(of_node, prop_name, &len)) return NULL; len /= sizeof(data); @@ -439,13 +444,13 @@ static struct core_dev_map *init_core_dev_map(struct device *dev, return NULL; for (i = 0, j = 0; i < nf; i++, j += 2) { - ret = of_property_read_u32_index(dev->of_node, prop_name, j, + ret = of_property_read_u32_index(of_node, prop_name, j, &data); if (ret < 0) return NULL; tbl[i].core_mhz = data / 1000; - ret = of_property_read_u32_index(dev->of_node, prop_name, j + 1, + ret = of_property_read_u32_index(of_node, prop_name, j + 1, &data); if (ret < 0) return NULL; @@ -462,6 +467,7 @@ static struct memlat_node *register_common(struct device *dev, struct memlat_hwmon *hw) { struct memlat_node *node; + struct device_node *of_child; if (!hw->dev && !hw->of_node) return ERR_PTR(-EINVAL); @@ -473,7 +479,14 @@ static struct memlat_node *register_common(struct device *dev, node->ratio_ceil = 10; node->hw = hw; - hw->freq_map = init_core_dev_map(dev, "qcom,core-dev-table"); + if (hw->get_child_of_node) { + of_child = hw->get_child_of_node(dev); + hw->freq_map = init_core_dev_map(dev, of_child, + "qcom,core-dev-table"); + } else { + hw->freq_map = init_core_dev_map(dev, NULL, + "qcom,core-dev-table"); + } if (!hw->freq_map) { dev_err(dev, "Couldn't find the core-dev freq table!\n"); return ERR_PTR(-EINVAL); diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h index c43edddb2695..b1c334018f16 100644 --- a/drivers/devfreq/governor_memlat.h +++ b/drivers/devfreq/governor_memlat.h @@ -54,6 +54,7 @@ struct memlat_hwmon { int (*start_hwmon)(struct memlat_hwmon *hw); void (*stop_hwmon)(struct memlat_hwmon *hw); unsigned long (*get_cnt)(struct memlat_hwmon *hw); + struct device_node *(*get_child_of_node)(struct device *dev); struct device *dev; struct device_node *of_node; From 74b58a68aae3a0459e6cd3c4a53d781d487562fa Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Mon, 8 Apr 2019 11:38:35 -0700 Subject: [PATCH 15/27] PM / devfreq: memlat: Aggregate memlat mons under a controller Currently, each individual arm-memlat-mon instance separately reads from a few shared PMUs, which leads to unnecessary IPI overhead. Tweak the arm-memlat-mon design to allow multiple monitors to share certain event counts, lowering overhead. Change-Id: Ic3358afd1853efd84a4c3d5f3d66c3a86df6e5e7 Signed-off-by: Jonathan Avila Signed-off-by: Amir Vajid [avajid@codeaurora.org: squashed minor fixes and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/arm-memlat-mon.c | 729 ++++++++++++++++++++++-------- drivers/devfreq/governor_memlat.c | 18 +- drivers/devfreq/governor_memlat.h | 3 + 3 files changed, 559 insertions(+), 191 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index f1d2f3d38fa9..cd1f4c84fe6d 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -25,137 +25,200 @@ #include "governor_memlat.h" #include #include +#include -enum ev_index { +enum common_ev_idx { INST_IDX, - CM_IDX, CYC_IDX, - STALL_CYC_IDX, - NUM_EVENTS + STALL_IDX, + NUM_COMMON_EVS }; #define INST_EV 0x08 -#define L2DM_EV 0x17 #define CYC_EV 0x11 +enum mon_type { + MEMLAT_CPU_GRP, + MEMLAT_MON, + COMPUTE_MON, + NUM_MON_TYPES +}; + struct event_data { - struct perf_event *pevent; - unsigned long prev_count; + struct perf_event *pevent; + unsigned long prev_count; + unsigned long last_delta; }; -struct cpu_pmu_stats { - struct event_data events[NUM_EVENTS]; - ktime_t prev_ts; +/** + * struct memlat_mon - A specific consumer of cpu_grp generic counters. + * + * @is_active: Whether or not this mon is currently running + * memlat. + * @cpus: CPUs this mon votes on behalf of. Must be a + * subset of @cpu_grp's CPUs. If no CPUs provided, + * defaults to using all of @cpu_grp's CPUs. + * @miss_ev_id: The event code corresponding to the @miss_ev + * perf event. Will be 0 for compute. + * @miss_ev: The cache miss perf event exclusive to this + * mon. Will be NULL for compute. + * @requested_update_ms: The mon's desired polling rate. The lowest + * @requested_update_ms of all mons determines + * @cpu_grp's update_ms. + * @hw: The memlat_hwmon struct corresponding to this + * mon's specific memlat instance. + * @cpu_grp: The cpu_grp who owns this mon. + */ +struct memlat_mon { + bool is_active; + cpumask_t cpus; + unsigned int miss_ev_id; + unsigned int requested_update_ms; + struct event_data *miss_ev; + struct memlat_hwmon hw; + + struct memlat_cpu_grp *cpu_grp; }; -struct cpu_grp_info { - cpumask_t cpus; - unsigned int event_ids[NUM_EVENTS]; - struct cpu_pmu_stats *cpustats; - struct memlat_hwmon hw; +/** + * struct memlat_cpu_grp - A coordinator of both HW reads and devfreq updates + * for one or more memlat_mons. + * + * @cpus: The CPUs this cpu_grp will read events from. + * @common_ev_ids: The event codes of the events all mons need. + * @common_evs: The event data of the events all mons need. + * Organized as a 2d array of [cpu][event]. + * @last_update_ts: Used to avoid redundant reads. + * @last_ts_delta_us: The time difference between the most recent + * update and the one before that. Used to compute + * effective frequency. + * @work: The delayed_work used for handling updates. + * @update_ms: The frequency with which @work triggers. + * @num_mons: The number of @mons for this cpu_grp. + * @num_inited_mons: The number of @mons who have probed. + * @num_active_mons: The number of @mons currently running + * memlat. + * @mons: All of the memlat_mon structs representing + * the different voters who share this cpu_grp. + * @mons_lock: A lock used to protect the @mons. + */ +struct memlat_cpu_grp { + cpumask_t cpus; + unsigned int common_ev_ids[NUM_COMMON_EVS]; + struct event_data **common_evs; + ktime_t last_update_ts; + unsigned long last_ts_delta_us; + + struct delayed_work work; + unsigned int update_ms; + + unsigned int num_mons; + unsigned int num_inited_mons; + unsigned int num_active_mons; + struct memlat_mon *mons; + struct mutex mons_lock; }; struct memlat_mon_spec { - bool is_compute; + enum mon_type type; }; -#define to_cpustats(cpu_grp, cpu) \ - (&cpu_grp->cpustats[cpu - cpumask_first(&cpu_grp->cpus)]) -#define to_devstats(cpu_grp, cpu) \ - (&cpu_grp->hw.core_stats[cpu - cpumask_first(&cpu_grp->cpus)]) -#define to_cpu_grp(hwmon) container_of(hwmon, struct cpu_grp_info, hw) +#define to_common_cpustats(cpu_grp, cpu) \ + (cpu_grp->common_evs[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_devstats(mon, cpu) \ + (&mon->hw.core_stats[cpu - cpumask_first(&mon->cpus)]) +#define to_mon(hwmon) container_of(hwmon, struct memlat_mon, hw) - -static unsigned long compute_freq(struct cpu_pmu_stats *cpustats, - unsigned long cyc_cnt) -{ - ktime_t ts; - unsigned int diff; - uint64_t freq = 0; - - ts = ktime_get(); - diff = ktime_to_us(ktime_sub(ts, cpustats->prev_ts)); - if (!diff) - diff = 1; - cpustats->prev_ts = ts; - freq = cyc_cnt; - do_div(freq, diff); - - return freq; -} +static struct workqueue_struct *memlat_wq; #define MAX_COUNT_LIM 0xFFFFFFFFFFFFFFFF -static inline unsigned long read_event(struct event_data *event) +static inline void read_event(struct event_data *event) { - unsigned long ev_count; + unsigned long ev_count = 0; u64 total, enabled, running; if (!event->pevent) - return 0; + return; total = perf_event_read_value(event->pevent, &enabled, &running); ev_count = total - event->prev_count; event->prev_count = total; - return ev_count; + event->last_delta = ev_count; } -static void read_perf_counters(int cpu, struct cpu_grp_info *cpu_grp) +static void update_counts(struct memlat_cpu_grp *cpu_grp) { - struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); - struct dev_stats *devstats = to_devstats(cpu_grp, cpu); - unsigned long cyc_cnt, stall_cnt; + unsigned int cpu, i; + struct memlat_mon *mon; + ktime_t now = ktime_get(); + unsigned long delta = ktime_us_delta(now, cpu_grp->last_update_ts); - devstats->inst_count = read_event(&cpustats->events[INST_IDX]); - devstats->mem_count = read_event(&cpustats->events[CM_IDX]); - cyc_cnt = read_event(&cpustats->events[CYC_IDX]); - devstats->freq = compute_freq(cpustats, cyc_cnt); - if (cpustats->events[STALL_CYC_IDX].pevent) { - stall_cnt = read_event(&cpustats->events[STALL_CYC_IDX]); - stall_cnt = min(stall_cnt, cyc_cnt); - devstats->stall_pct = mult_frac(100, stall_cnt, cyc_cnt); - } else { - devstats->stall_pct = 100; + cpu_grp->last_ts_delta_us = delta; + cpu_grp->last_update_ts = now; + + for_each_cpu(cpu, &cpu_grp->cpus) { + struct event_data *cpu_grp_stats = + to_common_cpustats(cpu_grp, cpu); + + for (i = 0; i < NUM_COMMON_EVS; i++) + read_event(&cpu_grp_stats[i]); + + if (!cpu_grp_stats[STALL_IDX].pevent) + cpu_grp_stats[STALL_IDX].last_delta = + cpu_grp_stats[CYC_IDX].last_delta; + } + + for (i = 0; i < cpu_grp->num_mons; i++) { + mon = &cpu_grp->mons[i]; + + if (!mon->is_active || !mon->miss_ev) + continue; + + for_each_cpu(cpu, &mon->cpus) { + unsigned int mon_idx = + cpu - cpumask_first(&mon->cpus); + read_event(&mon->miss_ev[mon_idx]); + } } } static unsigned long get_cnt(struct memlat_hwmon *hw) { - int cpu; - struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + struct memlat_mon *mon = to_mon(hw); + struct memlat_cpu_grp *cpu_grp = mon->cpu_grp; + unsigned int cpu; - for_each_cpu(cpu, &cpu_grp->cpus) - read_perf_counters(cpu, cpu_grp); + for_each_cpu(cpu, &mon->cpus) { + struct event_data *cpu_grp_stats = + to_common_cpustats(cpu_grp, cpu); + unsigned int mon_idx = + cpu - cpumask_first(&mon->cpus); + struct dev_stats *devstats = to_devstats(mon, cpu); + + devstats->freq = cpu_grp_stats[CYC_IDX].last_delta / + cpu_grp->last_ts_delta_us; + devstats->stall_pct = + mult_frac(100, cpu_grp_stats[STALL_IDX].last_delta, + cpu_grp_stats[CYC_IDX].last_delta); + devstats->inst_count = cpu_grp_stats[INST_IDX].last_delta; + + if (mon->miss_ev) + devstats->mem_count = + mon->miss_ev[mon_idx].last_delta; + else { + devstats->inst_count = 0; + devstats->mem_count = 1; + } + } return 0; } -static void delete_events(struct cpu_pmu_stats *cpustats) +static void delete_event(struct event_data *event) { - int i; - - for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { - cpustats->events[i].prev_count = 0; - if (cpustats->events[i].pevent) { - perf_event_release_kernel(cpustats->events[i].pevent); - cpustats->events[i].pevent = NULL; - } - } -} - -static void stop_hwmon(struct memlat_hwmon *hw) -{ - int cpu; - struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); - struct dev_stats *devstats; - - for_each_cpu(cpu, &cpu_grp->cpus) { - delete_events(to_cpustats(cpu_grp, cpu)); - - /* Clear governor data */ - devstats = to_devstats(cpu_grp, cpu); - devstats->inst_count = 0; - devstats->mem_count = 0; - devstats->freq = 0; - devstats->stall_pct = 0; + event->prev_count = event->last_delta = 0; + if (event->pevent) { + perf_event_release_kernel(event->pevent); + event->pevent = NULL; } } @@ -174,59 +237,214 @@ static struct perf_event_attr *alloc_attr(void) return attr; } -static int set_events(struct cpu_grp_info *cpu_grp, int cpu) +static int set_event(struct event_data *ev, int cpu, unsigned int event_id, + struct perf_event_attr *attr) { struct perf_event *pevent; - struct perf_event_attr *attr; - int err, i; - unsigned int event_id; - struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); - /* Allocate an attribute for event initialization */ - attr = alloc_attr(); - if (!attr) - return -ENOMEM; + if (!event_id) + return 0; - for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { - event_id = cpu_grp->event_ids[i]; - if (!event_id) - continue; + attr->config = event_id; + pevent = perf_event_create_kernel_counter(attr, cpu, NULL, NULL, NULL); + if (IS_ERR(pevent)) + return PTR_ERR(pevent); - attr->config = event_id; - pevent = perf_event_create_kernel_counter(attr, cpu, NULL, - NULL, NULL); - if (IS_ERR(pevent)) - goto err_out; - cpustats->events[i].pevent = pevent; - perf_event_enable(pevent); - } + ev->pevent = pevent; + perf_event_enable(pevent); - kfree(attr); return 0; - -err_out: - err = PTR_ERR(pevent); - kfree(attr); - return err; } -static int start_hwmon(struct memlat_hwmon *hw) +static int init_common_evs(struct memlat_cpu_grp *cpu_grp, + struct perf_event_attr *attr) { - int cpu, ret = 0; - struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + unsigned int cpu, i; + int ret = 0; for_each_cpu(cpu, &cpu_grp->cpus) { - ret = set_events(cpu_grp, cpu); - if (ret < 0) { - pr_warn("Perf event init failed on CPU%d: %d\n", cpu, - ret); - break; + struct event_data *cpustats = to_common_cpustats(cpu_grp, cpu); + + for (i = 0; i < NUM_COMMON_EVS; i++) { + ret = set_event(&cpustats[i], cpu, + cpu_grp->common_ev_ids[i], attr); + if (ret < 0) + break; } } return ret; } +static void free_common_evs(struct memlat_cpu_grp *cpu_grp) +{ + unsigned int cpu, i; + + for_each_cpu(cpu, &cpu_grp->cpus) { + struct event_data *cpustats = to_common_cpustats(cpu_grp, cpu); + + for (i = 0; i < NUM_COMMON_EVS; i++) + delete_event(&cpustats[i]); + } +} + +static void memlat_monitor_work(struct work_struct *work) +{ + int err; + struct memlat_cpu_grp *cpu_grp = + container_of(work, struct memlat_cpu_grp, work.work); + struct memlat_mon *mon; + unsigned int i; + + mutex_lock(&cpu_grp->mons_lock); + if (!cpu_grp->num_active_mons) + goto unlock_out; + update_counts(cpu_grp); + for (i = 0; i < cpu_grp->num_mons; i++) { + struct devfreq *df; + + mon = &cpu_grp->mons[i]; + + if (!mon->is_active) + continue; + + df = mon->hw.df; + mutex_lock(&df->lock); + err = update_devfreq(df); + if (err < 0) + dev_err(mon->hw.dev, "Memlat update failed: %d\n", err); + mutex_unlock(&df->lock); + } + + queue_delayed_work(memlat_wq, &cpu_grp->work, + msecs_to_jiffies(cpu_grp->update_ms)); + +unlock_out: + mutex_unlock(&cpu_grp->mons_lock); +} + +static int start_hwmon(struct memlat_hwmon *hw) +{ + int ret = 0; + unsigned int cpu; + struct memlat_mon *mon = to_mon(hw); + struct memlat_cpu_grp *cpu_grp = mon->cpu_grp; + bool should_init_cpu_grp; + struct perf_event_attr *attr = alloc_attr(); + + if (!attr) + return -ENOMEM; + + mutex_lock(&cpu_grp->mons_lock); + should_init_cpu_grp = !(cpu_grp->num_active_mons++); + if (should_init_cpu_grp) { + ret = init_common_evs(cpu_grp, attr); + if (ret < 0) + goto unlock_out; + + INIT_DEFERRABLE_WORK(&cpu_grp->work, &memlat_monitor_work); + } + + if (mon->miss_ev) { + for_each_cpu(cpu, &mon->cpus) { + unsigned int idx = cpu - cpumask_first(&mon->cpus); + + ret = set_event(&mon->miss_ev[idx], cpu, + mon->miss_ev_id, attr); + if (ret < 0) + goto unlock_out; + } + } + + mon->is_active = true; + + if (should_init_cpu_grp) + queue_delayed_work(memlat_wq, &cpu_grp->work, + msecs_to_jiffies(cpu_grp->update_ms)); + +unlock_out: + mutex_unlock(&cpu_grp->mons_lock); + kfree(attr); + + return ret; +} + +static void stop_hwmon(struct memlat_hwmon *hw) +{ + unsigned int cpu; + struct memlat_mon *mon = to_mon(hw); + struct memlat_cpu_grp *cpu_grp = mon->cpu_grp; + + mutex_lock(&cpu_grp->mons_lock); + mon->is_active = false; + cpu_grp->num_active_mons--; + + for_each_cpu(cpu, &mon->cpus) { + unsigned int idx = cpu - cpumask_first(&mon->cpus); + struct dev_stats *devstats = to_devstats(mon, cpu); + + if (mon->miss_ev) + delete_event(&mon->miss_ev[idx]); + devstats->inst_count = 0; + devstats->mem_count = 0; + devstats->freq = 0; + devstats->stall_pct = 0; + } + + if (!cpu_grp->num_active_mons) { + cancel_delayed_work(&cpu_grp->work); + free_common_evs(cpu_grp); + } + mutex_unlock(&cpu_grp->mons_lock); +} + +/** + * We should set update_ms to the lowest requested_update_ms of all of the + * active mons, or 0 (i.e. stop polling) if ALL active mons have 0. + * This is expected to be called with cpu_grp->mons_lock taken. + */ +static void set_update_ms(struct memlat_cpu_grp *cpu_grp) +{ + struct memlat_mon *mon; + unsigned int i, new_update_ms = UINT_MAX; + + for (i = 0; i < cpu_grp->num_mons; i++) { + mon = &cpu_grp->mons[i]; + if (mon->is_active && mon->requested_update_ms) + new_update_ms = + min(new_update_ms, mon->requested_update_ms); + } + + if (new_update_ms == UINT_MAX) { + cancel_delayed_work(&cpu_grp->work); + } else if (cpu_grp->update_ms == UINT_MAX) { + queue_delayed_work(memlat_wq, &cpu_grp->work, + msecs_to_jiffies(new_update_ms)); + } else if (new_update_ms > cpu_grp->update_ms) { + cancel_delayed_work(&cpu_grp->work); + queue_delayed_work(memlat_wq, &cpu_grp->work, + msecs_to_jiffies(new_update_ms)); + } + + cpu_grp->update_ms = new_update_ms; +} + +static void request_update_ms(struct memlat_hwmon *hw, unsigned int update_ms) +{ + struct devfreq *df = hw->df; + struct memlat_mon *mon = to_mon(hw); + struct memlat_cpu_grp *cpu_grp = mon->cpu_grp; + + mutex_lock(&df->lock); + df->profile->polling_ms = update_ms; + mutex_unlock(&df->lock); + + mutex_lock(&cpu_grp->mons_lock); + mon->requested_update_ms = update_ms; + set_update_ms(cpu_grp); + mutex_unlock(&cpu_grp->mons_lock); +} + static int get_mask_from_dev_handle(struct platform_device *pdev, cpumask_t *mask) { @@ -273,71 +491,38 @@ static struct device_node *parse_child_nodes(struct device *dev) return NULL; } -static int arm_memlat_mon_driver_probe(struct platform_device *pdev) +#define DEFAULT_UPDATE_MS 100 +static int memlat_cpu_grp_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; - struct memlat_hwmon *hw; - struct cpu_grp_info *cpu_grp; - const struct memlat_mon_spec *spec; - int cpu, ret; - u32 event_id; + struct memlat_cpu_grp *cpu_grp; + int ret = 0; + unsigned int i, event_id, num_cpus, num_mons; cpu_grp = devm_kzalloc(dev, sizeof(*cpu_grp), GFP_KERNEL); if (!cpu_grp) return -ENOMEM; - hw = &cpu_grp->hw; - - hw->dev = dev; - hw->of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); - if (!hw->of_node) { - dev_err(dev, "Couldn't find a target device\n"); - return -ENODEV; - } if (get_mask_from_dev_handle(pdev, &cpu_grp->cpus)) { - dev_err(dev, "CPU list is empty\n"); + dev_err(dev, "No CPUs specified.\n"); return -ENODEV; } - hw->num_cores = cpumask_weight(&cpu_grp->cpus); - hw->core_stats = devm_kzalloc(dev, hw->num_cores * - sizeof(*(hw->core_stats)), GFP_KERNEL); - if (!hw->core_stats) - return -ENOMEM; + num_mons = of_get_available_child_count(dev->of_node); - cpu_grp->cpustats = devm_kzalloc(dev, hw->num_cores * - sizeof(*(cpu_grp->cpustats)), GFP_KERNEL); - if (!cpu_grp->cpustats) - return -ENOMEM; - - cpu_grp->event_ids[CYC_IDX] = CYC_EV; - - for_each_cpu(cpu, &cpu_grp->cpus) - to_devstats(cpu_grp, cpu)->id = cpu; - - hw->start_hwmon = &start_hwmon; - hw->stop_hwmon = &stop_hwmon; - hw->get_cnt = &get_cnt; - if (of_get_child_count(dev->of_node)) - hw->get_child_of_node = &parse_child_nodes; - - spec = of_device_get_match_data(dev); - if (spec && spec->is_compute) { - ret = register_compute(dev, hw); - if (ret) - pr_err("Compute Gov registration failed\n"); - - return ret; + if (!num_mons) { + dev_err(dev, "No mons provided.\n"); + return -ENODEV; } - ret = of_property_read_u32(dev->of_node, "qcom,cachemiss-ev", - &event_id); - if (ret < 0) { - dev_dbg(dev, "Cache Miss event not specified. Using def:0x%x\n", - L2DM_EV); - event_id = L2DM_EV; - } - cpu_grp->event_ids[CM_IDX] = event_id; + cpu_grp->num_mons = num_mons; + cpu_grp->num_inited_mons = 0; + + cpu_grp->mons = + devm_kzalloc(dev, num_mons * sizeof(*cpu_grp->mons), + GFP_KERNEL); + if (!cpu_grp->mons) + return -ENOMEM; ret = of_property_read_u32(dev->of_node, "qcom,inst-ev", &event_id); if (ret < 0) { @@ -345,30 +530,200 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) INST_EV); event_id = INST_EV; } - cpu_grp->event_ids[INST_IDX] = event_id; + cpu_grp->common_ev_ids[INST_IDX] = event_id; - ret = of_property_read_u32(dev->of_node, "qcom,stall-cycle-ev", - &event_id); - if (ret) - dev_dbg(dev, "Stall cycle event not specified. Event ignored.\n"); - else - cpu_grp->event_ids[STALL_CYC_IDX] = event_id; + ret = of_property_read_u32(dev->of_node, "qcom,cyc-ev", &event_id); + if (ret < 0) { + dev_dbg(dev, "Cyc event not specified. Using def:0x%x\n", + CYC_EV); + event_id = CYC_EV; + } + cpu_grp->common_ev_ids[CYC_IDX] = event_id; - ret = register_memlat(dev, hw); + ret = of_property_read_u32(dev->of_node, "qcom,stall-ev", &event_id); if (ret < 0) - pr_err("Mem Latency Gov registration failed: %d\n", ret); + dev_dbg(dev, "Stall event not specified. Skipping.\n"); + else + cpu_grp->common_ev_ids[STALL_IDX] = event_id; + num_cpus = cpumask_weight(&cpu_grp->cpus); + cpu_grp->common_evs = + devm_kzalloc(dev, num_cpus * sizeof(*cpu_grp->common_evs), + GFP_KERNEL); + if (!cpu_grp->common_evs) + return -ENOMEM; + + for (i = 0; i < num_cpus; i++) { + cpu_grp->common_evs[i] = + devm_kzalloc(dev, NUM_COMMON_EVS * + sizeof(*cpu_grp->common_evs[i]), + GFP_KERNEL); + if (!cpu_grp->common_evs[i]) + return -ENOMEM; + } + + mutex_init(&cpu_grp->mons_lock); + cpu_grp->update_ms = DEFAULT_UPDATE_MS; + + dev_set_drvdata(dev, cpu_grp); + + return 0; +} + +static int memlat_mon_probe(struct platform_device *pdev, bool is_compute) +{ + struct device *dev = &pdev->dev; + int ret = 0; + struct memlat_cpu_grp *cpu_grp; + struct memlat_mon *mon; + struct memlat_hwmon *hw; + unsigned int event_id, num_cpus, cpu; + + if (!memlat_wq) + memlat_wq = create_freezable_workqueue("memlat_wq"); + + if (!memlat_wq) { + dev_err(dev, "Couldn't create memlat workqueue.\n"); + return -ENOMEM; + } + + cpu_grp = dev_get_drvdata(dev->parent); + if (!cpu_grp) { + dev_err(dev, "Mon initialized without cpu_grp.\n"); + return -ENODEV; + } + + mutex_lock(&cpu_grp->mons_lock); + mon = &cpu_grp->mons[cpu_grp->num_inited_mons]; + mon->is_active = false; + mon->requested_update_ms = 0; + mon->cpu_grp = cpu_grp; + + if (get_mask_from_dev_handle(pdev, &mon->cpus)) { + cpumask_copy(&mon->cpus, &cpu_grp->cpus); + } else { + if (!cpumask_subset(&mon->cpus, &cpu_grp->cpus)) { + dev_err(dev, + "Mon CPUs must be a subset of cpu_grp CPUs. mon=%*pbl cpu_grp=%*pbl\n", + mon->cpus, cpu_grp->cpus); + ret = -EINVAL; + goto unlock_out; + } + } + + num_cpus = cpumask_weight(&mon->cpus); + + hw = &mon->hw; + hw->of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); + if (!hw->of_node) { + dev_err(dev, "Couldn't find a target device.\n"); + ret = -ENODEV; + goto unlock_out; + } + hw->dev = dev; + hw->num_cores = num_cpus; + hw->should_ignore_df_monitor = true; + hw->core_stats = devm_kzalloc(dev, num_cpus * sizeof(*(hw->core_stats)), + GFP_KERNEL); + if (!hw->core_stats) { + ret = -ENOMEM; + goto unlock_out; + } + + for_each_cpu(cpu, &mon->cpus) + to_devstats(mon, cpu)->id = cpu; + + hw->start_hwmon = &start_hwmon; + hw->stop_hwmon = &stop_hwmon; + hw->get_cnt = &get_cnt; + if (of_get_child_count(dev->of_node)) + hw->get_child_of_node = &parse_child_nodes; + hw->request_update_ms = &request_update_ms; + + /* + * Compute mons rely solely on common events. + */ + if (is_compute) { + mon->miss_ev_id = 0; + ret = register_compute(dev, hw); + } else { + mon->miss_ev = + devm_kzalloc(dev, num_cpus * sizeof(*mon->miss_ev), + GFP_KERNEL); + if (!mon->miss_ev) { + ret = -ENOMEM; + goto unlock_out; + } + + ret = of_property_read_u32(dev->of_node, "qcom,cachemiss-ev", + &event_id); + if (ret < 0) { + dev_err(dev, "Cache miss event missing for mon: %d\n", + ret); + ret = -EINVAL; + goto unlock_out; + } + mon->miss_ev_id = event_id; + + ret = register_memlat(dev, hw); + } + + if (!ret) + cpu_grp->num_inited_mons++; + +unlock_out: + mutex_unlock(&cpu_grp->mons_lock); return ret; } +static int arm_memlat_mon_driver_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + int ret = 0; + const struct memlat_mon_spec *spec = of_device_get_match_data(dev); + enum mon_type type = NUM_MON_TYPES; + + if (spec) + type = spec->type; + + switch (type) { + case MEMLAT_CPU_GRP: + ret = memlat_cpu_grp_probe(pdev); + if (of_get_available_child_count(dev->of_node)) + of_platform_populate(dev->of_node, NULL, NULL, dev); + break; + case MEMLAT_MON: + ret = memlat_mon_probe(pdev, false); + break; + case COMPUTE_MON: + ret = memlat_mon_probe(pdev, true); + break; + default: + /* + * This should never happen. + */ + dev_err(dev, "Invalid memlat mon type specified: %u\n", type); + return -EINVAL; + } + + if (ret < 0) { + dev_err(dev, "Failure to probe memlat device: %d\n", ret); + return ret; + } + + return 0; +} + static const struct memlat_mon_spec spec[] = { - [0] = { false }, - [1] = { true }, + [0] = { MEMLAT_CPU_GRP }, + [1] = { MEMLAT_MON }, + [2] = { COMPUTE_MON }, }; static const struct of_device_id memlat_match_table[] = { - { .compatible = "qcom,arm-memlat-mon", .data = &spec[0] }, - { .compatible = "qcom,arm-compute-mon", .data = &spec[1] }, + { .compatible = "qcom,arm-memlat-cpugrp", .data = &spec[0] }, + { .compatible = "qcom,arm-memlat-mon", .data = &spec[1] }, + { .compatible = "qcom,arm-compute-mon", .data = &spec[2] }, {} }; diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index 04ff9abdefb9..3194527e06a2 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -147,7 +147,8 @@ static int start_monitor(struct devfreq *df) return ret; } - devfreq_monitor_start(df); + if (!hw->should_ignore_df_monitor) + devfreq_monitor_start(df); node->mon_started = true; @@ -161,7 +162,9 @@ static void stop_monitor(struct devfreq *df) node->mon_started = false; - devfreq_monitor_stop(df); + if (!hw->should_ignore_df_monitor) + devfreq_monitor_stop(df); + hw->stop_hwmon(hw); } @@ -341,13 +344,15 @@ static struct attribute_group compute_dev_attr_group = { .attrs = compute_dev_attr, }; -#define MIN_MS 10U +#define MIN_MS 0U #define MAX_MS 500U static int devfreq_memlat_ev_handler(struct devfreq *df, unsigned int event, void *data) { int ret; unsigned int sample_ms; + struct memlat_node *node; + struct memlat_hwmon *hw; switch (event) { case DEVFREQ_GOV_START: @@ -395,10 +400,15 @@ static int devfreq_memlat_ev_handler(struct devfreq *df, break; case DEVFREQ_GOV_INTERVAL: + node = df->data; + hw = node->hw; sample_ms = *(unsigned int *)data; sample_ms = max(MIN_MS, sample_ms); sample_ms = min(MAX_MS, sample_ms); - devfreq_interval_update(df, &sample_ms); + if (hw->request_update_ms) + hw->request_update_ms(hw, sample_ms); + if (!hw->should_ignore_df_monitor) + devfreq_interval_update(df, &sample_ms); break; } diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h index b1c334018f16..7fe65f4caf8c 100644 --- a/drivers/devfreq/governor_memlat.h +++ b/drivers/devfreq/governor_memlat.h @@ -55,6 +55,8 @@ struct memlat_hwmon { void (*stop_hwmon)(struct memlat_hwmon *hw); unsigned long (*get_cnt)(struct memlat_hwmon *hw); struct device_node *(*get_child_of_node)(struct device *dev); + void (*request_update_ms)(struct memlat_hwmon *hw, + unsigned int update_ms); struct device *dev; struct device_node *of_node; @@ -63,6 +65,7 @@ struct memlat_hwmon { struct devfreq *df; struct core_dev_map *freq_map; + bool should_ignore_df_monitor; }; #ifdef CONFIG_DEVFREQ_GOV_MEMLAT From 3c55215ea57ecb8996047d8b3cea6a2b6f97236b Mon Sep 17 00:00:00 2001 From: Amir Vajid Date: Thu, 13 Jun 2019 11:31:25 -0700 Subject: [PATCH 16/27] PM / devfreq: memlat: optimize freq and stall_pct calculations Currently, we calculate freq and stall_pct multiple times for the same shared PMU counters. Calculate these once per cpu_grp to prevent repeating the same calculation multiple times later. Change-Id: I3b594e01ca50a583f45174e69a24f0b0eca32dde Signed-off-by: Amir Vajid [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/arm-memlat-mon.c | 74 ++++++++++++++++---------------- 1 file changed, 38 insertions(+), 36 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index cd1f4c84fe6d..092736cd101d 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -49,6 +49,12 @@ struct event_data { unsigned long last_delta; }; +struct cpu_data { + struct event_data common_evs[NUM_COMMON_EVS]; + unsigned long freq; + unsigned long stall_pct; +}; + /** * struct memlat_mon - A specific consumer of cpu_grp generic counters. * @@ -85,8 +91,9 @@ struct memlat_mon { * * @cpus: The CPUs this cpu_grp will read events from. * @common_ev_ids: The event codes of the events all mons need. - * @common_evs: The event data of the events all mons need. - * Organized as a 2d array of [cpu][event]. + * @cpus_data: The cpus data array of length #cpus. Includes + * event_data of all the events all mons need as + * well as common computed cpu data like freq. * @last_update_ts: Used to avoid redundant reads. * @last_ts_delta_us: The time difference between the most recent * update and the one before that. Used to compute @@ -104,7 +111,7 @@ struct memlat_mon { struct memlat_cpu_grp { cpumask_t cpus; unsigned int common_ev_ids[NUM_COMMON_EVS]; - struct event_data **common_evs; + struct cpu_data *cpus_data; ktime_t last_update_ts; unsigned long last_ts_delta_us; @@ -122,8 +129,10 @@ struct memlat_mon_spec { enum mon_type type; }; -#define to_common_cpustats(cpu_grp, cpu) \ - (cpu_grp->common_evs[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_cpu_data(cpu_grp, cpu) \ + (&cpu_grp->cpus_data[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_common_evs(cpu_grp, cpu) \ + (cpu_grp->cpus_data[cpu - cpumask_first(&cpu_grp->cpus)].common_evs) #define to_devstats(mon, cpu) \ (&mon->hw.core_stats[cpu - cpumask_first(&mon->cpus)]) #define to_mon(hwmon) container_of(hwmon, struct memlat_mon, hw) @@ -156,15 +165,20 @@ static void update_counts(struct memlat_cpu_grp *cpu_grp) cpu_grp->last_update_ts = now; for_each_cpu(cpu, &cpu_grp->cpus) { - struct event_data *cpu_grp_stats = - to_common_cpustats(cpu_grp, cpu); + struct cpu_data *cpu_data = to_cpu_data(cpu_grp, cpu); + struct event_data *common_evs = cpu_data->common_evs; for (i = 0; i < NUM_COMMON_EVS; i++) - read_event(&cpu_grp_stats[i]); + read_event(&common_evs[i]); - if (!cpu_grp_stats[STALL_IDX].pevent) - cpu_grp_stats[STALL_IDX].last_delta = - cpu_grp_stats[CYC_IDX].last_delta; + if (!common_evs[STALL_IDX].pevent) + common_evs[STALL_IDX].last_delta = + common_evs[CYC_IDX].last_delta; + + cpu_data->freq = common_evs[CYC_IDX].last_delta / delta; + cpu_data->stall_pct = mult_frac(100, + common_evs[STALL_IDX].last_delta, + common_evs[CYC_IDX].last_delta); } for (i = 0; i < cpu_grp->num_mons; i++) { @@ -188,18 +202,15 @@ static unsigned long get_cnt(struct memlat_hwmon *hw) unsigned int cpu; for_each_cpu(cpu, &mon->cpus) { - struct event_data *cpu_grp_stats = - to_common_cpustats(cpu_grp, cpu); + struct cpu_data *cpu_data = to_cpu_data(cpu_grp, cpu); + struct event_data *common_evs = cpu_data->common_evs; unsigned int mon_idx = cpu - cpumask_first(&mon->cpus); struct dev_stats *devstats = to_devstats(mon, cpu); - devstats->freq = cpu_grp_stats[CYC_IDX].last_delta / - cpu_grp->last_ts_delta_us; - devstats->stall_pct = - mult_frac(100, cpu_grp_stats[STALL_IDX].last_delta, - cpu_grp_stats[CYC_IDX].last_delta); - devstats->inst_count = cpu_grp_stats[INST_IDX].last_delta; + devstats->freq = cpu_data->freq; + devstats->stall_pct = cpu_data->stall_pct; + devstats->inst_count = common_evs[INST_IDX].last_delta; if (mon->miss_ev) devstats->mem_count = @@ -263,10 +274,10 @@ static int init_common_evs(struct memlat_cpu_grp *cpu_grp, int ret = 0; for_each_cpu(cpu, &cpu_grp->cpus) { - struct event_data *cpustats = to_common_cpustats(cpu_grp, cpu); + struct event_data *common_evs = to_common_evs(cpu_grp, cpu); for (i = 0; i < NUM_COMMON_EVS; i++) { - ret = set_event(&cpustats[i], cpu, + ret = set_event(&common_evs[i], cpu, cpu_grp->common_ev_ids[i], attr); if (ret < 0) break; @@ -281,10 +292,10 @@ static void free_common_evs(struct memlat_cpu_grp *cpu_grp) unsigned int cpu, i; for_each_cpu(cpu, &cpu_grp->cpus) { - struct event_data *cpustats = to_common_cpustats(cpu_grp, cpu); + struct event_data *common_evs = to_common_evs(cpu_grp, cpu); for (i = 0; i < NUM_COMMON_EVS; i++) - delete_event(&cpustats[i]); + delete_event(&common_evs[i]); } } @@ -497,7 +508,7 @@ static int memlat_cpu_grp_probe(struct platform_device *pdev) struct device *dev = &pdev->dev; struct memlat_cpu_grp *cpu_grp; int ret = 0; - unsigned int i, event_id, num_cpus, num_mons; + unsigned int event_id, num_cpus, num_mons; cpu_grp = devm_kzalloc(dev, sizeof(*cpu_grp), GFP_KERNEL); if (!cpu_grp) @@ -547,21 +558,12 @@ static int memlat_cpu_grp_probe(struct platform_device *pdev) cpu_grp->common_ev_ids[STALL_IDX] = event_id; num_cpus = cpumask_weight(&cpu_grp->cpus); - cpu_grp->common_evs = - devm_kzalloc(dev, num_cpus * sizeof(*cpu_grp->common_evs), + cpu_grp->cpus_data = + devm_kzalloc(dev, num_cpus * sizeof(*cpu_grp->cpus_data), GFP_KERNEL); - if (!cpu_grp->common_evs) + if (!cpu_grp->cpus_data) return -ENOMEM; - for (i = 0; i < num_cpus; i++) { - cpu_grp->common_evs[i] = - devm_kzalloc(dev, NUM_COMMON_EVS * - sizeof(*cpu_grp->common_evs[i]), - GFP_KERNEL); - if (!cpu_grp->common_evs[i]) - return -ENOMEM; - } - mutex_init(&cpu_grp->mons_lock); cpu_grp->update_ms = DEFAULT_UPDATE_MS; From 71876da229a1820e87e3dc77be58fc209ede92c8 Mon Sep 17 00:00:00 2001 From: Amir Vajid Date: Thu, 23 May 2019 14:17:34 -0700 Subject: [PATCH 17/27] PM / devfreq: Add support for memory latency QoS voting Add a devfreq device that has the ability to vote for a QoS level to ensure better memory latency. Change-Id: I33094b45ec250a58f244ecb1b7b8476df8791425 Signed-off-by: Amir Vajid --- drivers/devfreq/Kconfig | 13 +++ drivers/devfreq/Makefile | 1 + drivers/devfreq/devfreq_qcom_qoslat.c | 147 ++++++++++++++++++++++++++ 3 files changed, 161 insertions(+) create mode 100644 drivers/devfreq/devfreq_qcom_qoslat.c diff --git a/drivers/devfreq/Kconfig b/drivers/devfreq/Kconfig index 624ad55216b0..c4cfa29a777a 100644 --- a/drivers/devfreq/Kconfig +++ b/drivers/devfreq/Kconfig @@ -197,6 +197,19 @@ config QCOM_DEVFREQ_ICC agnostic interface to so that some of the devfreq governors can be shared across SoCs. +config ARM_QCOM_DEVFREQ_QOSLAT + bool "Qualcomm Technologies Inc. DEVFREQ QOSLAT device driver" + depends on ARCH_QCOM + select DEVFREQ_GOV_PERFORMANCE + select DEVFREQ_GOV_POWERSAVE + select DEVFREQ_GOV_USERSPACE + default n + help + Some Qualcomm Technologies, Inc. (QTI) chipsets have an + interface to vote for a memory latency QoS level. This + driver votes on this interface to request a particular + memory latency QoS level. + source "drivers/devfreq/event/Kconfig" endif # PM_DEVFREQ diff --git a/drivers/devfreq/Makefile b/drivers/devfreq/Makefile index 359b04eb21be..9385d503f68f 100644 --- a/drivers/devfreq/Makefile +++ b/drivers/devfreq/Makefile @@ -19,6 +19,7 @@ obj-$(CONFIG_ARM_TEGRA_DEVFREQ) += tegra30-devfreq.o obj-$(CONFIG_ARM_TEGRA20_DEVFREQ) += tegra20-devfreq.o obj-$(CONFIG_QCOM_DEVFREQ_ICC) += devfreq_icc.o obj-$(CONFIG_DEVFREQ_SIMPLE_DEV) += devfreq_simple_dev.o +obj-$(CONFIG_ARM_QCOM_DEVFREQ_QOSLAT) += devfreq_qcom_qoslat.o # DEVFREQ Event Drivers obj-$(CONFIG_PM_DEVFREQ_EVENT) += event/ diff --git a/drivers/devfreq/devfreq_qcom_qoslat.c b/drivers/devfreq/devfreq_qcom_qoslat.c new file mode 100644 index 000000000000..3c739858ba0c --- /dev/null +++ b/drivers/devfreq/devfreq_qcom_qoslat.c @@ -0,0 +1,147 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "devfreq-qcom-qoslat: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +struct qoslat_data { + struct mbox_client mbox_cl; + struct mbox_chan *mbox; + struct devfreq *df; + struct devfreq_dev_profile profile; + unsigned int qos_level; +}; + +#define MAX_MSG_LEN 96 +static int update_qos_level(struct device *dev, struct qoslat_data *d) +{ + struct qmp_pkt pkt; + char mbox_msg[MAX_MSG_LEN + 1] = {0}; + char *qos_msg = "off"; + int ret; + + if (d->qos_level) + qos_msg = "on"; + + snprintf(mbox_msg, MAX_MSG_LEN, "{class: ddr, perfmode: %s}", qos_msg); + pkt.size = MAX_MSG_LEN; + pkt.data = mbox_msg; + + ret = mbox_send_message(d->mbox, &pkt); + if (ret < 0) { + dev_err(dev, "Failed to send mbox message: %d\n", ret); + return ret; + } + + return 0; +} + +static int dev_target(struct device *dev, unsigned long *freq, u32 flags) +{ + struct qoslat_data *d = dev_get_drvdata(dev); + struct dev_pm_opp *opp; + + opp = devfreq_recommended_opp(dev, freq, flags); + if (!IS_ERR(opp)) + dev_pm_opp_put(opp); + else + return PTR_ERR(opp); + + if (*freq == d->qos_level) + return 0; + + d->qos_level = *freq; + + return update_qos_level(dev, d); +} + +static int dev_get_cur_freq(struct device *dev, unsigned long *freq) +{ + struct qoslat_data *d = dev_get_drvdata(dev); + + *freq = d->qos_level; + + return 0; +} + +static int devfreq_qcom_qoslat_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct qoslat_data *d; + struct devfreq_dev_profile *p; + const char *gov_name; + int ret = 0; + + d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); + if (!d) + return -ENOMEM; + dev_set_drvdata(dev, d); + + if (!of_find_property(dev->of_node, "mboxes", NULL)) { + dev_err(dev, "Couldn't find AOP mbox\n"); + return -EINVAL; + } + d->mbox_cl.dev = dev; + d->mbox_cl.tx_block = true; + d->mbox_cl.tx_tout = 1000; + d->mbox_cl.knows_txdone = false; + d->mbox = mbox_request_channel(&d->mbox_cl, 0); + if (IS_ERR(d->mbox)) { + ret = PTR_ERR(d->mbox); + dev_err(dev, "Failed to get mailbox channel: %d\n", ret); + return ret; + } + d->qos_level = 0; + + p = &d->profile; + p->target = dev_target; + p->get_cur_freq = dev_get_cur_freq; + p->polling_ms = 10; + + ret = dev_pm_opp_of_add_table(dev); + if (ret < 0) + dev_err(dev, "Couldn't parse OPP table: %d\n", ret); + + if (of_property_read_string(dev->of_node, "governor", &gov_name)) + gov_name = "powersave"; + + d->df = devfreq_add_device(dev, p, gov_name, NULL); + if (IS_ERR(d->df)) { + ret = PTR_ERR(d->df); + dev_err(dev, "Failed to add devfreq device: %d\n", ret); + return ret; + } + + return 0; +} + +static const struct of_device_id devfreq_qoslat_match_table[] = { + { .compatible = "qcom,devfreq-qoslat" }, + {} +}; + +static struct platform_driver devfreq_qcom_qoslat_driver = { + .probe = devfreq_qcom_qoslat_probe, + .driver = { + .name = "devfreq-qcom-qoslat", + .of_match_table = devfreq_qoslat_match_table, + }, +}; +module_platform_driver(devfreq_qcom_qoslat_driver); +MODULE_DESCRIPTION("Device driver for setting memory latency qos level"); +MODULE_LICENSE("GPL v2"); From e43263bdea3eece3d543627f5e8c326235bdf8b6 Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Wed, 16 Jan 2019 20:47:00 -0800 Subject: [PATCH 18/27] perf: Introduce a LLCC PMU Some chips have hardware that can count misses for LLCC at a per-CPU level. This PMU serves as an intermediary that allows us to retrieve these values for use in other drivers. Change-Id: I1dc3090a64ec7d5b12a36b0395c42930128287fe Signed-off-by: Jonathan Avila [avajid@codeaurora.org: removed BEAC registers, unused variables and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/perf/Kconfig | 9 ++ drivers/perf/Makefile | 1 + drivers/perf/qcom_llcc_pmu.c | 176 +++++++++++++++++++++++++++++++++++ 3 files changed, 186 insertions(+) create mode 100644 drivers/perf/qcom_llcc_pmu.c diff --git a/drivers/perf/Kconfig b/drivers/perf/Kconfig index 09ae8a970880..b052c29b6ffb 100644 --- a/drivers/perf/Kconfig +++ b/drivers/perf/Kconfig @@ -105,6 +105,15 @@ config QCOM_L3_PMU Adds the L3 cache PMU into the perf events subsystem for monitoring L3 cache events. +config QCOM_LLCC_PMU + bool "Qualcomm Technologies LLCC PMU" + depends on ARCH_QCOM && ARM64 + help + Provides support for the LLCC performance monitor unit (PMU) in + Qualcomm Technologies processors. + Adds the LLCC PMU into the perf events subsystem for monitoring + LLCC miss events. + config THUNDERX2_PMU tristate "Cavium ThunderX2 SoC PMU UNCORE" depends on ARCH_THUNDER2 && ARM64 && ACPI && NUMA diff --git a/drivers/perf/Makefile b/drivers/perf/Makefile index 2ebb4de17815..36b514fcde94 100644 --- a/drivers/perf/Makefile +++ b/drivers/perf/Makefile @@ -9,6 +9,7 @@ obj-$(CONFIG_FSL_IMX8_DDR_PMU) += fsl_imx8_ddr_perf.o obj-$(CONFIG_HISI_PMU) += hisilicon/ obj-$(CONFIG_QCOM_L2_PMU) += qcom_l2_pmu.o obj-$(CONFIG_QCOM_L3_PMU) += qcom_l3_pmu.o +obj-$(CONFIG_QCOM_LLCC_PMU) += qcom_llcc_pmu.o obj-$(CONFIG_THUNDERX2_PMU) += thunderx2_pmu.o obj-$(CONFIG_XGENE_PMU) += xgene_pmu.o obj-$(CONFIG_ARM_SPE_PMU) += arm_spe_pmu.o diff --git a/drivers/perf/qcom_llcc_pmu.c b/drivers/perf/qcom_llcc_pmu.c new file mode 100644 index 000000000000..e5e50244a35a --- /dev/null +++ b/drivers/perf/qcom_llcc_pmu.c @@ -0,0 +1,176 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2017-2019, The Linux Foundation. All rights reserved. + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +struct llcc_pmu { + struct pmu pmu; + struct hlist_node node; + void __iomem *lagg_base; + struct perf_event event; +}; + +#define MON_CFG(m) ((m)->lagg_base + 0x200) +#define MON_CNT(m, cpu) ((m)->lagg_base + 0x220 + 0x4 * cpu) +#define to_llcc_pmu(ptr) (container_of(ptr, struct llcc_pmu, pmu)) + +#define LLCC_RD_EV 0x1000 +#define ENABLE 0x01 +#define CLEAR 0x10 +#define DISABLE 0x00 +#define SCALING_FACTOR 0x3 +#define NUM_COUNTERS NR_CPUS +#define VALUE_MASK 0xFFFFFF + +static u64 llcc_stats[NUM_COUNTERS]; +static unsigned int users; +static raw_spinlock_t counter_lock; +static raw_spinlock_t users_lock; +static ktime_t last_read; + +static int qcom_llcc_event_init(struct perf_event *event) +{ + u64 config = event->attr.config; + + if (config == LLCC_RD_EV) { + event->hw.config_base = event->attr.config; + return 0; + } else + return -ENOENT; +} + +static void qcom_llcc_event_read(struct perf_event *event) +{ + int i = 0, cpu = event->cpu; + unsigned long raw, irq_flags; + struct llcc_pmu *llccpmu = to_llcc_pmu(event->pmu); + ktime_t cur; + + raw_spin_lock_irqsave(&counter_lock, irq_flags); + cur = ktime_get(); + if (ktime_ms_delta(cur, last_read) > 1) { + writel_relaxed(DISABLE, MON_CFG(llccpmu)); + for (i = 0; i < NUM_COUNTERS; i++) { + raw = readl_relaxed(MON_CNT(llccpmu, i)); + raw &= VALUE_MASK; + llcc_stats[i] += (u64) raw << SCALING_FACTOR; + } + last_read = cur; + writel_relaxed(CLEAR, MON_CFG(llccpmu)); + writel_relaxed(ENABLE, MON_CFG(llccpmu)); + } + + if (!(event->hw.state & PERF_HES_STOPPED)) + local64_set(&event->count, llcc_stats[cpu]); + raw_spin_unlock_irqrestore(&counter_lock, irq_flags); +} + +static void qcom_llcc_event_start(struct perf_event *event, int flags) +{ + if (flags & PERF_EF_RELOAD) + WARN_ON(!(event->hw.state & PERF_HES_UPTODATE)); + event->hw.state = 0; +} + +static void qcom_llcc_event_stop(struct perf_event *event, int flags) +{ + qcom_llcc_event_read(event); + event->hw.state |= PERF_HES_STOPPED | PERF_HES_UPTODATE; +} + +static int qcom_llcc_event_add(struct perf_event *event, int flags) +{ + struct llcc_pmu *llccpmu = to_llcc_pmu(event->pmu); + + raw_spin_lock(&users_lock); + if (!users) + writel_relaxed(ENABLE, MON_CFG(llccpmu)); + users++; + raw_spin_unlock(&users_lock); + + event->hw.state = PERF_HES_STOPPED | PERF_HES_UPTODATE; + + if (flags & PERF_EF_START) + qcom_llcc_event_start(event, PERF_EF_RELOAD); + + return 0; +} + +static void qcom_llcc_event_del(struct perf_event *event, int flags) +{ + struct llcc_pmu *llccpmu = to_llcc_pmu(event->pmu); + + raw_spin_lock(&users_lock); + users--; + if (!users) + writel_relaxed(DISABLE, MON_CFG(llccpmu)); + raw_spin_unlock(&users_lock); +} + +static int qcom_llcc_pmu_probe(struct platform_device *pdev) +{ + struct llcc_pmu *llccpmu; + struct resource *res; + int ret; + + llccpmu = devm_kzalloc(&pdev->dev, sizeof(struct llcc_pmu), GFP_KERNEL); + if (!llccpmu) + return -ENOMEM; + + llccpmu->pmu = (struct pmu) { + .task_ctx_nr = perf_invalid_context, + + .event_init = qcom_llcc_event_init, + .add = qcom_llcc_event_add, + .del = qcom_llcc_event_del, + .start = qcom_llcc_event_start, + .stop = qcom_llcc_event_stop, + .read = qcom_llcc_event_read, + }; + + res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "lagg-base"); + llccpmu->lagg_base = devm_ioremap_resource(&pdev->dev, res); + if (IS_ERR(llccpmu->lagg_base)) { + dev_err(&pdev->dev, "Can't map PMU lagg base: @%pa\n", + &res->start); + return PTR_ERR(llccpmu->lagg_base); + } + + raw_spin_lock_init(&counter_lock); + raw_spin_lock_init(&users_lock); + + ret = perf_pmu_register(&llccpmu->pmu, "llcc-pmu", -1); + if (ret < 0) + dev_err(&pdev->dev, "Failed to register LLCC PMU (%d)\n", ret); + + dev_info(&pdev->dev, "Registered llcc_pmu, type: %d\n", + llccpmu->pmu.type); + + return 0; +} + +static const struct of_device_id qcom_llcc_pmu_match_table[] = { + { .compatible = "qcom,qcom-llcc-pmu" }, + {} +}; + +static struct platform_driver qcom_llcc_pmu_driver = { + .driver = { + .name = "qcom-llcc-pmu", + .of_match_table = qcom_llcc_pmu_match_table, + }, + .probe = qcom_llcc_pmu_probe, +}; + +module_platform_driver(qcom_llcc_pmu_driver); From b6d80a179b926891beb98414f28480a4197cb279 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Wed, 23 Jan 2019 12:04:14 -0800 Subject: [PATCH 19/27] qcom-llcc-pmu: Update the LLCC PMU configurations for kona Kona has new LLCC PMU version. This brought in changes in the monitor enable, disable and configuration registers and hence made code changes accordingly to read cache misses for multiple CPUs. Change-Id: I4d25487966a31f12143e2f99264a25d8c38f0188 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: updated version enum, fixed disable/clear masks and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/perf/qcom_llcc_pmu.c | 172 +++++++++++++++++++++++++++++++---- 1 file changed, 153 insertions(+), 19 deletions(-) diff --git a/drivers/perf/qcom_llcc_pmu.c b/drivers/perf/qcom_llcc_pmu.c index e5e50244a35a..ed82c97ed37d 100644 --- a/drivers/perf/qcom_llcc_pmu.c +++ b/drivers/perf/qcom_llcc_pmu.c @@ -4,6 +4,7 @@ */ #include +#include #include #include #include @@ -14,11 +15,17 @@ #include #include +enum llcc_pmu_version { + LLCC_PMU_VER1 = 1, + LLCC_PMU_VER2, +}; + struct llcc_pmu { struct pmu pmu; struct hlist_node node; void __iomem *lagg_base; struct perf_event event; + enum llcc_pmu_version ver; }; #define MON_CFG(m) ((m)->lagg_base + 0x200) @@ -26,9 +33,10 @@ struct llcc_pmu { #define to_llcc_pmu(ptr) (container_of(ptr, struct llcc_pmu, pmu)) #define LLCC_RD_EV 0x1000 -#define ENABLE 0x01 +#define ENABLE 0x1 #define CLEAR 0x10 -#define DISABLE 0x00 +#define CLEAR_POS 16 +#define DISABLE 0x0 #define SCALING_FACTOR 0x3 #define NUM_COUNTERS NR_CPUS #define VALUE_MASK 0xFFFFFF @@ -38,6 +46,93 @@ static unsigned int users; static raw_spinlock_t counter_lock; static raw_spinlock_t users_lock; static ktime_t last_read; +static DEFINE_PER_CPU(unsigned int, users_alive); + +static void mon_disable(struct llcc_pmu *llccpmu, int cpu) +{ + u32 reg; + + if (!llccpmu->ver) { + pr_err("LLCCPMU version not correct\n"); + return; + } + + switch (llccpmu->ver) { + case LLCC_PMU_VER1: + writel_relaxed(DISABLE, MON_CFG(llccpmu)); + break; + case LLCC_PMU_VER2: + reg = readl_relaxed(MON_CFG(llccpmu)); + reg &= ~(ENABLE << cpu); + writel_relaxed(reg, MON_CFG(llccpmu)); + break; + } +} + +static void mon_clear(struct llcc_pmu *llccpmu, int cpu) +{ + int clear_bit = CLEAR_POS + cpu; + u32 reg; + + if (!llccpmu->ver) { + pr_err("LLCCPMU version not correct\n"); + return; + } + + switch (llccpmu->ver) { + case LLCC_PMU_VER1: + writel_relaxed(CLEAR, MON_CFG(llccpmu)); + break; + case LLCC_PMU_VER2: + reg = readl_relaxed(MON_CFG(llccpmu)); + reg |= (ENABLE << clear_bit); + writel_relaxed(reg, MON_CFG(llccpmu)); + reg &= ~(ENABLE << clear_bit); + writel_relaxed(reg, MON_CFG(llccpmu)); + break; + } +} + +static void mon_enable(struct llcc_pmu *llccpmu, int cpu) +{ + u32 reg; + + if (!llccpmu->ver) { + pr_err("LLCCPMU version not correct\n"); + return; + } + + switch (llccpmu->ver) { + case LLCC_PMU_VER1: + writel_relaxed(ENABLE, MON_CFG(llccpmu)); + break; + case LLCC_PMU_VER2: + reg = readl_relaxed(MON_CFG(llccpmu)); + reg |= (ENABLE << cpu); + writel_relaxed(reg, MON_CFG(llccpmu)); + break; + } +} + +static unsigned long read_cnt(struct llcc_pmu *llccpmu, int cpu) +{ + unsigned long value; + + if (!llccpmu->ver) { + pr_err("LLCCPMU version not correct\n"); + return -EINVAL; + } + + switch (llccpmu->ver) { + case LLCC_PMU_VER1: + value = readl_relaxed(MON_CNT(llccpmu, cpu)); + break; + case LLCC_PMU_VER2: + value = readl_relaxed(MON_CNT(llccpmu, cpu)); + break; + } + return value; +} static int qcom_llcc_event_init(struct perf_event *event) { @@ -58,17 +153,26 @@ static void qcom_llcc_event_read(struct perf_event *event) ktime_t cur; raw_spin_lock_irqsave(&counter_lock, irq_flags); - cur = ktime_get(); - if (ktime_ms_delta(cur, last_read) > 1) { - writel_relaxed(DISABLE, MON_CFG(llccpmu)); - for (i = 0; i < NUM_COUNTERS; i++) { - raw = readl_relaxed(MON_CNT(llccpmu, i)); - raw &= VALUE_MASK; - llcc_stats[i] += (u64) raw << SCALING_FACTOR; + if (llccpmu->ver == LLCC_PMU_VER1) { + cur = ktime_get(); + if (ktime_ms_delta(cur, last_read) > 1) { + mon_disable(llccpmu, cpu); + for (i = 0; i < NUM_COUNTERS; i++) { + raw = read_cnt(llccpmu, i); + raw &= VALUE_MASK; + llcc_stats[i] += (u64) raw << SCALING_FACTOR; + } + last_read = cur; + mon_clear(llccpmu, cpu); + mon_enable(llccpmu, cpu); } - last_read = cur; - writel_relaxed(CLEAR, MON_CFG(llccpmu)); - writel_relaxed(ENABLE, MON_CFG(llccpmu)); + } else { + mon_disable(llccpmu, cpu); + raw = read_cnt(llccpmu, cpu); + raw &= VALUE_MASK; + llcc_stats[cpu] += (u64) raw << SCALING_FACTOR; + mon_clear(llccpmu, cpu); + mon_enable(llccpmu, cpu); } if (!(event->hw.state & PERF_HES_STOPPED)) @@ -92,11 +196,22 @@ static void qcom_llcc_event_stop(struct perf_event *event, int flags) static int qcom_llcc_event_add(struct perf_event *event, int flags) { struct llcc_pmu *llccpmu = to_llcc_pmu(event->pmu); + unsigned int cpu_users; raw_spin_lock(&users_lock); - if (!users) - writel_relaxed(ENABLE, MON_CFG(llccpmu)); - users++; + + if (llccpmu->ver == LLCC_PMU_VER1) { + if (!users) + mon_enable(llccpmu, event->cpu); + users++; + } else { + cpu_users = per_cpu(users_alive, event->cpu); + if (!cpu_users) + mon_enable(llccpmu, event->cpu); + cpu_users++; + per_cpu(users_alive, event->cpu) = cpu_users; + } + raw_spin_unlock(&users_lock); event->hw.state = PERF_HES_STOPPED | PERF_HES_UPTODATE; @@ -110,11 +225,22 @@ static int qcom_llcc_event_add(struct perf_event *event, int flags) static void qcom_llcc_event_del(struct perf_event *event, int flags) { struct llcc_pmu *llccpmu = to_llcc_pmu(event->pmu); + unsigned int cpu_users; raw_spin_lock(&users_lock); - users--; - if (!users) - writel_relaxed(DISABLE, MON_CFG(llccpmu)); + + if (llccpmu->ver == LLCC_PMU_VER1) { + users--; + if (!users) + mon_disable(llccpmu, event->cpu); + } else { + cpu_users = per_cpu(users_alive, event->cpu); + cpu_users--; + if (!cpu_users) + mon_disable(llccpmu, event->cpu); + per_cpu(users_alive, event->cpu) = cpu_users; + } + raw_spin_unlock(&users_lock); } @@ -128,6 +254,13 @@ static int qcom_llcc_pmu_probe(struct platform_device *pdev) if (!llccpmu) return -ENOMEM; + llccpmu->ver = (enum llcc_pmu_version) + of_device_get_match_data(&pdev->dev); + if (!llccpmu->ver) { + pr_err("Unknown device type!\n"); + return -ENODEV; + } + llccpmu->pmu = (struct pmu) { .task_ctx_nr = perf_invalid_context, @@ -161,7 +294,8 @@ static int qcom_llcc_pmu_probe(struct platform_device *pdev) } static const struct of_device_id qcom_llcc_pmu_match_table[] = { - { .compatible = "qcom,qcom-llcc-pmu" }, + { .compatible = "qcom,llcc-pmu-ver1", .data = (void *) LLCC_PMU_VER1 }, + { .compatible = "qcom,llcc-pmu-ver2", .data = (void *) LLCC_PMU_VER2 }, {} }; From 6f5e9162ed1a62fbd8e5c906f99852275cd657a2 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Thu, 19 Sep 2019 17:02:06 -0700 Subject: [PATCH 20/27] devfreq: bwmon: Increase the IOPercentage limits to 400 The IO percentage limits the values to 100, increase it since some devices might need a higher value. Change-Id: I7f19017b426832d9a1a11bd02195a4d15634c041 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: resolved trivial merge conflict] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_bw_hwmon.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index fd95f4f145a8..f78118e32e5b 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -754,7 +754,7 @@ show_attr(decay_rate); store_attr(decay_rate, 0U, 100U); static DEVICE_ATTR_RW(decay_rate); show_attr(io_percent); -store_attr(io_percent, 1U, 100U); +store_attr(io_percent, 1U, 400U); static DEVICE_ATTR_RW(io_percent); show_attr(bw_step); store_attr(bw_step, 50U, 1000U); From 6b1a168b7eab8d66fc50239d9928e72e8ff7718e Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Tue, 17 Sep 2019 15:40:55 -0700 Subject: [PATCH 21/27] devfreq: Allow bw_hwmon resume with zero resume freq Currently bw_hwmon governor does not allow the device to be resume if the resume_freq is set to zero. But there can be devices that vote for zero frequency when there is no bandwidth on the device. This does not mean that the device should not be resume from suspend. Remove the check with prevents resume, instead add a check in the devfreq framework to check if device is suspended or not before calling the governor resume. Change-Id: Ib82a6a36308aee52e8bb989fddca92265de6c4a4 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: removed suspended check in devfreq framework] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_bw_hwmon.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index f78118e32e5b..65d10f69b609 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -675,11 +675,6 @@ static int gov_resume(struct devfreq *df) if (!node->hw->resume_hwmon) return -EPERM; - if (!node->resume_freq) { - dev_warn(df->dev.parent, "Governor already resumed!\n"); - return -EBUSY; - } - mutex_lock(&df->lock); update_devfreq(df); mutex_unlock(&df->lock); From 1bb4b043b3e1c3edb79a8d3e3a934a363f171579 Mon Sep 17 00:00:00 2001 From: Amir Vajid Date: Thu, 19 Sep 2019 14:18:51 -0700 Subject: [PATCH 22/27] PM / devfreq: qoslat: Update voting level definitions Having OPP levels of 0 and 1 causes a minor error in OPP debugfs since an OPP level of 0 causes the OPP index to be used instead (1 in this case) which later clashes with an actual OPP of 1. Update the qoslat OPP levels to avoid this. Change-Id: I23c0e736c9f1bfe77a99af44ced180d488a52eeb Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq_qcom_qoslat.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/drivers/devfreq/devfreq_qcom_qoslat.c b/drivers/devfreq/devfreq_qcom_qoslat.c index 3c739858ba0c..a75154138cb5 100644 --- a/drivers/devfreq/devfreq_qcom_qoslat.c +++ b/drivers/devfreq/devfreq_qcom_qoslat.c @@ -27,6 +27,9 @@ struct qoslat_data { unsigned int qos_level; }; +#define QOS_LEVEL_OFF 1 +#define QOS_LEVEL_ON 2 + #define MAX_MSG_LEN 96 static int update_qos_level(struct device *dev, struct qoslat_data *d) { @@ -35,7 +38,7 @@ static int update_qos_level(struct device *dev, struct qoslat_data *d) char *qos_msg = "off"; int ret; - if (d->qos_level) + if (d->qos_level == QOS_LEVEL_ON) qos_msg = "on"; snprintf(mbox_msg, MAX_MSG_LEN, "{class: ddr, perfmode: %s}", qos_msg); @@ -106,7 +109,7 @@ static int devfreq_qcom_qoslat_probe(struct platform_device *pdev) dev_err(dev, "Failed to get mailbox channel: %d\n", ret); return ret; } - d->qos_level = 0; + d->qos_level = QOS_LEVEL_OFF; p = &d->profile; p->target = dev_target; From 52ab3535b24cd6464f9f59927892962b329e5f66 Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Wed, 16 Jan 2019 20:27:47 -0800 Subject: [PATCH 23/27] Revert "PM / devfreq: Modify the device name as devfreq(X) for sysfs" This reverts commit 4585fbcb5331fc910b7e553ad3efd0dd7b320d14. The change to be reverted would add probe order-dependent numbers to the end of each devfreq device, making our post boot script (among others) incredibly brittle. Change-Id: I8fdb679ae651d9f425ffd60e2141613ae20bb029 Signed-off-by: Jonathan Avila [avajid@codeaurora.org: resolved trival merge conflicts] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/drivers/devfreq/devfreq.c b/drivers/devfreq/devfreq.c index 446490c9d635..18285305b86f 100644 --- a/drivers/devfreq/devfreq.c +++ b/drivers/devfreq/devfreq.c @@ -613,7 +613,6 @@ struct devfreq *devfreq_add_device(struct device *dev, { struct devfreq *devfreq; struct devfreq_governor *governor; - static atomic_t devfreq_no = ATOMIC_INIT(-1); int err = 0; if (!dev || !profile || !governor_name) { @@ -676,8 +675,7 @@ struct devfreq *devfreq_add_device(struct device *dev, devfreq->suspend_freq = dev_pm_opp_get_suspend_opp_freq(dev); atomic_set(&devfreq->suspend_count, 0); - dev_set_name(&devfreq->dev, "devfreq%d", - atomic_inc_return(&devfreq_no)); + dev_set_name(&devfreq->dev, "%s", dev_name(dev)); err = device_register(&devfreq->dev); if (err) { mutex_unlock(&devfreq->lock); From 9693b4d62052f177b91bdd0ff8c7f2323f822385 Mon Sep 17 00:00:00 2001 From: Jonathan Avila Date: Wed, 16 Jan 2019 20:28:27 -0800 Subject: [PATCH 24/27] PM / devfreq: Introduce an event lock Currently, concurrent writes to sysfs entries leave the possibility for race conditions within the devfreq framework. For example, concurrently executing max_freq_store and governor_store can result in attempting to perform an update_devfreq() before the new governor's start handler can be executed. A more concrete case is a race between polling_interval_store and governor_store. Because no lock is used after calling into the event handler of the old governor and there's nothing preventing work from being queued after the monitor is stopped, it's possible to accidentally cause delayed work to be queued on the governor being switched to. This can be seen if you create two threads, one which changes a device's governor between simple_ondemand and performance, and one which changes its polling interval between 45 and 50. All of these races can be addressed with the introduction of a lock that prevents sysfs operations from interleaving in this fashion. Change-Id: Ia6887dcb2d69dc2576837a6c09fed55a28943abc Signed-off-by: Jonathan Avila [avajid@codeaurora.org: renamed to event lock and only used when CONFIG_QCOM_DEVFREQ_ICC is enabled] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq.c | 14 +++++++++++++- include/linux/devfreq.h | 28 ++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/devfreq.c b/drivers/devfreq/devfreq.c index 18285305b86f..a076b1dac541 100644 --- a/drivers/devfreq/devfreq.c +++ b/drivers/devfreq/devfreq.c @@ -595,6 +595,7 @@ static void devfreq_dev_release(struct device *dev) devfreq->profile->exit(devfreq->dev.parent); mutex_destroy(&devfreq->lock); + event_mutex_destroy(devfreq); kfree(devfreq); } @@ -637,6 +638,7 @@ struct devfreq *devfreq_add_device(struct device *dev, } mutex_init(&devfreq->lock); + event_mutex_init(devfreq); mutex_lock(&devfreq->lock); devfreq->dev.parent = dev; devfreq->dev.class = devfreq_class; @@ -1145,12 +1147,13 @@ static ssize_t governor_store(struct device *dev, struct device_attribute *attr, goto out; } + event_mutex_lock(df); if (df->governor) { ret = df->governor->event_handler(df, DEVFREQ_GOV_STOP, NULL); if (ret) { dev_warn(dev, "%s: Governor %s not stopped(%d)\n", __func__, df->governor->name, ret); - goto out; + goto gov_stop_out; } } prev_governor = df->governor; @@ -1171,6 +1174,9 @@ static ssize_t governor_store(struct device *dev, struct device_attribute *attr, df->governor = NULL; } } + +gov_stop_out: + event_mutex_unlock(df); out: mutex_unlock(&devfreq_list_lock); @@ -1265,8 +1271,10 @@ static ssize_t polling_interval_store(struct device *dev, if (ret != 1) return -EINVAL; + event_mutex_lock(df); df->governor->event_handler(df, DEVFREQ_GOV_INTERVAL, &value); ret = count; + event_mutex_unlock(df); return ret; } @@ -1283,6 +1291,7 @@ static ssize_t min_freq_store(struct device *dev, struct device_attribute *attr, if (ret != 1) return -EINVAL; + event_mutex_lock(df); mutex_lock(&df->lock); if (value) { @@ -1305,6 +1314,7 @@ static ssize_t min_freq_store(struct device *dev, struct device_attribute *attr, ret = count; unlock: mutex_unlock(&df->lock); + event_mutex_unlock(df); return ret; } @@ -1327,6 +1337,7 @@ static ssize_t max_freq_store(struct device *dev, struct device_attribute *attr, if (ret != 1) return -EINVAL; + event_mutex_lock(df); mutex_lock(&df->lock); if (value) { @@ -1349,6 +1360,7 @@ static ssize_t max_freq_store(struct device *dev, struct device_attribute *attr, ret = count; unlock: mutex_unlock(&df->lock); + event_mutex_unlock(df); return ret; } static DEVICE_ATTR_RW(min_freq); diff --git a/include/linux/devfreq.h b/include/linux/devfreq.h index 2bae9ed3c783..d19581dac8da 100644 --- a/include/linux/devfreq.h +++ b/include/linux/devfreq.h @@ -149,6 +149,9 @@ struct devfreq { struct list_head node; struct mutex lock; +#ifdef CONFIG_QCOM_DEVFREQ_ICC + struct mutex event_lock; +#endif struct device dev; struct devfreq_dev_profile *profile; const struct devfreq_governor *governor; @@ -185,6 +188,31 @@ struct devfreq_freqs { unsigned long new; }; +static inline void event_mutex_init(struct devfreq *devfreq) +{ +#ifdef CONFIG_QCOM_DEVFREQ_ICC + mutex_init(&devfreq->event_lock); +#endif +} +static inline void event_mutex_destroy(struct devfreq *devfreq) +{ +#ifdef CONFIG_QCOM_DEVFREQ_ICC + mutex_destroy(&devfreq->event_lock); +#endif +} +static inline void event_mutex_lock(struct devfreq *devfreq) +{ +#ifdef CONFIG_QCOM_DEVFREQ_ICC + mutex_lock(&devfreq->event_lock); +#endif +} +static inline void event_mutex_unlock(struct devfreq *devfreq) +{ +#ifdef CONFIG_QCOM_DEVFREQ_ICC + mutex_unlock(&devfreq->event_lock); +#endif +} + #if defined(CONFIG_PM_DEVFREQ) extern struct devfreq *devfreq_add_device(struct device *dev, struct devfreq_dev_profile *profile, From 96784c4520a8270de98f47be0e70586a8474a0d5 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Wed, 10 Oct 2018 14:53:32 -0700 Subject: [PATCH 25/27] PM / devfreq: Fix race condition between suspend/resume and governor_store There is a race condition when the event governor_store is being executed from sysfs and the device issues a suspend. The devfreq data structures would become stale when the suspend tries to access them in the middle of the governor_store operation. Fix this issue by taking a lock around suspend and resume operations so that these operations are not concurrent with the other events from sysfs. Change-Id: Ifa0e93915a920cec3e0429966328a1128d61098b Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: updated to merge better with upstream changes and removed renaming since done in previous commit] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/devfreq/devfreq.c b/drivers/devfreq/devfreq.c index a076b1dac541..55ef66a11266 100644 --- a/drivers/devfreq/devfreq.c +++ b/drivers/devfreq/devfreq.c @@ -896,8 +896,10 @@ int devfreq_suspend_device(struct devfreq *devfreq) return 0; if (devfreq->governor) { + event_mutex_lock(devfreq); ret = devfreq->governor->event_handler(devfreq, DEVFREQ_GOV_SUSPEND, NULL); + event_mutex_unlock(devfreq); if (ret) return ret; } @@ -937,8 +939,10 @@ int devfreq_resume_device(struct devfreq *devfreq) } if (devfreq->governor) { + event_mutex_lock(devfreq); ret = devfreq->governor->event_handler(devfreq, DEVFREQ_GOV_RESUME, NULL); + event_mutex_unlock(devfreq); if (ret) return ret; } From db4f4a0933ce2f34ceaa96e4e55de4d7b1c25f2b Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Sat, 13 Oct 2018 14:54:56 -0700 Subject: [PATCH 26/27] PM/devfreq: Do not switch governors from sysfs when device is suspended There is a possibility to switch the governors from sysfs even when the device is in suspended state. This can cause a NOC error at times when trying to access the device's monitor registers in a suspended state. This change fixes this issue. Utilize suspend_count to know if the device is in suspended state or not. Check if the device is suspended before switching the governor from sysfs. Change-Id: I15055aa51daa35272be4667e5bafb8ccd7933098 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: updated to utilize suspend_count from upstream, resolved minor merge conflicts and updated commit text] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq.c | 34 ++++++++++++++++++++-------------- 1 file changed, 20 insertions(+), 14 deletions(-) diff --git a/drivers/devfreq/devfreq.c b/drivers/devfreq/devfreq.c index 55ef66a11266..56b7fcbd2ed2 100644 --- a/drivers/devfreq/devfreq.c +++ b/drivers/devfreq/devfreq.c @@ -887,30 +887,31 @@ EXPORT_SYMBOL(devm_devfreq_remove_device); */ int devfreq_suspend_device(struct devfreq *devfreq) { - int ret; + int ret = 0; if (!devfreq) return -EINVAL; + event_mutex_lock(devfreq); if (atomic_inc_return(&devfreq->suspend_count) > 1) - return 0; + goto unlock_out; if (devfreq->governor) { - event_mutex_lock(devfreq); ret = devfreq->governor->event_handler(devfreq, DEVFREQ_GOV_SUSPEND, NULL); - event_mutex_unlock(devfreq); if (ret) - return ret; + goto unlock_out; } if (devfreq->suspend_freq) { ret = devfreq_set_target(devfreq, devfreq->suspend_freq, 0); if (ret) - return ret; + goto unlock_out; } - return 0; +unlock_out: + event_mutex_unlock(devfreq); + return ret; } EXPORT_SYMBOL(devfreq_suspend_device); @@ -924,30 +925,31 @@ EXPORT_SYMBOL(devfreq_suspend_device); */ int devfreq_resume_device(struct devfreq *devfreq) { - int ret; + int ret = 0; if (!devfreq) return -EINVAL; + event_mutex_lock(devfreq); if (atomic_dec_return(&devfreq->suspend_count) >= 1) - return 0; + goto unlock_out; if (devfreq->resume_freq) { ret = devfreq_set_target(devfreq, devfreq->resume_freq, 0); if (ret) - return ret; + goto unlock_out; } if (devfreq->governor) { - event_mutex_lock(devfreq); ret = devfreq->governor->event_handler(devfreq, DEVFREQ_GOV_RESUME, NULL); - event_mutex_unlock(devfreq); if (ret) - return ret; + goto unlock_out; } - return 0; +unlock_out: + event_mutex_unlock(devfreq); + return ret; } EXPORT_SYMBOL(devfreq_resume_device); @@ -1152,6 +1154,10 @@ static ssize_t governor_store(struct device *dev, struct device_attribute *attr, } event_mutex_lock(df); + if (atomic_read(&df->suspend_count) > 0) { + ret = -EINVAL; + goto gov_stop_out; + } if (df->governor) { ret = df->governor->event_handler(df, DEVFREQ_GOV_STOP, NULL); if (ret) { From 5e33579a6a9c6e7d255804b0cacad0ccb0c6a0d7 Mon Sep 17 00:00:00 2001 From: Rama Aparna Mallavarapu Date: Tue, 17 Sep 2019 16:21:39 -0700 Subject: [PATCH 27/27] devfreq: Do not allow tunable updates when device is suspended Tunables update accesses the monitor's registers for some devices. When the device is suspended, the clocks for those devices are turned off and updating the tunables of a suspended device causes the device to crash. Fix this by checking the flag dev_suspended before allowing the tunable updates. Change-Id: I3a0938edf761a9ea6d39ffeec19f35c228d5c306 Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: updated to utilize suspend_count from upstream and resolved minor merge conflicts] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq.c | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/drivers/devfreq/devfreq.c b/drivers/devfreq/devfreq.c index 56b7fcbd2ed2..631b38f57f39 100644 --- a/drivers/devfreq/devfreq.c +++ b/drivers/devfreq/devfreq.c @@ -1274,14 +1274,16 @@ static ssize_t polling_interval_store(struct device *dev, unsigned int value; int ret; - if (!df->governor) - return -EINVAL; - ret = sscanf(buf, "%u", &value); if (ret != 1) return -EINVAL; event_mutex_lock(df); + if (!df->governor || atomic_read(&df->suspend_count) > 0) { + dev_warn(dev, "device suspended, operation not allowed\n"); + event_mutex_unlock(df); + return -EINVAL; + } df->governor->event_handler(df, DEVFREQ_GOV_INTERVAL, &value); ret = count; event_mutex_unlock(df); @@ -1302,6 +1304,11 @@ static ssize_t min_freq_store(struct device *dev, struct device_attribute *attr, return -EINVAL; event_mutex_lock(df); + if (atomic_read(&df->suspend_count) > 0) { + dev_warn(dev, "device suspended, min freq not allowed\n"); + event_mutex_unlock(df); + return -EINVAL; + } mutex_lock(&df->lock); if (value) { @@ -1348,6 +1355,11 @@ static ssize_t max_freq_store(struct device *dev, struct device_attribute *attr, return -EINVAL; event_mutex_lock(df); + if (atomic_read(&df->suspend_count) > 0) { + event_mutex_unlock(df); + dev_warn(dev, "device suspended, max freq not allowed\n"); + return -EINVAL; + } mutex_lock(&df->lock); if (value) {