From f482e490771a91bde276437abfee14d7a6e8f9f6 Mon Sep 17 00:00:00 2001 From: Saravana Kannan Date: Wed, 16 Jan 2019 18:28:39 -0800 Subject: [PATCH 01/20] PM / devfreq: bimc-bwmon: Add support for version 2 The version 2 of the BIMC BWMON HW doesn't reset the counter to 0 when it hits the threshold. It also has support for an overflow status register. Change-Id: I9f18d2153a2e5e762ec9950f26e0e7601468a80a Signed-off-by: Saravana Kannan [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 73 ++++++++++++++++++++++++++---------- 1 file changed, 54 insertions(+), 19 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 8a9e4ef91824..cbe35fb6e97b 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -15,6 +15,7 @@ #include #include #include +#include #include #include "governor_bw_hwmon.h" @@ -31,13 +32,19 @@ #define MON_MASK(m) ((m)->base + 0x298) #define MON_MATCH(m) ((m)->base + 0x29C) +struct bwmon_spec { + bool wrap_on_thres; + bool overflow; +}; + struct bwmon { - void __iomem *base; - void __iomem *global_base; - unsigned int mport; - unsigned int irq; - struct device *dev; - struct bw_hwmon hw; + void __iomem *base; + void __iomem *global_base; + unsigned int mport; + unsigned int irq; + const struct bwmon_spec *spec; + struct device *dev; + struct bw_hwmon hw; }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) @@ -94,7 +101,7 @@ static void mon_irq_disable(struct bwmon *m) writel_relaxed(val, MON_INT_EN(m)); } -static int mon_irq_status(struct bwmon *m) +static unsigned int mon_irq_status(struct bwmon *m) { u32 mval; @@ -103,12 +110,12 @@ static int mon_irq_status(struct bwmon *m) dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, readl_relaxed(GLB_INT_STATUS(m))); - return mval & 0x1; + return mval; } static void mon_irq_clear(struct bwmon *m) { - writel_relaxed(0x1, MON_INT_CLR(m)); + writel_relaxed(0x3, MON_INT_CLR(m)); /* Ensure the monitor IRQ is clear before clearing GLB IRQ */ mb(); writel_relaxed(1 << m->mport, GLB_INT_CLR(m)); @@ -127,14 +134,22 @@ static u32 mon_get_limit(struct bwmon *m) return readl_relaxed(MON_THRES(m)); } +#define THRES_HIT(status) (status & BIT(0)) +#define OVERFLOW(status) (status & BIT(1)) static unsigned long mon_get_count(struct bwmon *m) { - unsigned long count; + unsigned long count, status; count = readl_relaxed(MON_CNT(m)); + status = mon_irq_status(m); + dev_dbg(m->dev, "Counter: %08lx\n", count); - if (mon_irq_status(m)) + + if (OVERFLOW(status) && m->spec->overflow) + count += 0xFFFFFFFF; + if (THRES_HIT(status) && m->spec->wrap_on_thres) count += mon_get_limit(m); + dev_dbg(m->dev, "Actual Count: %08lx\n", count); return count; @@ -173,11 +188,17 @@ static unsigned long meas_bw_and_set_irq(struct bw_hwmon *hw, mbps = mon_get_count(m); mbps = bytes_to_mbps(mbps, us); + /* - * The fudging of mbps when calculating limit is to workaround a HW - * design issue. Needs further tuning. + * If the counter wraps on thres, don't set the thres too low. + * Setting it too low runs the risk of the counter wrapping around + * multiple times before the IRQ is processed. */ - limit = mbps_to_bytes(max(mbps, 400UL), sample_ms, tol); + if (likely(!m->spec->wrap_on_thres)) + limit = mbps_to_bytes(mbps, sample_ms, tol); + else + limit = mbps_to_bytes(max(mbps, 400UL), sample_ms, tol); + mon_set_limit(m, limit); mon_clear(m); @@ -273,11 +294,23 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) /*************************************************************************/ +static const struct bwmon_spec spec[] = { + { .wrap_on_thres = true, .overflow = false }, + { .wrap_on_thres = false, .overflow = true }, +}; + +static const struct of_device_id bimc_bwmon_match_table[] = { + { .compatible = "qcom,bimc-bwmon", .data = &spec[0] }, + { .compatible = "qcom,bimc-bwmon2", .data = &spec[1] }, + {} +}; + static int bimc_bwmon_driver_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; struct resource *res; struct bwmon *m; + const struct of_device_id *id; int ret; u32 data; @@ -293,6 +326,13 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) } m->mport = data; + id = of_match_device(bimc_bwmon_match_table, dev); + if (!id) { + dev_err(dev, "Unknown device type!\n"); + return -ENODEV; + } + m->spec = id->data; + res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "base"); if (!res) { dev_err(dev, "base not found!\n"); @@ -339,11 +379,6 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return 0; } -static const struct of_device_id bimc_bwmon_match_table[] = { - { .compatible = "qcom,bimc-bwmon" }, - {} -}; - static struct platform_driver bimc_bwmon_driver = { .probe = bimc_bwmon_driver_probe, .driver = { From b29a2f00d317c84e57dabd594e2c7dba4d4b4beb Mon Sep 17 00:00:00 2001 From: Junjie Wu Date: Wed, 16 Jan 2019 18:29:30 -0800 Subject: [PATCH 02/20] PM / devfreq: Refactor Cache HWmon governor to be more generic The refactor allows the governor to support multiple devfreq devices. This is done by having different HW monitor instances register their capability to monitor different devfreq devices and then picking the right HW monitor based on which devfreq device is using this governor. Change-Id: I72c0542ce97f3965e422df521e0ce86cad218d93 Signed-off-by: Junjie Wu [avajid@codeaurora.org: resolved minor conflict and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_cache_hwmon.c | 227 ++++++++++++++++++------- drivers/devfreq/governor_cache_hwmon.h | 34 +++- 2 files changed, 190 insertions(+), 71 deletions(-) diff --git a/drivers/devfreq/governor_cache_hwmon.c b/drivers/devfreq/governor_cache_hwmon.c index 86b02aba6450..53fe8f4a75bf 100644 --- a/drivers/devfreq/governor_cache_hwmon.c +++ b/drivers/devfreq/governor_cache_hwmon.c @@ -23,11 +23,36 @@ #include "governor.h" #include "governor_cache_hwmon.h" +struct cache_hwmon_node { + unsigned int cycles_per_low_req; + unsigned int cycles_per_med_req; + unsigned int cycles_per_high_req; + unsigned int min_busy; + unsigned int max_busy; + unsigned int tolerance_mrps; + unsigned int guard_band_mhz; + unsigned int decay_rate; + unsigned long prev_mhz; + ktime_t prev_ts; + struct list_head list; + void *orig_data; + struct cache_hwmon *hw; + struct attribute_group *attr_grp; +}; + +static LIST_HEAD(cache_hwmon_list); +static DEFINE_MUTEX(list_lock); + +static int use_cnt; +static DEFINE_MUTEX(state_lock); + #define show_attr(name) \ static ssize_t name##_show(struct device *dev, \ struct device_attribute *attr, char *buf) \ { \ - return scnprintf(buf, PAGE_SIZE, "%u\n", name); \ + struct devfreq *df = to_devfreq(dev); \ + struct cache_hwmon_node *hw = df->data; \ + return scnprintf(buf, PAGE_SIZE, "%u\n", hw->name); \ } #define store_attr(name, _min, _max) \ @@ -37,36 +62,42 @@ static ssize_t name##_store(struct device *dev, \ { \ int ret; \ unsigned int val; \ + struct devfreq *df = to_devfreq(dev); \ + struct cache_hwmon_node *hw = df->data; \ ret = kstrtoint(buf, 10, &val); \ if (ret < 0) \ return ret; \ val = max(val, _min); \ val = min(val, _max); \ - name = val; \ + hw->name = val; \ return count; \ } -static struct cache_hwmon *hw; -static unsigned int cycles_per_low_req; -static unsigned int cycles_per_med_req = 20; -static unsigned int cycles_per_high_req = 35; -static unsigned int min_busy = 100; -static unsigned int max_busy = 100; -static unsigned int tolerance_mrps = 5; -static unsigned int guard_band_mhz = 100; -static unsigned int decay_rate = 90; - #define MIN_MS 10U #define MAX_MS 500U -static unsigned int sample_ms = 50; -static unsigned long prev_mhz; -static ktime_t prev_ts; -static unsigned long measure_mrps_and_set_irq(struct devfreq *df, +static struct cache_hwmon_node *find_hwmon_node(struct devfreq *df) +{ + struct cache_hwmon_node *node, *found = NULL; + + mutex_lock(&list_lock); + list_for_each_entry(node, &cache_hwmon_list, list) + if (node->hw->dev == df->dev.parent || + node->hw->of_node == df->dev.parent->of_node) { + found = node; + break; + } + mutex_unlock(&list_lock); + + return found; +} + +static unsigned long measure_mrps_and_set_irq(struct cache_hwmon_node *node, struct mrps_stats *stat) { ktime_t ts; unsigned int us; + struct cache_hwmon *hw = node->hw; /* * Since we are stopping the counters, we don't want this short work @@ -78,59 +109,63 @@ static unsigned long measure_mrps_and_set_irq(struct devfreq *df, preempt_disable(); ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, prev_ts)); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); if (!us) us = 1; - hw->meas_mrps_and_set_irq(df, tolerance_mrps, us, stat); - prev_ts = ts; + hw->meas_mrps_and_set_irq(hw, node->tolerance_mrps, us, stat); + node->prev_ts = ts; preempt_enable(); - pr_debug("stat H=%3lu, M=%3lu, T=%3lu, b=%3u, f=%4lu, us=%d\n", + dev_dbg(hw->df->dev.parent, + "stat H=%3lu, M=%3lu, T=%3lu, b=%3u, f=%4lu, us=%d\n", stat->high, stat->med, stat->high + stat->med, - stat->busy_percent, df->previous_freq / 1000, us); + stat->busy_percent, hw->df->previous_freq / 1000, us); return 0; } -static void compute_cache_freq(struct mrps_stats *mrps, unsigned long *freq) +static void compute_cache_freq(struct cache_hwmon_node *node, + struct mrps_stats *mrps, unsigned long *freq) { unsigned long new_mhz; unsigned int busy; - new_mhz = mrps->high * cycles_per_high_req - + mrps->med * cycles_per_med_req - + mrps->low * cycles_per_low_req; + new_mhz = mrps->high * node->cycles_per_high_req + + mrps->med * node->cycles_per_med_req + + mrps->low * node->cycles_per_low_req; - busy = max(min_busy, mrps->busy_percent); - busy = min(max_busy, busy); + busy = max(node->min_busy, mrps->busy_percent); + busy = min(node->max_busy, busy); new_mhz *= 100; new_mhz /= busy; - if (new_mhz < prev_mhz) { - new_mhz = new_mhz * decay_rate + prev_mhz * (100 - decay_rate); + if (new_mhz < node->prev_mhz) { + new_mhz = new_mhz * node->decay_rate + node->prev_mhz + * (100 - node->decay_rate); new_mhz /= 100; } - prev_mhz = new_mhz; + node->prev_mhz = new_mhz; - new_mhz += guard_band_mhz; + new_mhz += node->guard_band_mhz; *freq = new_mhz * 1000; } #define TOO_SOON_US (1 * USEC_PER_MSEC) static irqreturn_t mon_intr_handler(int irq, void *dev) { - struct devfreq *df = dev; + struct cache_hwmon_node *node = dev; + struct devfreq *df = node->hw->df; ktime_t ts; unsigned int us; int ret; - if (!hw->is_valid_irq(df)) + if (!node->hw->is_valid_irq(node->hw)) return IRQ_NONE; - pr_debug("Got interrupt\n"); + dev_dbg(df->dev.parent, "Got interrupt\n"); devfreq_monitor_stop(df); /* @@ -146,12 +181,13 @@ static irqreturn_t mon_intr_handler(int irq, void *dev) * readjusted. */ ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, prev_ts)); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); if (us > TOO_SOON_US) { mutex_lock(&df->lock); ret = update_devfreq(df); if (ret < 0) - pr_err("Unable to update freq on IRQ! (%d)\n", ret); + dev_err(df->dev.parent, + "Unable to update freq on IRQ! (%d)\n", ret); mutex_unlock(&df->lock); } @@ -164,9 +200,11 @@ static int devfreq_cache_hwmon_get_freq(struct devfreq *df, unsigned long *freq) { struct mrps_stats stat; + struct cache_hwmon_node *node = df->data; - measure_mrps_and_set_irq(df, &stat); - compute_cache_freq(&stat, freq); + memset(&stat, 0, sizeof(stat)); + measure_mrps_and_set_irq(node, &stat); + compute_cache_freq(node, &stat, freq); return 0; } @@ -217,58 +255,89 @@ static int start_monitoring(struct devfreq *df) { int ret; struct mrps_stats mrps; + struct device *dev = df->dev.parent; + struct cache_hwmon_node *node; + struct cache_hwmon *hw; - prev_ts = ktime_get(); - prev_mhz = 0; - mrps.high = (df->previous_freq / 1000) - guard_band_mhz; - mrps.high /= cycles_per_high_req; + node = find_hwmon_node(df); + if (!node) { + dev_err(dev, "Unable to find HW monitor!\n"); + return -ENODEV; + } + hw = node->hw; + hw->df = df; + node->orig_data = df->data; + df->data = node; - ret = hw->start_hwmon(df, &mrps); + node->prev_ts = ktime_get(); + node->prev_mhz = 0; + mrps.high = (df->previous_freq / 1000) - node->guard_band_mhz; + mrps.high /= node->cycles_per_high_req; + mrps.med = mrps.low = 0; + + ret = hw->start_hwmon(hw, &mrps); if (ret < 0) { - pr_err("Unable to start HW monitor! (%d)\n", ret); - return ret; + dev_err(dev, "Unable to start HW monitor! (%d)\n", ret); + goto err_start; } devfreq_monitor_start(df); - ret = request_threaded_irq(hw->irq, NULL, mon_intr_handler, + if (hw->irq) + ret = request_threaded_irq(hw->irq, NULL, mon_intr_handler, IRQF_ONESHOT | IRQF_SHARED, - "cache_hwmon", df); + "cache_hwmon", node); if (ret < 0) { - pr_err("Unable to register interrupt handler! (%d)\n", ret); + dev_err(dev, "Unable to register interrupt handler! (%d)\n", + ret); goto req_irq_fail; } ret = sysfs_create_group(&df->dev.kobj, &dev_attr_group); if (ret < 0) { - pr_err("Error creating sys entries! (%d)\n", ret); + dev_err(dev, "Error creating sys entries! (%d)\n", ret); goto sysfs_fail; } return 0; sysfs_fail: - disable_irq(hw->irq); - free_irq(hw->irq, df); + if (hw->irq) { + disable_irq(hw->irq); + free_irq(hw->irq, node); + } req_irq_fail: devfreq_monitor_stop(df); - hw->stop_hwmon(df); + hw->stop_hwmon(hw); +err_start: + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; return ret; } static void stop_monitoring(struct devfreq *df) { + struct cache_hwmon_node *node = df->data; + struct cache_hwmon *hw = node->hw; + sysfs_remove_group(&df->dev.kobj, &dev_attr_group); - disable_irq(hw->irq); - free_irq(hw->irq, df); + if (hw->irq) { + disable_irq(hw->irq); + free_irq(hw->irq, node); + } devfreq_monitor_stop(df); - hw->stop_hwmon(df); + hw->stop_hwmon(hw); + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; } static int devfreq_cache_hwmon_ev_handler(struct devfreq *df, unsigned int event, void *data) { int ret; + unsigned int sample_ms; switch (event) { case DEVFREQ_GOV_START: @@ -281,11 +350,11 @@ static int devfreq_cache_hwmon_ev_handler(struct devfreq *df, if (ret < 0) return ret; - pr_debug("Enabled Cache HW monitor governor\n"); + dev_dbg(df->dev.parent, "Enabled Cache HW monitor governor\n"); break; case DEVFREQ_GOV_STOP: stop_monitoring(df); - pr_debug("Disabled Cache HW monitor governor\n"); + dev_dbg(df->dev.parent, "Disabled Cache HW monitor governor\n"); break; case DEVFREQ_GOV_INTERVAL: sample_ms = *(unsigned int *)data; @@ -304,18 +373,48 @@ static struct devfreq_governor devfreq_cache_hwmon = { .event_handler = devfreq_cache_hwmon_ev_handler, }; -int register_cache_hwmon(struct cache_hwmon *hwmon) +int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon) { - int ret; + int ret = 0; + struct cache_hwmon_node *node; - hw = hwmon; - ret = devfreq_add_governor(&devfreq_cache_hwmon); - if (ret < 0) { - pr_err("devfreq governor registration failed: %d\n", ret); + if (!hwmon->dev && !hwmon->of_node) + return -EINVAL; + + node = devm_kzalloc(dev, sizeof(*node), GFP_KERNEL); + if (!node) + return -ENOMEM; + + node->cycles_per_med_req = 20; + node->cycles_per_high_req = 35; + node->min_busy = 100; + node->max_busy = 100; + node->tolerance_mrps = 5; + node->guard_band_mhz = 100; + node->decay_rate = 90; + node->hw = hwmon; + node->attr_grp = &dev_attr_group; + + mutex_lock(&state_lock); + if (!use_cnt) { + ret = devfreq_add_governor(&devfreq_cache_hwmon); + if (!ret) + use_cnt++; + } + mutex_unlock(&state_lock); + + if (!ret) { + dev_info(dev, "Cache HWmon governor registered.\n"); + } else { + dev_err(dev, "Failed to add Cache HWmon governor: %d\n", ret); return ret; } - return 0; + mutex_lock(&list_lock); + list_add_tail(&node->list, &cache_hwmon_list); + mutex_unlock(&list_lock); + + return ret; } MODULE_DESCRIPTION("HW monitor based cache freq driver"); diff --git a/drivers/devfreq/governor_cache_hwmon.h b/drivers/devfreq/governor_cache_hwmon.h index 75fbbdedba0f..8c492c58ee41 100644 --- a/drivers/devfreq/governor_cache_hwmon.h +++ b/drivers/devfreq/governor_cache_hwmon.h @@ -16,21 +16,41 @@ struct mrps_stats { unsigned int busy_percent; }; +/** + * struct cache_hwmon - devfreq Cache HW monitor info + * @start_hwmon: Start the HW monitoring + * @stop_hwmon: Stop the HW monitoring + * @is_valid_irq: Check whether the IRQ was triggered by the counter + * used to monitor cache activity. + * @meas_mrps_and_set_irq: Return the measured count and set up the + * IRQ to fire if usage exceeds current + * measurement by @tol percent. + * @irq: IRQ number that corresponds to this HW monitor. + * @dev: device that this HW monitor can monitor. + * @of_node: OF node of device that this HW monitor can monitor. + * @df: Devfreq node that this HW montior is being used + * for. NULL when not actively in use, and non-NULL + * when in use. + */ struct cache_hwmon { - int (*start_hwmon)(struct devfreq *df, + int (*start_hwmon)(struct cache_hwmon *hw, struct mrps_stats *mrps); - void (*stop_hwmon)(struct devfreq *df); - bool (*is_valid_irq)(struct devfreq *df); - unsigned long (*meas_mrps_and_set_irq)(struct devfreq *df, + void (*stop_hwmon)(struct cache_hwmon *hw); + bool (*is_valid_irq)(struct cache_hwmon *hw); + unsigned long (*meas_mrps_and_set_irq)(struct cache_hwmon *hw, unsigned int tol, unsigned int us, struct mrps_stats *mrps); - int irq; + int irq; + struct device *dev; + struct device_node *of_node; + struct devfreq *df; }; #ifdef CONFIG_DEVFREQ_GOV_QCOM_CACHE_HWMON -int register_cache_hwmon(struct cache_hwmon *hwmon); +int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon); #else -static inline int register_cache_hwmon(struct cache_hwmon *hwmon) +static inline int register_cache_hwmon(struct device *dev, + struct cache_hwmon *hwmon) { return 0; } From 939d572b127c7cd0c7102f05c3beecd575fa5236 Mon Sep 17 00:00:00 2001 From: Junjie Wu Date: Wed, 16 Jan 2019 18:30:32 -0800 Subject: [PATCH 03/20] PM / devfreq: cache_hwmon: Move IRQ handling to device drivers The cache monitoring devices might have more than one IRQ to handle or might have notifications from other drivers instead of using actual IRQs. So, refactor the governor to move the IRQ handling to the cache monitoring device specific drivers and just provide an API that can be used to request a re-evaluation. The device specific driver can call this API to request an immediate re-evaluation whenever the cache request has exceeded the previously set limit instead of waiting for the periodic update. Change-Id: Ib2e9f53f95749d659f440739a1b074b5a0d94fd8 Signed-off-by: Junjie Wu [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_cache_hwmon.c | 47 +++++++++++--------------- drivers/devfreq/governor_cache_hwmon.h | 10 +++--- 2 files changed, 25 insertions(+), 32 deletions(-) diff --git a/drivers/devfreq/governor_cache_hwmon.c b/drivers/devfreq/governor_cache_hwmon.c index 53fe8f4a75bf..e670ee9e7548 100644 --- a/drivers/devfreq/governor_cache_hwmon.c +++ b/drivers/devfreq/governor_cache_hwmon.c @@ -34,6 +34,7 @@ struct cache_hwmon_node { unsigned int decay_rate; unsigned long prev_mhz; ktime_t prev_ts; + bool mon_started; struct list_head list; void *orig_data; struct cache_hwmon *hw; @@ -154,18 +155,26 @@ static void compute_cache_freq(struct cache_hwmon_node *node, } #define TOO_SOON_US (1 * USEC_PER_MSEC) -static irqreturn_t mon_intr_handler(int irq, void *dev) +int update_cache_hwmon(struct cache_hwmon *hwmon) { - struct cache_hwmon_node *node = dev; - struct devfreq *df = node->hw->df; + struct cache_hwmon_node *node; + struct devfreq *df; ktime_t ts; unsigned int us; int ret; - if (!node->hw->is_valid_irq(node->hw)) - return IRQ_NONE; + if (!hwmon) + return -EINVAL; + df = hwmon->df; + if (!df) + return -ENODEV; + node = df->data; + if (!node) + return -ENODEV; + if (!node->mon_started) + return -EBUSY; - dev_dbg(df->dev.parent, "Got interrupt\n"); + dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); /* @@ -187,13 +196,13 @@ static irqreturn_t mon_intr_handler(int irq, void *dev) ret = update_devfreq(df); if (ret < 0) dev_err(df->dev.parent, - "Unable to update freq on IRQ! (%d)\n", ret); + "Unable to update freq on req! (%d)\n", ret); mutex_unlock(&df->lock); } devfreq_monitor_start(df); - return IRQ_HANDLED; + return 0; } static int devfreq_cache_hwmon_get_freq(struct devfreq *df, @@ -282,16 +291,7 @@ static int start_monitoring(struct devfreq *df) } devfreq_monitor_start(df); - - if (hw->irq) - ret = request_threaded_irq(hw->irq, NULL, mon_intr_handler, - IRQF_ONESHOT | IRQF_SHARED, - "cache_hwmon", node); - if (ret < 0) { - dev_err(dev, "Unable to register interrupt handler! (%d)\n", - ret); - goto req_irq_fail; - } + node->mon_started = true; ret = sysfs_create_group(&df->dev.kobj, &dev_attr_group); if (ret < 0) { @@ -302,11 +302,7 @@ static int start_monitoring(struct devfreq *df) return 0; sysfs_fail: - if (hw->irq) { - disable_irq(hw->irq); - free_irq(hw->irq, node); - } -req_irq_fail: + node->mon_started = false; devfreq_monitor_stop(df); hw->stop_hwmon(hw); err_start: @@ -322,10 +318,7 @@ static void stop_monitoring(struct devfreq *df) struct cache_hwmon *hw = node->hw; sysfs_remove_group(&df->dev.kobj, &dev_attr_group); - if (hw->irq) { - disable_irq(hw->irq); - free_irq(hw->irq, node); - } + node->mon_started = false; devfreq_monitor_stop(df); hw->stop_hwmon(hw); df->data = node->orig_data; diff --git a/drivers/devfreq/governor_cache_hwmon.h b/drivers/devfreq/governor_cache_hwmon.h index 8c492c58ee41..f78a1d7cd028 100644 --- a/drivers/devfreq/governor_cache_hwmon.h +++ b/drivers/devfreq/governor_cache_hwmon.h @@ -20,12 +20,9 @@ struct mrps_stats { * struct cache_hwmon - devfreq Cache HW monitor info * @start_hwmon: Start the HW monitoring * @stop_hwmon: Stop the HW monitoring - * @is_valid_irq: Check whether the IRQ was triggered by the counter - * used to monitor cache activity. * @meas_mrps_and_set_irq: Return the measured count and set up the * IRQ to fire if usage exceeds current * measurement by @tol percent. - * @irq: IRQ number that corresponds to this HW monitor. * @dev: device that this HW monitor can monitor. * @of_node: OF node of device that this HW monitor can monitor. * @df: Devfreq node that this HW montior is being used @@ -36,11 +33,9 @@ struct cache_hwmon { int (*start_hwmon)(struct cache_hwmon *hw, struct mrps_stats *mrps); void (*stop_hwmon)(struct cache_hwmon *hw); - bool (*is_valid_irq)(struct cache_hwmon *hw); unsigned long (*meas_mrps_and_set_irq)(struct cache_hwmon *hw, unsigned int tol, unsigned int us, struct mrps_stats *mrps); - int irq; struct device *dev; struct device_node *of_node; struct devfreq *df; @@ -48,12 +43,17 @@ struct cache_hwmon { #ifdef CONFIG_DEVFREQ_GOV_QCOM_CACHE_HWMON int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon); +int update_cache_hwmon(struct cache_hwmon *hwmon); #else static inline int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon) { return 0; } +int update_cache_hwmon(struct cache_hwmon *hwmon) +{ + return 0; +} #endif #endif /* _GOVERNOR_CACHE_HWMON_H */ From 7123adfcc050635124de2db868a8f2f9849678d1 Mon Sep 17 00:00:00 2001 From: Junjie Wu Date: Wed, 16 Jan 2019 18:31:11 -0800 Subject: [PATCH 04/20] PM / devfreq: cache_hwmon: Use array for reporting monitor stats Using an array to report monitor stats instead of hard coded variable names would allow for cleaner implementations of some cache hwmon device drivers. Change-Id: I787bdc12f10a0c8ff3c4195ce229a2987acdfce7 Signed-off-by: Junjie Wu [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_cache_hwmon.c | 22 ++++++------- drivers/devfreq/governor_cache_hwmon.h | 11 +++++-- include/trace/events/power.h | 43 ++++++++++++++++++++++++++ 3 files changed, 62 insertions(+), 14 deletions(-) diff --git a/drivers/devfreq/governor_cache_hwmon.c b/drivers/devfreq/governor_cache_hwmon.c index e670ee9e7548..25cccb41af8d 100644 --- a/drivers/devfreq/governor_cache_hwmon.c +++ b/drivers/devfreq/governor_cache_hwmon.c @@ -20,6 +20,7 @@ #include #include #include +#include #include "governor.h" #include "governor_cache_hwmon.h" @@ -119,11 +120,9 @@ static unsigned long measure_mrps_and_set_irq(struct cache_hwmon_node *node, preempt_enable(); - dev_dbg(hw->df->dev.parent, - "stat H=%3lu, M=%3lu, T=%3lu, b=%3u, f=%4lu, us=%d\n", - stat->high, stat->med, stat->high + stat->med, - stat->busy_percent, hw->df->previous_freq / 1000, us); - + trace_cache_hwmon_meas(dev_name(hw->df->dev.parent), stat->mrps[HIGH], + stat->mrps[MED], stat->mrps[LOW], + stat->busy_percent, us); return 0; } @@ -133,9 +132,9 @@ static void compute_cache_freq(struct cache_hwmon_node *node, unsigned long new_mhz; unsigned int busy; - new_mhz = mrps->high * node->cycles_per_high_req - + mrps->med * node->cycles_per_med_req - + mrps->low * node->cycles_per_low_req; + new_mhz = mrps->mrps[HIGH] * node->cycles_per_high_req + + mrps->mrps[MED] * node->cycles_per_med_req + + mrps->mrps[LOW] * node->cycles_per_low_req; busy = max(node->min_busy, mrps->busy_percent); busy = min(node->max_busy, busy); @@ -152,6 +151,7 @@ static void compute_cache_freq(struct cache_hwmon_node *node, new_mhz += node->guard_band_mhz; *freq = new_mhz * 1000; + trace_cache_hwmon_update(dev_name(node->hw->df->dev.parent), *freq); } #define TOO_SOON_US (1 * USEC_PER_MSEC) @@ -280,9 +280,9 @@ static int start_monitoring(struct devfreq *df) node->prev_ts = ktime_get(); node->prev_mhz = 0; - mrps.high = (df->previous_freq / 1000) - node->guard_band_mhz; - mrps.high /= node->cycles_per_high_req; - mrps.med = mrps.low = 0; + mrps.mrps[HIGH] = (df->previous_freq / 1000) - node->guard_band_mhz; + mrps.mrps[HIGH] /= node->cycles_per_high_req; + mrps.mrps[MED] = mrps.mrps[LOW] = 0; ret = hw->start_hwmon(hw, &mrps); if (ret < 0) { diff --git a/drivers/devfreq/governor_cache_hwmon.h b/drivers/devfreq/governor_cache_hwmon.h index f78a1d7cd028..26e7313a841f 100644 --- a/drivers/devfreq/governor_cache_hwmon.h +++ b/drivers/devfreq/governor_cache_hwmon.h @@ -9,10 +9,15 @@ #include #include +enum request_group { + HIGH, + MED, + LOW, + MAX_NUM_GROUPS, +}; + struct mrps_stats { - unsigned long high; - unsigned long med; - unsigned long low; + unsigned long mrps[MAX_NUM_GROUPS]; unsigned int busy_percent; }; diff --git a/include/trace/events/power.h b/include/trace/events/power.h index bd6a9606647b..cb523728964c 100644 --- a/include/trace/events/power.h +++ b/include/trace/events/power.h @@ -585,6 +585,49 @@ TRACE_EVENT(bw_hwmon_update, __entry->down_thres) ); +TRACE_EVENT(cache_hwmon_meas, + TP_PROTO(const char *name, unsigned long high_mrps, + unsigned long med_mrps, unsigned long low_mrps, + unsigned int busy_percent, unsigned int us), + TP_ARGS(name, high_mrps, med_mrps, low_mrps, busy_percent, us), + TP_STRUCT__entry( + __string(name, name) + __field(unsigned long, high_mrps) + __field(unsigned long, med_mrps) + __field(unsigned long, low_mrps) + __field(unsigned long, total_mrps) + __field(unsigned int, busy_percent) + __field(unsigned int, us) + ), + TP_fast_assign( + __assign_str(name, name); + __entry->high_mrps = high_mrps; + __entry->med_mrps = med_mrps; + __entry->low_mrps = low_mrps; + __entry->total_mrps = high_mrps + med_mrps + low_mrps; + __entry->busy_percent = busy_percent; + __entry->us = us; + ), + TP_printk("dev=%s H=%lu M=%lu L=%lu T=%lu busy_pct=%u period=%u", + __get_str(name), __entry->high_mrps, __entry->med_mrps, + __entry->low_mrps, __entry->total_mrps, + __entry->busy_percent, __entry->us) +); + +TRACE_EVENT(cache_hwmon_update, + TP_PROTO(const char *name, unsigned long freq_mhz), + TP_ARGS(name, freq_mhz), + TP_STRUCT__entry( + __string(name, name) + __field(unsigned long, freq) + ), + TP_fast_assign( + __assign_str(name, name); + __entry->freq = freq_mhz; + ), + TP_printk("dev=%s freq=%lu", __get_str(name), __entry->freq) +); + #endif /* _TRACE_POWER_H */ /* This part must be outside protection */ From b0dfbcd1e9aa2a56666175d7b6e2dbfb67924d1f Mon Sep 17 00:00:00 2001 From: Hanumath Prasad Date: Wed, 16 Jan 2019 18:31:57 -0800 Subject: [PATCH 05/20] PM / devfreq: bimc-bwmon: set a floor_mbps for irq threshold Interrupt storm happens when bwmon is enabled for GPU. This is mainly due to constant low traffic observed with GPU while doing memory read/write. So as the data rates read from counters are low and so the threshold set for triggering the interrupt also set as low, which in turn causes huge number of interrupts. Avoid this by setting a minimum floor for the irq threshold. Change-Id: I190fad5108bc24afcb67bec5809485380ee3662e Signed-off-by: Hanumath Prasad --- drivers/devfreq/bimc-bwmon.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index cbe35fb6e97b..68d31ca738a9 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -32,6 +32,13 @@ #define MON_MASK(m) ((m)->base + 0x298) #define MON_MATCH(m) ((m)->base + 0x29C) +/* + * Don't set the threshold lower than this value. This helps avoid + * threshold IRQs when the traffic is close to zero and even small + * changes can exceed the threshold percentage. + */ +#define FLOOR_MBPS 100UL + struct bwmon_spec { bool wrap_on_thres; bool overflow; @@ -195,7 +202,7 @@ static unsigned long meas_bw_and_set_irq(struct bw_hwmon *hw, * multiple times before the IRQ is processed. */ if (likely(!m->spec->wrap_on_thres)) - limit = mbps_to_bytes(mbps, sample_ms, tol); + limit = mbps_to_bytes(max(mbps, FLOOR_MBPS), sample_ms, tol); else limit = mbps_to_bytes(max(mbps, 400UL), sample_ms, tol); From a3b03ca5532c907339be4d253b195d3590896ed7 Mon Sep 17 00:00:00 2001 From: Junjie Wu Date: Wed, 16 Jan 2019 18:32:33 -0800 Subject: [PATCH 06/20] PM / devfreq: governor_cache_hwmon: Fix race in monitor start/stop Some cache_hwmon devices can have interrupts firing at any time. The interrupt handler would stop devfreq monitor, update its vote and restart the monitor again. This introduces a race if devfreq_supend/resume() or devfreq_interval_update() is called at the same time. Since devfreq_monitor_start() re-initializes the work, it could cause corruption while the work is being used elsewhere. Protect governor monitor start/stops with a new lock. Change-Id: I143aaaea86494b4c617df46e2c521a19b43861d5 Signed-off-by: Junjie Wu --- drivers/devfreq/governor_cache_hwmon.c | 23 ++++++++++++++++++----- 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/drivers/devfreq/governor_cache_hwmon.c b/drivers/devfreq/governor_cache_hwmon.c index 25cccb41af8d..fa4c0bce532f 100644 --- a/drivers/devfreq/governor_cache_hwmon.c +++ b/drivers/devfreq/governor_cache_hwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014, 2019 The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "cache-hwmon: " fmt @@ -46,7 +46,9 @@ static LIST_HEAD(cache_hwmon_list); static DEFINE_MUTEX(list_lock); static int use_cnt; -static DEFINE_MUTEX(state_lock); +static DEFINE_MUTEX(register_lock); + +static DEFINE_MUTEX(monitor_lock); #define show_attr(name) \ static ssize_t name##_show(struct device *dev, \ @@ -171,8 +173,12 @@ int update_cache_hwmon(struct cache_hwmon *hwmon) node = df->data; if (!node) return -ENODEV; - if (!node->mon_started) + + mutex_lock(&monitor_lock); + if (!node->mon_started) { + mutex_unlock(&monitor_lock); return -EBUSY; + } dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); @@ -202,6 +208,7 @@ int update_cache_hwmon(struct cache_hwmon *hwmon) devfreq_monitor_start(df); + mutex_unlock(&monitor_lock); return 0; } @@ -290,8 +297,10 @@ static int start_monitoring(struct devfreq *df) goto err_start; } + mutex_lock(&monitor_lock); devfreq_monitor_start(df); node->mon_started = true; + mutex_unlock(&monitor_lock); ret = sysfs_create_group(&df->dev.kobj, &dev_attr_group); if (ret < 0) { @@ -302,8 +311,10 @@ static int start_monitoring(struct devfreq *df) return 0; sysfs_fail: + mutex_lock(&monitor_lock); node->mon_started = false; devfreq_monitor_stop(df); + mutex_unlock(&monitor_lock); hw->stop_hwmon(hw); err_start: df->data = node->orig_data; @@ -318,8 +329,10 @@ static void stop_monitoring(struct devfreq *df) struct cache_hwmon *hw = node->hw; sysfs_remove_group(&df->dev.kobj, &dev_attr_group); + mutex_lock(&monitor_lock); node->mon_started = false; devfreq_monitor_stop(df); + mutex_unlock(&monitor_lock); hw->stop_hwmon(hw); df->data = node->orig_data; node->orig_data = NULL; @@ -388,13 +401,13 @@ int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon) node->hw = hwmon; node->attr_grp = &dev_attr_group; - mutex_lock(&state_lock); + mutex_lock(®ister_lock); if (!use_cnt) { ret = devfreq_add_governor(&devfreq_cache_hwmon); if (!ret) use_cnt++; } - mutex_unlock(&state_lock); + mutex_unlock(®ister_lock); if (!ret) { dev_info(dev, "Cache HWmon governor registered.\n"); From 5121f0a5c310c687c6014d90a72e5908f48efe47 Mon Sep 17 00:00:00 2001 From: Saravana Kannan Date: Wed, 16 Jan 2019 18:33:34 -0800 Subject: [PATCH 07/20] PM / devfreq: bw_hwmon: Update to low latency, high sampling rate algorithm The existing bw_hwmon governor samples the bandwidth every polling_interval milliseconds and makes decisions. Polling interval of 50ms or even 10ms gives a very low resolution picture of the DDR/bus traffic. Due to the lower resolution picture, the existing governor algorithm has to be biased aggressively towards performance to avoid any performance degradation compared to using a static mapping between bus master (CPU, GPU, etc) frequency to DDR/bus BW votes. While the existing governor uses IRQ to get early notification of traffic increase, even a 4x early notification for a 50ms polling interval still takes 12.5ms. This kind of reaction time is still too slow for some bus masters like CPU. To take care of these limitations, rewrite the governor algorithm to take multiple short samples of BW within a decision window (polling interval) and use that higher resolution picture to make much better and faster decisions. Doing so allows the governor to have the following features: - Very low reaction time - Over voting to stay ahead of increasing traffic - Historic peak tracking to limit over voting - Being power aware when doing over voting - Pattern detection and intelligent hysteresis - Detection low traffic modes and being less aggressive about BW votes. Change-Id: I69886b7fbeea0b64d10b5a1fb23fcb5f3918f0ce Signed-off-by: Saravana Kannan [aparnam@codeaurora.org: Replaced snprintf with scnprintf] Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: updated attr definitions, made to_mbps_zone() static and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 117 ++++--- drivers/devfreq/governor_bw_hwmon.c | 472 +++++++++++++++++++++++----- drivers/devfreq/governor_bw_hwmon.h | 22 +- 3 files changed, 472 insertions(+), 139 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 68d31ca738a9..05e0ce1715b5 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -32,13 +32,6 @@ #define MON_MASK(m) ((m)->base + 0x298) #define MON_MATCH(m) ((m)->base + 0x29C) -/* - * Don't set the threshold lower than this value. This helps avoid - * threshold IRQs when the traffic is close to zero and even small - * changes can exceed the threshold percentage. - */ -#define FLOOR_MBPS 100UL - struct bwmon_spec { bool wrap_on_thres; bool overflow; @@ -65,6 +58,12 @@ static void mon_enable(struct bwmon *m) static void mon_disable(struct bwmon *m) { writel_relaxed(0x0, MON_EN(m)); + /* + * mon_disable() and mon_irq_clear(), + * If latter goes first and count happen to trigger irq, we would + * have the irq line high but no one handling it. + */ + mb(); } static void mon_clear(struct bwmon *m) @@ -91,6 +90,11 @@ static void mon_irq_enable(struct bwmon *m) val = readl_relaxed(MON_INT_EN(m)); val |= 0x1; writel_relaxed(val, MON_INT_EN(m)); + /* + * make Sure irq enable complete for local and global + * to avoid race with other monitor calls + */ + mb(); } static void mon_irq_disable(struct bwmon *m) @@ -106,6 +110,11 @@ static void mon_irq_disable(struct bwmon *m) val = readl_relaxed(MON_INT_EN(m)); val &= ~0x1; writel_relaxed(val, MON_INT_EN(m)); + /* + * make Sure irq disable complete for local and global + * to avoid race with other monitor calls + */ + mb(); } static unsigned int mon_irq_status(struct bwmon *m) @@ -165,14 +174,6 @@ static unsigned long mon_get_count(struct bwmon *m) /* ********** CPUBW specific code ********** */ /* Returns MBps of read/writes for the sampling window. */ -static unsigned int bytes_to_mbps(long long bytes, unsigned int us) -{ - bytes *= USEC_PER_SEC; - do_div(bytes, us); - bytes = DIV_ROUND_UP_ULL(bytes, SZ_1M); - return bytes; -} - static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms, unsigned int tolerance_percent) { @@ -183,49 +184,61 @@ static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms, return mbps; } -static unsigned long meas_bw_and_set_irq(struct bw_hwmon *hw, - unsigned int tol, unsigned int us) +static unsigned long get_bytes_and_clear(struct bw_hwmon *hw) { - unsigned long mbps; - u32 limit; - unsigned int sample_ms = hw->df->profile->polling_ms; struct bwmon *m = to_bwmon(hw); + unsigned long count; mon_disable(m); - - mbps = mon_get_count(m); - mbps = bytes_to_mbps(mbps, us); - - /* - * If the counter wraps on thres, don't set the thres too low. - * Setting it too low runs the risk of the counter wrapping around - * multiple times before the IRQ is processed. - */ - if (likely(!m->spec->wrap_on_thres)) - limit = mbps_to_bytes(max(mbps, FLOOR_MBPS), sample_ms, tol); - else - limit = mbps_to_bytes(max(mbps, 400UL), sample_ms, tol); - - mon_set_limit(m, limit); - + count = mon_get_count(m); mon_clear(m); mon_irq_clear(m); mon_enable(m); - dev_dbg(m->dev, "MBps = %lu\n", mbps); - return mbps; + return count; +} + +static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) +{ + unsigned long count; + u32 limit; + struct bwmon *m = to_bwmon(hw); + + mon_disable(m); + count = mon_get_count(m); + mon_clear(m); + mon_irq_clear(m); + + if (likely(!m->spec->wrap_on_thres)) + limit = bytes; + else + limit = max(bytes, 500000UL); + + mon_set_limit(m, limit); + mon_enable(m); + + return count; } static irqreturn_t bwmon_intr_handler(int irq, void *dev) { struct bwmon *m = dev; - if (mon_irq_status(m)) { - update_bw_hwmon(&m->hw); - return IRQ_HANDLED; - } + if (!mon_irq_status(m)) + return IRQ_NONE; - return IRQ_NONE; + if (bw_hwmon_sample_end(&m->hw) > 0) + return IRQ_WAKE_THREAD; + + return IRQ_HANDLED; +} + +static irqreturn_t bwmon_intr_thread(int irq, void *dev) +{ + struct bwmon *m = dev; + + update_bw_hwmon(&m->hw); + return IRQ_HANDLED; } static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) @@ -234,7 +247,8 @@ static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) u32 limit; int ret; - ret = request_threaded_irq(m->irq, NULL, bwmon_intr_handler, + ret = request_threaded_irq(m->irq, bwmon_intr_handler, + bwmon_intr_thread, IRQF_ONESHOT | IRQF_SHARED, dev_name(m->dev), m); if (ret < 0) { @@ -260,9 +274,9 @@ static void stop_bw_hwmon(struct bw_hwmon *hw) { struct bwmon *m = to_bwmon(hw); + mon_irq_disable(m); free_irq(m->irq, m); mon_disable(m); - mon_irq_disable(m); mon_clear(m); mon_irq_clear(m); } @@ -271,9 +285,9 @@ static int suspend_bw_hwmon(struct bw_hwmon *hw) { struct bwmon *m = to_bwmon(hw); + mon_irq_disable(m); free_irq(m->irq, m); mon_disable(m); - mon_irq_disable(m); mon_irq_clear(m); return 0; @@ -285,9 +299,8 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) int ret; mon_clear(m); - mon_irq_enable(m); - mon_enable(m); - ret = request_threaded_irq(m->irq, NULL, bwmon_intr_handler, + ret = request_threaded_irq(m->irq, bwmon_intr_handler, + bwmon_intr_thread, IRQF_ONESHOT | IRQF_SHARED, dev_name(m->dev), m); if (ret < 0) { @@ -296,6 +309,9 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) return ret; } + mon_irq_enable(m); + mon_enable(m); + return 0; } @@ -375,7 +391,8 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->hw.stop_hwmon = &stop_bw_hwmon; m->hw.suspend_hwmon = &suspend_bw_hwmon; m->hw.resume_hwmon = &resume_bw_hwmon; - m->hw.meas_bw_and_set_irq = &meas_bw_and_set_irq; + m->hw.get_bytes_and_clear = &get_bytes_and_clear; + m->hw.set_thres = &set_thres; ret = register_bw_hwmon(dev, &m->hw); if (ret < 0) { diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 1a5a5ba9bd0d..139923a5a7cb 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2013-2015, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2013-2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bw-hwmon: " fmt @@ -17,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -24,17 +25,43 @@ #include "governor.h" #include "governor_bw_hwmon.h" +#define NUM_MBPS_ZONES 10 struct hwmon_node { - unsigned int tolerance_percent; unsigned int guard_band_mbps; unsigned int decay_rate; unsigned int io_percent; unsigned int bw_step; + unsigned int sample_ms; + unsigned int up_scale; + unsigned int up_thres; + unsigned int down_thres; + unsigned int down_count; + unsigned int hist_memory; + unsigned int hyst_trigger_count; + unsigned int hyst_length; + unsigned int idle_mbps; + unsigned int mbps_zones[NUM_MBPS_ZONES]; + unsigned long prev_ab; unsigned long *dev_ab; unsigned long resume_freq; unsigned long resume_ab; + unsigned long bytes; + unsigned long max_mbps; + unsigned long hist_max_mbps; + unsigned long hist_mem; + unsigned long hyst_peak; + unsigned long hyst_mbps; + unsigned long hyst_trig_win; + unsigned long hyst_en; + unsigned long prev_req; + unsigned long up_wake_mbps; + unsigned long down_wake_mbps; + unsigned int wake; + unsigned int down_cnt; ktime_t prev_ts; + ktime_t hist_max_ts; + bool sampled; bool mon_started; struct list_head list; void *orig_data; @@ -43,6 +70,10 @@ struct hwmon_node { struct attribute_group *attr_grp; }; +#define UP_WAKE 1 +#define DOWN_WAKE 2 +static DEFINE_SPINLOCK(irq_lock); + static LIST_HEAD(hwmon_list); static DEFINE_MUTEX(list_lock); @@ -76,55 +107,288 @@ static ssize_t name##_store(struct device *dev, \ return count; \ } +#define show_list_attr(name, n) \ +static ssize_t name##_show(struct device *dev, \ + struct device_attribute *attr, char *buf) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct hwmon_node *hw = df->data; \ + unsigned int i, cnt = 0; \ + \ + for (i = 0; i < n && hw->name[i]; i++) \ + cnt += scnprintf(buf + cnt, PAGE_SIZE, "%u ", hw->name[i]);\ + cnt += scnprintf(buf + cnt, PAGE_SIZE, "\n"); \ + return cnt; \ +} + +#define store_list_attr(name, n, _min, _max) \ +static ssize_t name##_store(struct device *dev, \ + struct device_attribute *attr, const char *buf, \ + size_t count) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct hwmon_node *hw = df->data; \ + int ret, numvals; \ + unsigned int i = 0, val; \ + char **strlist; \ + \ + strlist = argv_split(GFP_KERNEL, buf, &numvals); \ + if (!strlist) \ + return -ENOMEM; \ + numvals = min(numvals, n - 1); \ + for (i = 0; i < numvals; i++) { \ + ret = kstrtouint(strlist[i], 10, &val); \ + if (ret < 0) \ + goto out; \ + val = max(val, _min); \ + val = min(val, _max); \ + hw->name[i] = val; \ + } \ + ret = count; \ +out: \ + argv_free(strlist); \ + hw->name[i] = 0; \ + return ret; \ +} + #define MIN_MS 10U #define MAX_MS 500U -static unsigned long measure_bw_and_set_irq(struct hwmon_node *node) +/* Returns MBps of read/writes for the sampling window. */ +static unsigned int bytes_to_mbps(long long bytes, unsigned int us) { - ktime_t ts; - unsigned int us; - unsigned long mbps; - struct bw_hwmon *hw = node->hw; - - /* - * Since we are stopping the counters, we don't want this short work - * to be interrupted by other tasks and cause the measurements to be - * wrong. Not blocking interrupts to avoid affecting interrupt - * latency and since they should be short anyway because they run in - * atomic context. - */ - preempt_disable(); - - ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, node->prev_ts)); - if (!us) - us = 1; - - mbps = hw->meas_bw_and_set_irq(hw, node->tolerance_percent, us); - node->prev_ts = ts; - - preempt_enable(); - - dev_dbg(hw->df->dev.parent, "BW MBps = %6lu, period = %u\n", mbps, us); - trace_bw_hwmon_meas(dev_name(hw->df->dev.parent), - mbps, - us, - 0); + bytes *= USEC_PER_SEC; + do_div(bytes, us); + bytes = DIV_ROUND_UP_ULL(bytes, SZ_1M); + return bytes; +} +static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms) +{ + mbps *= ms; + mbps = DIV_ROUND_UP(mbps, MSEC_PER_SEC); + mbps *= SZ_1M; return mbps; } -static void compute_bw(struct hwmon_node *node, int mbps, - unsigned long *freq, unsigned long *ab) +static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) { - int new_bw; + struct devfreq *df; + struct hwmon_node *node; + ktime_t ts; + unsigned long bytes, mbps; + unsigned int us; + int wake = 0; - mbps += node->guard_band_mbps; + df = hwmon->df; + node = df->data; - if (mbps > node->prev_ab) { - new_bw = mbps; + ts = ktime_get(); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); + + bytes = hwmon->get_bytes_and_clear(hwmon); + bytes += node->bytes; + node->bytes = 0; + + mbps = bytes_to_mbps(bytes, us); + node->max_mbps = max(node->max_mbps, mbps); + + /* + * If the measured bandwidth in a micro sample is greater than the + * wake up threshold, it indicates an increase in load that's non + * trivial. So, have the governor ignore historical idle time or low + * bandwidth usage and do the bandwidth calculation based on just + * this micro sample. + */ + if (mbps > node->up_wake_mbps) { + wake = UP_WAKE; + } else if (mbps < node->down_wake_mbps) { + if (node->down_cnt) + node->down_cnt--; + if (node->down_cnt <= 0) + wake = DOWN_WAKE; + } + + node->prev_ts = ts; + node->wake = wake; + node->sampled = true; + + trace_bw_hwmon_meas(dev_name(df->dev.parent), + mbps, + us, + wake); + + return wake; +} + +int bw_hwmon_sample_end(struct bw_hwmon *hwmon) +{ + unsigned long flags; + int wake; + + spin_lock_irqsave(&irq_lock, flags); + wake = __bw_hwmon_sample_end(hwmon); + spin_unlock_irqrestore(&irq_lock, flags); + + return wake; +} + +static unsigned long to_mbps_zone(struct hwmon_node *node, unsigned long mbps) +{ + int i; + + for (i = 0; i < NUM_MBPS_ZONES && node->mbps_zones[i]; i++) + if (node->mbps_zones[i] >= mbps) + return node->mbps_zones[i]; + + return node->hw->df->max_freq; +} + +#define MIN_MBPS 500UL +#define HIST_PEAK_TOL 60 +static unsigned long get_bw_and_set_irq(struct hwmon_node *node, + unsigned long *freq, unsigned long *ab) +{ + unsigned long meas_mbps, thres, flags, req_mbps, adj_mbps; + unsigned long meas_mbps_zone; + unsigned long hist_lo_tol, hyst_lo_tol; + struct bw_hwmon *hw = node->hw; + unsigned int new_bw, io_percent = node->io_percent; + ktime_t ts; + unsigned int ms; + + spin_lock_irqsave(&irq_lock, flags); + + ts = ktime_get(); + ms = ktime_to_ms(ktime_sub(ts, node->prev_ts)); + if (!node->sampled || ms >= node->sample_ms) + __bw_hwmon_sample_end(node->hw); + node->sampled = false; + + req_mbps = meas_mbps = node->max_mbps; + node->max_mbps = 0; + + hist_lo_tol = (node->hist_max_mbps * HIST_PEAK_TOL) / 100; + /* Remember historic peak in the past hist_mem decision windows. */ + if (meas_mbps > node->hist_max_mbps || !node->hist_mem) { + /* If new max or no history */ + node->hist_max_mbps = meas_mbps; + node->hist_mem = node->hist_memory; + } else if (meas_mbps >= hist_lo_tol) { + /* + * If subsequent peaks come close (within tolerance) to but + * less than the historic peak, then reset the history start, + * but not the peak value. + */ + node->hist_mem = node->hist_memory; } else { - new_bw = mbps * node->decay_rate + /* Count down history expiration. */ + if (node->hist_mem) + node->hist_mem--; + } + + /* + * The AB value that corresponds to the lowest mbps zone greater than + * or equal to the "frequency" the current measurement will pick. + * This upper limit is useful for balancing out any prediction + * mechanisms to be power friendly. + */ + meas_mbps_zone = (meas_mbps * 100) / io_percent; + meas_mbps_zone = to_mbps_zone(node, meas_mbps_zone); + meas_mbps_zone = (meas_mbps_zone * io_percent) / 100; + meas_mbps_zone = max(meas_mbps, meas_mbps_zone); + + /* + * If this is a wake up due to BW increase, vote much higher BW than + * what we measure to stay ahead of increasing traffic and then set + * it up to vote for measured BW if we see down_count short sample + * windows of low traffic. + */ + if (node->wake == UP_WAKE) { + req_mbps += ((meas_mbps - node->prev_req) + * node->up_scale) / 100; + /* + * However if the measured load is less than the historic + * peak, but the over request is higher than the historic + * peak, then we could limit the over requesting to the + * historic peak. + */ + if (req_mbps > node->hist_max_mbps + && meas_mbps < node->hist_max_mbps) + req_mbps = node->hist_max_mbps; + + req_mbps = min(req_mbps, meas_mbps_zone); + } + + hyst_lo_tol = (node->hyst_mbps * HIST_PEAK_TOL) / 100; + if (meas_mbps > node->hyst_mbps && meas_mbps > MIN_MBPS) { + hyst_lo_tol = (meas_mbps * HIST_PEAK_TOL) / 100; + node->hyst_peak = 0; + node->hyst_trig_win = node->hyst_length; + node->hyst_mbps = meas_mbps; + } + + /* + * Check node->max_mbps to avoid double counting peaks that cause + * early termination of a window. + */ + if (meas_mbps >= hyst_lo_tol && meas_mbps > MIN_MBPS + && !node->max_mbps) { + node->hyst_peak++; + if (node->hyst_peak >= node->hyst_trigger_count + || node->hyst_en) + node->hyst_en = node->hyst_length; + } + + if (node->hyst_trig_win) + node->hyst_trig_win--; + if (node->hyst_en) + node->hyst_en--; + + if (!node->hyst_trig_win && !node->hyst_en) { + node->hyst_peak = 0; + node->hyst_mbps = 0; + } + + if (node->hyst_en) { + if (meas_mbps > node->idle_mbps) + req_mbps = max(req_mbps, node->hyst_mbps); + } + + /* Stretch the short sample window size, if the traffic is too low */ + if (meas_mbps < MIN_MBPS) { + node->up_wake_mbps = (max(MIN_MBPS, req_mbps) + * (100 + node->up_thres)) / 100; + node->down_wake_mbps = 0; + thres = mbps_to_bytes(max(MIN_MBPS, req_mbps / 2), + node->sample_ms); + } else { + /* + * Up wake vs down wake are intentionally a percentage of + * req_mbps vs meas_mbps to make sure the over requesting + * phase is handled properly. We only want to wake up and + * reduce the vote based on the measured mbps being less than + * the previous measurement that caused the "over request". + */ + node->up_wake_mbps = (req_mbps * (100 + node->up_thres)) / 100; + node->down_wake_mbps = (meas_mbps * node->down_thres) / 100; + thres = mbps_to_bytes(meas_mbps, node->sample_ms); + } + node->down_cnt = node->down_count; + + node->bytes = hw->set_thres(hw, thres); + + node->wake = 0; + node->prev_req = req_mbps; + + spin_unlock_irqrestore(&irq_lock, flags); + + adj_mbps = req_mbps + node->guard_band_mbps; + + if (adj_mbps > node->prev_ab) { + new_bw = adj_mbps; + } else { + new_bw = adj_mbps * node->decay_rate + node->prev_ab * (100 - node->decay_rate); new_bw /= 100; } @@ -132,12 +396,14 @@ static void compute_bw(struct hwmon_node *node, int mbps, node->prev_ab = new_bw; if (ab) *ab = roundup(new_bw, node->bw_step); - *freq = (new_bw * 100) / node->io_percent; + + *freq = (new_bw * 100) / io_percent; trace_bw_hwmon_update(dev_name(node->hw->df->dev.parent), new_bw, *freq, - 0, - 0); + node->up_wake_mbps, + node->down_wake_mbps); + return req_mbps; } static struct hwmon_node *find_hwmon_node(struct devfreq *df) @@ -158,13 +424,10 @@ static struct hwmon_node *find_hwmon_node(struct devfreq *df) return found; } -#define TOO_SOON_US (1 * USEC_PER_MSEC) int update_bw_hwmon(struct bw_hwmon *hwmon) { struct devfreq *df; struct hwmon_node *node; - ktime_t ts; - unsigned int us; int ret; if (!hwmon) @@ -172,7 +435,7 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) df = hwmon->df; if (!df) return -ENODEV; - node = find_hwmon_node(df); + node = df->data; if (!node) return -ENODEV; @@ -182,26 +445,12 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); - /* - * Don't recalc bandwidth if the interrupt comes right after a - * previous bandwidth calculation. This is done for two reasons: - * - * 1. Sampling the BW during a very short duration can result in a - * very inaccurate measurement due to very short bursts. - * 2. This can only happen if the limit was hit very close to the end - * of the previous sample period. Which means the current BW - * estimate is not very off and doesn't need to be readjusted. - */ - ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, node->prev_ts)); - if (us > TOO_SOON_US) { - mutex_lock(&df->lock); - ret = update_devfreq(df); - if (ret < 0) - dev_err(df->dev.parent, - "Unable to update freq on request: %d\n", ret); - mutex_unlock(&df->lock); - } + mutex_lock(&df->lock); + ret = update_devfreq(df); + if (ret < 0) + dev_err(df->dev.parent, + "Unable to update freq on request! (%d)\n", ret); + mutex_unlock(&df->lock); devfreq_monitor_start(df); @@ -380,7 +629,6 @@ static int gov_resume(struct devfreq *df) static int devfreq_bw_hwmon_get_freq(struct devfreq *df, unsigned long *freq) { - unsigned long mbps; struct hwmon_node *node = df->data; /* Suspend/resume sequence */ @@ -390,15 +638,11 @@ static int devfreq_bw_hwmon_get_freq(struct devfreq *df, return 0; } - mbps = measure_bw_and_set_irq(node); - compute_bw(node, mbps, freq, node->dev_ab); + get_bw_and_set_irq(node, freq, node->dev_ab); return 0; } -show_attr(tolerance_percent); -store_attr(tolerance_percent, 0U, 30U); -static DEVICE_ATTR_RW(tolerance_percent); show_attr(guard_band_mbps); store_attr(guard_band_mbps, 0U, 2000U); static DEVICE_ATTR_RW(guard_band_mbps); @@ -411,13 +655,52 @@ static DEVICE_ATTR_RW(io_percent); show_attr(bw_step); store_attr(bw_step, 50U, 1000U); static DEVICE_ATTR_RW(bw_step); +show_attr(sample_ms); +store_attr(sample_ms, 1U, 50U); +static DEVICE_ATTR_RW(sample_ms); +show_attr(up_scale); +store_attr(up_scale, 0U, 500U); +static DEVICE_ATTR_RW(up_scale); +show_attr(up_thres); +store_attr(up_thres, 1U, 100U); +static DEVICE_ATTR_RW(up_thres); +show_attr(down_thres); +store_attr(down_thres, 0U, 90U); +static DEVICE_ATTR_RW(down_thres); +show_attr(down_count); +store_attr(down_count, 0U, 90U); +static DEVICE_ATTR_RW(down_count); +show_attr(hist_memory); +store_attr(hist_memory, 0U, 90U); +static DEVICE_ATTR_RW(hist_memory); +show_attr(hyst_trigger_count); +store_attr(hyst_trigger_count, 0U, 90U); +static DEVICE_ATTR_RW(hyst_trigger_count); +show_attr(hyst_length); +store_attr(hyst_length, 0U, 90U); +static DEVICE_ATTR_RW(hyst_length); +show_attr(idle_mbps); +store_attr(idle_mbps, 0U, 2000U); +static DEVICE_ATTR_RW(idle_mbps); +show_list_attr(mbps_zones, NUM_MBPS_ZONES); +store_list_attr(mbps_zones, NUM_MBPS_ZONES, 0U, UINT_MAX); +static DEVICE_ATTR_RW(mbps_zones); static struct attribute *dev_attr[] = { - &dev_attr_tolerance_percent.attr, &dev_attr_guard_band_mbps.attr, &dev_attr_decay_rate.attr, &dev_attr_io_percent.attr, &dev_attr_bw_step.attr, + &dev_attr_sample_ms.attr, + &dev_attr_up_scale.attr, + &dev_attr_up_thres.attr, + &dev_attr_down_thres.attr, + &dev_attr_down_count.attr, + &dev_attr_hist_memory.attr, + &dev_attr_hyst_trigger_count.attr, + &dev_attr_hyst_length.attr, + &dev_attr_idle_mbps.attr, + &dev_attr_mbps_zones.attr, NULL, }; @@ -429,8 +712,12 @@ static struct attribute_group dev_attr_group = { static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, unsigned int event, void *data) { - int ret; + int ret = 0; unsigned int sample_ms; + struct hwmon_node *node; + struct bw_hwmon *hw; + + mutex_lock(&state_lock); switch (event) { case DEVFREQ_GOV_START: @@ -441,7 +728,7 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, ret = gov_start(df); if (ret < 0) - return ret; + goto out; dev_dbg(df->dev.parent, "Enabled dev BW HW monitor governor\n"); @@ -455,7 +742,22 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, sample_ms = *(unsigned int *)data; sample_ms = max(MIN_MS, sample_ms); sample_ms = min(MAX_MS, sample_ms); + /* + * Suspend/resume the HW monitor around the interval update + * to prevent the HW monitor IRQ from trying to change + * stop/start the delayed workqueue while the interval update + * is happening. + */ + node = df->data; + hw = node->hw; + hw->suspend_hwmon(hw); devfreq_interval_update(df, &sample_ms); + ret = hw->resume_hwmon(hw); + if (ret < 0) { + dev_err(df->dev.parent, + "Unable to resume HW monitor (%d)\n", ret); + goto out; + } break; case DEVFREQ_GOV_SUSPEND: @@ -464,7 +766,7 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, dev_err(df->dev.parent, "Unable to suspend BW HW mon governor (%d)\n", ret); - return ret; + goto out; } dev_dbg(df->dev.parent, "Suspended BW HW mon governor\n"); @@ -476,14 +778,17 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, dev_err(df->dev.parent, "Unable to resume BW HW mon governor (%d)\n", ret); - return ret; + goto out; } dev_dbg(df->dev.parent, "Resumed BW HW mon governor\n"); break; } - return 0; +out: + mutex_unlock(&state_lock); + + return ret; } static struct devfreq_governor devfreq_gov_bw_hwmon = { @@ -522,11 +827,20 @@ int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon) node->attr_grp = &dev_attr_group; } - node->tolerance_percent = 10; node->guard_band_mbps = 100; node->decay_rate = 90; node->io_percent = 16; node->bw_step = 190; + node->sample_ms = 50; + node->up_scale = 0; + node->up_thres = 10; + node->down_thres = 0; + node->down_count = 3; + node->hist_memory = 0; + node->hyst_trigger_count = 3; + node->hyst_length = 0; + node->idle_mbps = 400; + node->mbps_zones[0] = 0; node->hw = hwmon; mutex_lock(&list_lock); diff --git a/drivers/devfreq/governor_bw_hwmon.h b/drivers/devfreq/governor_bw_hwmon.h index 0737bc5fe41f..4b9c19bfd5b0 100644 --- a/drivers/devfreq/governor_bw_hwmon.h +++ b/drivers/devfreq/governor_bw_hwmon.h @@ -13,13 +13,9 @@ * struct bw_hwmon - dev BW HW monitor info * @start_hwmon: Start the HW monitoring of the dev BW * @stop_hwmon: Stop the HW monitoring of dev BW - * @is_valid_irq: Check whether the IRQ was triggered by the - * counters used to monitor dev BW. - * @meas_bw_and_set_irq: Return the measured bandwidth and set up the - * IRQ to fire if the usage exceeds current - * measurement by @tol percent. - * @irq: IRQ number that corresponds to this HW - * monitor. + * @set_thres: Set the count threshold to generate an IRQ + * @get_bytes_and_clear: Get the bytes transferred since the last call + * and reset the counter to start over. * @dev: Pointer to device that this HW monitor can * monitor. * @of_node: OF node of device that this HW monitor can @@ -42,8 +38,9 @@ struct bw_hwmon { void (*stop_hwmon)(struct bw_hwmon *hw); int (*suspend_hwmon)(struct bw_hwmon *hw); int (*resume_hwmon)(struct bw_hwmon *hw); - unsigned long (*meas_bw_and_set_irq)(struct bw_hwmon *hw, - unsigned int tol, unsigned int us); + unsigned long (*set_thres)(struct bw_hwmon *hw, + unsigned long bytes); + unsigned long (*get_bytes_and_clear)(struct bw_hwmon *hw); struct device *dev; struct device_node *of_node; struct devfreq_governor *gov; @@ -53,13 +50,18 @@ struct bw_hwmon { #ifdef CONFIG_DEVFREQ_GOV_QCOM_BW_HWMON int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon); int update_bw_hwmon(struct bw_hwmon *hwmon); +int bw_hwmon_sample_end(struct bw_hwmon *hwmon); #else static inline int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon) { return 0; } -int update_bw_hwmon(struct bw_hwmon *hwmon) +static inline int update_bw_hwmon(struct bw_hwmon *hwmon) +{ + return 0; +} +static inline int bw_hwmon_sample_end(struct bw_hwmon *hwmon) { return 0; } From f4c2ac968a695adef4f312e18f689d965b248b83 Mon Sep 17 00:00:00 2001 From: Junjie Wu Date: Wed, 16 Jan 2019 18:47:51 -0800 Subject: [PATCH 08/20] devfreq: devfreq_simple_dev: Add support for preparing device clock For certain implementation, device clock needs to be prepared before rate voting taking effect. Add support for preparing device clock during initialization. Change-Id: Ib22e83952187118342ff2546d4c79d3970a288f9 Signed-off-by: Junjie Wu [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq_simple_dev.c | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/drivers/devfreq/devfreq_simple_dev.c b/drivers/devfreq/devfreq_simple_dev.c index 076055757268..c3010cf177ba 100644 --- a/drivers/devfreq/devfreq_simple_dev.c +++ b/drivers/devfreq/devfreq_simple_dev.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "devfreq-simple-dev: " fmt @@ -147,11 +147,23 @@ static int devfreq_clock_probe(struct platform_device *pdev) if (of_property_read_string(dev->of_node, "governor", &gov_name)) gov_name = "performance"; + if (of_property_read_bool(dev->of_node, "qcom,prepare-clk")) { + ret = clk_prepare(d->clk); + if (ret < 0) + return ret; + } + d->df = devfreq_add_device(dev, p, gov_name, NULL); - if (IS_ERR(d->df)) - return PTR_ERR_OR_ZERO(d->df); + if (IS_ERR(d->df)) { + ret = PTR_ERR_OR_ZERO(d->df); + goto add_err; + } return 0; +add_err: + if (of_property_read_bool(dev->of_node, "qcom,prepare-clk")) + clk_unprepare(d->clk); + return ret; } static int devfreq_clock_remove(struct platform_device *pdev) From 4fe3e2f24216428853b043f07641659d91684e98 Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 18:56:01 -0800 Subject: [PATCH 09/20] PM / devfreq: bw_hwmon: Expose a throttle adjust tunable Newer versions of bimc-bwmon counters have the capability to fake higher byte count than what's actually transferred between a bus master and DDR if the bus master is being throttled by QoS hardware logic. Add support to set the throttle adjust field that comes with this newer version of bimc-bwmon. Change-Id: I33376c825fb11ab2e378f828b1d2ae46dd582836 Signed-off-by: Rohit Gupta [aparnam@codeaurora.org: Renamed throttle_adj functions] Signed-off-by: Rama Aparna Mallavarapu [avajid@codeaurora.org: resolved minor merge conflicts made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 40 +++++++++++++++++++++++++--- drivers/devfreq/governor_bw_hwmon.c | 41 +++++++++++++++++++++++++++++ drivers/devfreq/governor_bw_hwmon.h | 5 ++++ 3 files changed, 82 insertions(+), 4 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 05e0ce1715b5..39040fac3171 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -35,6 +35,7 @@ struct bwmon_spec { bool wrap_on_thres; bool overflow; + bool throt_adj; }; struct bwmon { @@ -45,19 +46,24 @@ struct bwmon { const struct bwmon_spec *spec; struct device *dev; struct bw_hwmon hw; + u32 throttle_adj; }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) +#define ENABLE_MASK BIT(0) +#define THROTTLE_MASK 0x1F +#define THROTTLE_SHIFT 16 + static DEFINE_SPINLOCK(glb_lock); static void mon_enable(struct bwmon *m) { - writel_relaxed(0x1, MON_EN(m)); + writel_relaxed((ENABLE_MASK | m->throttle_adj), MON_EN(m)); } static void mon_disable(struct bwmon *m) { - writel_relaxed(0x0, MON_EN(m)); + writel_relaxed(m->throttle_adj, MON_EN(m)); /* * mon_disable() and mon_irq_clear(), * If latter goes first and count happen to trigger irq, we would @@ -139,6 +145,26 @@ static void mon_irq_clear(struct bwmon *m) mb(); } +static int mon_set_throttle_adj(struct bw_hwmon *hw, uint adj) +{ + struct bwmon *m = to_bwmon(hw); + + if (adj > THROTTLE_MASK) + return -EINVAL; + + adj = (adj & THROTTLE_MASK) << THROTTLE_SHIFT; + m->throttle_adj = adj; + + return 0; +} + +static u32 mon_get_throttle_adj(struct bw_hwmon *hw) +{ + struct bwmon *m = to_bwmon(hw); + + return m->throttle_adj >> THROTTLE_SHIFT; +} + static void mon_set_limit(struct bwmon *m, u32 count) { writel_relaxed(count, MON_THRES(m)); @@ -318,13 +344,15 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) /*************************************************************************/ static const struct bwmon_spec spec[] = { - { .wrap_on_thres = true, .overflow = false }, - { .wrap_on_thres = false, .overflow = true }, + { .wrap_on_thres = true, .overflow = false, .throt_adj = false}, + { .wrap_on_thres = false, .overflow = true, .throt_adj = false}, + { .wrap_on_thres = false, .overflow = true, .throt_adj = true}, }; static const struct of_device_id bimc_bwmon_match_table[] = { { .compatible = "qcom,bimc-bwmon", .data = &spec[0] }, { .compatible = "qcom,bimc-bwmon2", .data = &spec[1] }, + { .compatible = "qcom,bimc-bwmon3", .data = &spec[2] }, {} }; @@ -393,6 +421,10 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->hw.resume_hwmon = &resume_bw_hwmon; m->hw.get_bytes_and_clear = &get_bytes_and_clear; m->hw.set_thres = &set_thres; + if (m->spec->throt_adj) { + m->hw.set_throttle_adj = &mon_set_throttle_adj; + m->hw.get_throttle_adj = &mon_get_throttle_adj; + } ret = register_bw_hwmon(dev, &m->hw); if (ret < 0) { diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 139923a5a7cb..56906c055981 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -643,6 +643,46 @@ static int devfreq_bw_hwmon_get_freq(struct devfreq *df, return 0; } +static ssize_t throttle_adj_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t count) +{ + struct devfreq *df = to_devfreq(dev); + struct hwmon_node *node = df->data; + int ret; + unsigned int val; + + if (!node->hw->set_throttle_adj) + return -EPERM; + + ret = kstrtouint(buf, 10, &val); + if (ret < 0) + return ret; + + ret = node->hw->set_throttle_adj(node->hw, val); + + if (!ret) + return count; + else + return ret; +} + +static ssize_t throttle_adj_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct devfreq *df = to_devfreq(dev); + struct hwmon_node *node = df->data; + unsigned int val; + + if (!node->hw->get_throttle_adj) + val = 0; + else + val = node->hw->get_throttle_adj(node->hw); + + return snprintf(buf, PAGE_SIZE, "%u\n", val); +} + +static DEVICE_ATTR_RW(throttle_adj); + show_attr(guard_band_mbps); store_attr(guard_band_mbps, 0U, 2000U); static DEVICE_ATTR_RW(guard_band_mbps); @@ -701,6 +741,7 @@ static struct attribute *dev_attr[] = { &dev_attr_hyst_length.attr, &dev_attr_idle_mbps.attr, &dev_attr_mbps_zones.attr, + &dev_attr_throttle_adj.attr, NULL, }; diff --git a/drivers/devfreq/governor_bw_hwmon.h b/drivers/devfreq/governor_bw_hwmon.h index 4b9c19bfd5b0..832081f41f70 100644 --- a/drivers/devfreq/governor_bw_hwmon.h +++ b/drivers/devfreq/governor_bw_hwmon.h @@ -16,6 +16,8 @@ * @set_thres: Set the count threshold to generate an IRQ * @get_bytes_and_clear: Get the bytes transferred since the last call * and reset the counter to start over. + * @set_throttle_adj: Set throttle adjust field to the given value + * @get_throttle_adj: Get the value written to throttle adjust field * @dev: Pointer to device that this HW monitor can * monitor. * @of_node: OF node of device that this HW monitor can @@ -41,6 +43,9 @@ struct bw_hwmon { unsigned long (*set_thres)(struct bw_hwmon *hw, unsigned long bytes); unsigned long (*get_bytes_and_clear)(struct bw_hwmon *hw); + int (*set_throttle_adj)(struct bw_hwmon *hw, + uint adj); + u32 (*get_throttle_adj)(struct bw_hwmon *hw); struct device *dev; struct device_node *of_node; struct devfreq_governor *gov; From a4e0d26f52906b7a83f9d49c6ea414fba0fdf89a Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 19:00:20 -0800 Subject: [PATCH 10/20] PM / devfreq: Introduce a memory-latency governor Use performance counters to detect the memory latency sensitivity of CPU workloads and vote for higher DDR frequency if required. Change-Id: Ie77a3523bc5713fc0315bd0abc3913f485a96e0e Signed-off-by: Rohit Gupta [avajid@codeaurora.org: updated attr definitions, removed exclude_idle flag and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/Kconfig | 19 ++ drivers/devfreq/Makefile | 2 + drivers/devfreq/arm-memlat-mon.c | 314 +++++++++++++++++++++++ drivers/devfreq/governor_memlat.c | 411 ++++++++++++++++++++++++++++++ drivers/devfreq/governor_memlat.h | 81 ++++++ include/trace/events/power.h | 68 +++++ 6 files changed, 895 insertions(+) create mode 100644 drivers/devfreq/arm-memlat-mon.c create mode 100644 drivers/devfreq/governor_memlat.c create mode 100644 drivers/devfreq/governor_memlat.h diff --git a/drivers/devfreq/Kconfig b/drivers/devfreq/Kconfig index 1f713deff5f4..624ad55216b0 100644 --- a/drivers/devfreq/Kconfig +++ b/drivers/devfreq/Kconfig @@ -83,6 +83,15 @@ config QCOM_BIMC_BWMON has the capability to raise an IRQ when the count exceeds a programmable limit. +config ARM_MEMLAT_MON + tristate "ARM CPU Memory Latency monitor hardware" + depends on ARCH_QCOM + help + The PMU present on these ARM cores allow for the use of counters to + monitor the memory latency characteristics of an ARM CPU workload. + This driver uses these counters to implement the APIs needed by + the mem_latency devfreq governor. + config DEVFREQ_GOV_QCOM_BW_HWMON tristate "HW monitor based governor for device BW" depends on QCOM_BIMC_BWMON @@ -102,6 +111,16 @@ config DEVFREQ_GOV_QCOM_CACHE_HWMON it can conflict with existing profiling tools. This governor is unlikely to be useful for other devices. +config DEVFREQ_GOV_MEMLAT + tristate "HW monitor based governor for device BW" + depends on ARM_MEMLAT_MON + help + HW monitor based governor for device to DDR bandwidth voting. + This governor sets the CPU BW vote based on stats obtained from memalat + monitor if it determines that a workload is memory latency bound. Since + this uses target specific counters it can conflict with existing profiling + tools. + comment "DEVFREQ Drivers" config ARM_EXYNOS_BUS_DEVFREQ diff --git a/drivers/devfreq/Makefile b/drivers/devfreq/Makefile index c1cdd38b1a5c..359b04eb21be 100644 --- a/drivers/devfreq/Makefile +++ b/drivers/devfreq/Makefile @@ -7,8 +7,10 @@ obj-$(CONFIG_DEVFREQ_GOV_POWERSAVE) += governor_powersave.o obj-$(CONFIG_DEVFREQ_GOV_USERSPACE) += governor_userspace.o obj-$(CONFIG_DEVFREQ_GOV_PASSIVE) += governor_passive.o obj-$(CONFIG_QCOM_BIMC_BWMON) += bimc-bwmon.o +obj-$(CONFIG_ARM_MEMLAT_MON) += arm-memlat-mon.o obj-$(CONFIG_DEVFREQ_GOV_QCOM_BW_HWMON) += governor_bw_hwmon.o obj-$(CONFIG_DEVFREQ_GOV_QCOM_CACHE_HWMON) += governor_cache_hwmon.o +obj-$(CONFIG_DEVFREQ_GOV_MEMLAT) += governor_memlat.o # DEVFREQ Drivers obj-$(CONFIG_ARM_EXYNOS_BUS_DEVFREQ) += exynos-bus.o diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c new file mode 100644 index 000000000000..76f13daa5565 --- /dev/null +++ b/drivers/devfreq/arm-memlat-mon.c @@ -0,0 +1,314 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "arm-memlat-mon: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "governor.h" +#include "governor_memlat.h" +#include + +enum ev_index { + INST_IDX, + CM_IDX, + CYC_IDX, + NUM_EVENTS +}; +#define INST_EV 0x08 +#define L2DM_EV 0x17 +#define CYC_EV 0x11 + +struct event_data { + struct perf_event *pevent; + unsigned long prev_count; +}; + +struct cpu_pmu_stats { + struct event_data events[NUM_EVENTS]; + ktime_t prev_ts; +}; + +struct cpu_grp_info { + cpumask_t cpus; + unsigned int event_ids[NUM_EVENTS]; + struct cpu_pmu_stats *cpustats; + struct memlat_hwmon hw; +}; + +#define to_cpustats(cpu_grp, cpu) \ + (&cpu_grp->cpustats[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_devstats(cpu_grp, cpu) \ + (&cpu_grp->hw.core_stats[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_cpu_grp(hwmon) container_of(hwmon, struct cpu_grp_info, hw) + + +static unsigned long compute_freq(struct cpu_pmu_stats *cpustats, + unsigned long cyc_cnt) +{ + ktime_t ts; + unsigned int diff; + unsigned long freq = 0; + + ts = ktime_get(); + diff = ktime_to_us(ktime_sub(ts, cpustats->prev_ts)); + if (!diff) + diff = 1; + cpustats->prev_ts = ts; + freq = cyc_cnt; + do_div(freq, diff); + + return freq; +} + +#define MAX_COUNT_LIM 0xFFFFFFFFFFFFFFFF +static inline unsigned long read_event(struct event_data *event) +{ + unsigned long ev_count; + u64 total, enabled, running; + + total = perf_event_read_value(event->pevent, &enabled, &running); + ev_count = total - event->prev_count; + event->prev_count = total; + return ev_count; +} + +static void read_perf_counters(int cpu, struct cpu_grp_info *cpu_grp) +{ + struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); + struct dev_stats *devstats = to_devstats(cpu_grp, cpu); + unsigned long cyc_cnt; + + devstats->inst_count = read_event(&cpustats->events[INST_IDX]); + devstats->mem_count = read_event(&cpustats->events[CM_IDX]); + cyc_cnt = read_event(&cpustats->events[CYC_IDX]); + devstats->freq = compute_freq(cpustats, cyc_cnt); +} + +static unsigned long get_cnt(struct memlat_hwmon *hw) +{ + int cpu; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + + for_each_cpu(cpu, &cpu_grp->cpus) + read_perf_counters(cpu, cpu_grp); + + return 0; +} + +static void delete_events(struct cpu_pmu_stats *cpustats) +{ + int i; + + for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { + cpustats->events[i].prev_count = 0; + perf_event_release_kernel(cpustats->events[i].pevent); + } +} + +static void stop_hwmon(struct memlat_hwmon *hw) +{ + int cpu; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + struct dev_stats *devstats; + + for_each_cpu(cpu, &cpu_grp->cpus) { + delete_events(to_cpustats(cpu_grp, cpu)); + + /* Clear governor data */ + devstats = to_devstats(cpu_grp, cpu); + devstats->inst_count = 0; + devstats->mem_count = 0; + devstats->freq = 0; + } +} + +static struct perf_event_attr *alloc_attr(void) +{ + struct perf_event_attr *attr; + + attr = kzalloc(sizeof(struct perf_event_attr), GFP_KERNEL); + if (!attr) + return attr; + + attr->type = PERF_TYPE_RAW; + attr->size = sizeof(struct perf_event_attr); + attr->pinned = 1; + + return attr; +} + +static int set_events(struct cpu_grp_info *cpu_grp, int cpu) +{ + struct perf_event *pevent; + struct perf_event_attr *attr; + int err, i; + struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); + + /* Allocate an attribute for event initialization */ + attr = alloc_attr(); + if (!attr) + return -ENOMEM; + + for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { + attr->config = cpu_grp->event_ids[i]; + pevent = perf_event_create_kernel_counter(attr, cpu, NULL, + NULL, NULL); + if (IS_ERR(pevent)) + goto err_out; + cpustats->events[i].pevent = pevent; + perf_event_enable(pevent); + } + + kfree(attr); + return 0; + +err_out: + err = PTR_ERR(pevent); + kfree(attr); + return err; +} + +static int start_hwmon(struct memlat_hwmon *hw) +{ + int cpu, ret = 0; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + + for_each_cpu(cpu, &cpu_grp->cpus) { + ret = set_events(cpu_grp, cpu); + if (ret < 0) { + pr_warn("Perf event init failed on CPU%d: %d\n", cpu, + ret); + break; + } + } + + return ret; +} + +static int get_mask_from_dev_handle(struct platform_device *pdev, + cpumask_t *mask) +{ + struct device *dev = &pdev->dev; + struct device_node *dev_phandle; + struct device *cpu_dev; + int cpu, i = 0; + int ret = -ENOENT; + + dev_phandle = of_parse_phandle(dev->of_node, "qcom,cpulist", i++); + while (dev_phandle) { + for_each_possible_cpu(cpu) { + cpu_dev = get_cpu_device(cpu); + if (cpu_dev && cpu_dev->of_node == dev_phandle) { + cpumask_set_cpu(cpu, mask); + ret = 0; + break; + } + } + dev_phandle = of_parse_phandle(dev->of_node, + "qcom,cpulist", i++); + } + + return ret; +} + +static int arm_memlat_mon_driver_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct memlat_hwmon *hw; + struct cpu_grp_info *cpu_grp; + int cpu, ret; + u32 event_id; + + cpu_grp = devm_kzalloc(dev, sizeof(*cpu_grp), GFP_KERNEL); + if (!cpu_grp) + return -ENOMEM; + hw = &cpu_grp->hw; + + hw->dev = dev; + hw->of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); + if (!hw->of_node) { + dev_err(dev, "Couldn't find a target device\n"); + return -ENODEV; + } + + if (get_mask_from_dev_handle(pdev, &cpu_grp->cpus)) { + dev_err(dev, "CPU list is empty\n"); + return -ENODEV; + } + + hw->num_cores = cpumask_weight(&cpu_grp->cpus); + hw->core_stats = devm_kzalloc(dev, hw->num_cores * + sizeof(*(hw->core_stats)), GFP_KERNEL); + if (!hw->core_stats) + return -ENOMEM; + + cpu_grp->cpustats = devm_kzalloc(dev, hw->num_cores * + sizeof(*(cpu_grp->cpustats)), GFP_KERNEL); + if (!cpu_grp->cpustats) + return -ENOMEM; + + cpu_grp->event_ids[CYC_IDX] = CYC_EV; + + ret = of_property_read_u32(dev->of_node, "qcom,cachemiss-ev", + &event_id); + if (ret < 0) { + dev_dbg(dev, "Cache Miss event not specified. Using def:0x%x\n", + L2DM_EV); + event_id = L2DM_EV; + } + cpu_grp->event_ids[CM_IDX] = event_id; + + ret = of_property_read_u32(dev->of_node, "qcom,inst-ev", &event_id); + if (ret < 0) { + dev_dbg(dev, "Inst event not specified. Using def:0x%x\n", + INST_EV); + event_id = INST_EV; + } + cpu_grp->event_ids[INST_IDX] = event_id; + + for_each_cpu(cpu, &cpu_grp->cpus) + to_devstats(cpu_grp, cpu)->id = cpu; + + hw->start_hwmon = &start_hwmon; + hw->stop_hwmon = &stop_hwmon; + hw->get_cnt = &get_cnt; + + ret = register_memlat(dev, hw); + if (ret < 0) { + pr_err("Mem Latency Gov registration failed: %d\n", ret); + return ret; + } + + return 0; +} + +static const struct of_device_id memlat_match_table[] = { + { .compatible = "qcom,arm-memlat-mon" }, + {} +}; + +static struct platform_driver arm_memlat_mon_driver = { + .probe = arm_memlat_mon_driver_probe, + .driver = { + .name = "arm-memlat-mon", + .of_match_table = memlat_match_table, + }, +}; + +module_platform_driver(arm_memlat_mon_driver); diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c new file mode 100644 index 000000000000..b7415b0eaeef --- /dev/null +++ b/drivers/devfreq/governor_memlat.c @@ -0,0 +1,411 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2015-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "mem_lat: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "governor.h" +#include "governor_memlat.h" + +#include + +struct memlat_node { + unsigned int ratio_ceil; + bool mon_started; + bool already_zero; + struct list_head list; + void *orig_data; + struct memlat_hwmon *hw; + struct devfreq_governor *gov; + struct attribute_group *attr_grp; +}; + +static LIST_HEAD(memlat_list); +static DEFINE_MUTEX(list_lock); + +static int use_cnt; +static DEFINE_MUTEX(state_lock); + +#define show_attr(name) \ +static ssize_t name##_show(struct device *dev, \ + struct device_attribute *attr, char *buf) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct memlat_node *hw = df->data; \ + return scnprintf(buf, PAGE_SIZE, "%u\n", hw->name); \ +} + +#define store_attr(name, _min, _max) \ +static ssize_t name##_store(struct device *dev, \ + struct device_attribute *attr, const char *buf, \ + size_t count) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct memlat_node *hw = df->data; \ + int ret; \ + unsigned int val; \ + ret = kstrtouint(buf, 10, &val); \ + if (ret < 0) \ + return ret; \ + val = max(val, _min); \ + val = min(val, _max); \ + hw->name = val; \ + return count; \ +} + +static ssize_t freq_map_show(struct device *dev, struct device_attribute *attr, + char *buf) +{ + struct devfreq *df = to_devfreq(dev); + struct memlat_node *n = df->data; + struct core_dev_map *map = n->hw->freq_map; + unsigned int cnt = 0; + + cnt += scnprintf(buf, PAGE_SIZE, "Core freq (MHz)\tDevice BW\n"); + + while (map->core_mhz && cnt < PAGE_SIZE) { + cnt += scnprintf(buf + cnt, PAGE_SIZE - cnt, "%15u\t%9u\n", + map->core_mhz, map->target_freq); + map++; + } + if (cnt < PAGE_SIZE) + cnt += scnprintf(buf + cnt, PAGE_SIZE - cnt, "\n"); + + return cnt; +} + +static DEVICE_ATTR_RO(freq_map); + +static unsigned long core_to_dev_freq(struct memlat_node *node, + unsigned long coref) +{ + struct memlat_hwmon *hw = node->hw; + struct core_dev_map *map = hw->freq_map; + unsigned long freq = 0; + + if (!map) + goto out; + + while (map->core_mhz && map->core_mhz < coref) + map++; + if (!map->core_mhz) + map--; + freq = map->target_freq; + +out: + pr_debug("freq: %lu -> dev: %lu\n", coref, freq); + return freq; +} + +static struct memlat_node *find_memlat_node(struct devfreq *df) +{ + struct memlat_node *node, *found = NULL; + + mutex_lock(&list_lock); + list_for_each_entry(node, &memlat_list, list) + if (node->hw->dev == df->dev.parent || + node->hw->of_node == df->dev.parent->of_node) { + found = node; + break; + } + mutex_unlock(&list_lock); + + return found; +} + +static int start_monitor(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + struct device *dev = df->dev.parent; + int ret; + + ret = hw->start_hwmon(hw); + + if (ret < 0) { + dev_err(dev, "Unable to start HW monitor! (%d)\n", ret); + return ret; + } + + devfreq_monitor_start(df); + + node->mon_started = true; + + return 0; +} + +static void stop_monitor(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + + node->mon_started = false; + + devfreq_monitor_stop(df); + hw->stop_hwmon(hw); +} + +static int gov_start(struct devfreq *df) +{ + int ret = 0; + struct device *dev = df->dev.parent; + struct memlat_node *node; + struct memlat_hwmon *hw; + + node = find_memlat_node(df); + if (!node) { + dev_err(dev, "Unable to find HW monitor!\n"); + return -ENODEV; + } + hw = node->hw; + + hw->df = df; + node->orig_data = df->data; + df->data = node; + + if (start_monitor(df)) + goto err_start; + + ret = sysfs_create_group(&df->dev.kobj, node->attr_grp); + if (ret < 0) + goto err_sysfs; + + return 0; + +err_sysfs: + stop_monitor(df); +err_start: + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; + return ret; +} + +static void gov_stop(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + + sysfs_remove_group(&df->dev.kobj, node->attr_grp); + stop_monitor(df); + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; +} + +static int devfreq_memlat_get_freq(struct devfreq *df, + unsigned long *freq) +{ + int i, lat_dev = 0; + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + unsigned long max_freq = 0; + unsigned int ratio; + + hw->get_cnt(hw); + + for (i = 0; i < hw->num_cores; i++) { + ratio = hw->core_stats[i].inst_count; + + if (hw->core_stats[i].mem_count) + ratio /= hw->core_stats[i].mem_count; + + if (!hw->core_stats[i].inst_count + || !hw->core_stats[i].freq) + continue; + + trace_memlat_dev_meas(dev_name(df->dev.parent), + hw->core_stats[i].id, + hw->core_stats[i].inst_count, + hw->core_stats[i].mem_count, + hw->core_stats[i].freq, ratio); + + if (ratio <= node->ratio_ceil + && hw->core_stats[i].freq > max_freq) { + lat_dev = i; + max_freq = hw->core_stats[i].freq; + } + } + + if (max_freq) + max_freq = core_to_dev_freq(node, max_freq); + + if (max_freq || !node->already_zero) { + trace_memlat_dev_update(dev_name(df->dev.parent), + hw->core_stats[lat_dev].id, + hw->core_stats[lat_dev].inst_count, + hw->core_stats[lat_dev].mem_count, + hw->core_stats[lat_dev].freq, + max_freq); + } + + node->already_zero = !max_freq; + + *freq = max_freq; + return 0; +} + +show_attr(ratio_ceil); +store_attr(ratio_ceil, 1U, 20000U); +static DEVICE_ATTR_RW(ratio_ceil); + +static struct attribute *dev_attr[] = { + &dev_attr_ratio_ceil.attr, + &dev_attr_freq_map.attr, + NULL, +}; + +static struct attribute_group dev_attr_group = { + .name = "mem_latency", + .attrs = dev_attr, +}; + +#define MIN_MS 10U +#define MAX_MS 500U +static int devfreq_memlat_ev_handler(struct devfreq *df, + unsigned int event, void *data) +{ + int ret; + unsigned int sample_ms; + + switch (event) { + case DEVFREQ_GOV_START: + sample_ms = df->profile->polling_ms; + sample_ms = max(MIN_MS, sample_ms); + sample_ms = min(MAX_MS, sample_ms); + df->profile->polling_ms = sample_ms; + + ret = gov_start(df); + if (ret < 0) + return ret; + + dev_dbg(df->dev.parent, + "Enabled Memory Latency governor\n"); + break; + + case DEVFREQ_GOV_STOP: + gov_stop(df); + dev_dbg(df->dev.parent, + "Disabled Memory Latency governor\n"); + break; + + case DEVFREQ_GOV_INTERVAL: + sample_ms = *(unsigned int *)data; + sample_ms = max(MIN_MS, sample_ms); + sample_ms = min(MAX_MS, sample_ms); + devfreq_interval_update(df, &sample_ms); + break; + } + + return 0; +} + +static struct devfreq_governor devfreq_gov_memlat = { + .name = "mem_latency", + .get_target_freq = devfreq_memlat_get_freq, + .event_handler = devfreq_memlat_ev_handler, +}; + +#define NUM_COLS 2 +static struct core_dev_map *init_core_dev_map(struct device *dev, + char *prop_name) +{ + int len, nf, i, j; + u32 data; + struct core_dev_map *tbl; + int ret; + + if (!of_find_property(dev->of_node, prop_name, &len)) + return NULL; + len /= sizeof(data); + + if (len % NUM_COLS || len == 0) + return NULL; + nf = len / NUM_COLS; + + tbl = devm_kzalloc(dev, (nf + 1) * sizeof(struct core_dev_map), + GFP_KERNEL); + if (!tbl) + return NULL; + + for (i = 0, j = 0; i < nf; i++, j += 2) { + ret = of_property_read_u32_index(dev->of_node, prop_name, j, + &data); + if (ret < 0) + return NULL; + tbl[i].core_mhz = data / 1000; + + ret = of_property_read_u32_index(dev->of_node, prop_name, j + 1, + &data); + if (ret < 0) + return NULL; + tbl[i].target_freq = data; + pr_debug("Entry%d CPU:%u, Dev:%u\n", i, tbl[i].core_mhz, + tbl[i].target_freq); + } + tbl[i].core_mhz = 0; + + return tbl; +} + +int register_memlat(struct device *dev, struct memlat_hwmon *hw) +{ + int ret = 0; + struct memlat_node *node; + + if (!hw->dev && !hw->of_node) + return -EINVAL; + + node = devm_kzalloc(dev, sizeof(*node), GFP_KERNEL); + if (!node) + return -ENOMEM; + + node->gov = &devfreq_gov_memlat; + node->attr_grp = &dev_attr_group; + + node->ratio_ceil = 10; + node->hw = hw; + + hw->freq_map = init_core_dev_map(dev, "qcom,core-dev-table"); + if (!hw->freq_map) { + dev_err(dev, "Couldn't find the core-dev freq table!\n"); + return -EINVAL; + } + + mutex_lock(&list_lock); + list_add_tail(&node->list, &memlat_list); + mutex_unlock(&list_lock); + + mutex_lock(&state_lock); + if (!use_cnt) + ret = devfreq_add_governor(&devfreq_gov_memlat); + if (!ret) + use_cnt++; + mutex_unlock(&state_lock); + + if (!ret) + dev_info(dev, "Memory Latency governor registered.\n"); + else + dev_err(dev, "Memory Latency governor registration failed!\n"); + + return ret; +} + +MODULE_DESCRIPTION("HW monitor based dev DDR bandwidth voting driver"); +MODULE_LICENSE("GPL v2"); diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h new file mode 100644 index 000000000000..6eb3dee6ae0f --- /dev/null +++ b/drivers/devfreq/governor_memlat.h @@ -0,0 +1,81 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2015-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#ifndef _GOVERNOR_MEMLAT_H +#define _GOVERNOR_MEMLAT_H + +#include +#include + +/** + * struct dev_stats - Device stats + * @inst_count: Number of instructions executed. + * @mem_count: Number of memory accesses made. + * @freq: Effective frequency of the device in the + * last interval. + */ +struct dev_stats { + int id; + unsigned long inst_count; + unsigned long mem_count; + unsigned long freq; +}; + +struct core_dev_map { + unsigned int core_mhz; + unsigned int target_freq; +}; + +/** + * struct memlat_hwmon - Memory Latency HW monitor info + * @start_hwmon: Start the HW monitoring + * @stop_hwmon: Stop the HW monitoring + * @get_cnt: Return the number of intructions executed, + * memory accesses and effective frequency + * @dev: Pointer to device that this HW monitor can + * monitor. + * @of_node: OF node of device that this HW monitor can + * monitor. + * @df: Devfreq node that this HW monitor is being + * used for. NULL when not actively in use and + * non-NULL when in use. + * @num_cores: Number of cores that are monitored by the + * hardware monitor. + * @core_stats: Array containing instruction count, memory + * accesses and effective frequency for each core. + * + * One of dev or of_node needs to be specified for a successful registration. + * + */ +struct memlat_hwmon { + int (*start_hwmon)(struct memlat_hwmon *hw); + void (*stop_hwmon)(struct memlat_hwmon *hw); + unsigned long (*get_cnt)(struct memlat_hwmon *hw); + struct device *dev; + struct device_node *of_node; + + unsigned int num_cores; + struct dev_stats *core_stats; + + struct devfreq *df; + struct core_dev_map *freq_map; +}; + +#ifdef CONFIG_DEVFREQ_GOV_MEMLAT +int register_memlat(struct device *dev, struct memlat_hwmon *hw); +int update_memlat(struct memlat_hwmon *hw); +#else +static inline int register_memlat(struct device *dev, + struct memlat_hwmon *hw) +{ + return 0; +} +static inline int update_memlat(struct memlat_hwmon *hw) +{ + return 0; +} +#endif + +#endif /* _GOVERNOR_BW_HWMON_H */ diff --git a/include/trace/events/power.h b/include/trace/events/power.h index cb523728964c..f95eb5bba10a 100644 --- a/include/trace/events/power.h +++ b/include/trace/events/power.h @@ -628,6 +628,74 @@ TRACE_EVENT(cache_hwmon_update, TP_printk("dev=%s freq=%lu", __get_str(name), __entry->freq) ); +TRACE_EVENT(memlat_dev_meas, + + TP_PROTO(const char *name, unsigned int dev_id, unsigned long inst, + unsigned long mem, unsigned long freq, unsigned int ratio), + + TP_ARGS(name, dev_id, inst, mem, freq, ratio), + + TP_STRUCT__entry( + __string(name, name) + __field(unsigned int, dev_id) + __field(unsigned long, inst) + __field(unsigned long, mem) + __field(unsigned long, freq) + __field(unsigned int, ratio) + ), + + TP_fast_assign( + __assign_str(name, name); + __entry->dev_id = dev_id; + __entry->inst = inst; + __entry->mem = mem; + __entry->freq = freq; + __entry->ratio = ratio; + ), + + TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, ratio=%u", + __get_str(name), + __entry->dev_id, + __entry->inst, + __entry->mem, + __entry->freq, + __entry->ratio) +); + +TRACE_EVENT(memlat_dev_update, + + TP_PROTO(const char *name, unsigned int dev_id, unsigned long inst, + unsigned long mem, unsigned long freq, unsigned long vote), + + TP_ARGS(name, dev_id, inst, mem, freq, vote), + + TP_STRUCT__entry( + __string(name, name) + __field(unsigned int, dev_id) + __field(unsigned long, inst) + __field(unsigned long, mem) + __field(unsigned long, freq) + __field(unsigned long, vote) + ), + + TP_fast_assign( + __assign_str(name, name); + __entry->dev_id = dev_id; + __entry->inst = inst; + __entry->mem = mem; + __entry->freq = freq; + __entry->vote = vote; + ), + + TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, vote=%lu", + __get_str(name), + __entry->dev_id, + __entry->inst, + __entry->mem, + __entry->freq, + __entry->vote) +); + #endif /* _TRACE_POWER_H */ /* This part must be outside protection */ From 7236ae6387dfbae040f2546c7260c4cf26501952 Mon Sep 17 00:00:00 2001 From: Saravana Kannan Date: Wed, 16 Jan 2019 19:02:43 -0800 Subject: [PATCH 11/20] PM / devfreq: bw_hwmon: Add HW offload support to governor Some HW monitors can do a better job of the sampling and the threshold checking than the SW implementation in the governor. Update the governor's API to add support for them. Change-Id: Id4b5593a5ed3290684ba43ebebe2466ba0b730b6 Signed-off-by: Saravana Kannan [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/governor_bw_hwmon.c | 89 +++++++++++++++++++++++------ drivers/devfreq/governor_bw_hwmon.h | 6 ++ 2 files changed, 79 insertions(+), 16 deletions(-) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 56906c055981..77a91e8f65ca 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -55,8 +55,6 @@ struct hwmon_node { unsigned long hyst_trig_win; unsigned long hyst_en; unsigned long prev_req; - unsigned long up_wake_mbps; - unsigned long down_wake_mbps; unsigned int wake; unsigned int down_cnt; ktime_t prev_ts; @@ -171,7 +169,7 @@ static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms) return mbps; } -static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) +static int __bw_hwmon_sw_sample_end(struct bw_hwmon *hwmon) { struct devfreq *df; struct hwmon_node *node; @@ -200,9 +198,9 @@ static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) * bandwidth usage and do the bandwidth calculation based on just * this micro sample. */ - if (mbps > node->up_wake_mbps) { + if (mbps > node->hw->up_wake_mbps) { wake = UP_WAKE; - } else if (mbps < node->down_wake_mbps) { + } else if (mbps < node->hw->down_wake_mbps) { if (node->down_cnt) node->down_cnt--; if (node->down_cnt <= 0) @@ -221,6 +219,50 @@ static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) return wake; } +static int __bw_hwmon_hw_sample_end(struct bw_hwmon *hwmon) +{ + struct devfreq *df; + struct hwmon_node *node; + unsigned long bytes, mbps; + int wake = 0; + + df = hwmon->df; + node = df->data; + + /* + * If this read is in response to an IRQ, the HW monitor should + * return the measurement in the micro sample that triggered the IRQ. + * Otherwise, it should return the maximum measured value in any + * micro sample since the last time we called get_bytes_and_clear() + */ + bytes = hwmon->get_bytes_and_clear(hwmon); + mbps = bytes_to_mbps(bytes, node->sample_ms * USEC_PER_MSEC); + node->max_mbps = mbps; + + if (mbps > node->hw->up_wake_mbps) + wake = UP_WAKE; + else if (mbps < node->hw->down_wake_mbps) + wake = DOWN_WAKE; + + node->wake = wake; + node->sampled = true; + + trace_bw_hwmon_meas(dev_name(df->dev.parent), + mbps, + node->sample_ms * USEC_PER_MSEC, + wake); + + return 1; +} + +static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) +{ + if (hwmon->set_hw_events) + return __bw_hwmon_hw_sample_end(hwmon); + else + return __bw_hwmon_sw_sample_end(hwmon); +} + int bw_hwmon_sample_end(struct bw_hwmon *hwmon) { unsigned long flags; @@ -255,12 +297,14 @@ static unsigned long get_bw_and_set_irq(struct hwmon_node *node, struct bw_hwmon *hw = node->hw; unsigned int new_bw, io_percent = node->io_percent; ktime_t ts; - unsigned int ms; + unsigned int ms = 0; spin_lock_irqsave(&irq_lock, flags); - ts = ktime_get(); - ms = ktime_to_ms(ktime_sub(ts, node->prev_ts)); + if (!hw->set_hw_events) { + ts = ktime_get(); + ms = ktime_to_ms(ktime_sub(ts, node->prev_ts)); + } if (!node->sampled || ms >= node->sample_ms) __bw_hwmon_sample_end(node->hw); node->sampled = false; @@ -357,9 +401,10 @@ static unsigned long get_bw_and_set_irq(struct hwmon_node *node, /* Stretch the short sample window size, if the traffic is too low */ if (meas_mbps < MIN_MBPS) { - node->up_wake_mbps = (max(MIN_MBPS, req_mbps) + hw->up_wake_mbps = (max(MIN_MBPS, req_mbps) * (100 + node->up_thres)) / 100; - node->down_wake_mbps = 0; + hw->down_wake_mbps = 0; + hw->undo_over_req_mbps = 0; thres = mbps_to_bytes(max(MIN_MBPS, req_mbps / 2), node->sample_ms); } else { @@ -370,13 +415,22 @@ static unsigned long get_bw_and_set_irq(struct hwmon_node *node, * reduce the vote based on the measured mbps being less than * the previous measurement that caused the "over request". */ - node->up_wake_mbps = (req_mbps * (100 + node->up_thres)) / 100; - node->down_wake_mbps = (meas_mbps * node->down_thres) / 100; + hw->up_wake_mbps = (req_mbps * (100 + node->up_thres)) / 100; + hw->down_wake_mbps = (meas_mbps * node->down_thres) / 100; + if (node->wake == UP_WAKE) + hw->undo_over_req_mbps = min(req_mbps, meas_mbps_zone); + else + hw->undo_over_req_mbps = 0; thres = mbps_to_bytes(meas_mbps, node->sample_ms); } - node->down_cnt = node->down_count; - node->bytes = hw->set_thres(hw, thres); + if (hw->set_hw_events) { + hw->down_cnt = node->down_count; + hw->set_hw_events(hw, node->sample_ms); + } else { + node->down_cnt = node->down_count; + node->bytes = hw->set_thres(hw, thres); + } node->wake = 0; node->prev_req = req_mbps; @@ -401,8 +455,8 @@ static unsigned long get_bw_and_set_irq(struct hwmon_node *node, trace_bw_hwmon_update(dev_name(node->hw->df->dev.parent), new_bw, *freq, - node->up_wake_mbps, - node->down_wake_mbps); + hw->up_wake_mbps, + hw->down_wake_mbps); return req_mbps; } @@ -472,6 +526,9 @@ static int start_monitor(struct devfreq *df, bool init) node->resume_freq = 0; node->resume_ab = 0; mbps = (df->previous_freq * node->io_percent) / 100; + hw->up_wake_mbps = mbps; + hw->down_wake_mbps = MIN_MBPS; + hw->undo_over_req_mbps = 0; ret = hw->start_hwmon(hw, mbps); } else { ret = hw->resume_hwmon(hw); diff --git a/drivers/devfreq/governor_bw_hwmon.h b/drivers/devfreq/governor_bw_hwmon.h index 832081f41f70..576dafbc651f 100644 --- a/drivers/devfreq/governor_bw_hwmon.h +++ b/drivers/devfreq/governor_bw_hwmon.h @@ -42,6 +42,8 @@ struct bw_hwmon { int (*resume_hwmon)(struct bw_hwmon *hw); unsigned long (*set_thres)(struct bw_hwmon *hw, unsigned long bytes); + unsigned long (*set_hw_events)(struct bw_hwmon *hw, + unsigned int sample_ms); unsigned long (*get_bytes_and_clear)(struct bw_hwmon *hw); int (*set_throttle_adj)(struct bw_hwmon *hw, uint adj); @@ -49,6 +51,10 @@ struct bw_hwmon { struct device *dev; struct device_node *of_node; struct devfreq_governor *gov; + unsigned long up_wake_mbps; + unsigned long undo_over_req_mbps; + unsigned long down_wake_mbps; + unsigned int down_cnt; struct devfreq *df; }; From f30d9fbf4c5001991972bbb09a81e4d0c6665c74 Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 20:07:30 -0800 Subject: [PATCH 12/20] PM / devfreq: bimc-bwmon: Add support for version 4 The version 4 of the BIMC BWMON hardware now has provisions for counting bytes transferred at a high sampling rate. Modify the existing driver and governor algorithm to take advantage of that. Change-Id: I5080297aef7e310d5c1a19098c177ddecb729c25 Signed-off-by: Rohit Gupta [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 267 ++++++++++++++++++++++++++++++++--- 1 file changed, 245 insertions(+), 22 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 39040fac3171..6b0bae90ba24 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2016, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bimc-bwmon: " fmt @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -32,10 +33,24 @@ #define MON_MASK(m) ((m)->base + 0x298) #define MON_MATCH(m) ((m)->base + 0x29C) +#define MON2_EN(m) ((m)->base + 0x2A0) +#define MON2_CLEAR(m) ((m)->base + 0x2A4) +#define MON2_SW(m) ((m)->base + 0x2A8) +#define MON2_THRES_HI(m) ((m)->base + 0x2AC) +#define MON2_THRES_MED(m) ((m)->base + 0x2B0) +#define MON2_THRES_LO(m) ((m)->base + 0x2B4) +#define MON2_ZONE_ACTIONS(m) ((m)->base + 0x2B8) +#define MON2_ZONE_CNT_THRES(m) ((m)->base + 0x2BC) +#define MON2_BYTE_CNT(m) ((m)->base + 0x2D0) +#define MON2_WIN_TIMER(m) ((m)->base + 0x2D4) +#define MON2_ZONE_CNT(m) ((m)->base + 0x2D8) +#define MON2_ZONE_MAX(m, zone) ((m)->base + 0x2E0 + 0x4 * zone) + struct bwmon_spec { bool wrap_on_thres; bool overflow; bool throt_adj; + bool hw_sampling; }; struct bwmon { @@ -46,24 +61,37 @@ struct bwmon { const struct bwmon_spec *spec; struct device *dev; struct bw_hwmon hw; + u32 hw_timer_hz; u32 throttle_adj; + u32 sample_size_ms; + u32 intr_status; }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) +#define has_hw_sampling(m) (m->spec->hw_sampling) #define ENABLE_MASK BIT(0) #define THROTTLE_MASK 0x1F #define THROTTLE_SHIFT 16 +#define INT_ENABLE_V1 0x1 +#define INT_STATUS_MASK 0x03 +#define INT_STATUS_MASK_HWS 0xF0 static DEFINE_SPINLOCK(glb_lock); static void mon_enable(struct bwmon *m) { - writel_relaxed((ENABLE_MASK | m->throttle_adj), MON_EN(m)); + if (has_hw_sampling(m)) + writel_relaxed((ENABLE_MASK | m->throttle_adj), MON2_EN(m)); + else + writel_relaxed((ENABLE_MASK | m->throttle_adj), MON_EN(m)); } static void mon_disable(struct bwmon *m) { - writel_relaxed(m->throttle_adj, MON_EN(m)); + if (has_hw_sampling(m)) + writel_relaxed(m->throttle_adj, MON2_EN(m)); + else + writel_relaxed(m->throttle_adj, MON_EN(m)); /* * mon_disable() and mon_irq_clear(), * If latter goes first and count happen to trigger irq, we would @@ -72,17 +100,46 @@ static void mon_disable(struct bwmon *m) mb(); } -static void mon_clear(struct bwmon *m) +#define MON_CLEAR_BIT 0x1 +#define MON_CLEAR_ALL_BIT 0x2 +static void mon_clear(struct bwmon *m, bool clear_all) { - writel_relaxed(0x1, MON_CLEAR(m)); + if (!has_hw_sampling(m)) { + writel_relaxed(MON_CLEAR_BIT, MON_CLEAR(m)); + goto out; + } + + if (clear_all) + writel_relaxed(MON_CLEAR_ALL_BIT, MON2_CLEAR(m)); + else + writel_relaxed(MON_CLEAR_BIT, MON2_CLEAR(m)); + /* * The counter clear and IRQ clear bits are not in the same 4KB * region. So, we need to make sure the counter clear is completed * before we try to clear the IRQ or do any other counter operations. */ +out: mb(); } +#define SAMPLE_WIN_LIM 0xFFFFF +static void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms) +{ + u32 rate; + + if (unlikely(sample_ms != m->sample_size_ms)) { + rate = mult_frac(sample_ms, m->hw_timer_hz, MSEC_PER_SEC); + m->sample_size_ms = sample_ms; + if (unlikely(rate > SAMPLE_WIN_LIM)) { + rate = SAMPLE_WIN_LIM; + pr_warn("Sample window %u larger than hw limit: %u\n", + rate, SAMPLE_WIN_LIM); + } + writel_relaxed(rate, MON2_SW(m)); + } +} + static void mon_irq_enable(struct bwmon *m) { u32 val; @@ -91,11 +148,11 @@ static void mon_irq_enable(struct bwmon *m) val = readl_relaxed(GLB_INT_EN(m)); val |= 1 << m->mport; writel_relaxed(val, GLB_INT_EN(m)); - spin_unlock(&glb_lock); val = readl_relaxed(MON_INT_EN(m)); - val |= 0x1; + val |= has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_ENABLE_V1; writel_relaxed(val, MON_INT_EN(m)); + spin_unlock(&glb_lock); /* * make Sure irq enable complete for local and global * to avoid race with other monitor calls @@ -111,11 +168,11 @@ static void mon_irq_disable(struct bwmon *m) val = readl_relaxed(GLB_INT_EN(m)); val &= ~(1 << m->mport); writel_relaxed(val, GLB_INT_EN(m)); - spin_unlock(&glb_lock); val = readl_relaxed(MON_INT_EN(m)); - val &= ~0x1; + val &= has_hw_sampling(m) ? ~INT_STATUS_MASK_HWS : ~INT_ENABLE_V1; writel_relaxed(val, MON_INT_EN(m)); + spin_unlock(&glb_lock); /* * make Sure irq disable complete for local and global * to avoid race with other monitor calls @@ -132,12 +189,18 @@ static unsigned int mon_irq_status(struct bwmon *m) dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, readl_relaxed(GLB_INT_STATUS(m))); + mval &= has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_STATUS_MASK; + return mval; } static void mon_irq_clear(struct bwmon *m) { - writel_relaxed(0x3, MON_INT_CLR(m)); + u32 intclr; + + intclr = has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_STATUS_MASK; + + writel_relaxed(intclr, MON_INT_CLR(m)); /* Ensure the monitor IRQ is clear before clearing GLB IRQ */ mb(); writel_relaxed(1 << m->mport, GLB_INT_CLR(m)); @@ -165,6 +228,90 @@ static u32 mon_get_throttle_adj(struct bw_hwmon *hw) return m->throttle_adj >> THROTTLE_SHIFT; } +#define ZONE1_SHIFT 8 +#define ZONE2_SHIFT 16 +#define ZONE3_SHIFT 24 +#define ZONE0_ACTION 0x01 /* Increment zone 0 count */ +#define ZONE1_ACTION 0x09 /* Increment zone 1 & clear lower zones */ +#define ZONE2_ACTION 0x25 /* Increment zone 2 & clear lower zones */ +#define ZONE3_ACTION 0x95 /* Increment zone 3 & clear lower zones */ +static u32 calc_zone_actions(void) +{ + u32 zone_actions; + + zone_actions = ZONE0_ACTION; + zone_actions |= ZONE1_ACTION << ZONE1_SHIFT; + zone_actions |= ZONE2_ACTION << ZONE2_SHIFT; + zone_actions |= ZONE3_ACTION << ZONE3_SHIFT; + + return zone_actions; +} + +#define ZONE_CNT_LIM 0xFFU +#define UP_CNT_1 1 +static u32 calc_zone_counts(struct bw_hwmon *hw) +{ + u32 zone_counts; + + zone_counts = ZONE_CNT_LIM; + zone_counts |= min(hw->down_cnt, ZONE_CNT_LIM) << ZONE1_SHIFT; + zone_counts |= ZONE_CNT_LIM << ZONE2_SHIFT; + zone_counts |= UP_CNT_1 << ZONE3_SHIFT; + + return zone_counts; +} + +static unsigned int mbps_to_mb(unsigned long mbps, unsigned int ms) +{ + mbps *= ms; + mbps = DIV_ROUND_UP(mbps, MSEC_PER_SEC); + return mbps; +} + +/* + * Define the 4 zones using HI, MED & LO thresholds: + * Zone 0: byte count < THRES_LO + * Zone 1: THRES_LO < byte count < THRES_MED + * Zone 2: THRES_MED < byte count < THRES_HI + * Zone 3: byte count > THRES_HI + */ +#define THRES_LIM 0x7FFU +static void set_zone_thres(struct bwmon *m, unsigned int sample_ms) +{ + struct bw_hwmon *hw = &(m->hw); + u32 hi, med, lo; + + hi = mbps_to_mb(hw->up_wake_mbps, sample_ms); + med = mbps_to_mb(hw->down_wake_mbps, sample_ms); + lo = 0; + + if (unlikely((hi > THRES_LIM) || (med > hi) || (lo > med))) { + pr_warn("Zone thres larger than hw limit: hi:%u med:%u lo:%u\n", + hi, med, lo); + hi = min(hi, THRES_LIM); + med = min(med, hi - 1); + lo = min(lo, med-1); + } + + writel_relaxed(hi, MON2_THRES_HI(m)); + writel_relaxed(med, MON2_THRES_MED(m)); + writel_relaxed(lo, MON2_THRES_LO(m)); + dev_dbg(m->dev, "Thres: hi:%u med:%u lo:%u\n", hi, med, lo); +} + +static void mon_set_zones(struct bwmon *m, unsigned int sample_ms) +{ + struct bw_hwmon *hw = &(m->hw); + u32 zone_cnt_thres = calc_zone_counts(hw); + + mon_set_hw_sampling_window(m, sample_ms); + set_zone_thres(m, sample_ms); + /* Set the zone count thresholds for interrupts */ + writel_relaxed(zone_cnt_thres, MON2_ZONE_CNT_THRES(m)); + + dev_dbg(m->dev, "Zone Count Thres: %0x\n", zone_cnt_thres); +} + static void mon_set_limit(struct bwmon *m, u32 count) { writel_relaxed(count, MON_THRES(m)); @@ -197,6 +344,41 @@ static unsigned long mon_get_count(struct bwmon *m) return count; } +static unsigned int get_zone(struct bwmon *m) +{ + u32 zone_counts; + u32 zone; + + zone = get_bitmask_order((m->intr_status & INT_STATUS_MASK_HWS) >> 4); + if (zone) { + zone--; + } else { + zone_counts = readl_relaxed(MON2_ZONE_CNT(m)); + if (zone_counts) { + zone = get_bitmask_order(zone_counts) - 1; + zone /= 8; + } + } + + m->intr_status = 0; + return zone; +} + +static unsigned long mon_get_zone_stats(struct bwmon *m) +{ + unsigned int zone; + unsigned long count = 0; + + zone = get_zone(m); + + count = readl_relaxed(MON2_ZONE_MAX(m, zone)) + 1; + count *= SZ_1M; + + dev_dbg(m->dev, "Zone%d Max byte count: %08lx\n", zone, count); + + return count; +} + /* ********** CPUBW specific code ********** */ /* Returns MBps of read/writes for the sampling window. */ @@ -216,8 +398,8 @@ static unsigned long get_bytes_and_clear(struct bw_hwmon *hw) unsigned long count; mon_disable(m); - count = mon_get_count(m); - mon_clear(m); + count = has_hw_sampling(m) ? mon_get_zone_stats(m) : mon_get_count(m); + mon_clear(m, false); mon_irq_clear(m); mon_enable(m); @@ -232,7 +414,7 @@ static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) mon_disable(m); count = mon_get_count(m); - mon_clear(m); + mon_clear(m, false); mon_irq_clear(m); if (likely(!m->spec->wrap_on_thres)) @@ -246,11 +428,26 @@ static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) return count; } +static unsigned long set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms) +{ + struct bwmon *m = to_bwmon(hw); + + mon_disable(m); + mon_clear(m, false); + mon_irq_clear(m); + + mon_set_zones(m, sample_ms); + mon_enable(m); + + return 0; +} + static irqreturn_t bwmon_intr_handler(int irq, void *dev) { struct bwmon *m = dev; - if (!mon_irq_status(m)) + m->intr_status = mon_irq_status(m); + if (!m->intr_status) return IRQ_NONE; if (bw_hwmon_sample_end(&m->hw) > 0) @@ -271,6 +468,7 @@ static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) { struct bwmon *m = to_bwmon(hw); u32 limit; + u32 zone_actions = calc_zone_actions(); int ret; ret = request_threaded_irq(m->irq, bwmon_intr_handler, @@ -285,10 +483,16 @@ static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) mon_disable(m); + mon_clear(m, true); limit = mbps_to_bytes(mbps, hw->df->profile->polling_ms, 0); - mon_set_limit(m, limit); + if (has_hw_sampling(m)) { + mon_set_zones(m, hw->df->profile->polling_ms); + /* Set the zone actions to increment appropriate counters */ + writel_relaxed(zone_actions, MON2_ZONE_ACTIONS(m)); + } else { + mon_set_limit(m, limit); + } - mon_clear(m); mon_irq_clear(m); mon_irq_enable(m); mon_enable(m); @@ -303,7 +507,7 @@ static void stop_bw_hwmon(struct bw_hwmon *hw) mon_irq_disable(m); free_irq(m->irq, m); mon_disable(m); - mon_clear(m); + mon_clear(m, true); mon_irq_clear(m); } @@ -324,7 +528,7 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) struct bwmon *m = to_bwmon(hw); int ret; - mon_clear(m); + mon_clear(m, false); ret = request_threaded_irq(m->irq, bwmon_intr_handler, bwmon_intr_thread, IRQF_ONESHOT | IRQF_SHARED, @@ -344,15 +548,21 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) /*************************************************************************/ static const struct bwmon_spec spec[] = { - { .wrap_on_thres = true, .overflow = false, .throt_adj = false}, - { .wrap_on_thres = false, .overflow = true, .throt_adj = false}, - { .wrap_on_thres = false, .overflow = true, .throt_adj = true}, + { .wrap_on_thres = true, .overflow = false, .throt_adj = false, + .hw_sampling = false}, + { .wrap_on_thres = false, .overflow = true, .throt_adj = false, + .hw_sampling = false}, + { .wrap_on_thres = false, .overflow = true, .throt_adj = true, + .hw_sampling = false}, + { .wrap_on_thres = false, .overflow = true, .throt_adj = true, + .hw_sampling = true}, }; static const struct of_device_id bimc_bwmon_match_table[] = { { .compatible = "qcom,bimc-bwmon", .data = &spec[0] }, { .compatible = "qcom,bimc-bwmon2", .data = &spec[1] }, { .compatible = "qcom,bimc-bwmon3", .data = &spec[2] }, + { .compatible = "qcom,bimc-bwmon4", .data = &spec[3] }, {} }; @@ -384,6 +594,17 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) } m->spec = id->data; + if (has_hw_sampling(m)) { + ret = of_property_read_u32(dev->of_node, + "qcom,hw-timer-hz", &data); + if (ret < 0) { + dev_err(dev, "HW sampling rate not specified! (%d)\n", + ret); + return ret; + } + m->hw_timer_hz = data; + } + res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "base"); if (!res) { dev_err(dev, "base not found!\n"); @@ -420,7 +641,9 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->hw.suspend_hwmon = &suspend_bw_hwmon; m->hw.resume_hwmon = &resume_bw_hwmon; m->hw.get_bytes_and_clear = &get_bytes_and_clear; - m->hw.set_thres = &set_thres; + m->hw.set_thres = &set_thres; + if (has_hw_sampling(m)) + m->hw.set_hw_events = &set_hw_events; if (m->spec->throt_adj) { m->hw.set_throttle_adj = &mon_set_throttle_adj; m->hw.get_throttle_adj = &mon_get_throttle_adj; From 9ff1db3cc7796f3182cb2a9e577220298bd6d333 Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:13:46 -0800 Subject: [PATCH 13/20] PM / devfreq: bw_hwmon: irq can be negative platform_get_irq() can return a negative number, but we assign it to an unsigned integer, which can never be negative. Change the type to int here so that we can detect irq errors. Change-Id: I997063abfe5c9966f99014a099619e6bfe7aafe7 Signed-off-by: Stephen Boyd [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 6b0bae90ba24..b93ad308057f 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2016, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bimc-bwmon: " fmt @@ -57,7 +57,7 @@ struct bwmon { void __iomem *base; void __iomem *global_base; unsigned int mport; - unsigned int irq; + int irq; const struct bwmon_spec *spec; struct device *dev; struct bw_hwmon hw; From e1b5bffc445a7d3d0025bc3a784404296f9b0bfa Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:14:46 -0800 Subject: [PATCH 14/20] PM / devfreq: bw_hwmon: Reflow some code Fix up style in this driver probe routine to use newer mechanisms like of_device_get_match_data() and to not use things like addresses of functions to assign function pointers. This makes the code more readable. Change-Id: Ib07edc0051fe9319e9a40cda7ba79646ce59b7e4 Signed-off-by: Stephen Boyd [avajid@codeaurora.org: resolved trivial merge conflict] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 63 +++++++++++++++++++++++------------- 1 file changed, 40 insertions(+), 23 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index b93ad308057f..6b9df0478b1f 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -548,14 +548,30 @@ static int resume_bw_hwmon(struct bw_hwmon *hw) /*************************************************************************/ static const struct bwmon_spec spec[] = { - { .wrap_on_thres = true, .overflow = false, .throt_adj = false, - .hw_sampling = false}, - { .wrap_on_thres = false, .overflow = true, .throt_adj = false, - .hw_sampling = false}, - { .wrap_on_thres = false, .overflow = true, .throt_adj = true, - .hw_sampling = false}, - { .wrap_on_thres = false, .overflow = true, .throt_adj = true, - .hw_sampling = true}, + [0] = { + .wrap_on_thres = true, + .overflow = false, + .throt_adj = false, + .hw_sampling = false + }, + [1] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = false, + .hw_sampling = false + }, + [2] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = true, + .hw_sampling = false + }, + [3] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = true, + .hw_sampling = true + }, }; static const struct of_device_id bimc_bwmon_match_table[] = { @@ -571,7 +587,6 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) struct device *dev = &pdev->dev; struct resource *res; struct bwmon *m; - const struct of_device_id *id; int ret; u32 data; @@ -587,16 +602,15 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) } m->mport = data; - id = of_match_device(bimc_bwmon_match_table, dev); - if (!id) { + m->spec = of_device_get_match_data(dev); + if (!m->spec) { dev_err(dev, "Unknown device type!\n"); return -ENODEV; } - m->spec = id->data; if (has_hw_sampling(m)) { - ret = of_property_read_u32(dev->of_node, - "qcom,hw-timer-hz", &data); + ret = of_property_read_u32(dev->of_node, "qcom,hw-timer-hz", + &data); if (ret < 0) { dev_err(dev, "HW sampling rate not specified! (%d)\n", ret); @@ -636,17 +650,20 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->hw.of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); if (!m->hw.of_node) return -EINVAL; - m->hw.start_hwmon = &start_bw_hwmon; - m->hw.stop_hwmon = &stop_bw_hwmon; - m->hw.suspend_hwmon = &suspend_bw_hwmon; - m->hw.resume_hwmon = &resume_bw_hwmon; - m->hw.get_bytes_and_clear = &get_bytes_and_clear; - m->hw.set_thres = &set_thres; + + m->hw.start_hwmon = start_bw_hwmon; + m->hw.stop_hwmon = stop_bw_hwmon; + m->hw.suspend_hwmon = suspend_bw_hwmon; + m->hw.resume_hwmon = resume_bw_hwmon; + m->hw.get_bytes_and_clear = get_bytes_and_clear; + m->hw.set_thres = set_thres; + if (has_hw_sampling(m)) - m->hw.set_hw_events = &set_hw_events; + m->hw.set_hw_events = set_hw_events; + if (m->spec->throt_adj) { - m->hw.set_throttle_adj = &mon_set_throttle_adj; - m->hw.get_throttle_adj = &mon_get_throttle_adj; + m->hw.set_throttle_adj = mon_set_throttle_adj; + m->hw.get_throttle_adj = mon_get_throttle_adj; } ret = register_bw_hwmon(dev, &m->hw); From 77d4c0bb9c54e9c17b9d2ea66465a3f00d425694 Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:15:44 -0800 Subject: [PATCH 15/20] PM / devfreq: bw_hwmon: Split out sw and hw paths Let's split out the sw and hw counter configuration code paths indicated by has_hw_sampling() into inline functions for the two different types of monitors. This allows us to add different types of monitors in the future with minimal changes. Change-Id: I5cf6a1fe4d84ee0958fe68601cb1e76836d10256 Signed-off-by: Stephen Boyd [avajid@codeaurora.org: minor change to check ret < 0] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 457 ++++++++++++++++++++++++----------- 1 file changed, 322 insertions(+), 135 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 6b9df0478b1f..5a4305822993 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -46,6 +46,11 @@ #define MON2_ZONE_CNT(m) ((m)->base + 0x2D8) #define MON2_ZONE_MAX(m, zone) ((m)->base + 0x2E0 + 0x4 * zone) +enum mon_reg_type { + MON1, + MON2, +}; + struct bwmon_spec { bool wrap_on_thres; bool overflow; @@ -68,7 +73,6 @@ struct bwmon { }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) -#define has_hw_sampling(m) (m->spec->hw_sampling) #define ENABLE_MASK BIT(0) #define THROTTLE_MASK 0x1F @@ -78,20 +82,29 @@ struct bwmon { #define INT_STATUS_MASK_HWS 0xF0 static DEFINE_SPINLOCK(glb_lock); -static void mon_enable(struct bwmon *m) + +static __always_inline void mon_enable(struct bwmon *m, enum mon_reg_type type) { - if (has_hw_sampling(m)) - writel_relaxed((ENABLE_MASK | m->throttle_adj), MON2_EN(m)); - else - writel_relaxed((ENABLE_MASK | m->throttle_adj), MON_EN(m)); + switch (type) { + case MON1: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON_EN(m)); + break; + case MON2: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON2_EN(m)); + break; + } } -static void mon_disable(struct bwmon *m) +static __always_inline void mon_disable(struct bwmon *m, enum mon_reg_type type) { - if (has_hw_sampling(m)) - writel_relaxed(m->throttle_adj, MON2_EN(m)); - else + switch (type) { + case MON1: writel_relaxed(m->throttle_adj, MON_EN(m)); + break; + case MON2: + writel_relaxed(m->throttle_adj, MON2_EN(m)); + break; + } /* * mon_disable() and mon_irq_clear(), * If latter goes first and count happen to trigger irq, we would @@ -102,24 +115,25 @@ static void mon_disable(struct bwmon *m) #define MON_CLEAR_BIT 0x1 #define MON_CLEAR_ALL_BIT 0x2 -static void mon_clear(struct bwmon *m, bool clear_all) +static __always_inline +void mon_clear(struct bwmon *m, bool clear_all, enum mon_reg_type type) { - if (!has_hw_sampling(m)) { + switch (type) { + case MON1: writel_relaxed(MON_CLEAR_BIT, MON_CLEAR(m)); - goto out; + break; + case MON2: + if (clear_all) + writel_relaxed(MON_CLEAR_ALL_BIT, MON2_CLEAR(m)); + else + writel_relaxed(MON_CLEAR_BIT, MON2_CLEAR(m)); + break; } - - if (clear_all) - writel_relaxed(MON_CLEAR_ALL_BIT, MON2_CLEAR(m)); - else - writel_relaxed(MON_CLEAR_BIT, MON2_CLEAR(m)); - /* * The counter clear and IRQ clear bits are not in the same 4KB * region. So, we need to make sure the counter clear is completed * before we try to clear the IRQ or do any other counter operations. */ -out: mb(); } @@ -140,74 +154,141 @@ static void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms) } } -static void mon_irq_enable(struct bwmon *m) +static void mon_glb_irq_enable(struct bwmon *m) { u32 val; - spin_lock(&glb_lock); val = readl_relaxed(GLB_INT_EN(m)); val |= 1 << m->mport; writel_relaxed(val, GLB_INT_EN(m)); - - val = readl_relaxed(MON_INT_EN(m)); - val |= has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_ENABLE_V1; - writel_relaxed(val, MON_INT_EN(m)); - spin_unlock(&glb_lock); - /* - * make Sure irq enable complete for local and global - * to avoid race with other monitor calls - */ - mb(); } -static void mon_irq_disable(struct bwmon *m) +static __always_inline +void mon_irq_enable(struct bwmon *m, enum mon_reg_type type) { u32 val; spin_lock(&glb_lock); - val = readl_relaxed(GLB_INT_EN(m)); - val &= ~(1 << m->mport); - writel_relaxed(val, GLB_INT_EN(m)); - - val = readl_relaxed(MON_INT_EN(m)); - val &= has_hw_sampling(m) ? ~INT_STATUS_MASK_HWS : ~INT_ENABLE_V1; - writel_relaxed(val, MON_INT_EN(m)); + switch (type) { + case MON1: + mon_glb_irq_enable(m); + val = readl_relaxed(MON_INT_EN(m)); + val |= INT_ENABLE_V1; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON2: + mon_glb_irq_enable(m); + val = readl_relaxed(MON_INT_EN(m)); + val |= INT_STATUS_MASK_HWS; + writel_relaxed(val, MON_INT_EN(m)); + break; + } spin_unlock(&glb_lock); /* - * make Sure irq disable complete for local and global + * make sure irq enable complete for local and global * to avoid race with other monitor calls */ mb(); } -static unsigned int mon_irq_status(struct bwmon *m) +static void mon_glb_irq_disable(struct bwmon *m) +{ + u32 val; + + val = readl_relaxed(GLB_INT_EN(m)); + val &= ~(1 << m->mport); + writel_relaxed(val, GLB_INT_EN(m)); +} + +static __always_inline +void mon_irq_disable(struct bwmon *m, enum mon_reg_type type) +{ + u32 val; + + spin_lock(&glb_lock); + + switch (type) { + case MON1: + mon_glb_irq_disable(m); + val = readl_relaxed(MON_INT_EN(m)); + val &= ~INT_ENABLE_V1; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON2: + mon_glb_irq_disable(m); + val = readl_relaxed(MON_INT_EN(m)); + val &= ~INT_STATUS_MASK_HWS; + writel_relaxed(val, MON_INT_EN(m)); + break; + } + spin_unlock(&glb_lock); + /* + * make sure irq disable complete for local and global + * to avoid race with other monitor calls + */ + mb(); +} + +static __always_inline +unsigned int mon_irq_status(struct bwmon *m, enum mon_reg_type type) { u32 mval; - mval = readl_relaxed(MON_INT_STATUS(m)); - - dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, - readl_relaxed(GLB_INT_STATUS(m))); - - mval &= has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_STATUS_MASK; + switch (type) { + case MON1: + mval = readl_relaxed(MON_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, + readl_relaxed(GLB_INT_STATUS(m))); + mval &= INT_STATUS_MASK; + break; + case MON2: + mval = readl_relaxed(MON_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, + readl_relaxed(GLB_INT_STATUS(m))); + mval &= INT_STATUS_MASK_HWS; + break; + } return mval; } -static void mon_irq_clear(struct bwmon *m) + +static void mon_glb_irq_clear(struct bwmon *m) { - u32 intclr; - - intclr = has_hw_sampling(m) ? INT_STATUS_MASK_HWS : INT_STATUS_MASK; - - writel_relaxed(intclr, MON_INT_CLR(m)); - /* Ensure the monitor IRQ is clear before clearing GLB IRQ */ + /* + * Synchronize the local interrupt clear in mon_irq_clear() + * with the global interrupt clear here. Otherwise, the CPU + * may reorder the two writes and clear the global interrupt + * before the local interrupt, causing the global interrupt + * to be retriggered by the local interrupt still being high. + */ mb(); writel_relaxed(1 << m->mport, GLB_INT_CLR(m)); - /* Ensure the GLB IRQ clear is complete */ + /* + * Similarly, because the global registers are in a different + * region than the local registers, we need to ensure any register + * writes to enable the monitor after this call are ordered with the + * clearing here so that local writes don't happen before the + * interrupt is cleared. + */ mb(); } +static __always_inline +void mon_irq_clear(struct bwmon *m, enum mon_reg_type type) +{ + switch (type) { + case MON1: + writel_relaxed(INT_STATUS_MASK, MON_INT_CLR(m)); + mon_glb_irq_clear(m); + break; + case MON2: + writel_relaxed(INT_STATUS_MASK_HWS, MON_INT_CLR(m)); + mon_glb_irq_clear(m); + break; + } +} + static int mon_set_throttle_adj(struct bw_hwmon *hw, uint adj) { struct bwmon *m = to_bwmon(hw); @@ -325,12 +406,12 @@ static u32 mon_get_limit(struct bwmon *m) #define THRES_HIT(status) (status & BIT(0)) #define OVERFLOW(status) (status & BIT(1)) -static unsigned long mon_get_count(struct bwmon *m) +static unsigned long mon_get_count1(struct bwmon *m) { unsigned long count, status; count = readl_relaxed(MON_CNT(m)); - status = mon_irq_status(m); + status = mon_irq_status(m, MON1); dev_dbg(m->dev, "Counter: %08lx\n", count); @@ -379,6 +460,23 @@ static unsigned long mon_get_zone_stats(struct bwmon *m) return count; } +static __always_inline +unsigned long mon_get_count(struct bwmon *m, enum mon_reg_type type) +{ + unsigned long count; + + switch (type) { + case MON1: + count = mon_get_count1(m); + break; + case MON2: + count = mon_get_zone_stats(m); + break; + } + + return count; +} + /* ********** CPUBW specific code ********** */ /* Returns MBps of read/writes for the sampling window. */ @@ -392,30 +490,41 @@ static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms, return mbps; } -static unsigned long get_bytes_and_clear(struct bw_hwmon *hw) +static __always_inline +unsigned long __get_bytes_and_clear(struct bw_hwmon *hw, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); unsigned long count; - mon_disable(m); - count = has_hw_sampling(m) ? mon_get_zone_stats(m) : mon_get_count(m); - mon_clear(m, false); - mon_irq_clear(m); - mon_enable(m); + mon_disable(m, type); + count = mon_get_count(m, type); + mon_clear(m, false, type); + mon_irq_clear(m, type); + mon_enable(m, type); return count; } +static unsigned long get_bytes_and_clear(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON1); +} + +static unsigned long get_bytes_and_clear2(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON2); +} + static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) { unsigned long count; u32 limit; struct bwmon *m = to_bwmon(hw); - mon_disable(m); - count = mon_get_count(m); - mon_clear(m, false); - mon_irq_clear(m); + mon_disable(m, MON1); + count = mon_get_count1(m); + mon_clear(m, false, MON1); + mon_irq_clear(m, MON1); if (likely(!m->spec->wrap_on_thres)) limit = bytes; @@ -423,7 +532,7 @@ static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) limit = max(bytes, 500000UL); mon_set_limit(m, limit); - mon_enable(m); + mon_enable(m, MON1); return count; } @@ -432,21 +541,22 @@ static unsigned long set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms) { struct bwmon *m = to_bwmon(hw); - mon_disable(m); - mon_clear(m, false); - mon_irq_clear(m); + mon_disable(m, MON2); + mon_clear(m, false, MON2); + mon_irq_clear(m, MON2); mon_set_zones(m, sample_ms); - mon_enable(m); + mon_enable(m, MON2); return 0; } -static irqreturn_t bwmon_intr_handler(int irq, void *dev) +static irqreturn_t +__bwmon_intr_handler(int irq, void *dev, enum mon_reg_type type) { struct bwmon *m = dev; - m->intr_status = mon_irq_status(m); + m->intr_status = mon_irq_status(m, type); if (!m->intr_status) return IRQ_NONE; @@ -456,6 +566,16 @@ static irqreturn_t bwmon_intr_handler(int irq, void *dev) return IRQ_HANDLED; } +static irqreturn_t bwmon_intr_handler(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON1); +} + +static irqreturn_t bwmon_intr_handler2(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON2); +} + static irqreturn_t bwmon_intr_thread(int irq, void *dev) { struct bwmon *m = dev; @@ -464,85 +584,150 @@ static irqreturn_t bwmon_intr_thread(int irq, void *dev) return IRQ_HANDLED; } -static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) +static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, + unsigned long mbps, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); - u32 limit; - u32 zone_actions = calc_zone_actions(); + u32 limit, zone_actions; int ret; + irq_handler_t handler; - ret = request_threaded_irq(m->irq, bwmon_intr_handler, - bwmon_intr_thread, + switch (type) { + case MON1: + handler = bwmon_intr_handler; + limit = mbps_to_bytes(mbps, hw->df->profile->polling_ms, 0); + break; + case MON2: + zone_actions = calc_zone_actions(); + handler = bwmon_intr_handler2; + break; + } + + ret = request_threaded_irq(m->irq, handler, bwmon_intr_thread, IRQF_ONESHOT | IRQF_SHARED, dev_name(m->dev), m); if (ret < 0) { dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", - ret); + ret); return ret; } - mon_disable(m); + mon_disable(m, type); - mon_clear(m, true); - limit = mbps_to_bytes(mbps, hw->df->profile->polling_ms, 0); - if (has_hw_sampling(m)) { + mon_clear(m, false, type); + + switch (type) { + case MON1: + mon_set_limit(m, limit); + break; + case MON2: mon_set_zones(m, hw->df->profile->polling_ms); /* Set the zone actions to increment appropriate counters */ writel_relaxed(zone_actions, MON2_ZONE_ACTIONS(m)); - } else { - mon_set_limit(m, limit); + break; } - mon_irq_clear(m); - mon_irq_enable(m); - mon_enable(m); + mon_irq_clear(m, type); + mon_irq_enable(m, type); + mon_enable(m, type); return 0; } -static void stop_bw_hwmon(struct bw_hwmon *hw) +static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON1); +} + +static int start_bw_hwmon2(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON2); +} + +static __always_inline +void __stop_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); - mon_irq_disable(m); + mon_irq_disable(m, type); free_irq(m->irq, m); - mon_disable(m); - mon_clear(m, true); - mon_irq_clear(m); + mon_disable(m, type); + mon_clear(m, true, type); + mon_irq_clear(m, type); +} + +static void stop_bw_hwmon(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON1); +} + +static void stop_bw_hwmon2(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON2); +} + +static __always_inline +int __suspend_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + + mon_irq_disable(m, type); + free_irq(m->irq, m); + mon_disable(m, type); + mon_irq_clear(m, type); + + return 0; } static int suspend_bw_hwmon(struct bw_hwmon *hw) { - struct bwmon *m = to_bwmon(hw); + return __suspend_bw_hwmon(hw, MON1); +} - mon_irq_disable(m); - free_irq(m->irq, m); - mon_disable(m); - mon_irq_clear(m); +static int suspend_bw_hwmon2(struct bw_hwmon *hw) +{ + return __suspend_bw_hwmon(hw, MON2); +} + +static int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + int ret; + irq_handler_t handler; + + switch (type) { + case MON1: + handler = bwmon_intr_handler; + break; + case MON2: + handler = bwmon_intr_handler2; + break; + } + + mon_clear(m, false, type); + ret = request_threaded_irq(m->irq, handler, bwmon_intr_thread, + IRQF_ONESHOT | IRQF_SHARED, + dev_name(m->dev), m); + if (ret < 0) { + dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", + ret); + return ret; + } + + mon_irq_enable(m, type); + mon_enable(m, type); return 0; } static int resume_bw_hwmon(struct bw_hwmon *hw) { - struct bwmon *m = to_bwmon(hw); - int ret; + return __resume_bw_hwmon(hw, MON1); +} - mon_clear(m, false); - ret = request_threaded_irq(m->irq, bwmon_intr_handler, - bwmon_intr_thread, - IRQF_ONESHOT | IRQF_SHARED, - dev_name(m->dev), m); - if (ret < 0) { - dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", - ret); - return ret; - } - - mon_irq_enable(m); - mon_enable(m); - - return 0; +static int resume_bw_hwmon2(struct bw_hwmon *hw) +{ + return __resume_bw_hwmon(hw, MON2); } /*************************************************************************/ @@ -608,17 +793,6 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return -ENODEV; } - if (has_hw_sampling(m)) { - ret = of_property_read_u32(dev->of_node, "qcom,hw-timer-hz", - &data); - if (ret < 0) { - dev_err(dev, "HW sampling rate not specified! (%d)\n", - ret); - return ret; - } - m->hw_timer_hz = data; - } - res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "base"); if (!res) { dev_err(dev, "base not found!\n"); @@ -651,15 +825,28 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) if (!m->hw.of_node) return -EINVAL; - m->hw.start_hwmon = start_bw_hwmon; - m->hw.stop_hwmon = stop_bw_hwmon; - m->hw.suspend_hwmon = suspend_bw_hwmon; - m->hw.resume_hwmon = resume_bw_hwmon; - m->hw.get_bytes_and_clear = get_bytes_and_clear; - m->hw.set_thres = set_thres; + if (m->spec->hw_sampling) { + ret = of_property_read_u32(dev->of_node, "qcom,hw-timer-hz", + &m->hw_timer_hz); + if (ret < 0) { + dev_err(dev, "HW sampling rate not specified!\n"); + return ret; + } - if (has_hw_sampling(m)) + m->hw.start_hwmon = start_bw_hwmon2; + m->hw.stop_hwmon = stop_bw_hwmon2; + m->hw.suspend_hwmon = suspend_bw_hwmon2; + m->hw.resume_hwmon = resume_bw_hwmon2; + m->hw.get_bytes_and_clear = get_bytes_and_clear2; m->hw.set_hw_events = set_hw_events; + } else { + m->hw.start_hwmon = start_bw_hwmon; + m->hw.stop_hwmon = stop_bw_hwmon; + m->hw.suspend_hwmon = suspend_bw_hwmon; + m->hw.resume_hwmon = resume_bw_hwmon; + m->hw.get_bytes_and_clear = get_bytes_and_clear; + m->hw.set_thres = set_thres; + } if (m->spec->throt_adj) { m->hw.set_throttle_adj = mon_set_throttle_adj; From 0dd9e33079120e21223288bef64af40111796fec Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:17:38 -0800 Subject: [PATCH 16/20] PM / devfreq: bw_hwmon: Add support for BWMON5 monitors Add support for the fifth type of bandwidth monitor. This monitor is similar to the other types of monitors, but the register offset is slightly different, and it doesn't have a global interrupt base. Change-Id: Ib05602832b6d4712b783ebaaa5aae9e60c00e30b Signed-off-by: Stephen Boyd [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 334 ++++++++++++++++++++++++++++------- 1 file changed, 269 insertions(+), 65 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 5a4305822993..742cbcf9d349 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -24,8 +24,12 @@ #define GLB_INT_CLR(m) ((m)->global_base + 0x108) #define GLB_INT_EN(m) ((m)->global_base + 0x10C) #define MON_INT_STATUS(m) ((m)->base + 0x100) +#define MON_INT_STATUS_MASK 0x03 +#define MON2_INT_STATUS_MASK 0xF0 +#define MON2_INT_STATUS_SHIFT 4 #define MON_INT_CLR(m) ((m)->base + 0x108) #define MON_INT_EN(m) ((m)->base + 0x10C) +#define MON_INT_ENABLE 0x1 #define MON_EN(m) ((m)->base + 0x280) #define MON_CLEAR(m) ((m)->base + 0x284) #define MON_CNT(m) ((m)->base + 0x288) @@ -46,9 +50,27 @@ #define MON2_ZONE_CNT(m) ((m)->base + 0x2D8) #define MON2_ZONE_MAX(m, zone) ((m)->base + 0x2E0 + 0x4 * zone) +#define MON3_INT_STATUS(m) ((m)->base + 0x00) +#define MON3_INT_CLR(m) ((m)->base + 0x08) +#define MON3_INT_EN(m) ((m)->base + 0x0C) +#define MON3_INT_STATUS_MASK 0x0F +#define MON3_EN(m) ((m)->base + 0x10) +#define MON3_CLEAR(m) ((m)->base + 0x14) +#define MON3_SW(m) ((m)->base + 0x20) +#define MON3_THRES_HI(m) ((m)->base + 0x24) +#define MON3_THRES_MED(m) ((m)->base + 0x28) +#define MON3_THRES_LO(m) ((m)->base + 0x2C) +#define MON3_ZONE_ACTIONS(m) ((m)->base + 0x30) +#define MON3_ZONE_CNT_THRES(m) ((m)->base + 0x34) +#define MON3_BYTE_CNT(m) ((m)->base + 0x38) +#define MON3_WIN_TIMER(m) ((m)->base + 0x3C) +#define MON3_ZONE_CNT(m) ((m)->base + 0x40) +#define MON3_ZONE_MAX(m, zone) ((m)->base + 0x44 + 0x4 * zone) + enum mon_reg_type { MON1, MON2, + MON3, }; struct bwmon_spec { @@ -56,6 +78,8 @@ struct bwmon_spec { bool overflow; bool throt_adj; bool hw_sampling; + bool has_global_base; + enum mon_reg_type reg_type; }; struct bwmon { @@ -77,9 +101,6 @@ struct bwmon { #define ENABLE_MASK BIT(0) #define THROTTLE_MASK 0x1F #define THROTTLE_SHIFT 16 -#define INT_ENABLE_V1 0x1 -#define INT_STATUS_MASK 0x03 -#define INT_STATUS_MASK_HWS 0xF0 static DEFINE_SPINLOCK(glb_lock); @@ -92,6 +113,9 @@ static __always_inline void mon_enable(struct bwmon *m, enum mon_reg_type type) case MON2: writel_relaxed(ENABLE_MASK | m->throttle_adj, MON2_EN(m)); break; + case MON3: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON3_EN(m)); + break; } } @@ -104,6 +128,9 @@ static __always_inline void mon_disable(struct bwmon *m, enum mon_reg_type type) case MON2: writel_relaxed(m->throttle_adj, MON2_EN(m)); break; + case MON3: + writel_relaxed(m->throttle_adj, MON3_EN(m)); + break; } /* * mon_disable() and mon_irq_clear(), @@ -128,6 +155,12 @@ void mon_clear(struct bwmon *m, bool clear_all, enum mon_reg_type type) else writel_relaxed(MON_CLEAR_BIT, MON2_CLEAR(m)); break; + case MON3: + if (clear_all) + writel_relaxed(MON_CLEAR_ALL_BIT, MON3_CLEAR(m)); + else + writel_relaxed(MON_CLEAR_BIT, MON3_CLEAR(m)); + break; } /* * The counter clear and IRQ clear bits are not in the same 4KB @@ -138,7 +171,9 @@ void mon_clear(struct bwmon *m, bool clear_all, enum mon_reg_type type) } #define SAMPLE_WIN_LIM 0xFFFFF -static void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms) +static __always_inline +void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) { u32 rate; @@ -150,7 +185,17 @@ static void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms) pr_warn("Sample window %u larger than hw limit: %u\n", rate, SAMPLE_WIN_LIM); } - writel_relaxed(rate, MON2_SW(m)); + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return; + case MON2: + writel_relaxed(rate, MON2_SW(m)); + break; + case MON3: + writel_relaxed(rate, MON3_SW(m)); + break; + } } } @@ -173,15 +218,20 @@ void mon_irq_enable(struct bwmon *m, enum mon_reg_type type) case MON1: mon_glb_irq_enable(m); val = readl_relaxed(MON_INT_EN(m)); - val |= INT_ENABLE_V1; + val |= MON_INT_ENABLE; writel_relaxed(val, MON_INT_EN(m)); break; case MON2: mon_glb_irq_enable(m); val = readl_relaxed(MON_INT_EN(m)); - val |= INT_STATUS_MASK_HWS; + val |= MON2_INT_STATUS_MASK; writel_relaxed(val, MON_INT_EN(m)); break; + case MON3: + val = readl_relaxed(MON3_INT_EN(m)); + val |= MON3_INT_STATUS_MASK; + writel_relaxed(val, MON3_INT_EN(m)); + break; } spin_unlock(&glb_lock); /* @@ -211,15 +261,20 @@ void mon_irq_disable(struct bwmon *m, enum mon_reg_type type) case MON1: mon_glb_irq_disable(m); val = readl_relaxed(MON_INT_EN(m)); - val &= ~INT_ENABLE_V1; + val &= ~MON_INT_ENABLE; writel_relaxed(val, MON_INT_EN(m)); break; case MON2: mon_glb_irq_disable(m); val = readl_relaxed(MON_INT_EN(m)); - val &= ~INT_STATUS_MASK_HWS; + val &= ~MON2_INT_STATUS_MASK; writel_relaxed(val, MON_INT_EN(m)); break; + case MON3: + val = readl_relaxed(MON3_INT_EN(m)); + val &= ~MON3_INT_STATUS_MASK; + writel_relaxed(val, MON3_INT_EN(m)); + break; } spin_unlock(&glb_lock); /* @@ -239,13 +294,19 @@ unsigned int mon_irq_status(struct bwmon *m, enum mon_reg_type type) mval = readl_relaxed(MON_INT_STATUS(m)); dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, readl_relaxed(GLB_INT_STATUS(m))); - mval &= INT_STATUS_MASK; + mval &= MON_INT_STATUS_MASK; break; case MON2: mval = readl_relaxed(MON_INT_STATUS(m)); dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, readl_relaxed(GLB_INT_STATUS(m))); - mval &= INT_STATUS_MASK_HWS; + mval &= MON2_INT_STATUS_MASK; + mval >>= MON2_INT_STATUS_SHIFT; + break; + case MON3: + mval = readl_relaxed(MON3_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x\n", mval); + mval &= MON3_INT_STATUS_MASK; break; } @@ -279,13 +340,16 @@ void mon_irq_clear(struct bwmon *m, enum mon_reg_type type) { switch (type) { case MON1: - writel_relaxed(INT_STATUS_MASK, MON_INT_CLR(m)); + writel_relaxed(MON_INT_STATUS_MASK, MON_INT_CLR(m)); mon_glb_irq_clear(m); break; case MON2: - writel_relaxed(INT_STATUS_MASK_HWS, MON_INT_CLR(m)); + writel_relaxed(MON2_INT_STATUS_MASK, MON_INT_CLR(m)); mon_glb_irq_clear(m); break; + case MON3: + writel_relaxed(MON3_INT_STATUS_MASK, MON3_INT_CLR(m)); + break; } } @@ -357,10 +421,13 @@ static unsigned int mbps_to_mb(unsigned long mbps, unsigned int ms) * Zone 3: byte count > THRES_HI */ #define THRES_LIM 0x7FFU -static void set_zone_thres(struct bwmon *m, unsigned int sample_ms) +static __always_inline +void set_zone_thres(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) { - struct bw_hwmon *hw = &(m->hw); + struct bw_hwmon *hw = &m->hw; u32 hi, med, lo; + u32 zone_cnt_thres = calc_zone_counts(hw); hi = mbps_to_mb(hw->up_wake_mbps, sample_ms); med = mbps_to_mb(hw->down_wake_mbps, sample_ms); @@ -374,23 +441,36 @@ static void set_zone_thres(struct bwmon *m, unsigned int sample_ms) lo = min(lo, med-1); } - writel_relaxed(hi, MON2_THRES_HI(m)); - writel_relaxed(med, MON2_THRES_MED(m)); - writel_relaxed(lo, MON2_THRES_LO(m)); + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return; + case MON2: + writel_relaxed(hi, MON2_THRES_HI(m)); + writel_relaxed(med, MON2_THRES_MED(m)); + writel_relaxed(lo, MON2_THRES_LO(m)); + /* Set the zone count thresholds for interrupts */ + writel_relaxed(zone_cnt_thres, MON2_ZONE_CNT_THRES(m)); + break; + case MON3: + writel_relaxed(hi, MON3_THRES_HI(m)); + writel_relaxed(med, MON3_THRES_MED(m)); + writel_relaxed(lo, MON3_THRES_LO(m)); + /* Set the zone count thresholds for interrupts */ + writel_relaxed(zone_cnt_thres, MON3_ZONE_CNT_THRES(m)); + break; + } + dev_dbg(m->dev, "Thres: hi:%u med:%u lo:%u\n", hi, med, lo); + dev_dbg(m->dev, "Zone Count Thres: %0x\n", zone_cnt_thres); } -static void mon_set_zones(struct bwmon *m, unsigned int sample_ms) +static __always_inline +void mon_set_zones(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) { - struct bw_hwmon *hw = &(m->hw); - u32 zone_cnt_thres = calc_zone_counts(hw); - - mon_set_hw_sampling_window(m, sample_ms); - set_zone_thres(m, sample_ms); - /* Set the zone count thresholds for interrupts */ - writel_relaxed(zone_cnt_thres, MON2_ZONE_CNT_THRES(m)); - - dev_dbg(m->dev, "Zone Count Thres: %0x\n", zone_cnt_thres); + mon_set_hw_sampling_window(m, sample_ms, type); + set_zone_thres(m, sample_ms, type); } static void mon_set_limit(struct bwmon *m, u32 count) @@ -425,16 +505,28 @@ static unsigned long mon_get_count1(struct bwmon *m) return count; } -static unsigned int get_zone(struct bwmon *m) +static __always_inline +unsigned int get_zone(struct bwmon *m, enum mon_reg_type type) { u32 zone_counts; u32 zone; - zone = get_bitmask_order((m->intr_status & INT_STATUS_MASK_HWS) >> 4); + zone = get_bitmask_order(m->intr_status); if (zone) { zone--; } else { - zone_counts = readl_relaxed(MON2_ZONE_CNT(m)); + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return 0; + case MON2: + zone_counts = readl_relaxed(MON2_ZONE_CNT(m)); + break; + case MON3: + zone_counts = readl_relaxed(MON3_ZONE_CNT(m)); + break; + } + if (zone_counts) { zone = get_bitmask_order(zone_counts) - 1; zone /= 8; @@ -445,14 +537,37 @@ static unsigned int get_zone(struct bwmon *m) return zone; } -static unsigned long mon_get_zone_stats(struct bwmon *m) +static __always_inline +unsigned long get_zone_count(struct bwmon *m, unsigned int zone, + enum mon_reg_type type) +{ + unsigned long count; + + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return 0; + case MON2: + count = readl_relaxed(MON2_ZONE_MAX(m, zone)) + 1; + break; + case MON3: + count = readl_relaxed(MON3_ZONE_MAX(m, zone)); + if (count) + count++; + break; + } + + return count; +} + +static __always_inline +unsigned long mon_get_zone_stats(struct bwmon *m, enum mon_reg_type type) { unsigned int zone; unsigned long count = 0; - zone = get_zone(m); - - count = readl_relaxed(MON2_ZONE_MAX(m, zone)) + 1; + zone = get_zone(m, type); + count = get_zone_count(m, zone, type); count *= SZ_1M; dev_dbg(m->dev, "Zone%d Max byte count: %08lx\n", zone, count); @@ -470,7 +585,8 @@ unsigned long mon_get_count(struct bwmon *m, enum mon_reg_type type) count = mon_get_count1(m); break; case MON2: - count = mon_get_zone_stats(m); + case MON3: + count = mon_get_zone_stats(m, type); break; } @@ -515,6 +631,11 @@ static unsigned long get_bytes_and_clear2(struct bw_hwmon *hw) return __get_bytes_and_clear(hw, MON2); } +static unsigned long get_bytes_and_clear3(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON3); +} + static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) { unsigned long count; @@ -537,20 +658,33 @@ static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) return count; } -static unsigned long set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms) +static unsigned long +__set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms, + enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); - mon_disable(m, MON2); - mon_clear(m, false, MON2); - mon_irq_clear(m, MON2); + mon_disable(m, type); + mon_clear(m, false, type); + mon_irq_clear(m, type); - mon_set_zones(m, sample_ms); - mon_enable(m, MON2); + mon_set_zones(m, sample_ms, type); + mon_enable(m, type); return 0; } +static unsigned long set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms) +{ + return __set_hw_events(hw, sample_ms, MON2); +} + +static unsigned long +set_hw_events3(struct bw_hwmon *hw, unsigned int sample_ms) +{ + return __set_hw_events(hw, sample_ms, MON3); +} + static irqreturn_t __bwmon_intr_handler(int irq, void *dev, enum mon_reg_type type) { @@ -576,6 +710,11 @@ static irqreturn_t bwmon_intr_handler2(int irq, void *dev) return __bwmon_intr_handler(irq, dev, MON2); } +static irqreturn_t bwmon_intr_handler3(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON3); +} + static irqreturn_t bwmon_intr_thread(int irq, void *dev) { struct bwmon *m = dev; @@ -601,6 +740,10 @@ static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, zone_actions = calc_zone_actions(); handler = bwmon_intr_handler2; break; + case MON3: + zone_actions = calc_zone_actions(); + handler = bwmon_intr_handler3; + break; } ret = request_threaded_irq(m->irq, handler, bwmon_intr_thread, @@ -621,10 +764,14 @@ static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, mon_set_limit(m, limit); break; case MON2: - mon_set_zones(m, hw->df->profile->polling_ms); + mon_set_zones(m, hw->df->profile->polling_ms, type); /* Set the zone actions to increment appropriate counters */ writel_relaxed(zone_actions, MON2_ZONE_ACTIONS(m)); break; + case MON3: + mon_set_zones(m, hw->df->profile->polling_ms, type); + /* Set the zone actions to increment appropriate counters */ + writel_relaxed(zone_actions, MON3_ZONE_ACTIONS(m)); } mon_irq_clear(m, type); @@ -644,6 +791,11 @@ static int start_bw_hwmon2(struct bw_hwmon *hw, unsigned long mbps) return __start_bw_hwmon(hw, mbps, MON2); } +static int start_bw_hwmon3(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON3); +} + static __always_inline void __stop_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { @@ -666,6 +818,11 @@ static void stop_bw_hwmon2(struct bw_hwmon *hw) return __stop_bw_hwmon(hw, MON2); } +static void stop_bw_hwmon3(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON3); +} + static __always_inline int __suspend_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { @@ -689,7 +846,13 @@ static int suspend_bw_hwmon2(struct bw_hwmon *hw) return __suspend_bw_hwmon(hw, MON2); } -static int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) +static int suspend_bw_hwmon3(struct bw_hwmon *hw) +{ + return __suspend_bw_hwmon(hw, MON3); +} + +static __always_inline +int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); int ret; @@ -702,6 +865,9 @@ static int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) case MON2: handler = bwmon_intr_handler2; break; + case MON3: + handler = bwmon_intr_handler3; + break; } mon_clear(m, false, type); @@ -730,6 +896,11 @@ static int resume_bw_hwmon2(struct bw_hwmon *hw) return __resume_bw_hwmon(hw, MON2); } +static int resume_bw_hwmon3(struct bw_hwmon *hw) +{ + return __resume_bw_hwmon(hw, MON3); +} + /*************************************************************************/ static const struct bwmon_spec spec[] = { @@ -737,25 +908,40 @@ static const struct bwmon_spec spec[] = { .wrap_on_thres = true, .overflow = false, .throt_adj = false, - .hw_sampling = false + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, }, [1] = { .wrap_on_thres = false, .overflow = true, .throt_adj = false, - .hw_sampling = false + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, }, [2] = { .wrap_on_thres = false, .overflow = true, .throt_adj = true, - .hw_sampling = false + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, }, [3] = { .wrap_on_thres = false, .overflow = true, .throt_adj = true, - .hw_sampling = true + .hw_sampling = true, + .has_global_base = true, + .reg_type = MON2, + }, + [4] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = false, + .hw_sampling = true, + .reg_type = MON3, }, }; @@ -764,6 +950,7 @@ static const struct of_device_id bimc_bwmon_match_table[] = { { .compatible = "qcom,bimc-bwmon2", .data = &spec[1] }, { .compatible = "qcom,bimc-bwmon3", .data = &spec[2] }, { .compatible = "qcom,bimc-bwmon4", .data = &spec[3] }, + { .compatible = "qcom,bimc-bwmon5", .data = &spec[4] }, {} }; @@ -780,13 +967,6 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return -ENOMEM; m->dev = dev; - ret = of_property_read_u32(dev->of_node, "qcom,mport", &data); - if (ret < 0) { - dev_err(dev, "mport not found! (%d)\n", ret); - return ret; - } - m->mport = data; - m->spec = of_device_get_match_data(dev); if (!m->spec) { dev_err(dev, "Unknown device type!\n"); @@ -804,15 +984,26 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return -ENOMEM; } - res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "global_base"); - if (!res) { - dev_err(dev, "global_base not found!\n"); - return -EINVAL; - } - m->global_base = devm_ioremap(dev, res->start, resource_size(res)); - if (!m->global_base) { - dev_err(dev, "Unable map global_base!\n"); - return -ENOMEM; + if (m->spec->has_global_base) { + res = platform_get_resource_byname(pdev, IORESOURCE_MEM, + "global_base"); + if (!res) { + dev_err(dev, "global_base not found!\n"); + return -EINVAL; + } + m->global_base = devm_ioremap(dev, res->start, + resource_size(res)); + if (!m->global_base) { + dev_err(dev, "Unable map global_base!\n"); + return -ENOMEM; + } + + ret = of_property_read_u32(dev->of_node, "qcom,mport", &data); + if (ret < 0) { + dev_err(dev, "mport not found! (%d)\n", ret); + return ret; + } + m->mport = data; } m->irq = platform_get_irq(pdev, 0); @@ -832,20 +1023,33 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) dev_err(dev, "HW sampling rate not specified!\n"); return ret; } + } + switch (m->spec->reg_type) { + case MON3: + m->hw.start_hwmon = start_bw_hwmon3; + m->hw.stop_hwmon = stop_bw_hwmon3; + m->hw.suspend_hwmon = suspend_bw_hwmon3; + m->hw.resume_hwmon = resume_bw_hwmon3; + m->hw.get_bytes_and_clear = get_bytes_and_clear3; + m->hw.set_hw_events = set_hw_events3; + break; + case MON2: m->hw.start_hwmon = start_bw_hwmon2; m->hw.stop_hwmon = stop_bw_hwmon2; m->hw.suspend_hwmon = suspend_bw_hwmon2; m->hw.resume_hwmon = resume_bw_hwmon2; m->hw.get_bytes_and_clear = get_bytes_and_clear2; m->hw.set_hw_events = set_hw_events; - } else { + break; + case MON1: m->hw.start_hwmon = start_bw_hwmon; m->hw.stop_hwmon = stop_bw_hwmon; m->hw.suspend_hwmon = suspend_bw_hwmon; m->hw.resume_hwmon = resume_bw_hwmon; m->hw.get_bytes_and_clear = get_bytes_and_clear; m->hw.set_thres = set_thres; + break; } if (m->spec->throt_adj) { From d6ee9c7aa62c103524ce9d15a89fe7cc1343a2c5 Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:19:21 -0800 Subject: [PATCH 17/20] PM / devfreq: bw_hwmon: Add support for configuring byte MID match Add support to configure the byte MID match value from DT. Change-Id: I5848aef98f15c6de9fe0fae0a1188a17810a5ef5 Signed-off-by: Stephen Boyd [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 742cbcf9d349..34ec734f3317 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -56,6 +56,8 @@ #define MON3_INT_STATUS_MASK 0x0F #define MON3_EN(m) ((m)->base + 0x10) #define MON3_CLEAR(m) ((m)->base + 0x14) +#define MON3_MASK(m) ((m)->base + 0x18) +#define MON3_MATCH(m) ((m)->base + 0x1C) #define MON3_SW(m) ((m)->base + 0x20) #define MON3_THRES_HI(m) ((m)->base + 0x24) #define MON3_THRES_MED(m) ((m)->base + 0x28) @@ -94,6 +96,8 @@ struct bwmon { u32 throttle_adj; u32 sample_size_ms; u32 intr_status; + u32 byte_mask; + u32 byte_match; }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) @@ -723,6 +727,25 @@ static irqreturn_t bwmon_intr_thread(int irq, void *dev) return IRQ_HANDLED; } +static __always_inline +void mon_set_byte_count_filter(struct bwmon *m, enum mon_reg_type type) +{ + if (!m->byte_mask) + return; + + switch (type) { + case MON1: + case MON2: + writel_relaxed(m->byte_mask, MON_MASK(m)); + writel_relaxed(m->byte_match, MON_MATCH(m)); + break; + case MON3: + writel_relaxed(m->byte_mask, MON3_MASK(m)); + writel_relaxed(m->byte_match, MON3_MATCH(m)); + break; + } +} + static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps, enum mon_reg_type type) { @@ -774,6 +797,7 @@ static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, writel_relaxed(zone_actions, MON3_ZONE_ACTIONS(m)); } + mon_set_byte_count_filter(m, type); mon_irq_clear(m, type); mon_irq_enable(m, type); mon_enable(m, type); @@ -1052,6 +1076,11 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) break; } + of_property_read_u32(dev->of_node, "qcom,byte-mid-match", + &m->byte_match); + of_property_read_u32(dev->of_node, "qcom,byte-mid-mask", + &m->byte_mask); + if (m->spec->throt_adj) { m->hw.set_throttle_adj = mon_set_throttle_adj; m->hw.get_throttle_adj = mon_get_throttle_adj; From b89dc2a89136d72c5fa14ed13f583367cfecdd14 Mon Sep 17 00:00:00 2001 From: Stephen Boyd Date: Wed, 16 Jan 2019 20:20:33 -0800 Subject: [PATCH 18/20] PM / devfreq: bw_hwmon: Add support for specifying count factor Not all bandwidth monitors count in units of 1MB. Add a DT property that indicates the number of bytes the monitor counts in so we can properly scale the count we read from the hardware to reflect the traffic. By default, we will assume 1MB units if the property doesn't exist. Change-Id: I1b581b2fffda04ba151df6e87221628368ff5b17 Signed-off-by: Stephen Boyd [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/bimc-bwmon.c | 39 ++++++++++++++++++++++++++---------- 1 file changed, 28 insertions(+), 11 deletions(-) diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 34ec734f3317..0937cdf6c822 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -18,6 +18,8 @@ #include #include #include +#include +#include #include "governor_bw_hwmon.h" #define GLB_INT_STATUS(m) ((m)->global_base + 0x100) @@ -96,6 +98,8 @@ struct bwmon { u32 throttle_adj; u32 sample_size_ms; u32 intr_status; + u8 count_shift; + u32 thres_lim; u32 byte_mask; u32 byte_match; }; @@ -410,11 +414,18 @@ static u32 calc_zone_counts(struct bw_hwmon *hw) return zone_counts; } -static unsigned int mbps_to_mb(unsigned long mbps, unsigned int ms) +#define MB_SHIFT 20 + +static u32 mbps_to_count(unsigned long mbps, unsigned int ms, u8 shift) { mbps *= ms; - mbps = DIV_ROUND_UP(mbps, MSEC_PER_SEC); - return mbps; + + if (shift > MB_SHIFT) + mbps >>= shift - MB_SHIFT; + else + mbps <<= MB_SHIFT - shift; + + return DIV_ROUND_UP(mbps, MSEC_PER_SEC); } /* @@ -422,9 +433,10 @@ static unsigned int mbps_to_mb(unsigned long mbps, unsigned int ms) * Zone 0: byte count < THRES_LO * Zone 1: THRES_LO < byte count < THRES_MED * Zone 2: THRES_MED < byte count < THRES_HI - * Zone 3: byte count > THRES_HI + * Zone 3: THRES_LIM > byte count > THRES_HI */ -#define THRES_LIM 0x7FFU +#define THRES_LIM(shift) (0xFFFFFFFF >> shift) + static __always_inline void set_zone_thres(struct bwmon *m, unsigned int sample_ms, enum mon_reg_type type) @@ -433,14 +445,14 @@ void set_zone_thres(struct bwmon *m, unsigned int sample_ms, u32 hi, med, lo; u32 zone_cnt_thres = calc_zone_counts(hw); - hi = mbps_to_mb(hw->up_wake_mbps, sample_ms); - med = mbps_to_mb(hw->down_wake_mbps, sample_ms); + hi = mbps_to_count(hw->up_wake_mbps, sample_ms, m->count_shift); + med = mbps_to_count(hw->down_wake_mbps, sample_ms, m->count_shift); lo = 0; - if (unlikely((hi > THRES_LIM) || (med > hi) || (lo > med))) { + if (unlikely((hi > m->thres_lim) || (med > hi) || (lo > med))) { pr_warn("Zone thres larger than hw limit: hi:%u med:%u lo:%u\n", hi, med, lo); - hi = min(hi, THRES_LIM); + hi = min(hi, m->thres_lim); med = min(med, hi - 1); lo = min(lo, med-1); } @@ -572,7 +584,7 @@ unsigned long mon_get_zone_stats(struct bwmon *m, enum mon_reg_type type) zone = get_zone(m, type); count = get_zone_count(m, zone, type); - count *= SZ_1M; + count <<= m->count_shift; dev_dbg(m->dev, "Zone%d Max byte count: %08lx\n", zone, count); @@ -984,7 +996,7 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) struct resource *res; struct bwmon *m; int ret; - u32 data; + u32 data, count_unit; m = devm_kzalloc(dev, sizeof(*m), GFP_KERNEL); if (!m) @@ -1049,6 +1061,11 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) } } + if (of_property_read_u32(dev->of_node, "qcom,count-unit", &count_unit)) + count_unit = SZ_1M; + m->count_shift = order_base_2(count_unit); + m->thres_lim = THRES_LIM(m->count_shift); + switch (m->spec->reg_type) { case MON3: m->hw.start_hwmon = start_bw_hwmon3; From 6b881f24a92a9714187e756dd909018cf4ab1eed Mon Sep 17 00:00:00 2001 From: Rohit Gupta Date: Wed, 16 Jan 2019 20:22:37 -0800 Subject: [PATCH 19/20] devfreq: simple-dev: Make the freq-table property optional The call to devfreq_add_device() tries to initialize frequency tables for simple-dev. This can be a more efficient way of setting up frequency tables in case there are multiple tables for a single device depending on hardware version. Make 'freq-tbl-khz' property an optional one so that probe doesn't fail in its absence and devfreq_add_device() also gets a chance to set up the freq table for simple-dev. Change-Id: I96b5c2d4aef2085512d94dc6792f5ba2711de64b Signed-off-by: Rohit Gupta [avajid@codeaurora.org: made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/devfreq_simple_dev.c | 60 ++++++++++++++++++---------- 1 file changed, 39 insertions(+), 21 deletions(-) diff --git a/drivers/devfreq/devfreq_simple_dev.c b/drivers/devfreq/devfreq_simple_dev.c index c3010cf177ba..c5431755a176 100644 --- a/drivers/devfreq/devfreq_simple_dev.c +++ b/drivers/devfreq/devfreq_simple_dev.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "devfreq-simple-dev: " fmt @@ -26,6 +26,7 @@ struct dev_data { struct clk *clk; struct devfreq *df; struct devfreq_dev_profile profile; + bool freq_in_khz; }; static void find_freq(struct devfreq_dev_profile *p, unsigned long *freq, @@ -57,7 +58,7 @@ static int dev_target(struct device *dev, unsigned long *freq, u32 flags) find_freq(&d->profile, freq, flags); - rfreq = clk_round_rate(d->clk, *freq * 1000); + rfreq = clk_round_rate(d->clk, d->freq_in_khz ? *freq * 1000 : *freq); if (IS_ERR_VALUE(rfreq)) { dev_err(dev, "devfreq: Cannot find matching frequency for %lu\n", *freq); @@ -75,39 +76,30 @@ static int dev_get_cur_freq(struct device *dev, unsigned long *freq) f = clk_get_rate(d->clk); if (IS_ERR_VALUE(f)) return f; - *freq = f / 1000; + *freq = d->freq_in_khz ? f / 1000 : f; return 0; } #define PROP_TBL "freq-tbl-khz" -static int devfreq_clock_probe(struct platform_device *pdev) +static int parse_freq_table(struct device *dev, struct dev_data *d) { - struct device *dev = &pdev->dev; - struct dev_data *d; - struct devfreq_dev_profile *p; - u32 *data, poll; - const char *gov_name; + struct devfreq_dev_profile *p = &d->profile; int ret, len, i, j; + u32 *data; unsigned long f; - d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); - if (!d) - return -ENOMEM; - platform_set_drvdata(pdev, d); - - d->clk = devm_clk_get(dev, "devfreq_clk"); - if (IS_ERR(d->clk)) - return PTR_ERR(d->clk); - - if (!of_find_property(dev->of_node, PROP_TBL, &len)) - return -EINVAL; + if (!of_find_property(dev->of_node, PROP_TBL, &len)) { + if (dev_pm_opp_get_opp_count(dev) <= 0) + return -EPROBE_DEFER; + return 0; + } + d->freq_in_khz = true; len /= sizeof(*data); data = devm_kzalloc(dev, len * sizeof(*data), GFP_KERNEL); if (!data) return -ENOMEM; - p = &d->profile; p->freq_table = devm_kzalloc(dev, len * sizeof(*p->freq_table), GFP_KERNEL); if (!p->freq_table) @@ -134,6 +126,32 @@ static int devfreq_clock_probe(struct platform_device *pdev) return -EINVAL; } + return 0; +} + +static int devfreq_clock_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct dev_data *d; + struct devfreq_dev_profile *p; + u32 poll; + const char *gov_name; + int ret; + + d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); + if (!d) + return -ENOMEM; + platform_set_drvdata(pdev, d); + + d->clk = devm_clk_get(dev, "devfreq_clk"); + if (IS_ERR(d->clk)) + return PTR_ERR(d->clk); + + ret = parse_freq_table(dev, d); + if (ret < 0) + return ret; + + p = &d->profile; p->target = dev_target; p->get_cur_freq = dev_get_cur_freq; ret = dev_get_cur_freq(dev, &p->initial_freq); From 93c29d13974450cc18ebc66babada59964beec16 Mon Sep 17 00:00:00 2001 From: Saravana Kannan Date: Wed, 16 Jan 2019 20:24:29 -0800 Subject: [PATCH 20/20] PM / devfreq: memlat: Look for min stall% in addition to ratio criteria Some workloads doing memory access might appear memory latency bound even though they might not actually be memory latency bound. This error can happen when the core that's running the workload is very parallelized or can do out of order executions, etc so not all memory accesses would actually stall the core. This can also happen when the the memory access monitoring capabilities aren't ideal and end up counting more kinds of memory accesses than what would be ideal. In this case, the IPM ratio can be lower than what it would be if we had ideal monitoring capabilities. To account for these errors, if the core has a stall cycle counting capabilities, check for a minimum stall% before the workload is considered memory latency bound. This would help reduce the inaccuracies, but is not a replacement for IPM ratio scheme because the stall% method doesn't allow us to detect which level of memory the workload is latency bound on, but the IPM ratio does (based on which memory accesses we use for calculating the ratio). Change-Id: I4363d7848584e5562f6683b5ad6b0f99017ec71b Signed-off-by: Saravana Kannan [avajid@codeaurora.org: resolved minor merge conflicts and made minor styling changes] Signed-off-by: Amir Vajid --- drivers/devfreq/arm-memlat-mon.c | 30 +++++++++++++++++++++++++++--- drivers/devfreq/governor_memlat.c | 9 ++++++++- drivers/devfreq/governor_memlat.h | 1 + include/trace/events/power.h | 10 +++++++--- 4 files changed, 43 insertions(+), 7 deletions(-) diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c index 76f13daa5565..cbc51baf2991 100644 --- a/drivers/devfreq/arm-memlat-mon.c +++ b/drivers/devfreq/arm-memlat-mon.c @@ -28,6 +28,7 @@ enum ev_index { INST_IDX, CM_IDX, CYC_IDX, + STALL_CYC_IDX, NUM_EVENTS }; #define INST_EV 0x08 @@ -92,12 +93,19 @@ static void read_perf_counters(int cpu, struct cpu_grp_info *cpu_grp) { struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); struct dev_stats *devstats = to_devstats(cpu_grp, cpu); - unsigned long cyc_cnt; + unsigned long cyc_cnt, stall_cnt; devstats->inst_count = read_event(&cpustats->events[INST_IDX]); devstats->mem_count = read_event(&cpustats->events[CM_IDX]); cyc_cnt = read_event(&cpustats->events[CYC_IDX]); devstats->freq = compute_freq(cpustats, cyc_cnt); + if (cpustats->events[STALL_CYC_IDX].pevent) { + stall_cnt = read_event(&cpustats->events[STALL_CYC_IDX]); + stall_cnt = min(stall_cnt, cyc_cnt); + devstats->stall_pct = mult_frac(100, stall_cnt, cyc_cnt); + } else { + devstats->stall_pct = 100; + } } static unsigned long get_cnt(struct memlat_hwmon *hw) @@ -117,7 +125,10 @@ static void delete_events(struct cpu_pmu_stats *cpustats) for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { cpustats->events[i].prev_count = 0; - perf_event_release_kernel(cpustats->events[i].pevent); + if (cpustats->events[i].pevent) { + perf_event_release_kernel(cpustats->events[i].pevent); + cpustats->events[i].pevent = NULL; + } } } @@ -135,6 +146,7 @@ static void stop_hwmon(struct memlat_hwmon *hw) devstats->inst_count = 0; devstats->mem_count = 0; devstats->freq = 0; + devstats->stall_pct = 0; } } @@ -158,6 +170,7 @@ static int set_events(struct cpu_grp_info *cpu_grp, int cpu) struct perf_event *pevent; struct perf_event_attr *attr; int err, i; + unsigned int event_id; struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); /* Allocate an attribute for event initialization */ @@ -166,7 +179,11 @@ static int set_events(struct cpu_grp_info *cpu_grp, int cpu) return -ENOMEM; for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { - attr->config = cpu_grp->event_ids[i]; + event_id = cpu_grp->event_ids[i]; + if (!event_id) + continue; + + attr->config = event_id; pevent = perf_event_create_kernel_counter(attr, cpu, NULL, NULL, NULL); if (IS_ERR(pevent)) @@ -282,6 +299,13 @@ static int arm_memlat_mon_driver_probe(struct platform_device *pdev) } cpu_grp->event_ids[INST_IDX] = event_id; + ret = of_property_read_u32(dev->of_node, "qcom,stall-cycle-ev", + &event_id); + if (ret) + dev_dbg(dev, "Stall cycle event not specified. Event ignored.\n"); + else + cpu_grp->event_ids[STALL_CYC_IDX] = event_id; + for_each_cpu(cpu, &cpu_grp->cpus) to_devstats(cpu_grp, cpu)->id = cpu; diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c index b7415b0eaeef..9f1186542cea 100644 --- a/drivers/devfreq/governor_memlat.c +++ b/drivers/devfreq/governor_memlat.c @@ -27,6 +27,7 @@ struct memlat_node { unsigned int ratio_ceil; + unsigned int stall_floor; bool mon_started; bool already_zero; struct list_head list; @@ -234,9 +235,11 @@ static int devfreq_memlat_get_freq(struct devfreq *df, hw->core_stats[i].id, hw->core_stats[i].inst_count, hw->core_stats[i].mem_count, - hw->core_stats[i].freq, ratio); + hw->core_stats[i].freq, + hw->core_stats[i].stall_pct, ratio); if (ratio <= node->ratio_ceil + && hw->core_stats[i].stall_pct >= node->stall_floor && hw->core_stats[i].freq > max_freq) { lat_dev = i; max_freq = hw->core_stats[i].freq; @@ -264,9 +267,13 @@ static int devfreq_memlat_get_freq(struct devfreq *df, show_attr(ratio_ceil); store_attr(ratio_ceil, 1U, 20000U); static DEVICE_ATTR_RW(ratio_ceil); +show_attr(stall_floor); +store_attr(stall_floor, 0U, 100U); +static DEVICE_ATTR_RW(stall_floor); static struct attribute *dev_attr[] = { &dev_attr_ratio_ceil.attr, + &dev_attr_stall_floor.attr, &dev_attr_freq_map.attr, NULL, }; diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h index 6eb3dee6ae0f..335ba7598b6b 100644 --- a/drivers/devfreq/governor_memlat.h +++ b/drivers/devfreq/governor_memlat.h @@ -21,6 +21,7 @@ struct dev_stats { unsigned long inst_count; unsigned long mem_count; unsigned long freq; + unsigned long stall_pct; }; struct core_dev_map { diff --git a/include/trace/events/power.h b/include/trace/events/power.h index f95eb5bba10a..04bd8831dd2b 100644 --- a/include/trace/events/power.h +++ b/include/trace/events/power.h @@ -631,9 +631,10 @@ TRACE_EVENT(cache_hwmon_update, TRACE_EVENT(memlat_dev_meas, TP_PROTO(const char *name, unsigned int dev_id, unsigned long inst, - unsigned long mem, unsigned long freq, unsigned int ratio), + unsigned long mem, unsigned long freq, unsigned int stall, + unsigned int ratio), - TP_ARGS(name, dev_id, inst, mem, freq, ratio), + TP_ARGS(name, dev_id, inst, mem, freq, stall, ratio), TP_STRUCT__entry( __string(name, name) @@ -641,6 +642,7 @@ TRACE_EVENT(memlat_dev_meas, __field(unsigned long, inst) __field(unsigned long, mem) __field(unsigned long, freq) + __field(unsigned int, stall) __field(unsigned int, ratio) ), @@ -650,15 +652,17 @@ TRACE_EVENT(memlat_dev_meas, __entry->inst = inst; __entry->mem = mem; __entry->freq = freq; + __entry->stall = stall; __entry->ratio = ratio; ), - TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, ratio=%u", + TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, stall=%u, ratio=%u", __get_str(name), __entry->dev_id, __entry->inst, __entry->mem, __entry->freq, + __entry->stall, __entry->ratio) );