diff --git a/drivers/devfreq/Kconfig b/drivers/devfreq/Kconfig index 1f713deff5f4..624ad55216b0 100644 --- a/drivers/devfreq/Kconfig +++ b/drivers/devfreq/Kconfig @@ -83,6 +83,15 @@ config QCOM_BIMC_BWMON has the capability to raise an IRQ when the count exceeds a programmable limit. +config ARM_MEMLAT_MON + tristate "ARM CPU Memory Latency monitor hardware" + depends on ARCH_QCOM + help + The PMU present on these ARM cores allow for the use of counters to + monitor the memory latency characteristics of an ARM CPU workload. + This driver uses these counters to implement the APIs needed by + the mem_latency devfreq governor. + config DEVFREQ_GOV_QCOM_BW_HWMON tristate "HW monitor based governor for device BW" depends on QCOM_BIMC_BWMON @@ -102,6 +111,16 @@ config DEVFREQ_GOV_QCOM_CACHE_HWMON it can conflict with existing profiling tools. This governor is unlikely to be useful for other devices. +config DEVFREQ_GOV_MEMLAT + tristate "HW monitor based governor for device BW" + depends on ARM_MEMLAT_MON + help + HW monitor based governor for device to DDR bandwidth voting. + This governor sets the CPU BW vote based on stats obtained from memalat + monitor if it determines that a workload is memory latency bound. Since + this uses target specific counters it can conflict with existing profiling + tools. + comment "DEVFREQ Drivers" config ARM_EXYNOS_BUS_DEVFREQ diff --git a/drivers/devfreq/Makefile b/drivers/devfreq/Makefile index c1cdd38b1a5c..359b04eb21be 100644 --- a/drivers/devfreq/Makefile +++ b/drivers/devfreq/Makefile @@ -7,8 +7,10 @@ obj-$(CONFIG_DEVFREQ_GOV_POWERSAVE) += governor_powersave.o obj-$(CONFIG_DEVFREQ_GOV_USERSPACE) += governor_userspace.o obj-$(CONFIG_DEVFREQ_GOV_PASSIVE) += governor_passive.o obj-$(CONFIG_QCOM_BIMC_BWMON) += bimc-bwmon.o +obj-$(CONFIG_ARM_MEMLAT_MON) += arm-memlat-mon.o obj-$(CONFIG_DEVFREQ_GOV_QCOM_BW_HWMON) += governor_bw_hwmon.o obj-$(CONFIG_DEVFREQ_GOV_QCOM_CACHE_HWMON) += governor_cache_hwmon.o +obj-$(CONFIG_DEVFREQ_GOV_MEMLAT) += governor_memlat.o # DEVFREQ Drivers obj-$(CONFIG_ARM_EXYNOS_BUS_DEVFREQ) += exynos-bus.o diff --git a/drivers/devfreq/arm-memlat-mon.c b/drivers/devfreq/arm-memlat-mon.c new file mode 100644 index 000000000000..cbc51baf2991 --- /dev/null +++ b/drivers/devfreq/arm-memlat-mon.c @@ -0,0 +1,338 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "arm-memlat-mon: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "governor.h" +#include "governor_memlat.h" +#include + +enum ev_index { + INST_IDX, + CM_IDX, + CYC_IDX, + STALL_CYC_IDX, + NUM_EVENTS +}; +#define INST_EV 0x08 +#define L2DM_EV 0x17 +#define CYC_EV 0x11 + +struct event_data { + struct perf_event *pevent; + unsigned long prev_count; +}; + +struct cpu_pmu_stats { + struct event_data events[NUM_EVENTS]; + ktime_t prev_ts; +}; + +struct cpu_grp_info { + cpumask_t cpus; + unsigned int event_ids[NUM_EVENTS]; + struct cpu_pmu_stats *cpustats; + struct memlat_hwmon hw; +}; + +#define to_cpustats(cpu_grp, cpu) \ + (&cpu_grp->cpustats[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_devstats(cpu_grp, cpu) \ + (&cpu_grp->hw.core_stats[cpu - cpumask_first(&cpu_grp->cpus)]) +#define to_cpu_grp(hwmon) container_of(hwmon, struct cpu_grp_info, hw) + + +static unsigned long compute_freq(struct cpu_pmu_stats *cpustats, + unsigned long cyc_cnt) +{ + ktime_t ts; + unsigned int diff; + unsigned long freq = 0; + + ts = ktime_get(); + diff = ktime_to_us(ktime_sub(ts, cpustats->prev_ts)); + if (!diff) + diff = 1; + cpustats->prev_ts = ts; + freq = cyc_cnt; + do_div(freq, diff); + + return freq; +} + +#define MAX_COUNT_LIM 0xFFFFFFFFFFFFFFFF +static inline unsigned long read_event(struct event_data *event) +{ + unsigned long ev_count; + u64 total, enabled, running; + + total = perf_event_read_value(event->pevent, &enabled, &running); + ev_count = total - event->prev_count; + event->prev_count = total; + return ev_count; +} + +static void read_perf_counters(int cpu, struct cpu_grp_info *cpu_grp) +{ + struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); + struct dev_stats *devstats = to_devstats(cpu_grp, cpu); + unsigned long cyc_cnt, stall_cnt; + + devstats->inst_count = read_event(&cpustats->events[INST_IDX]); + devstats->mem_count = read_event(&cpustats->events[CM_IDX]); + cyc_cnt = read_event(&cpustats->events[CYC_IDX]); + devstats->freq = compute_freq(cpustats, cyc_cnt); + if (cpustats->events[STALL_CYC_IDX].pevent) { + stall_cnt = read_event(&cpustats->events[STALL_CYC_IDX]); + stall_cnt = min(stall_cnt, cyc_cnt); + devstats->stall_pct = mult_frac(100, stall_cnt, cyc_cnt); + } else { + devstats->stall_pct = 100; + } +} + +static unsigned long get_cnt(struct memlat_hwmon *hw) +{ + int cpu; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + + for_each_cpu(cpu, &cpu_grp->cpus) + read_perf_counters(cpu, cpu_grp); + + return 0; +} + +static void delete_events(struct cpu_pmu_stats *cpustats) +{ + int i; + + for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { + cpustats->events[i].prev_count = 0; + if (cpustats->events[i].pevent) { + perf_event_release_kernel(cpustats->events[i].pevent); + cpustats->events[i].pevent = NULL; + } + } +} + +static void stop_hwmon(struct memlat_hwmon *hw) +{ + int cpu; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + struct dev_stats *devstats; + + for_each_cpu(cpu, &cpu_grp->cpus) { + delete_events(to_cpustats(cpu_grp, cpu)); + + /* Clear governor data */ + devstats = to_devstats(cpu_grp, cpu); + devstats->inst_count = 0; + devstats->mem_count = 0; + devstats->freq = 0; + devstats->stall_pct = 0; + } +} + +static struct perf_event_attr *alloc_attr(void) +{ + struct perf_event_attr *attr; + + attr = kzalloc(sizeof(struct perf_event_attr), GFP_KERNEL); + if (!attr) + return attr; + + attr->type = PERF_TYPE_RAW; + attr->size = sizeof(struct perf_event_attr); + attr->pinned = 1; + + return attr; +} + +static int set_events(struct cpu_grp_info *cpu_grp, int cpu) +{ + struct perf_event *pevent; + struct perf_event_attr *attr; + int err, i; + unsigned int event_id; + struct cpu_pmu_stats *cpustats = to_cpustats(cpu_grp, cpu); + + /* Allocate an attribute for event initialization */ + attr = alloc_attr(); + if (!attr) + return -ENOMEM; + + for (i = 0; i < ARRAY_SIZE(cpustats->events); i++) { + event_id = cpu_grp->event_ids[i]; + if (!event_id) + continue; + + attr->config = event_id; + pevent = perf_event_create_kernel_counter(attr, cpu, NULL, + NULL, NULL); + if (IS_ERR(pevent)) + goto err_out; + cpustats->events[i].pevent = pevent; + perf_event_enable(pevent); + } + + kfree(attr); + return 0; + +err_out: + err = PTR_ERR(pevent); + kfree(attr); + return err; +} + +static int start_hwmon(struct memlat_hwmon *hw) +{ + int cpu, ret = 0; + struct cpu_grp_info *cpu_grp = to_cpu_grp(hw); + + for_each_cpu(cpu, &cpu_grp->cpus) { + ret = set_events(cpu_grp, cpu); + if (ret < 0) { + pr_warn("Perf event init failed on CPU%d: %d\n", cpu, + ret); + break; + } + } + + return ret; +} + +static int get_mask_from_dev_handle(struct platform_device *pdev, + cpumask_t *mask) +{ + struct device *dev = &pdev->dev; + struct device_node *dev_phandle; + struct device *cpu_dev; + int cpu, i = 0; + int ret = -ENOENT; + + dev_phandle = of_parse_phandle(dev->of_node, "qcom,cpulist", i++); + while (dev_phandle) { + for_each_possible_cpu(cpu) { + cpu_dev = get_cpu_device(cpu); + if (cpu_dev && cpu_dev->of_node == dev_phandle) { + cpumask_set_cpu(cpu, mask); + ret = 0; + break; + } + } + dev_phandle = of_parse_phandle(dev->of_node, + "qcom,cpulist", i++); + } + + return ret; +} + +static int arm_memlat_mon_driver_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct memlat_hwmon *hw; + struct cpu_grp_info *cpu_grp; + int cpu, ret; + u32 event_id; + + cpu_grp = devm_kzalloc(dev, sizeof(*cpu_grp), GFP_KERNEL); + if (!cpu_grp) + return -ENOMEM; + hw = &cpu_grp->hw; + + hw->dev = dev; + hw->of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); + if (!hw->of_node) { + dev_err(dev, "Couldn't find a target device\n"); + return -ENODEV; + } + + if (get_mask_from_dev_handle(pdev, &cpu_grp->cpus)) { + dev_err(dev, "CPU list is empty\n"); + return -ENODEV; + } + + hw->num_cores = cpumask_weight(&cpu_grp->cpus); + hw->core_stats = devm_kzalloc(dev, hw->num_cores * + sizeof(*(hw->core_stats)), GFP_KERNEL); + if (!hw->core_stats) + return -ENOMEM; + + cpu_grp->cpustats = devm_kzalloc(dev, hw->num_cores * + sizeof(*(cpu_grp->cpustats)), GFP_KERNEL); + if (!cpu_grp->cpustats) + return -ENOMEM; + + cpu_grp->event_ids[CYC_IDX] = CYC_EV; + + ret = of_property_read_u32(dev->of_node, "qcom,cachemiss-ev", + &event_id); + if (ret < 0) { + dev_dbg(dev, "Cache Miss event not specified. Using def:0x%x\n", + L2DM_EV); + event_id = L2DM_EV; + } + cpu_grp->event_ids[CM_IDX] = event_id; + + ret = of_property_read_u32(dev->of_node, "qcom,inst-ev", &event_id); + if (ret < 0) { + dev_dbg(dev, "Inst event not specified. Using def:0x%x\n", + INST_EV); + event_id = INST_EV; + } + cpu_grp->event_ids[INST_IDX] = event_id; + + ret = of_property_read_u32(dev->of_node, "qcom,stall-cycle-ev", + &event_id); + if (ret) + dev_dbg(dev, "Stall cycle event not specified. Event ignored.\n"); + else + cpu_grp->event_ids[STALL_CYC_IDX] = event_id; + + for_each_cpu(cpu, &cpu_grp->cpus) + to_devstats(cpu_grp, cpu)->id = cpu; + + hw->start_hwmon = &start_hwmon; + hw->stop_hwmon = &stop_hwmon; + hw->get_cnt = &get_cnt; + + ret = register_memlat(dev, hw); + if (ret < 0) { + pr_err("Mem Latency Gov registration failed: %d\n", ret); + return ret; + } + + return 0; +} + +static const struct of_device_id memlat_match_table[] = { + { .compatible = "qcom,arm-memlat-mon" }, + {} +}; + +static struct platform_driver arm_memlat_mon_driver = { + .probe = arm_memlat_mon_driver_probe, + .driver = { + .name = "arm-memlat-mon", + .of_match_table = memlat_match_table, + }, +}; + +module_platform_driver(arm_memlat_mon_driver); diff --git a/drivers/devfreq/bimc-bwmon.c b/drivers/devfreq/bimc-bwmon.c index 8a9e4ef91824..0937cdf6c822 100644 --- a/drivers/devfreq/bimc-bwmon.c +++ b/drivers/devfreq/bimc-bwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bimc-bwmon: " fmt @@ -10,20 +10,28 @@ #include #include #include +#include #include #include #include #include #include +#include #include +#include +#include #include "governor_bw_hwmon.h" #define GLB_INT_STATUS(m) ((m)->global_base + 0x100) #define GLB_INT_CLR(m) ((m)->global_base + 0x108) #define GLB_INT_EN(m) ((m)->global_base + 0x10C) #define MON_INT_STATUS(m) ((m)->base + 0x100) +#define MON_INT_STATUS_MASK 0x03 +#define MON2_INT_STATUS_MASK 0xF0 +#define MON2_INT_STATUS_SHIFT 4 #define MON_INT_CLR(m) ((m)->base + 0x108) #define MON_INT_EN(m) ((m)->base + 0x10C) +#define MON_INT_ENABLE 0x1 #define MON_EN(m) ((m)->base + 0x280) #define MON_CLEAR(m) ((m)->base + 0x284) #define MON_CNT(m) ((m)->base + 0x288) @@ -31,31 +39,137 @@ #define MON_MASK(m) ((m)->base + 0x298) #define MON_MATCH(m) ((m)->base + 0x29C) +#define MON2_EN(m) ((m)->base + 0x2A0) +#define MON2_CLEAR(m) ((m)->base + 0x2A4) +#define MON2_SW(m) ((m)->base + 0x2A8) +#define MON2_THRES_HI(m) ((m)->base + 0x2AC) +#define MON2_THRES_MED(m) ((m)->base + 0x2B0) +#define MON2_THRES_LO(m) ((m)->base + 0x2B4) +#define MON2_ZONE_ACTIONS(m) ((m)->base + 0x2B8) +#define MON2_ZONE_CNT_THRES(m) ((m)->base + 0x2BC) +#define MON2_BYTE_CNT(m) ((m)->base + 0x2D0) +#define MON2_WIN_TIMER(m) ((m)->base + 0x2D4) +#define MON2_ZONE_CNT(m) ((m)->base + 0x2D8) +#define MON2_ZONE_MAX(m, zone) ((m)->base + 0x2E0 + 0x4 * zone) + +#define MON3_INT_STATUS(m) ((m)->base + 0x00) +#define MON3_INT_CLR(m) ((m)->base + 0x08) +#define MON3_INT_EN(m) ((m)->base + 0x0C) +#define MON3_INT_STATUS_MASK 0x0F +#define MON3_EN(m) ((m)->base + 0x10) +#define MON3_CLEAR(m) ((m)->base + 0x14) +#define MON3_MASK(m) ((m)->base + 0x18) +#define MON3_MATCH(m) ((m)->base + 0x1C) +#define MON3_SW(m) ((m)->base + 0x20) +#define MON3_THRES_HI(m) ((m)->base + 0x24) +#define MON3_THRES_MED(m) ((m)->base + 0x28) +#define MON3_THRES_LO(m) ((m)->base + 0x2C) +#define MON3_ZONE_ACTIONS(m) ((m)->base + 0x30) +#define MON3_ZONE_CNT_THRES(m) ((m)->base + 0x34) +#define MON3_BYTE_CNT(m) ((m)->base + 0x38) +#define MON3_WIN_TIMER(m) ((m)->base + 0x3C) +#define MON3_ZONE_CNT(m) ((m)->base + 0x40) +#define MON3_ZONE_MAX(m, zone) ((m)->base + 0x44 + 0x4 * zone) + +enum mon_reg_type { + MON1, + MON2, + MON3, +}; + +struct bwmon_spec { + bool wrap_on_thres; + bool overflow; + bool throt_adj; + bool hw_sampling; + bool has_global_base; + enum mon_reg_type reg_type; +}; + struct bwmon { - void __iomem *base; - void __iomem *global_base; - unsigned int mport; - unsigned int irq; - struct device *dev; - struct bw_hwmon hw; + void __iomem *base; + void __iomem *global_base; + unsigned int mport; + int irq; + const struct bwmon_spec *spec; + struct device *dev; + struct bw_hwmon hw; + u32 hw_timer_hz; + u32 throttle_adj; + u32 sample_size_ms; + u32 intr_status; + u8 count_shift; + u32 thres_lim; + u32 byte_mask; + u32 byte_match; }; #define to_bwmon(ptr) container_of(ptr, struct bwmon, hw) +#define ENABLE_MASK BIT(0) +#define THROTTLE_MASK 0x1F +#define THROTTLE_SHIFT 16 + static DEFINE_SPINLOCK(glb_lock); -static void mon_enable(struct bwmon *m) + +static __always_inline void mon_enable(struct bwmon *m, enum mon_reg_type type) { - writel_relaxed(0x1, MON_EN(m)); + switch (type) { + case MON1: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON_EN(m)); + break; + case MON2: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON2_EN(m)); + break; + case MON3: + writel_relaxed(ENABLE_MASK | m->throttle_adj, MON3_EN(m)); + break; + } } -static void mon_disable(struct bwmon *m) +static __always_inline void mon_disable(struct bwmon *m, enum mon_reg_type type) { - writel_relaxed(0x0, MON_EN(m)); + switch (type) { + case MON1: + writel_relaxed(m->throttle_adj, MON_EN(m)); + break; + case MON2: + writel_relaxed(m->throttle_adj, MON2_EN(m)); + break; + case MON3: + writel_relaxed(m->throttle_adj, MON3_EN(m)); + break; + } + /* + * mon_disable() and mon_irq_clear(), + * If latter goes first and count happen to trigger irq, we would + * have the irq line high but no one handling it. + */ + mb(); } -static void mon_clear(struct bwmon *m) +#define MON_CLEAR_BIT 0x1 +#define MON_CLEAR_ALL_BIT 0x2 +static __always_inline +void mon_clear(struct bwmon *m, bool clear_all, enum mon_reg_type type) { - writel_relaxed(0x1, MON_CLEAR(m)); + switch (type) { + case MON1: + writel_relaxed(MON_CLEAR_BIT, MON_CLEAR(m)); + break; + case MON2: + if (clear_all) + writel_relaxed(MON_CLEAR_ALL_BIT, MON2_CLEAR(m)); + else + writel_relaxed(MON_CLEAR_BIT, MON2_CLEAR(m)); + break; + case MON3: + if (clear_all) + writel_relaxed(MON_CLEAR_ALL_BIT, MON3_CLEAR(m)); + else + writel_relaxed(MON_CLEAR_BIT, MON3_CLEAR(m)); + break; + } /* * The counter clear and IRQ clear bits are not in the same 4KB * region. So, we need to make sure the counter clear is completed @@ -64,58 +178,317 @@ static void mon_clear(struct bwmon *m) mb(); } -static void mon_irq_enable(struct bwmon *m) +#define SAMPLE_WIN_LIM 0xFFFFF +static __always_inline +void mon_set_hw_sampling_window(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) +{ + u32 rate; + + if (unlikely(sample_ms != m->sample_size_ms)) { + rate = mult_frac(sample_ms, m->hw_timer_hz, MSEC_PER_SEC); + m->sample_size_ms = sample_ms; + if (unlikely(rate > SAMPLE_WIN_LIM)) { + rate = SAMPLE_WIN_LIM; + pr_warn("Sample window %u larger than hw limit: %u\n", + rate, SAMPLE_WIN_LIM); + } + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return; + case MON2: + writel_relaxed(rate, MON2_SW(m)); + break; + case MON3: + writel_relaxed(rate, MON3_SW(m)); + break; + } + } +} + +static void mon_glb_irq_enable(struct bwmon *m) { u32 val; - spin_lock(&glb_lock); val = readl_relaxed(GLB_INT_EN(m)); val |= 1 << m->mport; writel_relaxed(val, GLB_INT_EN(m)); - spin_unlock(&glb_lock); - - val = readl_relaxed(MON_INT_EN(m)); - val |= 0x1; - writel_relaxed(val, MON_INT_EN(m)); } -static void mon_irq_disable(struct bwmon *m) +static __always_inline +void mon_irq_enable(struct bwmon *m, enum mon_reg_type type) { u32 val; spin_lock(&glb_lock); + switch (type) { + case MON1: + mon_glb_irq_enable(m); + val = readl_relaxed(MON_INT_EN(m)); + val |= MON_INT_ENABLE; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON2: + mon_glb_irq_enable(m); + val = readl_relaxed(MON_INT_EN(m)); + val |= MON2_INT_STATUS_MASK; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON3: + val = readl_relaxed(MON3_INT_EN(m)); + val |= MON3_INT_STATUS_MASK; + writel_relaxed(val, MON3_INT_EN(m)); + break; + } + spin_unlock(&glb_lock); + /* + * make sure irq enable complete for local and global + * to avoid race with other monitor calls + */ + mb(); +} + +static void mon_glb_irq_disable(struct bwmon *m) +{ + u32 val; + val = readl_relaxed(GLB_INT_EN(m)); val &= ~(1 << m->mport); writel_relaxed(val, GLB_INT_EN(m)); - spin_unlock(&glb_lock); - - val = readl_relaxed(MON_INT_EN(m)); - val &= ~0x1; - writel_relaxed(val, MON_INT_EN(m)); } -static int mon_irq_status(struct bwmon *m) +static __always_inline +void mon_irq_disable(struct bwmon *m, enum mon_reg_type type) +{ + u32 val; + + spin_lock(&glb_lock); + + switch (type) { + case MON1: + mon_glb_irq_disable(m); + val = readl_relaxed(MON_INT_EN(m)); + val &= ~MON_INT_ENABLE; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON2: + mon_glb_irq_disable(m); + val = readl_relaxed(MON_INT_EN(m)); + val &= ~MON2_INT_STATUS_MASK; + writel_relaxed(val, MON_INT_EN(m)); + break; + case MON3: + val = readl_relaxed(MON3_INT_EN(m)); + val &= ~MON3_INT_STATUS_MASK; + writel_relaxed(val, MON3_INT_EN(m)); + break; + } + spin_unlock(&glb_lock); + /* + * make sure irq disable complete for local and global + * to avoid race with other monitor calls + */ + mb(); +} + +static __always_inline +unsigned int mon_irq_status(struct bwmon *m, enum mon_reg_type type) { u32 mval; - mval = readl_relaxed(MON_INT_STATUS(m)); + switch (type) { + case MON1: + mval = readl_relaxed(MON_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, + readl_relaxed(GLB_INT_STATUS(m))); + mval &= MON_INT_STATUS_MASK; + break; + case MON2: + mval = readl_relaxed(MON_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, + readl_relaxed(GLB_INT_STATUS(m))); + mval &= MON2_INT_STATUS_MASK; + mval >>= MON2_INT_STATUS_SHIFT; + break; + case MON3: + mval = readl_relaxed(MON3_INT_STATUS(m)); + dev_dbg(m->dev, "IRQ status p:%x\n", mval); + mval &= MON3_INT_STATUS_MASK; + break; + } - dev_dbg(m->dev, "IRQ status p:%x, g:%x\n", mval, - readl_relaxed(GLB_INT_STATUS(m))); - - return mval & 0x1; + return mval; } -static void mon_irq_clear(struct bwmon *m) + +static void mon_glb_irq_clear(struct bwmon *m) { - writel_relaxed(0x1, MON_INT_CLR(m)); - /* Ensure the monitor IRQ is clear before clearing GLB IRQ */ + /* + * Synchronize the local interrupt clear in mon_irq_clear() + * with the global interrupt clear here. Otherwise, the CPU + * may reorder the two writes and clear the global interrupt + * before the local interrupt, causing the global interrupt + * to be retriggered by the local interrupt still being high. + */ mb(); writel_relaxed(1 << m->mport, GLB_INT_CLR(m)); - /* Ensure the GLB IRQ clear is complete */ + /* + * Similarly, because the global registers are in a different + * region than the local registers, we need to ensure any register + * writes to enable the monitor after this call are ordered with the + * clearing here so that local writes don't happen before the + * interrupt is cleared. + */ mb(); } +static __always_inline +void mon_irq_clear(struct bwmon *m, enum mon_reg_type type) +{ + switch (type) { + case MON1: + writel_relaxed(MON_INT_STATUS_MASK, MON_INT_CLR(m)); + mon_glb_irq_clear(m); + break; + case MON2: + writel_relaxed(MON2_INT_STATUS_MASK, MON_INT_CLR(m)); + mon_glb_irq_clear(m); + break; + case MON3: + writel_relaxed(MON3_INT_STATUS_MASK, MON3_INT_CLR(m)); + break; + } +} + +static int mon_set_throttle_adj(struct bw_hwmon *hw, uint adj) +{ + struct bwmon *m = to_bwmon(hw); + + if (adj > THROTTLE_MASK) + return -EINVAL; + + adj = (adj & THROTTLE_MASK) << THROTTLE_SHIFT; + m->throttle_adj = adj; + + return 0; +} + +static u32 mon_get_throttle_adj(struct bw_hwmon *hw) +{ + struct bwmon *m = to_bwmon(hw); + + return m->throttle_adj >> THROTTLE_SHIFT; +} + +#define ZONE1_SHIFT 8 +#define ZONE2_SHIFT 16 +#define ZONE3_SHIFT 24 +#define ZONE0_ACTION 0x01 /* Increment zone 0 count */ +#define ZONE1_ACTION 0x09 /* Increment zone 1 & clear lower zones */ +#define ZONE2_ACTION 0x25 /* Increment zone 2 & clear lower zones */ +#define ZONE3_ACTION 0x95 /* Increment zone 3 & clear lower zones */ +static u32 calc_zone_actions(void) +{ + u32 zone_actions; + + zone_actions = ZONE0_ACTION; + zone_actions |= ZONE1_ACTION << ZONE1_SHIFT; + zone_actions |= ZONE2_ACTION << ZONE2_SHIFT; + zone_actions |= ZONE3_ACTION << ZONE3_SHIFT; + + return zone_actions; +} + +#define ZONE_CNT_LIM 0xFFU +#define UP_CNT_1 1 +static u32 calc_zone_counts(struct bw_hwmon *hw) +{ + u32 zone_counts; + + zone_counts = ZONE_CNT_LIM; + zone_counts |= min(hw->down_cnt, ZONE_CNT_LIM) << ZONE1_SHIFT; + zone_counts |= ZONE_CNT_LIM << ZONE2_SHIFT; + zone_counts |= UP_CNT_1 << ZONE3_SHIFT; + + return zone_counts; +} + +#define MB_SHIFT 20 + +static u32 mbps_to_count(unsigned long mbps, unsigned int ms, u8 shift) +{ + mbps *= ms; + + if (shift > MB_SHIFT) + mbps >>= shift - MB_SHIFT; + else + mbps <<= MB_SHIFT - shift; + + return DIV_ROUND_UP(mbps, MSEC_PER_SEC); +} + +/* + * Define the 4 zones using HI, MED & LO thresholds: + * Zone 0: byte count < THRES_LO + * Zone 1: THRES_LO < byte count < THRES_MED + * Zone 2: THRES_MED < byte count < THRES_HI + * Zone 3: THRES_LIM > byte count > THRES_HI + */ +#define THRES_LIM(shift) (0xFFFFFFFF >> shift) + +static __always_inline +void set_zone_thres(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) +{ + struct bw_hwmon *hw = &m->hw; + u32 hi, med, lo; + u32 zone_cnt_thres = calc_zone_counts(hw); + + hi = mbps_to_count(hw->up_wake_mbps, sample_ms, m->count_shift); + med = mbps_to_count(hw->down_wake_mbps, sample_ms, m->count_shift); + lo = 0; + + if (unlikely((hi > m->thres_lim) || (med > hi) || (lo > med))) { + pr_warn("Zone thres larger than hw limit: hi:%u med:%u lo:%u\n", + hi, med, lo); + hi = min(hi, m->thres_lim); + med = min(med, hi - 1); + lo = min(lo, med-1); + } + + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return; + case MON2: + writel_relaxed(hi, MON2_THRES_HI(m)); + writel_relaxed(med, MON2_THRES_MED(m)); + writel_relaxed(lo, MON2_THRES_LO(m)); + /* Set the zone count thresholds for interrupts */ + writel_relaxed(zone_cnt_thres, MON2_ZONE_CNT_THRES(m)); + break; + case MON3: + writel_relaxed(hi, MON3_THRES_HI(m)); + writel_relaxed(med, MON3_THRES_MED(m)); + writel_relaxed(lo, MON3_THRES_LO(m)); + /* Set the zone count thresholds for interrupts */ + writel_relaxed(zone_cnt_thres, MON3_ZONE_CNT_THRES(m)); + break; + } + + dev_dbg(m->dev, "Thres: hi:%u med:%u lo:%u\n", hi, med, lo); + dev_dbg(m->dev, "Zone Count Thres: %0x\n", zone_cnt_thres); +} + +static __always_inline +void mon_set_zones(struct bwmon *m, unsigned int sample_ms, + enum mon_reg_type type) +{ + mon_set_hw_sampling_window(m, sample_ms, type); + set_zone_thres(m, sample_ms, type); +} + static void mon_set_limit(struct bwmon *m, u32 count) { writel_relaxed(count, MON_THRES(m)); @@ -127,15 +500,111 @@ static u32 mon_get_limit(struct bwmon *m) return readl_relaxed(MON_THRES(m)); } -static unsigned long mon_get_count(struct bwmon *m) +#define THRES_HIT(status) (status & BIT(0)) +#define OVERFLOW(status) (status & BIT(1)) +static unsigned long mon_get_count1(struct bwmon *m) +{ + unsigned long count, status; + + count = readl_relaxed(MON_CNT(m)); + status = mon_irq_status(m, MON1); + + dev_dbg(m->dev, "Counter: %08lx\n", count); + + if (OVERFLOW(status) && m->spec->overflow) + count += 0xFFFFFFFF; + if (THRES_HIT(status) && m->spec->wrap_on_thres) + count += mon_get_limit(m); + + dev_dbg(m->dev, "Actual Count: %08lx\n", count); + + return count; +} + +static __always_inline +unsigned int get_zone(struct bwmon *m, enum mon_reg_type type) +{ + u32 zone_counts; + u32 zone; + + zone = get_bitmask_order(m->intr_status); + if (zone) { + zone--; + } else { + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return 0; + case MON2: + zone_counts = readl_relaxed(MON2_ZONE_CNT(m)); + break; + case MON3: + zone_counts = readl_relaxed(MON3_ZONE_CNT(m)); + break; + } + + if (zone_counts) { + zone = get_bitmask_order(zone_counts) - 1; + zone /= 8; + } + } + + m->intr_status = 0; + return zone; +} + +static __always_inline +unsigned long get_zone_count(struct bwmon *m, unsigned int zone, + enum mon_reg_type type) { unsigned long count; - count = readl_relaxed(MON_CNT(m)); - dev_dbg(m->dev, "Counter: %08lx\n", count); - if (mon_irq_status(m)) - count += mon_get_limit(m); - dev_dbg(m->dev, "Actual Count: %08lx\n", count); + switch (type) { + case MON1: + WARN(1, "Invalid\n"); + return 0; + case MON2: + count = readl_relaxed(MON2_ZONE_MAX(m, zone)) + 1; + break; + case MON3: + count = readl_relaxed(MON3_ZONE_MAX(m, zone)); + if (count) + count++; + break; + } + + return count; +} + +static __always_inline +unsigned long mon_get_zone_stats(struct bwmon *m, enum mon_reg_type type) +{ + unsigned int zone; + unsigned long count = 0; + + zone = get_zone(m, type); + count = get_zone_count(m, zone, type); + count <<= m->count_shift; + + dev_dbg(m->dev, "Zone%d Max byte count: %08lx\n", zone, count); + + return count; +} + +static __always_inline +unsigned long mon_get_count(struct bwmon *m, enum mon_reg_type type) +{ + unsigned long count; + + switch (type) { + case MON1: + count = mon_get_count1(m); + break; + case MON2: + case MON3: + count = mon_get_zone_stats(m, type); + break; + } return count; } @@ -143,14 +612,6 @@ static unsigned long mon_get_count(struct bwmon *m) /* ********** CPUBW specific code ********** */ /* Returns MBps of read/writes for the sampling window. */ -static unsigned int bytes_to_mbps(long long bytes, unsigned int us) -{ - bytes *= USEC_PER_SEC; - do_div(bytes, us); - bytes = DIV_ROUND_UP_ULL(bytes, SZ_1M); - return bytes; -} - static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms, unsigned int tolerance_percent) { @@ -161,137 +622,392 @@ static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms, return mbps; } -static unsigned long meas_bw_and_set_irq(struct bw_hwmon *hw, - unsigned int tol, unsigned int us) -{ - unsigned long mbps; - u32 limit; - unsigned int sample_ms = hw->df->profile->polling_ms; - struct bwmon *m = to_bwmon(hw); - - mon_disable(m); - - mbps = mon_get_count(m); - mbps = bytes_to_mbps(mbps, us); - /* - * The fudging of mbps when calculating limit is to workaround a HW - * design issue. Needs further tuning. - */ - limit = mbps_to_bytes(max(mbps, 400UL), sample_ms, tol); - mon_set_limit(m, limit); - - mon_clear(m); - mon_irq_clear(m); - mon_enable(m); - - dev_dbg(m->dev, "MBps = %lu\n", mbps); - return mbps; -} - -static irqreturn_t bwmon_intr_handler(int irq, void *dev) -{ - struct bwmon *m = dev; - - if (mon_irq_status(m)) { - update_bw_hwmon(&m->hw); - return IRQ_HANDLED; - } - - return IRQ_NONE; -} - -static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) +static __always_inline +unsigned long __get_bytes_and_clear(struct bw_hwmon *hw, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); + unsigned long count; + + mon_disable(m, type); + count = mon_get_count(m, type); + mon_clear(m, false, type); + mon_irq_clear(m, type); + mon_enable(m, type); + + return count; +} + +static unsigned long get_bytes_and_clear(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON1); +} + +static unsigned long get_bytes_and_clear2(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON2); +} + +static unsigned long get_bytes_and_clear3(struct bw_hwmon *hw) +{ + return __get_bytes_and_clear(hw, MON3); +} + +static unsigned long set_thres(struct bw_hwmon *hw, unsigned long bytes) +{ + unsigned long count; u32 limit; - int ret; + struct bwmon *m = to_bwmon(hw); - ret = request_threaded_irq(m->irq, NULL, bwmon_intr_handler, - IRQF_ONESHOT | IRQF_SHARED, - dev_name(m->dev), m); - if (ret < 0) { - dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", - ret); - return ret; - } + mon_disable(m, MON1); + count = mon_get_count1(m); + mon_clear(m, false, MON1); + mon_irq_clear(m, MON1); - mon_disable(m); + if (likely(!m->spec->wrap_on_thres)) + limit = bytes; + else + limit = max(bytes, 500000UL); - limit = mbps_to_bytes(mbps, hw->df->profile->polling_ms, 0); mon_set_limit(m, limit); + mon_enable(m, MON1); - mon_clear(m); - mon_irq_clear(m); - mon_irq_enable(m); - mon_enable(m); + return count; +} + +static unsigned long +__set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms, + enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + + mon_disable(m, type); + mon_clear(m, false, type); + mon_irq_clear(m, type); + + mon_set_zones(m, sample_ms, type); + mon_enable(m, type); return 0; } -static void stop_bw_hwmon(struct bw_hwmon *hw) +static unsigned long set_hw_events(struct bw_hwmon *hw, unsigned int sample_ms) +{ + return __set_hw_events(hw, sample_ms, MON2); +} + +static unsigned long +set_hw_events3(struct bw_hwmon *hw, unsigned int sample_ms) +{ + return __set_hw_events(hw, sample_ms, MON3); +} + +static irqreturn_t +__bwmon_intr_handler(int irq, void *dev, enum mon_reg_type type) +{ + struct bwmon *m = dev; + + m->intr_status = mon_irq_status(m, type); + if (!m->intr_status) + return IRQ_NONE; + + if (bw_hwmon_sample_end(&m->hw) > 0) + return IRQ_WAKE_THREAD; + + return IRQ_HANDLED; +} + +static irqreturn_t bwmon_intr_handler(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON1); +} + +static irqreturn_t bwmon_intr_handler2(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON2); +} + +static irqreturn_t bwmon_intr_handler3(int irq, void *dev) +{ + return __bwmon_intr_handler(irq, dev, MON3); +} + +static irqreturn_t bwmon_intr_thread(int irq, void *dev) +{ + struct bwmon *m = dev; + + update_bw_hwmon(&m->hw); + return IRQ_HANDLED; +} + +static __always_inline +void mon_set_byte_count_filter(struct bwmon *m, enum mon_reg_type type) +{ + if (!m->byte_mask) + return; + + switch (type) { + case MON1: + case MON2: + writel_relaxed(m->byte_mask, MON_MASK(m)); + writel_relaxed(m->byte_match, MON_MATCH(m)); + break; + case MON3: + writel_relaxed(m->byte_mask, MON3_MASK(m)); + writel_relaxed(m->byte_match, MON3_MATCH(m)); + break; + } +} + +static __always_inline int __start_bw_hwmon(struct bw_hwmon *hw, + unsigned long mbps, enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + u32 limit, zone_actions; + int ret; + irq_handler_t handler; + + switch (type) { + case MON1: + handler = bwmon_intr_handler; + limit = mbps_to_bytes(mbps, hw->df->profile->polling_ms, 0); + break; + case MON2: + zone_actions = calc_zone_actions(); + handler = bwmon_intr_handler2; + break; + case MON3: + zone_actions = calc_zone_actions(); + handler = bwmon_intr_handler3; + break; + } + + ret = request_threaded_irq(m->irq, handler, bwmon_intr_thread, + IRQF_ONESHOT | IRQF_SHARED, + dev_name(m->dev), m); + if (ret < 0) { + dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", + ret); + return ret; + } + + mon_disable(m, type); + + mon_clear(m, false, type); + + switch (type) { + case MON1: + mon_set_limit(m, limit); + break; + case MON2: + mon_set_zones(m, hw->df->profile->polling_ms, type); + /* Set the zone actions to increment appropriate counters */ + writel_relaxed(zone_actions, MON2_ZONE_ACTIONS(m)); + break; + case MON3: + mon_set_zones(m, hw->df->profile->polling_ms, type); + /* Set the zone actions to increment appropriate counters */ + writel_relaxed(zone_actions, MON3_ZONE_ACTIONS(m)); + } + + mon_set_byte_count_filter(m, type); + mon_irq_clear(m, type); + mon_irq_enable(m, type); + mon_enable(m, type); + + return 0; +} + +static int start_bw_hwmon(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON1); +} + +static int start_bw_hwmon2(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON2); +} + +static int start_bw_hwmon3(struct bw_hwmon *hw, unsigned long mbps) +{ + return __start_bw_hwmon(hw, mbps, MON3); +} + +static __always_inline +void __stop_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) { struct bwmon *m = to_bwmon(hw); + mon_irq_disable(m, type); free_irq(m->irq, m); - mon_disable(m); - mon_irq_disable(m); - mon_clear(m); - mon_irq_clear(m); + mon_disable(m, type); + mon_clear(m, true, type); + mon_irq_clear(m, type); +} + +static void stop_bw_hwmon(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON1); +} + +static void stop_bw_hwmon2(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON2); +} + +static void stop_bw_hwmon3(struct bw_hwmon *hw) +{ + return __stop_bw_hwmon(hw, MON3); +} + +static __always_inline +int __suspend_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + + mon_irq_disable(m, type); + free_irq(m->irq, m); + mon_disable(m, type); + mon_irq_clear(m, type); + + return 0; } static int suspend_bw_hwmon(struct bw_hwmon *hw) { - struct bwmon *m = to_bwmon(hw); + return __suspend_bw_hwmon(hw, MON1); +} - free_irq(m->irq, m); - mon_disable(m); - mon_irq_disable(m); - mon_irq_clear(m); +static int suspend_bw_hwmon2(struct bw_hwmon *hw) +{ + return __suspend_bw_hwmon(hw, MON2); +} + +static int suspend_bw_hwmon3(struct bw_hwmon *hw) +{ + return __suspend_bw_hwmon(hw, MON3); +} + +static __always_inline +int __resume_bw_hwmon(struct bw_hwmon *hw, enum mon_reg_type type) +{ + struct bwmon *m = to_bwmon(hw); + int ret; + irq_handler_t handler; + + switch (type) { + case MON1: + handler = bwmon_intr_handler; + break; + case MON2: + handler = bwmon_intr_handler2; + break; + case MON3: + handler = bwmon_intr_handler3; + break; + } + + mon_clear(m, false, type); + ret = request_threaded_irq(m->irq, handler, bwmon_intr_thread, + IRQF_ONESHOT | IRQF_SHARED, + dev_name(m->dev), m); + if (ret < 0) { + dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", + ret); + return ret; + } + + mon_irq_enable(m, type); + mon_enable(m, type); return 0; } static int resume_bw_hwmon(struct bw_hwmon *hw) { - struct bwmon *m = to_bwmon(hw); - int ret; + return __resume_bw_hwmon(hw, MON1); +} - mon_clear(m); - mon_irq_enable(m); - mon_enable(m); - ret = request_threaded_irq(m->irq, NULL, bwmon_intr_handler, - IRQF_ONESHOT | IRQF_SHARED, - dev_name(m->dev), m); - if (ret < 0) { - dev_err(m->dev, "Unable to register interrupt handler! (%d)\n", - ret); - return ret; - } +static int resume_bw_hwmon2(struct bw_hwmon *hw) +{ + return __resume_bw_hwmon(hw, MON2); +} - return 0; +static int resume_bw_hwmon3(struct bw_hwmon *hw) +{ + return __resume_bw_hwmon(hw, MON3); } /*************************************************************************/ +static const struct bwmon_spec spec[] = { + [0] = { + .wrap_on_thres = true, + .overflow = false, + .throt_adj = false, + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, + }, + [1] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = false, + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, + }, + [2] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = true, + .hw_sampling = false, + .has_global_base = true, + .reg_type = MON1, + }, + [3] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = true, + .hw_sampling = true, + .has_global_base = true, + .reg_type = MON2, + }, + [4] = { + .wrap_on_thres = false, + .overflow = true, + .throt_adj = false, + .hw_sampling = true, + .reg_type = MON3, + }, +}; + +static const struct of_device_id bimc_bwmon_match_table[] = { + { .compatible = "qcom,bimc-bwmon", .data = &spec[0] }, + { .compatible = "qcom,bimc-bwmon2", .data = &spec[1] }, + { .compatible = "qcom,bimc-bwmon3", .data = &spec[2] }, + { .compatible = "qcom,bimc-bwmon4", .data = &spec[3] }, + { .compatible = "qcom,bimc-bwmon5", .data = &spec[4] }, + {} +}; + static int bimc_bwmon_driver_probe(struct platform_device *pdev) { struct device *dev = &pdev->dev; struct resource *res; struct bwmon *m; int ret; - u32 data; + u32 data, count_unit; m = devm_kzalloc(dev, sizeof(*m), GFP_KERNEL); if (!m) return -ENOMEM; m->dev = dev; - ret = of_property_read_u32(dev->of_node, "qcom,mport", &data); - if (ret < 0) { - dev_err(dev, "mport not found! (%d)\n", ret); - return ret; + m->spec = of_device_get_match_data(dev); + if (!m->spec) { + dev_err(dev, "Unknown device type!\n"); + return -ENODEV; } - m->mport = data; res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "base"); if (!res) { @@ -304,15 +1020,26 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return -ENOMEM; } - res = platform_get_resource_byname(pdev, IORESOURCE_MEM, "global_base"); - if (!res) { - dev_err(dev, "global_base not found!\n"); - return -EINVAL; - } - m->global_base = devm_ioremap(dev, res->start, resource_size(res)); - if (!m->global_base) { - dev_err(dev, "Unable map global_base!\n"); - return -ENOMEM; + if (m->spec->has_global_base) { + res = platform_get_resource_byname(pdev, IORESOURCE_MEM, + "global_base"); + if (!res) { + dev_err(dev, "global_base not found!\n"); + return -EINVAL; + } + m->global_base = devm_ioremap(dev, res->start, + resource_size(res)); + if (!m->global_base) { + dev_err(dev, "Unable map global_base!\n"); + return -ENOMEM; + } + + ret = of_property_read_u32(dev->of_node, "qcom,mport", &data); + if (ret < 0) { + dev_err(dev, "mport not found! (%d)\n", ret); + return ret; + } + m->mport = data; } m->irq = platform_get_irq(pdev, 0); @@ -324,11 +1051,57 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) m->hw.of_node = of_parse_phandle(dev->of_node, "qcom,target-dev", 0); if (!m->hw.of_node) return -EINVAL; - m->hw.start_hwmon = &start_bw_hwmon; - m->hw.stop_hwmon = &stop_bw_hwmon; - m->hw.suspend_hwmon = &suspend_bw_hwmon; - m->hw.resume_hwmon = &resume_bw_hwmon; - m->hw.meas_bw_and_set_irq = &meas_bw_and_set_irq; + + if (m->spec->hw_sampling) { + ret = of_property_read_u32(dev->of_node, "qcom,hw-timer-hz", + &m->hw_timer_hz); + if (ret < 0) { + dev_err(dev, "HW sampling rate not specified!\n"); + return ret; + } + } + + if (of_property_read_u32(dev->of_node, "qcom,count-unit", &count_unit)) + count_unit = SZ_1M; + m->count_shift = order_base_2(count_unit); + m->thres_lim = THRES_LIM(m->count_shift); + + switch (m->spec->reg_type) { + case MON3: + m->hw.start_hwmon = start_bw_hwmon3; + m->hw.stop_hwmon = stop_bw_hwmon3; + m->hw.suspend_hwmon = suspend_bw_hwmon3; + m->hw.resume_hwmon = resume_bw_hwmon3; + m->hw.get_bytes_and_clear = get_bytes_and_clear3; + m->hw.set_hw_events = set_hw_events3; + break; + case MON2: + m->hw.start_hwmon = start_bw_hwmon2; + m->hw.stop_hwmon = stop_bw_hwmon2; + m->hw.suspend_hwmon = suspend_bw_hwmon2; + m->hw.resume_hwmon = resume_bw_hwmon2; + m->hw.get_bytes_and_clear = get_bytes_and_clear2; + m->hw.set_hw_events = set_hw_events; + break; + case MON1: + m->hw.start_hwmon = start_bw_hwmon; + m->hw.stop_hwmon = stop_bw_hwmon; + m->hw.suspend_hwmon = suspend_bw_hwmon; + m->hw.resume_hwmon = resume_bw_hwmon; + m->hw.get_bytes_and_clear = get_bytes_and_clear; + m->hw.set_thres = set_thres; + break; + } + + of_property_read_u32(dev->of_node, "qcom,byte-mid-match", + &m->byte_match); + of_property_read_u32(dev->of_node, "qcom,byte-mid-mask", + &m->byte_mask); + + if (m->spec->throt_adj) { + m->hw.set_throttle_adj = mon_set_throttle_adj; + m->hw.get_throttle_adj = mon_get_throttle_adj; + } ret = register_bw_hwmon(dev, &m->hw); if (ret < 0) { @@ -339,11 +1112,6 @@ static int bimc_bwmon_driver_probe(struct platform_device *pdev) return 0; } -static const struct of_device_id bimc_bwmon_match_table[] = { - { .compatible = "qcom,bimc-bwmon" }, - {} -}; - static struct platform_driver bimc_bwmon_driver = { .probe = bimc_bwmon_driver_probe, .driver = { diff --git a/drivers/devfreq/devfreq_simple_dev.c b/drivers/devfreq/devfreq_simple_dev.c index 076055757268..c5431755a176 100644 --- a/drivers/devfreq/devfreq_simple_dev.c +++ b/drivers/devfreq/devfreq_simple_dev.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "devfreq-simple-dev: " fmt @@ -26,6 +26,7 @@ struct dev_data { struct clk *clk; struct devfreq *df; struct devfreq_dev_profile profile; + bool freq_in_khz; }; static void find_freq(struct devfreq_dev_profile *p, unsigned long *freq, @@ -57,7 +58,7 @@ static int dev_target(struct device *dev, unsigned long *freq, u32 flags) find_freq(&d->profile, freq, flags); - rfreq = clk_round_rate(d->clk, *freq * 1000); + rfreq = clk_round_rate(d->clk, d->freq_in_khz ? *freq * 1000 : *freq); if (IS_ERR_VALUE(rfreq)) { dev_err(dev, "devfreq: Cannot find matching frequency for %lu\n", *freq); @@ -75,39 +76,30 @@ static int dev_get_cur_freq(struct device *dev, unsigned long *freq) f = clk_get_rate(d->clk); if (IS_ERR_VALUE(f)) return f; - *freq = f / 1000; + *freq = d->freq_in_khz ? f / 1000 : f; return 0; } #define PROP_TBL "freq-tbl-khz" -static int devfreq_clock_probe(struct platform_device *pdev) +static int parse_freq_table(struct device *dev, struct dev_data *d) { - struct device *dev = &pdev->dev; - struct dev_data *d; - struct devfreq_dev_profile *p; - u32 *data, poll; - const char *gov_name; + struct devfreq_dev_profile *p = &d->profile; int ret, len, i, j; + u32 *data; unsigned long f; - d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); - if (!d) - return -ENOMEM; - platform_set_drvdata(pdev, d); - - d->clk = devm_clk_get(dev, "devfreq_clk"); - if (IS_ERR(d->clk)) - return PTR_ERR(d->clk); - - if (!of_find_property(dev->of_node, PROP_TBL, &len)) - return -EINVAL; + if (!of_find_property(dev->of_node, PROP_TBL, &len)) { + if (dev_pm_opp_get_opp_count(dev) <= 0) + return -EPROBE_DEFER; + return 0; + } + d->freq_in_khz = true; len /= sizeof(*data); data = devm_kzalloc(dev, len * sizeof(*data), GFP_KERNEL); if (!data) return -ENOMEM; - p = &d->profile; p->freq_table = devm_kzalloc(dev, len * sizeof(*p->freq_table), GFP_KERNEL); if (!p->freq_table) @@ -134,6 +126,32 @@ static int devfreq_clock_probe(struct platform_device *pdev) return -EINVAL; } + return 0; +} + +static int devfreq_clock_probe(struct platform_device *pdev) +{ + struct device *dev = &pdev->dev; + struct dev_data *d; + struct devfreq_dev_profile *p; + u32 poll; + const char *gov_name; + int ret; + + d = devm_kzalloc(dev, sizeof(*d), GFP_KERNEL); + if (!d) + return -ENOMEM; + platform_set_drvdata(pdev, d); + + d->clk = devm_clk_get(dev, "devfreq_clk"); + if (IS_ERR(d->clk)) + return PTR_ERR(d->clk); + + ret = parse_freq_table(dev, d); + if (ret < 0) + return ret; + + p = &d->profile; p->target = dev_target; p->get_cur_freq = dev_get_cur_freq; ret = dev_get_cur_freq(dev, &p->initial_freq); @@ -147,11 +165,23 @@ static int devfreq_clock_probe(struct platform_device *pdev) if (of_property_read_string(dev->of_node, "governor", &gov_name)) gov_name = "performance"; + if (of_property_read_bool(dev->of_node, "qcom,prepare-clk")) { + ret = clk_prepare(d->clk); + if (ret < 0) + return ret; + } + d->df = devfreq_add_device(dev, p, gov_name, NULL); - if (IS_ERR(d->df)) - return PTR_ERR_OR_ZERO(d->df); + if (IS_ERR(d->df)) { + ret = PTR_ERR_OR_ZERO(d->df); + goto add_err; + } return 0; +add_err: + if (of_property_read_bool(dev->of_node, "qcom,prepare-clk")) + clk_unprepare(d->clk); + return ret; } static int devfreq_clock_remove(struct platform_device *pdev) diff --git a/drivers/devfreq/governor_bw_hwmon.c b/drivers/devfreq/governor_bw_hwmon.c index 1a5a5ba9bd0d..77a91e8f65ca 100644 --- a/drivers/devfreq/governor_bw_hwmon.c +++ b/drivers/devfreq/governor_bw_hwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2013-2015, 2019, The Linux Foundation. All rights reserved. + * Copyright (c) 2013-2017, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "bw-hwmon: " fmt @@ -17,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -24,17 +25,41 @@ #include "governor.h" #include "governor_bw_hwmon.h" +#define NUM_MBPS_ZONES 10 struct hwmon_node { - unsigned int tolerance_percent; unsigned int guard_band_mbps; unsigned int decay_rate; unsigned int io_percent; unsigned int bw_step; + unsigned int sample_ms; + unsigned int up_scale; + unsigned int up_thres; + unsigned int down_thres; + unsigned int down_count; + unsigned int hist_memory; + unsigned int hyst_trigger_count; + unsigned int hyst_length; + unsigned int idle_mbps; + unsigned int mbps_zones[NUM_MBPS_ZONES]; + unsigned long prev_ab; unsigned long *dev_ab; unsigned long resume_freq; unsigned long resume_ab; + unsigned long bytes; + unsigned long max_mbps; + unsigned long hist_max_mbps; + unsigned long hist_mem; + unsigned long hyst_peak; + unsigned long hyst_mbps; + unsigned long hyst_trig_win; + unsigned long hyst_en; + unsigned long prev_req; + unsigned int wake; + unsigned int down_cnt; ktime_t prev_ts; + ktime_t hist_max_ts; + bool sampled; bool mon_started; struct list_head list; void *orig_data; @@ -43,6 +68,10 @@ struct hwmon_node { struct attribute_group *attr_grp; }; +#define UP_WAKE 1 +#define DOWN_WAKE 2 +static DEFINE_SPINLOCK(irq_lock); + static LIST_HEAD(hwmon_list); static DEFINE_MUTEX(list_lock); @@ -76,55 +105,344 @@ static ssize_t name##_store(struct device *dev, \ return count; \ } +#define show_list_attr(name, n) \ +static ssize_t name##_show(struct device *dev, \ + struct device_attribute *attr, char *buf) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct hwmon_node *hw = df->data; \ + unsigned int i, cnt = 0; \ + \ + for (i = 0; i < n && hw->name[i]; i++) \ + cnt += scnprintf(buf + cnt, PAGE_SIZE, "%u ", hw->name[i]);\ + cnt += scnprintf(buf + cnt, PAGE_SIZE, "\n"); \ + return cnt; \ +} + +#define store_list_attr(name, n, _min, _max) \ +static ssize_t name##_store(struct device *dev, \ + struct device_attribute *attr, const char *buf, \ + size_t count) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct hwmon_node *hw = df->data; \ + int ret, numvals; \ + unsigned int i = 0, val; \ + char **strlist; \ + \ + strlist = argv_split(GFP_KERNEL, buf, &numvals); \ + if (!strlist) \ + return -ENOMEM; \ + numvals = min(numvals, n - 1); \ + for (i = 0; i < numvals; i++) { \ + ret = kstrtouint(strlist[i], 10, &val); \ + if (ret < 0) \ + goto out; \ + val = max(val, _min); \ + val = min(val, _max); \ + hw->name[i] = val; \ + } \ + ret = count; \ +out: \ + argv_free(strlist); \ + hw->name[i] = 0; \ + return ret; \ +} + #define MIN_MS 10U #define MAX_MS 500U -static unsigned long measure_bw_and_set_irq(struct hwmon_node *node) +/* Returns MBps of read/writes for the sampling window. */ +static unsigned int bytes_to_mbps(long long bytes, unsigned int us) { - ktime_t ts; - unsigned int us; - unsigned long mbps; - struct bw_hwmon *hw = node->hw; - - /* - * Since we are stopping the counters, we don't want this short work - * to be interrupted by other tasks and cause the measurements to be - * wrong. Not blocking interrupts to avoid affecting interrupt - * latency and since they should be short anyway because they run in - * atomic context. - */ - preempt_disable(); - - ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, node->prev_ts)); - if (!us) - us = 1; - - mbps = hw->meas_bw_and_set_irq(hw, node->tolerance_percent, us); - node->prev_ts = ts; - - preempt_enable(); - - dev_dbg(hw->df->dev.parent, "BW MBps = %6lu, period = %u\n", mbps, us); - trace_bw_hwmon_meas(dev_name(hw->df->dev.parent), - mbps, - us, - 0); + bytes *= USEC_PER_SEC; + do_div(bytes, us); + bytes = DIV_ROUND_UP_ULL(bytes, SZ_1M); + return bytes; +} +static unsigned int mbps_to_bytes(unsigned long mbps, unsigned int ms) +{ + mbps *= ms; + mbps = DIV_ROUND_UP(mbps, MSEC_PER_SEC); + mbps *= SZ_1M; return mbps; } -static void compute_bw(struct hwmon_node *node, int mbps, - unsigned long *freq, unsigned long *ab) +static int __bw_hwmon_sw_sample_end(struct bw_hwmon *hwmon) { - int new_bw; + struct devfreq *df; + struct hwmon_node *node; + ktime_t ts; + unsigned long bytes, mbps; + unsigned int us; + int wake = 0; - mbps += node->guard_band_mbps; + df = hwmon->df; + node = df->data; - if (mbps > node->prev_ab) { - new_bw = mbps; + ts = ktime_get(); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); + + bytes = hwmon->get_bytes_and_clear(hwmon); + bytes += node->bytes; + node->bytes = 0; + + mbps = bytes_to_mbps(bytes, us); + node->max_mbps = max(node->max_mbps, mbps); + + /* + * If the measured bandwidth in a micro sample is greater than the + * wake up threshold, it indicates an increase in load that's non + * trivial. So, have the governor ignore historical idle time or low + * bandwidth usage and do the bandwidth calculation based on just + * this micro sample. + */ + if (mbps > node->hw->up_wake_mbps) { + wake = UP_WAKE; + } else if (mbps < node->hw->down_wake_mbps) { + if (node->down_cnt) + node->down_cnt--; + if (node->down_cnt <= 0) + wake = DOWN_WAKE; + } + + node->prev_ts = ts; + node->wake = wake; + node->sampled = true; + + trace_bw_hwmon_meas(dev_name(df->dev.parent), + mbps, + us, + wake); + + return wake; +} + +static int __bw_hwmon_hw_sample_end(struct bw_hwmon *hwmon) +{ + struct devfreq *df; + struct hwmon_node *node; + unsigned long bytes, mbps; + int wake = 0; + + df = hwmon->df; + node = df->data; + + /* + * If this read is in response to an IRQ, the HW monitor should + * return the measurement in the micro sample that triggered the IRQ. + * Otherwise, it should return the maximum measured value in any + * micro sample since the last time we called get_bytes_and_clear() + */ + bytes = hwmon->get_bytes_and_clear(hwmon); + mbps = bytes_to_mbps(bytes, node->sample_ms * USEC_PER_MSEC); + node->max_mbps = mbps; + + if (mbps > node->hw->up_wake_mbps) + wake = UP_WAKE; + else if (mbps < node->hw->down_wake_mbps) + wake = DOWN_WAKE; + + node->wake = wake; + node->sampled = true; + + trace_bw_hwmon_meas(dev_name(df->dev.parent), + mbps, + node->sample_ms * USEC_PER_MSEC, + wake); + + return 1; +} + +static int __bw_hwmon_sample_end(struct bw_hwmon *hwmon) +{ + if (hwmon->set_hw_events) + return __bw_hwmon_hw_sample_end(hwmon); + else + return __bw_hwmon_sw_sample_end(hwmon); +} + +int bw_hwmon_sample_end(struct bw_hwmon *hwmon) +{ + unsigned long flags; + int wake; + + spin_lock_irqsave(&irq_lock, flags); + wake = __bw_hwmon_sample_end(hwmon); + spin_unlock_irqrestore(&irq_lock, flags); + + return wake; +} + +static unsigned long to_mbps_zone(struct hwmon_node *node, unsigned long mbps) +{ + int i; + + for (i = 0; i < NUM_MBPS_ZONES && node->mbps_zones[i]; i++) + if (node->mbps_zones[i] >= mbps) + return node->mbps_zones[i]; + + return node->hw->df->max_freq; +} + +#define MIN_MBPS 500UL +#define HIST_PEAK_TOL 60 +static unsigned long get_bw_and_set_irq(struct hwmon_node *node, + unsigned long *freq, unsigned long *ab) +{ + unsigned long meas_mbps, thres, flags, req_mbps, adj_mbps; + unsigned long meas_mbps_zone; + unsigned long hist_lo_tol, hyst_lo_tol; + struct bw_hwmon *hw = node->hw; + unsigned int new_bw, io_percent = node->io_percent; + ktime_t ts; + unsigned int ms = 0; + + spin_lock_irqsave(&irq_lock, flags); + + if (!hw->set_hw_events) { + ts = ktime_get(); + ms = ktime_to_ms(ktime_sub(ts, node->prev_ts)); + } + if (!node->sampled || ms >= node->sample_ms) + __bw_hwmon_sample_end(node->hw); + node->sampled = false; + + req_mbps = meas_mbps = node->max_mbps; + node->max_mbps = 0; + + hist_lo_tol = (node->hist_max_mbps * HIST_PEAK_TOL) / 100; + /* Remember historic peak in the past hist_mem decision windows. */ + if (meas_mbps > node->hist_max_mbps || !node->hist_mem) { + /* If new max or no history */ + node->hist_max_mbps = meas_mbps; + node->hist_mem = node->hist_memory; + } else if (meas_mbps >= hist_lo_tol) { + /* + * If subsequent peaks come close (within tolerance) to but + * less than the historic peak, then reset the history start, + * but not the peak value. + */ + node->hist_mem = node->hist_memory; } else { - new_bw = mbps * node->decay_rate + /* Count down history expiration. */ + if (node->hist_mem) + node->hist_mem--; + } + + /* + * The AB value that corresponds to the lowest mbps zone greater than + * or equal to the "frequency" the current measurement will pick. + * This upper limit is useful for balancing out any prediction + * mechanisms to be power friendly. + */ + meas_mbps_zone = (meas_mbps * 100) / io_percent; + meas_mbps_zone = to_mbps_zone(node, meas_mbps_zone); + meas_mbps_zone = (meas_mbps_zone * io_percent) / 100; + meas_mbps_zone = max(meas_mbps, meas_mbps_zone); + + /* + * If this is a wake up due to BW increase, vote much higher BW than + * what we measure to stay ahead of increasing traffic and then set + * it up to vote for measured BW if we see down_count short sample + * windows of low traffic. + */ + if (node->wake == UP_WAKE) { + req_mbps += ((meas_mbps - node->prev_req) + * node->up_scale) / 100; + /* + * However if the measured load is less than the historic + * peak, but the over request is higher than the historic + * peak, then we could limit the over requesting to the + * historic peak. + */ + if (req_mbps > node->hist_max_mbps + && meas_mbps < node->hist_max_mbps) + req_mbps = node->hist_max_mbps; + + req_mbps = min(req_mbps, meas_mbps_zone); + } + + hyst_lo_tol = (node->hyst_mbps * HIST_PEAK_TOL) / 100; + if (meas_mbps > node->hyst_mbps && meas_mbps > MIN_MBPS) { + hyst_lo_tol = (meas_mbps * HIST_PEAK_TOL) / 100; + node->hyst_peak = 0; + node->hyst_trig_win = node->hyst_length; + node->hyst_mbps = meas_mbps; + } + + /* + * Check node->max_mbps to avoid double counting peaks that cause + * early termination of a window. + */ + if (meas_mbps >= hyst_lo_tol && meas_mbps > MIN_MBPS + && !node->max_mbps) { + node->hyst_peak++; + if (node->hyst_peak >= node->hyst_trigger_count + || node->hyst_en) + node->hyst_en = node->hyst_length; + } + + if (node->hyst_trig_win) + node->hyst_trig_win--; + if (node->hyst_en) + node->hyst_en--; + + if (!node->hyst_trig_win && !node->hyst_en) { + node->hyst_peak = 0; + node->hyst_mbps = 0; + } + + if (node->hyst_en) { + if (meas_mbps > node->idle_mbps) + req_mbps = max(req_mbps, node->hyst_mbps); + } + + /* Stretch the short sample window size, if the traffic is too low */ + if (meas_mbps < MIN_MBPS) { + hw->up_wake_mbps = (max(MIN_MBPS, req_mbps) + * (100 + node->up_thres)) / 100; + hw->down_wake_mbps = 0; + hw->undo_over_req_mbps = 0; + thres = mbps_to_bytes(max(MIN_MBPS, req_mbps / 2), + node->sample_ms); + } else { + /* + * Up wake vs down wake are intentionally a percentage of + * req_mbps vs meas_mbps to make sure the over requesting + * phase is handled properly. We only want to wake up and + * reduce the vote based on the measured mbps being less than + * the previous measurement that caused the "over request". + */ + hw->up_wake_mbps = (req_mbps * (100 + node->up_thres)) / 100; + hw->down_wake_mbps = (meas_mbps * node->down_thres) / 100; + if (node->wake == UP_WAKE) + hw->undo_over_req_mbps = min(req_mbps, meas_mbps_zone); + else + hw->undo_over_req_mbps = 0; + thres = mbps_to_bytes(meas_mbps, node->sample_ms); + } + + if (hw->set_hw_events) { + hw->down_cnt = node->down_count; + hw->set_hw_events(hw, node->sample_ms); + } else { + node->down_cnt = node->down_count; + node->bytes = hw->set_thres(hw, thres); + } + + node->wake = 0; + node->prev_req = req_mbps; + + spin_unlock_irqrestore(&irq_lock, flags); + + adj_mbps = req_mbps + node->guard_band_mbps; + + if (adj_mbps > node->prev_ab) { + new_bw = adj_mbps; + } else { + new_bw = adj_mbps * node->decay_rate + node->prev_ab * (100 - node->decay_rate); new_bw /= 100; } @@ -132,12 +450,14 @@ static void compute_bw(struct hwmon_node *node, int mbps, node->prev_ab = new_bw; if (ab) *ab = roundup(new_bw, node->bw_step); - *freq = (new_bw * 100) / node->io_percent; + + *freq = (new_bw * 100) / io_percent; trace_bw_hwmon_update(dev_name(node->hw->df->dev.parent), new_bw, *freq, - 0, - 0); + hw->up_wake_mbps, + hw->down_wake_mbps); + return req_mbps; } static struct hwmon_node *find_hwmon_node(struct devfreq *df) @@ -158,13 +478,10 @@ static struct hwmon_node *find_hwmon_node(struct devfreq *df) return found; } -#define TOO_SOON_US (1 * USEC_PER_MSEC) int update_bw_hwmon(struct bw_hwmon *hwmon) { struct devfreq *df; struct hwmon_node *node; - ktime_t ts; - unsigned int us; int ret; if (!hwmon) @@ -172,7 +489,7 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) df = hwmon->df; if (!df) return -ENODEV; - node = find_hwmon_node(df); + node = df->data; if (!node) return -ENODEV; @@ -182,26 +499,12 @@ int update_bw_hwmon(struct bw_hwmon *hwmon) dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); - /* - * Don't recalc bandwidth if the interrupt comes right after a - * previous bandwidth calculation. This is done for two reasons: - * - * 1. Sampling the BW during a very short duration can result in a - * very inaccurate measurement due to very short bursts. - * 2. This can only happen if the limit was hit very close to the end - * of the previous sample period. Which means the current BW - * estimate is not very off and doesn't need to be readjusted. - */ - ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, node->prev_ts)); - if (us > TOO_SOON_US) { - mutex_lock(&df->lock); - ret = update_devfreq(df); - if (ret < 0) - dev_err(df->dev.parent, - "Unable to update freq on request: %d\n", ret); - mutex_unlock(&df->lock); - } + mutex_lock(&df->lock); + ret = update_devfreq(df); + if (ret < 0) + dev_err(df->dev.parent, + "Unable to update freq on request! (%d)\n", ret); + mutex_unlock(&df->lock); devfreq_monitor_start(df); @@ -223,6 +526,9 @@ static int start_monitor(struct devfreq *df, bool init) node->resume_freq = 0; node->resume_ab = 0; mbps = (df->previous_freq * node->io_percent) / 100; + hw->up_wake_mbps = mbps; + hw->down_wake_mbps = MIN_MBPS; + hw->undo_over_req_mbps = 0; ret = hw->start_hwmon(hw, mbps); } else { ret = hw->resume_hwmon(hw); @@ -380,7 +686,6 @@ static int gov_resume(struct devfreq *df) static int devfreq_bw_hwmon_get_freq(struct devfreq *df, unsigned long *freq) { - unsigned long mbps; struct hwmon_node *node = df->data; /* Suspend/resume sequence */ @@ -390,15 +695,51 @@ static int devfreq_bw_hwmon_get_freq(struct devfreq *df, return 0; } - mbps = measure_bw_and_set_irq(node); - compute_bw(node, mbps, freq, node->dev_ab); + get_bw_and_set_irq(node, freq, node->dev_ab); return 0; } -show_attr(tolerance_percent); -store_attr(tolerance_percent, 0U, 30U); -static DEVICE_ATTR_RW(tolerance_percent); +static ssize_t throttle_adj_store(struct device *dev, + struct device_attribute *attr, const char *buf, size_t count) +{ + struct devfreq *df = to_devfreq(dev); + struct hwmon_node *node = df->data; + int ret; + unsigned int val; + + if (!node->hw->set_throttle_adj) + return -EPERM; + + ret = kstrtouint(buf, 10, &val); + if (ret < 0) + return ret; + + ret = node->hw->set_throttle_adj(node->hw, val); + + if (!ret) + return count; + else + return ret; +} + +static ssize_t throttle_adj_show(struct device *dev, + struct device_attribute *attr, char *buf) +{ + struct devfreq *df = to_devfreq(dev); + struct hwmon_node *node = df->data; + unsigned int val; + + if (!node->hw->get_throttle_adj) + val = 0; + else + val = node->hw->get_throttle_adj(node->hw); + + return snprintf(buf, PAGE_SIZE, "%u\n", val); +} + +static DEVICE_ATTR_RW(throttle_adj); + show_attr(guard_band_mbps); store_attr(guard_band_mbps, 0U, 2000U); static DEVICE_ATTR_RW(guard_band_mbps); @@ -411,13 +752,53 @@ static DEVICE_ATTR_RW(io_percent); show_attr(bw_step); store_attr(bw_step, 50U, 1000U); static DEVICE_ATTR_RW(bw_step); +show_attr(sample_ms); +store_attr(sample_ms, 1U, 50U); +static DEVICE_ATTR_RW(sample_ms); +show_attr(up_scale); +store_attr(up_scale, 0U, 500U); +static DEVICE_ATTR_RW(up_scale); +show_attr(up_thres); +store_attr(up_thres, 1U, 100U); +static DEVICE_ATTR_RW(up_thres); +show_attr(down_thres); +store_attr(down_thres, 0U, 90U); +static DEVICE_ATTR_RW(down_thres); +show_attr(down_count); +store_attr(down_count, 0U, 90U); +static DEVICE_ATTR_RW(down_count); +show_attr(hist_memory); +store_attr(hist_memory, 0U, 90U); +static DEVICE_ATTR_RW(hist_memory); +show_attr(hyst_trigger_count); +store_attr(hyst_trigger_count, 0U, 90U); +static DEVICE_ATTR_RW(hyst_trigger_count); +show_attr(hyst_length); +store_attr(hyst_length, 0U, 90U); +static DEVICE_ATTR_RW(hyst_length); +show_attr(idle_mbps); +store_attr(idle_mbps, 0U, 2000U); +static DEVICE_ATTR_RW(idle_mbps); +show_list_attr(mbps_zones, NUM_MBPS_ZONES); +store_list_attr(mbps_zones, NUM_MBPS_ZONES, 0U, UINT_MAX); +static DEVICE_ATTR_RW(mbps_zones); static struct attribute *dev_attr[] = { - &dev_attr_tolerance_percent.attr, &dev_attr_guard_band_mbps.attr, &dev_attr_decay_rate.attr, &dev_attr_io_percent.attr, &dev_attr_bw_step.attr, + &dev_attr_sample_ms.attr, + &dev_attr_up_scale.attr, + &dev_attr_up_thres.attr, + &dev_attr_down_thres.attr, + &dev_attr_down_count.attr, + &dev_attr_hist_memory.attr, + &dev_attr_hyst_trigger_count.attr, + &dev_attr_hyst_length.attr, + &dev_attr_idle_mbps.attr, + &dev_attr_mbps_zones.attr, + &dev_attr_throttle_adj.attr, NULL, }; @@ -429,8 +810,12 @@ static struct attribute_group dev_attr_group = { static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, unsigned int event, void *data) { - int ret; + int ret = 0; unsigned int sample_ms; + struct hwmon_node *node; + struct bw_hwmon *hw; + + mutex_lock(&state_lock); switch (event) { case DEVFREQ_GOV_START: @@ -441,7 +826,7 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, ret = gov_start(df); if (ret < 0) - return ret; + goto out; dev_dbg(df->dev.parent, "Enabled dev BW HW monitor governor\n"); @@ -455,7 +840,22 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, sample_ms = *(unsigned int *)data; sample_ms = max(MIN_MS, sample_ms); sample_ms = min(MAX_MS, sample_ms); + /* + * Suspend/resume the HW monitor around the interval update + * to prevent the HW monitor IRQ from trying to change + * stop/start the delayed workqueue while the interval update + * is happening. + */ + node = df->data; + hw = node->hw; + hw->suspend_hwmon(hw); devfreq_interval_update(df, &sample_ms); + ret = hw->resume_hwmon(hw); + if (ret < 0) { + dev_err(df->dev.parent, + "Unable to resume HW monitor (%d)\n", ret); + goto out; + } break; case DEVFREQ_GOV_SUSPEND: @@ -464,7 +864,7 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, dev_err(df->dev.parent, "Unable to suspend BW HW mon governor (%d)\n", ret); - return ret; + goto out; } dev_dbg(df->dev.parent, "Suspended BW HW mon governor\n"); @@ -476,14 +876,17 @@ static int devfreq_bw_hwmon_ev_handler(struct devfreq *df, dev_err(df->dev.parent, "Unable to resume BW HW mon governor (%d)\n", ret); - return ret; + goto out; } dev_dbg(df->dev.parent, "Resumed BW HW mon governor\n"); break; } - return 0; +out: + mutex_unlock(&state_lock); + + return ret; } static struct devfreq_governor devfreq_gov_bw_hwmon = { @@ -522,11 +925,20 @@ int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon) node->attr_grp = &dev_attr_group; } - node->tolerance_percent = 10; node->guard_band_mbps = 100; node->decay_rate = 90; node->io_percent = 16; node->bw_step = 190; + node->sample_ms = 50; + node->up_scale = 0; + node->up_thres = 10; + node->down_thres = 0; + node->down_count = 3; + node->hist_memory = 0; + node->hyst_trigger_count = 3; + node->hyst_length = 0; + node->idle_mbps = 400; + node->mbps_zones[0] = 0; node->hw = hwmon; mutex_lock(&list_lock); diff --git a/drivers/devfreq/governor_bw_hwmon.h b/drivers/devfreq/governor_bw_hwmon.h index 0737bc5fe41f..576dafbc651f 100644 --- a/drivers/devfreq/governor_bw_hwmon.h +++ b/drivers/devfreq/governor_bw_hwmon.h @@ -13,13 +13,11 @@ * struct bw_hwmon - dev BW HW monitor info * @start_hwmon: Start the HW monitoring of the dev BW * @stop_hwmon: Stop the HW monitoring of dev BW - * @is_valid_irq: Check whether the IRQ was triggered by the - * counters used to monitor dev BW. - * @meas_bw_and_set_irq: Return the measured bandwidth and set up the - * IRQ to fire if the usage exceeds current - * measurement by @tol percent. - * @irq: IRQ number that corresponds to this HW - * monitor. + * @set_thres: Set the count threshold to generate an IRQ + * @get_bytes_and_clear: Get the bytes transferred since the last call + * and reset the counter to start over. + * @set_throttle_adj: Set throttle adjust field to the given value + * @get_throttle_adj: Get the value written to throttle adjust field * @dev: Pointer to device that this HW monitor can * monitor. * @of_node: OF node of device that this HW monitor can @@ -42,24 +40,39 @@ struct bw_hwmon { void (*stop_hwmon)(struct bw_hwmon *hw); int (*suspend_hwmon)(struct bw_hwmon *hw); int (*resume_hwmon)(struct bw_hwmon *hw); - unsigned long (*meas_bw_and_set_irq)(struct bw_hwmon *hw, - unsigned int tol, unsigned int us); + unsigned long (*set_thres)(struct bw_hwmon *hw, + unsigned long bytes); + unsigned long (*set_hw_events)(struct bw_hwmon *hw, + unsigned int sample_ms); + unsigned long (*get_bytes_and_clear)(struct bw_hwmon *hw); + int (*set_throttle_adj)(struct bw_hwmon *hw, + uint adj); + u32 (*get_throttle_adj)(struct bw_hwmon *hw); struct device *dev; struct device_node *of_node; struct devfreq_governor *gov; + unsigned long up_wake_mbps; + unsigned long undo_over_req_mbps; + unsigned long down_wake_mbps; + unsigned int down_cnt; struct devfreq *df; }; #ifdef CONFIG_DEVFREQ_GOV_QCOM_BW_HWMON int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon); int update_bw_hwmon(struct bw_hwmon *hwmon); +int bw_hwmon_sample_end(struct bw_hwmon *hwmon); #else static inline int register_bw_hwmon(struct device *dev, struct bw_hwmon *hwmon) { return 0; } -int update_bw_hwmon(struct bw_hwmon *hwmon) +static inline int update_bw_hwmon(struct bw_hwmon *hwmon) +{ + return 0; +} +static inline int bw_hwmon_sample_end(struct bw_hwmon *hwmon) { return 0; } diff --git a/drivers/devfreq/governor_cache_hwmon.c b/drivers/devfreq/governor_cache_hwmon.c index 86b02aba6450..fa4c0bce532f 100644 --- a/drivers/devfreq/governor_cache_hwmon.c +++ b/drivers/devfreq/governor_cache_hwmon.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only /* - * Copyright (c) 2014, 2019 The Linux Foundation. All rights reserved. + * Copyright (c) 2014-2015, 2019, The Linux Foundation. All rights reserved. */ #define pr_fmt(fmt) "cache-hwmon: " fmt @@ -20,14 +20,43 @@ #include #include #include +#include #include "governor.h" #include "governor_cache_hwmon.h" +struct cache_hwmon_node { + unsigned int cycles_per_low_req; + unsigned int cycles_per_med_req; + unsigned int cycles_per_high_req; + unsigned int min_busy; + unsigned int max_busy; + unsigned int tolerance_mrps; + unsigned int guard_band_mhz; + unsigned int decay_rate; + unsigned long prev_mhz; + ktime_t prev_ts; + bool mon_started; + struct list_head list; + void *orig_data; + struct cache_hwmon *hw; + struct attribute_group *attr_grp; +}; + +static LIST_HEAD(cache_hwmon_list); +static DEFINE_MUTEX(list_lock); + +static int use_cnt; +static DEFINE_MUTEX(register_lock); + +static DEFINE_MUTEX(monitor_lock); + #define show_attr(name) \ static ssize_t name##_show(struct device *dev, \ struct device_attribute *attr, char *buf) \ { \ - return scnprintf(buf, PAGE_SIZE, "%u\n", name); \ + struct devfreq *df = to_devfreq(dev); \ + struct cache_hwmon_node *hw = df->data; \ + return scnprintf(buf, PAGE_SIZE, "%u\n", hw->name); \ } #define store_attr(name, _min, _max) \ @@ -37,36 +66,42 @@ static ssize_t name##_store(struct device *dev, \ { \ int ret; \ unsigned int val; \ + struct devfreq *df = to_devfreq(dev); \ + struct cache_hwmon_node *hw = df->data; \ ret = kstrtoint(buf, 10, &val); \ if (ret < 0) \ return ret; \ val = max(val, _min); \ val = min(val, _max); \ - name = val; \ + hw->name = val; \ return count; \ } -static struct cache_hwmon *hw; -static unsigned int cycles_per_low_req; -static unsigned int cycles_per_med_req = 20; -static unsigned int cycles_per_high_req = 35; -static unsigned int min_busy = 100; -static unsigned int max_busy = 100; -static unsigned int tolerance_mrps = 5; -static unsigned int guard_band_mhz = 100; -static unsigned int decay_rate = 90; - #define MIN_MS 10U #define MAX_MS 500U -static unsigned int sample_ms = 50; -static unsigned long prev_mhz; -static ktime_t prev_ts; -static unsigned long measure_mrps_and_set_irq(struct devfreq *df, +static struct cache_hwmon_node *find_hwmon_node(struct devfreq *df) +{ + struct cache_hwmon_node *node, *found = NULL; + + mutex_lock(&list_lock); + list_for_each_entry(node, &cache_hwmon_list, list) + if (node->hw->dev == df->dev.parent || + node->hw->of_node == df->dev.parent->of_node) { + found = node; + break; + } + mutex_unlock(&list_lock); + + return found; +} + +static unsigned long measure_mrps_and_set_irq(struct cache_hwmon_node *node, struct mrps_stats *stat) { ktime_t ts; unsigned int us; + struct cache_hwmon *hw = node->hw; /* * Since we are stopping the counters, we don't want this short work @@ -78,59 +113,74 @@ static unsigned long measure_mrps_and_set_irq(struct devfreq *df, preempt_disable(); ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, prev_ts)); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); if (!us) us = 1; - hw->meas_mrps_and_set_irq(df, tolerance_mrps, us, stat); - prev_ts = ts; + hw->meas_mrps_and_set_irq(hw, node->tolerance_mrps, us, stat); + node->prev_ts = ts; preempt_enable(); - pr_debug("stat H=%3lu, M=%3lu, T=%3lu, b=%3u, f=%4lu, us=%d\n", - stat->high, stat->med, stat->high + stat->med, - stat->busy_percent, df->previous_freq / 1000, us); - + trace_cache_hwmon_meas(dev_name(hw->df->dev.parent), stat->mrps[HIGH], + stat->mrps[MED], stat->mrps[LOW], + stat->busy_percent, us); return 0; } -static void compute_cache_freq(struct mrps_stats *mrps, unsigned long *freq) +static void compute_cache_freq(struct cache_hwmon_node *node, + struct mrps_stats *mrps, unsigned long *freq) { unsigned long new_mhz; unsigned int busy; - new_mhz = mrps->high * cycles_per_high_req - + mrps->med * cycles_per_med_req - + mrps->low * cycles_per_low_req; + new_mhz = mrps->mrps[HIGH] * node->cycles_per_high_req + + mrps->mrps[MED] * node->cycles_per_med_req + + mrps->mrps[LOW] * node->cycles_per_low_req; - busy = max(min_busy, mrps->busy_percent); - busy = min(max_busy, busy); + busy = max(node->min_busy, mrps->busy_percent); + busy = min(node->max_busy, busy); new_mhz *= 100; new_mhz /= busy; - if (new_mhz < prev_mhz) { - new_mhz = new_mhz * decay_rate + prev_mhz * (100 - decay_rate); + if (new_mhz < node->prev_mhz) { + new_mhz = new_mhz * node->decay_rate + node->prev_mhz + * (100 - node->decay_rate); new_mhz /= 100; } - prev_mhz = new_mhz; + node->prev_mhz = new_mhz; - new_mhz += guard_band_mhz; + new_mhz += node->guard_band_mhz; *freq = new_mhz * 1000; + trace_cache_hwmon_update(dev_name(node->hw->df->dev.parent), *freq); } #define TOO_SOON_US (1 * USEC_PER_MSEC) -static irqreturn_t mon_intr_handler(int irq, void *dev) +int update_cache_hwmon(struct cache_hwmon *hwmon) { - struct devfreq *df = dev; + struct cache_hwmon_node *node; + struct devfreq *df; ktime_t ts; unsigned int us; int ret; - if (!hw->is_valid_irq(df)) - return IRQ_NONE; + if (!hwmon) + return -EINVAL; + df = hwmon->df; + if (!df) + return -ENODEV; + node = df->data; + if (!node) + return -ENODEV; - pr_debug("Got interrupt\n"); + mutex_lock(&monitor_lock); + if (!node->mon_started) { + mutex_unlock(&monitor_lock); + return -EBUSY; + } + + dev_dbg(df->dev.parent, "Got update request\n"); devfreq_monitor_stop(df); /* @@ -146,27 +196,31 @@ static irqreturn_t mon_intr_handler(int irq, void *dev) * readjusted. */ ts = ktime_get(); - us = ktime_to_us(ktime_sub(ts, prev_ts)); + us = ktime_to_us(ktime_sub(ts, node->prev_ts)); if (us > TOO_SOON_US) { mutex_lock(&df->lock); ret = update_devfreq(df); if (ret < 0) - pr_err("Unable to update freq on IRQ! (%d)\n", ret); + dev_err(df->dev.parent, + "Unable to update freq on req! (%d)\n", ret); mutex_unlock(&df->lock); } devfreq_monitor_start(df); - return IRQ_HANDLED; + mutex_unlock(&monitor_lock); + return 0; } static int devfreq_cache_hwmon_get_freq(struct devfreq *df, unsigned long *freq) { struct mrps_stats stat; + struct cache_hwmon_node *node = df->data; - measure_mrps_and_set_irq(df, &stat); - compute_cache_freq(&stat, freq); + memset(&stat, 0, sizeof(stat)); + measure_mrps_and_set_irq(node, &stat); + compute_cache_freq(node, &stat, freq); return 0; } @@ -217,58 +271,79 @@ static int start_monitoring(struct devfreq *df) { int ret; struct mrps_stats mrps; + struct device *dev = df->dev.parent; + struct cache_hwmon_node *node; + struct cache_hwmon *hw; - prev_ts = ktime_get(); - prev_mhz = 0; - mrps.high = (df->previous_freq / 1000) - guard_band_mhz; - mrps.high /= cycles_per_high_req; + node = find_hwmon_node(df); + if (!node) { + dev_err(dev, "Unable to find HW monitor!\n"); + return -ENODEV; + } + hw = node->hw; + hw->df = df; + node->orig_data = df->data; + df->data = node; - ret = hw->start_hwmon(df, &mrps); + node->prev_ts = ktime_get(); + node->prev_mhz = 0; + mrps.mrps[HIGH] = (df->previous_freq / 1000) - node->guard_band_mhz; + mrps.mrps[HIGH] /= node->cycles_per_high_req; + mrps.mrps[MED] = mrps.mrps[LOW] = 0; + + ret = hw->start_hwmon(hw, &mrps); if (ret < 0) { - pr_err("Unable to start HW monitor! (%d)\n", ret); - return ret; + dev_err(dev, "Unable to start HW monitor! (%d)\n", ret); + goto err_start; } + mutex_lock(&monitor_lock); devfreq_monitor_start(df); - - ret = request_threaded_irq(hw->irq, NULL, mon_intr_handler, - IRQF_ONESHOT | IRQF_SHARED, - "cache_hwmon", df); - if (ret < 0) { - pr_err("Unable to register interrupt handler! (%d)\n", ret); - goto req_irq_fail; - } + node->mon_started = true; + mutex_unlock(&monitor_lock); ret = sysfs_create_group(&df->dev.kobj, &dev_attr_group); if (ret < 0) { - pr_err("Error creating sys entries! (%d)\n", ret); + dev_err(dev, "Error creating sys entries! (%d)\n", ret); goto sysfs_fail; } return 0; sysfs_fail: - disable_irq(hw->irq); - free_irq(hw->irq, df); -req_irq_fail: + mutex_lock(&monitor_lock); + node->mon_started = false; devfreq_monitor_stop(df); - hw->stop_hwmon(df); + mutex_unlock(&monitor_lock); + hw->stop_hwmon(hw); +err_start: + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; return ret; } static void stop_monitoring(struct devfreq *df) { + struct cache_hwmon_node *node = df->data; + struct cache_hwmon *hw = node->hw; + sysfs_remove_group(&df->dev.kobj, &dev_attr_group); - disable_irq(hw->irq); - free_irq(hw->irq, df); + mutex_lock(&monitor_lock); + node->mon_started = false; devfreq_monitor_stop(df); - hw->stop_hwmon(df); + mutex_unlock(&monitor_lock); + hw->stop_hwmon(hw); + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; } static int devfreq_cache_hwmon_ev_handler(struct devfreq *df, unsigned int event, void *data) { int ret; + unsigned int sample_ms; switch (event) { case DEVFREQ_GOV_START: @@ -281,11 +356,11 @@ static int devfreq_cache_hwmon_ev_handler(struct devfreq *df, if (ret < 0) return ret; - pr_debug("Enabled Cache HW monitor governor\n"); + dev_dbg(df->dev.parent, "Enabled Cache HW monitor governor\n"); break; case DEVFREQ_GOV_STOP: stop_monitoring(df); - pr_debug("Disabled Cache HW monitor governor\n"); + dev_dbg(df->dev.parent, "Disabled Cache HW monitor governor\n"); break; case DEVFREQ_GOV_INTERVAL: sample_ms = *(unsigned int *)data; @@ -304,18 +379,48 @@ static struct devfreq_governor devfreq_cache_hwmon = { .event_handler = devfreq_cache_hwmon_ev_handler, }; -int register_cache_hwmon(struct cache_hwmon *hwmon) +int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon) { - int ret; + int ret = 0; + struct cache_hwmon_node *node; - hw = hwmon; - ret = devfreq_add_governor(&devfreq_cache_hwmon); - if (ret < 0) { - pr_err("devfreq governor registration failed: %d\n", ret); + if (!hwmon->dev && !hwmon->of_node) + return -EINVAL; + + node = devm_kzalloc(dev, sizeof(*node), GFP_KERNEL); + if (!node) + return -ENOMEM; + + node->cycles_per_med_req = 20; + node->cycles_per_high_req = 35; + node->min_busy = 100; + node->max_busy = 100; + node->tolerance_mrps = 5; + node->guard_band_mhz = 100; + node->decay_rate = 90; + node->hw = hwmon; + node->attr_grp = &dev_attr_group; + + mutex_lock(®ister_lock); + if (!use_cnt) { + ret = devfreq_add_governor(&devfreq_cache_hwmon); + if (!ret) + use_cnt++; + } + mutex_unlock(®ister_lock); + + if (!ret) { + dev_info(dev, "Cache HWmon governor registered.\n"); + } else { + dev_err(dev, "Failed to add Cache HWmon governor: %d\n", ret); return ret; } - return 0; + mutex_lock(&list_lock); + list_add_tail(&node->list, &cache_hwmon_list); + mutex_unlock(&list_lock); + + return ret; } MODULE_DESCRIPTION("HW monitor based cache freq driver"); diff --git a/drivers/devfreq/governor_cache_hwmon.h b/drivers/devfreq/governor_cache_hwmon.h index 75fbbdedba0f..26e7313a841f 100644 --- a/drivers/devfreq/governor_cache_hwmon.h +++ b/drivers/devfreq/governor_cache_hwmon.h @@ -9,28 +9,53 @@ #include #include +enum request_group { + HIGH, + MED, + LOW, + MAX_NUM_GROUPS, +}; + struct mrps_stats { - unsigned long high; - unsigned long med; - unsigned long low; + unsigned long mrps[MAX_NUM_GROUPS]; unsigned int busy_percent; }; +/** + * struct cache_hwmon - devfreq Cache HW monitor info + * @start_hwmon: Start the HW monitoring + * @stop_hwmon: Stop the HW monitoring + * @meas_mrps_and_set_irq: Return the measured count and set up the + * IRQ to fire if usage exceeds current + * measurement by @tol percent. + * @dev: device that this HW monitor can monitor. + * @of_node: OF node of device that this HW monitor can monitor. + * @df: Devfreq node that this HW montior is being used + * for. NULL when not actively in use, and non-NULL + * when in use. + */ struct cache_hwmon { - int (*start_hwmon)(struct devfreq *df, + int (*start_hwmon)(struct cache_hwmon *hw, struct mrps_stats *mrps); - void (*stop_hwmon)(struct devfreq *df); - bool (*is_valid_irq)(struct devfreq *df); - unsigned long (*meas_mrps_and_set_irq)(struct devfreq *df, + void (*stop_hwmon)(struct cache_hwmon *hw); + unsigned long (*meas_mrps_and_set_irq)(struct cache_hwmon *hw, unsigned int tol, unsigned int us, struct mrps_stats *mrps); - int irq; + struct device *dev; + struct device_node *of_node; + struct devfreq *df; }; #ifdef CONFIG_DEVFREQ_GOV_QCOM_CACHE_HWMON -int register_cache_hwmon(struct cache_hwmon *hwmon); +int register_cache_hwmon(struct device *dev, struct cache_hwmon *hwmon); +int update_cache_hwmon(struct cache_hwmon *hwmon); #else -static inline int register_cache_hwmon(struct cache_hwmon *hwmon) +static inline int register_cache_hwmon(struct device *dev, + struct cache_hwmon *hwmon) +{ + return 0; +} +int update_cache_hwmon(struct cache_hwmon *hwmon) { return 0; } diff --git a/drivers/devfreq/governor_memlat.c b/drivers/devfreq/governor_memlat.c new file mode 100644 index 000000000000..9f1186542cea --- /dev/null +++ b/drivers/devfreq/governor_memlat.c @@ -0,0 +1,418 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2015-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "mem_lat: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "governor.h" +#include "governor_memlat.h" + +#include + +struct memlat_node { + unsigned int ratio_ceil; + unsigned int stall_floor; + bool mon_started; + bool already_zero; + struct list_head list; + void *orig_data; + struct memlat_hwmon *hw; + struct devfreq_governor *gov; + struct attribute_group *attr_grp; +}; + +static LIST_HEAD(memlat_list); +static DEFINE_MUTEX(list_lock); + +static int use_cnt; +static DEFINE_MUTEX(state_lock); + +#define show_attr(name) \ +static ssize_t name##_show(struct device *dev, \ + struct device_attribute *attr, char *buf) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct memlat_node *hw = df->data; \ + return scnprintf(buf, PAGE_SIZE, "%u\n", hw->name); \ +} + +#define store_attr(name, _min, _max) \ +static ssize_t name##_store(struct device *dev, \ + struct device_attribute *attr, const char *buf, \ + size_t count) \ +{ \ + struct devfreq *df = to_devfreq(dev); \ + struct memlat_node *hw = df->data; \ + int ret; \ + unsigned int val; \ + ret = kstrtouint(buf, 10, &val); \ + if (ret < 0) \ + return ret; \ + val = max(val, _min); \ + val = min(val, _max); \ + hw->name = val; \ + return count; \ +} + +static ssize_t freq_map_show(struct device *dev, struct device_attribute *attr, + char *buf) +{ + struct devfreq *df = to_devfreq(dev); + struct memlat_node *n = df->data; + struct core_dev_map *map = n->hw->freq_map; + unsigned int cnt = 0; + + cnt += scnprintf(buf, PAGE_SIZE, "Core freq (MHz)\tDevice BW\n"); + + while (map->core_mhz && cnt < PAGE_SIZE) { + cnt += scnprintf(buf + cnt, PAGE_SIZE - cnt, "%15u\t%9u\n", + map->core_mhz, map->target_freq); + map++; + } + if (cnt < PAGE_SIZE) + cnt += scnprintf(buf + cnt, PAGE_SIZE - cnt, "\n"); + + return cnt; +} + +static DEVICE_ATTR_RO(freq_map); + +static unsigned long core_to_dev_freq(struct memlat_node *node, + unsigned long coref) +{ + struct memlat_hwmon *hw = node->hw; + struct core_dev_map *map = hw->freq_map; + unsigned long freq = 0; + + if (!map) + goto out; + + while (map->core_mhz && map->core_mhz < coref) + map++; + if (!map->core_mhz) + map--; + freq = map->target_freq; + +out: + pr_debug("freq: %lu -> dev: %lu\n", coref, freq); + return freq; +} + +static struct memlat_node *find_memlat_node(struct devfreq *df) +{ + struct memlat_node *node, *found = NULL; + + mutex_lock(&list_lock); + list_for_each_entry(node, &memlat_list, list) + if (node->hw->dev == df->dev.parent || + node->hw->of_node == df->dev.parent->of_node) { + found = node; + break; + } + mutex_unlock(&list_lock); + + return found; +} + +static int start_monitor(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + struct device *dev = df->dev.parent; + int ret; + + ret = hw->start_hwmon(hw); + + if (ret < 0) { + dev_err(dev, "Unable to start HW monitor! (%d)\n", ret); + return ret; + } + + devfreq_monitor_start(df); + + node->mon_started = true; + + return 0; +} + +static void stop_monitor(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + + node->mon_started = false; + + devfreq_monitor_stop(df); + hw->stop_hwmon(hw); +} + +static int gov_start(struct devfreq *df) +{ + int ret = 0; + struct device *dev = df->dev.parent; + struct memlat_node *node; + struct memlat_hwmon *hw; + + node = find_memlat_node(df); + if (!node) { + dev_err(dev, "Unable to find HW monitor!\n"); + return -ENODEV; + } + hw = node->hw; + + hw->df = df; + node->orig_data = df->data; + df->data = node; + + if (start_monitor(df)) + goto err_start; + + ret = sysfs_create_group(&df->dev.kobj, node->attr_grp); + if (ret < 0) + goto err_sysfs; + + return 0; + +err_sysfs: + stop_monitor(df); +err_start: + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; + return ret; +} + +static void gov_stop(struct devfreq *df) +{ + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + + sysfs_remove_group(&df->dev.kobj, node->attr_grp); + stop_monitor(df); + df->data = node->orig_data; + node->orig_data = NULL; + hw->df = NULL; +} + +static int devfreq_memlat_get_freq(struct devfreq *df, + unsigned long *freq) +{ + int i, lat_dev = 0; + struct memlat_node *node = df->data; + struct memlat_hwmon *hw = node->hw; + unsigned long max_freq = 0; + unsigned int ratio; + + hw->get_cnt(hw); + + for (i = 0; i < hw->num_cores; i++) { + ratio = hw->core_stats[i].inst_count; + + if (hw->core_stats[i].mem_count) + ratio /= hw->core_stats[i].mem_count; + + if (!hw->core_stats[i].inst_count + || !hw->core_stats[i].freq) + continue; + + trace_memlat_dev_meas(dev_name(df->dev.parent), + hw->core_stats[i].id, + hw->core_stats[i].inst_count, + hw->core_stats[i].mem_count, + hw->core_stats[i].freq, + hw->core_stats[i].stall_pct, ratio); + + if (ratio <= node->ratio_ceil + && hw->core_stats[i].stall_pct >= node->stall_floor + && hw->core_stats[i].freq > max_freq) { + lat_dev = i; + max_freq = hw->core_stats[i].freq; + } + } + + if (max_freq) + max_freq = core_to_dev_freq(node, max_freq); + + if (max_freq || !node->already_zero) { + trace_memlat_dev_update(dev_name(df->dev.parent), + hw->core_stats[lat_dev].id, + hw->core_stats[lat_dev].inst_count, + hw->core_stats[lat_dev].mem_count, + hw->core_stats[lat_dev].freq, + max_freq); + } + + node->already_zero = !max_freq; + + *freq = max_freq; + return 0; +} + +show_attr(ratio_ceil); +store_attr(ratio_ceil, 1U, 20000U); +static DEVICE_ATTR_RW(ratio_ceil); +show_attr(stall_floor); +store_attr(stall_floor, 0U, 100U); +static DEVICE_ATTR_RW(stall_floor); + +static struct attribute *dev_attr[] = { + &dev_attr_ratio_ceil.attr, + &dev_attr_stall_floor.attr, + &dev_attr_freq_map.attr, + NULL, +}; + +static struct attribute_group dev_attr_group = { + .name = "mem_latency", + .attrs = dev_attr, +}; + +#define MIN_MS 10U +#define MAX_MS 500U +static int devfreq_memlat_ev_handler(struct devfreq *df, + unsigned int event, void *data) +{ + int ret; + unsigned int sample_ms; + + switch (event) { + case DEVFREQ_GOV_START: + sample_ms = df->profile->polling_ms; + sample_ms = max(MIN_MS, sample_ms); + sample_ms = min(MAX_MS, sample_ms); + df->profile->polling_ms = sample_ms; + + ret = gov_start(df); + if (ret < 0) + return ret; + + dev_dbg(df->dev.parent, + "Enabled Memory Latency governor\n"); + break; + + case DEVFREQ_GOV_STOP: + gov_stop(df); + dev_dbg(df->dev.parent, + "Disabled Memory Latency governor\n"); + break; + + case DEVFREQ_GOV_INTERVAL: + sample_ms = *(unsigned int *)data; + sample_ms = max(MIN_MS, sample_ms); + sample_ms = min(MAX_MS, sample_ms); + devfreq_interval_update(df, &sample_ms); + break; + } + + return 0; +} + +static struct devfreq_governor devfreq_gov_memlat = { + .name = "mem_latency", + .get_target_freq = devfreq_memlat_get_freq, + .event_handler = devfreq_memlat_ev_handler, +}; + +#define NUM_COLS 2 +static struct core_dev_map *init_core_dev_map(struct device *dev, + char *prop_name) +{ + int len, nf, i, j; + u32 data; + struct core_dev_map *tbl; + int ret; + + if (!of_find_property(dev->of_node, prop_name, &len)) + return NULL; + len /= sizeof(data); + + if (len % NUM_COLS || len == 0) + return NULL; + nf = len / NUM_COLS; + + tbl = devm_kzalloc(dev, (nf + 1) * sizeof(struct core_dev_map), + GFP_KERNEL); + if (!tbl) + return NULL; + + for (i = 0, j = 0; i < nf; i++, j += 2) { + ret = of_property_read_u32_index(dev->of_node, prop_name, j, + &data); + if (ret < 0) + return NULL; + tbl[i].core_mhz = data / 1000; + + ret = of_property_read_u32_index(dev->of_node, prop_name, j + 1, + &data); + if (ret < 0) + return NULL; + tbl[i].target_freq = data; + pr_debug("Entry%d CPU:%u, Dev:%u\n", i, tbl[i].core_mhz, + tbl[i].target_freq); + } + tbl[i].core_mhz = 0; + + return tbl; +} + +int register_memlat(struct device *dev, struct memlat_hwmon *hw) +{ + int ret = 0; + struct memlat_node *node; + + if (!hw->dev && !hw->of_node) + return -EINVAL; + + node = devm_kzalloc(dev, sizeof(*node), GFP_KERNEL); + if (!node) + return -ENOMEM; + + node->gov = &devfreq_gov_memlat; + node->attr_grp = &dev_attr_group; + + node->ratio_ceil = 10; + node->hw = hw; + + hw->freq_map = init_core_dev_map(dev, "qcom,core-dev-table"); + if (!hw->freq_map) { + dev_err(dev, "Couldn't find the core-dev freq table!\n"); + return -EINVAL; + } + + mutex_lock(&list_lock); + list_add_tail(&node->list, &memlat_list); + mutex_unlock(&list_lock); + + mutex_lock(&state_lock); + if (!use_cnt) + ret = devfreq_add_governor(&devfreq_gov_memlat); + if (!ret) + use_cnt++; + mutex_unlock(&state_lock); + + if (!ret) + dev_info(dev, "Memory Latency governor registered.\n"); + else + dev_err(dev, "Memory Latency governor registration failed!\n"); + + return ret; +} + +MODULE_DESCRIPTION("HW monitor based dev DDR bandwidth voting driver"); +MODULE_LICENSE("GPL v2"); diff --git a/drivers/devfreq/governor_memlat.h b/drivers/devfreq/governor_memlat.h new file mode 100644 index 000000000000..335ba7598b6b --- /dev/null +++ b/drivers/devfreq/governor_memlat.h @@ -0,0 +1,82 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2015-2017, 2019, The Linux Foundation. All rights reserved. + */ + +#ifndef _GOVERNOR_MEMLAT_H +#define _GOVERNOR_MEMLAT_H + +#include +#include + +/** + * struct dev_stats - Device stats + * @inst_count: Number of instructions executed. + * @mem_count: Number of memory accesses made. + * @freq: Effective frequency of the device in the + * last interval. + */ +struct dev_stats { + int id; + unsigned long inst_count; + unsigned long mem_count; + unsigned long freq; + unsigned long stall_pct; +}; + +struct core_dev_map { + unsigned int core_mhz; + unsigned int target_freq; +}; + +/** + * struct memlat_hwmon - Memory Latency HW monitor info + * @start_hwmon: Start the HW monitoring + * @stop_hwmon: Stop the HW monitoring + * @get_cnt: Return the number of intructions executed, + * memory accesses and effective frequency + * @dev: Pointer to device that this HW monitor can + * monitor. + * @of_node: OF node of device that this HW monitor can + * monitor. + * @df: Devfreq node that this HW monitor is being + * used for. NULL when not actively in use and + * non-NULL when in use. + * @num_cores: Number of cores that are monitored by the + * hardware monitor. + * @core_stats: Array containing instruction count, memory + * accesses and effective frequency for each core. + * + * One of dev or of_node needs to be specified for a successful registration. + * + */ +struct memlat_hwmon { + int (*start_hwmon)(struct memlat_hwmon *hw); + void (*stop_hwmon)(struct memlat_hwmon *hw); + unsigned long (*get_cnt)(struct memlat_hwmon *hw); + struct device *dev; + struct device_node *of_node; + + unsigned int num_cores; + struct dev_stats *core_stats; + + struct devfreq *df; + struct core_dev_map *freq_map; +}; + +#ifdef CONFIG_DEVFREQ_GOV_MEMLAT +int register_memlat(struct device *dev, struct memlat_hwmon *hw); +int update_memlat(struct memlat_hwmon *hw); +#else +static inline int register_memlat(struct device *dev, + struct memlat_hwmon *hw) +{ + return 0; +} +static inline int update_memlat(struct memlat_hwmon *hw) +{ + return 0; +} +#endif + +#endif /* _GOVERNOR_BW_HWMON_H */ diff --git a/include/trace/events/power.h b/include/trace/events/power.h index bd6a9606647b..04bd8831dd2b 100644 --- a/include/trace/events/power.h +++ b/include/trace/events/power.h @@ -585,6 +585,121 @@ TRACE_EVENT(bw_hwmon_update, __entry->down_thres) ); +TRACE_EVENT(cache_hwmon_meas, + TP_PROTO(const char *name, unsigned long high_mrps, + unsigned long med_mrps, unsigned long low_mrps, + unsigned int busy_percent, unsigned int us), + TP_ARGS(name, high_mrps, med_mrps, low_mrps, busy_percent, us), + TP_STRUCT__entry( + __string(name, name) + __field(unsigned long, high_mrps) + __field(unsigned long, med_mrps) + __field(unsigned long, low_mrps) + __field(unsigned long, total_mrps) + __field(unsigned int, busy_percent) + __field(unsigned int, us) + ), + TP_fast_assign( + __assign_str(name, name); + __entry->high_mrps = high_mrps; + __entry->med_mrps = med_mrps; + __entry->low_mrps = low_mrps; + __entry->total_mrps = high_mrps + med_mrps + low_mrps; + __entry->busy_percent = busy_percent; + __entry->us = us; + ), + TP_printk("dev=%s H=%lu M=%lu L=%lu T=%lu busy_pct=%u period=%u", + __get_str(name), __entry->high_mrps, __entry->med_mrps, + __entry->low_mrps, __entry->total_mrps, + __entry->busy_percent, __entry->us) +); + +TRACE_EVENT(cache_hwmon_update, + TP_PROTO(const char *name, unsigned long freq_mhz), + TP_ARGS(name, freq_mhz), + TP_STRUCT__entry( + __string(name, name) + __field(unsigned long, freq) + ), + TP_fast_assign( + __assign_str(name, name); + __entry->freq = freq_mhz; + ), + TP_printk("dev=%s freq=%lu", __get_str(name), __entry->freq) +); + +TRACE_EVENT(memlat_dev_meas, + + TP_PROTO(const char *name, unsigned int dev_id, unsigned long inst, + unsigned long mem, unsigned long freq, unsigned int stall, + unsigned int ratio), + + TP_ARGS(name, dev_id, inst, mem, freq, stall, ratio), + + TP_STRUCT__entry( + __string(name, name) + __field(unsigned int, dev_id) + __field(unsigned long, inst) + __field(unsigned long, mem) + __field(unsigned long, freq) + __field(unsigned int, stall) + __field(unsigned int, ratio) + ), + + TP_fast_assign( + __assign_str(name, name); + __entry->dev_id = dev_id; + __entry->inst = inst; + __entry->mem = mem; + __entry->freq = freq; + __entry->stall = stall; + __entry->ratio = ratio; + ), + + TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, stall=%u, ratio=%u", + __get_str(name), + __entry->dev_id, + __entry->inst, + __entry->mem, + __entry->freq, + __entry->stall, + __entry->ratio) +); + +TRACE_EVENT(memlat_dev_update, + + TP_PROTO(const char *name, unsigned int dev_id, unsigned long inst, + unsigned long mem, unsigned long freq, unsigned long vote), + + TP_ARGS(name, dev_id, inst, mem, freq, vote), + + TP_STRUCT__entry( + __string(name, name) + __field(unsigned int, dev_id) + __field(unsigned long, inst) + __field(unsigned long, mem) + __field(unsigned long, freq) + __field(unsigned long, vote) + ), + + TP_fast_assign( + __assign_str(name, name); + __entry->dev_id = dev_id; + __entry->inst = inst; + __entry->mem = mem; + __entry->freq = freq; + __entry->vote = vote; + ), + + TP_printk("dev: %s, id=%u, inst=%lu, mem=%lu, freq=%lu, vote=%lu", + __get_str(name), + __entry->dev_id, + __entry->inst, + __entry->mem, + __entry->freq, + __entry->vote) +); + #endif /* _TRACE_POWER_H */ /* This part must be outside protection */