diff --git a/include/linux/sched/sysctl.h b/include/linux/sched/sysctl.h index 3d9b702f3f96..0e83233722d7 100644 --- a/include/linux/sched/sysctl.h +++ b/include/linux/sched/sysctl.h @@ -34,30 +34,30 @@ extern unsigned int sysctl_sched_force_lb_enable; extern unsigned int sysctl_hh_suspend_timeout_ms; #endif #ifdef CONFIG_SCHED_WALT -extern unsigned int __weak sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS]; -extern unsigned int __weak sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS]; -extern unsigned int __weak sysctl_sched_user_hint; -extern const int __weak sched_user_hint_max; -extern unsigned int __weak sysctl_sched_boost; -extern unsigned int __weak sysctl_sched_group_upmigrate_pct; -extern unsigned int __weak sysctl_sched_group_downmigrate_pct; -extern unsigned int __weak sysctl_sched_conservative_pl; -extern unsigned int __weak sysctl_sched_walt_rotate_big_tasks; -extern unsigned int __weak sysctl_sched_min_task_util_for_boost; -extern unsigned int __weak sysctl_sched_min_task_util_for_colocation; -extern unsigned int __weak sysctl_sched_asym_cap_sibling_freq_match_pct; -extern unsigned int __weak sysctl_sched_coloc_downmigrate_ns; -extern unsigned int __weak sysctl_sched_task_unfilter_period; -extern unsigned int __weak sysctl_sched_busy_hyst_enable_cpus; -extern unsigned int __weak sysctl_sched_busy_hyst; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_enable_cpus; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS]; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_max_ms; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS]; -extern unsigned int __weak sysctl_sched_window_stats_policy; -extern unsigned int __weak sysctl_sched_ravg_window_nr_ticks; -extern unsigned int __weak sysctl_sched_many_wakeup_threshold; -extern unsigned int __weak sysctl_sched_dynamic_ravg_window_enable; +extern unsigned int sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS]; +extern unsigned int sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS]; +extern unsigned int sysctl_sched_user_hint; +extern const int sched_user_hint_max; +extern unsigned int sysctl_sched_boost; +extern unsigned int sysctl_sched_group_upmigrate_pct; +extern unsigned int sysctl_sched_group_downmigrate_pct; +extern unsigned int sysctl_sched_conservative_pl; +extern unsigned int sysctl_sched_walt_rotate_big_tasks; +extern unsigned int sysctl_sched_min_task_util_for_boost; +extern unsigned int sysctl_sched_min_task_util_for_colocation; +extern unsigned int sysctl_sched_asym_cap_sibling_freq_match_pct; +extern unsigned int sysctl_sched_coloc_downmigrate_ns; +extern unsigned int sysctl_sched_task_unfilter_period; +extern unsigned int sysctl_sched_busy_hyst_enable_cpus; +extern unsigned int sysctl_sched_busy_hyst; +extern unsigned int sysctl_sched_coloc_busy_hyst_enable_cpus; +extern unsigned int sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS]; +extern unsigned int sysctl_sched_coloc_busy_hyst_max_ms; +extern unsigned int sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS]; +extern unsigned int sysctl_sched_window_stats_policy; +extern unsigned int sysctl_sched_ravg_window_nr_ticks; +extern unsigned int sysctl_sched_many_wakeup_threshold; +extern unsigned int sysctl_sched_dynamic_ravg_window_enable; extern unsigned int sysctl_sched_prefer_spread; extern unsigned int sysctl_walt_rtg_cfs_boost_prio; extern unsigned int sysctl_walt_low_latency_task_threshold; diff --git a/kernel/sched/Makefile b/kernel/sched/Makefile index 1175ed6ccf4d..dad16b665f11 100644 --- a/kernel/sched/Makefile +++ b/kernel/sched/Makefile @@ -20,7 +20,7 @@ obj-y += core.o loadavg.o clock.o cputime.o obj-y += idle.o fair.o rt.o deadline.o obj-y += wait.o wait_bit.o swait.o completion.o -obj-$(CONFIG_SCHED_WALT) += walt.o +obj-$(CONFIG_SCHED_WALT) += walt/ obj-$(CONFIG_SMP) += cpupri.o cpudeadline.o topology.o stop_task.o pelt.o obj-$(CONFIG_SCHED_AUTOGROUP) += autogroup.o obj-$(CONFIG_SCHEDSTATS) += stats.o diff --git a/kernel/sched/core.c b/kernel/sched/core.c index c5b5ec2ef436..e9339790ab85 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -21,7 +21,7 @@ #include "../smpboot.h" #include "pelt.h" -#include "walt.h" +#include "walt/walt.h" #define CREATE_TRACE_POINTS #include diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c index 8ea9e07784da..55dff2d61356 100644 --- a/kernel/sched/cputime.c +++ b/kernel/sched/cputime.c @@ -4,7 +4,7 @@ */ #include #include "sched.h" -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_IRQ_TIME_ACCOUNTING diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 58e0eeef0090..650d8b0ba510 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -17,7 +17,7 @@ */ #include "sched.h" #include "pelt.h" -#include "walt.h" +#include "walt/walt.h" struct dl_bandwidth def_dl_bandwidth; diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 43ddd00efe39..c81e1a5d64b2 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -25,7 +25,7 @@ #include #include -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_SMP static inline bool task_fits_max(struct task_struct *p, int cpu); diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 2c83330f9ea0..e8faf503a343 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -11,7 +11,7 @@ #include -#include "walt.h" +#include "walt/walt.h" #include diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index 6a0937d4884f..6d360473ab67 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -8,7 +8,7 @@ * See kernel/stop_machine.c */ #include "sched.h" -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_SMP static int diff --git a/kernel/sched/walt.c b/kernel/sched/walt.c deleted file mode 100644 index 1721a470c2f1..000000000000 --- a/kernel/sched/walt.c +++ /dev/null @@ -1,218 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -/* - * Copyright (c) 2016-2020, The Linux Foundation. All rights reserved. - */ - -#include "sched.h" -#include "walt.h" - -int __weak sched_wake_up_idle_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak sched_wake_up_idle_write(struct file *file, - const char __user *buf, size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_wake_up_idle_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_init_task_load_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak -sched_init_task_load_write(struct file *file, const char __user *buf, - size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_init_task_load_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_group_id_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak sched_group_id_write(struct file *file, const char __user *buf, - size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_group_id_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_isolate_cpu(int cpu) { return 0; } - -int __weak sched_unisolate_cpu(int cpu) { return 0; } - -int __weak sched_unisolate_cpu_unlocked(int cpu) { return 0; } - -int __weak register_cpu_cycle_counter_cb(struct cpu_cycle_counter_cb *cb) -{ - return 0; -} - -void __weak sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, - u32 fmax) { } - -void __weak free_task_load_ptrs(struct task_struct *p) { } - -int __weak core_ctl_set_boost(bool boost) { return 0; } - -void __weak core_ctl_notifier_register(struct notifier_block *n) { } - -void __weak core_ctl_notifier_unregister(struct notifier_block *n) { } - -void __weak sched_update_nr_prod(int cpu, long delta, bool inc) { } - -unsigned int __weak sched_get_cpu_util(int cpu) { return 0; } - -void __weak sched_update_hyst_times(void) { } - -u64 __weak sched_lpm_disallowed_time(int cpu) { return 0; } - -int __weak -walt_proc_group_thresholds_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -walt_proc_user_hint_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -sched_updown_migrate_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -sched_ravg_window_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak sched_boost_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak sched_busy_hyst_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -u64 __weak sched_ktime_clock(void) { return 0; } - -unsigned long __weak -cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) -{ - return cpu_util(cpu); -} - -int __weak update_preferred_cluster(struct walt_related_thread_group *grp, - struct task_struct *p, u32 old_load, bool from_tick) -{ - return 0; -} - -void __weak set_preferred_cluster(struct walt_related_thread_group *grp) { } - -void __weak add_new_task_to_grp(struct task_struct *new) { } - -int __weak -preferred_cluster(struct walt_sched_cluster *cluster, struct task_struct *p) -{ - return -1; -} - -int __weak sync_cgroup_colocation(struct task_struct *p, bool insert) -{ - return 0; -} - -int __weak alloc_related_thread_groups(void) { return 0; } - -void __weak check_for_migration(struct rq *rq, struct task_struct *p) { } - -unsigned long __weak thermal_cap(int cpu) -{ - return cpu_rq(cpu)->cpu_capacity_orig; -} - -void __weak clear_walt_request(int cpu) { } - -void __weak clear_ed_task(struct task_struct *p, struct rq *rq) { } - -bool __weak early_detection_notify(struct rq *rq, u64 wallclock) -{ - return 0; -} - -void __weak note_task_waking(struct task_struct *p, u64 wallclock) { } - -int __weak group_balance_cpu_not_isolated(struct sched_group *sg) -{ - return group_balance_cpu(sg); -} - -void __weak detach_one_task_core(struct task_struct *p, struct rq *rq, - struct list_head *tasks) { } - -void __weak attach_tasks_core(struct list_head *tasks, struct rq *rq) { } - -void __weak walt_update_task_ravg(struct task_struct *p, struct rq *rq, - int event, u64 wallclock, u64 irqtime) { } - -void __weak fixup_busy_time(struct task_struct *p, int new_cpu) { } - -void __weak init_new_task_load(struct task_struct *p) { } - -void __weak mark_task_starting(struct task_struct *p) { } - -void __weak set_window_start(struct rq *rq) { } - -bool __weak do_pl_notif(struct rq *rq) { return false; } - -void __weak walt_sched_account_irqstart(int cpu, struct task_struct *curr) { } -void __weak walt_sched_account_irqend(int cpu, struct task_struct *curr, - u64 delta) -{ -} - -void __weak update_cluster_topology(void) { } - -void __weak init_clusters(void) { } - -void __weak walt_sched_init_rq(struct rq *rq) { } - -void __weak walt_update_cluster_topology(void) { } - -void __weak walt_task_dead(struct task_struct *p) { } - -#if defined(CONFIG_UCLAMP_TASK_GROUP) -void __weak walt_init_sched_boost(struct task_group *tg) { } -#endif diff --git a/kernel/sched/walt/Makefile b/kernel/sched/walt/Makefile new file mode 100644 index 000000000000..8bf42cb669af --- /dev/null +++ b/kernel/sched/walt/Makefile @@ -0,0 +1,3 @@ +# SPDX-License-Identifier: GPL-2.0 +obj-$(CONFIG_SCHED_WALT) += walt.o boost.o sched_avg.o qc_vas.o core_ctl.o trace.o +obj-$(CONFIG_CPU_FREQ) += cpu-boost.o diff --git a/kernel/sched/walt/boost.c b/kernel/sched/walt/boost.c new file mode 100644 index 000000000000..c37185e293d0 --- /dev/null +++ b/kernel/sched/walt/boost.c @@ -0,0 +1,318 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2012-2021, The Linux Foundation. All rights reserved. + */ +#include "qc_vas.h" +#include +#include +#include + +/* + * Scheduler boost is a mechanism to temporarily place tasks on CPUs + * with higher capacity than those where a task would have normally + * ended up with their load characteristics. Any entity enabling + * boost is responsible for disabling it as well. + */ + +unsigned int sysctl_sched_boost; /* To/from userspace */ +unsigned int sched_boost_type; /* currently activated sched boost */ +enum sched_boost_policy boost_policy; + +static enum sched_boost_policy boost_policy_dt = SCHED_BOOST_NONE; +static DEFINE_MUTEX(boost_mutex); + +#if defined(CONFIG_UCLAMP_TASK_GROUP) +void walt_init_sched_boost(struct task_group *tg) +{ + tg->wtg.sched_boost_no_override = false; + tg->wtg.sched_boost_enabled = true; + tg->wtg.colocate = false; + tg->wtg.colocate_update_disabled = false; +} + +static void update_cgroup_boost_settings(void) +{ + struct task_group *tg; + + rcu_read_lock(); + list_for_each_entry_rcu(tg, &task_groups, list) { + if (tg->wtg.sched_boost_no_override) + continue; + + tg->wtg.sched_boost_enabled = false; + } + rcu_read_unlock(); +} + +static void restore_cgroup_boost_settings(void) +{ + struct task_group *tg; + + rcu_read_lock(); + list_for_each_entry_rcu(tg, &task_groups, list) + tg->wtg.sched_boost_enabled = true; + rcu_read_unlock(); +} + +#else +static void update_cgroup_boost_settings(void) { } +static void restore_cgroup_boost_settings(void) { } +#endif + +/* + * Scheduler boost type and boost policy might at first seem unrelated, + * however, there exists a connection between them that will allow us + * to use them interchangeably during placement decisions. We'll explain + * the connection here in one possible way so that the implications are + * clear when looking at placement policies. + * + * When policy = SCHED_BOOST_NONE, type is either none or RESTRAINED + * When policy = SCHED_BOOST_ON_ALL or SCHED_BOOST_ON_BIG, type can + * neither be none nor RESTRAINED. + */ +static void set_boost_policy(int type) +{ + if (type == NO_BOOST || type == RESTRAINED_BOOST) { + boost_policy = SCHED_BOOST_NONE; + return; + } + + if (boost_policy_dt) { + boost_policy = boost_policy_dt; + return; + } + + if (hmp_capable()) { + boost_policy = SCHED_BOOST_ON_BIG; + return; + } + + boost_policy = SCHED_BOOST_ON_ALL; +} + +static bool verify_boost_params(int type) +{ + return type >= RESTRAINED_BOOST_DISABLE && type <= RESTRAINED_BOOST; +} + +static void sched_no_boost_nop(void) +{ +} + +static void sched_full_throttle_boost_enter(void) +{ + core_ctl_set_boost(true); + walt_enable_frequency_aggregation(true); +} + +static void sched_full_throttle_boost_exit(void) +{ + core_ctl_set_boost(false); + walt_enable_frequency_aggregation(false); +} + +static void sched_conservative_boost_enter(void) +{ + update_cgroup_boost_settings(); +} + +static void sched_conservative_boost_exit(void) +{ + restore_cgroup_boost_settings(); +} + +static void sched_restrained_boost_enter(void) +{ + walt_enable_frequency_aggregation(true); +} + +static void sched_restrained_boost_exit(void) +{ + walt_enable_frequency_aggregation(false); +} + +struct sched_boost_data { + int refcount; + void (*enter)(void); + void (*exit)(void); +}; + +static struct sched_boost_data sched_boosts[] = { + [NO_BOOST] = { + .refcount = 0, + .enter = sched_no_boost_nop, + .exit = sched_no_boost_nop, + }, + [FULL_THROTTLE_BOOST] = { + .refcount = 0, + .enter = sched_full_throttle_boost_enter, + .exit = sched_full_throttle_boost_exit, + }, + [CONSERVATIVE_BOOST] = { + .refcount = 0, + .enter = sched_conservative_boost_enter, + .exit = sched_conservative_boost_exit, + }, + [RESTRAINED_BOOST] = { + .refcount = 0, + .enter = sched_restrained_boost_enter, + .exit = sched_restrained_boost_exit, + }, +}; + +#define SCHED_BOOST_START FULL_THROTTLE_BOOST +#define SCHED_BOOST_END (RESTRAINED_BOOST + 1) + +static int sched_effective_boost(void) +{ + int i; + + /* + * The boosts are sorted in descending order by + * priority. + */ + for (i = SCHED_BOOST_START; i < SCHED_BOOST_END; i++) { + if (sched_boosts[i].refcount >= 1) + return i; + } + + return NO_BOOST; +} + +static void sched_boost_disable(int type) +{ + struct sched_boost_data *sb = &sched_boosts[type]; + int next_boost; + + if (sb->refcount <= 0) + return; + + sb->refcount--; + + if (sb->refcount) + return; + + /* + * This boost's refcount becomes zero, so it must + * be disabled. Disable it first and then apply + * the next boost. + */ + sb->exit(); + + next_boost = sched_effective_boost(); + sched_boosts[next_boost].enter(); +} + +static void sched_boost_enable(int type) +{ + struct sched_boost_data *sb = &sched_boosts[type]; + int next_boost, prev_boost = sched_boost_type; + + sb->refcount++; + + if (sb->refcount != 1) + return; + + /* + * This boost enable request did not come before. + * Take this new request and find the next boost + * by aggregating all the enabled boosts. If there + * is a change, disable the previous boost and enable + * the next boost. + */ + + next_boost = sched_effective_boost(); + if (next_boost == prev_boost) + return; + + sched_boosts[prev_boost].exit(); + sched_boosts[next_boost].enter(); +} + +static void sched_boost_disable_all(void) +{ + int i; + + for (i = SCHED_BOOST_START; i < SCHED_BOOST_END; i++) { + if (sched_boosts[i].refcount > 0) { + sched_boosts[i].exit(); + sched_boosts[i].refcount = 0; + } + } +} + +static void _sched_set_boost(int type) +{ + if (type == 0) + sched_boost_disable_all(); + else if (type > 0) + sched_boost_enable(type); + else + sched_boost_disable(-type); + + /* + * sysctl_sched_boost holds the boost request from + * user space which could be different from the + * effectively enabled boost. Update the effective + * boost here. + */ + + sched_boost_type = sched_effective_boost(); + sysctl_sched_boost = sched_boost_type; + set_boost_policy(sysctl_sched_boost); + trace_sched_set_boost(sysctl_sched_boost); +} + +void sched_boost_parse_dt(void) +{ + struct device_node *sn; + const char *boost_policy; + + sn = of_find_node_by_path("/sched-hmp"); + if (!sn) + return; + + if (!of_property_read_string(sn, "boost-policy", &boost_policy)) { + if (!strcmp(boost_policy, "boost-on-big")) + boost_policy_dt = SCHED_BOOST_ON_BIG; + else if (!strcmp(boost_policy, "boost-on-all")) + boost_policy_dt = SCHED_BOOST_ON_ALL; + } +} + +int sched_set_boost(int type) +{ + int ret = 0; + + mutex_lock(&boost_mutex); + if (verify_boost_params(type)) + _sched_set_boost(type); + else + ret = -EINVAL; + mutex_unlock(&boost_mutex); + return ret; +} + +int sched_boost_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + unsigned int *data = (unsigned int *)table->data; + + mutex_lock(&boost_mutex); + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + + if (ret || !write) + goto done; + + if (verify_boost_params(*data)) + _sched_set_boost(*data); + else + ret = -EINVAL; + +done: + mutex_unlock(&boost_mutex); + return ret; +} diff --git a/kernel/sched/walt/core_ctl.c b/kernel/sched/walt/core_ctl.c new file mode 100644 index 000000000000..1413b399c98b --- /dev/null +++ b/kernel/sched/walt/core_ctl.c @@ -0,0 +1,1354 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2014-2021, The Linux Foundation. All rights reserved. + */ +#define pr_fmt(fmt) "core_ctl: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include "qc_vas.h" + +struct cluster_data { + bool inited; + unsigned int min_cpus; + unsigned int max_cpus; + unsigned int offline_delay_ms; + unsigned int busy_up_thres[MAX_CPUS_PER_CLUSTER]; + unsigned int busy_down_thres[MAX_CPUS_PER_CLUSTER]; + unsigned int active_cpus; + unsigned int num_cpus; + unsigned int nr_isolated_cpus; + unsigned int nr_not_preferred_cpus; + cpumask_t cpu_mask; + unsigned int need_cpus; + unsigned int task_thres; + unsigned int max_nr; + unsigned int nr_prev_assist; + unsigned int nr_prev_assist_thresh; + s64 need_ts; + struct list_head lru; + bool pending; + spinlock_t pending_lock; + bool enable; + int nrrun; + struct task_struct *core_ctl_thread; + unsigned int first_cpu; + unsigned int boost; + struct kobject kobj; + unsigned int strict_nrrun; +}; + +struct cpu_data { + bool is_busy; + unsigned int busy; + unsigned int cpu; + bool not_preferred; + struct cluster_data *cluster; + struct list_head sib; + bool isolated_by_us; +}; + +static DEFINE_PER_CPU(struct cpu_data, cpu_state); +static struct cluster_data cluster_state[MAX_CLUSTERS]; +static unsigned int num_clusters; + +#define for_each_cluster(cluster, idx) \ + for (; (idx) < num_clusters && ((cluster) = &cluster_state[idx]);\ + idx++) + + +static DEFINE_SPINLOCK(state_lock); +static void apply_need(struct cluster_data *state); +static void wake_up_core_ctl_thread(struct cluster_data *state); +static bool initialized; + +ATOMIC_NOTIFIER_HEAD(core_ctl_notifier); +static unsigned int last_nr_big; + +static unsigned int get_active_cpu_count(const struct cluster_data *cluster); + +/* ========================= sysfs interface =========================== */ + +static ssize_t store_min_cpus(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->min_cpus = min(val, state->num_cpus); + wake_up_core_ctl_thread(state); + + return count; +} + +static ssize_t show_min_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->min_cpus); +} + +static ssize_t store_max_cpus(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->max_cpus = min(val, state->num_cpus); + wake_up_core_ctl_thread(state); + + return count; +} + +static ssize_t show_max_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->max_cpus); +} + +static ssize_t store_offline_delay_ms(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->offline_delay_ms = val; + apply_need(state); + + return count; +} + +static ssize_t show_task_thres(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->task_thres); +} + +static ssize_t store_task_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + if (val < state->num_cpus) + return -EINVAL; + + state->task_thres = val; + apply_need(state); + + return count; +} + +static ssize_t show_nr_prev_assist_thresh(const struct cluster_data *state, + char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->nr_prev_assist_thresh); +} + +static ssize_t store_nr_prev_assist_thresh(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->nr_prev_assist_thresh = val; + apply_need(state); + + return count; +} + +static ssize_t show_offline_delay_ms(const struct cluster_data *state, + char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->offline_delay_ms); +} + +static ssize_t store_busy_up_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val[MAX_CPUS_PER_CLUSTER]; + int ret, i; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != 1 && ret != state->num_cpus) + return -EINVAL; + + if (ret == 1) { + for (i = 0; i < state->num_cpus; i++) + state->busy_up_thres[i] = val[0]; + } else { + for (i = 0; i < state->num_cpus; i++) + state->busy_up_thres[i] = val[i]; + } + apply_need(state); + return count; +} + +static ssize_t show_busy_up_thres(const struct cluster_data *state, char *buf) +{ + int i, count = 0; + + for (i = 0; i < state->num_cpus; i++) + count += snprintf(buf + count, PAGE_SIZE - count, "%u ", + state->busy_up_thres[i]); + + count += snprintf(buf + count, PAGE_SIZE - count, "\n"); + return count; +} + +static ssize_t store_busy_down_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val[MAX_CPUS_PER_CLUSTER]; + int ret, i; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != 1 && ret != state->num_cpus) + return -EINVAL; + + if (ret == 1) { + for (i = 0; i < state->num_cpus; i++) + state->busy_down_thres[i] = val[0]; + } else { + for (i = 0; i < state->num_cpus; i++) + state->busy_down_thres[i] = val[i]; + } + apply_need(state); + return count; +} + +static ssize_t show_busy_down_thres(const struct cluster_data *state, char *buf) +{ + int i, count = 0; + + for (i = 0; i < state->num_cpus; i++) + count += snprintf(buf + count, PAGE_SIZE - count, "%u ", + state->busy_down_thres[i]); + + count += snprintf(buf + count, PAGE_SIZE - count, "\n"); + return count; +} + +static ssize_t store_enable(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + bool bval; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + bval = !!val; + if (bval != state->enable) { + state->enable = bval; + apply_need(state); + } + + return count; +} + +static ssize_t show_enable(const struct cluster_data *state, char *buf) +{ + return scnprintf(buf, PAGE_SIZE, "%u\n", state->enable); +} + +static ssize_t show_need_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->need_cpus); +} + +static ssize_t show_active_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->active_cpus); +} + +static ssize_t show_global_state(const struct cluster_data *state, char *buf) +{ + struct cpu_data *c; + struct cluster_data *cluster; + ssize_t count = 0; + unsigned int cpu; + + spin_lock_irq(&state_lock); + for_each_possible_cpu(cpu) { + c = &per_cpu(cpu_state, cpu); + cluster = c->cluster; + if (!cluster || !cluster->inited) + continue; + + count += snprintf(buf + count, PAGE_SIZE - count, + "CPU%u\n", cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tCPU: %u\n", c->cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tOnline: %u\n", + cpu_online(c->cpu)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tIsolated: %u\n", + cpu_isolated(c->cpu)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tFirst CPU: %u\n", + cluster->first_cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tBusy%%: %u\n", c->busy); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tIs busy: %u\n", c->is_busy); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNot preferred: %u\n", + c->not_preferred); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNr running: %u\n", cluster->nrrun); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tActive CPUs: %u\n", get_active_cpu_count(cluster)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNeed CPUs: %u\n", cluster->need_cpus); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNr isolated CPUs: %u\n", + cluster->nr_isolated_cpus); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tBoost: %u\n", (unsigned int) cluster->boost); + } + spin_unlock_irq(&state_lock); + + return count; +} + +static ssize_t store_not_preferred(struct cluster_data *state, + const char *buf, size_t count) +{ + struct cpu_data *c; + unsigned int i; + unsigned int val[MAX_CPUS_PER_CLUSTER]; + unsigned long flags; + int ret; + int not_preferred_count = 0; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != state->num_cpus) + return -EINVAL; + + spin_lock_irqsave(&state_lock, flags); + for (i = 0; i < state->num_cpus; i++) { + c = &per_cpu(cpu_state, i + state->first_cpu); + c->not_preferred = val[i]; + not_preferred_count += !!val[i]; + } + state->nr_not_preferred_cpus = not_preferred_count; + spin_unlock_irqrestore(&state_lock, flags); + + return count; +} + +static ssize_t show_not_preferred(const struct cluster_data *state, char *buf) +{ + struct cpu_data *c; + ssize_t count = 0; + unsigned long flags; + int i; + + spin_lock_irqsave(&state_lock, flags); + for (i = 0; i < state->num_cpus; i++) { + c = &per_cpu(cpu_state, i + state->first_cpu); + count += scnprintf(buf + count, PAGE_SIZE - count, + "CPU#%d: %u\n", c->cpu, c->not_preferred); + } + spin_unlock_irqrestore(&state_lock, flags); + + return count; +} + + +struct core_ctl_attr { + struct attribute attr; + ssize_t (*show)(const struct cluster_data *, char *); + ssize_t (*store)(struct cluster_data *, const char *, size_t count); +}; + +#define core_ctl_attr_ro(_name) \ +static struct core_ctl_attr _name = \ +__ATTR(_name, 0444, show_##_name, NULL) + +#define core_ctl_attr_rw(_name) \ +static struct core_ctl_attr _name = \ +__ATTR(_name, 0644, show_##_name, store_##_name) + +core_ctl_attr_rw(min_cpus); +core_ctl_attr_rw(max_cpus); +core_ctl_attr_rw(offline_delay_ms); +core_ctl_attr_rw(busy_up_thres); +core_ctl_attr_rw(busy_down_thres); +core_ctl_attr_rw(task_thres); +core_ctl_attr_rw(nr_prev_assist_thresh); +core_ctl_attr_ro(need_cpus); +core_ctl_attr_ro(active_cpus); +core_ctl_attr_ro(global_state); +core_ctl_attr_rw(not_preferred); +core_ctl_attr_rw(enable); + +static struct attribute *default_attrs[] = { + &min_cpus.attr, + &max_cpus.attr, + &offline_delay_ms.attr, + &busy_up_thres.attr, + &busy_down_thres.attr, + &task_thres.attr, + &nr_prev_assist_thresh.attr, + &enable.attr, + &need_cpus.attr, + &active_cpus.attr, + &global_state.attr, + ¬_preferred.attr, + NULL +}; + +#define to_cluster_data(k) container_of(k, struct cluster_data, kobj) +#define to_attr(a) container_of(a, struct core_ctl_attr, attr) +static ssize_t show(struct kobject *kobj, struct attribute *attr, char *buf) +{ + struct cluster_data *data = to_cluster_data(kobj); + struct core_ctl_attr *cattr = to_attr(attr); + ssize_t ret = -EIO; + + if (cattr->show) + ret = cattr->show(data, buf); + + return ret; +} + +static ssize_t store(struct kobject *kobj, struct attribute *attr, + const char *buf, size_t count) +{ + struct cluster_data *data = to_cluster_data(kobj); + struct core_ctl_attr *cattr = to_attr(attr); + ssize_t ret = -EIO; + + if (cattr->store) + ret = cattr->store(data, buf, count); + + return ret; +} + +static const struct sysfs_ops sysfs_ops = { + .show = show, + .store = store, +}; + +static struct kobj_type ktype_core_ctl = { + .sysfs_ops = &sysfs_ops, + .default_attrs = default_attrs, +}; + +/* ==================== runqueue based core count =================== */ + +static struct sched_avg_stats nr_stats[NR_CPUS]; + +/* + * nr_need: + * Number of tasks running on this cluster plus + * tasks running on higher capacity clusters. + * To find out CPUs needed from this cluster. + * + * For example: + * On dual cluster system with 4 min capacity + * CPUs and 4 max capacity CPUs, if there are + * 4 small tasks running on min capacity CPUs + * and 2 big tasks running on 2 max capacity + * CPUs, nr_need has to be 6 for min capacity + * cluster and 2 for max capacity cluster. + * This is because, min capacity cluster has to + * account for tasks running on max capacity + * cluster, so that, the min capacity cluster + * can be ready to accommodate tasks running on max + * capacity CPUs if the demand of tasks goes down. + */ +static int compute_cluster_nr_need(int index) +{ + int cpu; + struct cluster_data *cluster; + int nr_need = 0; + + for_each_cluster(cluster, index) { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_need += nr_stats[cpu].nr; + } + + return nr_need; +} + +/* + * prev_misfit_need: + * Tasks running on smaller capacity cluster which + * needs to be migrated to higher capacity cluster. + * To find out how many tasks need higher capacity CPUs. + * + * For example: + * On dual cluster system with 4 min capacity + * CPUs and 4 max capacity CPUs, if there are + * 2 small tasks and 2 big tasks running on + * min capacity CPUs and no tasks running on + * max cpacity, prev_misfit_need of min capacity + * cluster will be 0 and prev_misfit_need of + * max capacity cluster will be 2. + */ +static int compute_prev_cluster_misfit_need(int index) +{ + int cpu; + struct cluster_data *prev_cluster; + int prev_misfit_need = 0; + + /* + * Lowest capacity cluster does not have to + * accommodate any misfit tasks. + */ + if (index == 0) + return 0; + + prev_cluster = &cluster_state[index - 1]; + + for_each_cpu(cpu, &prev_cluster->cpu_mask) + prev_misfit_need += nr_stats[cpu].nr_misfit; + + return prev_misfit_need; +} + +static int compute_cluster_max_nr(int index) +{ + int cpu; + struct cluster_data *cluster = &cluster_state[index]; + int max_nr = 0; + + for_each_cpu(cpu, &cluster->cpu_mask) + max_nr = max(max_nr, nr_stats[cpu].nr_max); + + return max_nr; +} + +static int cluster_real_big_tasks(int index) +{ + int nr_big = 0; + int cpu; + struct cluster_data *cluster = &cluster_state[index]; + + if (index == 0) { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_big += nr_stats[cpu].nr_misfit; + } else { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_big += nr_stats[cpu].nr; + } + + return nr_big; +} + +/* + * prev_nr_need_assist: + * Tasks that are eligible to run on the previous + * cluster but cannot run because of insufficient + * CPUs there. prev_nr_need_assist is indicative + * of number of CPUs in this cluster that should + * assist its previous cluster to makeup for + * insufficient CPUs there. + * + * For example: + * On tri-cluster system with 4 min capacity + * CPUs, 3 intermediate capacity CPUs and 1 + * max capacity CPU, if there are 4 small + * tasks running on min capacity CPUs, 4 big + * tasks running on intermediate capacity CPUs + * and no tasks running on max capacity CPU, + * prev_nr_need_assist for min & max capacity + * clusters will be 0, but, for intermediate + * capacity cluster prev_nr_need_assist will + * be 1 as it has 3 CPUs, but, there are 4 big + * tasks to be served. + */ +static int prev_cluster_nr_need_assist(int index) +{ + int need = 0; + int cpu; + struct cluster_data *prev_cluster; + + if (index == 0) + return 0; + + index--; + prev_cluster = &cluster_state[index]; + + /* + * Next cluster should not assist, while there are isolated cpus + * in this cluster. + */ + if (prev_cluster->nr_isolated_cpus) + return 0; + + for_each_cpu(cpu, &prev_cluster->cpu_mask) + need += nr_stats[cpu].nr; + + need += compute_prev_cluster_misfit_need(index); + + if (need > prev_cluster->active_cpus) + need = need - prev_cluster->active_cpus; + else + need = 0; + + return need; +} + +/* + * This is only implemented for min capacity cluster. + * + * Bringing a little CPU out of isolation and using it + * more does not hurt power as much as bringing big CPUs. + * + * little cluster provides help needed for the other clusters. + * we take nr_scaled (which gives better resolution) and find + * the total nr in the system. Then take out the active higher + * capacity CPUs from the nr and consider the remaining nr as + * strict and consider that many little CPUs are needed. + */ +static int compute_cluster_nr_strict_need(int index) +{ + int cpu; + struct cluster_data *cluster; + int nr_strict_need = 0; + + if (index != 0) + return 0; + + for_each_cluster(cluster, index) { + int nr_scaled = 0; + int active_cpus = cluster->active_cpus; + + for_each_cpu(cpu, &cluster->cpu_mask) + nr_scaled += nr_stats[cpu].nr_scaled; + + nr_scaled /= 100; + + /* + * For little cluster, nr_scaled becomes the nr_strict, + * for other cluster, overflow is counted towards + * the little cluster need. + */ + if (index == 0) + nr_strict_need += nr_scaled; + else + nr_strict_need += max(0, nr_scaled - active_cpus); + } + + return nr_strict_need; +} +static void update_running_avg(void) +{ + struct cluster_data *cluster; + unsigned int index = 0; + unsigned long flags; + int big_avg = 0; + + sched_get_nr_running_avg(nr_stats); + + spin_lock_irqsave(&state_lock, flags); + for_each_cluster(cluster, index) { + int nr_need, prev_misfit_need; + + if (!cluster->inited) + continue; + + nr_need = compute_cluster_nr_need(index); + prev_misfit_need = compute_prev_cluster_misfit_need(index); + + + cluster->nrrun = nr_need + prev_misfit_need; + cluster->max_nr = compute_cluster_max_nr(index); + cluster->nr_prev_assist = prev_cluster_nr_need_assist(index); + + cluster->strict_nrrun = compute_cluster_nr_strict_need(index); + + trace_core_ctl_update_nr_need(cluster->first_cpu, nr_need, + prev_misfit_need, + cluster->nrrun, cluster->max_nr, + cluster->nr_prev_assist); + + big_avg += cluster_real_big_tasks(index); + } + spin_unlock_irqrestore(&state_lock, flags); + + last_nr_big = big_avg; + walt_rotation_checkpoint(big_avg); +} + +#define MAX_NR_THRESHOLD 4 +/* adjust needed CPUs based on current runqueue information */ +static unsigned int apply_task_need(const struct cluster_data *cluster, + unsigned int new_need) +{ + /* unisolate all cores if there are enough tasks */ + if (cluster->nrrun >= cluster->task_thres) + return cluster->num_cpus; + + /* + * unisolate as many cores as the previous cluster + * needs assistance with. + */ + if (cluster->nr_prev_assist >= cluster->nr_prev_assist_thresh) + new_need = new_need + cluster->nr_prev_assist; + + /* only unisolate more cores if there are tasks to run */ + if (cluster->nrrun > new_need) + new_need = new_need + 1; + + /* + * We don't want tasks to be overcrowded in a cluster. + * If any CPU has more than MAX_NR_THRESHOLD in the last + * window, bring another CPU to help out. + */ + if (cluster->max_nr > MAX_NR_THRESHOLD) + new_need = new_need + 1; + + /* + * For little cluster, we use a bit more relaxed approach + * and impose the strict nr condition. Because all tasks can + * spill onto little if big cluster is crowded. + */ + if (new_need < cluster->strict_nrrun) + new_need = cluster->strict_nrrun; + + return new_need; +} + +/* ======================= load based core count ====================== */ + +static unsigned int apply_limits(const struct cluster_data *cluster, + unsigned int need_cpus) +{ + return min(max(cluster->min_cpus, need_cpus), cluster->max_cpus); +} + +static unsigned int get_active_cpu_count(const struct cluster_data *cluster) +{ + return cluster->num_cpus - + sched_isolate_count(&cluster->cpu_mask, true); +} + +static bool is_active(const struct cpu_data *state) +{ + return cpu_online(state->cpu) && !cpu_isolated(state->cpu); +} + +static bool adjustment_possible(const struct cluster_data *cluster, + unsigned int need) +{ + return (need < cluster->active_cpus || (need > cluster->active_cpus && + cluster->nr_isolated_cpus)); +} + +static bool need_all_cpus(const struct cluster_data *cluster) +{ + return (is_min_capacity_cpu(cluster->first_cpu) && + sched_ravg_window < DEFAULT_SCHED_RAVG_WINDOW); +} + +static bool eval_need(struct cluster_data *cluster) +{ + unsigned long flags; + struct cpu_data *c; + unsigned int need_cpus = 0, last_need, thres_idx; + int ret = 0; + bool need_flag = false; + unsigned int new_need; + s64 now, elapsed; + + if (unlikely(!cluster->inited)) + return 0; + + spin_lock_irqsave(&state_lock, flags); + + if (cluster->boost || !cluster->enable || need_all_cpus(cluster)) { + need_cpus = cluster->max_cpus; + } else { + cluster->active_cpus = get_active_cpu_count(cluster); + thres_idx = cluster->active_cpus ? cluster->active_cpus - 1 : 0; + list_for_each_entry(c, &cluster->lru, sib) { + bool old_is_busy = c->is_busy; + + if (c->busy >= cluster->busy_up_thres[thres_idx] || + sched_cpu_high_irqload(c->cpu)) + c->is_busy = true; + else if (c->busy < cluster->busy_down_thres[thres_idx]) + c->is_busy = false; + + trace_core_ctl_set_busy(c->cpu, c->busy, old_is_busy, + c->is_busy); + need_cpus += c->is_busy; + } + need_cpus = apply_task_need(cluster, need_cpus); + } + new_need = apply_limits(cluster, need_cpus); + need_flag = adjustment_possible(cluster, new_need); + + last_need = cluster->need_cpus; + now = ktime_to_ms(ktime_get()); + + if (new_need > cluster->active_cpus) { + ret = 1; + } else { + /* + * When there is no change in need and there are no more + * active CPUs than currently needed, just update the + * need time stamp and return. + */ + if (new_need == last_need && new_need == cluster->active_cpus) { + cluster->need_ts = now; + spin_unlock_irqrestore(&state_lock, flags); + return 0; + } + + elapsed = now - cluster->need_ts; + ret = elapsed >= cluster->offline_delay_ms; + } + + if (ret) { + cluster->need_ts = now; + cluster->need_cpus = new_need; + } + trace_core_ctl_eval_need(cluster->first_cpu, last_need, new_need, + ret && need_flag); + spin_unlock_irqrestore(&state_lock, flags); + + return ret && need_flag; +} + +static void apply_need(struct cluster_data *cluster) +{ + if (eval_need(cluster)) + wake_up_core_ctl_thread(cluster); +} + +/* ========================= core count enforcement ==================== */ + +static void wake_up_core_ctl_thread(struct cluster_data *cluster) +{ + unsigned long flags; + + spin_lock_irqsave(&cluster->pending_lock, flags); + cluster->pending = true; + spin_unlock_irqrestore(&cluster->pending_lock, flags); + + wake_up_process(cluster->core_ctl_thread); +} + +static u64 core_ctl_check_timestamp; + +int core_ctl_set_boost(bool boost) +{ + unsigned int index = 0; + struct cluster_data *cluster = NULL; + unsigned long flags; + int ret = 0; + bool boost_state_changed = false; + + if (unlikely(!initialized)) + return 0; + + spin_lock_irqsave(&state_lock, flags); + for_each_cluster(cluster, index) { + if (boost) { + boost_state_changed = !cluster->boost; + ++cluster->boost; + } else { + if (!cluster->boost) { + ret = -EINVAL; + break; + } else { + --cluster->boost; + boost_state_changed = !cluster->boost; + } + } + } + spin_unlock_irqrestore(&state_lock, flags); + + if (boost_state_changed) { + index = 0; + for_each_cluster(cluster, index) + apply_need(cluster); + } + + if (cluster) + trace_core_ctl_set_boost(cluster->boost, ret); + + return ret; +} +EXPORT_SYMBOL(core_ctl_set_boost); + +void core_ctl_notifier_register(struct notifier_block *n) +{ + atomic_notifier_chain_register(&core_ctl_notifier, n); +} + +void core_ctl_notifier_unregister(struct notifier_block *n) +{ + atomic_notifier_chain_unregister(&core_ctl_notifier, n); +} + +static void core_ctl_call_notifier(void) +{ + struct core_ctl_notif_data ndata = {0}; + struct notifier_block *nb; + + /* + * Don't bother querying the stats when the notifier + * chain is empty. + */ + rcu_read_lock(); + nb = rcu_dereference_raw(core_ctl_notifier.head); + rcu_read_unlock(); + + if (!nb) + return; + + ndata.nr_big = last_nr_big; + walt_fill_ta_data(&ndata); + trace_core_ctl_notif_data(ndata.nr_big, ndata.coloc_load_pct, + ndata.ta_util_pct, ndata.cur_cap_pct); + + atomic_notifier_call_chain(&core_ctl_notifier, 0, &ndata); +} + +void core_ctl_check(u64 window_start) +{ + int cpu; + struct cpu_data *c; + struct cluster_data *cluster; + unsigned int index = 0; + unsigned long flags; + + if (unlikely(!initialized)) + return; + + if (window_start == core_ctl_check_timestamp) + return; + + core_ctl_check_timestamp = window_start; + + spin_lock_irqsave(&state_lock, flags); + for_each_possible_cpu(cpu) { + + c = &per_cpu(cpu_state, cpu); + cluster = c->cluster; + + if (!cluster || !cluster->inited) + continue; + + c->busy = sched_get_cpu_util(cpu); + } + spin_unlock_irqrestore(&state_lock, flags); + + update_running_avg(); + + for_each_cluster(cluster, index) { + if (eval_need(cluster)) + wake_up_core_ctl_thread(cluster); + } + + core_ctl_call_notifier(); +} + +static void move_cpu_lru(struct cpu_data *cpu_data) +{ + unsigned long flags; + + spin_lock_irqsave(&state_lock, flags); + list_del(&cpu_data->sib); + list_add_tail(&cpu_data->sib, &cpu_data->cluster->lru); + spin_unlock_irqrestore(&state_lock, flags); +} + +static bool should_we_isolate(int cpu, struct cluster_data *cluster) +{ + return true; +} + +static void try_to_isolate(struct cluster_data *cluster, unsigned int need) +{ + struct cpu_data *c, *tmp; + unsigned long flags; + unsigned int num_cpus = cluster->num_cpus; + unsigned int nr_isolated = 0; + bool first_pass = cluster->nr_not_preferred_cpus; + + /* + * Protect against entry being removed (and added at tail) by other + * thread (hotplug). + */ + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!is_active(c)) + continue; + if (cluster->active_cpus == need) + break; + /* Don't isolate busy CPUs. */ + if (c->is_busy) + continue; + + /* + * We isolate only the not_preferred CPUs. If none + * of the CPUs are selected as not_preferred, then + * all CPUs are eligible for isolation. + */ + if (cluster->nr_not_preferred_cpus && !c->not_preferred) + continue; + + if (!should_we_isolate(c->cpu, cluster)) + continue; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to isolate CPU%u\n", c->cpu); + if (!sched_isolate_cpu(c->cpu)) { + c->isolated_by_us = true; + move_cpu_lru(c); + nr_isolated++; + } else { + pr_debug("Unable to isolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus += nr_isolated; + spin_unlock_irqrestore(&state_lock, flags); + +again: + /* + * If the number of active CPUs is within the limits, then + * don't force isolation of any busy CPUs. + */ + if (cluster->active_cpus <= cluster->max_cpus) + return; + + nr_isolated = 0; + num_cpus = cluster->num_cpus; + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!is_active(c)) + continue; + if (cluster->active_cpus <= cluster->max_cpus) + break; + + if (first_pass && !c->not_preferred) + continue; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to isolate CPU%u\n", c->cpu); + if (!sched_isolate_cpu(c->cpu)) { + c->isolated_by_us = true; + move_cpu_lru(c); + nr_isolated++; + } else { + pr_debug("Unable to isolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus += nr_isolated; + spin_unlock_irqrestore(&state_lock, flags); + + if (first_pass && cluster->active_cpus > cluster->max_cpus) { + first_pass = false; + goto again; + } +} + +static void __try_to_unisolate(struct cluster_data *cluster, + unsigned int need, bool force) +{ + struct cpu_data *c, *tmp; + unsigned long flags; + unsigned int num_cpus = cluster->num_cpus; + unsigned int nr_unisolated = 0; + + /* + * Protect against entry being removed (and added at tail) by other + * thread (hotplug). + */ + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!c->isolated_by_us) + continue; + if ((cpu_online(c->cpu) && !cpu_isolated(c->cpu)) || + (!force && c->not_preferred)) + continue; + if (cluster->active_cpus == need) + break; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to unisolate CPU%u\n", c->cpu); + if (!sched_unisolate_cpu(c->cpu)) { + c->isolated_by_us = false; + move_cpu_lru(c); + nr_unisolated++; + } else { + pr_debug("Unable to unisolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus -= nr_unisolated; + spin_unlock_irqrestore(&state_lock, flags); +} + +static void try_to_unisolate(struct cluster_data *cluster, unsigned int need) +{ + bool force_use_non_preferred = false; + + __try_to_unisolate(cluster, need, force_use_non_preferred); + + if (cluster->active_cpus == need) + return; + + force_use_non_preferred = true; + __try_to_unisolate(cluster, need, force_use_non_preferred); +} + +static void __ref do_core_ctl(struct cluster_data *cluster) +{ + unsigned int need; + + need = apply_limits(cluster, cluster->need_cpus); + + if (adjustment_possible(cluster, need)) { + pr_debug("Trying to adjust group %u from %u to %u\n", + cluster->first_cpu, cluster->active_cpus, need); + + if (cluster->active_cpus > need) + try_to_isolate(cluster, need); + else if (cluster->active_cpus < need) + try_to_unisolate(cluster, need); + } +} + +static int __ref try_core_ctl(void *data) +{ + struct cluster_data *cluster = data; + unsigned long flags; + + while (1) { + set_current_state(TASK_INTERRUPTIBLE); + spin_lock_irqsave(&cluster->pending_lock, flags); + if (!cluster->pending) { + spin_unlock_irqrestore(&cluster->pending_lock, flags); + schedule(); + if (kthread_should_stop()) + break; + spin_lock_irqsave(&cluster->pending_lock, flags); + } + set_current_state(TASK_RUNNING); + cluster->pending = false; + spin_unlock_irqrestore(&cluster->pending_lock, flags); + + do_core_ctl(cluster); + } + + return 0; +} + +static int isolation_cpuhp_state(unsigned int cpu, bool online) +{ + struct cpu_data *state = &per_cpu(cpu_state, cpu); + struct cluster_data *cluster = state->cluster; + unsigned int need; + bool do_wakeup = false, unisolated = false; + unsigned long flags; + + if (unlikely(!cluster || !cluster->inited)) + return 0; + + if (online) { + cluster->active_cpus = get_active_cpu_count(cluster); + + /* + * Moving to the end of the list should only happen in + * CPU_ONLINE and not on CPU_UP_PREPARE to prevent an + * infinite list traversal when thermal (or other entities) + * reject trying to online CPUs. + */ + move_cpu_lru(state); + } else { + /* + * We don't want to have a CPU both offline and isolated. + * So unisolate a CPU that went down if it was isolated by us. + */ + if (state->isolated_by_us) { + sched_unisolate_cpu_unlocked(cpu); + state->isolated_by_us = false; + unisolated = true; + } + + /* Move a CPU to the end of the LRU when it goes offline. */ + move_cpu_lru(state); + + state->busy = 0; + cluster->active_cpus = get_active_cpu_count(cluster); + } + + need = apply_limits(cluster, cluster->need_cpus); + spin_lock_irqsave(&state_lock, flags); + if (unisolated) + cluster->nr_isolated_cpus--; + do_wakeup = adjustment_possible(cluster, need); + spin_unlock_irqrestore(&state_lock, flags); + if (do_wakeup) + wake_up_core_ctl_thread(cluster); + + return 0; +} + +static int core_ctl_isolation_online_cpu(unsigned int cpu) +{ + return isolation_cpuhp_state(cpu, true); +} + +static int core_ctl_isolation_dead_cpu(unsigned int cpu) +{ + return isolation_cpuhp_state(cpu, false); +} + +/* ============================ init code ============================== */ + +static struct cluster_data *find_cluster_by_first_cpu(unsigned int first_cpu) +{ + unsigned int i; + + for (i = 0; i < num_clusters; ++i) { + if (cluster_state[i].first_cpu == first_cpu) + return &cluster_state[i]; + } + + return NULL; +} + +static int cluster_init(const struct cpumask *mask) +{ + struct device *dev; + unsigned int first_cpu = cpumask_first(mask); + struct cluster_data *cluster; + struct cpu_data *state; + unsigned int cpu; + struct sched_param param = { .sched_priority = MAX_RT_PRIO-1 }; + + if (find_cluster_by_first_cpu(first_cpu)) + return 0; + + dev = get_cpu_device(first_cpu); + if (!dev) + return -ENODEV; + + pr_info("Creating CPU group %d\n", first_cpu); + + if (num_clusters == MAX_CLUSTERS) { + pr_err("Unsupported number of clusters. Only %u supported\n", + MAX_CLUSTERS); + return -EINVAL; + } + cluster = &cluster_state[num_clusters]; + ++num_clusters; + + cpumask_copy(&cluster->cpu_mask, mask); + cluster->num_cpus = cpumask_weight(mask); + if (cluster->num_cpus > MAX_CPUS_PER_CLUSTER) { + pr_err("HW configuration not supported\n"); + return -EINVAL; + } + cluster->first_cpu = first_cpu; + cluster->min_cpus = 1; + cluster->max_cpus = cluster->num_cpus; + cluster->need_cpus = cluster->num_cpus; + cluster->offline_delay_ms = 100; + cluster->task_thres = UINT_MAX; + cluster->nr_prev_assist_thresh = UINT_MAX; + cluster->nrrun = cluster->num_cpus; + cluster->enable = true; + cluster->nr_not_preferred_cpus = 0; + cluster->strict_nrrun = 0; + INIT_LIST_HEAD(&cluster->lru); + spin_lock_init(&cluster->pending_lock); + + for_each_cpu(cpu, mask) { + pr_info("Init CPU%u state\n", cpu); + + state = &per_cpu(cpu_state, cpu); + state->cluster = cluster; + state->cpu = cpu; + list_add_tail(&state->sib, &cluster->lru); + } + cluster->active_cpus = get_active_cpu_count(cluster); + + cluster->core_ctl_thread = kthread_run(try_core_ctl, (void *) cluster, + "core_ctl/%d", first_cpu); + if (IS_ERR(cluster->core_ctl_thread)) + return PTR_ERR(cluster->core_ctl_thread); + + sched_setscheduler_nocheck(cluster->core_ctl_thread, SCHED_FIFO, + ¶m); + + cluster->inited = true; + + kobject_init(&cluster->kobj, &ktype_core_ctl); + return kobject_add(&cluster->kobj, &dev->kobj, "core_ctl"); +} + +static int __init core_ctl_init(void) +{ + struct walt_sched_cluster *cluster; + int ret; + + cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, + "core_ctl/isolation:online", + core_ctl_isolation_online_cpu, NULL); + + cpuhp_setup_state_nocalls(CPUHP_CORE_CTL_ISOLATION_DEAD, + "core_ctl/isolation:dead", + NULL, core_ctl_isolation_dead_cpu); + + for_each_sched_cluster(cluster) { + ret = cluster_init(&cluster->cpus); + if (ret) + pr_warn("unable to create core ctl group: %d\n", ret); + } + + initialized = true; + return 0; +} + +late_initcall(core_ctl_init); diff --git a/kernel/sched/walt/cpu-boost.c b/kernel/sched/walt/cpu-boost.c new file mode 100644 index 000000000000..006510c55f13 --- /dev/null +++ b/kernel/sched/walt/cpu-boost.c @@ -0,0 +1,389 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2013-2015,2017,2019-2021, The Linux Foundation. All rights reserved. + */ +#define pr_fmt(fmt) "cpu-boost: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "qc_vas.h" + +#define cpu_boost_attr_rw(_name) \ +static struct kobj_attribute _name##_attr = \ +__ATTR(_name, 0644, show_##_name, store_##_name) + +#define show_one(file_name) \ +static ssize_t show_##file_name \ +(struct kobject *kobj, struct kobj_attribute *attr, char *buf) \ +{ \ + return scnprintf(buf, PAGE_SIZE, "%u\n", file_name); \ +} + +#define store_one(file_name) \ +static ssize_t store_##file_name \ +(struct kobject *kobj, struct kobj_attribute *attr, \ +const char *buf, size_t count) \ +{ \ + \ + sscanf(buf, "%u", &file_name); \ + return count; \ +} + +struct cpu_sync { + int cpu; + unsigned int input_boost_min; + unsigned int input_boost_freq; +}; + +static DEFINE_PER_CPU(struct cpu_sync, sync_info); +static struct workqueue_struct *cpu_boost_wq; + +static struct work_struct input_boost_work; + +static bool input_boost_enabled; + +static unsigned int input_boost_ms = 40; +show_one(input_boost_ms); +store_one(input_boost_ms); +cpu_boost_attr_rw(input_boost_ms); + +static unsigned int sched_boost_on_input; +show_one(sched_boost_on_input); +store_one(sched_boost_on_input); +cpu_boost_attr_rw(sched_boost_on_input); + +static bool sched_boost_active; + +static struct delayed_work input_boost_rem; +static u64 last_input_time; +#define MIN_INPUT_INTERVAL (150 * USEC_PER_MSEC) + +static DEFINE_PER_CPU(struct freq_qos_request, qos_req); + +static ssize_t store_input_boost_freq(struct kobject *kobj, + struct kobj_attribute *attr, + const char *buf, size_t count) +{ + int i, ntokens = 0; + unsigned int val, cpu; + const char *cp = buf; + bool enabled = false; + + while ((cp = strpbrk(cp + 1, " :"))) + ntokens++; + + /* single number: apply to all CPUs */ + if (!ntokens) { + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + for_each_possible_cpu(i) + per_cpu(sync_info, i).input_boost_freq = val; + goto check_enable; + } + + /* CPU:value pair */ + if (!(ntokens % 2)) + return -EINVAL; + + cp = buf; + for (i = 0; i < ntokens; i += 2) { + if (sscanf(cp, "%u:%u", &cpu, &val) != 2) + return -EINVAL; + if (cpu >= num_possible_cpus()) + return -EINVAL; + + per_cpu(sync_info, cpu).input_boost_freq = val; + cp = strnchr(cp, PAGE_SIZE - (cp - buf), ' '); + cp++; + } + +check_enable: + for_each_possible_cpu(i) { + if (per_cpu(sync_info, i).input_boost_freq) { + enabled = true; + break; + } + } + input_boost_enabled = enabled; + + return count; +} + +static ssize_t show_input_boost_freq(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + int cnt = 0, cpu; + struct cpu_sync *s; + + for_each_possible_cpu(cpu) { + s = &per_cpu(sync_info, cpu); + cnt += snprintf(buf + cnt, PAGE_SIZE - cnt, + "%d:%u ", cpu, s->input_boost_freq); + } + cnt += snprintf(buf + cnt, PAGE_SIZE - cnt, "\n"); + return cnt; +} + +cpu_boost_attr_rw(input_boost_freq); + +static void boost_adjust_notify(struct cpufreq_policy *policy) +{ + unsigned int cpu = policy->cpu; + struct cpu_sync *s = &per_cpu(sync_info, cpu); + unsigned int ib_min = s->input_boost_min; + struct freq_qos_request *req = &per_cpu(qos_req, cpu); + int ret; + + pr_debug("CPU%u policy min before boost: %u kHz\n", + cpu, policy->min); + pr_debug("CPU%u boost min: %u kHz\n", cpu, ib_min); + + ret = freq_qos_update_request(req, ib_min); + + if (ret < 0) + pr_err("Failed to update freq constraint in boost_adjust: %d\n", + ib_min); + + pr_debug("CPU%u policy min after boost: %u kHz\n", + cpu, policy->min); + + return; +} + +static void update_policy_online(void) +{ + unsigned int i; + struct cpufreq_policy *policy; + struct cpumask online_cpus; + /* Re-evaluate policy to trigger adjust notifier for online CPUs */ + get_online_cpus(); + online_cpus = *cpu_online_mask; + for_each_cpu(i, &online_cpus) { + policy = cpufreq_cpu_get(i); + if (!policy) { + pr_err("%s: cpufreq policy not found for cpu%d\n", + __func__, i); + return; + } + + cpumask_andnot(&online_cpus, &online_cpus, + policy->related_cpus); + boost_adjust_notify(policy); + } + put_online_cpus(); +} + +static void do_input_boost_rem(struct work_struct *work) +{ + unsigned int i, ret; + struct cpu_sync *i_sync_info; + + /* Reset the input_boost_min for all CPUs in the system */ + pr_debug("Resetting input boost min for all CPUs\n"); + for_each_possible_cpu(i) { + i_sync_info = &per_cpu(sync_info, i); + i_sync_info->input_boost_min = 0; + } + + /* Update policies for all online CPUs */ + update_policy_online(); + + if (sched_boost_active) { + ret = sched_set_boost(0); + if (ret) + pr_err("cpu-boost: sched boost disable failed\n"); + sched_boost_active = false; + } +} + +static void do_input_boost(struct work_struct *work) +{ + unsigned int i, ret; + struct cpu_sync *i_sync_info; + + cancel_delayed_work_sync(&input_boost_rem); + if (sched_boost_active) { + sched_set_boost(0); + sched_boost_active = false; + } + + /* Set the input_boost_min for all CPUs in the system */ + pr_debug("Setting input boost min for all CPUs\n"); + for_each_possible_cpu(i) { + i_sync_info = &per_cpu(sync_info, i); + i_sync_info->input_boost_min = i_sync_info->input_boost_freq; + } + + /* Update policies for all online CPUs */ + update_policy_online(); + + /* Enable scheduler boost to migrate tasks to big cluster */ + if (sched_boost_on_input > 0) { + ret = sched_set_boost(sched_boost_on_input); + if (ret) + pr_err("cpu-boost: sched boost enable failed\n"); + else + sched_boost_active = true; + } + + queue_delayed_work(cpu_boost_wq, &input_boost_rem, + msecs_to_jiffies(input_boost_ms)); +} + +static void cpuboost_input_event(struct input_handle *handle, + unsigned int type, unsigned int code, int value) +{ + u64 now; + + if (!input_boost_enabled) + return; + + now = ktime_to_us(ktime_get()); + if (now - last_input_time < MIN_INPUT_INTERVAL) + return; + + if (work_pending(&input_boost_work)) + return; + + queue_work(cpu_boost_wq, &input_boost_work); + last_input_time = ktime_to_us(ktime_get()); +} + +static int cpuboost_input_connect(struct input_handler *handler, + struct input_dev *dev, const struct input_device_id *id) +{ + struct input_handle *handle; + int error; + + handle = kzalloc(sizeof(struct input_handle), GFP_KERNEL); + if (!handle) + return -ENOMEM; + + handle->dev = dev; + handle->handler = handler; + handle->name = "cpufreq"; + + error = input_register_handle(handle); + if (error) + goto err2; + + error = input_open_device(handle); + if (error) + goto err1; + + return 0; +err1: + input_unregister_handle(handle); +err2: + kfree(handle); + return error; +} + +static void cpuboost_input_disconnect(struct input_handle *handle) +{ + input_close_device(handle); + input_unregister_handle(handle); + kfree(handle); +} + +static const struct input_device_id cpuboost_ids[] = { + /* multi-touch touchscreen */ + { + .flags = INPUT_DEVICE_ID_MATCH_EVBIT | + INPUT_DEVICE_ID_MATCH_ABSBIT, + .evbit = { BIT_MASK(EV_ABS) }, + .absbit = { [BIT_WORD(ABS_MT_POSITION_X)] = + BIT_MASK(ABS_MT_POSITION_X) | + BIT_MASK(ABS_MT_POSITION_Y) }, + }, + /* touchpad */ + { + .flags = INPUT_DEVICE_ID_MATCH_KEYBIT | + INPUT_DEVICE_ID_MATCH_ABSBIT, + .keybit = { [BIT_WORD(BTN_TOUCH)] = BIT_MASK(BTN_TOUCH) }, + .absbit = { [BIT_WORD(ABS_X)] = + BIT_MASK(ABS_X) | BIT_MASK(ABS_Y) }, + }, + /* Keypad */ + { + .flags = INPUT_DEVICE_ID_MATCH_EVBIT, + .evbit = { BIT_MASK(EV_KEY) }, + }, + { }, +}; + +static struct input_handler cpuboost_input_handler = { + .event = cpuboost_input_event, + .connect = cpuboost_input_connect, + .disconnect = cpuboost_input_disconnect, + .name = "cpu-boost", + .id_table = cpuboost_ids, +}; + +struct kobject *cpu_boost_kobj; +static int cpu_boost_init(void) +{ + int cpu, ret; + struct cpu_sync *s; + struct cpufreq_policy *policy; + struct freq_qos_request *req; + + cpu_boost_wq = alloc_workqueue("cpuboost_wq", WQ_HIGHPRI, 0); + if (!cpu_boost_wq) + return -EFAULT; + + INIT_WORK(&input_boost_work, do_input_boost); + INIT_DELAYED_WORK(&input_boost_rem, do_input_boost_rem); + + for_each_possible_cpu(cpu) { + s = &per_cpu(sync_info, cpu); + s->cpu = cpu; + req = &per_cpu(qos_req, cpu); + policy = cpufreq_cpu_get(cpu); + if (!policy) { + pr_err("%s: cpufreq policy not found for cpu%d\n", + __func__, cpu); + return -ESRCH; + } + + ret = freq_qos_add_request(&policy->constraints, req, + FREQ_QOS_MIN, policy->min); + if (ret < 0) { + pr_err("%s: Failed to add freq constraint (%d)\n", + __func__, ret); + return ret; + } + + } + + cpu_boost_kobj = kobject_create_and_add("cpu_boost", + &cpu_subsys.dev_root->kobj); + if (!cpu_boost_kobj) + pr_err("Failed to initialize sysfs node for cpu_boost.\n"); + + ret = sysfs_create_file(cpu_boost_kobj, &input_boost_ms_attr.attr); + if (ret) + pr_err("Failed to create input_boost_ms node: %d\n", ret); + + ret = sysfs_create_file(cpu_boost_kobj, &input_boost_freq_attr.attr); + if (ret) + pr_err("Failed to create input_boost_freq node: %d\n", ret); + + ret = sysfs_create_file(cpu_boost_kobj, + &sched_boost_on_input_attr.attr); + if (ret) + pr_err("Failed to create sched_boost_on_input node: %d\n", ret); + + ret = input_register_handler(&cpuboost_input_handler); + return 0; +} +late_initcall(cpu_boost_init); diff --git a/kernel/sched/walt/qc_vas.c b/kernel/sched/walt/qc_vas.c new file mode 100644 index 000000000000..d4d6c838babe --- /dev/null +++ b/kernel/sched/walt/qc_vas.c @@ -0,0 +1,744 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ +#include +#include +#include + +#include "qc_vas.h" + +#ifdef CONFIG_SCHED_WALT +/* 1ms default for 20ms window size scaled to 1024 */ +unsigned int sysctl_sched_min_task_util_for_boost = 51; +/* 0.68ms default for 20ms window size scaled to 1024 */ +unsigned int sysctl_sched_min_task_util_for_colocation = 35; + +int +kick_active_balance(struct rq *rq, struct task_struct *p, int new_cpu) +{ + unsigned long flags; + int rc = 0; + + /* Invoke active balance to force migrate currently running task */ + raw_spin_lock_irqsave(&rq->lock, flags); + if (!rq->active_balance) { + rq->active_balance = 1; + rq->push_cpu = new_cpu; + get_task_struct(p); + rq->wrq.push_task = p; + rc = 1; + } + raw_spin_unlock_irqrestore(&rq->lock, flags); + + return rc; +} + +struct walt_rotate_work { + struct work_struct w; + struct task_struct *src_task; + struct task_struct *dst_task; + int src_cpu; + int dst_cpu; +}; + +DEFINE_PER_CPU(struct walt_rotate_work, walt_rotate_works); + +void walt_rotate_work_func(struct work_struct *work) +{ + struct walt_rotate_work *wr = container_of(work, + struct walt_rotate_work, w); + + migrate_swap(wr->src_task, wr->dst_task, wr->dst_cpu, wr->src_cpu); + + put_task_struct(wr->src_task); + put_task_struct(wr->dst_task); + + clear_reserved(wr->src_cpu); + clear_reserved(wr->dst_cpu); +} + +void walt_rotate_work_init(void) +{ + int i; + + for_each_possible_cpu(i) { + struct walt_rotate_work *wr = &per_cpu(walt_rotate_works, i); + + INIT_WORK(&wr->w, walt_rotate_work_func); + } +} + +#define WALT_ROTATION_THRESHOLD_NS 16000000 +void walt_check_for_rotation(struct rq *src_rq) +{ + u64 wc, wait, max_wait = 0, run, max_run = 0; + int deserved_cpu = nr_cpu_ids, dst_cpu = nr_cpu_ids; + int i, src_cpu = cpu_of(src_rq); + struct rq *dst_rq; + struct walt_rotate_work *wr = NULL; + + if (!walt_rotation_enabled) + return; + + if (!is_min_capacity_cpu(src_cpu)) + return; + + wc = sched_ktime_clock(); + for_each_possible_cpu(i) { + struct rq *rq = cpu_rq(i); + + if (!is_min_capacity_cpu(i)) + break; + + if (is_reserved(i)) + continue; + + if (!rq->misfit_task_load || rq->curr->sched_class != + &fair_sched_class) + continue; + + wait = wc - rq->curr->wts.last_enqueued_ts; + if (wait > max_wait) { + max_wait = wait; + deserved_cpu = i; + } + } + + if (deserved_cpu != src_cpu) + return; + + for_each_possible_cpu(i) { + struct rq *rq = cpu_rq(i); + + if (is_min_capacity_cpu(i)) + continue; + + if (is_reserved(i)) + continue; + + if (rq->curr->sched_class != &fair_sched_class) + continue; + + if (rq->nr_running > 1) + continue; + + run = wc - rq->curr->wts.last_enqueued_ts; + + if (run < WALT_ROTATION_THRESHOLD_NS) + continue; + + if (run > max_run) { + max_run = run; + dst_cpu = i; + } + } + + if (dst_cpu == nr_cpu_ids) + return; + + dst_rq = cpu_rq(dst_cpu); + + double_rq_lock(src_rq, dst_rq); + if (dst_rq->curr->sched_class == &fair_sched_class) { + get_task_struct(src_rq->curr); + get_task_struct(dst_rq->curr); + + mark_reserved(src_cpu); + mark_reserved(dst_cpu); + wr = &per_cpu(walt_rotate_works, src_cpu); + + wr->src_task = src_rq->curr; + wr->dst_task = dst_rq->curr; + + wr->src_cpu = src_cpu; + wr->dst_cpu = dst_cpu; + } + double_rq_unlock(src_rq, dst_rq); + + if (wr) + queue_work_on(src_cpu, system_highpri_wq, &wr->w); +} + +DEFINE_RAW_SPINLOCK(migration_lock); +void check_for_migration(struct rq *rq, struct task_struct *p) +{ + int active_balance; + int new_cpu = -1; + int prev_cpu = task_cpu(p); + int ret; + + if (rq->misfit_task_load) { + if (rq->curr->state != TASK_RUNNING || + rq->curr->nr_cpus_allowed == 1) + return; + + if (walt_rotation_enabled) { + raw_spin_lock(&migration_lock); + walt_check_for_rotation(rq); + raw_spin_unlock(&migration_lock); + return; + } + + raw_spin_lock(&migration_lock); + rcu_read_lock(); + new_cpu = find_energy_efficient_cpu(p, prev_cpu, 0, 1); + rcu_read_unlock(); + if ((new_cpu >= 0) && (new_cpu != prev_cpu) && + (capacity_orig_of(new_cpu) > capacity_orig_of(prev_cpu))) { + active_balance = kick_active_balance(rq, p, new_cpu); + if (active_balance) { + mark_reserved(new_cpu); + raw_spin_unlock(&migration_lock); + ret = stop_one_cpu_nowait(prev_cpu, + active_load_balance_cpu_stop, rq, + &rq->active_balance_work); + if (!ret) + clear_reserved(new_cpu); + else + wake_up_if_idle(new_cpu); + return; + } + } + raw_spin_unlock(&migration_lock); + } +} + +int sched_init_task_load_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_init_task_load(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_init_task_load_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int init_task_load, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &init_task_load); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_init_task_load(p, init_task_load); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_init_task_load_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_init_task_load_show, inode); +} + +int sched_group_id_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_group_id(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_group_id_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int group_id, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &group_id); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_group_id(p, group_id); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_group_id_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_group_id_show, inode); +} + +#ifdef CONFIG_SMP +/* + * Print out various scheduling related per-task fields: + */ +int sched_wake_up_idle_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_wake_up_idle(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_wake_up_idle_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int wake_up_idle, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &wake_up_idle); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_wake_up_idle(p, wake_up_idle); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_wake_up_idle_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_wake_up_idle_show, inode); +} + +int group_balance_cpu_not_isolated(struct sched_group *sg) +{ + cpumask_t cpus; + + cpumask_and(&cpus, sched_group_span(sg), group_balance_mask(sg)); + cpumask_andnot(&cpus, &cpus, cpu_isolated_mask); + return cpumask_first(&cpus); +} +#endif /* CONFIG_SMP */ + +#ifdef CONFIG_PROC_SYSCTL +static void sched_update_updown_migrate_values(bool up) +{ + int i = 0, cpu; + struct walt_sched_cluster *cluster; + int cap_margin_levels = num_sched_clusters - 1; + + if (cap_margin_levels > 1) { + /* + * No need to worry about CPUs in last cluster + * if there are more than 2 clusters in the system + */ + for_each_sched_cluster(cluster) { + for_each_cpu(cpu, &cluster->cpus) { + if (up) + sched_capacity_margin_up[cpu] = + sysctl_sched_capacity_margin_up[i]; + else + sched_capacity_margin_down[cpu] = + sysctl_sched_capacity_margin_down[i]; + } + + if (++i >= cap_margin_levels) + break; + } + } else { + for_each_possible_cpu(cpu) { + if (up) + sched_capacity_margin_up[cpu] = + sysctl_sched_capacity_margin_up[0]; + else + sched_capacity_margin_down[cpu] = + sysctl_sched_capacity_margin_down[0]; + } + } +} + +int sched_updown_migrate_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret, i; + unsigned int *data = (unsigned int *)table->data; + unsigned int *old_val; + static DEFINE_MUTEX(mutex); + int cap_margin_levels = num_sched_clusters ? num_sched_clusters - 1 : 0; + + if (cap_margin_levels <= 0) + return -EINVAL; + + mutex_lock(&mutex); + + if (table->maxlen != (sizeof(unsigned int) * cap_margin_levels)) + table->maxlen = sizeof(unsigned int) * cap_margin_levels; + + if (!write) { + ret = proc_douintvec_capacity(table, write, buffer, lenp, ppos); + goto unlock_mutex; + } + + /* + * Cache the old values so that they can be restored + * if either the write fails (for example out of range values) + * or the downmigrate and upmigrate are not in sync. + */ + old_val = kzalloc(table->maxlen, GFP_KERNEL); + if (!old_val) { + ret = -ENOMEM; + goto unlock_mutex; + } + + memcpy(old_val, data, table->maxlen); + + ret = proc_douintvec_capacity(table, write, buffer, lenp, ppos); + + if (ret) { + memcpy(data, old_val, table->maxlen); + goto free_old_val; + } + + for (i = 0; i < cap_margin_levels; i++) { + if (sysctl_sched_capacity_margin_up[i] > + sysctl_sched_capacity_margin_down[i]) { + memcpy(data, old_val, table->maxlen); + ret = -EINVAL; + goto free_old_val; + } + } + + sched_update_updown_migrate_values(data == + &sysctl_sched_capacity_margin_up[0]); + +free_old_val: + kfree(old_val); +unlock_mutex: + mutex_unlock(&mutex); + + return ret; +} +#endif /* CONFIG_PROC_SYSCTL */ + +int sched_isolate_count(const cpumask_t *mask, bool include_offline) +{ + cpumask_t count_mask = CPU_MASK_NONE; + + if (include_offline) { + cpumask_complement(&count_mask, cpu_online_mask); + cpumask_or(&count_mask, &count_mask, cpu_isolated_mask); + cpumask_and(&count_mask, &count_mask, mask); + } else { + cpumask_and(&count_mask, mask, cpu_isolated_mask); + } + + return cpumask_weight(&count_mask); +} + +#ifdef CONFIG_HOTPLUG_CPU +static int do_isolation_work_cpu_stop(void *data) +{ + unsigned int cpu = smp_processor_id(); + struct rq *rq = cpu_rq(cpu); + struct rq_flags rf; + + local_irq_disable(); + + irq_migrate_all_off_this_cpu(); + + sched_ttwu_pending(); + + /* Update our root-domain */ + rq_lock(rq, &rf); + + /* + * Temporarily mark the rq as offline. This will allow us to + * move tasks off the CPU. + */ + if (rq->rd) { + BUG_ON(!cpumask_test_cpu(cpu, rq->rd->span)); + set_rq_offline(rq); + } + + migrate_tasks(rq, &rf, false); + + if (rq->rd) + set_rq_online(rq); + rq_unlock(rq, &rf); + + clear_walt_request(cpu); + local_irq_enable(); + return 0; +} + +static int do_unisolation_work_cpu_stop(void *data) +{ + watchdog_enable(smp_processor_id()); + return 0; +} + +static void sched_update_group_capacities(int cpu) +{ + struct sched_domain *sd; + + mutex_lock(&sched_domains_mutex); + rcu_read_lock(); + + for_each_domain(cpu, sd) { + int balance_cpu = group_balance_cpu(sd->groups); + + init_sched_groups_capacity(cpu, sd); + /* + * Need to ensure this is also called with balancing + * cpu. + */ + if (cpu != balance_cpu) + init_sched_groups_capacity(balance_cpu, sd); + } + + rcu_read_unlock(); + mutex_unlock(&sched_domains_mutex); +} + +static unsigned int cpu_isolation_vote[NR_CPUS]; + +/* + * 1) CPU is isolated and cpu is offlined: + * Unisolate the core. + * 2) CPU is not isolated and CPU is offlined: + * No action taken. + * 3) CPU is offline and request to isolate + * Request ignored. + * 4) CPU is offline and isolated: + * Not a possible state. + * 5) CPU is online and request to isolate + * Normal case: Isolate the CPU + * 6) CPU is not isolated and comes back online + * Nothing to do + * + * Note: The client calling sched_isolate_cpu() is repsonsible for ONLY + * calling sched_unisolate_cpu() on a CPU that the client previously isolated. + * Client is also responsible for unisolating when a core goes offline + * (after CPU is marked offline). + */ +int sched_isolate_cpu(int cpu) +{ + struct rq *rq; + cpumask_t avail_cpus; + int ret_code = 0; + u64 start_time = 0; + + if (trace_sched_isolate_enabled()) + start_time = sched_clock(); + + cpu_maps_update_begin(); + + cpumask_andnot(&avail_cpus, cpu_online_mask, cpu_isolated_mask); + + if (cpu < 0 || cpu >= nr_cpu_ids || !cpu_possible(cpu) || + !cpu_online(cpu) || cpu >= NR_CPUS) { + ret_code = -EINVAL; + goto out; + } + + rq = cpu_rq(cpu); + + if (++cpu_isolation_vote[cpu] > 1) + goto out; + + /* We cannot isolate ALL cpus in the system */ + if (cpumask_weight(&avail_cpus) == 1) { + --cpu_isolation_vote[cpu]; + ret_code = -EINVAL; + goto out; + } + + /* + * There is a race between watchdog being enabled by hotplug and + * core isolation disabling the watchdog. When a CPU is hotplugged in + * and the hotplug lock has been released the watchdog thread might + * not have run yet to enable the watchdog. + * We have to wait for the watchdog to be enabled before proceeding. + */ + if (!watchdog_configured(cpu)) { + msleep(20); + if (!watchdog_configured(cpu)) { + --cpu_isolation_vote[cpu]; + ret_code = -EBUSY; + goto out; + } + } + + set_cpu_isolated(cpu, true); + cpumask_clear_cpu(cpu, &avail_cpus); + + /* Migrate timers */ + smp_call_function_any(&avail_cpus, hrtimer_quiesce_cpu, &cpu, 1); + smp_call_function_any(&avail_cpus, timer_quiesce_cpu, &cpu, 1); + + watchdog_disable(cpu); + irq_lock_sparse(); + stop_cpus(cpumask_of(cpu), do_isolation_work_cpu_stop, 0); + irq_unlock_sparse(); + + calc_load_migrate(rq); + update_max_interval(); + sched_update_group_capacities(cpu); + +out: + cpu_maps_update_done(); + trace_sched_isolate(cpu, cpumask_bits(cpu_isolated_mask)[0], + start_time, 1); + return ret_code; +} + +/* + * Note: The client calling sched_isolate_cpu() is repsonsible for ONLY + * calling sched_unisolate_cpu() on a CPU that the client previously isolated. + * Client is also responsible for unisolating when a core goes offline + * (after CPU is marked offline). + */ +int sched_unisolate_cpu_unlocked(int cpu) +{ + int ret_code = 0; + u64 start_time = 0; + + if (cpu < 0 || cpu >= nr_cpu_ids || !cpu_possible(cpu) + || cpu >= NR_CPUS) { + ret_code = -EINVAL; + goto out; + } + + if (trace_sched_isolate_enabled()) + start_time = sched_clock(); + + if (!cpu_isolation_vote[cpu]) { + ret_code = -EINVAL; + goto out; + } + + if (--cpu_isolation_vote[cpu]) + goto out; + + set_cpu_isolated(cpu, false); + update_max_interval(); + sched_update_group_capacities(cpu); + + if (cpu_online(cpu)) { + stop_cpus(cpumask_of(cpu), do_unisolation_work_cpu_stop, 0); + + /* Kick CPU to immediately do load balancing */ + if (!atomic_fetch_or(NOHZ_KICK_MASK, nohz_flags(cpu))) + smp_send_reschedule(cpu); + } + +out: + trace_sched_isolate(cpu, cpumask_bits(cpu_isolated_mask)[0], + start_time, 0); + return ret_code; +} + +int sched_unisolate_cpu(int cpu) +{ + int ret_code; + + cpu_maps_update_begin(); + ret_code = sched_unisolate_cpu_unlocked(cpu); + cpu_maps_update_done(); + return ret_code; +} + +/* + * Remove a task from the runqueue and pretend that it's migrating. This + * should prevent migrations for the detached task and disallow further + * changes to tsk_cpus_allowed. + */ +void +detach_one_task_core(struct task_struct *p, struct rq *rq, + struct list_head *tasks) +{ + lockdep_assert_held(&rq->lock); + + p->on_rq = TASK_ON_RQ_MIGRATING; + deactivate_task(rq, p, 0); + list_add(&p->se.group_node, tasks); +} + +void attach_tasks_core(struct list_head *tasks, struct rq *rq) +{ + struct task_struct *p; + + lockdep_assert_held(&rq->lock); + + while (!list_empty(tasks)) { + p = list_first_entry(tasks, struct task_struct, se.group_node); + list_del_init(&p->se.group_node); + + BUG_ON(task_rq(p) != rq); + activate_task(rq, p, 0); + p->on_rq = TASK_ON_RQ_QUEUED; + } +} +#endif /* CONFIG_HOTPLUG_CPU */ +#endif /* CONFIG_SCHED_WALT */ diff --git a/kernel/sched/walt/qc_vas.h b/kernel/sched/walt/qc_vas.h new file mode 100644 index 000000000000..3b67a5494655 --- /dev/null +++ b/kernel/sched/walt/qc_vas.h @@ -0,0 +1,81 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#include "../sched.h" +#include "../../../fs/proc/internal.h" + +#include "walt.h" +#include "trace.h" + +#ifdef CONFIG_SCHED_WALT +#ifdef CONFIG_HZ_300 +/* + * Tick interval becomes to 3333333 due to + * rounding error when HZ=300. + */ +#define DEFAULT_SCHED_RAVG_WINDOW (3333333 * 5) +#else +/* Min window size (in ns) = 16ms */ +#define DEFAULT_SCHED_RAVG_WINDOW 16000000 +#endif + +/* Max window size (in ns) = 1s */ +#define MAX_SCHED_RAVG_WINDOW 1000000000 + +#define NR_WINDOWS_PER_SEC (NSEC_PER_SEC / DEFAULT_SCHED_RAVG_WINDOW) + +extern int num_sched_clusters; + +extern unsigned int walt_big_tasks(int cpu); +extern void reset_task_stats(struct task_struct *p); +extern void walt_rotate_work_init(void); +extern void walt_rotation_checkpoint(int nr_big); +extern void walt_fill_ta_data(struct core_ctl_notif_data *data); +extern int sched_set_group_id(struct task_struct *p, unsigned int group_id); +extern unsigned int sched_get_group_id(struct task_struct *p); +extern int sched_set_init_task_load(struct task_struct *p, int init_load_pct); +extern u32 sched_get_init_task_load(struct task_struct *p); +extern void core_ctl_check(u64 wallclock); +extern int sched_set_boost(int enable); +extern int sched_isolate_count(const cpumask_t *mask, bool include_offline); + +extern struct list_head cluster_head; +#define for_each_sched_cluster(cluster) \ + list_for_each_entry_rcu(cluster, &cluster_head, list) + +static inline u32 cpu_cycles_to_freq(u64 cycles, u64 period) +{ + return div64_u64(cycles, period); +} + +static inline unsigned int sched_cpu_legacy_freq(int cpu) +{ + unsigned long curr_cap = arch_scale_freq_capacity(cpu); + + return (curr_cap * (u64) cpu_rq(cpu)->wrq.cluster->max_possible_freq) >> + SCHED_CAPACITY_SHIFT; +} + +extern __read_mostly bool sched_freq_aggr_en; +static inline void walt_enable_frequency_aggregation(bool enable) +{ + sched_freq_aggr_en = enable; +} + +#ifndef CONFIG_IRQ_TIME_ACCOUNTING +static inline u64 irq_time_read(int cpu) { return 0; } +#endif + +#else +static inline unsigned int walt_big_tasks(int cpu) +{ + return 0; +} + +static inline int sched_set_boost(int enable) +{ + return -EINVAL; +} +#endif diff --git a/kernel/sched/walt/sched_avg.c b/kernel/sched/walt/sched_avg.c new file mode 100644 index 000000000000..4a5efa4118b3 --- /dev/null +++ b/kernel/sched/walt/sched_avg.c @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2012, 2015-2021, The Linux Foundation. All rights reserved. + */ +/* + * Scheduler hook for average runqueue determination + */ +#include +#include +#include +#include +#include + +#include "qc_vas.h" +#include + +static DEFINE_PER_CPU(u64, nr_prod_sum); +static DEFINE_PER_CPU(u64, last_time); +static DEFINE_PER_CPU(u64, nr_big_prod_sum); +static DEFINE_PER_CPU(u64, nr); +static DEFINE_PER_CPU(u64, nr_max); + +static DEFINE_PER_CPU(spinlock_t, nr_lock) = __SPIN_LOCK_UNLOCKED(nr_lock); +static s64 last_get_time; + +unsigned int sysctl_sched_busy_hyst_enable_cpus; +unsigned int sysctl_sched_busy_hyst; +unsigned int sysctl_sched_coloc_busy_hyst_enable_cpus = 112; +unsigned int sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS] = { + [0 ... NR_CPUS-1] = 39000000 }; +unsigned int sysctl_sched_coloc_busy_hyst_max_ms = 5000; +unsigned int sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS] = { + [0 ... NR_CPUS-1] = 10 }; +static DEFINE_PER_CPU(atomic64_t, busy_hyst_end_time) = ATOMIC64_INIT(0); + +static DEFINE_PER_CPU(u64, hyst_time); +static DEFINE_PER_CPU(u64, coloc_hyst_busy); +static DEFINE_PER_CPU(u64, coloc_hyst_time); + +#define NR_THRESHOLD_PCT 15 +#define MAX_RTGB_TIME (sysctl_sched_coloc_busy_hyst_max_ms * NSEC_PER_MSEC) + +/** + * sched_get_nr_running_avg + * @return: Average nr_running, iowait and nr_big_tasks value since last poll. + * Returns the avg * 100 to return up to two decimal points + * of accuracy. + * + * Obtains the average nr_running value since the last poll. + * This function may not be called concurrently with itself + */ +void sched_get_nr_running_avg(struct sched_avg_stats *stats) +{ + int cpu; + u64 curr_time = sched_clock(); + u64 period = curr_time - last_get_time; + u64 tmp_nr, tmp_misfit; + bool any_hyst_time = false; + + if (!period) + return; + + /* read and reset nr_running counts */ + for_each_possible_cpu(cpu) { + unsigned long flags; + u64 diff; + + spin_lock_irqsave(&per_cpu(nr_lock, cpu), flags); + curr_time = sched_clock(); + diff = curr_time - per_cpu(last_time, cpu); + BUG_ON((s64)diff < 0); + + tmp_nr = per_cpu(nr_prod_sum, cpu); + tmp_nr += per_cpu(nr, cpu) * diff; + tmp_nr = div64_u64((tmp_nr * 100), period); + + tmp_misfit = per_cpu(nr_big_prod_sum, cpu); + tmp_misfit += walt_big_tasks(cpu) * diff; + tmp_misfit = div64_u64((tmp_misfit * 100), period); + + /* + * NR_THRESHOLD_PCT is to make sure that the task ran + * at least 85% in the last window to compensate any + * over estimating being done. + */ + stats[cpu].nr = (int)div64_u64((tmp_nr + NR_THRESHOLD_PCT), + 100); + stats[cpu].nr_misfit = (int)div64_u64((tmp_misfit + + NR_THRESHOLD_PCT), 100); + stats[cpu].nr_max = per_cpu(nr_max, cpu); + stats[cpu].nr_scaled = tmp_nr; + + trace_sched_get_nr_running_avg(cpu, stats[cpu].nr, + stats[cpu].nr_misfit, stats[cpu].nr_max, + stats[cpu].nr_scaled); + + per_cpu(last_time, cpu) = curr_time; + per_cpu(nr_prod_sum, cpu) = 0; + per_cpu(nr_big_prod_sum, cpu) = 0; + per_cpu(nr_max, cpu) = per_cpu(nr, cpu); + + spin_unlock_irqrestore(&per_cpu(nr_lock, cpu), flags); + } + + for_each_possible_cpu(cpu) { + if (per_cpu(coloc_hyst_time, cpu)) { + any_hyst_time = true; + break; + } + } + if (any_hyst_time && get_rtgb_active_time() >= MAX_RTGB_TIME) + sched_update_hyst_times(); + + last_get_time = curr_time; + +} +EXPORT_SYMBOL(sched_get_nr_running_avg); + +void sched_update_hyst_times(void) +{ + bool rtgb_active; + int cpu; + unsigned long cpu_cap, coloc_busy_pct; + + rtgb_active = is_rtgb_active() && (sched_boost() != CONSERVATIVE_BOOST) + && (get_rtgb_active_time() < MAX_RTGB_TIME); + + for_each_possible_cpu(cpu) { + cpu_cap = arch_scale_cpu_capacity(cpu); + coloc_busy_pct = sysctl_sched_coloc_busy_hyst_cpu_busy_pct[cpu]; + per_cpu(hyst_time, cpu) = (BIT(cpu) + & sysctl_sched_busy_hyst_enable_cpus) ? + sysctl_sched_busy_hyst : 0; + per_cpu(coloc_hyst_time, cpu) = ((BIT(cpu) + & sysctl_sched_coloc_busy_hyst_enable_cpus) + && rtgb_active) ? + sysctl_sched_coloc_busy_hyst_cpu[cpu] : 0; + per_cpu(coloc_hyst_busy, cpu) = mult_frac(cpu_cap, + coloc_busy_pct, 100); + } +} + +#define BUSY_NR_RUN 3 +#define BUSY_LOAD_FACTOR 10 +static inline void update_busy_hyst_end_time(int cpu, bool dequeue, + unsigned long prev_nr_run, u64 curr_time) +{ + bool nr_run_trigger = false; + bool load_trigger = false, coloc_load_trigger = false; + u64 agg_hyst_time; + + if (!per_cpu(hyst_time, cpu) && !per_cpu(coloc_hyst_time, cpu)) + return; + + if (prev_nr_run >= BUSY_NR_RUN && per_cpu(nr, cpu) < BUSY_NR_RUN) + nr_run_trigger = true; + + if (dequeue && (cpu_util(cpu) * BUSY_LOAD_FACTOR) > + capacity_orig_of(cpu)) + load_trigger = true; + + if (dequeue && cpu_util(cpu) > per_cpu(coloc_hyst_busy, cpu)) + coloc_load_trigger = true; + + agg_hyst_time = max((nr_run_trigger || load_trigger) ? + per_cpu(hyst_time, cpu) : 0, + (nr_run_trigger || coloc_load_trigger) ? + per_cpu(coloc_hyst_time, cpu) : 0); + + if (agg_hyst_time) + atomic64_set(&per_cpu(busy_hyst_end_time, cpu), + curr_time + agg_hyst_time); +} + +int sched_busy_hyst_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, loff_t *ppos) +{ + int ret; + + if (table->maxlen > (sizeof(unsigned int) * num_possible_cpus())) + table->maxlen = sizeof(unsigned int) * num_possible_cpus(); + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + + if (!ret && write) + sched_update_hyst_times(); + + return ret; +} + +/** + * sched_update_nr_prod + * @cpu: The core id of the nr running driver. + * @delta: Adjust nr by 'delta' amount + * @inc: Whether we are increasing or decreasing the count + * @return: N/A + * + * Update average with latest nr_running value for CPU + */ +void sched_update_nr_prod(int cpu, long delta, bool inc) +{ + u64 diff; + u64 curr_time; + unsigned long flags, nr_running; + + spin_lock_irqsave(&per_cpu(nr_lock, cpu), flags); + nr_running = per_cpu(nr, cpu); + curr_time = sched_clock(); + diff = curr_time - per_cpu(last_time, cpu); + BUG_ON((s64)diff < 0); + per_cpu(last_time, cpu) = curr_time; + per_cpu(nr, cpu) = nr_running + (inc ? delta : -delta); + + BUG_ON((s64)per_cpu(nr, cpu) < 0); + + if (per_cpu(nr, cpu) > per_cpu(nr_max, cpu)) + per_cpu(nr_max, cpu) = per_cpu(nr, cpu); + + update_busy_hyst_end_time(cpu, !inc, nr_running, curr_time); + + per_cpu(nr_prod_sum, cpu) += nr_running * diff; + per_cpu(nr_big_prod_sum, cpu) += walt_big_tasks(cpu) * diff; + spin_unlock_irqrestore(&per_cpu(nr_lock, cpu), flags); +} +EXPORT_SYMBOL(sched_update_nr_prod); + +/* + * Returns the CPU utilization % in the last window. + * + */ +unsigned int sched_get_cpu_util(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + u64 util; + unsigned long capacity, flags; + unsigned int busy; + + raw_spin_lock_irqsave(&rq->lock, flags); + + capacity = capacity_orig_of(cpu); + + util = rq->wrq.prev_runnable_sum + rq->wrq.grp_time.prev_runnable_sum; + util = div64_u64(util, sched_ravg_window >> SCHED_CAPACITY_SHIFT); + raw_spin_unlock_irqrestore(&rq->lock, flags); + + util = (util >= capacity) ? capacity : util; + busy = div64_ul((util * 100), capacity); + return busy; +} + +u64 sched_lpm_disallowed_time(int cpu) +{ + u64 now = sched_clock(); + u64 bias_end_time = atomic64_read(&per_cpu(busy_hyst_end_time, cpu)); + + if (now < bias_end_time) + return bias_end_time - now; + + return 0; +} diff --git a/kernel/sched/walt/trace.c b/kernel/sched/walt/trace.c new file mode 100644 index 000000000000..7c06ce1fda64 --- /dev/null +++ b/kernel/sched/walt/trace.c @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#include "qc_vas.h" + +#ifdef CONFIG_SCHED_WALT +static inline void __window_data(u32 *dst, u32 *src) +{ + if (src) + memcpy(dst, src, nr_cpu_ids * sizeof(u32)); + else + memset(dst, 0, nr_cpu_ids * sizeof(u32)); +} + +struct trace_seq; +const char *__window_print(struct trace_seq *p, const u32 *buf, int buf_len) +{ + int i; + const char *ret = p->buffer + seq_buf_used(&p->seq); + + for (i = 0; i < buf_len; i++) + trace_seq_printf(p, "%u ", buf[i]); + + trace_seq_putc(p, 0); + + return ret; +} + +static inline s64 __rq_update_sum(struct rq *rq, bool curr, bool new) +{ + if (curr) + if (new) + return rq->wrq.nt_curr_runnable_sum; + else + return rq->wrq.curr_runnable_sum; + else + if (new) + return rq->wrq.nt_prev_runnable_sum; + else + return rq->wrq.prev_runnable_sum; +} + +static inline s64 __grp_update_sum(struct rq *rq, bool curr, bool new) +{ + if (curr) + if (new) + return rq->wrq.grp_time.nt_curr_runnable_sum; + else + return rq->wrq.grp_time.curr_runnable_sum; + else + if (new) + return rq->wrq.grp_time.nt_prev_runnable_sum; + else + return rq->wrq.grp_time.prev_runnable_sum; +} + +static inline s64 +__get_update_sum(struct rq *rq, enum migrate_types migrate_type, + bool src, bool new, bool curr) +{ + switch (migrate_type) { + case RQ_TO_GROUP: + if (src) + return __rq_update_sum(rq, curr, new); + else + return __grp_update_sum(rq, curr, new); + case GROUP_TO_RQ: + if (src) + return __grp_update_sum(rq, curr, new); + else + return __rq_update_sum(rq, curr, new); + default: + WARN_ON_ONCE(1); + return -1; + } +} +#endif +#define CREATE_TRACE_POINTS +#include "trace.h" + diff --git a/kernel/sched/walt/trace.h b/kernel/sched/walt/trace.h new file mode 100644 index 000000000000..4b3c111c313a --- /dev/null +++ b/kernel/sched/walt/trace.h @@ -0,0 +1,669 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#undef TRACE_SYSTEM +#define TRACE_SYSTEM sched + +#if !defined(_TRACE_WALT_H) || defined(TRACE_HEADER_MULTI_READ) +#define _TRACE_WALT_H + +#include + +#ifdef CONFIG_SCHED_WALT +struct rq; +struct group_cpu_time; +extern const char __weak *task_event_names[]; + +TRACE_EVENT(sched_update_pred_demand, + + TP_PROTO(struct task_struct *p, u32 runtime, int pct, + unsigned int pred_demand), + + TP_ARGS(p, runtime, pct, pred_demand), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(unsigned int, runtime) + __field(int, pct) + __field(unsigned int, pred_demand) + __array(u8, bucket, NUM_BUSY_BUCKETS) + __field(int, cpu) + ), + + TP_fast_assign( + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->runtime = runtime; + __entry->pct = pct; + __entry->pred_demand = pred_demand; + memcpy(__entry->bucket, p->wts.busy_buckets, + NUM_BUSY_BUCKETS * sizeof(u8)); + __entry->cpu = task_cpu(p); + ), + + TP_printk("%d (%s): runtime %u pct %d cpu %d pred_demand %u (buckets: %u %u %u %u %u %u %u %u %u %u)", + __entry->pid, __entry->comm, + __entry->runtime, __entry->pct, __entry->cpu, + __entry->pred_demand, __entry->bucket[0], __entry->bucket[1], + __entry->bucket[2], __entry->bucket[3], __entry->bucket[4], + __entry->bucket[5], __entry->bucket[6], __entry->bucket[7], + __entry->bucket[8], __entry->bucket[9]) +); + +TRACE_EVENT(sched_update_history, + + TP_PROTO(struct rq *rq, struct task_struct *p, u32 runtime, int samples, + enum task_event evt), + + TP_ARGS(rq, p, runtime, samples, evt), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(unsigned int, runtime) + __field(int, samples) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(unsigned int, coloc_demand) + __field(unsigned int, pred_demand) + __array(u32, hist, RAVG_HIST_SIZE_MAX) + __field(unsigned int, nr_big_tasks) + __field(int, cpu) + ), + + TP_fast_assign( + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->runtime = runtime; + __entry->samples = samples; + __entry->evt = evt; + __entry->demand = p->wts.demand; + __entry->coloc_demand = p->wts.coloc_demand; + __entry->pred_demand = p->wts.pred_demand; + memcpy(__entry->hist, p->wts.sum_history, + RAVG_HIST_SIZE_MAX * sizeof(u32)); + __entry->nr_big_tasks = rq->wrq.walt_stats.nr_big_tasks; + __entry->cpu = rq->cpu; + ), + + TP_printk("%d (%s): runtime %u samples %d event %s demand %u coloc_demand %u pred_demand %u (hist: %u %u %u %u %u) cpu %d nr_big %u", + __entry->pid, __entry->comm, + __entry->runtime, __entry->samples, + task_event_names[__entry->evt], + __entry->demand, __entry->coloc_demand, __entry->pred_demand, + __entry->hist[0], __entry->hist[1], + __entry->hist[2], __entry->hist[3], + __entry->hist[4], __entry->cpu, __entry->nr_big_tasks) +); + +TRACE_EVENT(sched_get_task_cpu_cycles, + + TP_PROTO(int cpu, int event, u64 cycles, + u64 exec_time, struct task_struct *p), + + TP_ARGS(cpu, event, cycles, exec_time, p), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, event) + __field(u64, cycles) + __field(u64, exec_time) + __field(u32, freq) + __field(u32, legacy_freq) + __field(u32, max_freq) + __field(pid_t, pid) + __array(char, comm, TASK_COMM_LEN) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->event = event; + __entry->cycles = cycles; + __entry->exec_time = exec_time; + __entry->freq = cpu_cycles_to_freq(cycles, exec_time); + __entry->legacy_freq = sched_cpu_legacy_freq(cpu); + __entry->max_freq = cpu_max_freq(cpu); + __entry->pid = p->pid; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + ), + + TP_printk("cpu=%d event=%d cycles=%llu exec_time=%llu freq=%u legacy_freq=%u max_freq=%u task=%d (%s)", + __entry->cpu, __entry->event, __entry->cycles, + __entry->exec_time, __entry->freq, __entry->legacy_freq, + __entry->max_freq, __entry->pid, __entry->comm) +); + +TRACE_EVENT(sched_update_task_ravg, + + TP_PROTO(struct task_struct *p, struct rq *rq, enum task_event evt, + u64 wallclock, u64 irqtime, + struct group_cpu_time *cpu_time), + + TP_ARGS(p, rq, evt, wallclock, irqtime, cpu_time), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(pid_t, cur_pid) + __field(unsigned int, cur_freq) + __field(u64, wallclock) + __field(u64, mark_start) + __field(u64, delta_m) + __field(u64, win_start) + __field(u64, delta) + __field(u64, irqtime) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(unsigned int, coloc_demand) + __field(unsigned int, sum) + __field(int, cpu) + __field(unsigned int, pred_demand) + __field(u64, rq_cs) + __field(u64, rq_ps) + __field(u64, grp_cs) + __field(u64, grp_ps) + __field(u64, grp_nt_cs) + __field(u64, grp_nt_ps) + __field(u32, curr_window) + __field(u32, prev_window) + __dynamic_array(u32, curr_sum, nr_cpu_ids) + __dynamic_array(u32, prev_sum, nr_cpu_ids) + __field(u64, nt_cs) + __field(u64, nt_ps) + __field(u64, active_time) + __field(u32, curr_top) + __field(u32, prev_top) + ), + + TP_fast_assign( + __entry->wallclock = wallclock; + __entry->win_start = rq->wrq.window_start; + __entry->delta = (wallclock - rq->wrq.window_start); + __entry->evt = evt; + __entry->cpu = rq->cpu; + __entry->cur_pid = rq->curr->pid; + __entry->cur_freq = rq->wrq.task_exec_scale; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->mark_start = p->wts.mark_start; + __entry->delta_m = (wallclock - p->wts.mark_start); + __entry->demand = p->wts.demand; + __entry->coloc_demand = p->wts.coloc_demand; + __entry->sum = p->wts.sum; + __entry->irqtime = irqtime; + __entry->pred_demand = p->wts.pred_demand; + __entry->rq_cs = rq->wrq.curr_runnable_sum; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_cs = cpu_time ? cpu_time->curr_runnable_sum : 0; + __entry->grp_ps = cpu_time ? cpu_time->prev_runnable_sum : 0; + __entry->grp_nt_cs = cpu_time ? + cpu_time->nt_curr_runnable_sum : 0; + __entry->grp_nt_ps = cpu_time ? + cpu_time->nt_prev_runnable_sum : 0; + __entry->curr_window = p->wts.curr_window; + __entry->prev_window = p->wts.prev_window; + __window_data(__get_dynamic_array(curr_sum), + p->wts.curr_window_cpu); + __window_data(__get_dynamic_array(prev_sum), + p->wts.prev_window_cpu); + __entry->nt_cs = rq->wrq.nt_curr_runnable_sum; + __entry->nt_ps = rq->wrq.nt_prev_runnable_sum; + __entry->active_time = p->wts.active_time; + __entry->curr_top = rq->wrq.curr_top; + __entry->prev_top = rq->wrq.prev_top; + ), + + TP_printk("wc %llu ws %llu delta %llu event %s cpu %d cur_freq %u cur_pid %d task %d (%s) ms %llu delta %llu demand %u coloc_demand: %u sum %u irqtime %llu pred_demand %u rq_cs %llu rq_ps %llu cur_window %u (%s) prev_window %u (%s) nt_cs %llu nt_ps %llu active_time %u grp_cs %lld grp_ps %lld, grp_nt_cs %llu, grp_nt_ps: %llu curr_top %u prev_top %u", + __entry->wallclock, __entry->win_start, __entry->delta, + task_event_names[__entry->evt], __entry->cpu, + __entry->cur_freq, __entry->cur_pid, + __entry->pid, __entry->comm, __entry->mark_start, + __entry->delta_m, __entry->demand, __entry->coloc_demand, + __entry->sum, __entry->irqtime, __entry->pred_demand, + __entry->rq_cs, __entry->rq_ps, __entry->curr_window, + __window_print(p, __get_dynamic_array(curr_sum), nr_cpu_ids), + __entry->prev_window, + __window_print(p, __get_dynamic_array(prev_sum), nr_cpu_ids), + __entry->nt_cs, __entry->nt_ps, + __entry->active_time, __entry->grp_cs, + __entry->grp_ps, __entry->grp_nt_cs, __entry->grp_nt_ps, + __entry->curr_top, __entry->prev_top) +); + +TRACE_EVENT(sched_update_task_ravg_mini, + + TP_PROTO(struct task_struct *p, struct rq *rq, enum task_event evt, + u64 wallclock, u64 irqtime, + struct group_cpu_time *cpu_time), + + TP_ARGS(p, rq, evt, wallclock, irqtime, cpu_time), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(u64, wallclock) + __field(u64, mark_start) + __field(u64, delta_m) + __field(u64, win_start) + __field(u64, delta) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(int, cpu) + __field(u64, rq_cs) + __field(u64, rq_ps) + __field(u64, grp_cs) + __field(u64, grp_ps) + __field(u32, curr_window) + __field(u32, prev_window) + ), + + TP_fast_assign( + __entry->wallclock = wallclock; + __entry->win_start = rq->wrq.window_start; + __entry->delta = (wallclock - rq->wrq.window_start); + __entry->evt = evt; + __entry->cpu = rq->cpu; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->mark_start = p->wts.mark_start; + __entry->delta_m = (wallclock - p->wts.mark_start); + __entry->demand = p->wts.demand; + __entry->rq_cs = rq->wrq.curr_runnable_sum; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_cs = cpu_time ? cpu_time->curr_runnable_sum : 0; + __entry->grp_ps = cpu_time ? cpu_time->prev_runnable_sum : 0; + __entry->curr_window = p->wts.curr_window; + __entry->prev_window = p->wts.prev_window; + ), + + TP_printk("wc %llu ws %llu delta %llu event %s cpu %d task %d (%s) ms %llu delta %llu demand %u rq_cs %llu rq_ps %llu cur_window %u prev_window %u grp_cs %lld grp_ps %lld", + __entry->wallclock, __entry->win_start, __entry->delta, + task_event_names[__entry->evt], __entry->cpu, + __entry->pid, __entry->comm, __entry->mark_start, + __entry->delta_m, __entry->demand, + __entry->rq_cs, __entry->rq_ps, __entry->curr_window, + __entry->prev_window, __entry->grp_cs, __entry->grp_ps) +); + +struct migration_sum_data; +extern const char __weak *migrate_type_names[]; + +TRACE_EVENT(sched_set_preferred_cluster, + + TP_PROTO(struct walt_related_thread_group *grp, u64 total_demand), + + TP_ARGS(grp, total_demand), + + TP_STRUCT__entry( + __field(int, id) + __field(u64, total_demand) + __field(bool, skip_min) + ), + + TP_fast_assign( + __entry->id = grp->id; + __entry->total_demand = total_demand; + __entry->skip_min = grp->skip_min; + ), + + TP_printk("group_id %d total_demand %llu skip_min %d", + __entry->id, __entry->total_demand, + __entry->skip_min) +); + +TRACE_EVENT(sched_migration_update_sum, + + TP_PROTO(struct task_struct *p, enum migrate_types migrate_type, + struct rq *rq), + + TP_ARGS(p, migrate_type, rq), + + TP_STRUCT__entry( + __field(int, tcpu) + __field(int, pid) + __field(enum migrate_types, migrate_type) + __field(s64, src_cs) + __field(s64, src_ps) + __field(s64, dst_cs) + __field(s64, dst_ps) + __field(s64, src_nt_cs) + __field(s64, src_nt_ps) + __field(s64, dst_nt_cs) + __field(s64, dst_nt_ps) + ), + + TP_fast_assign( + __entry->tcpu = task_cpu(p); + __entry->pid = p->pid; + __entry->migrate_type = migrate_type; + __entry->src_cs = __get_update_sum(rq, migrate_type, + true, false, true); + __entry->src_ps = __get_update_sum(rq, migrate_type, + true, false, false); + __entry->dst_cs = __get_update_sum(rq, migrate_type, + false, false, true); + __entry->dst_ps = __get_update_sum(rq, migrate_type, + false, false, false); + __entry->src_nt_cs = __get_update_sum(rq, migrate_type, + true, true, true); + __entry->src_nt_ps = __get_update_sum(rq, migrate_type, + true, true, false); + __entry->dst_nt_cs = __get_update_sum(rq, migrate_type, + false, true, true); + __entry->dst_nt_ps = __get_update_sum(rq, migrate_type, + false, true, false); + ), + + TP_printk("pid %d task_cpu %d migrate_type %s src_cs %llu src_ps %llu dst_cs %lld dst_ps %lld src_nt_cs %llu src_nt_ps %llu dst_nt_cs %lld dst_nt_ps %lld", + __entry->pid, __entry->tcpu, + migrate_type_names[__entry->migrate_type], + __entry->src_cs, __entry->src_ps, __entry->dst_cs, + __entry->dst_ps, __entry->src_nt_cs, __entry->src_nt_ps, + __entry->dst_nt_cs, __entry->dst_nt_ps) +); + +TRACE_EVENT(sched_set_boost, + + TP_PROTO(int type), + + TP_ARGS(type), + + TP_STRUCT__entry( + __field(int, type) + ), + + TP_fast_assign( + __entry->type = type; + ), + + TP_printk("type %d", __entry->type) +); + +TRACE_EVENT(sched_load_to_gov, + + TP_PROTO(struct rq *rq, u64 aggr_grp_load, u32 tt_load, + int freq_aggr, u64 load, int policy, + int big_task_rotation, + unsigned int user_hint), + TP_ARGS(rq, aggr_grp_load, tt_load, freq_aggr, load, policy, + big_task_rotation, user_hint), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, policy) + __field(int, ed_task_pid) + __field(u64, aggr_grp_load) + __field(int, freq_aggr) + __field(u64, tt_load) + __field(u64, rq_ps) + __field(u64, grp_rq_ps) + __field(u64, nt_ps) + __field(u64, grp_nt_ps) + __field(u64, pl) + __field(u64, load) + __field(int, big_task_rotation) + __field(unsigned int, user_hint) + ), + + TP_fast_assign( + __entry->cpu = cpu_of(rq); + __entry->policy = policy; + __entry->ed_task_pid = + rq->wrq.ed_task ? rq->wrq.ed_task->pid : -1; + __entry->aggr_grp_load = aggr_grp_load; + __entry->freq_aggr = freq_aggr; + __entry->tt_load = tt_load; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_rq_ps = rq->wrq.grp_time.prev_runnable_sum; + __entry->nt_ps = rq->wrq.nt_prev_runnable_sum; + __entry->grp_nt_ps = rq->wrq.grp_time.nt_prev_runnable_sum; + __entry->pl = + rq->wrq.walt_stats.pred_demands_sum_scaled; + __entry->load = load; + __entry->big_task_rotation = big_task_rotation; + __entry->user_hint = user_hint; + ), + + TP_printk("cpu=%d policy=%d ed_task_pid=%d aggr_grp_load=%llu freq_aggr=%d tt_load=%llu rq_ps=%llu grp_rq_ps=%llu nt_ps=%llu grp_nt_ps=%llu pl=%llu load=%llu big_task_rotation=%d user_hint=%u", + __entry->cpu, __entry->policy, __entry->ed_task_pid, + __entry->aggr_grp_load, __entry->freq_aggr, + __entry->tt_load, __entry->rq_ps, __entry->grp_rq_ps, + __entry->nt_ps, __entry->grp_nt_ps, __entry->pl, __entry->load, + __entry->big_task_rotation, __entry->user_hint) +); + +TRACE_EVENT(core_ctl_eval_need, + + TP_PROTO(unsigned int cpu, unsigned int old_need, + unsigned int new_need, unsigned int updated), + TP_ARGS(cpu, old_need, new_need, updated), + TP_STRUCT__entry( + __field(u32, cpu) + __field(u32, old_need) + __field(u32, new_need) + __field(u32, updated) + ), + TP_fast_assign( + __entry->cpu = cpu; + __entry->old_need = old_need; + __entry->new_need = new_need; + __entry->updated = updated; + ), + TP_printk("cpu=%u, old_need=%u, new_need=%u, updated=%u", __entry->cpu, + __entry->old_need, __entry->new_need, __entry->updated) +); + +TRACE_EVENT(core_ctl_set_busy, + + TP_PROTO(unsigned int cpu, unsigned int busy, + unsigned int old_is_busy, unsigned int is_busy), + TP_ARGS(cpu, busy, old_is_busy, is_busy), + TP_STRUCT__entry( + __field(u32, cpu) + __field(u32, busy) + __field(u32, old_is_busy) + __field(u32, is_busy) + __field(bool, high_irqload) + ), + TP_fast_assign( + __entry->cpu = cpu; + __entry->busy = busy; + __entry->old_is_busy = old_is_busy; + __entry->is_busy = is_busy; + __entry->high_irqload = sched_cpu_high_irqload(cpu); + ), + TP_printk("cpu=%u, busy=%u, old_is_busy=%u, new_is_busy=%u high_irqload=%d", + __entry->cpu, __entry->busy, __entry->old_is_busy, + __entry->is_busy, __entry->high_irqload) +); + +TRACE_EVENT(core_ctl_set_boost, + + TP_PROTO(u32 refcount, s32 ret), + TP_ARGS(refcount, ret), + TP_STRUCT__entry( + __field(u32, refcount) + __field(s32, ret) + ), + TP_fast_assign( + __entry->refcount = refcount; + __entry->ret = ret; + ), + TP_printk("refcount=%u, ret=%d", __entry->refcount, __entry->ret) +); + +TRACE_EVENT(core_ctl_update_nr_need, + + TP_PROTO(int cpu, int nr_need, int prev_misfit_need, + int nrrun, int max_nr, int nr_prev_assist), + + TP_ARGS(cpu, nr_need, prev_misfit_need, nrrun, max_nr, nr_prev_assist), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, nr_need) + __field(int, prev_misfit_need) + __field(int, nrrun) + __field(int, max_nr) + __field(int, nr_prev_assist) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->nr_need = nr_need; + __entry->prev_misfit_need = prev_misfit_need; + __entry->nrrun = nrrun; + __entry->max_nr = max_nr; + __entry->nr_prev_assist = nr_prev_assist; + ), + + TP_printk("cpu=%d nr_need=%d prev_misfit_need=%d nrrun=%d max_nr=%d nr_prev_assist=%d", + __entry->cpu, __entry->nr_need, __entry->prev_misfit_need, + __entry->nrrun, __entry->max_nr, __entry->nr_prev_assist) +); + +TRACE_EVENT(core_ctl_notif_data, + + TP_PROTO(u32 nr_big, u32 ta_load, u32 *ta_util, u32 *cur_cap), + + TP_ARGS(nr_big, ta_load, ta_util, cur_cap), + + TP_STRUCT__entry( + __field(u32, nr_big) + __field(u32, ta_load) + __array(u32, ta_util, MAX_CLUSTERS) + __array(u32, cur_cap, MAX_CLUSTERS) + ), + + TP_fast_assign( + __entry->nr_big = nr_big; + __entry->ta_load = ta_load; + memcpy(__entry->ta_util, ta_util, MAX_CLUSTERS * sizeof(u32)); + memcpy(__entry->cur_cap, cur_cap, MAX_CLUSTERS * sizeof(u32)); + ), + + TP_printk("nr_big=%u ta_load=%u ta_util=(%u %u %u) cur_cap=(%u %u %u)", + __entry->nr_big, __entry->ta_load, + __entry->ta_util[0], __entry->ta_util[1], + __entry->ta_util[2], __entry->cur_cap[0], + __entry->cur_cap[1], __entry->cur_cap[2]) +); + +/* + * Tracepoint for sched_get_nr_running_avg + */ +TRACE_EVENT(sched_get_nr_running_avg, + + TP_PROTO(int cpu, int nr, int nr_misfit, int nr_max, int nr_scaled), + + TP_ARGS(cpu, nr, nr_misfit, nr_max, nr_scaled), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, nr) + __field(int, nr_misfit) + __field(int, nr_max) + __field(int, nr_scaled) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->nr = nr; + __entry->nr_misfit = nr_misfit; + __entry->nr_max = nr_max; + __entry->nr_scaled = nr_scaled; + ), + + TP_printk("cpu=%d nr=%d nr_misfit=%d nr_max=%d nr_scaled=%d", + __entry->cpu, __entry->nr, __entry->nr_misfit, __entry->nr_max, + __entry->nr_scaled) +); + +/* + * sched_isolate - called when cores are isolated/unisolated + * + * @acutal_mask: mask of cores actually isolated/unisolated + * @req_mask: mask of cores requested isolated/unisolated + * @online_mask: cpu online mask + * @time: amount of time in us it took to isolate/unisolate + * @isolate: 1 if isolating, 0 if unisolating + * + */ +TRACE_EVENT(sched_isolate, + + TP_PROTO(unsigned int requested_cpu, unsigned int isolated_cpus, + u64 start_time, unsigned char isolate), + + TP_ARGS(requested_cpu, isolated_cpus, start_time, isolate), + + TP_STRUCT__entry( + __field(u32, requested_cpu) + __field(u32, isolated_cpus) + __field(u32, time) + __field(unsigned char, isolate) + ), + + TP_fast_assign( + __entry->requested_cpu = requested_cpu; + __entry->isolated_cpus = isolated_cpus; + __entry->time = div64_u64(sched_clock() - start_time, 1000); + __entry->isolate = isolate; + ), + + TP_printk("iso cpu=%u cpus=0x%x time=%u us isolated=%d", + __entry->requested_cpu, __entry->isolated_cpus, + __entry->time, __entry->isolate) +); + +TRACE_EVENT(sched_ravg_window_change, + + TP_PROTO(unsigned int sched_ravg_window, unsigned int new_sched_ravg_window + , u64 change_time), + + TP_ARGS(sched_ravg_window, new_sched_ravg_window, change_time), + + TP_STRUCT__entry( + __field(unsigned int, sched_ravg_window) + __field(unsigned int, new_sched_ravg_window) + __field(u64, change_time) + ), + + TP_fast_assign( + __entry->sched_ravg_window = sched_ravg_window; + __entry->new_sched_ravg_window = new_sched_ravg_window; + __entry->change_time = change_time; + ), + + TP_printk("from=%u to=%u at=%lu", + __entry->sched_ravg_window, __entry->new_sched_ravg_window, + __entry->change_time) +); + +TRACE_EVENT(walt_window_rollover, + + TP_PROTO(u64 window_start), + + TP_ARGS(window_start), + + TP_STRUCT__entry( + __field(u64, window_start) + ), + + TP_fast_assign( + __entry->window_start = window_start; + ), + + TP_printk("window_start=%llu", __entry->window_start) +); + +#endif /* CONFIG_SCHED_WALT */ +#endif /* _TRACE_WALT_H */ + +#undef TRACE_INCLUDE_PATH +#define TRACE_INCLUDE_PATH . +#define TRACE_INCLUDE_FILE trace + +#include diff --git a/kernel/sched/walt/walt.c b/kernel/sched/walt/walt.c new file mode 100644 index 000000000000..92a3d27b3895 --- /dev/null +++ b/kernel/sched/walt/walt.c @@ -0,0 +1,3793 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2016-2021, The Linux Foundation. All rights reserved. + */ +#include +#include +#include +#include +#include +#include +#include "qc_vas.h" + +#include + +const char *task_event_names[] = {"PUT_PREV_TASK", "PICK_NEXT_TASK", + "TASK_WAKE", "TASK_MIGRATE", "TASK_UPDATE", + "IRQ_UPDATE"}; + +const char *migrate_type_names[] = {"GROUP_TO_RQ", "RQ_TO_GROUP", + "RQ_TO_RQ", "GROUP_TO_GROUP"}; + +#define SCHED_FREQ_ACCOUNT_WAIT_TIME 0 +#define SCHED_ACCOUNT_WAIT_TIME 1 + +#define EARLY_DETECTION_DURATION 9500000 +#define MAX_NUM_CGROUP_COLOC_ID 20 + +#define WINDOW_STATS_RECENT 0 +#define WINDOW_STATS_MAX 1 +#define WINDOW_STATS_MAX_RECENT_AVG 2 +#define WINDOW_STATS_AVG 3 +#define WINDOW_STATS_INVALID_POLICY 4 + +#define MAX_NR_CLUSTERS 3 + +#define FREQ_REPORT_MAX_CPU_LOAD_TOP_TASK 0 +#define FREQ_REPORT_CPU_LOAD 1 +#define FREQ_REPORT_TOP_TASK 2 + +#define NEW_TASK_ACTIVE_TIME 100000000 + +static ktime_t ktime_last; +static bool sched_ktime_suspended; +static struct cpu_cycle_counter_cb cpu_cycle_counter_cb; +static bool use_cycle_counter; +static DEFINE_MUTEX(cluster_lock); +static atomic64_t walt_irq_work_lastq_ws; +static u64 walt_load_reported_window; + +static struct irq_work walt_cpufreq_irq_work; +static struct irq_work walt_migration_irq_work; + +u64 sched_ktime_clock(void) +{ + if (unlikely(sched_ktime_suspended)) + return ktime_to_ns(ktime_last); + return ktime_get_ns(); +} + +static void sched_resume(void) +{ + sched_ktime_suspended = false; +} + +static int sched_suspend(void) +{ + ktime_last = ktime_get(); + sched_ktime_suspended = true; + return 0; +} + +static struct syscore_ops sched_syscore_ops = { + .resume = sched_resume, + .suspend = sched_suspend +}; + +static int __init sched_init_ops(void) +{ + register_syscore_ops(&sched_syscore_ops); + return 0; +} +late_initcall(sched_init_ops); + +static void acquire_rq_locks_irqsave(const cpumask_t *cpus, + unsigned long *flags) +{ + int cpu; + int level = 0; + + local_irq_save(*flags); + for_each_cpu(cpu, cpus) { + if (level == 0) + raw_spin_lock(&cpu_rq(cpu)->lock); + else + raw_spin_lock_nested(&cpu_rq(cpu)->lock, level); + level++; + } +} + +static void release_rq_locks_irqrestore(const cpumask_t *cpus, + unsigned long *flags) +{ + int cpu; + + for_each_cpu(cpu, cpus) + raw_spin_unlock(&cpu_rq(cpu)->lock); + local_irq_restore(*flags); +} + +unsigned int sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS] = { + [0 ... MAX_MARGIN_LEVELS-1] = 1078}; /* ~5% margin */ +unsigned int sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS] = { + [0 ... MAX_MARGIN_LEVELS-1] = 1205}; /* ~15% margin */ +static unsigned int walt_cpu_high_irqload; + +unsigned int sysctl_sched_walt_rotate_big_tasks; +unsigned int walt_rotation_enabled; + +__read_mostly unsigned int sysctl_sched_asym_cap_sibling_freq_match_pct = 100; +__read_mostly unsigned int sched_ravg_hist_size = 5; + +static __read_mostly unsigned int sched_io_is_busy = 1; + +__read_mostly unsigned int sysctl_sched_window_stats_policy = + WINDOW_STATS_MAX_RECENT_AVG; + +unsigned int sysctl_sched_ravg_window_nr_ticks = (HZ / NR_WINDOWS_PER_SEC); + +unsigned int sysctl_sched_dynamic_ravg_window_enable = (HZ == 250); + +/* Window size (in ns) */ +__read_mostly unsigned int sched_ravg_window = DEFAULT_SCHED_RAVG_WINDOW; +__read_mostly unsigned int new_sched_ravg_window = DEFAULT_SCHED_RAVG_WINDOW; + +static DEFINE_SPINLOCK(sched_ravg_window_lock); +u64 sched_ravg_window_change_time; + +/* + * A after-boot constant divisor for cpu_util_freq_walt() to apply the load + * boost. + */ +static __read_mostly unsigned int walt_cpu_util_freq_divisor; + +/* Initial task load. Newly created tasks are assigned this load. */ +unsigned int __read_mostly sched_init_task_load_windows; +unsigned int __read_mostly sched_init_task_load_windows_scaled; +unsigned int __read_mostly sysctl_sched_init_task_load_pct = 15; + +unsigned int max_possible_capacity = 1024; /* max(rq->max_possible_capacity) */ +unsigned int +min_max_possible_capacity = 1024; /* min(rq->max_possible_capacity) */ + +/* + * Task load is categorized into buckets for the purpose of top task tracking. + * The entire range of load from 0 to sched_ravg_window needs to be covered + * in NUM_LOAD_INDICES number of buckets. Therefore the size of each bucket + * is given by sched_ravg_window / NUM_LOAD_INDICES. Since the default value + * of sched_ravg_window is DEFAULT_SCHED_RAVG_WINDOW, use that to compute + * sched_load_granule. + */ +__read_mostly unsigned int sched_load_granule = + DEFAULT_SCHED_RAVG_WINDOW / NUM_LOAD_INDICES; +/* Size of bitmaps maintained to track top tasks */ +static const unsigned int top_tasks_bitmap_size = + BITS_TO_LONGS(NUM_LOAD_INDICES + 1) * sizeof(unsigned long); + +/* + * This governs what load needs to be used when reporting CPU busy time + * to the cpufreq governor. + */ +__read_mostly unsigned int sysctl_sched_freq_reporting_policy; + +static int __init set_sched_ravg_window(char *str) +{ + unsigned int window_size; + + get_option(&str, &window_size); + + if (window_size < DEFAULT_SCHED_RAVG_WINDOW || + window_size > MAX_SCHED_RAVG_WINDOW) { + WARN_ON(1); + return -EINVAL; + } + + sched_ravg_window = window_size; + return 0; +} + +early_param("sched_ravg_window", set_sched_ravg_window); + +static int __init set_sched_predl(char *str) +{ + unsigned int predl; + + get_option(&str, &predl); + sched_predl = !!predl; + return 0; +} +early_param("sched_predl", set_sched_predl); + +__read_mostly unsigned int walt_scale_demand_divisor; +#define scale_demand(d) ((d)/walt_scale_demand_divisor) + +#define SCHED_PRINT(arg) printk_deferred("%s=%llu", #arg, arg) +#define STRG(arg) #arg + +static inline void walt_task_dump(struct task_struct *p) +{ + char buff[NR_CPUS * 16]; + int i, j = 0; + int buffsz = NR_CPUS * 16; + + SCHED_PRINT(p->pid); + SCHED_PRINT(p->wts.mark_start); + SCHED_PRINT(p->wts.demand); + SCHED_PRINT(p->wts.coloc_demand); + SCHED_PRINT(sched_ravg_window); + SCHED_PRINT(new_sched_ravg_window); + + for (i = 0 ; i < nr_cpu_ids; i++) + j += scnprintf(buff + j, buffsz - j, "%u ", + p->wts.curr_window_cpu[i]); + printk_deferred("%s=%d (%s)\n", STRG(p->wts.curr_window), + p->wts.curr_window, buff); + + for (i = 0, j = 0 ; i < nr_cpu_ids; i++) + j += scnprintf(buff + j, buffsz - j, "%u ", + p->wts.prev_window_cpu[i]); + printk_deferred("%s=%d (%s)\n", STRG(p->wts.prev_window), + p->wts.prev_window, buff); + + SCHED_PRINT(p->wts.last_wake_ts); + SCHED_PRINT(p->wts.last_enqueued_ts); + SCHED_PRINT(p->wts.misfit); + SCHED_PRINT(p->wts.unfilter); +} + +static inline void walt_rq_dump(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + struct task_struct *tsk = cpu_curr(cpu); + int i; + + /* + * Increment the task reference so that it can't be + * freed on a remote CPU. Since we are going to + * enter panic, there is no need to decrement the + * task reference. Decrementing the task reference + * can't be done in atomic context, especially with + * rq locks held. + */ + get_task_struct(tsk); + printk_deferred("CPU:%d nr_running:%u current: %d (%s)\n", + cpu, rq->nr_running, tsk->pid, tsk->comm); + + printk_deferred("=========================================="); + SCHED_PRINT(rq->wrq.window_start); + SCHED_PRINT(rq->wrq.prev_window_size); + SCHED_PRINT(rq->wrq.curr_runnable_sum); + SCHED_PRINT(rq->wrq.prev_runnable_sum); + SCHED_PRINT(rq->wrq.nt_curr_runnable_sum); + SCHED_PRINT(rq->wrq.nt_prev_runnable_sum); + SCHED_PRINT(rq->wrq.cum_window_demand_scaled); + SCHED_PRINT(rq->wrq.task_exec_scale); + SCHED_PRINT(rq->wrq.grp_time.curr_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.prev_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.nt_curr_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.nt_prev_runnable_sum); + for (i = 0 ; i < NUM_TRACKED_WINDOWS; i++) { + printk_deferred("rq->wrq.load_subs[%d].window_start=%llu)\n", i, + rq->wrq.load_subs[i].window_start); + printk_deferred("rq->wrq.load_subs[%d].subs=%llu)\n", i, + rq->wrq.load_subs[i].subs); + printk_deferred("rq->wrq.load_subs[%d].new_subs=%llu)\n", i, + rq->wrq.load_subs[i].new_subs); + } + walt_task_dump(tsk); + SCHED_PRINT(sched_capacity_margin_up[cpu]); + SCHED_PRINT(sched_capacity_margin_down[cpu]); +} + +static inline void walt_dump(void) +{ + int cpu; + + printk_deferred("============ WALT RQ DUMP START ==============\n"); + printk_deferred("Sched ktime_get: %llu\n", sched_ktime_clock()); + printk_deferred("Time last window changed=%lu\n", + sched_ravg_window_change_time); + for_each_online_cpu(cpu) + walt_rq_dump(cpu); + SCHED_PRINT(max_possible_capacity); + SCHED_PRINT(min_max_possible_capacity); + printk_deferred("============ WALT RQ DUMP END ==============\n"); +} + +static int in_sched_bug; +#define SCHED_BUG_ON(condition) \ +({ \ + if (unlikely(!!(condition)) && !in_sched_bug) { \ + in_sched_bug = 1; \ + walt_dump(); \ + BUG_ON(condition); \ + } \ +}) + +static void fixup_walt_sched_stats_common(struct rq *rq, struct task_struct *p, + u16 updated_demand_scaled, + u16 updated_pred_demand_scaled) +{ + s64 task_load_delta = (s64)updated_demand_scaled - + p->wts.demand_scaled; + s64 pred_demand_delta = (s64)updated_pred_demand_scaled - + p->wts.pred_demand_scaled; + + fixup_cumulative_runnable_avg(&rq->wrq.walt_stats, task_load_delta, + pred_demand_delta); + + walt_fixup_cum_window_demand(rq, task_load_delta); +} + +/* + * Demand aggregation for frequency purpose: + * + * CPU demand of tasks from various related groups is aggregated per-cluster and + * added to the "max_busy_cpu" in that cluster, where max_busy_cpu is determined + * by just rq->wrq.prev_runnable_sum. + * + * Some examples follow, which assume: + * Cluster0 = CPU0-3, Cluster1 = CPU4-7 + * One related thread group A that has tasks A0, A1, A2 + * + * A->cpu_time[X].curr/prev_sum = counters in which cpu execution stats of + * tasks belonging to group A are accumulated when they run on cpu X. + * + * CX->curr/prev_sum = counters in which cpu execution stats of all tasks + * not belonging to group A are accumulated when they run on cpu X + * + * Lets say the stats for window M was as below: + * + * C0->prev_sum = 1ms, A->cpu_time[0].prev_sum = 5ms + * Task A0 ran 5ms on CPU0 + * Task B0 ran 1ms on CPU0 + * + * C1->prev_sum = 5ms, A->cpu_time[1].prev_sum = 6ms + * Task A1 ran 4ms on CPU1 + * Task A2 ran 2ms on CPU1 + * Task B1 ran 5ms on CPU1 + * + * C2->prev_sum = 0ms, A->cpu_time[2].prev_sum = 0 + * CPU2 idle + * + * C3->prev_sum = 0ms, A->cpu_time[3].prev_sum = 0 + * CPU3 idle + * + * In this case, CPU1 was most busy going by just its prev_sum counter. Demand + * from all group A tasks are added to CPU1. IOW, at end of window M, cpu busy + * time reported to governor will be: + * + * + * C0 busy time = 1ms + * C1 busy time = 5 + 5 + 6 = 16ms + * + */ +__read_mostly bool sched_freq_aggr_en; + +static u64 +update_window_start(struct rq *rq, u64 wallclock, int event) +{ + s64 delta; + int nr_windows; + u64 old_window_start = rq->wrq.window_start; + + delta = wallclock - rq->wrq.window_start; + if (delta < 0) { + printk_deferred("WALT-BUG CPU%d; wallclock=%llu is lesser than window_start=%llu", + rq->cpu, wallclock, rq->wrq.window_start); + SCHED_BUG_ON(1); + } + if (delta < sched_ravg_window) + return old_window_start; + + nr_windows = div64_u64(delta, sched_ravg_window); + rq->wrq.window_start += (u64)nr_windows * (u64)sched_ravg_window; + + rq->wrq.cum_window_demand_scaled = + rq->wrq.walt_stats.cumulative_runnable_avg_scaled; + rq->wrq.prev_window_size = sched_ravg_window; + + return old_window_start; +} + +/* + * Assumes rq_lock is held and wallclock was recorded in the same critical + * section as this function's invocation. + */ +static inline u64 read_cycle_counter(int cpu, u64 wallclock) +{ + struct rq *rq = cpu_rq(cpu); + + if (rq->wrq.last_cc_update != wallclock) { + rq->wrq.cycles = + cpu_cycle_counter_cb.get_cpu_cycle_counter(cpu); + rq->wrq.last_cc_update = wallclock; + } + + return rq->wrq.cycles; +} + +static void update_task_cpu_cycles(struct task_struct *p, int cpu, + u64 wallclock) +{ + if (use_cycle_counter) + p->wts.cpu_cycles = read_cycle_counter(cpu, wallclock); +} + +static inline bool is_ed_enabled(void) +{ + return (walt_rotation_enabled || (sched_boost_policy() != + SCHED_BOOST_NONE)); +} + +void clear_ed_task(struct task_struct *p, struct rq *rq) +{ + if (p == rq->wrq.ed_task) + rq->wrq.ed_task = NULL; +} + +static inline bool is_ed_task(struct task_struct *p, u64 wallclock) +{ + return (wallclock - p->wts.last_wake_ts >= EARLY_DETECTION_DURATION); +} + +bool early_detection_notify(struct rq *rq, u64 wallclock) +{ + struct task_struct *p; + int loop_max = 10; + + rq->wrq.ed_task = NULL; + + if (!is_ed_enabled() || !rq->cfs.h_nr_running) + return 0; + + list_for_each_entry(p, &rq->cfs_tasks, se.group_node) { + if (!loop_max) + break; + + if (is_ed_task(p, wallclock)) { + rq->wrq.ed_task = p; + return 1; + } + + loop_max--; + } + + return 0; +} + +void walt_sched_account_irqstart(int cpu, struct task_struct *curr) +{ + struct rq *rq = cpu_rq(cpu); + + if (!rq->wrq.window_start) + return; + + /* We're here without rq->lock held, IRQ disabled */ + raw_spin_lock(&rq->lock); + update_task_cpu_cycles(curr, cpu, sched_ktime_clock()); + raw_spin_unlock(&rq->lock); +} + +void walt_sched_account_irqend(int cpu, struct task_struct *curr, u64 delta) +{ + struct rq *rq = cpu_rq(cpu); + unsigned long flags; + + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_task_ravg(curr, rq, IRQ_UPDATE, sched_ktime_clock(), delta); + raw_spin_unlock_irqrestore(&rq->lock, flags); +} + +/* + * Return total number of tasks "eligible" to run on higher capacity cpus + */ +unsigned int walt_big_tasks(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + + return rq->wrq.walt_stats.nr_big_tasks; +} + +void clear_walt_request(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + unsigned long flags; + + clear_reserved(cpu); + if (rq->wrq.push_task) { + struct task_struct *push_task = NULL; + + raw_spin_lock_irqsave(&rq->lock, flags); + if (rq->wrq.push_task) { + clear_reserved(rq->push_cpu); + push_task = rq->wrq.push_task; + rq->wrq.push_task = NULL; + } + rq->active_balance = 0; + raw_spin_unlock_irqrestore(&rq->lock, flags); + if (push_task) + put_task_struct(push_task); + } +} + +/* + * Special case the last index and provide a fast path for index = 0. + * Note that sched_load_granule can change underneath us if we are not + * holding any runqueue locks while calling the two functions below. + */ +static u32 top_task_load(struct rq *rq) +{ + int index = rq->wrq.prev_top; + u8 prev = 1 - rq->wrq.curr_table; + + if (!index) { + int msb = NUM_LOAD_INDICES - 1; + + if (!test_bit(msb, rq->wrq.top_tasks_bitmap[prev])) + return 0; + else + return sched_load_granule; + } else if (index == NUM_LOAD_INDICES - 1) { + return sched_ravg_window; + } else { + return (index + 1) * sched_load_granule; + } +} + +unsigned int sysctl_sched_user_hint; +static unsigned long sched_user_hint_reset_time; +static bool is_cluster_hosting_top_app(struct walt_sched_cluster *cluster); + +static inline bool +should_apply_suh_freq_boost(struct walt_sched_cluster *cluster) +{ + if (sched_freq_aggr_en || !sysctl_sched_user_hint || + !cluster->aggr_grp_load) + return false; + + return is_cluster_hosting_top_app(cluster); +} + +static inline u64 freq_policy_load(struct rq *rq) +{ + unsigned int reporting_policy = sysctl_sched_freq_reporting_policy; + struct walt_sched_cluster *cluster = rq->wrq.cluster; + u64 aggr_grp_load = cluster->aggr_grp_load; + u64 load, tt_load = 0; + struct task_struct *cpu_ksoftirqd = per_cpu(ksoftirqd, cpu_of(rq)); + + if (rq->wrq.ed_task != NULL) { + load = sched_ravg_window; + goto done; + } + + if (sched_freq_aggr_en) + load = rq->wrq.prev_runnable_sum + aggr_grp_load; + else + load = rq->wrq.prev_runnable_sum + + rq->wrq.grp_time.prev_runnable_sum; + + if (cpu_ksoftirqd && cpu_ksoftirqd->state == TASK_RUNNING) + load = max_t(u64, load, task_load(cpu_ksoftirqd)); + + tt_load = top_task_load(rq); + switch (reporting_policy) { + case FREQ_REPORT_MAX_CPU_LOAD_TOP_TASK: + load = max_t(u64, load, tt_load); + break; + case FREQ_REPORT_TOP_TASK: + load = tt_load; + break; + case FREQ_REPORT_CPU_LOAD: + break; + default: + break; + } + + if (should_apply_suh_freq_boost(cluster)) { + if (is_suh_max()) + load = sched_ravg_window; + else + load = div64_u64(load * sysctl_sched_user_hint, + (u64)100); + } + +done: + trace_sched_load_to_gov(rq, aggr_grp_load, tt_load, sched_freq_aggr_en, + load, reporting_policy, walt_rotation_enabled, + sysctl_sched_user_hint); + return load; +} + +static bool rtgb_active; + +static inline unsigned long +__cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) +{ + u64 util, util_unboosted; + struct rq *rq = cpu_rq(cpu); + unsigned long capacity = capacity_orig_of(cpu); + int boost; + + boost = per_cpu(sched_load_boost, cpu); + util_unboosted = util = freq_policy_load(rq); + util = div64_u64(util * (100 + boost), + walt_cpu_util_freq_divisor); + + if (walt_load) { + u64 nl = cpu_rq(cpu)->wrq.nt_prev_runnable_sum + + rq->wrq.grp_time.nt_prev_runnable_sum; + u64 pl = rq->wrq.walt_stats.pred_demands_sum_scaled; + + /* do_pl_notif() needs unboosted signals */ + rq->wrq.old_busy_time = div64_u64(util_unboosted, + sched_ravg_window >> + SCHED_CAPACITY_SHIFT); + rq->wrq.old_estimated_time = pl; + + nl = div64_u64(nl * (100 + boost), walt_cpu_util_freq_divisor); + + walt_load->nl = nl; + walt_load->pl = pl; + walt_load->ws = walt_load_reported_window; + walt_load->rtgb_active = rtgb_active; + } + + return (util >= capacity) ? capacity : util; +} + +#define ADJUSTED_ASYM_CAP_CPU_UTIL(orig, other, x) \ + (max(orig, mult_frac(other, x, 100))) + +unsigned long +cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) +{ + struct walt_cpu_load wl_other = {0}; + unsigned long util = 0, util_other = 0; + unsigned long capacity = capacity_orig_of(cpu); + int i, mpct = sysctl_sched_asym_cap_sibling_freq_match_pct; + + if (!cpumask_test_cpu(cpu, &asym_cap_sibling_cpus)) + return __cpu_util_freq_walt(cpu, walt_load); + + for_each_cpu(i, &asym_cap_sibling_cpus) { + if (i == cpu) + util = __cpu_util_freq_walt(cpu, walt_load); + else + util_other = __cpu_util_freq_walt(i, &wl_other); + } + + if (cpu == cpumask_last(&asym_cap_sibling_cpus)) + mpct = 100; + + util = ADJUSTED_ASYM_CAP_CPU_UTIL(util, util_other, mpct); + + walt_load->nl = ADJUSTED_ASYM_CAP_CPU_UTIL(walt_load->nl, wl_other.nl, + mpct); + walt_load->pl = ADJUSTED_ASYM_CAP_CPU_UTIL(walt_load->pl, wl_other.pl, + mpct); + + return (util >= capacity) ? capacity : util; +} + +/* + * In this function we match the accumulated subtractions with the current + * and previous windows we are operating with. Ignore any entries where + * the window start in the load_subtraction struct does not match either + * the curent or the previous window. This could happen whenever CPUs + * become idle or busy with interrupts disabled for an extended period. + */ +static inline void account_load_subtractions(struct rq *rq) +{ + u64 ws = rq->wrq.window_start; + u64 prev_ws = ws - rq->wrq.prev_window_size; + struct load_subtractions *ls = rq->wrq.load_subs; + int i; + + for (i = 0; i < NUM_TRACKED_WINDOWS; i++) { + if (ls[i].window_start == ws) { + rq->wrq.curr_runnable_sum -= ls[i].subs; + rq->wrq.nt_curr_runnable_sum -= ls[i].new_subs; + } else if (ls[i].window_start == prev_ws) { + rq->wrq.prev_runnable_sum -= ls[i].subs; + rq->wrq.nt_prev_runnable_sum -= ls[i].new_subs; + } + + ls[i].subs = 0; + ls[i].new_subs = 0; + } + + SCHED_BUG_ON((s64)rq->wrq.prev_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.curr_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.nt_prev_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.nt_curr_runnable_sum < 0); +} + +static inline void create_subtraction_entry(struct rq *rq, u64 ws, int index) +{ + rq->wrq.load_subs[index].window_start = ws; + rq->wrq.load_subs[index].subs = 0; + rq->wrq.load_subs[index].new_subs = 0; +} + +static int get_top_index(unsigned long *bitmap, unsigned long old_top) +{ + int index = find_next_bit(bitmap, NUM_LOAD_INDICES, old_top); + + if (index == NUM_LOAD_INDICES) + return 0; + + return NUM_LOAD_INDICES - 1 - index; +} + +static bool get_subtraction_index(struct rq *rq, u64 ws) +{ + int i; + u64 oldest = ULLONG_MAX; + int oldest_index = 0; + + for (i = 0; i < NUM_TRACKED_WINDOWS; i++) { + u64 entry_ws = rq->wrq.load_subs[i].window_start; + + if (ws == entry_ws) + return i; + + if (entry_ws < oldest) { + oldest = entry_ws; + oldest_index = i; + } + } + + create_subtraction_entry(rq, ws, oldest_index); + return oldest_index; +} + +static void update_rq_load_subtractions(int index, struct rq *rq, + u32 sub_load, bool new_task) +{ + rq->wrq.load_subs[index].subs += sub_load; + if (new_task) + rq->wrq.load_subs[index].new_subs += sub_load; +} + +static inline struct walt_sched_cluster *cpu_cluster(int cpu) +{ + return cpu_rq(cpu)->wrq.cluster; +} + +void update_cluster_load_subtractions(struct task_struct *p, + int cpu, u64 ws, bool new_task) +{ + struct walt_sched_cluster *cluster = cpu_cluster(cpu); + struct cpumask cluster_cpus = cluster->cpus; + u64 prev_ws = ws - cpu_rq(cpu)->wrq.prev_window_size; + int i; + + cpumask_clear_cpu(cpu, &cluster_cpus); + raw_spin_lock(&cluster->load_lock); + + for_each_cpu(i, &cluster_cpus) { + struct rq *rq = cpu_rq(i); + int index; + + if (p->wts.curr_window_cpu[i]) { + index = get_subtraction_index(rq, ws); + update_rq_load_subtractions(index, rq, + p->wts.curr_window_cpu[i], new_task); + p->wts.curr_window_cpu[i] = 0; + } + + if (p->wts.prev_window_cpu[i]) { + index = get_subtraction_index(rq, prev_ws); + update_rq_load_subtractions(index, rq, + p->wts.prev_window_cpu[i], new_task); + p->wts.prev_window_cpu[i] = 0; + } + } + + raw_spin_unlock(&cluster->load_lock); +} + +static inline void inter_cluster_migration_fixup + (struct task_struct *p, int new_cpu, int task_cpu, bool new_task) +{ + struct rq *dest_rq = cpu_rq(new_cpu); + struct rq *src_rq = cpu_rq(task_cpu); + + if (same_freq_domain(new_cpu, task_cpu)) + return; + + p->wts.curr_window_cpu[new_cpu] = p->wts.curr_window; + p->wts.prev_window_cpu[new_cpu] = p->wts.prev_window; + + dest_rq->wrq.curr_runnable_sum += p->wts.curr_window; + dest_rq->wrq.prev_runnable_sum += p->wts.prev_window; + + if (src_rq->wrq.curr_runnable_sum < p->wts.curr_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.curr_runnable_sum, + p->wts.curr_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.curr_runnable_sum -= p->wts.curr_window_cpu[task_cpu]; + + if (src_rq->wrq.prev_runnable_sum < p->wts.prev_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.prev_runnable_sum, + p->wts.prev_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.prev_runnable_sum -= p->wts.prev_window_cpu[task_cpu]; + + if (new_task) { + dest_rq->wrq.nt_curr_runnable_sum += p->wts.curr_window; + dest_rq->wrq.nt_prev_runnable_sum += p->wts.prev_window; + + if (src_rq->wrq.nt_curr_runnable_sum < + p->wts.curr_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.nt_curr_runnable_sum, + p->wts.curr_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.nt_curr_runnable_sum -= + p->wts.curr_window_cpu[task_cpu]; + + if (src_rq->wrq.nt_prev_runnable_sum < + p->wts.prev_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.nt_prev_runnable_sum, + p->wts.prev_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.nt_prev_runnable_sum -= + p->wts.prev_window_cpu[task_cpu]; + } + + p->wts.curr_window_cpu[task_cpu] = 0; + p->wts.prev_window_cpu[task_cpu] = 0; + + update_cluster_load_subtractions(p, task_cpu, + src_rq->wrq.window_start, new_task); +} + +static u32 load_to_index(u32 load) +{ + u32 index = load / sched_load_granule; + + return min(index, (u32)(NUM_LOAD_INDICES - 1)); +} + +static void +migrate_top_tasks(struct task_struct *p, struct rq *src_rq, struct rq *dst_rq) +{ + int index; + int top_index; + u32 curr_window = p->wts.curr_window; + u32 prev_window = p->wts.prev_window; + u8 src = src_rq->wrq.curr_table; + u8 dst = dst_rq->wrq.curr_table; + u8 *src_table; + u8 *dst_table; + + if (curr_window) { + src_table = src_rq->wrq.top_tasks[src]; + dst_table = dst_rq->wrq.top_tasks[dst]; + index = load_to_index(curr_window); + src_table[index] -= 1; + dst_table[index] += 1; + + if (!src_table[index]) + __clear_bit(NUM_LOAD_INDICES - index - 1, + src_rq->wrq.top_tasks_bitmap[src]); + + if (dst_table[index] == 1) + __set_bit(NUM_LOAD_INDICES - index - 1, + dst_rq->wrq.top_tasks_bitmap[dst]); + + if (index > dst_rq->wrq.curr_top) + dst_rq->wrq.curr_top = index; + + top_index = src_rq->wrq.curr_top; + if (index == top_index && !src_table[index]) + src_rq->wrq.curr_top = get_top_index( + src_rq->wrq.top_tasks_bitmap[src], top_index); + } + + if (prev_window) { + src = 1 - src; + dst = 1 - dst; + src_table = src_rq->wrq.top_tasks[src]; + dst_table = dst_rq->wrq.top_tasks[dst]; + index = load_to_index(prev_window); + src_table[index] -= 1; + dst_table[index] += 1; + + if (!src_table[index]) + __clear_bit(NUM_LOAD_INDICES - index - 1, + src_rq->wrq.top_tasks_bitmap[src]); + + if (dst_table[index] == 1) + __set_bit(NUM_LOAD_INDICES - index - 1, + dst_rq->wrq.top_tasks_bitmap[dst]); + + if (index > dst_rq->wrq.prev_top) + dst_rq->wrq.prev_top = index; + + top_index = src_rq->wrq.prev_top; + if (index == top_index && !src_table[index]) + src_rq->wrq.prev_top = get_top_index( + src_rq->wrq.top_tasks_bitmap[src], top_index); + } +} + +static inline bool is_new_task(struct task_struct *p) +{ + return p->wts.active_time < NEW_TASK_ACTIVE_TIME; +} + +void fixup_busy_time(struct task_struct *p, int new_cpu) +{ + struct rq *src_rq = task_rq(p); + struct rq *dest_rq = cpu_rq(new_cpu); + u64 wallclock; + u64 *src_curr_runnable_sum, *dst_curr_runnable_sum; + u64 *src_prev_runnable_sum, *dst_prev_runnable_sum; + u64 *src_nt_curr_runnable_sum, *dst_nt_curr_runnable_sum; + u64 *src_nt_prev_runnable_sum, *dst_nt_prev_runnable_sum; + bool new_task; + struct walt_related_thread_group *grp; + long pstate; + + if (!p->on_rq && p->state != TASK_WAKING) + return; + + pstate = p->state; + + if (pstate == TASK_WAKING) + double_rq_lock(src_rq, dest_rq); + + wallclock = sched_ktime_clock(); + + walt_update_task_ravg(task_rq(p)->curr, task_rq(p), + TASK_UPDATE, + wallclock, 0); + walt_update_task_ravg(dest_rq->curr, dest_rq, + TASK_UPDATE, wallclock, 0); + + walt_update_task_ravg(p, task_rq(p), TASK_MIGRATE, + wallclock, 0); + + update_task_cpu_cycles(p, new_cpu, wallclock); + + /* + * When a task is migrating during the wakeup, adjust + * the task's contribution towards cumulative window + * demand. + */ + if (pstate == TASK_WAKING && p->wts.last_sleep_ts >= + src_rq->wrq.window_start) { + walt_fixup_cum_window_demand(src_rq, + -(s64)p->wts.demand_scaled); + walt_fixup_cum_window_demand(dest_rq, p->wts.demand_scaled); + } + + new_task = is_new_task(p); + /* Protected by rq_lock */ + grp = p->wts.grp; + + /* + * For frequency aggregation, we continue to do migration fixups + * even for intra cluster migrations. This is because, the aggregated + * load has to reported on a single CPU regardless. + */ + if (grp) { + struct group_cpu_time *cpu_time; + + cpu_time = &src_rq->wrq.grp_time; + src_curr_runnable_sum = &cpu_time->curr_runnable_sum; + src_prev_runnable_sum = &cpu_time->prev_runnable_sum; + src_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + cpu_time = &dest_rq->wrq.grp_time; + dst_curr_runnable_sum = &cpu_time->curr_runnable_sum; + dst_prev_runnable_sum = &cpu_time->prev_runnable_sum; + dst_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + dst_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + if (p->wts.curr_window) { + *src_curr_runnable_sum -= p->wts.curr_window; + *dst_curr_runnable_sum += p->wts.curr_window; + if (new_task) { + *src_nt_curr_runnable_sum -= + p->wts.curr_window; + *dst_nt_curr_runnable_sum += + p->wts.curr_window; + } + } + + if (p->wts.prev_window) { + *src_prev_runnable_sum -= p->wts.prev_window; + *dst_prev_runnable_sum += p->wts.prev_window; + if (new_task) { + *src_nt_prev_runnable_sum -= + p->wts.prev_window; + *dst_nt_prev_runnable_sum += + p->wts.prev_window; + } + } + } else { + inter_cluster_migration_fixup(p, new_cpu, + task_cpu(p), new_task); + } + + migrate_top_tasks(p, src_rq, dest_rq); + + if (!same_freq_domain(new_cpu, task_cpu(p))) { + src_rq->wrq.notif_pending = true; + dest_rq->wrq.notif_pending = true; + walt_irq_work_queue(&walt_migration_irq_work); + } + + if (is_ed_enabled()) { + if (p == src_rq->wrq.ed_task) { + src_rq->wrq.ed_task = NULL; + dest_rq->wrq.ed_task = p; + } else if (is_ed_task(p, wallclock)) { + dest_rq->wrq.ed_task = p; + } + } + + if (pstate == TASK_WAKING) + double_rq_unlock(src_rq, dest_rq); +} + +void set_window_start(struct rq *rq) +{ + static int sync_cpu_available; + + if (likely(rq->wrq.window_start)) + return; + + if (!sync_cpu_available) { + rq->wrq.window_start = 1; + sync_cpu_available = 1; + atomic64_set(&walt_irq_work_lastq_ws, rq->wrq.window_start); + walt_load_reported_window = + atomic64_read(&walt_irq_work_lastq_ws); + + } else { + struct rq *sync_rq = cpu_rq(cpumask_any(cpu_online_mask)); + + raw_spin_unlock(&rq->lock); + double_rq_lock(rq, sync_rq); + rq->wrq.window_start = sync_rq->wrq.window_start; + rq->wrq.curr_runnable_sum = rq->wrq.prev_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = rq->wrq.nt_prev_runnable_sum = 0; + raw_spin_unlock(&sync_rq->lock); + } + + rq->curr->wts.mark_start = rq->wrq.window_start; +} + +unsigned int sysctl_sched_conservative_pl; +unsigned int sysctl_sched_many_wakeup_threshold = WALT_MANY_WAKEUP_DEFAULT; + +#define INC_STEP 8 +#define DEC_STEP 2 +#define CONSISTENT_THRES 16 +#define INC_STEP_BIG 16 +/* + * bucket_increase - update the count of all buckets + * + * @buckets: array of buckets tracking busy time of a task + * @idx: the index of bucket to be incremented + * + * Each time a complete window finishes, count of bucket that runtime + * falls in (@idx) is incremented. Counts of all other buckets are + * decayed. The rate of increase and decay could be different based + * on current count in the bucket. + */ +static inline void bucket_increase(u8 *buckets, int idx) +{ + int i, step; + + for (i = 0; i < NUM_BUSY_BUCKETS; i++) { + if (idx != i) { + if (buckets[i] > DEC_STEP) + buckets[i] -= DEC_STEP; + else + buckets[i] = 0; + } else { + step = buckets[i] >= CONSISTENT_THRES ? + INC_STEP_BIG : INC_STEP; + if (buckets[i] > U8_MAX - step) + buckets[i] = U8_MAX; + else + buckets[i] += step; + } + } +} + +static inline int busy_to_bucket(u32 normalized_rt) +{ + int bidx; + + bidx = mult_frac(normalized_rt, NUM_BUSY_BUCKETS, max_task_load()); + bidx = min(bidx, NUM_BUSY_BUCKETS - 1); + + /* + * Combine lowest two buckets. The lowest frequency falls into + * 2nd bucket and thus keep predicting lowest bucket is not + * useful. + */ + if (!bidx) + bidx++; + + return bidx; +} + +/* + * get_pred_busy - calculate predicted demand for a task on runqueue + * + * @p: task whose prediction is being updated + * @start: starting bucket. returned prediction should not be lower than + * this bucket. + * @runtime: runtime of the task. returned prediction should not be lower + * than this runtime. + * Note: @start can be derived from @runtime. It's passed in only to + * avoid duplicated calculation in some cases. + * + * A new predicted busy time is returned for task @p based on @runtime + * passed in. The function searches through buckets that represent busy + * time equal to or bigger than @runtime and attempts to find the bucket to + * to use for prediction. Once found, it searches through historical busy + * time and returns the latest that falls into the bucket. If no such busy + * time exists, it returns the medium of that bucket. + */ +static u32 get_pred_busy(struct task_struct *p, + int start, u32 runtime) +{ + int i; + u8 *buckets = p->wts.busy_buckets; + u32 *hist = p->wts.sum_history; + u32 dmin, dmax; + u64 cur_freq_runtime = 0; + int first = NUM_BUSY_BUCKETS, final; + u32 ret = runtime; + + /* skip prediction for new tasks due to lack of history */ + if (unlikely(is_new_task(p))) + goto out; + + /* find minimal bucket index to pick */ + for (i = start; i < NUM_BUSY_BUCKETS; i++) { + if (buckets[i]) { + first = i; + break; + } + } + /* if no higher buckets are filled, predict runtime */ + if (first >= NUM_BUSY_BUCKETS) + goto out; + + /* compute the bucket for prediction */ + final = first; + + /* determine demand range for the predicted bucket */ + if (final < 2) { + /* lowest two buckets are combined */ + dmin = 0; + final = 1; + } else { + dmin = mult_frac(final, max_task_load(), NUM_BUSY_BUCKETS); + } + dmax = mult_frac(final + 1, max_task_load(), NUM_BUSY_BUCKETS); + + /* + * search through runtime history and return first runtime that falls + * into the range of predicted bucket. + */ + for (i = 0; i < sched_ravg_hist_size; i++) { + if (hist[i] >= dmin && hist[i] < dmax) { + ret = hist[i]; + break; + } + } + /* no historical runtime within bucket found, use average of the bin */ + if (ret < dmin) + ret = (dmin + dmax) / 2; + /* + * when updating in middle of a window, runtime could be higher + * than all recorded history. Always predict at least runtime. + */ + ret = max(runtime, ret); +out: + trace_sched_update_pred_demand(p, runtime, + mult_frac((unsigned int)cur_freq_runtime, 100, + sched_ravg_window), ret); + return ret; +} + +static inline u32 calc_pred_demand(struct task_struct *p) +{ + if (p->wts.pred_demand >= p->wts.curr_window) + return p->wts.pred_demand; + + return get_pred_busy(p, busy_to_bucket(p->wts.curr_window), + p->wts.curr_window); +} + +/* + * predictive demand of a task is calculated at the window roll-over. + * if the task current window busy time exceeds the predicted + * demand, update it here to reflect the task needs. + */ +void update_task_pred_demand(struct rq *rq, struct task_struct *p, int event) +{ + u32 new, old; + u16 new_scaled; + + if (!sched_predl) + return; + + if (is_idle_task(p)) + return; + + if (event != PUT_PREV_TASK && event != TASK_UPDATE && + (!SCHED_FREQ_ACCOUNT_WAIT_TIME || + (event != TASK_MIGRATE && + event != PICK_NEXT_TASK))) + return; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (!p->on_rq && !SCHED_FREQ_ACCOUNT_WAIT_TIME) + return; + } + + new = calc_pred_demand(p); + old = p->wts.pred_demand; + + if (old >= new) + return; + + new_scaled = scale_demand(new); + if (task_on_rq_queued(p) && (!task_has_dl_policy(p) || + !p->dl.dl_throttled)) + fixup_walt_sched_stats_common(rq, p, + p->wts.demand_scaled, + new_scaled); + + p->wts.pred_demand = new; + p->wts.pred_demand_scaled = new_scaled; +} + +void clear_top_tasks_bitmap(unsigned long *bitmap) +{ + memset(bitmap, 0, top_tasks_bitmap_size); + __set_bit(NUM_LOAD_INDICES, bitmap); +} + +static inline void clear_top_tasks_table(u8 *table) +{ + memset(table, 0, NUM_LOAD_INDICES * sizeof(u8)); +} + +static void update_top_tasks(struct task_struct *p, struct rq *rq, + u32 old_curr_window, int new_window, bool full_window) +{ + u8 curr = rq->wrq.curr_table; + u8 prev = 1 - curr; + u8 *curr_table = rq->wrq.top_tasks[curr]; + u8 *prev_table = rq->wrq.top_tasks[prev]; + int old_index, new_index, update_index; + u32 curr_window = p->wts.curr_window; + u32 prev_window = p->wts.prev_window; + bool zero_index_update; + + if (old_curr_window == curr_window && !new_window) + return; + + old_index = load_to_index(old_curr_window); + new_index = load_to_index(curr_window); + + if (!new_window) { + zero_index_update = !old_curr_window && curr_window; + if (old_index != new_index || zero_index_update) { + if (old_curr_window) + curr_table[old_index] -= 1; + if (curr_window) + curr_table[new_index] += 1; + if (new_index > rq->wrq.curr_top) + rq->wrq.curr_top = new_index; + } + + if (!curr_table[old_index]) + __clear_bit(NUM_LOAD_INDICES - old_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + + if (curr_table[new_index] == 1) + __set_bit(NUM_LOAD_INDICES - new_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + + return; + } + + /* + * The window has rolled over for this task. By the time we get + * here, curr/prev swaps would has already occurred. So we need + * to use prev_window for the new index. + */ + update_index = load_to_index(prev_window); + + if (full_window) { + /* + * Two cases here. Either 'p' ran for the entire window or + * it didn't run at all. In either case there is no entry + * in the prev table. If 'p' ran the entire window, we just + * need to create a new entry in the prev table. In this case + * update_index will be correspond to sched_ravg_window + * so we can unconditionally update the top index. + */ + if (prev_window) { + prev_table[update_index] += 1; + rq->wrq.prev_top = update_index; + } + + if (prev_table[update_index] == 1) + __set_bit(NUM_LOAD_INDICES - update_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + } else { + zero_index_update = !old_curr_window && prev_window; + if (old_index != update_index || zero_index_update) { + if (old_curr_window) + prev_table[old_index] -= 1; + + prev_table[update_index] += 1; + + if (update_index > rq->wrq.prev_top) + rq->wrq.prev_top = update_index; + + if (!prev_table[old_index]) + __clear_bit(NUM_LOAD_INDICES - old_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + + if (prev_table[update_index] == 1) + __set_bit(NUM_LOAD_INDICES - update_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + } + } + + if (curr_window) { + curr_table[new_index] += 1; + + if (new_index > rq->wrq.curr_top) + rq->wrq.curr_top = new_index; + + if (curr_table[new_index] == 1) + __set_bit(NUM_LOAD_INDICES - new_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + } +} + +static void rollover_top_tasks(struct rq *rq, bool full_window) +{ + u8 curr_table = rq->wrq.curr_table; + u8 prev_table = 1 - curr_table; + int curr_top = rq->wrq.curr_top; + + clear_top_tasks_table(rq->wrq.top_tasks[prev_table]); + clear_top_tasks_bitmap(rq->wrq.top_tasks_bitmap[prev_table]); + + if (full_window) { + curr_top = 0; + clear_top_tasks_table(rq->wrq.top_tasks[curr_table]); + clear_top_tasks_bitmap( + rq->wrq.top_tasks_bitmap[curr_table]); + } + + rq->wrq.curr_table = prev_table; + rq->wrq.prev_top = curr_top; + rq->wrq.curr_top = 0; +} + +static u32 empty_windows[NR_CPUS]; + +static void rollover_task_window(struct task_struct *p, bool full_window) +{ + u32 *curr_cpu_windows = empty_windows; + u32 curr_window; + int i; + + /* Rollover the sum */ + curr_window = 0; + + if (!full_window) { + curr_window = p->wts.curr_window; + curr_cpu_windows = p->wts.curr_window_cpu; + } + + p->wts.prev_window = curr_window; + p->wts.curr_window = 0; + + /* Roll over individual CPU contributions */ + for (i = 0; i < nr_cpu_ids; i++) { + p->wts.prev_window_cpu[i] = curr_cpu_windows[i]; + p->wts.curr_window_cpu[i] = 0; + } + + if (is_new_task(p)) + p->wts.active_time += task_rq(p)->wrq.prev_window_size; +} + +void sched_set_io_is_busy(int val) +{ + sched_io_is_busy = val; +} + +static inline int cpu_is_waiting_on_io(struct rq *rq) +{ + if (!sched_io_is_busy) + return 0; + + return atomic_read(&rq->nr_iowait); +} + +static int account_busy_for_cpu_time(struct rq *rq, struct task_struct *p, + u64 irqtime, int event) +{ + if (is_idle_task(p)) { + /* TASK_WAKE && TASK_MIGRATE is not possible on idle task! */ + if (event == PICK_NEXT_TASK) + return 0; + + /* PUT_PREV_TASK, TASK_UPDATE && IRQ_UPDATE are left */ + return irqtime || cpu_is_waiting_on_io(rq); + } + + if (event == TASK_WAKE) + return 0; + + if (event == PUT_PREV_TASK || event == IRQ_UPDATE) + return 1; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (rq->curr == p) + return 1; + + return p->on_rq ? SCHED_FREQ_ACCOUNT_WAIT_TIME : 0; + } + + /* TASK_MIGRATE, PICK_NEXT_TASK left */ + return SCHED_FREQ_ACCOUNT_WAIT_TIME; +} + +#define DIV64_U64_ROUNDUP(X, Y) div64_u64((X) + (Y - 1), Y) + +static inline u64 scale_exec_time(u64 delta, struct rq *rq) +{ + return (delta * rq->wrq.task_exec_scale) >> 10; +} + +/* Convert busy time to frequency equivalent + * Assumes load is scaled to 1024 + */ +static inline unsigned int load_to_freq(struct rq *rq, unsigned int load) +{ + return mult_frac(cpu_max_possible_freq(cpu_of(rq)), load, + (unsigned int)arch_scale_cpu_capacity(cpu_of(rq))); +} + +bool do_pl_notif(struct rq *rq) +{ + u64 prev = rq->wrq.old_busy_time; + u64 pl = rq->wrq.walt_stats.pred_demands_sum_scaled; + int cpu = cpu_of(rq); + + /* If already at max freq, bail out */ + if (capacity_orig_of(cpu) == capacity_curr_of(cpu)) + return false; + + prev = max(prev, rq->wrq.old_estimated_time); + + /* 400 MHz filter. */ + return (pl > prev) && (load_to_freq(rq, pl - prev) > 400000); +} + +static void rollover_cpu_window(struct rq *rq, bool full_window) +{ + u64 curr_sum = rq->wrq.curr_runnable_sum; + u64 nt_curr_sum = rq->wrq.nt_curr_runnable_sum; + u64 grp_curr_sum = rq->wrq.grp_time.curr_runnable_sum; + u64 grp_nt_curr_sum = rq->wrq.grp_time.nt_curr_runnable_sum; + + if (unlikely(full_window)) { + curr_sum = 0; + nt_curr_sum = 0; + grp_curr_sum = 0; + grp_nt_curr_sum = 0; + } + + rq->wrq.prev_runnable_sum = curr_sum; + rq->wrq.nt_prev_runnable_sum = nt_curr_sum; + rq->wrq.grp_time.prev_runnable_sum = grp_curr_sum; + rq->wrq.grp_time.nt_prev_runnable_sum = grp_nt_curr_sum; + + rq->wrq.curr_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = 0; + rq->wrq.grp_time.curr_runnable_sum = 0; + rq->wrq.grp_time.nt_curr_runnable_sum = 0; +} + +/* + * Account cpu activity in its + * busy time counters(rq->wrq.curr/prev_runnable_sum) + */ +static void update_cpu_busy_time(struct task_struct *p, struct rq *rq, + int event, u64 wallclock, u64 irqtime) +{ + int new_window, full_window = 0; + int p_is_curr_task = (p == rq->curr); + u64 mark_start = p->wts.mark_start; + u64 window_start = rq->wrq.window_start; + u32 window_size = rq->wrq.prev_window_size; + u64 delta; + u64 *curr_runnable_sum = &rq->wrq.curr_runnable_sum; + u64 *prev_runnable_sum = &rq->wrq.prev_runnable_sum; + u64 *nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + u64 *nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + bool new_task; + struct walt_related_thread_group *grp; + int cpu = rq->cpu; + u32 old_curr_window = p->wts.curr_window; + + new_window = mark_start < window_start; + if (new_window) + full_window = (window_start - mark_start) >= window_size; + + /* + * Handle per-task window rollover. We don't care about the + * idle task. + */ + if (!is_idle_task(p)) { + if (new_window) + rollover_task_window(p, full_window); + } + + new_task = is_new_task(p); + + if (p_is_curr_task && new_window) { + rollover_cpu_window(rq, full_window); + rollover_top_tasks(rq, full_window); + } + + if (!account_busy_for_cpu_time(rq, p, irqtime, event)) + goto done; + + grp = p->wts.grp; + if (grp) { + struct group_cpu_time *cpu_time = &rq->wrq.grp_time; + + curr_runnable_sum = &cpu_time->curr_runnable_sum; + prev_runnable_sum = &cpu_time->prev_runnable_sum; + + nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + } + + if (!new_window) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. No rollover + * since we didn't start a new window. An example of this is + * when a task starts execution and then sleeps within the + * same window. + */ + + if (!irqtime || !is_idle_task(p) || cpu_is_waiting_on_io(rq)) + delta = wallclock - mark_start; + else + delta = irqtime; + delta = scale_exec_time(delta, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + if (!is_idle_task(p)) { + p->wts.curr_window += delta; + p->wts.curr_window_cpu[cpu] += delta; + } + + goto done; + } + + if (!p_is_curr_task) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has also started, but p is not the current task, so the + * window is not rolled over - just split up and account + * as necessary into curr and prev. The window is only + * rolled over when a new window is processed for the current + * task. + * + * Irqtime can't be accounted by a task that isn't the + * currently running task. + */ + + if (!full_window) { + /* + * A full window hasn't elapsed, account partial + * contribution to previous completed window. + */ + delta = scale_exec_time(window_start - mark_start, rq); + p->wts.prev_window += delta; + p->wts.prev_window_cpu[cpu] += delta; + } else { + /* + * Since at least one full window has elapsed, + * the contribution to the previous window is the + * full window (window_size). + */ + delta = scale_exec_time(window_size, rq); + p->wts.prev_window = delta; + p->wts.prev_window_cpu[cpu] = delta; + } + + *prev_runnable_sum += delta; + if (new_task) + *nt_prev_runnable_sum += delta; + + /* Account piece of busy time in the current window. */ + delta = scale_exec_time(wallclock - window_start, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + p->wts.curr_window = delta; + p->wts.curr_window_cpu[cpu] = delta; + + goto done; + } + + if (!irqtime || !is_idle_task(p) || cpu_is_waiting_on_io(rq)) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has started and p is the current task so rollover is + * needed. If any of these three above conditions are true + * then this busy time can't be accounted as irqtime. + * + * Busy time for the idle task need not be accounted. + * + * An example of this would be a task that starts execution + * and then sleeps once a new window has begun. + */ + + if (!full_window) { + /* + * A full window hasn't elapsed, account partial + * contribution to previous completed window. + */ + delta = scale_exec_time(window_start - mark_start, rq); + if (!is_idle_task(p)) { + p->wts.prev_window += delta; + p->wts.prev_window_cpu[cpu] += delta; + } + } else { + /* + * Since at least one full window has elapsed, + * the contribution to the previous window is the + * full window (window_size). + */ + delta = scale_exec_time(window_size, rq); + if (!is_idle_task(p)) { + p->wts.prev_window = delta; + p->wts.prev_window_cpu[cpu] = delta; + } + } + + /* + * Rollover is done here by overwriting the values in + * prev_runnable_sum and curr_runnable_sum. + */ + *prev_runnable_sum += delta; + if (new_task) + *nt_prev_runnable_sum += delta; + + /* Account piece of busy time in the current window. */ + delta = scale_exec_time(wallclock - window_start, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + if (!is_idle_task(p)) { + p->wts.curr_window = delta; + p->wts.curr_window_cpu[cpu] = delta; + } + + goto done; + } + + if (irqtime) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has started and p is the current task so rollover is + * needed. The current task must be the idle task because + * irqtime is not accounted for any other task. + * + * Irqtime will be accounted each time we process IRQ activity + * after a period of idleness, so we know the IRQ busy time + * started at wallclock - irqtime. + */ + + SCHED_BUG_ON(!is_idle_task(p)); + mark_start = wallclock - irqtime; + + /* + * Roll window over. If IRQ busy time was just in the current + * window then that is all that need be accounted. + */ + if (mark_start > window_start) { + *curr_runnable_sum = scale_exec_time(irqtime, rq); + return; + } + + /* + * The IRQ busy time spanned multiple windows. Process the + * busy time preceding the current window start first. + */ + delta = window_start - mark_start; + if (delta > window_size) + delta = window_size; + delta = scale_exec_time(delta, rq); + *prev_runnable_sum += delta; + + /* Process the remaining IRQ busy time in the current window. */ + delta = wallclock - window_start; + rq->wrq.curr_runnable_sum = scale_exec_time(delta, rq); + + return; + } + +done: + if (!is_idle_task(p)) + update_top_tasks(p, rq, old_curr_window, + new_window, full_window); +} + + +static inline u32 predict_and_update_buckets( + struct task_struct *p, u32 runtime) { + + int bidx; + u32 pred_demand; + + if (!sched_predl) + return 0; + + bidx = busy_to_bucket(runtime); + pred_demand = get_pred_busy(p, bidx, runtime); + bucket_increase(p->wts.busy_buckets, bidx); + + return pred_demand; +} + +static int +account_busy_for_task_demand(struct rq *rq, struct task_struct *p, int event) +{ + /* + * No need to bother updating task demand for the idle task. + */ + if (is_idle_task(p)) + return 0; + + /* + * When a task is waking up it is completing a segment of non-busy + * time. Likewise, if wait time is not treated as busy time, then + * when a task begins to run or is migrated, it is not running and + * is completing a segment of non-busy time. + */ + if (event == TASK_WAKE || (!SCHED_ACCOUNT_WAIT_TIME && + (event == PICK_NEXT_TASK || event == TASK_MIGRATE))) + return 0; + + /* + * The idle exit time is not accounted for the first task _picked_ up to + * run on the idle CPU. + */ + if (event == PICK_NEXT_TASK && rq->curr == rq->idle) + return 0; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (rq->curr == p) + return 1; + + return p->on_rq ? SCHED_ACCOUNT_WAIT_TIME : 0; + } + + return 1; +} + +unsigned int sysctl_sched_task_unfilter_period = 100000000; + +/* + * Called when new window is starting for a task, to record cpu usage over + * recently concluded window(s). Normally 'samples' should be 1. It can be > 1 + * when, say, a real-time task runs without preemption for several windows at a + * stretch. + */ +static void update_history(struct rq *rq, struct task_struct *p, + u32 runtime, int samples, int event) +{ + u32 *hist = &p->wts.sum_history[0]; + int ridx, widx; + u32 max = 0, avg, demand, pred_demand; + u64 sum = 0; + u16 demand_scaled, pred_demand_scaled; + + /* Ignore windows where task had no activity */ + if (!runtime || is_idle_task(p) || !samples) + goto done; + + /* Push new 'runtime' value onto stack */ + widx = sched_ravg_hist_size - 1; + ridx = widx - samples; + for (; ridx >= 0; --widx, --ridx) { + hist[widx] = hist[ridx]; + sum += hist[widx]; + if (hist[widx] > max) + max = hist[widx]; + } + + for (widx = 0; widx < samples && widx < sched_ravg_hist_size; widx++) { + hist[widx] = runtime; + sum += hist[widx]; + if (hist[widx] > max) + max = hist[widx]; + } + + p->wts.sum = 0; + + if (sysctl_sched_window_stats_policy == WINDOW_STATS_RECENT) { + demand = runtime; + } else if (sysctl_sched_window_stats_policy == WINDOW_STATS_MAX) { + demand = max; + } else { + avg = div64_u64(sum, sched_ravg_hist_size); + if (sysctl_sched_window_stats_policy == WINDOW_STATS_AVG) + demand = avg; + else + demand = max(avg, runtime); + } + pred_demand = predict_and_update_buckets(p, runtime); + demand_scaled = scale_demand(demand); + pred_demand_scaled = scale_demand(pred_demand); + + /* + * A throttled deadline sched class task gets dequeued without + * changing p->on_rq. Since the dequeue decrements walt stats + * avoid decrementing it here again. + * + * When window is rolled over, the cumulative window demand + * is reset to the cumulative runnable average (contribution from + * the tasks on the runqueue). If the current task is dequeued + * already, it's demand is not included in the cumulative runnable + * average. So add the task demand separately to cumulative window + * demand. + */ + if (!task_has_dl_policy(p) || !p->dl.dl_throttled) { + if (task_on_rq_queued(p)) + fixup_walt_sched_stats_common(rq, p, + demand_scaled, pred_demand_scaled); + else if (rq->curr == p) + walt_fixup_cum_window_demand(rq, demand_scaled); + } + + p->wts.demand = demand; + p->wts.demand_scaled = demand_scaled; + p->wts.coloc_demand = div64_u64(sum, sched_ravg_hist_size); + p->wts.pred_demand = pred_demand; + p->wts.pred_demand_scaled = pred_demand_scaled; + + if (demand_scaled > sysctl_sched_min_task_util_for_colocation) + p->wts.unfilter = sysctl_sched_task_unfilter_period; + else + if (p->wts.unfilter) + p->wts.unfilter = max_t(int, 0, + p->wts.unfilter - rq->wrq.prev_window_size); + +done: + trace_sched_update_history(rq, p, runtime, samples, event); +} + +static u64 add_to_task_demand(struct rq *rq, struct task_struct *p, u64 delta) +{ + delta = scale_exec_time(delta, rq); + p->wts.sum += delta; + if (unlikely(p->wts.sum > sched_ravg_window)) + p->wts.sum = sched_ravg_window; + + return delta; +} + +/* + * Account cpu demand of task and/or update task's cpu demand history + * + * ms = p->wts.mark_start; + * wc = wallclock + * ws = rq->wrq.window_start + * + * Three possibilities: + * + * a) Task event is contained within one window. + * window_start < mark_start < wallclock + * + * ws ms wc + * | | | + * V V V + * |---------------| + * + * In this case, p->wts.sum is updated *iff* event is appropriate + * (ex: event == PUT_PREV_TASK) + * + * b) Task event spans two windows. + * mark_start < window_start < wallclock + * + * ms ws wc + * | | | + * V V V + * -----|------------------- + * + * In this case, p->wts.sum is updated with (ws - ms) *iff* event + * is appropriate, then a new window sample is recorded followed + * by p->wts.sum being set to (wc - ws) *iff* event is appropriate. + * + * c) Task event spans more than two windows. + * + * ms ws_tmp ws wc + * | | | | + * V V V V + * ---|-------|-------|-------|-------|------ + * | | + * |<------ nr_full_windows ------>| + * + * In this case, p->wts.sum is updated with (ws_tmp - ms) first *iff* + * event is appropriate, window sample of p->wts.sum is recorded, + * 'nr_full_window' samples of window_size is also recorded *iff* + * event is appropriate and finally p->wts.sum is set to (wc - ws) + * *iff* event is appropriate. + * + * IMPORTANT : Leave p->wts.mark_start unchanged, as update_cpu_busy_time() + * depends on it! + */ +static u64 update_task_demand(struct task_struct *p, struct rq *rq, + int event, u64 wallclock) +{ + u64 mark_start = p->wts.mark_start; + u64 delta, window_start = rq->wrq.window_start; + int new_window, nr_full_windows; + u32 window_size = sched_ravg_window; + u64 runtime; + + new_window = mark_start < window_start; + if (!account_busy_for_task_demand(rq, p, event)) { + if (new_window) + /* + * If the time accounted isn't being accounted as + * busy time, and a new window started, only the + * previous window need be closed out with the + * pre-existing demand. Multiple windows may have + * elapsed, but since empty windows are dropped, + * it is not necessary to account those. + */ + update_history(rq, p, p->wts.sum, 1, event); + return 0; + } + + if (!new_window) { + /* + * The simple case - busy time contained within the existing + * window. + */ + return add_to_task_demand(rq, p, wallclock - mark_start); + } + + /* + * Busy time spans at least two windows. Temporarily rewind + * window_start to first window boundary after mark_start. + */ + delta = window_start - mark_start; + nr_full_windows = div64_u64(delta, window_size); + window_start -= (u64)nr_full_windows * (u64)window_size; + + /* Process (window_start - mark_start) first */ + runtime = add_to_task_demand(rq, p, window_start - mark_start); + + /* Push new sample(s) into task's demand history */ + update_history(rq, p, p->wts.sum, 1, event); + if (nr_full_windows) { + u64 scaled_window = scale_exec_time(window_size, rq); + + update_history(rq, p, scaled_window, nr_full_windows, event); + runtime += nr_full_windows * scaled_window; + } + + /* + * Roll window_start back to current to process any remainder + * in current window. + */ + window_start += (u64)nr_full_windows * (u64)window_size; + + /* Process (wallclock - window_start) next */ + mark_start = window_start; + runtime += add_to_task_demand(rq, p, wallclock - mark_start); + + return runtime; +} + +static inline unsigned int cpu_cur_freq(int cpu) +{ + return cpu_rq(cpu)->wrq.cluster->cur_freq; +} + +static void +update_task_rq_cpu_cycles(struct task_struct *p, struct rq *rq, int event, + u64 wallclock, u64 irqtime) +{ + u64 cur_cycles; + u64 cycles_delta; + u64 time_delta; + int cpu = cpu_of(rq); + + lockdep_assert_held(&rq->lock); + + if (!use_cycle_counter) { + rq->wrq.task_exec_scale = DIV64_U64_ROUNDUP(cpu_cur_freq(cpu) * + arch_scale_cpu_capacity(cpu), + rq->wrq.cluster->max_possible_freq); + return; + } + + cur_cycles = read_cycle_counter(cpu, wallclock); + + /* + * If current task is idle task and irqtime == 0 CPU was + * indeed idle and probably its cycle counter was not + * increasing. We still need estimatied CPU frequency + * for IO wait time accounting. Use the previously + * calculated frequency in such a case. + */ + if (!is_idle_task(rq->curr) || irqtime) { + if (unlikely(cur_cycles < p->wts.cpu_cycles)) + cycles_delta = cur_cycles + (U64_MAX - + p->wts.cpu_cycles); + else + cycles_delta = cur_cycles - p->wts.cpu_cycles; + cycles_delta = cycles_delta * NSEC_PER_MSEC; + + if (event == IRQ_UPDATE && is_idle_task(p)) + /* + * Time between mark_start of idle task and IRQ handler + * entry time is CPU cycle counter stall period. + * Upon IRQ handler entry walt_sched_account_irqstart() + * replenishes idle task's cpu cycle counter so + * cycles_delta now represents increased cycles during + * IRQ handler rather than time between idle entry and + * IRQ exit. Thus use irqtime as time delta. + */ + time_delta = irqtime; + else + time_delta = wallclock - p->wts.mark_start; + SCHED_BUG_ON((s64)time_delta < 0); + + rq->wrq.task_exec_scale = DIV64_U64_ROUNDUP(cycles_delta * + arch_scale_cpu_capacity(cpu), + time_delta * + rq->wrq.cluster->max_possible_freq); + + trace_sched_get_task_cpu_cycles(cpu, event, + cycles_delta, time_delta, p); + } + + p->wts.cpu_cycles = cur_cycles; +} + +static inline void run_walt_irq_work(u64 old_window_start, struct rq *rq) +{ + u64 result; + + if (old_window_start == rq->wrq.window_start) + return; + + result = atomic64_cmpxchg(&walt_irq_work_lastq_ws, old_window_start, + rq->wrq.window_start); + if (result == old_window_start) { + walt_irq_work_queue(&walt_cpufreq_irq_work); + trace_walt_window_rollover(rq->wrq.window_start); + } +} + +/* Reflect task activity on its demand and cpu's busy time statistics */ +void walt_update_task_ravg(struct task_struct *p, struct rq *rq, int event, + u64 wallclock, u64 irqtime) +{ + u64 old_window_start; + + if (!rq->wrq.window_start || p->wts.mark_start == wallclock) + return; + + lockdep_assert_held(&rq->lock); + + old_window_start = update_window_start(rq, wallclock, event); + + if (!p->wts.mark_start) { + update_task_cpu_cycles(p, cpu_of(rq), wallclock); + goto done; + } + + update_task_rq_cpu_cycles(p, rq, event, wallclock, irqtime); + update_task_demand(p, rq, event, wallclock); + update_cpu_busy_time(p, rq, event, wallclock, irqtime); + update_task_pred_demand(rq, p, event); + if (event == PUT_PREV_TASK && p->state) + p->wts.iowaited = p->in_iowait; + + trace_sched_update_task_ravg(p, rq, event, wallclock, irqtime, + &rq->wrq.grp_time); + trace_sched_update_task_ravg_mini(p, rq, event, wallclock, irqtime, + &rq->wrq.grp_time); + +done: + p->wts.mark_start = wallclock; + + run_walt_irq_work(old_window_start, rq); +} + +u32 sched_get_init_task_load(struct task_struct *p) +{ + return p->wts.init_load_pct; +} + +int sched_set_init_task_load(struct task_struct *p, int init_load_pct) +{ + if (init_load_pct < 0 || init_load_pct > 100) + return -EINVAL; + + p->wts.init_load_pct = init_load_pct; + + return 0; +} + +void init_new_task_load(struct task_struct *p) +{ + int i; + u32 init_load_windows = sched_init_task_load_windows; + u32 init_load_windows_scaled = sched_init_task_load_windows_scaled; + u32 init_load_pct = current->wts.init_load_pct; + + p->wts.init_load_pct = 0; + rcu_assign_pointer(p->wts.grp, NULL); + INIT_LIST_HEAD(&p->wts.grp_list); + + p->wts.mark_start = 0; + p->wts.sum = 0; + p->wts.curr_window = 0; + p->wts.prev_window = 0; + p->wts.active_time = 0; + for (i = 0; i < NUM_BUSY_BUCKETS; ++i) + p->wts.busy_buckets[i] = 0; + + p->wts.cpu_cycles = 0; + + p->wts.curr_window_cpu = kcalloc(nr_cpu_ids, sizeof(u32), + GFP_KERNEL | __GFP_NOFAIL); + p->wts.prev_window_cpu = kcalloc(nr_cpu_ids, sizeof(u32), + GFP_KERNEL | __GFP_NOFAIL); + + if (init_load_pct) { + init_load_windows = div64_u64((u64)init_load_pct * + (u64)sched_ravg_window, 100); + init_load_windows_scaled = scale_demand(init_load_windows); + } + + p->wts.demand = init_load_windows; + p->wts.demand_scaled = init_load_windows_scaled; + p->wts.coloc_demand = init_load_windows; + p->wts.pred_demand = 0; + p->wts.pred_demand_scaled = 0; + for (i = 0; i < RAVG_HIST_SIZE_MAX; ++i) + p->wts.sum_history[i] = init_load_windows; + p->wts.misfit = false; + p->wts.rtg_high_prio = false; + p->wts.unfilter = sysctl_sched_task_unfilter_period; +} + +/* + * kfree() may wakeup kswapd. So this function should NOT be called + * with any CPU's rq->lock acquired. + */ +void free_task_load_ptrs(struct task_struct *p) +{ + kfree(p->wts.curr_window_cpu); + kfree(p->wts.prev_window_cpu); + + /* + * walt_update_task_ravg() can be called for exiting tasks. While the + * function itself ensures correct behavior, the corresponding + * trace event requires that these pointers be NULL. + */ + p->wts.curr_window_cpu = NULL; + p->wts.prev_window_cpu = NULL; +} + +void walt_task_dead(struct task_struct *p) +{ + sched_set_group_id(p, 0); + free_task_load_ptrs(p); +} + +void reset_task_stats(struct task_struct *p) +{ + int i = 0; + u32 *curr_window_ptr; + u32 *prev_window_ptr; + + curr_window_ptr = p->wts.curr_window_cpu; + prev_window_ptr = p->wts.prev_window_cpu; + memset(curr_window_ptr, 0, sizeof(u32) * nr_cpu_ids); + memset(prev_window_ptr, 0, sizeof(u32) * nr_cpu_ids); + + p->wts.mark_start = 0; + p->wts.sum = 0; + p->wts.demand = 0; + p->wts.coloc_demand = 0; + for (i = 0; i < RAVG_HIST_SIZE_MAX; ++i) + p->wts.sum_history[i] = 0; + p->wts.curr_window = 0; + p->wts.prev_window = 0; + p->wts.pred_demand = 0; + for (i = 0; i < NUM_BUSY_BUCKETS; ++i) + p->wts.busy_buckets[i] = 0; + p->wts.demand_scaled = 0; + p->wts.pred_demand_scaled = 0; + p->wts.active_time = 0; + + p->wts.curr_window_cpu = curr_window_ptr; + p->wts.prev_window_cpu = prev_window_ptr; +} + +void mark_task_starting(struct task_struct *p) +{ + u64 wallclock; + struct rq *rq = task_rq(p); + + if (!rq->wrq.window_start) { + reset_task_stats(p); + return; + } + + wallclock = sched_ktime_clock(); + p->wts.mark_start = p->wts.last_wake_ts = wallclock; + p->wts.last_enqueued_ts = wallclock; + update_task_cpu_cycles(p, cpu_of(rq), wallclock); +} + +/* + * Task groups whose aggregate demand on a cpu is more than + * sched_group_upmigrate need to be up-migrated if possible. + */ +unsigned int __read_mostly sched_group_upmigrate = 20000000; +unsigned int __read_mostly sysctl_sched_group_upmigrate_pct = 100; + +/* + * Task groups, once up-migrated, will need to drop their aggregate + * demand to less than sched_group_downmigrate before they are "down" + * migrated. + */ +unsigned int __read_mostly sched_group_downmigrate = 19000000; +unsigned int __read_mostly sysctl_sched_group_downmigrate_pct = 95; + +static inline void walt_update_group_thresholds(void) +{ + unsigned int min_scale = arch_scale_cpu_capacity( + cluster_first_cpu(sched_cluster[0])); + u64 min_ms = min_scale * (sched_ravg_window >> SCHED_CAPACITY_SHIFT); + + sched_group_upmigrate = div64_ul(min_ms * + sysctl_sched_group_upmigrate_pct, 100); + sched_group_downmigrate = div64_ul(min_ms * + sysctl_sched_group_downmigrate_pct, 100); +} + +struct walt_sched_cluster *sched_cluster[NR_CPUS]; +__read_mostly int num_sched_clusters; + +struct list_head cluster_head; +cpumask_t asym_cap_sibling_cpus = CPU_MASK_NONE; + +static struct walt_sched_cluster init_cluster = { + .list = LIST_HEAD_INIT(init_cluster.list), + .id = 0, + .cur_freq = 1, + .max_possible_freq = 1, + .aggr_grp_load = 0, +}; + +void init_clusters(void) +{ + init_cluster.cpus = *cpu_possible_mask; + raw_spin_lock_init(&init_cluster.load_lock); + INIT_LIST_HEAD(&cluster_head); + list_add(&init_cluster.list, &cluster_head); +} + +static void +insert_cluster(struct walt_sched_cluster *cluster, struct list_head *head) +{ + struct walt_sched_cluster *tmp; + struct list_head *iter = head; + + list_for_each_entry(tmp, head, list) { + if (arch_scale_cpu_capacity(cluster_first_cpu(cluster)) + < arch_scale_cpu_capacity(cluster_first_cpu(tmp))) + break; + iter = &tmp->list; + } + + list_add(&cluster->list, iter); +} + +static struct walt_sched_cluster *alloc_new_cluster(const struct cpumask *cpus) +{ + struct walt_sched_cluster *cluster = NULL; + + cluster = kzalloc(sizeof(struct walt_sched_cluster), GFP_ATOMIC); + if (!cluster) { + pr_warn("Cluster allocation failed. Possible bad scheduling\n"); + return NULL; + } + + INIT_LIST_HEAD(&cluster->list); + cluster->cur_freq = 1; + cluster->max_possible_freq = 1; + + raw_spin_lock_init(&cluster->load_lock); + cluster->cpus = *cpus; + + return cluster; +} + +static void add_cluster(const struct cpumask *cpus, struct list_head *head) +{ + struct walt_sched_cluster *cluster = alloc_new_cluster(cpus); + int i; + + if (!cluster) + return; + + for_each_cpu(i, cpus) + cpu_rq(i)->wrq.cluster = cluster; + + insert_cluster(cluster, head); + num_sched_clusters++; +} + +static void cleanup_clusters(struct list_head *head) +{ + struct walt_sched_cluster *cluster, *tmp; + int i; + + list_for_each_entry_safe(cluster, tmp, head, list) { + for_each_cpu(i, &cluster->cpus) + cpu_rq(i)->wrq.cluster = &init_cluster; + + list_del(&cluster->list); + num_sched_clusters--; + kfree(cluster); + } +} + +static inline void assign_cluster_ids(struct list_head *head) +{ + struct walt_sched_cluster *cluster; + int pos = 0; + + list_for_each_entry(cluster, head, list) { + cluster->id = pos; + sched_cluster[pos++] = cluster; + } + + WARN_ON(pos > MAX_NR_CLUSTERS); +} + +static inline void +move_list(struct list_head *dst, struct list_head *src, bool sync_rcu) +{ + struct list_head *first, *last; + + first = src->next; + last = src->prev; + + if (sync_rcu) { + INIT_LIST_HEAD_RCU(src); + synchronize_rcu(); + } + + first->prev = dst; + dst->prev = last; + last->next = dst; + + /* Ensure list sanity before making the head visible to all CPUs. */ + smp_mb(); + dst->next = first; +} + +static void update_all_clusters_stats(void) +{ + struct walt_sched_cluster *cluster; + u64 highest_mpc = 0, lowest_mpc = U64_MAX; + unsigned long flags; + + acquire_rq_locks_irqsave(cpu_possible_mask, &flags); + + for_each_sched_cluster(cluster) { + u64 mpc = arch_scale_cpu_capacity( + cluster_first_cpu(cluster)); + + if (mpc > highest_mpc) + highest_mpc = mpc; + + if (mpc < lowest_mpc) + lowest_mpc = mpc; + } + + max_possible_capacity = highest_mpc; + min_max_possible_capacity = lowest_mpc; + walt_update_group_thresholds(); + + release_rq_locks_irqrestore(cpu_possible_mask, &flags); +} + +static bool walt_clusters_parsed; +__read_mostly cpumask_t **cpu_array; + +static cpumask_t **init_cpu_array(void) +{ + int i; + cpumask_t **tmp_array; + + tmp_array = kcalloc(num_sched_clusters, sizeof(cpumask_t *), + GFP_ATOMIC); + if (!tmp_array) + return NULL; + for (i = 0; i < num_sched_clusters; i++) { + tmp_array[i] = kcalloc(num_sched_clusters, sizeof(cpumask_t), + GFP_ATOMIC); + if (!tmp_array[i]) + return NULL; + } + + return tmp_array; +} + +static cpumask_t **build_cpu_array(void) +{ + int i; + cpumask_t **tmp_array = init_cpu_array(); + + if (!tmp_array) + return NULL; + + /*Construct cpu_array row by row*/ + for (i = 0; i < num_sched_clusters; i++) { + int j, k = 1; + + /* Fill out first column with appropriate cpu arrays*/ + cpumask_copy(&tmp_array[i][0], &sched_cluster[i]->cpus); + + /* + * k starts from column 1 because 0 is filled + * Fill clusters for the rest of the row, + * above i in ascending order + */ + for (j = i + 1; j < num_sched_clusters; j++) { + cpumask_copy(&tmp_array[i][k], + &sched_cluster[j]->cpus); + k++; + } + + /* + * k starts from where we left off above. + * Fill clusters below i in descending order. + */ + for (j = i - 1; j >= 0; j--) { + cpumask_copy(&tmp_array[i][k], + &sched_cluster[j]->cpus); + k++; + } + } + return tmp_array; +} + +static void walt_get_possible_siblings(int cpuid, struct cpumask *cluster_cpus) +{ + int cpu; + struct cpu_topology *cpu_topo, *cpuid_topo = &cpu_topology[cpuid]; + + if (cpuid_topo->package_id == -1) + return; + + for_each_possible_cpu(cpu) { + cpu_topo = &cpu_topology[cpu]; + + if (cpuid_topo->package_id != cpu_topo->package_id) + continue; + cpumask_set_cpu(cpu, cluster_cpus); + } +} + +void walt_update_cluster_topology(void) +{ + struct cpumask cpus = *cpu_possible_mask; + struct cpumask cluster_cpus; + struct walt_sched_cluster *cluster; + struct list_head new_head; + cpumask_t **tmp; + int i; + + INIT_LIST_HEAD(&new_head); + + for_each_cpu(i, &cpus) { + cpumask_clear(&cluster_cpus); + walt_get_possible_siblings(i, &cluster_cpus); + if (cpumask_empty(&cluster_cpus)) { + WARN(1, "WALT: Invalid cpu topology!!"); + cleanup_clusters(&new_head); + return; + } + cpumask_andnot(&cpus, &cpus, &cluster_cpus); + add_cluster(&cluster_cpus, &new_head); + } + + assign_cluster_ids(&new_head); + + list_for_each_entry(cluster, &new_head, list) { + struct cpufreq_policy *policy; + + policy = cpufreq_cpu_get_raw(cluster_first_cpu(cluster)); + /* + * walt_update_cluster_topology() must be called AFTER policies + * for all cpus are initialized. If not, simply BUG(). + */ + SCHED_BUG_ON(!policy); + + if (policy) { + cluster->max_possible_freq = policy->cpuinfo.max_freq; + + for_each_cpu(i, &cluster->cpus) + cpumask_copy(&cpu_rq(i)->wrq.freq_domain_cpumask, + policy->related_cpus); + } + } + + /* + * Ensure cluster ids are visible to all CPUs before making + * cluster_head visible. + */ + move_list(&cluster_head, &new_head, false); + update_all_clusters_stats(); + + for_each_sched_cluster(cluster) { + if (cpumask_weight(&cluster->cpus) == 1) + cpumask_or(&asym_cap_sibling_cpus, + &asym_cap_sibling_cpus, &cluster->cpus); + } + + if (cpumask_weight(&asym_cap_sibling_cpus) == 1) + cpumask_clear(&asym_cap_sibling_cpus); + + tmp = build_cpu_array(); + if (!tmp) { + BUG_ON(1); + return; + } + smp_store_release(&cpu_array, tmp); + walt_clusters_parsed = true; +} + +static int cpufreq_notifier_trans(struct notifier_block *nb, + unsigned long val, void *data) +{ + struct cpufreq_freqs *freq = (struct cpufreq_freqs *)data; + unsigned int cpu = freq->policy->cpu, new_freq = freq->new; + unsigned long flags; + struct walt_sched_cluster *cluster; + struct cpumask policy_cpus = cpu_rq(cpu)->wrq.freq_domain_cpumask; + int i, j; + + if (use_cycle_counter) + return NOTIFY_DONE; + + if (cpu_rq(cpumask_first(&policy_cpus))->wrq.cluster == &init_cluster) + return NOTIFY_DONE; + + if (val != CPUFREQ_POSTCHANGE) + return NOTIFY_DONE; + + if (cpu_cur_freq(cpu) == new_freq) + return NOTIFY_OK; + + for_each_cpu(i, &policy_cpus) { + cluster = cpu_rq(i)->wrq.cluster; + + for_each_cpu(j, &cluster->cpus) { + struct rq *rq = cpu_rq(j); + + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_task_ravg(rq->curr, rq, TASK_UPDATE, + sched_ktime_clock(), 0); + raw_spin_unlock_irqrestore(&rq->lock, flags); + } + + cluster->cur_freq = new_freq; + cpumask_andnot(&policy_cpus, &policy_cpus, &cluster->cpus); + } + + return NOTIFY_OK; +} + +static struct notifier_block notifier_trans_block = { + .notifier_call = cpufreq_notifier_trans +}; + +static int register_walt_callback(void) +{ + return cpufreq_register_notifier(¬ifier_trans_block, + CPUFREQ_TRANSITION_NOTIFIER); +} +/* + * cpufreq callbacks can be registered at core_initcall or later time. + * Any registration done prior to that is "forgotten" by cpufreq. See + * initialization of variable init_cpufreq_transition_notifier_list_called + * for further information. + */ +core_initcall(register_walt_callback); + +int register_cpu_cycle_counter_cb(struct cpu_cycle_counter_cb *cb) +{ + unsigned long flags; + + mutex_lock(&cluster_lock); + if (!cb->get_cpu_cycle_counter) { + mutex_unlock(&cluster_lock); + return -EINVAL; + } + + acquire_rq_locks_irqsave(cpu_possible_mask, &flags); + cpu_cycle_counter_cb = *cb; + use_cycle_counter = true; + release_rq_locks_irqrestore(cpu_possible_mask, &flags); + + mutex_unlock(&cluster_lock); + + cpufreq_unregister_notifier(¬ifier_trans_block, + CPUFREQ_TRANSITION_NOTIFIER); + return 0; +} + +static void transfer_busy_time(struct rq *rq, + struct walt_related_thread_group *grp, + struct task_struct *p, int event); + +/* + * Enable colocation and frequency aggregation for all threads in a process. + * The children inherits the group id from the parent. + */ +unsigned int __read_mostly sysctl_sched_coloc_downmigrate_ns; + +struct walt_related_thread_group + *related_thread_groups[MAX_NUM_CGROUP_COLOC_ID]; +static LIST_HEAD(active_related_thread_groups); +static DEFINE_RWLOCK(related_thread_group_lock); + +static inline +void update_best_cluster(struct walt_related_thread_group *grp, + u64 demand, bool boost) +{ + if (boost) { + /* + * since we are in boost, we can keep grp on min, the boosts + * will ensure tasks get to bigs + */ + grp->skip_min = false; + return; + } + + if (is_suh_max()) + demand = sched_group_upmigrate; + + if (!grp->skip_min) { + if (demand >= sched_group_upmigrate) { + grp->skip_min = true; + } + return; + } + if (demand < sched_group_downmigrate) { + if (!sysctl_sched_coloc_downmigrate_ns) { + grp->skip_min = false; + return; + } + if (!grp->downmigrate_ts) { + grp->downmigrate_ts = grp->last_update; + return; + } + if (grp->last_update - grp->downmigrate_ts > + sysctl_sched_coloc_downmigrate_ns) { + grp->downmigrate_ts = 0; + grp->skip_min = false; + } + } else if (grp->downmigrate_ts) + grp->downmigrate_ts = 0; +} + +int preferred_cluster(struct walt_sched_cluster *cluster, struct task_struct *p) +{ + struct walt_related_thread_group *grp; + int rc = -1; + + rcu_read_lock(); + + grp = task_related_thread_group(p); + if (grp) + rc = (sched_cluster[(int)grp->skip_min] == cluster || + cpumask_subset(&cluster->cpus, &asym_cap_sibling_cpus)); + + rcu_read_unlock(); + return rc; +} + +static void _set_preferred_cluster(struct walt_related_thread_group *grp) +{ + struct task_struct *p; + u64 combined_demand = 0; + bool group_boost = false; + u64 wallclock; + bool prev_skip_min = grp->skip_min; + + if (list_empty(&grp->tasks)) { + grp->skip_min = false; + goto out; + } + + if (!hmp_capable()) { + grp->skip_min = false; + goto out; + } + + wallclock = sched_ktime_clock(); + + /* + * wakeup of two or more related tasks could race with each other and + * could result in multiple calls to _set_preferred_cluster being issued + * at same time. Avoid overhead in such cases of rechecking preferred + * cluster + */ + if (wallclock - grp->last_update < sched_ravg_window / 10) + return; + + list_for_each_entry(p, &grp->tasks, wts.grp_list) { + if (task_boost_policy(p) == SCHED_BOOST_ON_BIG) { + group_boost = true; + break; + } + + if (p->wts.mark_start < wallclock - + (sched_ravg_window * sched_ravg_hist_size)) + continue; + + combined_demand += p->wts.coloc_demand; + if (!trace_sched_set_preferred_cluster_enabled()) { + if (combined_demand > sched_group_upmigrate) + break; + } + } + + grp->last_update = wallclock; + update_best_cluster(grp, combined_demand, group_boost); + trace_sched_set_preferred_cluster(grp, combined_demand); + +out: + if (grp->id == DEFAULT_CGROUP_COLOC_ID + && grp->skip_min != prev_skip_min) { + if (grp->skip_min) + grp->start_ts = sched_clock(); + sched_update_hyst_times(); + } +} + +void set_preferred_cluster(struct walt_related_thread_group *grp) +{ + raw_spin_lock(&grp->lock); + _set_preferred_cluster(grp); + raw_spin_unlock(&grp->lock); +} + +int update_preferred_cluster(struct walt_related_thread_group *grp, + struct task_struct *p, u32 old_load, bool from_tick) +{ + u32 new_load = task_load(p); + + if (!grp) + return 0; + + if (unlikely(from_tick && is_suh_max())) + return 1; + + /* + * Update if task's load has changed significantly or a complete window + * has passed since we last updated preference + */ + if (abs(new_load - old_load) > sched_ravg_window / 4 || + sched_ktime_clock() - grp->last_update > sched_ravg_window) + return 1; + + return 0; +} + +#define ADD_TASK 0 +#define REM_TASK 1 + +static inline struct walt_related_thread_group* +lookup_related_thread_group(unsigned int group_id) +{ + return related_thread_groups[group_id]; +} + +int alloc_related_thread_groups(void) +{ + int i, ret; + struct walt_related_thread_group *grp; + + /* groupd_id = 0 is invalid as it's special id to remove group. */ + for (i = 1; i < MAX_NUM_CGROUP_COLOC_ID; i++) { + grp = kzalloc(sizeof(*grp), GFP_NOWAIT); + if (!grp) { + ret = -ENOMEM; + goto err; + } + + grp->id = i; + INIT_LIST_HEAD(&grp->tasks); + INIT_LIST_HEAD(&grp->list); + raw_spin_lock_init(&grp->lock); + + related_thread_groups[i] = grp; + } + + return 0; + +err: + for (i = 1; i < MAX_NUM_CGROUP_COLOC_ID; i++) { + grp = lookup_related_thread_group(i); + if (grp) { + kfree(grp); + related_thread_groups[i] = NULL; + } else { + break; + } + } + + return ret; +} + +static void remove_task_from_group(struct task_struct *p) +{ + struct walt_related_thread_group *grp = p->wts.grp; + struct rq *rq; + int empty_group = 1; + struct rq_flags rf; + + raw_spin_lock(&grp->lock); + + rq = __task_rq_lock(p, &rf); + transfer_busy_time(rq, p->wts.grp, p, REM_TASK); + list_del_init(&p->wts.grp_list); + rcu_assign_pointer(p->wts.grp, NULL); + __task_rq_unlock(rq, &rf); + + + if (!list_empty(&grp->tasks)) { + empty_group = 0; + _set_preferred_cluster(grp); + } + + raw_spin_unlock(&grp->lock); + + /* Reserved groups cannot be destroyed */ + if (empty_group && grp->id != DEFAULT_CGROUP_COLOC_ID) + /* + * We test whether grp->list is attached with list_empty() + * hence re-init the list after deletion. + */ + list_del_init(&grp->list); +} + +static int +add_task_to_group(struct task_struct *p, struct walt_related_thread_group *grp) +{ + struct rq *rq; + struct rq_flags rf; + + raw_spin_lock(&grp->lock); + + /* + * Change p->wts.grp under rq->lock. Will prevent races with read-side + * reference of p->wts.grp in various hot-paths + */ + rq = __task_rq_lock(p, &rf); + transfer_busy_time(rq, grp, p, ADD_TASK); + list_add(&p->wts.grp_list, &grp->tasks); + rcu_assign_pointer(p->wts.grp, grp); + __task_rq_unlock(rq, &rf); + + _set_preferred_cluster(grp); + + raw_spin_unlock(&grp->lock); + + return 0; +} + +#ifdef CONFIG_UCLAMP_TASK_GROUP +static inline bool uclamp_task_colocated(struct task_struct *p) +{ + struct cgroup_subsys_state *css; + struct task_group *tg; + bool colocate; + + rcu_read_lock(); + css = task_css(p, cpu_cgrp_id); + if (!css) { + rcu_read_unlock(); + return false; + } + tg = container_of(css, struct task_group, css); + colocate = tg->wtg.colocate; + rcu_read_unlock(); + + return colocate; +} +#else +static inline bool uclamp_task_colocated(struct task_struct *p) +{ + return false; +} +#endif /* CONFIG_UCLAMP_TASK_GROUP */ + +void add_new_task_to_grp(struct task_struct *new) +{ + unsigned long flags; + struct walt_related_thread_group *grp; + + /* + * If the task does not belong to colocated schedtune + * cgroup, nothing to do. We are checking this without + * lock. Even if there is a race, it will be added + * to the co-located cgroup via cgroup attach. + */ + if (!uclamp_task_colocated(new)) + return; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + write_lock_irqsave(&related_thread_group_lock, flags); + + /* + * It's possible that someone already added the new task to the + * group. or it might have taken out from the colocated schedtune + * cgroup. check these conditions under lock. + */ + if (!uclamp_task_colocated(new) || new->wts.grp) { + write_unlock_irqrestore(&related_thread_group_lock, flags); + return; + } + + raw_spin_lock(&grp->lock); + + rcu_assign_pointer(new->wts.grp, grp); + list_add(&new->wts.grp_list, &grp->tasks); + + raw_spin_unlock(&grp->lock); + write_unlock_irqrestore(&related_thread_group_lock, flags); +} + +static int __sched_set_group_id(struct task_struct *p, unsigned int group_id) +{ + int rc = 0; + unsigned long flags; + struct walt_related_thread_group *grp = NULL; + + if (group_id >= MAX_NUM_CGROUP_COLOC_ID) + return -EINVAL; + + raw_spin_lock_irqsave(&p->pi_lock, flags); + write_lock(&related_thread_group_lock); + + /* Switching from one group to another directly is not permitted */ + if ((!p->wts.grp && !group_id) || (p->wts.grp && group_id)) + goto done; + + if (!group_id) { + remove_task_from_group(p); + goto done; + } + + grp = lookup_related_thread_group(group_id); + if (list_empty(&grp->list)) + list_add(&grp->list, &active_related_thread_groups); + + rc = add_task_to_group(p, grp); +done: + write_unlock(&related_thread_group_lock); + raw_spin_unlock_irqrestore(&p->pi_lock, flags); + + return rc; +} + +int sched_set_group_id(struct task_struct *p, unsigned int group_id) +{ + /* DEFAULT_CGROUP_COLOC_ID is a reserved id */ + if (group_id == DEFAULT_CGROUP_COLOC_ID) + return -EINVAL; + + return __sched_set_group_id(p, group_id); +} + +unsigned int sched_get_group_id(struct task_struct *p) +{ + unsigned int group_id; + struct walt_related_thread_group *grp; + + rcu_read_lock(); + grp = task_related_thread_group(p); + group_id = grp ? grp->id : 0; + rcu_read_unlock(); + + return group_id; +} + +#if defined(CONFIG_UCLAMP_TASK_GROUP) +/* + * We create a default colocation group at boot. There is no need to + * synchronize tasks between cgroups at creation time because the + * correct cgroup hierarchy is not available at boot. Therefore cgroup + * colocation is turned off by default even though the colocation group + * itself has been allocated. Furthermore this colocation group cannot + * be destroyted once it has been created. All of this has been as part + * of runtime optimizations. + * + * The job of synchronizing tasks to the colocation group is done when + * the colocation flag in the cgroup is turned on. + */ +static int __init create_default_coloc_group(void) +{ + struct walt_related_thread_group *grp = NULL; + unsigned long flags; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + write_lock_irqsave(&related_thread_group_lock, flags); + list_add(&grp->list, &active_related_thread_groups); + write_unlock_irqrestore(&related_thread_group_lock, flags); + + return 0; +} +late_initcall(create_default_coloc_group); + +int sync_cgroup_colocation(struct task_struct *p, bool insert) +{ + unsigned int grp_id = insert ? DEFAULT_CGROUP_COLOC_ID : 0; + + return __sched_set_group_id(p, grp_id); +} +#endif + +static bool is_cluster_hosting_top_app(struct walt_sched_cluster *cluster) +{ + struct walt_related_thread_group *grp; + bool grp_on_min; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + if (!grp) + return false; + + grp_on_min = !grp->skip_min && + (sched_boost_policy() != SCHED_BOOST_ON_BIG); + + return (is_min_capacity_cluster(cluster) == grp_on_min); +} + +static unsigned long thermal_cap_cpu[NR_CPUS]; + +unsigned long thermal_cap(int cpu) +{ + return thermal_cap_cpu[cpu] ?: SCHED_CAPACITY_SCALE; +} + +static inline unsigned long +do_thermal_cap(int cpu, unsigned long thermal_max_freq) +{ + if (unlikely(!walt_clusters_parsed)) + return capacity_orig_of(cpu); + + return mult_frac(arch_scale_cpu_capacity(cpu), thermal_max_freq, + cpu_max_possible_freq(cpu)); +} + +static DEFINE_SPINLOCK(cpu_freq_min_max_lock); +void sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, u32 fmax) +{ + struct cpumask cpumask; + int i; + unsigned long flags; + + spin_lock_irqsave(&cpu_freq_min_max_lock, flags); + cpumask_copy(&cpumask, cpus); + + for_each_cpu(i, &cpumask) + thermal_cap_cpu[i] = do_thermal_cap(i, fmax); + + spin_unlock_irqrestore(&cpu_freq_min_max_lock, flags); +} + +void note_task_waking(struct task_struct *p, u64 wallclock) +{ + p->wts.last_wake_ts = wallclock; +} + +/* + * Task's cpu usage is accounted in: + * rq->wrq.curr/prev_runnable_sum, when its ->grp is NULL + * grp->cpu_time[cpu]->curr/prev_runnable_sum, when its ->grp is !NULL + * + * Transfer task's cpu usage between those counters when transitioning between + * groups + */ +static void transfer_busy_time(struct rq *rq, + struct walt_related_thread_group *grp, + struct task_struct *p, int event) +{ + u64 wallclock; + struct group_cpu_time *cpu_time; + u64 *src_curr_runnable_sum, *dst_curr_runnable_sum; + u64 *src_prev_runnable_sum, *dst_prev_runnable_sum; + u64 *src_nt_curr_runnable_sum, *dst_nt_curr_runnable_sum; + u64 *src_nt_prev_runnable_sum, *dst_nt_prev_runnable_sum; + int migrate_type; + int cpu = cpu_of(rq); + bool new_task; + int i; + + wallclock = sched_ktime_clock(); + + walt_update_task_ravg(rq->curr, rq, TASK_UPDATE, wallclock, 0); + walt_update_task_ravg(p, rq, TASK_UPDATE, wallclock, 0); + new_task = is_new_task(p); + + cpu_time = &rq->wrq.grp_time; + if (event == ADD_TASK) { + migrate_type = RQ_TO_GROUP; + + src_curr_runnable_sum = &rq->wrq.curr_runnable_sum; + dst_curr_runnable_sum = &cpu_time->curr_runnable_sum; + src_prev_runnable_sum = &rq->wrq.prev_runnable_sum; + dst_prev_runnable_sum = &cpu_time->prev_runnable_sum; + + src_nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + dst_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + dst_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + if (*src_curr_runnable_sum < p->wts.curr_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_curr_runnable_sum, + p->wts.curr_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_curr_runnable_sum -= p->wts.curr_window_cpu[cpu]; + + if (*src_prev_runnable_sum < p->wts.prev_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_prev_runnable_sum, + p->wts.prev_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_prev_runnable_sum -= p->wts.prev_window_cpu[cpu]; + + if (new_task) { + if (*src_nt_curr_runnable_sum < + p->wts.curr_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_curr_runnable_sum, + p->wts.curr_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_curr_runnable_sum -= + p->wts.curr_window_cpu[cpu]; + + if (*src_nt_prev_runnable_sum < + p->wts.prev_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_prev_runnable_sum, + p->wts.prev_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_prev_runnable_sum -= + p->wts.prev_window_cpu[cpu]; + } + + update_cluster_load_subtractions(p, cpu, + rq->wrq.window_start, new_task); + + } else { + migrate_type = GROUP_TO_RQ; + + src_curr_runnable_sum = &cpu_time->curr_runnable_sum; + dst_curr_runnable_sum = &rq->wrq.curr_runnable_sum; + src_prev_runnable_sum = &cpu_time->prev_runnable_sum; + dst_prev_runnable_sum = &rq->wrq.prev_runnable_sum; + + src_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + dst_nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + dst_nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + + if (*src_curr_runnable_sum < p->wts.curr_window) { + printk_deferred("WALT-UG pid=%u CPU=%d event=%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_curr_runnable_sum, + p->wts.curr_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_curr_runnable_sum -= p->wts.curr_window; + + if (*src_prev_runnable_sum < p->wts.prev_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_prev_runnable_sum, + p->wts.prev_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_prev_runnable_sum -= p->wts.prev_window; + + if (new_task) { + if (*src_nt_curr_runnable_sum < p->wts.curr_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_curr_runnable_sum, + p->wts.curr_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_curr_runnable_sum -= p->wts.curr_window; + + if (*src_nt_prev_runnable_sum < p->wts.prev_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_prev_runnable_sum, + p->wts.prev_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_prev_runnable_sum -= p->wts.prev_window; + } + + /* + * Need to reset curr/prev windows for all CPUs, not just the + * ones in the same cluster. Since inter cluster migrations + * did not result in the appropriate book keeping, the values + * per CPU would be inaccurate. + */ + for_each_possible_cpu(i) { + p->wts.curr_window_cpu[i] = 0; + p->wts.prev_window_cpu[i] = 0; + } + } + + *dst_curr_runnable_sum += p->wts.curr_window; + *dst_prev_runnable_sum += p->wts.prev_window; + if (new_task) { + *dst_nt_curr_runnable_sum += p->wts.curr_window; + *dst_nt_prev_runnable_sum += p->wts.prev_window; + } + + /* + * When a task enter or exits a group, it's curr and prev windows are + * moved to a single CPU. This behavior might be sub-optimal in the + * exit case, however, it saves us the overhead of handling inter + * cluster migration fixups while the task is part of a related group. + */ + p->wts.curr_window_cpu[cpu] = p->wts.curr_window; + p->wts.prev_window_cpu[cpu] = p->wts.prev_window; + + trace_sched_migration_update_sum(p, migrate_type, rq); +} + +bool is_rtgb_active(void) +{ + struct walt_related_thread_group *grp; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + return grp && grp->skip_min; +} + +u64 get_rtgb_active_time(void) +{ + struct walt_related_thread_group *grp; + u64 now = sched_clock(); + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + if (grp && grp->skip_min && grp->start_ts) + return now - grp->start_ts; + + return 0; +} + +static void walt_init_window_dep(void); +static void walt_tunables_fixup(void) +{ + if (likely(num_sched_clusters > 0)) + walt_update_group_thresholds(); + walt_init_window_dep(); +} + +static void walt_update_irqload(struct rq *rq) +{ + u64 irq_delta = 0; + unsigned int nr_windows = 0; + u64 cur_irq_time; + u64 last_irq_window = READ_ONCE(rq->wrq.last_irq_window); + + if (rq->wrq.window_start > last_irq_window) + nr_windows = div64_u64(rq->wrq.window_start - last_irq_window, + sched_ravg_window); + + /* Decay CPU's irqload by 3/4 for each window. */ + if (nr_windows < 10) + rq->wrq.avg_irqload = mult_frac(rq->wrq.avg_irqload, 3, 4); + else + rq->wrq.avg_irqload = 0; + + cur_irq_time = irq_time_read(cpu_of(rq)); + if (cur_irq_time > rq->wrq.prev_irq_time) + irq_delta = cur_irq_time - rq->wrq.prev_irq_time; + + rq->wrq.avg_irqload += irq_delta; + rq->wrq.prev_irq_time = cur_irq_time; + + if (nr_windows < SCHED_HIGH_IRQ_TIMEOUT) + rq->wrq.high_irqload = (rq->wrq.avg_irqload >= + walt_cpu_high_irqload); + else + rq->wrq.high_irqload = 0; +} + +/* + * Runs in hard-irq context. This should ideally run just after the latest + * window roll-over. + */ +void walt_irq_work(struct irq_work *irq_work) +{ + struct walt_sched_cluster *cluster; + struct rq *rq; + int cpu; + u64 wc; + bool is_migration = false, is_asym_migration = false; + u64 total_grp_load = 0, min_cluster_grp_load = 0; + int level = 0; + u64 cur_jiffies_ts; + unsigned long flags; + + /* Am I the window rollover work or the migration work? */ + if (irq_work == &walt_migration_irq_work) + is_migration = true; + + for_each_cpu(cpu, cpu_possible_mask) { + if (level == 0) + raw_spin_lock(&cpu_rq(cpu)->lock); + else + raw_spin_lock_nested(&cpu_rq(cpu)->lock, level); + level++; + } + + wc = sched_ktime_clock(); + cur_jiffies_ts = get_jiffies_64(); + walt_load_reported_window = atomic64_read(&walt_irq_work_lastq_ws); + for_each_sched_cluster(cluster) { + u64 aggr_grp_load = 0; + + raw_spin_lock(&cluster->load_lock); + + for_each_cpu(cpu, &cluster->cpus) { + rq = cpu_rq(cpu); + if (rq->curr) { + walt_update_task_ravg(rq->curr, rq, + TASK_UPDATE, wc, 0); + account_load_subtractions(rq); + aggr_grp_load += + rq->wrq.grp_time.prev_runnable_sum; + } + if (is_migration && rq->wrq.notif_pending && + cpumask_test_cpu(cpu, &asym_cap_sibling_cpus)) { + is_asym_migration = true; + rq->wrq.notif_pending = false; + } + } + + cluster->aggr_grp_load = aggr_grp_load; + total_grp_load += aggr_grp_load; + + if (is_min_capacity_cluster(cluster)) + min_cluster_grp_load = aggr_grp_load; + raw_spin_unlock(&cluster->load_lock); + } + + if (total_grp_load) { + if (cpumask_weight(&asym_cap_sibling_cpus)) { + u64 big_grp_load = + total_grp_load - min_cluster_grp_load; + + for_each_cpu(cpu, &asym_cap_sibling_cpus) + cpu_cluster(cpu)->aggr_grp_load = big_grp_load; + } + rtgb_active = is_rtgb_active(); + } else { + rtgb_active = false; + } + + if (!is_migration && sysctl_sched_user_hint && time_after(jiffies, + sched_user_hint_reset_time)) + sysctl_sched_user_hint = 0; + + for_each_sched_cluster(cluster) { + cpumask_t cluster_online_cpus; + unsigned int num_cpus, i = 1; + + cpumask_and(&cluster_online_cpus, &cluster->cpus, + cpu_online_mask); + num_cpus = cpumask_weight(&cluster_online_cpus); + for_each_cpu(cpu, &cluster_online_cpus) { + int flag = SCHED_CPUFREQ_WALT; + + rq = cpu_rq(cpu); + + if (is_migration) { + if (rq->wrq.notif_pending) { + flag |= SCHED_CPUFREQ_INTERCLUSTER_MIG; + rq->wrq.notif_pending = false; + } + } + + if (is_asym_migration && cpumask_test_cpu(cpu, + &asym_cap_sibling_cpus)) + flag |= SCHED_CPUFREQ_INTERCLUSTER_MIG; + + if (i == num_cpus) + cpufreq_update_util(cpu_rq(cpu), flag); + else + cpufreq_update_util(cpu_rq(cpu), flag | + SCHED_CPUFREQ_CONTINUE); + i++; + + if (!is_migration) + walt_update_irqload(rq); + } + } + + /* + * If the window change request is in pending, good place to + * change sched_ravg_window since all rq locks are acquired. + * + * If the current window roll over is delayed such that the + * mark_start (current wallclock with which roll over is done) + * of the current task went past the window start with the + * updated new window size, delay the update to the next + * window roll over. Otherwise the CPU counters (prs and crs) are + * not rolled over properly as mark_start > window_start. + */ + if (!is_migration) { + spin_lock_irqsave(&sched_ravg_window_lock, flags); + + if ((sched_ravg_window != new_sched_ravg_window) && + (wc < this_rq()->wrq.window_start + new_sched_ravg_window)) { + sched_ravg_window_change_time = sched_ktime_clock(); + printk_deferred("ALERT: changing window size from %u to %u at %lu\n", + sched_ravg_window, + new_sched_ravg_window, + sched_ravg_window_change_time); + trace_sched_ravg_window_change(sched_ravg_window, + new_sched_ravg_window, + sched_ravg_window_change_time); + sched_ravg_window = new_sched_ravg_window; + walt_tunables_fixup(); + } + spin_unlock_irqrestore(&sched_ravg_window_lock, flags); + } + + for_each_cpu(cpu, cpu_possible_mask) + raw_spin_unlock(&cpu_rq(cpu)->lock); + + if (!is_migration) + core_ctl_check(this_rq()->wrq.window_start); +} + +void walt_rotation_checkpoint(int nr_big) +{ + if (!hmp_capable()) + return; + + if (!sysctl_sched_walt_rotate_big_tasks || sched_boost() != NO_BOOST) { + walt_rotation_enabled = 0; + return; + } + + walt_rotation_enabled = nr_big >= num_possible_cpus(); +} + +void walt_fill_ta_data(struct core_ctl_notif_data *data) +{ + struct walt_related_thread_group *grp; + unsigned long flags; + u64 total_demand = 0, wallclock; + struct task_struct *p; + int min_cap_cpu, scale = 1024; + struct walt_sched_cluster *cluster; + int i = 0; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + raw_spin_lock_irqsave(&grp->lock, flags); + if (list_empty(&grp->tasks)) { + raw_spin_unlock_irqrestore(&grp->lock, flags); + goto fill_util; + } + + wallclock = sched_ktime_clock(); + + list_for_each_entry(p, &grp->tasks, wts.grp_list) { + if (p->wts.mark_start < wallclock - + (sched_ravg_window * sched_ravg_hist_size)) + continue; + + total_demand += p->wts.coloc_demand; + } + + raw_spin_unlock_irqrestore(&grp->lock, flags); + + /* + * Scale the total demand to the lowest capacity CPU and + * convert into percentage. + * + * P = total_demand/sched_ravg_window * 1024/scale * 100 + */ + + min_cap_cpu = this_rq()->rd->wrd.min_cap_orig_cpu; + if (min_cap_cpu != -1) + scale = arch_scale_cpu_capacity(min_cap_cpu); + + data->coloc_load_pct = div64_u64(total_demand * 1024 * 100, + (u64)sched_ravg_window * scale); + +fill_util: + for_each_sched_cluster(cluster) { + int fcpu = cluster_first_cpu(cluster); + + if (i == MAX_CLUSTERS) + break; + + scale = arch_scale_cpu_capacity(fcpu); + data->ta_util_pct[i] = div64_u64(cluster->aggr_grp_load * 1024 * + 100, (u64)sched_ravg_window * scale); + + scale = arch_scale_freq_capacity(fcpu); + data->cur_cap_pct[i] = (scale * 100)/1024; + i++; + } +} + +int walt_proc_group_thresholds_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + static DEFINE_MUTEX(mutex); + struct rq *rq = cpu_rq(cpumask_first(cpu_possible_mask)); + unsigned long flags; + + if (unlikely(num_sched_clusters <= 0)) + return -EPERM; + + mutex_lock(&mutex); + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (ret || !write) { + mutex_unlock(&mutex); + return ret; + } + + /* + * The load scale factor update happens with all + * rqs locked. so acquiring 1 CPU rq lock and + * updating the thresholds is sufficient for + * an atomic update. + */ + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_group_thresholds(); + raw_spin_unlock_irqrestore(&rq->lock, flags); + + mutex_unlock(&mutex); + + return ret; +} + +static void walt_init_window_dep(void) +{ + walt_cpu_util_freq_divisor = + (sched_ravg_window >> SCHED_CAPACITY_SHIFT) * 100; + walt_scale_demand_divisor = sched_ravg_window >> SCHED_CAPACITY_SHIFT; + + sched_init_task_load_windows = + div64_u64((u64)sysctl_sched_init_task_load_pct * + (u64)sched_ravg_window, 100); + sched_init_task_load_windows_scaled = + scale_demand(sched_init_task_load_windows); + + walt_cpu_high_irqload = div64_u64((u64)sched_ravg_window * 95, (u64) 100); +} + +static void walt_init_once(void) +{ + init_irq_work(&walt_migration_irq_work, walt_irq_work); + init_irq_work(&walt_cpufreq_irq_work, walt_irq_work); + walt_rotate_work_init(); + walt_init_window_dep(); +} + +void walt_sched_init_rq(struct rq *rq) +{ + int j; + + if (cpu_of(rq) == 0) + walt_init_once(); + + cpumask_set_cpu(cpu_of(rq), &rq->wrq.freq_domain_cpumask); + + rq->wrq.walt_stats.cumulative_runnable_avg_scaled = 0; + rq->wrq.prev_window_size = sched_ravg_window; + rq->wrq.window_start = 0; + rq->wrq.walt_stats.nr_big_tasks = 0; + rq->wrq.walt_flags = 0; + rq->wrq.avg_irqload = 0; + rq->wrq.prev_irq_time = 0; + rq->wrq.last_irq_window = 0; + rq->wrq.high_irqload = false; + rq->wrq.task_exec_scale = 1024; + rq->wrq.push_task = NULL; + + /* + * All cpus part of same cluster by default. This avoids the + * need to check for rq->wrq.cluster being non-NULL in hot-paths + * like select_best_cpu() + */ + rq->wrq.cluster = &init_cluster; + rq->wrq.curr_runnable_sum = rq->wrq.prev_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = rq->wrq.nt_prev_runnable_sum = 0; + memset(&rq->wrq.grp_time, 0, sizeof(struct group_cpu_time)); + rq->wrq.old_busy_time = 0; + rq->wrq.old_estimated_time = 0; + rq->wrq.walt_stats.pred_demands_sum_scaled = 0; + rq->wrq.walt_stats.nr_rtg_high_prio_tasks = 0; + rq->wrq.ed_task = NULL; + rq->wrq.curr_table = 0; + rq->wrq.prev_top = 0; + rq->wrq.curr_top = 0; + rq->wrq.last_cc_update = 0; + rq->wrq.cycles = 0; + for (j = 0; j < NUM_TRACKED_WINDOWS; j++) { + memset(&rq->wrq.load_subs[j], 0, + sizeof(struct load_subtractions)); + rq->wrq.top_tasks[j] = kcalloc(NUM_LOAD_INDICES, + sizeof(u8), GFP_NOWAIT); + /* No other choice */ + BUG_ON(!rq->wrq.top_tasks[j]); + clear_top_tasks_bitmap(rq->wrq.top_tasks_bitmap[j]); + } + rq->wrq.cum_window_demand_scaled = 0; + rq->wrq.notif_pending = false; +} + +int walt_proc_user_hint_handler(struct ctl_table *table, + int write, void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + unsigned int old_value; + static DEFINE_MUTEX(mutex); + + mutex_lock(&mutex); + + sched_user_hint_reset_time = jiffies + HZ; + old_value = sysctl_sched_user_hint; + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (ret || !write || (old_value == sysctl_sched_user_hint)) + goto unlock; + + walt_irq_work_queue(&walt_migration_irq_work); + +unlock: + mutex_unlock(&mutex); + return ret; +} + +static inline void sched_window_nr_ticks_change(void) +{ + unsigned long flags; + + spin_lock_irqsave(&sched_ravg_window_lock, flags); + new_sched_ravg_window = mult_frac(sysctl_sched_ravg_window_nr_ticks, + NSEC_PER_SEC, HZ); + spin_unlock_irqrestore(&sched_ravg_window_lock, flags); +} + +int sched_ravg_window_handler(struct ctl_table *table, + int write, void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret = -EPERM; + static DEFINE_MUTEX(mutex); + unsigned int prev_value; + + mutex_lock(&mutex); + + if (write && (HZ != 250 || !sysctl_sched_dynamic_ravg_window_enable)) + goto unlock; + + prev_value = sysctl_sched_ravg_window_nr_ticks; + ret = proc_douintvec_ravg_window(table, write, buffer, lenp, ppos); + if (ret || !write || (prev_value == sysctl_sched_ravg_window_nr_ticks)) + goto unlock; + + sched_window_nr_ticks_change(); + +unlock: + mutex_unlock(&mutex); + return ret; +} diff --git a/kernel/sched/walt.h b/kernel/sched/walt/walt.h similarity index 99% rename from kernel/sched/walt.h rename to kernel/sched/walt/walt.h index e20992b70532..b5985cb63f27 100644 --- a/kernel/sched/walt.h +++ b/kernel/sched/walt/walt.h @@ -1,6 +1,6 @@ /* SPDX-License-Identifier: GPL-2.0-only */ /* - * Copyright (c) 2016-2020, The Linux Foundation. All rights reserved. + * Copyright (c) 2016-2021, The Linux Foundation. All rights reserved. */ #ifndef __WALT_H