From 5b3db45955726387e258579a9b188a5c9cb99ca5 Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Thu, 24 Sep 2020 11:19:20 +0530 Subject: [PATCH] sched/walt: Move scheduler techpack to kernel The scheduler techpack sources are moved to kernel to ease the development. The weak symbol definitions for WALT functions are no longer required to cover the case of compiling kernel without syncing the scheduler techpack. So remove all the weak symbol references. Change-Id: Ief85bccd3dceaf60dda44aef9893b4138dc63380 Signed-off-by: Pavankumar Kondeti --- include/linux/sched/sysctl.h | 48 +- kernel/sched/Makefile | 2 +- kernel/sched/core.c | 2 +- kernel/sched/cputime.c | 2 +- kernel/sched/deadline.c | 2 +- kernel/sched/fair.c | 2 +- kernel/sched/rt.c | 2 +- kernel/sched/stop_task.c | 2 +- kernel/sched/walt.c | 218 -- kernel/sched/walt/Makefile | 3 + kernel/sched/walt/boost.c | 318 +++ kernel/sched/walt/core_ctl.c | 1354 ++++++++++++ kernel/sched/walt/cpu-boost.c | 389 ++++ kernel/sched/walt/qc_vas.c | 744 +++++++ kernel/sched/walt/qc_vas.h | 81 + kernel/sched/walt/sched_avg.c | 260 +++ kernel/sched/walt/trace.c | 82 + kernel/sched/walt/trace.h | 669 ++++++ kernel/sched/walt/walt.c | 3793 ++++++++++++++++++++++++++++++++ kernel/sched/{ => walt}/walt.h | 2 +- 20 files changed, 7725 insertions(+), 250 deletions(-) delete mode 100644 kernel/sched/walt.c create mode 100644 kernel/sched/walt/Makefile create mode 100644 kernel/sched/walt/boost.c create mode 100644 kernel/sched/walt/core_ctl.c create mode 100644 kernel/sched/walt/cpu-boost.c create mode 100644 kernel/sched/walt/qc_vas.c create mode 100644 kernel/sched/walt/qc_vas.h create mode 100644 kernel/sched/walt/sched_avg.c create mode 100644 kernel/sched/walt/trace.c create mode 100644 kernel/sched/walt/trace.h create mode 100644 kernel/sched/walt/walt.c rename kernel/sched/{ => walt}/walt.h (99%) diff --git a/include/linux/sched/sysctl.h b/include/linux/sched/sysctl.h index 3d9b702f3f96..0e83233722d7 100644 --- a/include/linux/sched/sysctl.h +++ b/include/linux/sched/sysctl.h @@ -34,30 +34,30 @@ extern unsigned int sysctl_sched_force_lb_enable; extern unsigned int sysctl_hh_suspend_timeout_ms; #endif #ifdef CONFIG_SCHED_WALT -extern unsigned int __weak sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS]; -extern unsigned int __weak sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS]; -extern unsigned int __weak sysctl_sched_user_hint; -extern const int __weak sched_user_hint_max; -extern unsigned int __weak sysctl_sched_boost; -extern unsigned int __weak sysctl_sched_group_upmigrate_pct; -extern unsigned int __weak sysctl_sched_group_downmigrate_pct; -extern unsigned int __weak sysctl_sched_conservative_pl; -extern unsigned int __weak sysctl_sched_walt_rotate_big_tasks; -extern unsigned int __weak sysctl_sched_min_task_util_for_boost; -extern unsigned int __weak sysctl_sched_min_task_util_for_colocation; -extern unsigned int __weak sysctl_sched_asym_cap_sibling_freq_match_pct; -extern unsigned int __weak sysctl_sched_coloc_downmigrate_ns; -extern unsigned int __weak sysctl_sched_task_unfilter_period; -extern unsigned int __weak sysctl_sched_busy_hyst_enable_cpus; -extern unsigned int __weak sysctl_sched_busy_hyst; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_enable_cpus; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS]; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_max_ms; -extern unsigned int __weak sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS]; -extern unsigned int __weak sysctl_sched_window_stats_policy; -extern unsigned int __weak sysctl_sched_ravg_window_nr_ticks; -extern unsigned int __weak sysctl_sched_many_wakeup_threshold; -extern unsigned int __weak sysctl_sched_dynamic_ravg_window_enable; +extern unsigned int sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS]; +extern unsigned int sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS]; +extern unsigned int sysctl_sched_user_hint; +extern const int sched_user_hint_max; +extern unsigned int sysctl_sched_boost; +extern unsigned int sysctl_sched_group_upmigrate_pct; +extern unsigned int sysctl_sched_group_downmigrate_pct; +extern unsigned int sysctl_sched_conservative_pl; +extern unsigned int sysctl_sched_walt_rotate_big_tasks; +extern unsigned int sysctl_sched_min_task_util_for_boost; +extern unsigned int sysctl_sched_min_task_util_for_colocation; +extern unsigned int sysctl_sched_asym_cap_sibling_freq_match_pct; +extern unsigned int sysctl_sched_coloc_downmigrate_ns; +extern unsigned int sysctl_sched_task_unfilter_period; +extern unsigned int sysctl_sched_busy_hyst_enable_cpus; +extern unsigned int sysctl_sched_busy_hyst; +extern unsigned int sysctl_sched_coloc_busy_hyst_enable_cpus; +extern unsigned int sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS]; +extern unsigned int sysctl_sched_coloc_busy_hyst_max_ms; +extern unsigned int sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS]; +extern unsigned int sysctl_sched_window_stats_policy; +extern unsigned int sysctl_sched_ravg_window_nr_ticks; +extern unsigned int sysctl_sched_many_wakeup_threshold; +extern unsigned int sysctl_sched_dynamic_ravg_window_enable; extern unsigned int sysctl_sched_prefer_spread; extern unsigned int sysctl_walt_rtg_cfs_boost_prio; extern unsigned int sysctl_walt_low_latency_task_threshold; diff --git a/kernel/sched/Makefile b/kernel/sched/Makefile index 1175ed6ccf4d..dad16b665f11 100644 --- a/kernel/sched/Makefile +++ b/kernel/sched/Makefile @@ -20,7 +20,7 @@ obj-y += core.o loadavg.o clock.o cputime.o obj-y += idle.o fair.o rt.o deadline.o obj-y += wait.o wait_bit.o swait.o completion.o -obj-$(CONFIG_SCHED_WALT) += walt.o +obj-$(CONFIG_SCHED_WALT) += walt/ obj-$(CONFIG_SMP) += cpupri.o cpudeadline.o topology.o stop_task.o pelt.o obj-$(CONFIG_SCHED_AUTOGROUP) += autogroup.o obj-$(CONFIG_SCHEDSTATS) += stats.o diff --git a/kernel/sched/core.c b/kernel/sched/core.c index c5b5ec2ef436..e9339790ab85 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -21,7 +21,7 @@ #include "../smpboot.h" #include "pelt.h" -#include "walt.h" +#include "walt/walt.h" #define CREATE_TRACE_POINTS #include diff --git a/kernel/sched/cputime.c b/kernel/sched/cputime.c index 8ea9e07784da..55dff2d61356 100644 --- a/kernel/sched/cputime.c +++ b/kernel/sched/cputime.c @@ -4,7 +4,7 @@ */ #include #include "sched.h" -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_IRQ_TIME_ACCOUNTING diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 58e0eeef0090..650d8b0ba510 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -17,7 +17,7 @@ */ #include "sched.h" #include "pelt.h" -#include "walt.h" +#include "walt/walt.h" struct dl_bandwidth def_dl_bandwidth; diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 43ddd00efe39..c81e1a5d64b2 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -25,7 +25,7 @@ #include #include -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_SMP static inline bool task_fits_max(struct task_struct *p, int cpu); diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 2c83330f9ea0..e8faf503a343 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -11,7 +11,7 @@ #include -#include "walt.h" +#include "walt/walt.h" #include diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index 6a0937d4884f..6d360473ab67 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -8,7 +8,7 @@ * See kernel/stop_machine.c */ #include "sched.h" -#include "walt.h" +#include "walt/walt.h" #ifdef CONFIG_SMP static int diff --git a/kernel/sched/walt.c b/kernel/sched/walt.c deleted file mode 100644 index 1721a470c2f1..000000000000 --- a/kernel/sched/walt.c +++ /dev/null @@ -1,218 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -/* - * Copyright (c) 2016-2020, The Linux Foundation. All rights reserved. - */ - -#include "sched.h" -#include "walt.h" - -int __weak sched_wake_up_idle_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak sched_wake_up_idle_write(struct file *file, - const char __user *buf, size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_wake_up_idle_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_init_task_load_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak -sched_init_task_load_write(struct file *file, const char __user *buf, - size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_init_task_load_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_group_id_show(struct seq_file *m, void *v) -{ - return -EPERM; -} - -ssize_t __weak sched_group_id_write(struct file *file, const char __user *buf, - size_t count, loff_t *offset) -{ - return -EPERM; -} - -int __weak sched_group_id_open(struct inode *inode, struct file *filp) -{ - return -EPERM; -} - -int __weak sched_isolate_cpu(int cpu) { return 0; } - -int __weak sched_unisolate_cpu(int cpu) { return 0; } - -int __weak sched_unisolate_cpu_unlocked(int cpu) { return 0; } - -int __weak register_cpu_cycle_counter_cb(struct cpu_cycle_counter_cb *cb) -{ - return 0; -} - -void __weak sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, - u32 fmax) { } - -void __weak free_task_load_ptrs(struct task_struct *p) { } - -int __weak core_ctl_set_boost(bool boost) { return 0; } - -void __weak core_ctl_notifier_register(struct notifier_block *n) { } - -void __weak core_ctl_notifier_unregister(struct notifier_block *n) { } - -void __weak sched_update_nr_prod(int cpu, long delta, bool inc) { } - -unsigned int __weak sched_get_cpu_util(int cpu) { return 0; } - -void __weak sched_update_hyst_times(void) { } - -u64 __weak sched_lpm_disallowed_time(int cpu) { return 0; } - -int __weak -walt_proc_group_thresholds_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -walt_proc_user_hint_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -sched_updown_migrate_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak -sched_ravg_window_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak sched_boost_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -int __weak sched_busy_hyst_handler(struct ctl_table *table, int write, - void __user *buffer, size_t *lenp, loff_t *ppos) -{ - return -ENOSYS; -} - -u64 __weak sched_ktime_clock(void) { return 0; } - -unsigned long __weak -cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) -{ - return cpu_util(cpu); -} - -int __weak update_preferred_cluster(struct walt_related_thread_group *grp, - struct task_struct *p, u32 old_load, bool from_tick) -{ - return 0; -} - -void __weak set_preferred_cluster(struct walt_related_thread_group *grp) { } - -void __weak add_new_task_to_grp(struct task_struct *new) { } - -int __weak -preferred_cluster(struct walt_sched_cluster *cluster, struct task_struct *p) -{ - return -1; -} - -int __weak sync_cgroup_colocation(struct task_struct *p, bool insert) -{ - return 0; -} - -int __weak alloc_related_thread_groups(void) { return 0; } - -void __weak check_for_migration(struct rq *rq, struct task_struct *p) { } - -unsigned long __weak thermal_cap(int cpu) -{ - return cpu_rq(cpu)->cpu_capacity_orig; -} - -void __weak clear_walt_request(int cpu) { } - -void __weak clear_ed_task(struct task_struct *p, struct rq *rq) { } - -bool __weak early_detection_notify(struct rq *rq, u64 wallclock) -{ - return 0; -} - -void __weak note_task_waking(struct task_struct *p, u64 wallclock) { } - -int __weak group_balance_cpu_not_isolated(struct sched_group *sg) -{ - return group_balance_cpu(sg); -} - -void __weak detach_one_task_core(struct task_struct *p, struct rq *rq, - struct list_head *tasks) { } - -void __weak attach_tasks_core(struct list_head *tasks, struct rq *rq) { } - -void __weak walt_update_task_ravg(struct task_struct *p, struct rq *rq, - int event, u64 wallclock, u64 irqtime) { } - -void __weak fixup_busy_time(struct task_struct *p, int new_cpu) { } - -void __weak init_new_task_load(struct task_struct *p) { } - -void __weak mark_task_starting(struct task_struct *p) { } - -void __weak set_window_start(struct rq *rq) { } - -bool __weak do_pl_notif(struct rq *rq) { return false; } - -void __weak walt_sched_account_irqstart(int cpu, struct task_struct *curr) { } -void __weak walt_sched_account_irqend(int cpu, struct task_struct *curr, - u64 delta) -{ -} - -void __weak update_cluster_topology(void) { } - -void __weak init_clusters(void) { } - -void __weak walt_sched_init_rq(struct rq *rq) { } - -void __weak walt_update_cluster_topology(void) { } - -void __weak walt_task_dead(struct task_struct *p) { } - -#if defined(CONFIG_UCLAMP_TASK_GROUP) -void __weak walt_init_sched_boost(struct task_group *tg) { } -#endif diff --git a/kernel/sched/walt/Makefile b/kernel/sched/walt/Makefile new file mode 100644 index 000000000000..8bf42cb669af --- /dev/null +++ b/kernel/sched/walt/Makefile @@ -0,0 +1,3 @@ +# SPDX-License-Identifier: GPL-2.0 +obj-$(CONFIG_SCHED_WALT) += walt.o boost.o sched_avg.o qc_vas.o core_ctl.o trace.o +obj-$(CONFIG_CPU_FREQ) += cpu-boost.o diff --git a/kernel/sched/walt/boost.c b/kernel/sched/walt/boost.c new file mode 100644 index 000000000000..c37185e293d0 --- /dev/null +++ b/kernel/sched/walt/boost.c @@ -0,0 +1,318 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2012-2021, The Linux Foundation. All rights reserved. + */ +#include "qc_vas.h" +#include +#include +#include + +/* + * Scheduler boost is a mechanism to temporarily place tasks on CPUs + * with higher capacity than those where a task would have normally + * ended up with their load characteristics. Any entity enabling + * boost is responsible for disabling it as well. + */ + +unsigned int sysctl_sched_boost; /* To/from userspace */ +unsigned int sched_boost_type; /* currently activated sched boost */ +enum sched_boost_policy boost_policy; + +static enum sched_boost_policy boost_policy_dt = SCHED_BOOST_NONE; +static DEFINE_MUTEX(boost_mutex); + +#if defined(CONFIG_UCLAMP_TASK_GROUP) +void walt_init_sched_boost(struct task_group *tg) +{ + tg->wtg.sched_boost_no_override = false; + tg->wtg.sched_boost_enabled = true; + tg->wtg.colocate = false; + tg->wtg.colocate_update_disabled = false; +} + +static void update_cgroup_boost_settings(void) +{ + struct task_group *tg; + + rcu_read_lock(); + list_for_each_entry_rcu(tg, &task_groups, list) { + if (tg->wtg.sched_boost_no_override) + continue; + + tg->wtg.sched_boost_enabled = false; + } + rcu_read_unlock(); +} + +static void restore_cgroup_boost_settings(void) +{ + struct task_group *tg; + + rcu_read_lock(); + list_for_each_entry_rcu(tg, &task_groups, list) + tg->wtg.sched_boost_enabled = true; + rcu_read_unlock(); +} + +#else +static void update_cgroup_boost_settings(void) { } +static void restore_cgroup_boost_settings(void) { } +#endif + +/* + * Scheduler boost type and boost policy might at first seem unrelated, + * however, there exists a connection between them that will allow us + * to use them interchangeably during placement decisions. We'll explain + * the connection here in one possible way so that the implications are + * clear when looking at placement policies. + * + * When policy = SCHED_BOOST_NONE, type is either none or RESTRAINED + * When policy = SCHED_BOOST_ON_ALL or SCHED_BOOST_ON_BIG, type can + * neither be none nor RESTRAINED. + */ +static void set_boost_policy(int type) +{ + if (type == NO_BOOST || type == RESTRAINED_BOOST) { + boost_policy = SCHED_BOOST_NONE; + return; + } + + if (boost_policy_dt) { + boost_policy = boost_policy_dt; + return; + } + + if (hmp_capable()) { + boost_policy = SCHED_BOOST_ON_BIG; + return; + } + + boost_policy = SCHED_BOOST_ON_ALL; +} + +static bool verify_boost_params(int type) +{ + return type >= RESTRAINED_BOOST_DISABLE && type <= RESTRAINED_BOOST; +} + +static void sched_no_boost_nop(void) +{ +} + +static void sched_full_throttle_boost_enter(void) +{ + core_ctl_set_boost(true); + walt_enable_frequency_aggregation(true); +} + +static void sched_full_throttle_boost_exit(void) +{ + core_ctl_set_boost(false); + walt_enable_frequency_aggregation(false); +} + +static void sched_conservative_boost_enter(void) +{ + update_cgroup_boost_settings(); +} + +static void sched_conservative_boost_exit(void) +{ + restore_cgroup_boost_settings(); +} + +static void sched_restrained_boost_enter(void) +{ + walt_enable_frequency_aggregation(true); +} + +static void sched_restrained_boost_exit(void) +{ + walt_enable_frequency_aggregation(false); +} + +struct sched_boost_data { + int refcount; + void (*enter)(void); + void (*exit)(void); +}; + +static struct sched_boost_data sched_boosts[] = { + [NO_BOOST] = { + .refcount = 0, + .enter = sched_no_boost_nop, + .exit = sched_no_boost_nop, + }, + [FULL_THROTTLE_BOOST] = { + .refcount = 0, + .enter = sched_full_throttle_boost_enter, + .exit = sched_full_throttle_boost_exit, + }, + [CONSERVATIVE_BOOST] = { + .refcount = 0, + .enter = sched_conservative_boost_enter, + .exit = sched_conservative_boost_exit, + }, + [RESTRAINED_BOOST] = { + .refcount = 0, + .enter = sched_restrained_boost_enter, + .exit = sched_restrained_boost_exit, + }, +}; + +#define SCHED_BOOST_START FULL_THROTTLE_BOOST +#define SCHED_BOOST_END (RESTRAINED_BOOST + 1) + +static int sched_effective_boost(void) +{ + int i; + + /* + * The boosts are sorted in descending order by + * priority. + */ + for (i = SCHED_BOOST_START; i < SCHED_BOOST_END; i++) { + if (sched_boosts[i].refcount >= 1) + return i; + } + + return NO_BOOST; +} + +static void sched_boost_disable(int type) +{ + struct sched_boost_data *sb = &sched_boosts[type]; + int next_boost; + + if (sb->refcount <= 0) + return; + + sb->refcount--; + + if (sb->refcount) + return; + + /* + * This boost's refcount becomes zero, so it must + * be disabled. Disable it first and then apply + * the next boost. + */ + sb->exit(); + + next_boost = sched_effective_boost(); + sched_boosts[next_boost].enter(); +} + +static void sched_boost_enable(int type) +{ + struct sched_boost_data *sb = &sched_boosts[type]; + int next_boost, prev_boost = sched_boost_type; + + sb->refcount++; + + if (sb->refcount != 1) + return; + + /* + * This boost enable request did not come before. + * Take this new request and find the next boost + * by aggregating all the enabled boosts. If there + * is a change, disable the previous boost and enable + * the next boost. + */ + + next_boost = sched_effective_boost(); + if (next_boost == prev_boost) + return; + + sched_boosts[prev_boost].exit(); + sched_boosts[next_boost].enter(); +} + +static void sched_boost_disable_all(void) +{ + int i; + + for (i = SCHED_BOOST_START; i < SCHED_BOOST_END; i++) { + if (sched_boosts[i].refcount > 0) { + sched_boosts[i].exit(); + sched_boosts[i].refcount = 0; + } + } +} + +static void _sched_set_boost(int type) +{ + if (type == 0) + sched_boost_disable_all(); + else if (type > 0) + sched_boost_enable(type); + else + sched_boost_disable(-type); + + /* + * sysctl_sched_boost holds the boost request from + * user space which could be different from the + * effectively enabled boost. Update the effective + * boost here. + */ + + sched_boost_type = sched_effective_boost(); + sysctl_sched_boost = sched_boost_type; + set_boost_policy(sysctl_sched_boost); + trace_sched_set_boost(sysctl_sched_boost); +} + +void sched_boost_parse_dt(void) +{ + struct device_node *sn; + const char *boost_policy; + + sn = of_find_node_by_path("/sched-hmp"); + if (!sn) + return; + + if (!of_property_read_string(sn, "boost-policy", &boost_policy)) { + if (!strcmp(boost_policy, "boost-on-big")) + boost_policy_dt = SCHED_BOOST_ON_BIG; + else if (!strcmp(boost_policy, "boost-on-all")) + boost_policy_dt = SCHED_BOOST_ON_ALL; + } +} + +int sched_set_boost(int type) +{ + int ret = 0; + + mutex_lock(&boost_mutex); + if (verify_boost_params(type)) + _sched_set_boost(type); + else + ret = -EINVAL; + mutex_unlock(&boost_mutex); + return ret; +} + +int sched_boost_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + unsigned int *data = (unsigned int *)table->data; + + mutex_lock(&boost_mutex); + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + + if (ret || !write) + goto done; + + if (verify_boost_params(*data)) + _sched_set_boost(*data); + else + ret = -EINVAL; + +done: + mutex_unlock(&boost_mutex); + return ret; +} diff --git a/kernel/sched/walt/core_ctl.c b/kernel/sched/walt/core_ctl.c new file mode 100644 index 000000000000..1413b399c98b --- /dev/null +++ b/kernel/sched/walt/core_ctl.c @@ -0,0 +1,1354 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2014-2021, The Linux Foundation. All rights reserved. + */ +#define pr_fmt(fmt) "core_ctl: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include "qc_vas.h" + +struct cluster_data { + bool inited; + unsigned int min_cpus; + unsigned int max_cpus; + unsigned int offline_delay_ms; + unsigned int busy_up_thres[MAX_CPUS_PER_CLUSTER]; + unsigned int busy_down_thres[MAX_CPUS_PER_CLUSTER]; + unsigned int active_cpus; + unsigned int num_cpus; + unsigned int nr_isolated_cpus; + unsigned int nr_not_preferred_cpus; + cpumask_t cpu_mask; + unsigned int need_cpus; + unsigned int task_thres; + unsigned int max_nr; + unsigned int nr_prev_assist; + unsigned int nr_prev_assist_thresh; + s64 need_ts; + struct list_head lru; + bool pending; + spinlock_t pending_lock; + bool enable; + int nrrun; + struct task_struct *core_ctl_thread; + unsigned int first_cpu; + unsigned int boost; + struct kobject kobj; + unsigned int strict_nrrun; +}; + +struct cpu_data { + bool is_busy; + unsigned int busy; + unsigned int cpu; + bool not_preferred; + struct cluster_data *cluster; + struct list_head sib; + bool isolated_by_us; +}; + +static DEFINE_PER_CPU(struct cpu_data, cpu_state); +static struct cluster_data cluster_state[MAX_CLUSTERS]; +static unsigned int num_clusters; + +#define for_each_cluster(cluster, idx) \ + for (; (idx) < num_clusters && ((cluster) = &cluster_state[idx]);\ + idx++) + + +static DEFINE_SPINLOCK(state_lock); +static void apply_need(struct cluster_data *state); +static void wake_up_core_ctl_thread(struct cluster_data *state); +static bool initialized; + +ATOMIC_NOTIFIER_HEAD(core_ctl_notifier); +static unsigned int last_nr_big; + +static unsigned int get_active_cpu_count(const struct cluster_data *cluster); + +/* ========================= sysfs interface =========================== */ + +static ssize_t store_min_cpus(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->min_cpus = min(val, state->num_cpus); + wake_up_core_ctl_thread(state); + + return count; +} + +static ssize_t show_min_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->min_cpus); +} + +static ssize_t store_max_cpus(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->max_cpus = min(val, state->num_cpus); + wake_up_core_ctl_thread(state); + + return count; +} + +static ssize_t show_max_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->max_cpus); +} + +static ssize_t store_offline_delay_ms(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->offline_delay_ms = val; + apply_need(state); + + return count; +} + +static ssize_t show_task_thres(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->task_thres); +} + +static ssize_t store_task_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + if (val < state->num_cpus) + return -EINVAL; + + state->task_thres = val; + apply_need(state); + + return count; +} + +static ssize_t show_nr_prev_assist_thresh(const struct cluster_data *state, + char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->nr_prev_assist_thresh); +} + +static ssize_t store_nr_prev_assist_thresh(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + state->nr_prev_assist_thresh = val; + apply_need(state); + + return count; +} + +static ssize_t show_offline_delay_ms(const struct cluster_data *state, + char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->offline_delay_ms); +} + +static ssize_t store_busy_up_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val[MAX_CPUS_PER_CLUSTER]; + int ret, i; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != 1 && ret != state->num_cpus) + return -EINVAL; + + if (ret == 1) { + for (i = 0; i < state->num_cpus; i++) + state->busy_up_thres[i] = val[0]; + } else { + for (i = 0; i < state->num_cpus; i++) + state->busy_up_thres[i] = val[i]; + } + apply_need(state); + return count; +} + +static ssize_t show_busy_up_thres(const struct cluster_data *state, char *buf) +{ + int i, count = 0; + + for (i = 0; i < state->num_cpus; i++) + count += snprintf(buf + count, PAGE_SIZE - count, "%u ", + state->busy_up_thres[i]); + + count += snprintf(buf + count, PAGE_SIZE - count, "\n"); + return count; +} + +static ssize_t store_busy_down_thres(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val[MAX_CPUS_PER_CLUSTER]; + int ret, i; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != 1 && ret != state->num_cpus) + return -EINVAL; + + if (ret == 1) { + for (i = 0; i < state->num_cpus; i++) + state->busy_down_thres[i] = val[0]; + } else { + for (i = 0; i < state->num_cpus; i++) + state->busy_down_thres[i] = val[i]; + } + apply_need(state); + return count; +} + +static ssize_t show_busy_down_thres(const struct cluster_data *state, char *buf) +{ + int i, count = 0; + + for (i = 0; i < state->num_cpus; i++) + count += snprintf(buf + count, PAGE_SIZE - count, "%u ", + state->busy_down_thres[i]); + + count += snprintf(buf + count, PAGE_SIZE - count, "\n"); + return count; +} + +static ssize_t store_enable(struct cluster_data *state, + const char *buf, size_t count) +{ + unsigned int val; + bool bval; + + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + + bval = !!val; + if (bval != state->enable) { + state->enable = bval; + apply_need(state); + } + + return count; +} + +static ssize_t show_enable(const struct cluster_data *state, char *buf) +{ + return scnprintf(buf, PAGE_SIZE, "%u\n", state->enable); +} + +static ssize_t show_need_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->need_cpus); +} + +static ssize_t show_active_cpus(const struct cluster_data *state, char *buf) +{ + return snprintf(buf, PAGE_SIZE, "%u\n", state->active_cpus); +} + +static ssize_t show_global_state(const struct cluster_data *state, char *buf) +{ + struct cpu_data *c; + struct cluster_data *cluster; + ssize_t count = 0; + unsigned int cpu; + + spin_lock_irq(&state_lock); + for_each_possible_cpu(cpu) { + c = &per_cpu(cpu_state, cpu); + cluster = c->cluster; + if (!cluster || !cluster->inited) + continue; + + count += snprintf(buf + count, PAGE_SIZE - count, + "CPU%u\n", cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tCPU: %u\n", c->cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tOnline: %u\n", + cpu_online(c->cpu)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tIsolated: %u\n", + cpu_isolated(c->cpu)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tFirst CPU: %u\n", + cluster->first_cpu); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tBusy%%: %u\n", c->busy); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tIs busy: %u\n", c->is_busy); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNot preferred: %u\n", + c->not_preferred); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNr running: %u\n", cluster->nrrun); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tActive CPUs: %u\n", get_active_cpu_count(cluster)); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNeed CPUs: %u\n", cluster->need_cpus); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tNr isolated CPUs: %u\n", + cluster->nr_isolated_cpus); + count += snprintf(buf + count, PAGE_SIZE - count, + "\tBoost: %u\n", (unsigned int) cluster->boost); + } + spin_unlock_irq(&state_lock); + + return count; +} + +static ssize_t store_not_preferred(struct cluster_data *state, + const char *buf, size_t count) +{ + struct cpu_data *c; + unsigned int i; + unsigned int val[MAX_CPUS_PER_CLUSTER]; + unsigned long flags; + int ret; + int not_preferred_count = 0; + + ret = sscanf(buf, "%u %u %u %u %u %u\n", + &val[0], &val[1], &val[2], &val[3], + &val[4], &val[5]); + if (ret != state->num_cpus) + return -EINVAL; + + spin_lock_irqsave(&state_lock, flags); + for (i = 0; i < state->num_cpus; i++) { + c = &per_cpu(cpu_state, i + state->first_cpu); + c->not_preferred = val[i]; + not_preferred_count += !!val[i]; + } + state->nr_not_preferred_cpus = not_preferred_count; + spin_unlock_irqrestore(&state_lock, flags); + + return count; +} + +static ssize_t show_not_preferred(const struct cluster_data *state, char *buf) +{ + struct cpu_data *c; + ssize_t count = 0; + unsigned long flags; + int i; + + spin_lock_irqsave(&state_lock, flags); + for (i = 0; i < state->num_cpus; i++) { + c = &per_cpu(cpu_state, i + state->first_cpu); + count += scnprintf(buf + count, PAGE_SIZE - count, + "CPU#%d: %u\n", c->cpu, c->not_preferred); + } + spin_unlock_irqrestore(&state_lock, flags); + + return count; +} + + +struct core_ctl_attr { + struct attribute attr; + ssize_t (*show)(const struct cluster_data *, char *); + ssize_t (*store)(struct cluster_data *, const char *, size_t count); +}; + +#define core_ctl_attr_ro(_name) \ +static struct core_ctl_attr _name = \ +__ATTR(_name, 0444, show_##_name, NULL) + +#define core_ctl_attr_rw(_name) \ +static struct core_ctl_attr _name = \ +__ATTR(_name, 0644, show_##_name, store_##_name) + +core_ctl_attr_rw(min_cpus); +core_ctl_attr_rw(max_cpus); +core_ctl_attr_rw(offline_delay_ms); +core_ctl_attr_rw(busy_up_thres); +core_ctl_attr_rw(busy_down_thres); +core_ctl_attr_rw(task_thres); +core_ctl_attr_rw(nr_prev_assist_thresh); +core_ctl_attr_ro(need_cpus); +core_ctl_attr_ro(active_cpus); +core_ctl_attr_ro(global_state); +core_ctl_attr_rw(not_preferred); +core_ctl_attr_rw(enable); + +static struct attribute *default_attrs[] = { + &min_cpus.attr, + &max_cpus.attr, + &offline_delay_ms.attr, + &busy_up_thres.attr, + &busy_down_thres.attr, + &task_thres.attr, + &nr_prev_assist_thresh.attr, + &enable.attr, + &need_cpus.attr, + &active_cpus.attr, + &global_state.attr, + ¬_preferred.attr, + NULL +}; + +#define to_cluster_data(k) container_of(k, struct cluster_data, kobj) +#define to_attr(a) container_of(a, struct core_ctl_attr, attr) +static ssize_t show(struct kobject *kobj, struct attribute *attr, char *buf) +{ + struct cluster_data *data = to_cluster_data(kobj); + struct core_ctl_attr *cattr = to_attr(attr); + ssize_t ret = -EIO; + + if (cattr->show) + ret = cattr->show(data, buf); + + return ret; +} + +static ssize_t store(struct kobject *kobj, struct attribute *attr, + const char *buf, size_t count) +{ + struct cluster_data *data = to_cluster_data(kobj); + struct core_ctl_attr *cattr = to_attr(attr); + ssize_t ret = -EIO; + + if (cattr->store) + ret = cattr->store(data, buf, count); + + return ret; +} + +static const struct sysfs_ops sysfs_ops = { + .show = show, + .store = store, +}; + +static struct kobj_type ktype_core_ctl = { + .sysfs_ops = &sysfs_ops, + .default_attrs = default_attrs, +}; + +/* ==================== runqueue based core count =================== */ + +static struct sched_avg_stats nr_stats[NR_CPUS]; + +/* + * nr_need: + * Number of tasks running on this cluster plus + * tasks running on higher capacity clusters. + * To find out CPUs needed from this cluster. + * + * For example: + * On dual cluster system with 4 min capacity + * CPUs and 4 max capacity CPUs, if there are + * 4 small tasks running on min capacity CPUs + * and 2 big tasks running on 2 max capacity + * CPUs, nr_need has to be 6 for min capacity + * cluster and 2 for max capacity cluster. + * This is because, min capacity cluster has to + * account for tasks running on max capacity + * cluster, so that, the min capacity cluster + * can be ready to accommodate tasks running on max + * capacity CPUs if the demand of tasks goes down. + */ +static int compute_cluster_nr_need(int index) +{ + int cpu; + struct cluster_data *cluster; + int nr_need = 0; + + for_each_cluster(cluster, index) { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_need += nr_stats[cpu].nr; + } + + return nr_need; +} + +/* + * prev_misfit_need: + * Tasks running on smaller capacity cluster which + * needs to be migrated to higher capacity cluster. + * To find out how many tasks need higher capacity CPUs. + * + * For example: + * On dual cluster system with 4 min capacity + * CPUs and 4 max capacity CPUs, if there are + * 2 small tasks and 2 big tasks running on + * min capacity CPUs and no tasks running on + * max cpacity, prev_misfit_need of min capacity + * cluster will be 0 and prev_misfit_need of + * max capacity cluster will be 2. + */ +static int compute_prev_cluster_misfit_need(int index) +{ + int cpu; + struct cluster_data *prev_cluster; + int prev_misfit_need = 0; + + /* + * Lowest capacity cluster does not have to + * accommodate any misfit tasks. + */ + if (index == 0) + return 0; + + prev_cluster = &cluster_state[index - 1]; + + for_each_cpu(cpu, &prev_cluster->cpu_mask) + prev_misfit_need += nr_stats[cpu].nr_misfit; + + return prev_misfit_need; +} + +static int compute_cluster_max_nr(int index) +{ + int cpu; + struct cluster_data *cluster = &cluster_state[index]; + int max_nr = 0; + + for_each_cpu(cpu, &cluster->cpu_mask) + max_nr = max(max_nr, nr_stats[cpu].nr_max); + + return max_nr; +} + +static int cluster_real_big_tasks(int index) +{ + int nr_big = 0; + int cpu; + struct cluster_data *cluster = &cluster_state[index]; + + if (index == 0) { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_big += nr_stats[cpu].nr_misfit; + } else { + for_each_cpu(cpu, &cluster->cpu_mask) + nr_big += nr_stats[cpu].nr; + } + + return nr_big; +} + +/* + * prev_nr_need_assist: + * Tasks that are eligible to run on the previous + * cluster but cannot run because of insufficient + * CPUs there. prev_nr_need_assist is indicative + * of number of CPUs in this cluster that should + * assist its previous cluster to makeup for + * insufficient CPUs there. + * + * For example: + * On tri-cluster system with 4 min capacity + * CPUs, 3 intermediate capacity CPUs and 1 + * max capacity CPU, if there are 4 small + * tasks running on min capacity CPUs, 4 big + * tasks running on intermediate capacity CPUs + * and no tasks running on max capacity CPU, + * prev_nr_need_assist for min & max capacity + * clusters will be 0, but, for intermediate + * capacity cluster prev_nr_need_assist will + * be 1 as it has 3 CPUs, but, there are 4 big + * tasks to be served. + */ +static int prev_cluster_nr_need_assist(int index) +{ + int need = 0; + int cpu; + struct cluster_data *prev_cluster; + + if (index == 0) + return 0; + + index--; + prev_cluster = &cluster_state[index]; + + /* + * Next cluster should not assist, while there are isolated cpus + * in this cluster. + */ + if (prev_cluster->nr_isolated_cpus) + return 0; + + for_each_cpu(cpu, &prev_cluster->cpu_mask) + need += nr_stats[cpu].nr; + + need += compute_prev_cluster_misfit_need(index); + + if (need > prev_cluster->active_cpus) + need = need - prev_cluster->active_cpus; + else + need = 0; + + return need; +} + +/* + * This is only implemented for min capacity cluster. + * + * Bringing a little CPU out of isolation and using it + * more does not hurt power as much as bringing big CPUs. + * + * little cluster provides help needed for the other clusters. + * we take nr_scaled (which gives better resolution) and find + * the total nr in the system. Then take out the active higher + * capacity CPUs from the nr and consider the remaining nr as + * strict and consider that many little CPUs are needed. + */ +static int compute_cluster_nr_strict_need(int index) +{ + int cpu; + struct cluster_data *cluster; + int nr_strict_need = 0; + + if (index != 0) + return 0; + + for_each_cluster(cluster, index) { + int nr_scaled = 0; + int active_cpus = cluster->active_cpus; + + for_each_cpu(cpu, &cluster->cpu_mask) + nr_scaled += nr_stats[cpu].nr_scaled; + + nr_scaled /= 100; + + /* + * For little cluster, nr_scaled becomes the nr_strict, + * for other cluster, overflow is counted towards + * the little cluster need. + */ + if (index == 0) + nr_strict_need += nr_scaled; + else + nr_strict_need += max(0, nr_scaled - active_cpus); + } + + return nr_strict_need; +} +static void update_running_avg(void) +{ + struct cluster_data *cluster; + unsigned int index = 0; + unsigned long flags; + int big_avg = 0; + + sched_get_nr_running_avg(nr_stats); + + spin_lock_irqsave(&state_lock, flags); + for_each_cluster(cluster, index) { + int nr_need, prev_misfit_need; + + if (!cluster->inited) + continue; + + nr_need = compute_cluster_nr_need(index); + prev_misfit_need = compute_prev_cluster_misfit_need(index); + + + cluster->nrrun = nr_need + prev_misfit_need; + cluster->max_nr = compute_cluster_max_nr(index); + cluster->nr_prev_assist = prev_cluster_nr_need_assist(index); + + cluster->strict_nrrun = compute_cluster_nr_strict_need(index); + + trace_core_ctl_update_nr_need(cluster->first_cpu, nr_need, + prev_misfit_need, + cluster->nrrun, cluster->max_nr, + cluster->nr_prev_assist); + + big_avg += cluster_real_big_tasks(index); + } + spin_unlock_irqrestore(&state_lock, flags); + + last_nr_big = big_avg; + walt_rotation_checkpoint(big_avg); +} + +#define MAX_NR_THRESHOLD 4 +/* adjust needed CPUs based on current runqueue information */ +static unsigned int apply_task_need(const struct cluster_data *cluster, + unsigned int new_need) +{ + /* unisolate all cores if there are enough tasks */ + if (cluster->nrrun >= cluster->task_thres) + return cluster->num_cpus; + + /* + * unisolate as many cores as the previous cluster + * needs assistance with. + */ + if (cluster->nr_prev_assist >= cluster->nr_prev_assist_thresh) + new_need = new_need + cluster->nr_prev_assist; + + /* only unisolate more cores if there are tasks to run */ + if (cluster->nrrun > new_need) + new_need = new_need + 1; + + /* + * We don't want tasks to be overcrowded in a cluster. + * If any CPU has more than MAX_NR_THRESHOLD in the last + * window, bring another CPU to help out. + */ + if (cluster->max_nr > MAX_NR_THRESHOLD) + new_need = new_need + 1; + + /* + * For little cluster, we use a bit more relaxed approach + * and impose the strict nr condition. Because all tasks can + * spill onto little if big cluster is crowded. + */ + if (new_need < cluster->strict_nrrun) + new_need = cluster->strict_nrrun; + + return new_need; +} + +/* ======================= load based core count ====================== */ + +static unsigned int apply_limits(const struct cluster_data *cluster, + unsigned int need_cpus) +{ + return min(max(cluster->min_cpus, need_cpus), cluster->max_cpus); +} + +static unsigned int get_active_cpu_count(const struct cluster_data *cluster) +{ + return cluster->num_cpus - + sched_isolate_count(&cluster->cpu_mask, true); +} + +static bool is_active(const struct cpu_data *state) +{ + return cpu_online(state->cpu) && !cpu_isolated(state->cpu); +} + +static bool adjustment_possible(const struct cluster_data *cluster, + unsigned int need) +{ + return (need < cluster->active_cpus || (need > cluster->active_cpus && + cluster->nr_isolated_cpus)); +} + +static bool need_all_cpus(const struct cluster_data *cluster) +{ + return (is_min_capacity_cpu(cluster->first_cpu) && + sched_ravg_window < DEFAULT_SCHED_RAVG_WINDOW); +} + +static bool eval_need(struct cluster_data *cluster) +{ + unsigned long flags; + struct cpu_data *c; + unsigned int need_cpus = 0, last_need, thres_idx; + int ret = 0; + bool need_flag = false; + unsigned int new_need; + s64 now, elapsed; + + if (unlikely(!cluster->inited)) + return 0; + + spin_lock_irqsave(&state_lock, flags); + + if (cluster->boost || !cluster->enable || need_all_cpus(cluster)) { + need_cpus = cluster->max_cpus; + } else { + cluster->active_cpus = get_active_cpu_count(cluster); + thres_idx = cluster->active_cpus ? cluster->active_cpus - 1 : 0; + list_for_each_entry(c, &cluster->lru, sib) { + bool old_is_busy = c->is_busy; + + if (c->busy >= cluster->busy_up_thres[thres_idx] || + sched_cpu_high_irqload(c->cpu)) + c->is_busy = true; + else if (c->busy < cluster->busy_down_thres[thres_idx]) + c->is_busy = false; + + trace_core_ctl_set_busy(c->cpu, c->busy, old_is_busy, + c->is_busy); + need_cpus += c->is_busy; + } + need_cpus = apply_task_need(cluster, need_cpus); + } + new_need = apply_limits(cluster, need_cpus); + need_flag = adjustment_possible(cluster, new_need); + + last_need = cluster->need_cpus; + now = ktime_to_ms(ktime_get()); + + if (new_need > cluster->active_cpus) { + ret = 1; + } else { + /* + * When there is no change in need and there are no more + * active CPUs than currently needed, just update the + * need time stamp and return. + */ + if (new_need == last_need && new_need == cluster->active_cpus) { + cluster->need_ts = now; + spin_unlock_irqrestore(&state_lock, flags); + return 0; + } + + elapsed = now - cluster->need_ts; + ret = elapsed >= cluster->offline_delay_ms; + } + + if (ret) { + cluster->need_ts = now; + cluster->need_cpus = new_need; + } + trace_core_ctl_eval_need(cluster->first_cpu, last_need, new_need, + ret && need_flag); + spin_unlock_irqrestore(&state_lock, flags); + + return ret && need_flag; +} + +static void apply_need(struct cluster_data *cluster) +{ + if (eval_need(cluster)) + wake_up_core_ctl_thread(cluster); +} + +/* ========================= core count enforcement ==================== */ + +static void wake_up_core_ctl_thread(struct cluster_data *cluster) +{ + unsigned long flags; + + spin_lock_irqsave(&cluster->pending_lock, flags); + cluster->pending = true; + spin_unlock_irqrestore(&cluster->pending_lock, flags); + + wake_up_process(cluster->core_ctl_thread); +} + +static u64 core_ctl_check_timestamp; + +int core_ctl_set_boost(bool boost) +{ + unsigned int index = 0; + struct cluster_data *cluster = NULL; + unsigned long flags; + int ret = 0; + bool boost_state_changed = false; + + if (unlikely(!initialized)) + return 0; + + spin_lock_irqsave(&state_lock, flags); + for_each_cluster(cluster, index) { + if (boost) { + boost_state_changed = !cluster->boost; + ++cluster->boost; + } else { + if (!cluster->boost) { + ret = -EINVAL; + break; + } else { + --cluster->boost; + boost_state_changed = !cluster->boost; + } + } + } + spin_unlock_irqrestore(&state_lock, flags); + + if (boost_state_changed) { + index = 0; + for_each_cluster(cluster, index) + apply_need(cluster); + } + + if (cluster) + trace_core_ctl_set_boost(cluster->boost, ret); + + return ret; +} +EXPORT_SYMBOL(core_ctl_set_boost); + +void core_ctl_notifier_register(struct notifier_block *n) +{ + atomic_notifier_chain_register(&core_ctl_notifier, n); +} + +void core_ctl_notifier_unregister(struct notifier_block *n) +{ + atomic_notifier_chain_unregister(&core_ctl_notifier, n); +} + +static void core_ctl_call_notifier(void) +{ + struct core_ctl_notif_data ndata = {0}; + struct notifier_block *nb; + + /* + * Don't bother querying the stats when the notifier + * chain is empty. + */ + rcu_read_lock(); + nb = rcu_dereference_raw(core_ctl_notifier.head); + rcu_read_unlock(); + + if (!nb) + return; + + ndata.nr_big = last_nr_big; + walt_fill_ta_data(&ndata); + trace_core_ctl_notif_data(ndata.nr_big, ndata.coloc_load_pct, + ndata.ta_util_pct, ndata.cur_cap_pct); + + atomic_notifier_call_chain(&core_ctl_notifier, 0, &ndata); +} + +void core_ctl_check(u64 window_start) +{ + int cpu; + struct cpu_data *c; + struct cluster_data *cluster; + unsigned int index = 0; + unsigned long flags; + + if (unlikely(!initialized)) + return; + + if (window_start == core_ctl_check_timestamp) + return; + + core_ctl_check_timestamp = window_start; + + spin_lock_irqsave(&state_lock, flags); + for_each_possible_cpu(cpu) { + + c = &per_cpu(cpu_state, cpu); + cluster = c->cluster; + + if (!cluster || !cluster->inited) + continue; + + c->busy = sched_get_cpu_util(cpu); + } + spin_unlock_irqrestore(&state_lock, flags); + + update_running_avg(); + + for_each_cluster(cluster, index) { + if (eval_need(cluster)) + wake_up_core_ctl_thread(cluster); + } + + core_ctl_call_notifier(); +} + +static void move_cpu_lru(struct cpu_data *cpu_data) +{ + unsigned long flags; + + spin_lock_irqsave(&state_lock, flags); + list_del(&cpu_data->sib); + list_add_tail(&cpu_data->sib, &cpu_data->cluster->lru); + spin_unlock_irqrestore(&state_lock, flags); +} + +static bool should_we_isolate(int cpu, struct cluster_data *cluster) +{ + return true; +} + +static void try_to_isolate(struct cluster_data *cluster, unsigned int need) +{ + struct cpu_data *c, *tmp; + unsigned long flags; + unsigned int num_cpus = cluster->num_cpus; + unsigned int nr_isolated = 0; + bool first_pass = cluster->nr_not_preferred_cpus; + + /* + * Protect against entry being removed (and added at tail) by other + * thread (hotplug). + */ + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!is_active(c)) + continue; + if (cluster->active_cpus == need) + break; + /* Don't isolate busy CPUs. */ + if (c->is_busy) + continue; + + /* + * We isolate only the not_preferred CPUs. If none + * of the CPUs are selected as not_preferred, then + * all CPUs are eligible for isolation. + */ + if (cluster->nr_not_preferred_cpus && !c->not_preferred) + continue; + + if (!should_we_isolate(c->cpu, cluster)) + continue; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to isolate CPU%u\n", c->cpu); + if (!sched_isolate_cpu(c->cpu)) { + c->isolated_by_us = true; + move_cpu_lru(c); + nr_isolated++; + } else { + pr_debug("Unable to isolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus += nr_isolated; + spin_unlock_irqrestore(&state_lock, flags); + +again: + /* + * If the number of active CPUs is within the limits, then + * don't force isolation of any busy CPUs. + */ + if (cluster->active_cpus <= cluster->max_cpus) + return; + + nr_isolated = 0; + num_cpus = cluster->num_cpus; + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!is_active(c)) + continue; + if (cluster->active_cpus <= cluster->max_cpus) + break; + + if (first_pass && !c->not_preferred) + continue; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to isolate CPU%u\n", c->cpu); + if (!sched_isolate_cpu(c->cpu)) { + c->isolated_by_us = true; + move_cpu_lru(c); + nr_isolated++; + } else { + pr_debug("Unable to isolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus += nr_isolated; + spin_unlock_irqrestore(&state_lock, flags); + + if (first_pass && cluster->active_cpus > cluster->max_cpus) { + first_pass = false; + goto again; + } +} + +static void __try_to_unisolate(struct cluster_data *cluster, + unsigned int need, bool force) +{ + struct cpu_data *c, *tmp; + unsigned long flags; + unsigned int num_cpus = cluster->num_cpus; + unsigned int nr_unisolated = 0; + + /* + * Protect against entry being removed (and added at tail) by other + * thread (hotplug). + */ + spin_lock_irqsave(&state_lock, flags); + list_for_each_entry_safe(c, tmp, &cluster->lru, sib) { + if (!num_cpus--) + break; + + if (!c->isolated_by_us) + continue; + if ((cpu_online(c->cpu) && !cpu_isolated(c->cpu)) || + (!force && c->not_preferred)) + continue; + if (cluster->active_cpus == need) + break; + + spin_unlock_irqrestore(&state_lock, flags); + + pr_debug("Trying to unisolate CPU%u\n", c->cpu); + if (!sched_unisolate_cpu(c->cpu)) { + c->isolated_by_us = false; + move_cpu_lru(c); + nr_unisolated++; + } else { + pr_debug("Unable to unisolate CPU%u\n", c->cpu); + } + cluster->active_cpus = get_active_cpu_count(cluster); + spin_lock_irqsave(&state_lock, flags); + } + cluster->nr_isolated_cpus -= nr_unisolated; + spin_unlock_irqrestore(&state_lock, flags); +} + +static void try_to_unisolate(struct cluster_data *cluster, unsigned int need) +{ + bool force_use_non_preferred = false; + + __try_to_unisolate(cluster, need, force_use_non_preferred); + + if (cluster->active_cpus == need) + return; + + force_use_non_preferred = true; + __try_to_unisolate(cluster, need, force_use_non_preferred); +} + +static void __ref do_core_ctl(struct cluster_data *cluster) +{ + unsigned int need; + + need = apply_limits(cluster, cluster->need_cpus); + + if (adjustment_possible(cluster, need)) { + pr_debug("Trying to adjust group %u from %u to %u\n", + cluster->first_cpu, cluster->active_cpus, need); + + if (cluster->active_cpus > need) + try_to_isolate(cluster, need); + else if (cluster->active_cpus < need) + try_to_unisolate(cluster, need); + } +} + +static int __ref try_core_ctl(void *data) +{ + struct cluster_data *cluster = data; + unsigned long flags; + + while (1) { + set_current_state(TASK_INTERRUPTIBLE); + spin_lock_irqsave(&cluster->pending_lock, flags); + if (!cluster->pending) { + spin_unlock_irqrestore(&cluster->pending_lock, flags); + schedule(); + if (kthread_should_stop()) + break; + spin_lock_irqsave(&cluster->pending_lock, flags); + } + set_current_state(TASK_RUNNING); + cluster->pending = false; + spin_unlock_irqrestore(&cluster->pending_lock, flags); + + do_core_ctl(cluster); + } + + return 0; +} + +static int isolation_cpuhp_state(unsigned int cpu, bool online) +{ + struct cpu_data *state = &per_cpu(cpu_state, cpu); + struct cluster_data *cluster = state->cluster; + unsigned int need; + bool do_wakeup = false, unisolated = false; + unsigned long flags; + + if (unlikely(!cluster || !cluster->inited)) + return 0; + + if (online) { + cluster->active_cpus = get_active_cpu_count(cluster); + + /* + * Moving to the end of the list should only happen in + * CPU_ONLINE and not on CPU_UP_PREPARE to prevent an + * infinite list traversal when thermal (or other entities) + * reject trying to online CPUs. + */ + move_cpu_lru(state); + } else { + /* + * We don't want to have a CPU both offline and isolated. + * So unisolate a CPU that went down if it was isolated by us. + */ + if (state->isolated_by_us) { + sched_unisolate_cpu_unlocked(cpu); + state->isolated_by_us = false; + unisolated = true; + } + + /* Move a CPU to the end of the LRU when it goes offline. */ + move_cpu_lru(state); + + state->busy = 0; + cluster->active_cpus = get_active_cpu_count(cluster); + } + + need = apply_limits(cluster, cluster->need_cpus); + spin_lock_irqsave(&state_lock, flags); + if (unisolated) + cluster->nr_isolated_cpus--; + do_wakeup = adjustment_possible(cluster, need); + spin_unlock_irqrestore(&state_lock, flags); + if (do_wakeup) + wake_up_core_ctl_thread(cluster); + + return 0; +} + +static int core_ctl_isolation_online_cpu(unsigned int cpu) +{ + return isolation_cpuhp_state(cpu, true); +} + +static int core_ctl_isolation_dead_cpu(unsigned int cpu) +{ + return isolation_cpuhp_state(cpu, false); +} + +/* ============================ init code ============================== */ + +static struct cluster_data *find_cluster_by_first_cpu(unsigned int first_cpu) +{ + unsigned int i; + + for (i = 0; i < num_clusters; ++i) { + if (cluster_state[i].first_cpu == first_cpu) + return &cluster_state[i]; + } + + return NULL; +} + +static int cluster_init(const struct cpumask *mask) +{ + struct device *dev; + unsigned int first_cpu = cpumask_first(mask); + struct cluster_data *cluster; + struct cpu_data *state; + unsigned int cpu; + struct sched_param param = { .sched_priority = MAX_RT_PRIO-1 }; + + if (find_cluster_by_first_cpu(first_cpu)) + return 0; + + dev = get_cpu_device(first_cpu); + if (!dev) + return -ENODEV; + + pr_info("Creating CPU group %d\n", first_cpu); + + if (num_clusters == MAX_CLUSTERS) { + pr_err("Unsupported number of clusters. Only %u supported\n", + MAX_CLUSTERS); + return -EINVAL; + } + cluster = &cluster_state[num_clusters]; + ++num_clusters; + + cpumask_copy(&cluster->cpu_mask, mask); + cluster->num_cpus = cpumask_weight(mask); + if (cluster->num_cpus > MAX_CPUS_PER_CLUSTER) { + pr_err("HW configuration not supported\n"); + return -EINVAL; + } + cluster->first_cpu = first_cpu; + cluster->min_cpus = 1; + cluster->max_cpus = cluster->num_cpus; + cluster->need_cpus = cluster->num_cpus; + cluster->offline_delay_ms = 100; + cluster->task_thres = UINT_MAX; + cluster->nr_prev_assist_thresh = UINT_MAX; + cluster->nrrun = cluster->num_cpus; + cluster->enable = true; + cluster->nr_not_preferred_cpus = 0; + cluster->strict_nrrun = 0; + INIT_LIST_HEAD(&cluster->lru); + spin_lock_init(&cluster->pending_lock); + + for_each_cpu(cpu, mask) { + pr_info("Init CPU%u state\n", cpu); + + state = &per_cpu(cpu_state, cpu); + state->cluster = cluster; + state->cpu = cpu; + list_add_tail(&state->sib, &cluster->lru); + } + cluster->active_cpus = get_active_cpu_count(cluster); + + cluster->core_ctl_thread = kthread_run(try_core_ctl, (void *) cluster, + "core_ctl/%d", first_cpu); + if (IS_ERR(cluster->core_ctl_thread)) + return PTR_ERR(cluster->core_ctl_thread); + + sched_setscheduler_nocheck(cluster->core_ctl_thread, SCHED_FIFO, + ¶m); + + cluster->inited = true; + + kobject_init(&cluster->kobj, &ktype_core_ctl); + return kobject_add(&cluster->kobj, &dev->kobj, "core_ctl"); +} + +static int __init core_ctl_init(void) +{ + struct walt_sched_cluster *cluster; + int ret; + + cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, + "core_ctl/isolation:online", + core_ctl_isolation_online_cpu, NULL); + + cpuhp_setup_state_nocalls(CPUHP_CORE_CTL_ISOLATION_DEAD, + "core_ctl/isolation:dead", + NULL, core_ctl_isolation_dead_cpu); + + for_each_sched_cluster(cluster) { + ret = cluster_init(&cluster->cpus); + if (ret) + pr_warn("unable to create core ctl group: %d\n", ret); + } + + initialized = true; + return 0; +} + +late_initcall(core_ctl_init); diff --git a/kernel/sched/walt/cpu-boost.c b/kernel/sched/walt/cpu-boost.c new file mode 100644 index 000000000000..006510c55f13 --- /dev/null +++ b/kernel/sched/walt/cpu-boost.c @@ -0,0 +1,389 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2013-2015,2017,2019-2021, The Linux Foundation. All rights reserved. + */ +#define pr_fmt(fmt) "cpu-boost: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "qc_vas.h" + +#define cpu_boost_attr_rw(_name) \ +static struct kobj_attribute _name##_attr = \ +__ATTR(_name, 0644, show_##_name, store_##_name) + +#define show_one(file_name) \ +static ssize_t show_##file_name \ +(struct kobject *kobj, struct kobj_attribute *attr, char *buf) \ +{ \ + return scnprintf(buf, PAGE_SIZE, "%u\n", file_name); \ +} + +#define store_one(file_name) \ +static ssize_t store_##file_name \ +(struct kobject *kobj, struct kobj_attribute *attr, \ +const char *buf, size_t count) \ +{ \ + \ + sscanf(buf, "%u", &file_name); \ + return count; \ +} + +struct cpu_sync { + int cpu; + unsigned int input_boost_min; + unsigned int input_boost_freq; +}; + +static DEFINE_PER_CPU(struct cpu_sync, sync_info); +static struct workqueue_struct *cpu_boost_wq; + +static struct work_struct input_boost_work; + +static bool input_boost_enabled; + +static unsigned int input_boost_ms = 40; +show_one(input_boost_ms); +store_one(input_boost_ms); +cpu_boost_attr_rw(input_boost_ms); + +static unsigned int sched_boost_on_input; +show_one(sched_boost_on_input); +store_one(sched_boost_on_input); +cpu_boost_attr_rw(sched_boost_on_input); + +static bool sched_boost_active; + +static struct delayed_work input_boost_rem; +static u64 last_input_time; +#define MIN_INPUT_INTERVAL (150 * USEC_PER_MSEC) + +static DEFINE_PER_CPU(struct freq_qos_request, qos_req); + +static ssize_t store_input_boost_freq(struct kobject *kobj, + struct kobj_attribute *attr, + const char *buf, size_t count) +{ + int i, ntokens = 0; + unsigned int val, cpu; + const char *cp = buf; + bool enabled = false; + + while ((cp = strpbrk(cp + 1, " :"))) + ntokens++; + + /* single number: apply to all CPUs */ + if (!ntokens) { + if (sscanf(buf, "%u\n", &val) != 1) + return -EINVAL; + for_each_possible_cpu(i) + per_cpu(sync_info, i).input_boost_freq = val; + goto check_enable; + } + + /* CPU:value pair */ + if (!(ntokens % 2)) + return -EINVAL; + + cp = buf; + for (i = 0; i < ntokens; i += 2) { + if (sscanf(cp, "%u:%u", &cpu, &val) != 2) + return -EINVAL; + if (cpu >= num_possible_cpus()) + return -EINVAL; + + per_cpu(sync_info, cpu).input_boost_freq = val; + cp = strnchr(cp, PAGE_SIZE - (cp - buf), ' '); + cp++; + } + +check_enable: + for_each_possible_cpu(i) { + if (per_cpu(sync_info, i).input_boost_freq) { + enabled = true; + break; + } + } + input_boost_enabled = enabled; + + return count; +} + +static ssize_t show_input_boost_freq(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + int cnt = 0, cpu; + struct cpu_sync *s; + + for_each_possible_cpu(cpu) { + s = &per_cpu(sync_info, cpu); + cnt += snprintf(buf + cnt, PAGE_SIZE - cnt, + "%d:%u ", cpu, s->input_boost_freq); + } + cnt += snprintf(buf + cnt, PAGE_SIZE - cnt, "\n"); + return cnt; +} + +cpu_boost_attr_rw(input_boost_freq); + +static void boost_adjust_notify(struct cpufreq_policy *policy) +{ + unsigned int cpu = policy->cpu; + struct cpu_sync *s = &per_cpu(sync_info, cpu); + unsigned int ib_min = s->input_boost_min; + struct freq_qos_request *req = &per_cpu(qos_req, cpu); + int ret; + + pr_debug("CPU%u policy min before boost: %u kHz\n", + cpu, policy->min); + pr_debug("CPU%u boost min: %u kHz\n", cpu, ib_min); + + ret = freq_qos_update_request(req, ib_min); + + if (ret < 0) + pr_err("Failed to update freq constraint in boost_adjust: %d\n", + ib_min); + + pr_debug("CPU%u policy min after boost: %u kHz\n", + cpu, policy->min); + + return; +} + +static void update_policy_online(void) +{ + unsigned int i; + struct cpufreq_policy *policy; + struct cpumask online_cpus; + /* Re-evaluate policy to trigger adjust notifier for online CPUs */ + get_online_cpus(); + online_cpus = *cpu_online_mask; + for_each_cpu(i, &online_cpus) { + policy = cpufreq_cpu_get(i); + if (!policy) { + pr_err("%s: cpufreq policy not found for cpu%d\n", + __func__, i); + return; + } + + cpumask_andnot(&online_cpus, &online_cpus, + policy->related_cpus); + boost_adjust_notify(policy); + } + put_online_cpus(); +} + +static void do_input_boost_rem(struct work_struct *work) +{ + unsigned int i, ret; + struct cpu_sync *i_sync_info; + + /* Reset the input_boost_min for all CPUs in the system */ + pr_debug("Resetting input boost min for all CPUs\n"); + for_each_possible_cpu(i) { + i_sync_info = &per_cpu(sync_info, i); + i_sync_info->input_boost_min = 0; + } + + /* Update policies for all online CPUs */ + update_policy_online(); + + if (sched_boost_active) { + ret = sched_set_boost(0); + if (ret) + pr_err("cpu-boost: sched boost disable failed\n"); + sched_boost_active = false; + } +} + +static void do_input_boost(struct work_struct *work) +{ + unsigned int i, ret; + struct cpu_sync *i_sync_info; + + cancel_delayed_work_sync(&input_boost_rem); + if (sched_boost_active) { + sched_set_boost(0); + sched_boost_active = false; + } + + /* Set the input_boost_min for all CPUs in the system */ + pr_debug("Setting input boost min for all CPUs\n"); + for_each_possible_cpu(i) { + i_sync_info = &per_cpu(sync_info, i); + i_sync_info->input_boost_min = i_sync_info->input_boost_freq; + } + + /* Update policies for all online CPUs */ + update_policy_online(); + + /* Enable scheduler boost to migrate tasks to big cluster */ + if (sched_boost_on_input > 0) { + ret = sched_set_boost(sched_boost_on_input); + if (ret) + pr_err("cpu-boost: sched boost enable failed\n"); + else + sched_boost_active = true; + } + + queue_delayed_work(cpu_boost_wq, &input_boost_rem, + msecs_to_jiffies(input_boost_ms)); +} + +static void cpuboost_input_event(struct input_handle *handle, + unsigned int type, unsigned int code, int value) +{ + u64 now; + + if (!input_boost_enabled) + return; + + now = ktime_to_us(ktime_get()); + if (now - last_input_time < MIN_INPUT_INTERVAL) + return; + + if (work_pending(&input_boost_work)) + return; + + queue_work(cpu_boost_wq, &input_boost_work); + last_input_time = ktime_to_us(ktime_get()); +} + +static int cpuboost_input_connect(struct input_handler *handler, + struct input_dev *dev, const struct input_device_id *id) +{ + struct input_handle *handle; + int error; + + handle = kzalloc(sizeof(struct input_handle), GFP_KERNEL); + if (!handle) + return -ENOMEM; + + handle->dev = dev; + handle->handler = handler; + handle->name = "cpufreq"; + + error = input_register_handle(handle); + if (error) + goto err2; + + error = input_open_device(handle); + if (error) + goto err1; + + return 0; +err1: + input_unregister_handle(handle); +err2: + kfree(handle); + return error; +} + +static void cpuboost_input_disconnect(struct input_handle *handle) +{ + input_close_device(handle); + input_unregister_handle(handle); + kfree(handle); +} + +static const struct input_device_id cpuboost_ids[] = { + /* multi-touch touchscreen */ + { + .flags = INPUT_DEVICE_ID_MATCH_EVBIT | + INPUT_DEVICE_ID_MATCH_ABSBIT, + .evbit = { BIT_MASK(EV_ABS) }, + .absbit = { [BIT_WORD(ABS_MT_POSITION_X)] = + BIT_MASK(ABS_MT_POSITION_X) | + BIT_MASK(ABS_MT_POSITION_Y) }, + }, + /* touchpad */ + { + .flags = INPUT_DEVICE_ID_MATCH_KEYBIT | + INPUT_DEVICE_ID_MATCH_ABSBIT, + .keybit = { [BIT_WORD(BTN_TOUCH)] = BIT_MASK(BTN_TOUCH) }, + .absbit = { [BIT_WORD(ABS_X)] = + BIT_MASK(ABS_X) | BIT_MASK(ABS_Y) }, + }, + /* Keypad */ + { + .flags = INPUT_DEVICE_ID_MATCH_EVBIT, + .evbit = { BIT_MASK(EV_KEY) }, + }, + { }, +}; + +static struct input_handler cpuboost_input_handler = { + .event = cpuboost_input_event, + .connect = cpuboost_input_connect, + .disconnect = cpuboost_input_disconnect, + .name = "cpu-boost", + .id_table = cpuboost_ids, +}; + +struct kobject *cpu_boost_kobj; +static int cpu_boost_init(void) +{ + int cpu, ret; + struct cpu_sync *s; + struct cpufreq_policy *policy; + struct freq_qos_request *req; + + cpu_boost_wq = alloc_workqueue("cpuboost_wq", WQ_HIGHPRI, 0); + if (!cpu_boost_wq) + return -EFAULT; + + INIT_WORK(&input_boost_work, do_input_boost); + INIT_DELAYED_WORK(&input_boost_rem, do_input_boost_rem); + + for_each_possible_cpu(cpu) { + s = &per_cpu(sync_info, cpu); + s->cpu = cpu; + req = &per_cpu(qos_req, cpu); + policy = cpufreq_cpu_get(cpu); + if (!policy) { + pr_err("%s: cpufreq policy not found for cpu%d\n", + __func__, cpu); + return -ESRCH; + } + + ret = freq_qos_add_request(&policy->constraints, req, + FREQ_QOS_MIN, policy->min); + if (ret < 0) { + pr_err("%s: Failed to add freq constraint (%d)\n", + __func__, ret); + return ret; + } + + } + + cpu_boost_kobj = kobject_create_and_add("cpu_boost", + &cpu_subsys.dev_root->kobj); + if (!cpu_boost_kobj) + pr_err("Failed to initialize sysfs node for cpu_boost.\n"); + + ret = sysfs_create_file(cpu_boost_kobj, &input_boost_ms_attr.attr); + if (ret) + pr_err("Failed to create input_boost_ms node: %d\n", ret); + + ret = sysfs_create_file(cpu_boost_kobj, &input_boost_freq_attr.attr); + if (ret) + pr_err("Failed to create input_boost_freq node: %d\n", ret); + + ret = sysfs_create_file(cpu_boost_kobj, + &sched_boost_on_input_attr.attr); + if (ret) + pr_err("Failed to create sched_boost_on_input node: %d\n", ret); + + ret = input_register_handler(&cpuboost_input_handler); + return 0; +} +late_initcall(cpu_boost_init); diff --git a/kernel/sched/walt/qc_vas.c b/kernel/sched/walt/qc_vas.c new file mode 100644 index 000000000000..d4d6c838babe --- /dev/null +++ b/kernel/sched/walt/qc_vas.c @@ -0,0 +1,744 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ +#include +#include +#include + +#include "qc_vas.h" + +#ifdef CONFIG_SCHED_WALT +/* 1ms default for 20ms window size scaled to 1024 */ +unsigned int sysctl_sched_min_task_util_for_boost = 51; +/* 0.68ms default for 20ms window size scaled to 1024 */ +unsigned int sysctl_sched_min_task_util_for_colocation = 35; + +int +kick_active_balance(struct rq *rq, struct task_struct *p, int new_cpu) +{ + unsigned long flags; + int rc = 0; + + /* Invoke active balance to force migrate currently running task */ + raw_spin_lock_irqsave(&rq->lock, flags); + if (!rq->active_balance) { + rq->active_balance = 1; + rq->push_cpu = new_cpu; + get_task_struct(p); + rq->wrq.push_task = p; + rc = 1; + } + raw_spin_unlock_irqrestore(&rq->lock, flags); + + return rc; +} + +struct walt_rotate_work { + struct work_struct w; + struct task_struct *src_task; + struct task_struct *dst_task; + int src_cpu; + int dst_cpu; +}; + +DEFINE_PER_CPU(struct walt_rotate_work, walt_rotate_works); + +void walt_rotate_work_func(struct work_struct *work) +{ + struct walt_rotate_work *wr = container_of(work, + struct walt_rotate_work, w); + + migrate_swap(wr->src_task, wr->dst_task, wr->dst_cpu, wr->src_cpu); + + put_task_struct(wr->src_task); + put_task_struct(wr->dst_task); + + clear_reserved(wr->src_cpu); + clear_reserved(wr->dst_cpu); +} + +void walt_rotate_work_init(void) +{ + int i; + + for_each_possible_cpu(i) { + struct walt_rotate_work *wr = &per_cpu(walt_rotate_works, i); + + INIT_WORK(&wr->w, walt_rotate_work_func); + } +} + +#define WALT_ROTATION_THRESHOLD_NS 16000000 +void walt_check_for_rotation(struct rq *src_rq) +{ + u64 wc, wait, max_wait = 0, run, max_run = 0; + int deserved_cpu = nr_cpu_ids, dst_cpu = nr_cpu_ids; + int i, src_cpu = cpu_of(src_rq); + struct rq *dst_rq; + struct walt_rotate_work *wr = NULL; + + if (!walt_rotation_enabled) + return; + + if (!is_min_capacity_cpu(src_cpu)) + return; + + wc = sched_ktime_clock(); + for_each_possible_cpu(i) { + struct rq *rq = cpu_rq(i); + + if (!is_min_capacity_cpu(i)) + break; + + if (is_reserved(i)) + continue; + + if (!rq->misfit_task_load || rq->curr->sched_class != + &fair_sched_class) + continue; + + wait = wc - rq->curr->wts.last_enqueued_ts; + if (wait > max_wait) { + max_wait = wait; + deserved_cpu = i; + } + } + + if (deserved_cpu != src_cpu) + return; + + for_each_possible_cpu(i) { + struct rq *rq = cpu_rq(i); + + if (is_min_capacity_cpu(i)) + continue; + + if (is_reserved(i)) + continue; + + if (rq->curr->sched_class != &fair_sched_class) + continue; + + if (rq->nr_running > 1) + continue; + + run = wc - rq->curr->wts.last_enqueued_ts; + + if (run < WALT_ROTATION_THRESHOLD_NS) + continue; + + if (run > max_run) { + max_run = run; + dst_cpu = i; + } + } + + if (dst_cpu == nr_cpu_ids) + return; + + dst_rq = cpu_rq(dst_cpu); + + double_rq_lock(src_rq, dst_rq); + if (dst_rq->curr->sched_class == &fair_sched_class) { + get_task_struct(src_rq->curr); + get_task_struct(dst_rq->curr); + + mark_reserved(src_cpu); + mark_reserved(dst_cpu); + wr = &per_cpu(walt_rotate_works, src_cpu); + + wr->src_task = src_rq->curr; + wr->dst_task = dst_rq->curr; + + wr->src_cpu = src_cpu; + wr->dst_cpu = dst_cpu; + } + double_rq_unlock(src_rq, dst_rq); + + if (wr) + queue_work_on(src_cpu, system_highpri_wq, &wr->w); +} + +DEFINE_RAW_SPINLOCK(migration_lock); +void check_for_migration(struct rq *rq, struct task_struct *p) +{ + int active_balance; + int new_cpu = -1; + int prev_cpu = task_cpu(p); + int ret; + + if (rq->misfit_task_load) { + if (rq->curr->state != TASK_RUNNING || + rq->curr->nr_cpus_allowed == 1) + return; + + if (walt_rotation_enabled) { + raw_spin_lock(&migration_lock); + walt_check_for_rotation(rq); + raw_spin_unlock(&migration_lock); + return; + } + + raw_spin_lock(&migration_lock); + rcu_read_lock(); + new_cpu = find_energy_efficient_cpu(p, prev_cpu, 0, 1); + rcu_read_unlock(); + if ((new_cpu >= 0) && (new_cpu != prev_cpu) && + (capacity_orig_of(new_cpu) > capacity_orig_of(prev_cpu))) { + active_balance = kick_active_balance(rq, p, new_cpu); + if (active_balance) { + mark_reserved(new_cpu); + raw_spin_unlock(&migration_lock); + ret = stop_one_cpu_nowait(prev_cpu, + active_load_balance_cpu_stop, rq, + &rq->active_balance_work); + if (!ret) + clear_reserved(new_cpu); + else + wake_up_if_idle(new_cpu); + return; + } + } + raw_spin_unlock(&migration_lock); + } +} + +int sched_init_task_load_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_init_task_load(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_init_task_load_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int init_task_load, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &init_task_load); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_init_task_load(p, init_task_load); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_init_task_load_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_init_task_load_show, inode); +} + +int sched_group_id_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_group_id(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_group_id_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int group_id, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &group_id); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_group_id(p, group_id); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_group_id_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_group_id_show, inode); +} + +#ifdef CONFIG_SMP +/* + * Print out various scheduling related per-task fields: + */ +int sched_wake_up_idle_show(struct seq_file *m, void *v) +{ + struct inode *inode = m->private; + struct task_struct *p; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + seq_printf(m, "%d\n", sched_get_wake_up_idle(p)); + + put_task_struct(p); + + return 0; +} + +ssize_t +sched_wake_up_idle_write(struct file *file, const char __user *buf, + size_t count, loff_t *offset) +{ + struct inode *inode = file_inode(file); + struct task_struct *p; + char buffer[PROC_NUMBUF]; + int wake_up_idle, err; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &wake_up_idle); + if (err) + goto out; + + p = get_proc_task(inode); + if (!p) + return -ESRCH; + + err = sched_set_wake_up_idle(p, wake_up_idle); + + put_task_struct(p); + +out: + return err < 0 ? err : count; +} + +int sched_wake_up_idle_open(struct inode *inode, struct file *filp) +{ + return single_open(filp, sched_wake_up_idle_show, inode); +} + +int group_balance_cpu_not_isolated(struct sched_group *sg) +{ + cpumask_t cpus; + + cpumask_and(&cpus, sched_group_span(sg), group_balance_mask(sg)); + cpumask_andnot(&cpus, &cpus, cpu_isolated_mask); + return cpumask_first(&cpus); +} +#endif /* CONFIG_SMP */ + +#ifdef CONFIG_PROC_SYSCTL +static void sched_update_updown_migrate_values(bool up) +{ + int i = 0, cpu; + struct walt_sched_cluster *cluster; + int cap_margin_levels = num_sched_clusters - 1; + + if (cap_margin_levels > 1) { + /* + * No need to worry about CPUs in last cluster + * if there are more than 2 clusters in the system + */ + for_each_sched_cluster(cluster) { + for_each_cpu(cpu, &cluster->cpus) { + if (up) + sched_capacity_margin_up[cpu] = + sysctl_sched_capacity_margin_up[i]; + else + sched_capacity_margin_down[cpu] = + sysctl_sched_capacity_margin_down[i]; + } + + if (++i >= cap_margin_levels) + break; + } + } else { + for_each_possible_cpu(cpu) { + if (up) + sched_capacity_margin_up[cpu] = + sysctl_sched_capacity_margin_up[0]; + else + sched_capacity_margin_down[cpu] = + sysctl_sched_capacity_margin_down[0]; + } + } +} + +int sched_updown_migrate_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret, i; + unsigned int *data = (unsigned int *)table->data; + unsigned int *old_val; + static DEFINE_MUTEX(mutex); + int cap_margin_levels = num_sched_clusters ? num_sched_clusters - 1 : 0; + + if (cap_margin_levels <= 0) + return -EINVAL; + + mutex_lock(&mutex); + + if (table->maxlen != (sizeof(unsigned int) * cap_margin_levels)) + table->maxlen = sizeof(unsigned int) * cap_margin_levels; + + if (!write) { + ret = proc_douintvec_capacity(table, write, buffer, lenp, ppos); + goto unlock_mutex; + } + + /* + * Cache the old values so that they can be restored + * if either the write fails (for example out of range values) + * or the downmigrate and upmigrate are not in sync. + */ + old_val = kzalloc(table->maxlen, GFP_KERNEL); + if (!old_val) { + ret = -ENOMEM; + goto unlock_mutex; + } + + memcpy(old_val, data, table->maxlen); + + ret = proc_douintvec_capacity(table, write, buffer, lenp, ppos); + + if (ret) { + memcpy(data, old_val, table->maxlen); + goto free_old_val; + } + + for (i = 0; i < cap_margin_levels; i++) { + if (sysctl_sched_capacity_margin_up[i] > + sysctl_sched_capacity_margin_down[i]) { + memcpy(data, old_val, table->maxlen); + ret = -EINVAL; + goto free_old_val; + } + } + + sched_update_updown_migrate_values(data == + &sysctl_sched_capacity_margin_up[0]); + +free_old_val: + kfree(old_val); +unlock_mutex: + mutex_unlock(&mutex); + + return ret; +} +#endif /* CONFIG_PROC_SYSCTL */ + +int sched_isolate_count(const cpumask_t *mask, bool include_offline) +{ + cpumask_t count_mask = CPU_MASK_NONE; + + if (include_offline) { + cpumask_complement(&count_mask, cpu_online_mask); + cpumask_or(&count_mask, &count_mask, cpu_isolated_mask); + cpumask_and(&count_mask, &count_mask, mask); + } else { + cpumask_and(&count_mask, mask, cpu_isolated_mask); + } + + return cpumask_weight(&count_mask); +} + +#ifdef CONFIG_HOTPLUG_CPU +static int do_isolation_work_cpu_stop(void *data) +{ + unsigned int cpu = smp_processor_id(); + struct rq *rq = cpu_rq(cpu); + struct rq_flags rf; + + local_irq_disable(); + + irq_migrate_all_off_this_cpu(); + + sched_ttwu_pending(); + + /* Update our root-domain */ + rq_lock(rq, &rf); + + /* + * Temporarily mark the rq as offline. This will allow us to + * move tasks off the CPU. + */ + if (rq->rd) { + BUG_ON(!cpumask_test_cpu(cpu, rq->rd->span)); + set_rq_offline(rq); + } + + migrate_tasks(rq, &rf, false); + + if (rq->rd) + set_rq_online(rq); + rq_unlock(rq, &rf); + + clear_walt_request(cpu); + local_irq_enable(); + return 0; +} + +static int do_unisolation_work_cpu_stop(void *data) +{ + watchdog_enable(smp_processor_id()); + return 0; +} + +static void sched_update_group_capacities(int cpu) +{ + struct sched_domain *sd; + + mutex_lock(&sched_domains_mutex); + rcu_read_lock(); + + for_each_domain(cpu, sd) { + int balance_cpu = group_balance_cpu(sd->groups); + + init_sched_groups_capacity(cpu, sd); + /* + * Need to ensure this is also called with balancing + * cpu. + */ + if (cpu != balance_cpu) + init_sched_groups_capacity(balance_cpu, sd); + } + + rcu_read_unlock(); + mutex_unlock(&sched_domains_mutex); +} + +static unsigned int cpu_isolation_vote[NR_CPUS]; + +/* + * 1) CPU is isolated and cpu is offlined: + * Unisolate the core. + * 2) CPU is not isolated and CPU is offlined: + * No action taken. + * 3) CPU is offline and request to isolate + * Request ignored. + * 4) CPU is offline and isolated: + * Not a possible state. + * 5) CPU is online and request to isolate + * Normal case: Isolate the CPU + * 6) CPU is not isolated and comes back online + * Nothing to do + * + * Note: The client calling sched_isolate_cpu() is repsonsible for ONLY + * calling sched_unisolate_cpu() on a CPU that the client previously isolated. + * Client is also responsible for unisolating when a core goes offline + * (after CPU is marked offline). + */ +int sched_isolate_cpu(int cpu) +{ + struct rq *rq; + cpumask_t avail_cpus; + int ret_code = 0; + u64 start_time = 0; + + if (trace_sched_isolate_enabled()) + start_time = sched_clock(); + + cpu_maps_update_begin(); + + cpumask_andnot(&avail_cpus, cpu_online_mask, cpu_isolated_mask); + + if (cpu < 0 || cpu >= nr_cpu_ids || !cpu_possible(cpu) || + !cpu_online(cpu) || cpu >= NR_CPUS) { + ret_code = -EINVAL; + goto out; + } + + rq = cpu_rq(cpu); + + if (++cpu_isolation_vote[cpu] > 1) + goto out; + + /* We cannot isolate ALL cpus in the system */ + if (cpumask_weight(&avail_cpus) == 1) { + --cpu_isolation_vote[cpu]; + ret_code = -EINVAL; + goto out; + } + + /* + * There is a race between watchdog being enabled by hotplug and + * core isolation disabling the watchdog. When a CPU is hotplugged in + * and the hotplug lock has been released the watchdog thread might + * not have run yet to enable the watchdog. + * We have to wait for the watchdog to be enabled before proceeding. + */ + if (!watchdog_configured(cpu)) { + msleep(20); + if (!watchdog_configured(cpu)) { + --cpu_isolation_vote[cpu]; + ret_code = -EBUSY; + goto out; + } + } + + set_cpu_isolated(cpu, true); + cpumask_clear_cpu(cpu, &avail_cpus); + + /* Migrate timers */ + smp_call_function_any(&avail_cpus, hrtimer_quiesce_cpu, &cpu, 1); + smp_call_function_any(&avail_cpus, timer_quiesce_cpu, &cpu, 1); + + watchdog_disable(cpu); + irq_lock_sparse(); + stop_cpus(cpumask_of(cpu), do_isolation_work_cpu_stop, 0); + irq_unlock_sparse(); + + calc_load_migrate(rq); + update_max_interval(); + sched_update_group_capacities(cpu); + +out: + cpu_maps_update_done(); + trace_sched_isolate(cpu, cpumask_bits(cpu_isolated_mask)[0], + start_time, 1); + return ret_code; +} + +/* + * Note: The client calling sched_isolate_cpu() is repsonsible for ONLY + * calling sched_unisolate_cpu() on a CPU that the client previously isolated. + * Client is also responsible for unisolating when a core goes offline + * (after CPU is marked offline). + */ +int sched_unisolate_cpu_unlocked(int cpu) +{ + int ret_code = 0; + u64 start_time = 0; + + if (cpu < 0 || cpu >= nr_cpu_ids || !cpu_possible(cpu) + || cpu >= NR_CPUS) { + ret_code = -EINVAL; + goto out; + } + + if (trace_sched_isolate_enabled()) + start_time = sched_clock(); + + if (!cpu_isolation_vote[cpu]) { + ret_code = -EINVAL; + goto out; + } + + if (--cpu_isolation_vote[cpu]) + goto out; + + set_cpu_isolated(cpu, false); + update_max_interval(); + sched_update_group_capacities(cpu); + + if (cpu_online(cpu)) { + stop_cpus(cpumask_of(cpu), do_unisolation_work_cpu_stop, 0); + + /* Kick CPU to immediately do load balancing */ + if (!atomic_fetch_or(NOHZ_KICK_MASK, nohz_flags(cpu))) + smp_send_reschedule(cpu); + } + +out: + trace_sched_isolate(cpu, cpumask_bits(cpu_isolated_mask)[0], + start_time, 0); + return ret_code; +} + +int sched_unisolate_cpu(int cpu) +{ + int ret_code; + + cpu_maps_update_begin(); + ret_code = sched_unisolate_cpu_unlocked(cpu); + cpu_maps_update_done(); + return ret_code; +} + +/* + * Remove a task from the runqueue and pretend that it's migrating. This + * should prevent migrations for the detached task and disallow further + * changes to tsk_cpus_allowed. + */ +void +detach_one_task_core(struct task_struct *p, struct rq *rq, + struct list_head *tasks) +{ + lockdep_assert_held(&rq->lock); + + p->on_rq = TASK_ON_RQ_MIGRATING; + deactivate_task(rq, p, 0); + list_add(&p->se.group_node, tasks); +} + +void attach_tasks_core(struct list_head *tasks, struct rq *rq) +{ + struct task_struct *p; + + lockdep_assert_held(&rq->lock); + + while (!list_empty(tasks)) { + p = list_first_entry(tasks, struct task_struct, se.group_node); + list_del_init(&p->se.group_node); + + BUG_ON(task_rq(p) != rq); + activate_task(rq, p, 0); + p->on_rq = TASK_ON_RQ_QUEUED; + } +} +#endif /* CONFIG_HOTPLUG_CPU */ +#endif /* CONFIG_SCHED_WALT */ diff --git a/kernel/sched/walt/qc_vas.h b/kernel/sched/walt/qc_vas.h new file mode 100644 index 000000000000..3b67a5494655 --- /dev/null +++ b/kernel/sched/walt/qc_vas.h @@ -0,0 +1,81 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#include "../sched.h" +#include "../../../fs/proc/internal.h" + +#include "walt.h" +#include "trace.h" + +#ifdef CONFIG_SCHED_WALT +#ifdef CONFIG_HZ_300 +/* + * Tick interval becomes to 3333333 due to + * rounding error when HZ=300. + */ +#define DEFAULT_SCHED_RAVG_WINDOW (3333333 * 5) +#else +/* Min window size (in ns) = 16ms */ +#define DEFAULT_SCHED_RAVG_WINDOW 16000000 +#endif + +/* Max window size (in ns) = 1s */ +#define MAX_SCHED_RAVG_WINDOW 1000000000 + +#define NR_WINDOWS_PER_SEC (NSEC_PER_SEC / DEFAULT_SCHED_RAVG_WINDOW) + +extern int num_sched_clusters; + +extern unsigned int walt_big_tasks(int cpu); +extern void reset_task_stats(struct task_struct *p); +extern void walt_rotate_work_init(void); +extern void walt_rotation_checkpoint(int nr_big); +extern void walt_fill_ta_data(struct core_ctl_notif_data *data); +extern int sched_set_group_id(struct task_struct *p, unsigned int group_id); +extern unsigned int sched_get_group_id(struct task_struct *p); +extern int sched_set_init_task_load(struct task_struct *p, int init_load_pct); +extern u32 sched_get_init_task_load(struct task_struct *p); +extern void core_ctl_check(u64 wallclock); +extern int sched_set_boost(int enable); +extern int sched_isolate_count(const cpumask_t *mask, bool include_offline); + +extern struct list_head cluster_head; +#define for_each_sched_cluster(cluster) \ + list_for_each_entry_rcu(cluster, &cluster_head, list) + +static inline u32 cpu_cycles_to_freq(u64 cycles, u64 period) +{ + return div64_u64(cycles, period); +} + +static inline unsigned int sched_cpu_legacy_freq(int cpu) +{ + unsigned long curr_cap = arch_scale_freq_capacity(cpu); + + return (curr_cap * (u64) cpu_rq(cpu)->wrq.cluster->max_possible_freq) >> + SCHED_CAPACITY_SHIFT; +} + +extern __read_mostly bool sched_freq_aggr_en; +static inline void walt_enable_frequency_aggregation(bool enable) +{ + sched_freq_aggr_en = enable; +} + +#ifndef CONFIG_IRQ_TIME_ACCOUNTING +static inline u64 irq_time_read(int cpu) { return 0; } +#endif + +#else +static inline unsigned int walt_big_tasks(int cpu) +{ + return 0; +} + +static inline int sched_set_boost(int enable) +{ + return -EINVAL; +} +#endif diff --git a/kernel/sched/walt/sched_avg.c b/kernel/sched/walt/sched_avg.c new file mode 100644 index 000000000000..4a5efa4118b3 --- /dev/null +++ b/kernel/sched/walt/sched_avg.c @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2012, 2015-2021, The Linux Foundation. All rights reserved. + */ +/* + * Scheduler hook for average runqueue determination + */ +#include +#include +#include +#include +#include + +#include "qc_vas.h" +#include + +static DEFINE_PER_CPU(u64, nr_prod_sum); +static DEFINE_PER_CPU(u64, last_time); +static DEFINE_PER_CPU(u64, nr_big_prod_sum); +static DEFINE_PER_CPU(u64, nr); +static DEFINE_PER_CPU(u64, nr_max); + +static DEFINE_PER_CPU(spinlock_t, nr_lock) = __SPIN_LOCK_UNLOCKED(nr_lock); +static s64 last_get_time; + +unsigned int sysctl_sched_busy_hyst_enable_cpus; +unsigned int sysctl_sched_busy_hyst; +unsigned int sysctl_sched_coloc_busy_hyst_enable_cpus = 112; +unsigned int sysctl_sched_coloc_busy_hyst_cpu[NR_CPUS] = { + [0 ... NR_CPUS-1] = 39000000 }; +unsigned int sysctl_sched_coloc_busy_hyst_max_ms = 5000; +unsigned int sysctl_sched_coloc_busy_hyst_cpu_busy_pct[NR_CPUS] = { + [0 ... NR_CPUS-1] = 10 }; +static DEFINE_PER_CPU(atomic64_t, busy_hyst_end_time) = ATOMIC64_INIT(0); + +static DEFINE_PER_CPU(u64, hyst_time); +static DEFINE_PER_CPU(u64, coloc_hyst_busy); +static DEFINE_PER_CPU(u64, coloc_hyst_time); + +#define NR_THRESHOLD_PCT 15 +#define MAX_RTGB_TIME (sysctl_sched_coloc_busy_hyst_max_ms * NSEC_PER_MSEC) + +/** + * sched_get_nr_running_avg + * @return: Average nr_running, iowait and nr_big_tasks value since last poll. + * Returns the avg * 100 to return up to two decimal points + * of accuracy. + * + * Obtains the average nr_running value since the last poll. + * This function may not be called concurrently with itself + */ +void sched_get_nr_running_avg(struct sched_avg_stats *stats) +{ + int cpu; + u64 curr_time = sched_clock(); + u64 period = curr_time - last_get_time; + u64 tmp_nr, tmp_misfit; + bool any_hyst_time = false; + + if (!period) + return; + + /* read and reset nr_running counts */ + for_each_possible_cpu(cpu) { + unsigned long flags; + u64 diff; + + spin_lock_irqsave(&per_cpu(nr_lock, cpu), flags); + curr_time = sched_clock(); + diff = curr_time - per_cpu(last_time, cpu); + BUG_ON((s64)diff < 0); + + tmp_nr = per_cpu(nr_prod_sum, cpu); + tmp_nr += per_cpu(nr, cpu) * diff; + tmp_nr = div64_u64((tmp_nr * 100), period); + + tmp_misfit = per_cpu(nr_big_prod_sum, cpu); + tmp_misfit += walt_big_tasks(cpu) * diff; + tmp_misfit = div64_u64((tmp_misfit * 100), period); + + /* + * NR_THRESHOLD_PCT is to make sure that the task ran + * at least 85% in the last window to compensate any + * over estimating being done. + */ + stats[cpu].nr = (int)div64_u64((tmp_nr + NR_THRESHOLD_PCT), + 100); + stats[cpu].nr_misfit = (int)div64_u64((tmp_misfit + + NR_THRESHOLD_PCT), 100); + stats[cpu].nr_max = per_cpu(nr_max, cpu); + stats[cpu].nr_scaled = tmp_nr; + + trace_sched_get_nr_running_avg(cpu, stats[cpu].nr, + stats[cpu].nr_misfit, stats[cpu].nr_max, + stats[cpu].nr_scaled); + + per_cpu(last_time, cpu) = curr_time; + per_cpu(nr_prod_sum, cpu) = 0; + per_cpu(nr_big_prod_sum, cpu) = 0; + per_cpu(nr_max, cpu) = per_cpu(nr, cpu); + + spin_unlock_irqrestore(&per_cpu(nr_lock, cpu), flags); + } + + for_each_possible_cpu(cpu) { + if (per_cpu(coloc_hyst_time, cpu)) { + any_hyst_time = true; + break; + } + } + if (any_hyst_time && get_rtgb_active_time() >= MAX_RTGB_TIME) + sched_update_hyst_times(); + + last_get_time = curr_time; + +} +EXPORT_SYMBOL(sched_get_nr_running_avg); + +void sched_update_hyst_times(void) +{ + bool rtgb_active; + int cpu; + unsigned long cpu_cap, coloc_busy_pct; + + rtgb_active = is_rtgb_active() && (sched_boost() != CONSERVATIVE_BOOST) + && (get_rtgb_active_time() < MAX_RTGB_TIME); + + for_each_possible_cpu(cpu) { + cpu_cap = arch_scale_cpu_capacity(cpu); + coloc_busy_pct = sysctl_sched_coloc_busy_hyst_cpu_busy_pct[cpu]; + per_cpu(hyst_time, cpu) = (BIT(cpu) + & sysctl_sched_busy_hyst_enable_cpus) ? + sysctl_sched_busy_hyst : 0; + per_cpu(coloc_hyst_time, cpu) = ((BIT(cpu) + & sysctl_sched_coloc_busy_hyst_enable_cpus) + && rtgb_active) ? + sysctl_sched_coloc_busy_hyst_cpu[cpu] : 0; + per_cpu(coloc_hyst_busy, cpu) = mult_frac(cpu_cap, + coloc_busy_pct, 100); + } +} + +#define BUSY_NR_RUN 3 +#define BUSY_LOAD_FACTOR 10 +static inline void update_busy_hyst_end_time(int cpu, bool dequeue, + unsigned long prev_nr_run, u64 curr_time) +{ + bool nr_run_trigger = false; + bool load_trigger = false, coloc_load_trigger = false; + u64 agg_hyst_time; + + if (!per_cpu(hyst_time, cpu) && !per_cpu(coloc_hyst_time, cpu)) + return; + + if (prev_nr_run >= BUSY_NR_RUN && per_cpu(nr, cpu) < BUSY_NR_RUN) + nr_run_trigger = true; + + if (dequeue && (cpu_util(cpu) * BUSY_LOAD_FACTOR) > + capacity_orig_of(cpu)) + load_trigger = true; + + if (dequeue && cpu_util(cpu) > per_cpu(coloc_hyst_busy, cpu)) + coloc_load_trigger = true; + + agg_hyst_time = max((nr_run_trigger || load_trigger) ? + per_cpu(hyst_time, cpu) : 0, + (nr_run_trigger || coloc_load_trigger) ? + per_cpu(coloc_hyst_time, cpu) : 0); + + if (agg_hyst_time) + atomic64_set(&per_cpu(busy_hyst_end_time, cpu), + curr_time + agg_hyst_time); +} + +int sched_busy_hyst_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, loff_t *ppos) +{ + int ret; + + if (table->maxlen > (sizeof(unsigned int) * num_possible_cpus())) + table->maxlen = sizeof(unsigned int) * num_possible_cpus(); + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + + if (!ret && write) + sched_update_hyst_times(); + + return ret; +} + +/** + * sched_update_nr_prod + * @cpu: The core id of the nr running driver. + * @delta: Adjust nr by 'delta' amount + * @inc: Whether we are increasing or decreasing the count + * @return: N/A + * + * Update average with latest nr_running value for CPU + */ +void sched_update_nr_prod(int cpu, long delta, bool inc) +{ + u64 diff; + u64 curr_time; + unsigned long flags, nr_running; + + spin_lock_irqsave(&per_cpu(nr_lock, cpu), flags); + nr_running = per_cpu(nr, cpu); + curr_time = sched_clock(); + diff = curr_time - per_cpu(last_time, cpu); + BUG_ON((s64)diff < 0); + per_cpu(last_time, cpu) = curr_time; + per_cpu(nr, cpu) = nr_running + (inc ? delta : -delta); + + BUG_ON((s64)per_cpu(nr, cpu) < 0); + + if (per_cpu(nr, cpu) > per_cpu(nr_max, cpu)) + per_cpu(nr_max, cpu) = per_cpu(nr, cpu); + + update_busy_hyst_end_time(cpu, !inc, nr_running, curr_time); + + per_cpu(nr_prod_sum, cpu) += nr_running * diff; + per_cpu(nr_big_prod_sum, cpu) += walt_big_tasks(cpu) * diff; + spin_unlock_irqrestore(&per_cpu(nr_lock, cpu), flags); +} +EXPORT_SYMBOL(sched_update_nr_prod); + +/* + * Returns the CPU utilization % in the last window. + * + */ +unsigned int sched_get_cpu_util(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + u64 util; + unsigned long capacity, flags; + unsigned int busy; + + raw_spin_lock_irqsave(&rq->lock, flags); + + capacity = capacity_orig_of(cpu); + + util = rq->wrq.prev_runnable_sum + rq->wrq.grp_time.prev_runnable_sum; + util = div64_u64(util, sched_ravg_window >> SCHED_CAPACITY_SHIFT); + raw_spin_unlock_irqrestore(&rq->lock, flags); + + util = (util >= capacity) ? capacity : util; + busy = div64_ul((util * 100), capacity); + return busy; +} + +u64 sched_lpm_disallowed_time(int cpu) +{ + u64 now = sched_clock(); + u64 bias_end_time = atomic64_read(&per_cpu(busy_hyst_end_time, cpu)); + + if (now < bias_end_time) + return bias_end_time - now; + + return 0; +} diff --git a/kernel/sched/walt/trace.c b/kernel/sched/walt/trace.c new file mode 100644 index 000000000000..7c06ce1fda64 --- /dev/null +++ b/kernel/sched/walt/trace.c @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#include "qc_vas.h" + +#ifdef CONFIG_SCHED_WALT +static inline void __window_data(u32 *dst, u32 *src) +{ + if (src) + memcpy(dst, src, nr_cpu_ids * sizeof(u32)); + else + memset(dst, 0, nr_cpu_ids * sizeof(u32)); +} + +struct trace_seq; +const char *__window_print(struct trace_seq *p, const u32 *buf, int buf_len) +{ + int i; + const char *ret = p->buffer + seq_buf_used(&p->seq); + + for (i = 0; i < buf_len; i++) + trace_seq_printf(p, "%u ", buf[i]); + + trace_seq_putc(p, 0); + + return ret; +} + +static inline s64 __rq_update_sum(struct rq *rq, bool curr, bool new) +{ + if (curr) + if (new) + return rq->wrq.nt_curr_runnable_sum; + else + return rq->wrq.curr_runnable_sum; + else + if (new) + return rq->wrq.nt_prev_runnable_sum; + else + return rq->wrq.prev_runnable_sum; +} + +static inline s64 __grp_update_sum(struct rq *rq, bool curr, bool new) +{ + if (curr) + if (new) + return rq->wrq.grp_time.nt_curr_runnable_sum; + else + return rq->wrq.grp_time.curr_runnable_sum; + else + if (new) + return rq->wrq.grp_time.nt_prev_runnable_sum; + else + return rq->wrq.grp_time.prev_runnable_sum; +} + +static inline s64 +__get_update_sum(struct rq *rq, enum migrate_types migrate_type, + bool src, bool new, bool curr) +{ + switch (migrate_type) { + case RQ_TO_GROUP: + if (src) + return __rq_update_sum(rq, curr, new); + else + return __grp_update_sum(rq, curr, new); + case GROUP_TO_RQ: + if (src) + return __grp_update_sum(rq, curr, new); + else + return __rq_update_sum(rq, curr, new); + default: + WARN_ON_ONCE(1); + return -1; + } +} +#endif +#define CREATE_TRACE_POINTS +#include "trace.h" + diff --git a/kernel/sched/walt/trace.h b/kernel/sched/walt/trace.h new file mode 100644 index 000000000000..4b3c111c313a --- /dev/null +++ b/kernel/sched/walt/trace.h @@ -0,0 +1,669 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * Copyright (c) 2019-2021, The Linux Foundation. All rights reserved. + */ + +#undef TRACE_SYSTEM +#define TRACE_SYSTEM sched + +#if !defined(_TRACE_WALT_H) || defined(TRACE_HEADER_MULTI_READ) +#define _TRACE_WALT_H + +#include + +#ifdef CONFIG_SCHED_WALT +struct rq; +struct group_cpu_time; +extern const char __weak *task_event_names[]; + +TRACE_EVENT(sched_update_pred_demand, + + TP_PROTO(struct task_struct *p, u32 runtime, int pct, + unsigned int pred_demand), + + TP_ARGS(p, runtime, pct, pred_demand), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(unsigned int, runtime) + __field(int, pct) + __field(unsigned int, pred_demand) + __array(u8, bucket, NUM_BUSY_BUCKETS) + __field(int, cpu) + ), + + TP_fast_assign( + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->runtime = runtime; + __entry->pct = pct; + __entry->pred_demand = pred_demand; + memcpy(__entry->bucket, p->wts.busy_buckets, + NUM_BUSY_BUCKETS * sizeof(u8)); + __entry->cpu = task_cpu(p); + ), + + TP_printk("%d (%s): runtime %u pct %d cpu %d pred_demand %u (buckets: %u %u %u %u %u %u %u %u %u %u)", + __entry->pid, __entry->comm, + __entry->runtime, __entry->pct, __entry->cpu, + __entry->pred_demand, __entry->bucket[0], __entry->bucket[1], + __entry->bucket[2], __entry->bucket[3], __entry->bucket[4], + __entry->bucket[5], __entry->bucket[6], __entry->bucket[7], + __entry->bucket[8], __entry->bucket[9]) +); + +TRACE_EVENT(sched_update_history, + + TP_PROTO(struct rq *rq, struct task_struct *p, u32 runtime, int samples, + enum task_event evt), + + TP_ARGS(rq, p, runtime, samples, evt), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(unsigned int, runtime) + __field(int, samples) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(unsigned int, coloc_demand) + __field(unsigned int, pred_demand) + __array(u32, hist, RAVG_HIST_SIZE_MAX) + __field(unsigned int, nr_big_tasks) + __field(int, cpu) + ), + + TP_fast_assign( + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->runtime = runtime; + __entry->samples = samples; + __entry->evt = evt; + __entry->demand = p->wts.demand; + __entry->coloc_demand = p->wts.coloc_demand; + __entry->pred_demand = p->wts.pred_demand; + memcpy(__entry->hist, p->wts.sum_history, + RAVG_HIST_SIZE_MAX * sizeof(u32)); + __entry->nr_big_tasks = rq->wrq.walt_stats.nr_big_tasks; + __entry->cpu = rq->cpu; + ), + + TP_printk("%d (%s): runtime %u samples %d event %s demand %u coloc_demand %u pred_demand %u (hist: %u %u %u %u %u) cpu %d nr_big %u", + __entry->pid, __entry->comm, + __entry->runtime, __entry->samples, + task_event_names[__entry->evt], + __entry->demand, __entry->coloc_demand, __entry->pred_demand, + __entry->hist[0], __entry->hist[1], + __entry->hist[2], __entry->hist[3], + __entry->hist[4], __entry->cpu, __entry->nr_big_tasks) +); + +TRACE_EVENT(sched_get_task_cpu_cycles, + + TP_PROTO(int cpu, int event, u64 cycles, + u64 exec_time, struct task_struct *p), + + TP_ARGS(cpu, event, cycles, exec_time, p), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, event) + __field(u64, cycles) + __field(u64, exec_time) + __field(u32, freq) + __field(u32, legacy_freq) + __field(u32, max_freq) + __field(pid_t, pid) + __array(char, comm, TASK_COMM_LEN) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->event = event; + __entry->cycles = cycles; + __entry->exec_time = exec_time; + __entry->freq = cpu_cycles_to_freq(cycles, exec_time); + __entry->legacy_freq = sched_cpu_legacy_freq(cpu); + __entry->max_freq = cpu_max_freq(cpu); + __entry->pid = p->pid; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + ), + + TP_printk("cpu=%d event=%d cycles=%llu exec_time=%llu freq=%u legacy_freq=%u max_freq=%u task=%d (%s)", + __entry->cpu, __entry->event, __entry->cycles, + __entry->exec_time, __entry->freq, __entry->legacy_freq, + __entry->max_freq, __entry->pid, __entry->comm) +); + +TRACE_EVENT(sched_update_task_ravg, + + TP_PROTO(struct task_struct *p, struct rq *rq, enum task_event evt, + u64 wallclock, u64 irqtime, + struct group_cpu_time *cpu_time), + + TP_ARGS(p, rq, evt, wallclock, irqtime, cpu_time), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(pid_t, cur_pid) + __field(unsigned int, cur_freq) + __field(u64, wallclock) + __field(u64, mark_start) + __field(u64, delta_m) + __field(u64, win_start) + __field(u64, delta) + __field(u64, irqtime) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(unsigned int, coloc_demand) + __field(unsigned int, sum) + __field(int, cpu) + __field(unsigned int, pred_demand) + __field(u64, rq_cs) + __field(u64, rq_ps) + __field(u64, grp_cs) + __field(u64, grp_ps) + __field(u64, grp_nt_cs) + __field(u64, grp_nt_ps) + __field(u32, curr_window) + __field(u32, prev_window) + __dynamic_array(u32, curr_sum, nr_cpu_ids) + __dynamic_array(u32, prev_sum, nr_cpu_ids) + __field(u64, nt_cs) + __field(u64, nt_ps) + __field(u64, active_time) + __field(u32, curr_top) + __field(u32, prev_top) + ), + + TP_fast_assign( + __entry->wallclock = wallclock; + __entry->win_start = rq->wrq.window_start; + __entry->delta = (wallclock - rq->wrq.window_start); + __entry->evt = evt; + __entry->cpu = rq->cpu; + __entry->cur_pid = rq->curr->pid; + __entry->cur_freq = rq->wrq.task_exec_scale; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->mark_start = p->wts.mark_start; + __entry->delta_m = (wallclock - p->wts.mark_start); + __entry->demand = p->wts.demand; + __entry->coloc_demand = p->wts.coloc_demand; + __entry->sum = p->wts.sum; + __entry->irqtime = irqtime; + __entry->pred_demand = p->wts.pred_demand; + __entry->rq_cs = rq->wrq.curr_runnable_sum; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_cs = cpu_time ? cpu_time->curr_runnable_sum : 0; + __entry->grp_ps = cpu_time ? cpu_time->prev_runnable_sum : 0; + __entry->grp_nt_cs = cpu_time ? + cpu_time->nt_curr_runnable_sum : 0; + __entry->grp_nt_ps = cpu_time ? + cpu_time->nt_prev_runnable_sum : 0; + __entry->curr_window = p->wts.curr_window; + __entry->prev_window = p->wts.prev_window; + __window_data(__get_dynamic_array(curr_sum), + p->wts.curr_window_cpu); + __window_data(__get_dynamic_array(prev_sum), + p->wts.prev_window_cpu); + __entry->nt_cs = rq->wrq.nt_curr_runnable_sum; + __entry->nt_ps = rq->wrq.nt_prev_runnable_sum; + __entry->active_time = p->wts.active_time; + __entry->curr_top = rq->wrq.curr_top; + __entry->prev_top = rq->wrq.prev_top; + ), + + TP_printk("wc %llu ws %llu delta %llu event %s cpu %d cur_freq %u cur_pid %d task %d (%s) ms %llu delta %llu demand %u coloc_demand: %u sum %u irqtime %llu pred_demand %u rq_cs %llu rq_ps %llu cur_window %u (%s) prev_window %u (%s) nt_cs %llu nt_ps %llu active_time %u grp_cs %lld grp_ps %lld, grp_nt_cs %llu, grp_nt_ps: %llu curr_top %u prev_top %u", + __entry->wallclock, __entry->win_start, __entry->delta, + task_event_names[__entry->evt], __entry->cpu, + __entry->cur_freq, __entry->cur_pid, + __entry->pid, __entry->comm, __entry->mark_start, + __entry->delta_m, __entry->demand, __entry->coloc_demand, + __entry->sum, __entry->irqtime, __entry->pred_demand, + __entry->rq_cs, __entry->rq_ps, __entry->curr_window, + __window_print(p, __get_dynamic_array(curr_sum), nr_cpu_ids), + __entry->prev_window, + __window_print(p, __get_dynamic_array(prev_sum), nr_cpu_ids), + __entry->nt_cs, __entry->nt_ps, + __entry->active_time, __entry->grp_cs, + __entry->grp_ps, __entry->grp_nt_cs, __entry->grp_nt_ps, + __entry->curr_top, __entry->prev_top) +); + +TRACE_EVENT(sched_update_task_ravg_mini, + + TP_PROTO(struct task_struct *p, struct rq *rq, enum task_event evt, + u64 wallclock, u64 irqtime, + struct group_cpu_time *cpu_time), + + TP_ARGS(p, rq, evt, wallclock, irqtime, cpu_time), + + TP_STRUCT__entry( + __array(char, comm, TASK_COMM_LEN) + __field(pid_t, pid) + __field(u64, wallclock) + __field(u64, mark_start) + __field(u64, delta_m) + __field(u64, win_start) + __field(u64, delta) + __field(enum task_event, evt) + __field(unsigned int, demand) + __field(int, cpu) + __field(u64, rq_cs) + __field(u64, rq_ps) + __field(u64, grp_cs) + __field(u64, grp_ps) + __field(u32, curr_window) + __field(u32, prev_window) + ), + + TP_fast_assign( + __entry->wallclock = wallclock; + __entry->win_start = rq->wrq.window_start; + __entry->delta = (wallclock - rq->wrq.window_start); + __entry->evt = evt; + __entry->cpu = rq->cpu; + memcpy(__entry->comm, p->comm, TASK_COMM_LEN); + __entry->pid = p->pid; + __entry->mark_start = p->wts.mark_start; + __entry->delta_m = (wallclock - p->wts.mark_start); + __entry->demand = p->wts.demand; + __entry->rq_cs = rq->wrq.curr_runnable_sum; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_cs = cpu_time ? cpu_time->curr_runnable_sum : 0; + __entry->grp_ps = cpu_time ? cpu_time->prev_runnable_sum : 0; + __entry->curr_window = p->wts.curr_window; + __entry->prev_window = p->wts.prev_window; + ), + + TP_printk("wc %llu ws %llu delta %llu event %s cpu %d task %d (%s) ms %llu delta %llu demand %u rq_cs %llu rq_ps %llu cur_window %u prev_window %u grp_cs %lld grp_ps %lld", + __entry->wallclock, __entry->win_start, __entry->delta, + task_event_names[__entry->evt], __entry->cpu, + __entry->pid, __entry->comm, __entry->mark_start, + __entry->delta_m, __entry->demand, + __entry->rq_cs, __entry->rq_ps, __entry->curr_window, + __entry->prev_window, __entry->grp_cs, __entry->grp_ps) +); + +struct migration_sum_data; +extern const char __weak *migrate_type_names[]; + +TRACE_EVENT(sched_set_preferred_cluster, + + TP_PROTO(struct walt_related_thread_group *grp, u64 total_demand), + + TP_ARGS(grp, total_demand), + + TP_STRUCT__entry( + __field(int, id) + __field(u64, total_demand) + __field(bool, skip_min) + ), + + TP_fast_assign( + __entry->id = grp->id; + __entry->total_demand = total_demand; + __entry->skip_min = grp->skip_min; + ), + + TP_printk("group_id %d total_demand %llu skip_min %d", + __entry->id, __entry->total_demand, + __entry->skip_min) +); + +TRACE_EVENT(sched_migration_update_sum, + + TP_PROTO(struct task_struct *p, enum migrate_types migrate_type, + struct rq *rq), + + TP_ARGS(p, migrate_type, rq), + + TP_STRUCT__entry( + __field(int, tcpu) + __field(int, pid) + __field(enum migrate_types, migrate_type) + __field(s64, src_cs) + __field(s64, src_ps) + __field(s64, dst_cs) + __field(s64, dst_ps) + __field(s64, src_nt_cs) + __field(s64, src_nt_ps) + __field(s64, dst_nt_cs) + __field(s64, dst_nt_ps) + ), + + TP_fast_assign( + __entry->tcpu = task_cpu(p); + __entry->pid = p->pid; + __entry->migrate_type = migrate_type; + __entry->src_cs = __get_update_sum(rq, migrate_type, + true, false, true); + __entry->src_ps = __get_update_sum(rq, migrate_type, + true, false, false); + __entry->dst_cs = __get_update_sum(rq, migrate_type, + false, false, true); + __entry->dst_ps = __get_update_sum(rq, migrate_type, + false, false, false); + __entry->src_nt_cs = __get_update_sum(rq, migrate_type, + true, true, true); + __entry->src_nt_ps = __get_update_sum(rq, migrate_type, + true, true, false); + __entry->dst_nt_cs = __get_update_sum(rq, migrate_type, + false, true, true); + __entry->dst_nt_ps = __get_update_sum(rq, migrate_type, + false, true, false); + ), + + TP_printk("pid %d task_cpu %d migrate_type %s src_cs %llu src_ps %llu dst_cs %lld dst_ps %lld src_nt_cs %llu src_nt_ps %llu dst_nt_cs %lld dst_nt_ps %lld", + __entry->pid, __entry->tcpu, + migrate_type_names[__entry->migrate_type], + __entry->src_cs, __entry->src_ps, __entry->dst_cs, + __entry->dst_ps, __entry->src_nt_cs, __entry->src_nt_ps, + __entry->dst_nt_cs, __entry->dst_nt_ps) +); + +TRACE_EVENT(sched_set_boost, + + TP_PROTO(int type), + + TP_ARGS(type), + + TP_STRUCT__entry( + __field(int, type) + ), + + TP_fast_assign( + __entry->type = type; + ), + + TP_printk("type %d", __entry->type) +); + +TRACE_EVENT(sched_load_to_gov, + + TP_PROTO(struct rq *rq, u64 aggr_grp_load, u32 tt_load, + int freq_aggr, u64 load, int policy, + int big_task_rotation, + unsigned int user_hint), + TP_ARGS(rq, aggr_grp_load, tt_load, freq_aggr, load, policy, + big_task_rotation, user_hint), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, policy) + __field(int, ed_task_pid) + __field(u64, aggr_grp_load) + __field(int, freq_aggr) + __field(u64, tt_load) + __field(u64, rq_ps) + __field(u64, grp_rq_ps) + __field(u64, nt_ps) + __field(u64, grp_nt_ps) + __field(u64, pl) + __field(u64, load) + __field(int, big_task_rotation) + __field(unsigned int, user_hint) + ), + + TP_fast_assign( + __entry->cpu = cpu_of(rq); + __entry->policy = policy; + __entry->ed_task_pid = + rq->wrq.ed_task ? rq->wrq.ed_task->pid : -1; + __entry->aggr_grp_load = aggr_grp_load; + __entry->freq_aggr = freq_aggr; + __entry->tt_load = tt_load; + __entry->rq_ps = rq->wrq.prev_runnable_sum; + __entry->grp_rq_ps = rq->wrq.grp_time.prev_runnable_sum; + __entry->nt_ps = rq->wrq.nt_prev_runnable_sum; + __entry->grp_nt_ps = rq->wrq.grp_time.nt_prev_runnable_sum; + __entry->pl = + rq->wrq.walt_stats.pred_demands_sum_scaled; + __entry->load = load; + __entry->big_task_rotation = big_task_rotation; + __entry->user_hint = user_hint; + ), + + TP_printk("cpu=%d policy=%d ed_task_pid=%d aggr_grp_load=%llu freq_aggr=%d tt_load=%llu rq_ps=%llu grp_rq_ps=%llu nt_ps=%llu grp_nt_ps=%llu pl=%llu load=%llu big_task_rotation=%d user_hint=%u", + __entry->cpu, __entry->policy, __entry->ed_task_pid, + __entry->aggr_grp_load, __entry->freq_aggr, + __entry->tt_load, __entry->rq_ps, __entry->grp_rq_ps, + __entry->nt_ps, __entry->grp_nt_ps, __entry->pl, __entry->load, + __entry->big_task_rotation, __entry->user_hint) +); + +TRACE_EVENT(core_ctl_eval_need, + + TP_PROTO(unsigned int cpu, unsigned int old_need, + unsigned int new_need, unsigned int updated), + TP_ARGS(cpu, old_need, new_need, updated), + TP_STRUCT__entry( + __field(u32, cpu) + __field(u32, old_need) + __field(u32, new_need) + __field(u32, updated) + ), + TP_fast_assign( + __entry->cpu = cpu; + __entry->old_need = old_need; + __entry->new_need = new_need; + __entry->updated = updated; + ), + TP_printk("cpu=%u, old_need=%u, new_need=%u, updated=%u", __entry->cpu, + __entry->old_need, __entry->new_need, __entry->updated) +); + +TRACE_EVENT(core_ctl_set_busy, + + TP_PROTO(unsigned int cpu, unsigned int busy, + unsigned int old_is_busy, unsigned int is_busy), + TP_ARGS(cpu, busy, old_is_busy, is_busy), + TP_STRUCT__entry( + __field(u32, cpu) + __field(u32, busy) + __field(u32, old_is_busy) + __field(u32, is_busy) + __field(bool, high_irqload) + ), + TP_fast_assign( + __entry->cpu = cpu; + __entry->busy = busy; + __entry->old_is_busy = old_is_busy; + __entry->is_busy = is_busy; + __entry->high_irqload = sched_cpu_high_irqload(cpu); + ), + TP_printk("cpu=%u, busy=%u, old_is_busy=%u, new_is_busy=%u high_irqload=%d", + __entry->cpu, __entry->busy, __entry->old_is_busy, + __entry->is_busy, __entry->high_irqload) +); + +TRACE_EVENT(core_ctl_set_boost, + + TP_PROTO(u32 refcount, s32 ret), + TP_ARGS(refcount, ret), + TP_STRUCT__entry( + __field(u32, refcount) + __field(s32, ret) + ), + TP_fast_assign( + __entry->refcount = refcount; + __entry->ret = ret; + ), + TP_printk("refcount=%u, ret=%d", __entry->refcount, __entry->ret) +); + +TRACE_EVENT(core_ctl_update_nr_need, + + TP_PROTO(int cpu, int nr_need, int prev_misfit_need, + int nrrun, int max_nr, int nr_prev_assist), + + TP_ARGS(cpu, nr_need, prev_misfit_need, nrrun, max_nr, nr_prev_assist), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, nr_need) + __field(int, prev_misfit_need) + __field(int, nrrun) + __field(int, max_nr) + __field(int, nr_prev_assist) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->nr_need = nr_need; + __entry->prev_misfit_need = prev_misfit_need; + __entry->nrrun = nrrun; + __entry->max_nr = max_nr; + __entry->nr_prev_assist = nr_prev_assist; + ), + + TP_printk("cpu=%d nr_need=%d prev_misfit_need=%d nrrun=%d max_nr=%d nr_prev_assist=%d", + __entry->cpu, __entry->nr_need, __entry->prev_misfit_need, + __entry->nrrun, __entry->max_nr, __entry->nr_prev_assist) +); + +TRACE_EVENT(core_ctl_notif_data, + + TP_PROTO(u32 nr_big, u32 ta_load, u32 *ta_util, u32 *cur_cap), + + TP_ARGS(nr_big, ta_load, ta_util, cur_cap), + + TP_STRUCT__entry( + __field(u32, nr_big) + __field(u32, ta_load) + __array(u32, ta_util, MAX_CLUSTERS) + __array(u32, cur_cap, MAX_CLUSTERS) + ), + + TP_fast_assign( + __entry->nr_big = nr_big; + __entry->ta_load = ta_load; + memcpy(__entry->ta_util, ta_util, MAX_CLUSTERS * sizeof(u32)); + memcpy(__entry->cur_cap, cur_cap, MAX_CLUSTERS * sizeof(u32)); + ), + + TP_printk("nr_big=%u ta_load=%u ta_util=(%u %u %u) cur_cap=(%u %u %u)", + __entry->nr_big, __entry->ta_load, + __entry->ta_util[0], __entry->ta_util[1], + __entry->ta_util[2], __entry->cur_cap[0], + __entry->cur_cap[1], __entry->cur_cap[2]) +); + +/* + * Tracepoint for sched_get_nr_running_avg + */ +TRACE_EVENT(sched_get_nr_running_avg, + + TP_PROTO(int cpu, int nr, int nr_misfit, int nr_max, int nr_scaled), + + TP_ARGS(cpu, nr, nr_misfit, nr_max, nr_scaled), + + TP_STRUCT__entry( + __field(int, cpu) + __field(int, nr) + __field(int, nr_misfit) + __field(int, nr_max) + __field(int, nr_scaled) + ), + + TP_fast_assign( + __entry->cpu = cpu; + __entry->nr = nr; + __entry->nr_misfit = nr_misfit; + __entry->nr_max = nr_max; + __entry->nr_scaled = nr_scaled; + ), + + TP_printk("cpu=%d nr=%d nr_misfit=%d nr_max=%d nr_scaled=%d", + __entry->cpu, __entry->nr, __entry->nr_misfit, __entry->nr_max, + __entry->nr_scaled) +); + +/* + * sched_isolate - called when cores are isolated/unisolated + * + * @acutal_mask: mask of cores actually isolated/unisolated + * @req_mask: mask of cores requested isolated/unisolated + * @online_mask: cpu online mask + * @time: amount of time in us it took to isolate/unisolate + * @isolate: 1 if isolating, 0 if unisolating + * + */ +TRACE_EVENT(sched_isolate, + + TP_PROTO(unsigned int requested_cpu, unsigned int isolated_cpus, + u64 start_time, unsigned char isolate), + + TP_ARGS(requested_cpu, isolated_cpus, start_time, isolate), + + TP_STRUCT__entry( + __field(u32, requested_cpu) + __field(u32, isolated_cpus) + __field(u32, time) + __field(unsigned char, isolate) + ), + + TP_fast_assign( + __entry->requested_cpu = requested_cpu; + __entry->isolated_cpus = isolated_cpus; + __entry->time = div64_u64(sched_clock() - start_time, 1000); + __entry->isolate = isolate; + ), + + TP_printk("iso cpu=%u cpus=0x%x time=%u us isolated=%d", + __entry->requested_cpu, __entry->isolated_cpus, + __entry->time, __entry->isolate) +); + +TRACE_EVENT(sched_ravg_window_change, + + TP_PROTO(unsigned int sched_ravg_window, unsigned int new_sched_ravg_window + , u64 change_time), + + TP_ARGS(sched_ravg_window, new_sched_ravg_window, change_time), + + TP_STRUCT__entry( + __field(unsigned int, sched_ravg_window) + __field(unsigned int, new_sched_ravg_window) + __field(u64, change_time) + ), + + TP_fast_assign( + __entry->sched_ravg_window = sched_ravg_window; + __entry->new_sched_ravg_window = new_sched_ravg_window; + __entry->change_time = change_time; + ), + + TP_printk("from=%u to=%u at=%lu", + __entry->sched_ravg_window, __entry->new_sched_ravg_window, + __entry->change_time) +); + +TRACE_EVENT(walt_window_rollover, + + TP_PROTO(u64 window_start), + + TP_ARGS(window_start), + + TP_STRUCT__entry( + __field(u64, window_start) + ), + + TP_fast_assign( + __entry->window_start = window_start; + ), + + TP_printk("window_start=%llu", __entry->window_start) +); + +#endif /* CONFIG_SCHED_WALT */ +#endif /* _TRACE_WALT_H */ + +#undef TRACE_INCLUDE_PATH +#define TRACE_INCLUDE_PATH . +#define TRACE_INCLUDE_FILE trace + +#include diff --git a/kernel/sched/walt/walt.c b/kernel/sched/walt/walt.c new file mode 100644 index 000000000000..92a3d27b3895 --- /dev/null +++ b/kernel/sched/walt/walt.c @@ -0,0 +1,3793 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2016-2021, The Linux Foundation. All rights reserved. + */ +#include +#include +#include +#include +#include +#include +#include "qc_vas.h" + +#include + +const char *task_event_names[] = {"PUT_PREV_TASK", "PICK_NEXT_TASK", + "TASK_WAKE", "TASK_MIGRATE", "TASK_UPDATE", + "IRQ_UPDATE"}; + +const char *migrate_type_names[] = {"GROUP_TO_RQ", "RQ_TO_GROUP", + "RQ_TO_RQ", "GROUP_TO_GROUP"}; + +#define SCHED_FREQ_ACCOUNT_WAIT_TIME 0 +#define SCHED_ACCOUNT_WAIT_TIME 1 + +#define EARLY_DETECTION_DURATION 9500000 +#define MAX_NUM_CGROUP_COLOC_ID 20 + +#define WINDOW_STATS_RECENT 0 +#define WINDOW_STATS_MAX 1 +#define WINDOW_STATS_MAX_RECENT_AVG 2 +#define WINDOW_STATS_AVG 3 +#define WINDOW_STATS_INVALID_POLICY 4 + +#define MAX_NR_CLUSTERS 3 + +#define FREQ_REPORT_MAX_CPU_LOAD_TOP_TASK 0 +#define FREQ_REPORT_CPU_LOAD 1 +#define FREQ_REPORT_TOP_TASK 2 + +#define NEW_TASK_ACTIVE_TIME 100000000 + +static ktime_t ktime_last; +static bool sched_ktime_suspended; +static struct cpu_cycle_counter_cb cpu_cycle_counter_cb; +static bool use_cycle_counter; +static DEFINE_MUTEX(cluster_lock); +static atomic64_t walt_irq_work_lastq_ws; +static u64 walt_load_reported_window; + +static struct irq_work walt_cpufreq_irq_work; +static struct irq_work walt_migration_irq_work; + +u64 sched_ktime_clock(void) +{ + if (unlikely(sched_ktime_suspended)) + return ktime_to_ns(ktime_last); + return ktime_get_ns(); +} + +static void sched_resume(void) +{ + sched_ktime_suspended = false; +} + +static int sched_suspend(void) +{ + ktime_last = ktime_get(); + sched_ktime_suspended = true; + return 0; +} + +static struct syscore_ops sched_syscore_ops = { + .resume = sched_resume, + .suspend = sched_suspend +}; + +static int __init sched_init_ops(void) +{ + register_syscore_ops(&sched_syscore_ops); + return 0; +} +late_initcall(sched_init_ops); + +static void acquire_rq_locks_irqsave(const cpumask_t *cpus, + unsigned long *flags) +{ + int cpu; + int level = 0; + + local_irq_save(*flags); + for_each_cpu(cpu, cpus) { + if (level == 0) + raw_spin_lock(&cpu_rq(cpu)->lock); + else + raw_spin_lock_nested(&cpu_rq(cpu)->lock, level); + level++; + } +} + +static void release_rq_locks_irqrestore(const cpumask_t *cpus, + unsigned long *flags) +{ + int cpu; + + for_each_cpu(cpu, cpus) + raw_spin_unlock(&cpu_rq(cpu)->lock); + local_irq_restore(*flags); +} + +unsigned int sysctl_sched_capacity_margin_up[MAX_MARGIN_LEVELS] = { + [0 ... MAX_MARGIN_LEVELS-1] = 1078}; /* ~5% margin */ +unsigned int sysctl_sched_capacity_margin_down[MAX_MARGIN_LEVELS] = { + [0 ... MAX_MARGIN_LEVELS-1] = 1205}; /* ~15% margin */ +static unsigned int walt_cpu_high_irqload; + +unsigned int sysctl_sched_walt_rotate_big_tasks; +unsigned int walt_rotation_enabled; + +__read_mostly unsigned int sysctl_sched_asym_cap_sibling_freq_match_pct = 100; +__read_mostly unsigned int sched_ravg_hist_size = 5; + +static __read_mostly unsigned int sched_io_is_busy = 1; + +__read_mostly unsigned int sysctl_sched_window_stats_policy = + WINDOW_STATS_MAX_RECENT_AVG; + +unsigned int sysctl_sched_ravg_window_nr_ticks = (HZ / NR_WINDOWS_PER_SEC); + +unsigned int sysctl_sched_dynamic_ravg_window_enable = (HZ == 250); + +/* Window size (in ns) */ +__read_mostly unsigned int sched_ravg_window = DEFAULT_SCHED_RAVG_WINDOW; +__read_mostly unsigned int new_sched_ravg_window = DEFAULT_SCHED_RAVG_WINDOW; + +static DEFINE_SPINLOCK(sched_ravg_window_lock); +u64 sched_ravg_window_change_time; + +/* + * A after-boot constant divisor for cpu_util_freq_walt() to apply the load + * boost. + */ +static __read_mostly unsigned int walt_cpu_util_freq_divisor; + +/* Initial task load. Newly created tasks are assigned this load. */ +unsigned int __read_mostly sched_init_task_load_windows; +unsigned int __read_mostly sched_init_task_load_windows_scaled; +unsigned int __read_mostly sysctl_sched_init_task_load_pct = 15; + +unsigned int max_possible_capacity = 1024; /* max(rq->max_possible_capacity) */ +unsigned int +min_max_possible_capacity = 1024; /* min(rq->max_possible_capacity) */ + +/* + * Task load is categorized into buckets for the purpose of top task tracking. + * The entire range of load from 0 to sched_ravg_window needs to be covered + * in NUM_LOAD_INDICES number of buckets. Therefore the size of each bucket + * is given by sched_ravg_window / NUM_LOAD_INDICES. Since the default value + * of sched_ravg_window is DEFAULT_SCHED_RAVG_WINDOW, use that to compute + * sched_load_granule. + */ +__read_mostly unsigned int sched_load_granule = + DEFAULT_SCHED_RAVG_WINDOW / NUM_LOAD_INDICES; +/* Size of bitmaps maintained to track top tasks */ +static const unsigned int top_tasks_bitmap_size = + BITS_TO_LONGS(NUM_LOAD_INDICES + 1) * sizeof(unsigned long); + +/* + * This governs what load needs to be used when reporting CPU busy time + * to the cpufreq governor. + */ +__read_mostly unsigned int sysctl_sched_freq_reporting_policy; + +static int __init set_sched_ravg_window(char *str) +{ + unsigned int window_size; + + get_option(&str, &window_size); + + if (window_size < DEFAULT_SCHED_RAVG_WINDOW || + window_size > MAX_SCHED_RAVG_WINDOW) { + WARN_ON(1); + return -EINVAL; + } + + sched_ravg_window = window_size; + return 0; +} + +early_param("sched_ravg_window", set_sched_ravg_window); + +static int __init set_sched_predl(char *str) +{ + unsigned int predl; + + get_option(&str, &predl); + sched_predl = !!predl; + return 0; +} +early_param("sched_predl", set_sched_predl); + +__read_mostly unsigned int walt_scale_demand_divisor; +#define scale_demand(d) ((d)/walt_scale_demand_divisor) + +#define SCHED_PRINT(arg) printk_deferred("%s=%llu", #arg, arg) +#define STRG(arg) #arg + +static inline void walt_task_dump(struct task_struct *p) +{ + char buff[NR_CPUS * 16]; + int i, j = 0; + int buffsz = NR_CPUS * 16; + + SCHED_PRINT(p->pid); + SCHED_PRINT(p->wts.mark_start); + SCHED_PRINT(p->wts.demand); + SCHED_PRINT(p->wts.coloc_demand); + SCHED_PRINT(sched_ravg_window); + SCHED_PRINT(new_sched_ravg_window); + + for (i = 0 ; i < nr_cpu_ids; i++) + j += scnprintf(buff + j, buffsz - j, "%u ", + p->wts.curr_window_cpu[i]); + printk_deferred("%s=%d (%s)\n", STRG(p->wts.curr_window), + p->wts.curr_window, buff); + + for (i = 0, j = 0 ; i < nr_cpu_ids; i++) + j += scnprintf(buff + j, buffsz - j, "%u ", + p->wts.prev_window_cpu[i]); + printk_deferred("%s=%d (%s)\n", STRG(p->wts.prev_window), + p->wts.prev_window, buff); + + SCHED_PRINT(p->wts.last_wake_ts); + SCHED_PRINT(p->wts.last_enqueued_ts); + SCHED_PRINT(p->wts.misfit); + SCHED_PRINT(p->wts.unfilter); +} + +static inline void walt_rq_dump(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + struct task_struct *tsk = cpu_curr(cpu); + int i; + + /* + * Increment the task reference so that it can't be + * freed on a remote CPU. Since we are going to + * enter panic, there is no need to decrement the + * task reference. Decrementing the task reference + * can't be done in atomic context, especially with + * rq locks held. + */ + get_task_struct(tsk); + printk_deferred("CPU:%d nr_running:%u current: %d (%s)\n", + cpu, rq->nr_running, tsk->pid, tsk->comm); + + printk_deferred("=========================================="); + SCHED_PRINT(rq->wrq.window_start); + SCHED_PRINT(rq->wrq.prev_window_size); + SCHED_PRINT(rq->wrq.curr_runnable_sum); + SCHED_PRINT(rq->wrq.prev_runnable_sum); + SCHED_PRINT(rq->wrq.nt_curr_runnable_sum); + SCHED_PRINT(rq->wrq.nt_prev_runnable_sum); + SCHED_PRINT(rq->wrq.cum_window_demand_scaled); + SCHED_PRINT(rq->wrq.task_exec_scale); + SCHED_PRINT(rq->wrq.grp_time.curr_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.prev_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.nt_curr_runnable_sum); + SCHED_PRINT(rq->wrq.grp_time.nt_prev_runnable_sum); + for (i = 0 ; i < NUM_TRACKED_WINDOWS; i++) { + printk_deferred("rq->wrq.load_subs[%d].window_start=%llu)\n", i, + rq->wrq.load_subs[i].window_start); + printk_deferred("rq->wrq.load_subs[%d].subs=%llu)\n", i, + rq->wrq.load_subs[i].subs); + printk_deferred("rq->wrq.load_subs[%d].new_subs=%llu)\n", i, + rq->wrq.load_subs[i].new_subs); + } + walt_task_dump(tsk); + SCHED_PRINT(sched_capacity_margin_up[cpu]); + SCHED_PRINT(sched_capacity_margin_down[cpu]); +} + +static inline void walt_dump(void) +{ + int cpu; + + printk_deferred("============ WALT RQ DUMP START ==============\n"); + printk_deferred("Sched ktime_get: %llu\n", sched_ktime_clock()); + printk_deferred("Time last window changed=%lu\n", + sched_ravg_window_change_time); + for_each_online_cpu(cpu) + walt_rq_dump(cpu); + SCHED_PRINT(max_possible_capacity); + SCHED_PRINT(min_max_possible_capacity); + printk_deferred("============ WALT RQ DUMP END ==============\n"); +} + +static int in_sched_bug; +#define SCHED_BUG_ON(condition) \ +({ \ + if (unlikely(!!(condition)) && !in_sched_bug) { \ + in_sched_bug = 1; \ + walt_dump(); \ + BUG_ON(condition); \ + } \ +}) + +static void fixup_walt_sched_stats_common(struct rq *rq, struct task_struct *p, + u16 updated_demand_scaled, + u16 updated_pred_demand_scaled) +{ + s64 task_load_delta = (s64)updated_demand_scaled - + p->wts.demand_scaled; + s64 pred_demand_delta = (s64)updated_pred_demand_scaled - + p->wts.pred_demand_scaled; + + fixup_cumulative_runnable_avg(&rq->wrq.walt_stats, task_load_delta, + pred_demand_delta); + + walt_fixup_cum_window_demand(rq, task_load_delta); +} + +/* + * Demand aggregation for frequency purpose: + * + * CPU demand of tasks from various related groups is aggregated per-cluster and + * added to the "max_busy_cpu" in that cluster, where max_busy_cpu is determined + * by just rq->wrq.prev_runnable_sum. + * + * Some examples follow, which assume: + * Cluster0 = CPU0-3, Cluster1 = CPU4-7 + * One related thread group A that has tasks A0, A1, A2 + * + * A->cpu_time[X].curr/prev_sum = counters in which cpu execution stats of + * tasks belonging to group A are accumulated when they run on cpu X. + * + * CX->curr/prev_sum = counters in which cpu execution stats of all tasks + * not belonging to group A are accumulated when they run on cpu X + * + * Lets say the stats for window M was as below: + * + * C0->prev_sum = 1ms, A->cpu_time[0].prev_sum = 5ms + * Task A0 ran 5ms on CPU0 + * Task B0 ran 1ms on CPU0 + * + * C1->prev_sum = 5ms, A->cpu_time[1].prev_sum = 6ms + * Task A1 ran 4ms on CPU1 + * Task A2 ran 2ms on CPU1 + * Task B1 ran 5ms on CPU1 + * + * C2->prev_sum = 0ms, A->cpu_time[2].prev_sum = 0 + * CPU2 idle + * + * C3->prev_sum = 0ms, A->cpu_time[3].prev_sum = 0 + * CPU3 idle + * + * In this case, CPU1 was most busy going by just its prev_sum counter. Demand + * from all group A tasks are added to CPU1. IOW, at end of window M, cpu busy + * time reported to governor will be: + * + * + * C0 busy time = 1ms + * C1 busy time = 5 + 5 + 6 = 16ms + * + */ +__read_mostly bool sched_freq_aggr_en; + +static u64 +update_window_start(struct rq *rq, u64 wallclock, int event) +{ + s64 delta; + int nr_windows; + u64 old_window_start = rq->wrq.window_start; + + delta = wallclock - rq->wrq.window_start; + if (delta < 0) { + printk_deferred("WALT-BUG CPU%d; wallclock=%llu is lesser than window_start=%llu", + rq->cpu, wallclock, rq->wrq.window_start); + SCHED_BUG_ON(1); + } + if (delta < sched_ravg_window) + return old_window_start; + + nr_windows = div64_u64(delta, sched_ravg_window); + rq->wrq.window_start += (u64)nr_windows * (u64)sched_ravg_window; + + rq->wrq.cum_window_demand_scaled = + rq->wrq.walt_stats.cumulative_runnable_avg_scaled; + rq->wrq.prev_window_size = sched_ravg_window; + + return old_window_start; +} + +/* + * Assumes rq_lock is held and wallclock was recorded in the same critical + * section as this function's invocation. + */ +static inline u64 read_cycle_counter(int cpu, u64 wallclock) +{ + struct rq *rq = cpu_rq(cpu); + + if (rq->wrq.last_cc_update != wallclock) { + rq->wrq.cycles = + cpu_cycle_counter_cb.get_cpu_cycle_counter(cpu); + rq->wrq.last_cc_update = wallclock; + } + + return rq->wrq.cycles; +} + +static void update_task_cpu_cycles(struct task_struct *p, int cpu, + u64 wallclock) +{ + if (use_cycle_counter) + p->wts.cpu_cycles = read_cycle_counter(cpu, wallclock); +} + +static inline bool is_ed_enabled(void) +{ + return (walt_rotation_enabled || (sched_boost_policy() != + SCHED_BOOST_NONE)); +} + +void clear_ed_task(struct task_struct *p, struct rq *rq) +{ + if (p == rq->wrq.ed_task) + rq->wrq.ed_task = NULL; +} + +static inline bool is_ed_task(struct task_struct *p, u64 wallclock) +{ + return (wallclock - p->wts.last_wake_ts >= EARLY_DETECTION_DURATION); +} + +bool early_detection_notify(struct rq *rq, u64 wallclock) +{ + struct task_struct *p; + int loop_max = 10; + + rq->wrq.ed_task = NULL; + + if (!is_ed_enabled() || !rq->cfs.h_nr_running) + return 0; + + list_for_each_entry(p, &rq->cfs_tasks, se.group_node) { + if (!loop_max) + break; + + if (is_ed_task(p, wallclock)) { + rq->wrq.ed_task = p; + return 1; + } + + loop_max--; + } + + return 0; +} + +void walt_sched_account_irqstart(int cpu, struct task_struct *curr) +{ + struct rq *rq = cpu_rq(cpu); + + if (!rq->wrq.window_start) + return; + + /* We're here without rq->lock held, IRQ disabled */ + raw_spin_lock(&rq->lock); + update_task_cpu_cycles(curr, cpu, sched_ktime_clock()); + raw_spin_unlock(&rq->lock); +} + +void walt_sched_account_irqend(int cpu, struct task_struct *curr, u64 delta) +{ + struct rq *rq = cpu_rq(cpu); + unsigned long flags; + + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_task_ravg(curr, rq, IRQ_UPDATE, sched_ktime_clock(), delta); + raw_spin_unlock_irqrestore(&rq->lock, flags); +} + +/* + * Return total number of tasks "eligible" to run on higher capacity cpus + */ +unsigned int walt_big_tasks(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + + return rq->wrq.walt_stats.nr_big_tasks; +} + +void clear_walt_request(int cpu) +{ + struct rq *rq = cpu_rq(cpu); + unsigned long flags; + + clear_reserved(cpu); + if (rq->wrq.push_task) { + struct task_struct *push_task = NULL; + + raw_spin_lock_irqsave(&rq->lock, flags); + if (rq->wrq.push_task) { + clear_reserved(rq->push_cpu); + push_task = rq->wrq.push_task; + rq->wrq.push_task = NULL; + } + rq->active_balance = 0; + raw_spin_unlock_irqrestore(&rq->lock, flags); + if (push_task) + put_task_struct(push_task); + } +} + +/* + * Special case the last index and provide a fast path for index = 0. + * Note that sched_load_granule can change underneath us if we are not + * holding any runqueue locks while calling the two functions below. + */ +static u32 top_task_load(struct rq *rq) +{ + int index = rq->wrq.prev_top; + u8 prev = 1 - rq->wrq.curr_table; + + if (!index) { + int msb = NUM_LOAD_INDICES - 1; + + if (!test_bit(msb, rq->wrq.top_tasks_bitmap[prev])) + return 0; + else + return sched_load_granule; + } else if (index == NUM_LOAD_INDICES - 1) { + return sched_ravg_window; + } else { + return (index + 1) * sched_load_granule; + } +} + +unsigned int sysctl_sched_user_hint; +static unsigned long sched_user_hint_reset_time; +static bool is_cluster_hosting_top_app(struct walt_sched_cluster *cluster); + +static inline bool +should_apply_suh_freq_boost(struct walt_sched_cluster *cluster) +{ + if (sched_freq_aggr_en || !sysctl_sched_user_hint || + !cluster->aggr_grp_load) + return false; + + return is_cluster_hosting_top_app(cluster); +} + +static inline u64 freq_policy_load(struct rq *rq) +{ + unsigned int reporting_policy = sysctl_sched_freq_reporting_policy; + struct walt_sched_cluster *cluster = rq->wrq.cluster; + u64 aggr_grp_load = cluster->aggr_grp_load; + u64 load, tt_load = 0; + struct task_struct *cpu_ksoftirqd = per_cpu(ksoftirqd, cpu_of(rq)); + + if (rq->wrq.ed_task != NULL) { + load = sched_ravg_window; + goto done; + } + + if (sched_freq_aggr_en) + load = rq->wrq.prev_runnable_sum + aggr_grp_load; + else + load = rq->wrq.prev_runnable_sum + + rq->wrq.grp_time.prev_runnable_sum; + + if (cpu_ksoftirqd && cpu_ksoftirqd->state == TASK_RUNNING) + load = max_t(u64, load, task_load(cpu_ksoftirqd)); + + tt_load = top_task_load(rq); + switch (reporting_policy) { + case FREQ_REPORT_MAX_CPU_LOAD_TOP_TASK: + load = max_t(u64, load, tt_load); + break; + case FREQ_REPORT_TOP_TASK: + load = tt_load; + break; + case FREQ_REPORT_CPU_LOAD: + break; + default: + break; + } + + if (should_apply_suh_freq_boost(cluster)) { + if (is_suh_max()) + load = sched_ravg_window; + else + load = div64_u64(load * sysctl_sched_user_hint, + (u64)100); + } + +done: + trace_sched_load_to_gov(rq, aggr_grp_load, tt_load, sched_freq_aggr_en, + load, reporting_policy, walt_rotation_enabled, + sysctl_sched_user_hint); + return load; +} + +static bool rtgb_active; + +static inline unsigned long +__cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) +{ + u64 util, util_unboosted; + struct rq *rq = cpu_rq(cpu); + unsigned long capacity = capacity_orig_of(cpu); + int boost; + + boost = per_cpu(sched_load_boost, cpu); + util_unboosted = util = freq_policy_load(rq); + util = div64_u64(util * (100 + boost), + walt_cpu_util_freq_divisor); + + if (walt_load) { + u64 nl = cpu_rq(cpu)->wrq.nt_prev_runnable_sum + + rq->wrq.grp_time.nt_prev_runnable_sum; + u64 pl = rq->wrq.walt_stats.pred_demands_sum_scaled; + + /* do_pl_notif() needs unboosted signals */ + rq->wrq.old_busy_time = div64_u64(util_unboosted, + sched_ravg_window >> + SCHED_CAPACITY_SHIFT); + rq->wrq.old_estimated_time = pl; + + nl = div64_u64(nl * (100 + boost), walt_cpu_util_freq_divisor); + + walt_load->nl = nl; + walt_load->pl = pl; + walt_load->ws = walt_load_reported_window; + walt_load->rtgb_active = rtgb_active; + } + + return (util >= capacity) ? capacity : util; +} + +#define ADJUSTED_ASYM_CAP_CPU_UTIL(orig, other, x) \ + (max(orig, mult_frac(other, x, 100))) + +unsigned long +cpu_util_freq_walt(int cpu, struct walt_cpu_load *walt_load) +{ + struct walt_cpu_load wl_other = {0}; + unsigned long util = 0, util_other = 0; + unsigned long capacity = capacity_orig_of(cpu); + int i, mpct = sysctl_sched_asym_cap_sibling_freq_match_pct; + + if (!cpumask_test_cpu(cpu, &asym_cap_sibling_cpus)) + return __cpu_util_freq_walt(cpu, walt_load); + + for_each_cpu(i, &asym_cap_sibling_cpus) { + if (i == cpu) + util = __cpu_util_freq_walt(cpu, walt_load); + else + util_other = __cpu_util_freq_walt(i, &wl_other); + } + + if (cpu == cpumask_last(&asym_cap_sibling_cpus)) + mpct = 100; + + util = ADJUSTED_ASYM_CAP_CPU_UTIL(util, util_other, mpct); + + walt_load->nl = ADJUSTED_ASYM_CAP_CPU_UTIL(walt_load->nl, wl_other.nl, + mpct); + walt_load->pl = ADJUSTED_ASYM_CAP_CPU_UTIL(walt_load->pl, wl_other.pl, + mpct); + + return (util >= capacity) ? capacity : util; +} + +/* + * In this function we match the accumulated subtractions with the current + * and previous windows we are operating with. Ignore any entries where + * the window start in the load_subtraction struct does not match either + * the curent or the previous window. This could happen whenever CPUs + * become idle or busy with interrupts disabled for an extended period. + */ +static inline void account_load_subtractions(struct rq *rq) +{ + u64 ws = rq->wrq.window_start; + u64 prev_ws = ws - rq->wrq.prev_window_size; + struct load_subtractions *ls = rq->wrq.load_subs; + int i; + + for (i = 0; i < NUM_TRACKED_WINDOWS; i++) { + if (ls[i].window_start == ws) { + rq->wrq.curr_runnable_sum -= ls[i].subs; + rq->wrq.nt_curr_runnable_sum -= ls[i].new_subs; + } else if (ls[i].window_start == prev_ws) { + rq->wrq.prev_runnable_sum -= ls[i].subs; + rq->wrq.nt_prev_runnable_sum -= ls[i].new_subs; + } + + ls[i].subs = 0; + ls[i].new_subs = 0; + } + + SCHED_BUG_ON((s64)rq->wrq.prev_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.curr_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.nt_prev_runnable_sum < 0); + SCHED_BUG_ON((s64)rq->wrq.nt_curr_runnable_sum < 0); +} + +static inline void create_subtraction_entry(struct rq *rq, u64 ws, int index) +{ + rq->wrq.load_subs[index].window_start = ws; + rq->wrq.load_subs[index].subs = 0; + rq->wrq.load_subs[index].new_subs = 0; +} + +static int get_top_index(unsigned long *bitmap, unsigned long old_top) +{ + int index = find_next_bit(bitmap, NUM_LOAD_INDICES, old_top); + + if (index == NUM_LOAD_INDICES) + return 0; + + return NUM_LOAD_INDICES - 1 - index; +} + +static bool get_subtraction_index(struct rq *rq, u64 ws) +{ + int i; + u64 oldest = ULLONG_MAX; + int oldest_index = 0; + + for (i = 0; i < NUM_TRACKED_WINDOWS; i++) { + u64 entry_ws = rq->wrq.load_subs[i].window_start; + + if (ws == entry_ws) + return i; + + if (entry_ws < oldest) { + oldest = entry_ws; + oldest_index = i; + } + } + + create_subtraction_entry(rq, ws, oldest_index); + return oldest_index; +} + +static void update_rq_load_subtractions(int index, struct rq *rq, + u32 sub_load, bool new_task) +{ + rq->wrq.load_subs[index].subs += sub_load; + if (new_task) + rq->wrq.load_subs[index].new_subs += sub_load; +} + +static inline struct walt_sched_cluster *cpu_cluster(int cpu) +{ + return cpu_rq(cpu)->wrq.cluster; +} + +void update_cluster_load_subtractions(struct task_struct *p, + int cpu, u64 ws, bool new_task) +{ + struct walt_sched_cluster *cluster = cpu_cluster(cpu); + struct cpumask cluster_cpus = cluster->cpus; + u64 prev_ws = ws - cpu_rq(cpu)->wrq.prev_window_size; + int i; + + cpumask_clear_cpu(cpu, &cluster_cpus); + raw_spin_lock(&cluster->load_lock); + + for_each_cpu(i, &cluster_cpus) { + struct rq *rq = cpu_rq(i); + int index; + + if (p->wts.curr_window_cpu[i]) { + index = get_subtraction_index(rq, ws); + update_rq_load_subtractions(index, rq, + p->wts.curr_window_cpu[i], new_task); + p->wts.curr_window_cpu[i] = 0; + } + + if (p->wts.prev_window_cpu[i]) { + index = get_subtraction_index(rq, prev_ws); + update_rq_load_subtractions(index, rq, + p->wts.prev_window_cpu[i], new_task); + p->wts.prev_window_cpu[i] = 0; + } + } + + raw_spin_unlock(&cluster->load_lock); +} + +static inline void inter_cluster_migration_fixup + (struct task_struct *p, int new_cpu, int task_cpu, bool new_task) +{ + struct rq *dest_rq = cpu_rq(new_cpu); + struct rq *src_rq = cpu_rq(task_cpu); + + if (same_freq_domain(new_cpu, task_cpu)) + return; + + p->wts.curr_window_cpu[new_cpu] = p->wts.curr_window; + p->wts.prev_window_cpu[new_cpu] = p->wts.prev_window; + + dest_rq->wrq.curr_runnable_sum += p->wts.curr_window; + dest_rq->wrq.prev_runnable_sum += p->wts.prev_window; + + if (src_rq->wrq.curr_runnable_sum < p->wts.curr_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.curr_runnable_sum, + p->wts.curr_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.curr_runnable_sum -= p->wts.curr_window_cpu[task_cpu]; + + if (src_rq->wrq.prev_runnable_sum < p->wts.prev_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.prev_runnable_sum, + p->wts.prev_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.prev_runnable_sum -= p->wts.prev_window_cpu[task_cpu]; + + if (new_task) { + dest_rq->wrq.nt_curr_runnable_sum += p->wts.curr_window; + dest_rq->wrq.nt_prev_runnable_sum += p->wts.prev_window; + + if (src_rq->wrq.nt_curr_runnable_sum < + p->wts.curr_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.nt_curr_runnable_sum, + p->wts.curr_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.nt_curr_runnable_sum -= + p->wts.curr_window_cpu[task_cpu]; + + if (src_rq->wrq.nt_prev_runnable_sum < + p->wts.prev_window_cpu[task_cpu]) { + printk_deferred("WALT-BUG pid=%u CPU%d -> CPU%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, src_rq->cpu, dest_rq->cpu, + src_rq->wrq.nt_prev_runnable_sum, + p->wts.prev_window_cpu[task_cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + src_rq->wrq.nt_prev_runnable_sum -= + p->wts.prev_window_cpu[task_cpu]; + } + + p->wts.curr_window_cpu[task_cpu] = 0; + p->wts.prev_window_cpu[task_cpu] = 0; + + update_cluster_load_subtractions(p, task_cpu, + src_rq->wrq.window_start, new_task); +} + +static u32 load_to_index(u32 load) +{ + u32 index = load / sched_load_granule; + + return min(index, (u32)(NUM_LOAD_INDICES - 1)); +} + +static void +migrate_top_tasks(struct task_struct *p, struct rq *src_rq, struct rq *dst_rq) +{ + int index; + int top_index; + u32 curr_window = p->wts.curr_window; + u32 prev_window = p->wts.prev_window; + u8 src = src_rq->wrq.curr_table; + u8 dst = dst_rq->wrq.curr_table; + u8 *src_table; + u8 *dst_table; + + if (curr_window) { + src_table = src_rq->wrq.top_tasks[src]; + dst_table = dst_rq->wrq.top_tasks[dst]; + index = load_to_index(curr_window); + src_table[index] -= 1; + dst_table[index] += 1; + + if (!src_table[index]) + __clear_bit(NUM_LOAD_INDICES - index - 1, + src_rq->wrq.top_tasks_bitmap[src]); + + if (dst_table[index] == 1) + __set_bit(NUM_LOAD_INDICES - index - 1, + dst_rq->wrq.top_tasks_bitmap[dst]); + + if (index > dst_rq->wrq.curr_top) + dst_rq->wrq.curr_top = index; + + top_index = src_rq->wrq.curr_top; + if (index == top_index && !src_table[index]) + src_rq->wrq.curr_top = get_top_index( + src_rq->wrq.top_tasks_bitmap[src], top_index); + } + + if (prev_window) { + src = 1 - src; + dst = 1 - dst; + src_table = src_rq->wrq.top_tasks[src]; + dst_table = dst_rq->wrq.top_tasks[dst]; + index = load_to_index(prev_window); + src_table[index] -= 1; + dst_table[index] += 1; + + if (!src_table[index]) + __clear_bit(NUM_LOAD_INDICES - index - 1, + src_rq->wrq.top_tasks_bitmap[src]); + + if (dst_table[index] == 1) + __set_bit(NUM_LOAD_INDICES - index - 1, + dst_rq->wrq.top_tasks_bitmap[dst]); + + if (index > dst_rq->wrq.prev_top) + dst_rq->wrq.prev_top = index; + + top_index = src_rq->wrq.prev_top; + if (index == top_index && !src_table[index]) + src_rq->wrq.prev_top = get_top_index( + src_rq->wrq.top_tasks_bitmap[src], top_index); + } +} + +static inline bool is_new_task(struct task_struct *p) +{ + return p->wts.active_time < NEW_TASK_ACTIVE_TIME; +} + +void fixup_busy_time(struct task_struct *p, int new_cpu) +{ + struct rq *src_rq = task_rq(p); + struct rq *dest_rq = cpu_rq(new_cpu); + u64 wallclock; + u64 *src_curr_runnable_sum, *dst_curr_runnable_sum; + u64 *src_prev_runnable_sum, *dst_prev_runnable_sum; + u64 *src_nt_curr_runnable_sum, *dst_nt_curr_runnable_sum; + u64 *src_nt_prev_runnable_sum, *dst_nt_prev_runnable_sum; + bool new_task; + struct walt_related_thread_group *grp; + long pstate; + + if (!p->on_rq && p->state != TASK_WAKING) + return; + + pstate = p->state; + + if (pstate == TASK_WAKING) + double_rq_lock(src_rq, dest_rq); + + wallclock = sched_ktime_clock(); + + walt_update_task_ravg(task_rq(p)->curr, task_rq(p), + TASK_UPDATE, + wallclock, 0); + walt_update_task_ravg(dest_rq->curr, dest_rq, + TASK_UPDATE, wallclock, 0); + + walt_update_task_ravg(p, task_rq(p), TASK_MIGRATE, + wallclock, 0); + + update_task_cpu_cycles(p, new_cpu, wallclock); + + /* + * When a task is migrating during the wakeup, adjust + * the task's contribution towards cumulative window + * demand. + */ + if (pstate == TASK_WAKING && p->wts.last_sleep_ts >= + src_rq->wrq.window_start) { + walt_fixup_cum_window_demand(src_rq, + -(s64)p->wts.demand_scaled); + walt_fixup_cum_window_demand(dest_rq, p->wts.demand_scaled); + } + + new_task = is_new_task(p); + /* Protected by rq_lock */ + grp = p->wts.grp; + + /* + * For frequency aggregation, we continue to do migration fixups + * even for intra cluster migrations. This is because, the aggregated + * load has to reported on a single CPU regardless. + */ + if (grp) { + struct group_cpu_time *cpu_time; + + cpu_time = &src_rq->wrq.grp_time; + src_curr_runnable_sum = &cpu_time->curr_runnable_sum; + src_prev_runnable_sum = &cpu_time->prev_runnable_sum; + src_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + cpu_time = &dest_rq->wrq.grp_time; + dst_curr_runnable_sum = &cpu_time->curr_runnable_sum; + dst_prev_runnable_sum = &cpu_time->prev_runnable_sum; + dst_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + dst_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + if (p->wts.curr_window) { + *src_curr_runnable_sum -= p->wts.curr_window; + *dst_curr_runnable_sum += p->wts.curr_window; + if (new_task) { + *src_nt_curr_runnable_sum -= + p->wts.curr_window; + *dst_nt_curr_runnable_sum += + p->wts.curr_window; + } + } + + if (p->wts.prev_window) { + *src_prev_runnable_sum -= p->wts.prev_window; + *dst_prev_runnable_sum += p->wts.prev_window; + if (new_task) { + *src_nt_prev_runnable_sum -= + p->wts.prev_window; + *dst_nt_prev_runnable_sum += + p->wts.prev_window; + } + } + } else { + inter_cluster_migration_fixup(p, new_cpu, + task_cpu(p), new_task); + } + + migrate_top_tasks(p, src_rq, dest_rq); + + if (!same_freq_domain(new_cpu, task_cpu(p))) { + src_rq->wrq.notif_pending = true; + dest_rq->wrq.notif_pending = true; + walt_irq_work_queue(&walt_migration_irq_work); + } + + if (is_ed_enabled()) { + if (p == src_rq->wrq.ed_task) { + src_rq->wrq.ed_task = NULL; + dest_rq->wrq.ed_task = p; + } else if (is_ed_task(p, wallclock)) { + dest_rq->wrq.ed_task = p; + } + } + + if (pstate == TASK_WAKING) + double_rq_unlock(src_rq, dest_rq); +} + +void set_window_start(struct rq *rq) +{ + static int sync_cpu_available; + + if (likely(rq->wrq.window_start)) + return; + + if (!sync_cpu_available) { + rq->wrq.window_start = 1; + sync_cpu_available = 1; + atomic64_set(&walt_irq_work_lastq_ws, rq->wrq.window_start); + walt_load_reported_window = + atomic64_read(&walt_irq_work_lastq_ws); + + } else { + struct rq *sync_rq = cpu_rq(cpumask_any(cpu_online_mask)); + + raw_spin_unlock(&rq->lock); + double_rq_lock(rq, sync_rq); + rq->wrq.window_start = sync_rq->wrq.window_start; + rq->wrq.curr_runnable_sum = rq->wrq.prev_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = rq->wrq.nt_prev_runnable_sum = 0; + raw_spin_unlock(&sync_rq->lock); + } + + rq->curr->wts.mark_start = rq->wrq.window_start; +} + +unsigned int sysctl_sched_conservative_pl; +unsigned int sysctl_sched_many_wakeup_threshold = WALT_MANY_WAKEUP_DEFAULT; + +#define INC_STEP 8 +#define DEC_STEP 2 +#define CONSISTENT_THRES 16 +#define INC_STEP_BIG 16 +/* + * bucket_increase - update the count of all buckets + * + * @buckets: array of buckets tracking busy time of a task + * @idx: the index of bucket to be incremented + * + * Each time a complete window finishes, count of bucket that runtime + * falls in (@idx) is incremented. Counts of all other buckets are + * decayed. The rate of increase and decay could be different based + * on current count in the bucket. + */ +static inline void bucket_increase(u8 *buckets, int idx) +{ + int i, step; + + for (i = 0; i < NUM_BUSY_BUCKETS; i++) { + if (idx != i) { + if (buckets[i] > DEC_STEP) + buckets[i] -= DEC_STEP; + else + buckets[i] = 0; + } else { + step = buckets[i] >= CONSISTENT_THRES ? + INC_STEP_BIG : INC_STEP; + if (buckets[i] > U8_MAX - step) + buckets[i] = U8_MAX; + else + buckets[i] += step; + } + } +} + +static inline int busy_to_bucket(u32 normalized_rt) +{ + int bidx; + + bidx = mult_frac(normalized_rt, NUM_BUSY_BUCKETS, max_task_load()); + bidx = min(bidx, NUM_BUSY_BUCKETS - 1); + + /* + * Combine lowest two buckets. The lowest frequency falls into + * 2nd bucket and thus keep predicting lowest bucket is not + * useful. + */ + if (!bidx) + bidx++; + + return bidx; +} + +/* + * get_pred_busy - calculate predicted demand for a task on runqueue + * + * @p: task whose prediction is being updated + * @start: starting bucket. returned prediction should not be lower than + * this bucket. + * @runtime: runtime of the task. returned prediction should not be lower + * than this runtime. + * Note: @start can be derived from @runtime. It's passed in only to + * avoid duplicated calculation in some cases. + * + * A new predicted busy time is returned for task @p based on @runtime + * passed in. The function searches through buckets that represent busy + * time equal to or bigger than @runtime and attempts to find the bucket to + * to use for prediction. Once found, it searches through historical busy + * time and returns the latest that falls into the bucket. If no such busy + * time exists, it returns the medium of that bucket. + */ +static u32 get_pred_busy(struct task_struct *p, + int start, u32 runtime) +{ + int i; + u8 *buckets = p->wts.busy_buckets; + u32 *hist = p->wts.sum_history; + u32 dmin, dmax; + u64 cur_freq_runtime = 0; + int first = NUM_BUSY_BUCKETS, final; + u32 ret = runtime; + + /* skip prediction for new tasks due to lack of history */ + if (unlikely(is_new_task(p))) + goto out; + + /* find minimal bucket index to pick */ + for (i = start; i < NUM_BUSY_BUCKETS; i++) { + if (buckets[i]) { + first = i; + break; + } + } + /* if no higher buckets are filled, predict runtime */ + if (first >= NUM_BUSY_BUCKETS) + goto out; + + /* compute the bucket for prediction */ + final = first; + + /* determine demand range for the predicted bucket */ + if (final < 2) { + /* lowest two buckets are combined */ + dmin = 0; + final = 1; + } else { + dmin = mult_frac(final, max_task_load(), NUM_BUSY_BUCKETS); + } + dmax = mult_frac(final + 1, max_task_load(), NUM_BUSY_BUCKETS); + + /* + * search through runtime history and return first runtime that falls + * into the range of predicted bucket. + */ + for (i = 0; i < sched_ravg_hist_size; i++) { + if (hist[i] >= dmin && hist[i] < dmax) { + ret = hist[i]; + break; + } + } + /* no historical runtime within bucket found, use average of the bin */ + if (ret < dmin) + ret = (dmin + dmax) / 2; + /* + * when updating in middle of a window, runtime could be higher + * than all recorded history. Always predict at least runtime. + */ + ret = max(runtime, ret); +out: + trace_sched_update_pred_demand(p, runtime, + mult_frac((unsigned int)cur_freq_runtime, 100, + sched_ravg_window), ret); + return ret; +} + +static inline u32 calc_pred_demand(struct task_struct *p) +{ + if (p->wts.pred_demand >= p->wts.curr_window) + return p->wts.pred_demand; + + return get_pred_busy(p, busy_to_bucket(p->wts.curr_window), + p->wts.curr_window); +} + +/* + * predictive demand of a task is calculated at the window roll-over. + * if the task current window busy time exceeds the predicted + * demand, update it here to reflect the task needs. + */ +void update_task_pred_demand(struct rq *rq, struct task_struct *p, int event) +{ + u32 new, old; + u16 new_scaled; + + if (!sched_predl) + return; + + if (is_idle_task(p)) + return; + + if (event != PUT_PREV_TASK && event != TASK_UPDATE && + (!SCHED_FREQ_ACCOUNT_WAIT_TIME || + (event != TASK_MIGRATE && + event != PICK_NEXT_TASK))) + return; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (!p->on_rq && !SCHED_FREQ_ACCOUNT_WAIT_TIME) + return; + } + + new = calc_pred_demand(p); + old = p->wts.pred_demand; + + if (old >= new) + return; + + new_scaled = scale_demand(new); + if (task_on_rq_queued(p) && (!task_has_dl_policy(p) || + !p->dl.dl_throttled)) + fixup_walt_sched_stats_common(rq, p, + p->wts.demand_scaled, + new_scaled); + + p->wts.pred_demand = new; + p->wts.pred_demand_scaled = new_scaled; +} + +void clear_top_tasks_bitmap(unsigned long *bitmap) +{ + memset(bitmap, 0, top_tasks_bitmap_size); + __set_bit(NUM_LOAD_INDICES, bitmap); +} + +static inline void clear_top_tasks_table(u8 *table) +{ + memset(table, 0, NUM_LOAD_INDICES * sizeof(u8)); +} + +static void update_top_tasks(struct task_struct *p, struct rq *rq, + u32 old_curr_window, int new_window, bool full_window) +{ + u8 curr = rq->wrq.curr_table; + u8 prev = 1 - curr; + u8 *curr_table = rq->wrq.top_tasks[curr]; + u8 *prev_table = rq->wrq.top_tasks[prev]; + int old_index, new_index, update_index; + u32 curr_window = p->wts.curr_window; + u32 prev_window = p->wts.prev_window; + bool zero_index_update; + + if (old_curr_window == curr_window && !new_window) + return; + + old_index = load_to_index(old_curr_window); + new_index = load_to_index(curr_window); + + if (!new_window) { + zero_index_update = !old_curr_window && curr_window; + if (old_index != new_index || zero_index_update) { + if (old_curr_window) + curr_table[old_index] -= 1; + if (curr_window) + curr_table[new_index] += 1; + if (new_index > rq->wrq.curr_top) + rq->wrq.curr_top = new_index; + } + + if (!curr_table[old_index]) + __clear_bit(NUM_LOAD_INDICES - old_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + + if (curr_table[new_index] == 1) + __set_bit(NUM_LOAD_INDICES - new_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + + return; + } + + /* + * The window has rolled over for this task. By the time we get + * here, curr/prev swaps would has already occurred. So we need + * to use prev_window for the new index. + */ + update_index = load_to_index(prev_window); + + if (full_window) { + /* + * Two cases here. Either 'p' ran for the entire window or + * it didn't run at all. In either case there is no entry + * in the prev table. If 'p' ran the entire window, we just + * need to create a new entry in the prev table. In this case + * update_index will be correspond to sched_ravg_window + * so we can unconditionally update the top index. + */ + if (prev_window) { + prev_table[update_index] += 1; + rq->wrq.prev_top = update_index; + } + + if (prev_table[update_index] == 1) + __set_bit(NUM_LOAD_INDICES - update_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + } else { + zero_index_update = !old_curr_window && prev_window; + if (old_index != update_index || zero_index_update) { + if (old_curr_window) + prev_table[old_index] -= 1; + + prev_table[update_index] += 1; + + if (update_index > rq->wrq.prev_top) + rq->wrq.prev_top = update_index; + + if (!prev_table[old_index]) + __clear_bit(NUM_LOAD_INDICES - old_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + + if (prev_table[update_index] == 1) + __set_bit(NUM_LOAD_INDICES - update_index - 1, + rq->wrq.top_tasks_bitmap[prev]); + } + } + + if (curr_window) { + curr_table[new_index] += 1; + + if (new_index > rq->wrq.curr_top) + rq->wrq.curr_top = new_index; + + if (curr_table[new_index] == 1) + __set_bit(NUM_LOAD_INDICES - new_index - 1, + rq->wrq.top_tasks_bitmap[curr]); + } +} + +static void rollover_top_tasks(struct rq *rq, bool full_window) +{ + u8 curr_table = rq->wrq.curr_table; + u8 prev_table = 1 - curr_table; + int curr_top = rq->wrq.curr_top; + + clear_top_tasks_table(rq->wrq.top_tasks[prev_table]); + clear_top_tasks_bitmap(rq->wrq.top_tasks_bitmap[prev_table]); + + if (full_window) { + curr_top = 0; + clear_top_tasks_table(rq->wrq.top_tasks[curr_table]); + clear_top_tasks_bitmap( + rq->wrq.top_tasks_bitmap[curr_table]); + } + + rq->wrq.curr_table = prev_table; + rq->wrq.prev_top = curr_top; + rq->wrq.curr_top = 0; +} + +static u32 empty_windows[NR_CPUS]; + +static void rollover_task_window(struct task_struct *p, bool full_window) +{ + u32 *curr_cpu_windows = empty_windows; + u32 curr_window; + int i; + + /* Rollover the sum */ + curr_window = 0; + + if (!full_window) { + curr_window = p->wts.curr_window; + curr_cpu_windows = p->wts.curr_window_cpu; + } + + p->wts.prev_window = curr_window; + p->wts.curr_window = 0; + + /* Roll over individual CPU contributions */ + for (i = 0; i < nr_cpu_ids; i++) { + p->wts.prev_window_cpu[i] = curr_cpu_windows[i]; + p->wts.curr_window_cpu[i] = 0; + } + + if (is_new_task(p)) + p->wts.active_time += task_rq(p)->wrq.prev_window_size; +} + +void sched_set_io_is_busy(int val) +{ + sched_io_is_busy = val; +} + +static inline int cpu_is_waiting_on_io(struct rq *rq) +{ + if (!sched_io_is_busy) + return 0; + + return atomic_read(&rq->nr_iowait); +} + +static int account_busy_for_cpu_time(struct rq *rq, struct task_struct *p, + u64 irqtime, int event) +{ + if (is_idle_task(p)) { + /* TASK_WAKE && TASK_MIGRATE is not possible on idle task! */ + if (event == PICK_NEXT_TASK) + return 0; + + /* PUT_PREV_TASK, TASK_UPDATE && IRQ_UPDATE are left */ + return irqtime || cpu_is_waiting_on_io(rq); + } + + if (event == TASK_WAKE) + return 0; + + if (event == PUT_PREV_TASK || event == IRQ_UPDATE) + return 1; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (rq->curr == p) + return 1; + + return p->on_rq ? SCHED_FREQ_ACCOUNT_WAIT_TIME : 0; + } + + /* TASK_MIGRATE, PICK_NEXT_TASK left */ + return SCHED_FREQ_ACCOUNT_WAIT_TIME; +} + +#define DIV64_U64_ROUNDUP(X, Y) div64_u64((X) + (Y - 1), Y) + +static inline u64 scale_exec_time(u64 delta, struct rq *rq) +{ + return (delta * rq->wrq.task_exec_scale) >> 10; +} + +/* Convert busy time to frequency equivalent + * Assumes load is scaled to 1024 + */ +static inline unsigned int load_to_freq(struct rq *rq, unsigned int load) +{ + return mult_frac(cpu_max_possible_freq(cpu_of(rq)), load, + (unsigned int)arch_scale_cpu_capacity(cpu_of(rq))); +} + +bool do_pl_notif(struct rq *rq) +{ + u64 prev = rq->wrq.old_busy_time; + u64 pl = rq->wrq.walt_stats.pred_demands_sum_scaled; + int cpu = cpu_of(rq); + + /* If already at max freq, bail out */ + if (capacity_orig_of(cpu) == capacity_curr_of(cpu)) + return false; + + prev = max(prev, rq->wrq.old_estimated_time); + + /* 400 MHz filter. */ + return (pl > prev) && (load_to_freq(rq, pl - prev) > 400000); +} + +static void rollover_cpu_window(struct rq *rq, bool full_window) +{ + u64 curr_sum = rq->wrq.curr_runnable_sum; + u64 nt_curr_sum = rq->wrq.nt_curr_runnable_sum; + u64 grp_curr_sum = rq->wrq.grp_time.curr_runnable_sum; + u64 grp_nt_curr_sum = rq->wrq.grp_time.nt_curr_runnable_sum; + + if (unlikely(full_window)) { + curr_sum = 0; + nt_curr_sum = 0; + grp_curr_sum = 0; + grp_nt_curr_sum = 0; + } + + rq->wrq.prev_runnable_sum = curr_sum; + rq->wrq.nt_prev_runnable_sum = nt_curr_sum; + rq->wrq.grp_time.prev_runnable_sum = grp_curr_sum; + rq->wrq.grp_time.nt_prev_runnable_sum = grp_nt_curr_sum; + + rq->wrq.curr_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = 0; + rq->wrq.grp_time.curr_runnable_sum = 0; + rq->wrq.grp_time.nt_curr_runnable_sum = 0; +} + +/* + * Account cpu activity in its + * busy time counters(rq->wrq.curr/prev_runnable_sum) + */ +static void update_cpu_busy_time(struct task_struct *p, struct rq *rq, + int event, u64 wallclock, u64 irqtime) +{ + int new_window, full_window = 0; + int p_is_curr_task = (p == rq->curr); + u64 mark_start = p->wts.mark_start; + u64 window_start = rq->wrq.window_start; + u32 window_size = rq->wrq.prev_window_size; + u64 delta; + u64 *curr_runnable_sum = &rq->wrq.curr_runnable_sum; + u64 *prev_runnable_sum = &rq->wrq.prev_runnable_sum; + u64 *nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + u64 *nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + bool new_task; + struct walt_related_thread_group *grp; + int cpu = rq->cpu; + u32 old_curr_window = p->wts.curr_window; + + new_window = mark_start < window_start; + if (new_window) + full_window = (window_start - mark_start) >= window_size; + + /* + * Handle per-task window rollover. We don't care about the + * idle task. + */ + if (!is_idle_task(p)) { + if (new_window) + rollover_task_window(p, full_window); + } + + new_task = is_new_task(p); + + if (p_is_curr_task && new_window) { + rollover_cpu_window(rq, full_window); + rollover_top_tasks(rq, full_window); + } + + if (!account_busy_for_cpu_time(rq, p, irqtime, event)) + goto done; + + grp = p->wts.grp; + if (grp) { + struct group_cpu_time *cpu_time = &rq->wrq.grp_time; + + curr_runnable_sum = &cpu_time->curr_runnable_sum; + prev_runnable_sum = &cpu_time->prev_runnable_sum; + + nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + } + + if (!new_window) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. No rollover + * since we didn't start a new window. An example of this is + * when a task starts execution and then sleeps within the + * same window. + */ + + if (!irqtime || !is_idle_task(p) || cpu_is_waiting_on_io(rq)) + delta = wallclock - mark_start; + else + delta = irqtime; + delta = scale_exec_time(delta, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + if (!is_idle_task(p)) { + p->wts.curr_window += delta; + p->wts.curr_window_cpu[cpu] += delta; + } + + goto done; + } + + if (!p_is_curr_task) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has also started, but p is not the current task, so the + * window is not rolled over - just split up and account + * as necessary into curr and prev. The window is only + * rolled over when a new window is processed for the current + * task. + * + * Irqtime can't be accounted by a task that isn't the + * currently running task. + */ + + if (!full_window) { + /* + * A full window hasn't elapsed, account partial + * contribution to previous completed window. + */ + delta = scale_exec_time(window_start - mark_start, rq); + p->wts.prev_window += delta; + p->wts.prev_window_cpu[cpu] += delta; + } else { + /* + * Since at least one full window has elapsed, + * the contribution to the previous window is the + * full window (window_size). + */ + delta = scale_exec_time(window_size, rq); + p->wts.prev_window = delta; + p->wts.prev_window_cpu[cpu] = delta; + } + + *prev_runnable_sum += delta; + if (new_task) + *nt_prev_runnable_sum += delta; + + /* Account piece of busy time in the current window. */ + delta = scale_exec_time(wallclock - window_start, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + p->wts.curr_window = delta; + p->wts.curr_window_cpu[cpu] = delta; + + goto done; + } + + if (!irqtime || !is_idle_task(p) || cpu_is_waiting_on_io(rq)) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has started and p is the current task so rollover is + * needed. If any of these three above conditions are true + * then this busy time can't be accounted as irqtime. + * + * Busy time for the idle task need not be accounted. + * + * An example of this would be a task that starts execution + * and then sleeps once a new window has begun. + */ + + if (!full_window) { + /* + * A full window hasn't elapsed, account partial + * contribution to previous completed window. + */ + delta = scale_exec_time(window_start - mark_start, rq); + if (!is_idle_task(p)) { + p->wts.prev_window += delta; + p->wts.prev_window_cpu[cpu] += delta; + } + } else { + /* + * Since at least one full window has elapsed, + * the contribution to the previous window is the + * full window (window_size). + */ + delta = scale_exec_time(window_size, rq); + if (!is_idle_task(p)) { + p->wts.prev_window = delta; + p->wts.prev_window_cpu[cpu] = delta; + } + } + + /* + * Rollover is done here by overwriting the values in + * prev_runnable_sum and curr_runnable_sum. + */ + *prev_runnable_sum += delta; + if (new_task) + *nt_prev_runnable_sum += delta; + + /* Account piece of busy time in the current window. */ + delta = scale_exec_time(wallclock - window_start, rq); + *curr_runnable_sum += delta; + if (new_task) + *nt_curr_runnable_sum += delta; + + if (!is_idle_task(p)) { + p->wts.curr_window = delta; + p->wts.curr_window_cpu[cpu] = delta; + } + + goto done; + } + + if (irqtime) { + /* + * account_busy_for_cpu_time() = 1 so busy time needs + * to be accounted to the current window. A new window + * has started and p is the current task so rollover is + * needed. The current task must be the idle task because + * irqtime is not accounted for any other task. + * + * Irqtime will be accounted each time we process IRQ activity + * after a period of idleness, so we know the IRQ busy time + * started at wallclock - irqtime. + */ + + SCHED_BUG_ON(!is_idle_task(p)); + mark_start = wallclock - irqtime; + + /* + * Roll window over. If IRQ busy time was just in the current + * window then that is all that need be accounted. + */ + if (mark_start > window_start) { + *curr_runnable_sum = scale_exec_time(irqtime, rq); + return; + } + + /* + * The IRQ busy time spanned multiple windows. Process the + * busy time preceding the current window start first. + */ + delta = window_start - mark_start; + if (delta > window_size) + delta = window_size; + delta = scale_exec_time(delta, rq); + *prev_runnable_sum += delta; + + /* Process the remaining IRQ busy time in the current window. */ + delta = wallclock - window_start; + rq->wrq.curr_runnable_sum = scale_exec_time(delta, rq); + + return; + } + +done: + if (!is_idle_task(p)) + update_top_tasks(p, rq, old_curr_window, + new_window, full_window); +} + + +static inline u32 predict_and_update_buckets( + struct task_struct *p, u32 runtime) { + + int bidx; + u32 pred_demand; + + if (!sched_predl) + return 0; + + bidx = busy_to_bucket(runtime); + pred_demand = get_pred_busy(p, bidx, runtime); + bucket_increase(p->wts.busy_buckets, bidx); + + return pred_demand; +} + +static int +account_busy_for_task_demand(struct rq *rq, struct task_struct *p, int event) +{ + /* + * No need to bother updating task demand for the idle task. + */ + if (is_idle_task(p)) + return 0; + + /* + * When a task is waking up it is completing a segment of non-busy + * time. Likewise, if wait time is not treated as busy time, then + * when a task begins to run or is migrated, it is not running and + * is completing a segment of non-busy time. + */ + if (event == TASK_WAKE || (!SCHED_ACCOUNT_WAIT_TIME && + (event == PICK_NEXT_TASK || event == TASK_MIGRATE))) + return 0; + + /* + * The idle exit time is not accounted for the first task _picked_ up to + * run on the idle CPU. + */ + if (event == PICK_NEXT_TASK && rq->curr == rq->idle) + return 0; + + /* + * TASK_UPDATE can be called on sleeping task, when its moved between + * related groups + */ + if (event == TASK_UPDATE) { + if (rq->curr == p) + return 1; + + return p->on_rq ? SCHED_ACCOUNT_WAIT_TIME : 0; + } + + return 1; +} + +unsigned int sysctl_sched_task_unfilter_period = 100000000; + +/* + * Called when new window is starting for a task, to record cpu usage over + * recently concluded window(s). Normally 'samples' should be 1. It can be > 1 + * when, say, a real-time task runs without preemption for several windows at a + * stretch. + */ +static void update_history(struct rq *rq, struct task_struct *p, + u32 runtime, int samples, int event) +{ + u32 *hist = &p->wts.sum_history[0]; + int ridx, widx; + u32 max = 0, avg, demand, pred_demand; + u64 sum = 0; + u16 demand_scaled, pred_demand_scaled; + + /* Ignore windows where task had no activity */ + if (!runtime || is_idle_task(p) || !samples) + goto done; + + /* Push new 'runtime' value onto stack */ + widx = sched_ravg_hist_size - 1; + ridx = widx - samples; + for (; ridx >= 0; --widx, --ridx) { + hist[widx] = hist[ridx]; + sum += hist[widx]; + if (hist[widx] > max) + max = hist[widx]; + } + + for (widx = 0; widx < samples && widx < sched_ravg_hist_size; widx++) { + hist[widx] = runtime; + sum += hist[widx]; + if (hist[widx] > max) + max = hist[widx]; + } + + p->wts.sum = 0; + + if (sysctl_sched_window_stats_policy == WINDOW_STATS_RECENT) { + demand = runtime; + } else if (sysctl_sched_window_stats_policy == WINDOW_STATS_MAX) { + demand = max; + } else { + avg = div64_u64(sum, sched_ravg_hist_size); + if (sysctl_sched_window_stats_policy == WINDOW_STATS_AVG) + demand = avg; + else + demand = max(avg, runtime); + } + pred_demand = predict_and_update_buckets(p, runtime); + demand_scaled = scale_demand(demand); + pred_demand_scaled = scale_demand(pred_demand); + + /* + * A throttled deadline sched class task gets dequeued without + * changing p->on_rq. Since the dequeue decrements walt stats + * avoid decrementing it here again. + * + * When window is rolled over, the cumulative window demand + * is reset to the cumulative runnable average (contribution from + * the tasks on the runqueue). If the current task is dequeued + * already, it's demand is not included in the cumulative runnable + * average. So add the task demand separately to cumulative window + * demand. + */ + if (!task_has_dl_policy(p) || !p->dl.dl_throttled) { + if (task_on_rq_queued(p)) + fixup_walt_sched_stats_common(rq, p, + demand_scaled, pred_demand_scaled); + else if (rq->curr == p) + walt_fixup_cum_window_demand(rq, demand_scaled); + } + + p->wts.demand = demand; + p->wts.demand_scaled = demand_scaled; + p->wts.coloc_demand = div64_u64(sum, sched_ravg_hist_size); + p->wts.pred_demand = pred_demand; + p->wts.pred_demand_scaled = pred_demand_scaled; + + if (demand_scaled > sysctl_sched_min_task_util_for_colocation) + p->wts.unfilter = sysctl_sched_task_unfilter_period; + else + if (p->wts.unfilter) + p->wts.unfilter = max_t(int, 0, + p->wts.unfilter - rq->wrq.prev_window_size); + +done: + trace_sched_update_history(rq, p, runtime, samples, event); +} + +static u64 add_to_task_demand(struct rq *rq, struct task_struct *p, u64 delta) +{ + delta = scale_exec_time(delta, rq); + p->wts.sum += delta; + if (unlikely(p->wts.sum > sched_ravg_window)) + p->wts.sum = sched_ravg_window; + + return delta; +} + +/* + * Account cpu demand of task and/or update task's cpu demand history + * + * ms = p->wts.mark_start; + * wc = wallclock + * ws = rq->wrq.window_start + * + * Three possibilities: + * + * a) Task event is contained within one window. + * window_start < mark_start < wallclock + * + * ws ms wc + * | | | + * V V V + * |---------------| + * + * In this case, p->wts.sum is updated *iff* event is appropriate + * (ex: event == PUT_PREV_TASK) + * + * b) Task event spans two windows. + * mark_start < window_start < wallclock + * + * ms ws wc + * | | | + * V V V + * -----|------------------- + * + * In this case, p->wts.sum is updated with (ws - ms) *iff* event + * is appropriate, then a new window sample is recorded followed + * by p->wts.sum being set to (wc - ws) *iff* event is appropriate. + * + * c) Task event spans more than two windows. + * + * ms ws_tmp ws wc + * | | | | + * V V V V + * ---|-------|-------|-------|-------|------ + * | | + * |<------ nr_full_windows ------>| + * + * In this case, p->wts.sum is updated with (ws_tmp - ms) first *iff* + * event is appropriate, window sample of p->wts.sum is recorded, + * 'nr_full_window' samples of window_size is also recorded *iff* + * event is appropriate and finally p->wts.sum is set to (wc - ws) + * *iff* event is appropriate. + * + * IMPORTANT : Leave p->wts.mark_start unchanged, as update_cpu_busy_time() + * depends on it! + */ +static u64 update_task_demand(struct task_struct *p, struct rq *rq, + int event, u64 wallclock) +{ + u64 mark_start = p->wts.mark_start; + u64 delta, window_start = rq->wrq.window_start; + int new_window, nr_full_windows; + u32 window_size = sched_ravg_window; + u64 runtime; + + new_window = mark_start < window_start; + if (!account_busy_for_task_demand(rq, p, event)) { + if (new_window) + /* + * If the time accounted isn't being accounted as + * busy time, and a new window started, only the + * previous window need be closed out with the + * pre-existing demand. Multiple windows may have + * elapsed, but since empty windows are dropped, + * it is not necessary to account those. + */ + update_history(rq, p, p->wts.sum, 1, event); + return 0; + } + + if (!new_window) { + /* + * The simple case - busy time contained within the existing + * window. + */ + return add_to_task_demand(rq, p, wallclock - mark_start); + } + + /* + * Busy time spans at least two windows. Temporarily rewind + * window_start to first window boundary after mark_start. + */ + delta = window_start - mark_start; + nr_full_windows = div64_u64(delta, window_size); + window_start -= (u64)nr_full_windows * (u64)window_size; + + /* Process (window_start - mark_start) first */ + runtime = add_to_task_demand(rq, p, window_start - mark_start); + + /* Push new sample(s) into task's demand history */ + update_history(rq, p, p->wts.sum, 1, event); + if (nr_full_windows) { + u64 scaled_window = scale_exec_time(window_size, rq); + + update_history(rq, p, scaled_window, nr_full_windows, event); + runtime += nr_full_windows * scaled_window; + } + + /* + * Roll window_start back to current to process any remainder + * in current window. + */ + window_start += (u64)nr_full_windows * (u64)window_size; + + /* Process (wallclock - window_start) next */ + mark_start = window_start; + runtime += add_to_task_demand(rq, p, wallclock - mark_start); + + return runtime; +} + +static inline unsigned int cpu_cur_freq(int cpu) +{ + return cpu_rq(cpu)->wrq.cluster->cur_freq; +} + +static void +update_task_rq_cpu_cycles(struct task_struct *p, struct rq *rq, int event, + u64 wallclock, u64 irqtime) +{ + u64 cur_cycles; + u64 cycles_delta; + u64 time_delta; + int cpu = cpu_of(rq); + + lockdep_assert_held(&rq->lock); + + if (!use_cycle_counter) { + rq->wrq.task_exec_scale = DIV64_U64_ROUNDUP(cpu_cur_freq(cpu) * + arch_scale_cpu_capacity(cpu), + rq->wrq.cluster->max_possible_freq); + return; + } + + cur_cycles = read_cycle_counter(cpu, wallclock); + + /* + * If current task is idle task and irqtime == 0 CPU was + * indeed idle and probably its cycle counter was not + * increasing. We still need estimatied CPU frequency + * for IO wait time accounting. Use the previously + * calculated frequency in such a case. + */ + if (!is_idle_task(rq->curr) || irqtime) { + if (unlikely(cur_cycles < p->wts.cpu_cycles)) + cycles_delta = cur_cycles + (U64_MAX - + p->wts.cpu_cycles); + else + cycles_delta = cur_cycles - p->wts.cpu_cycles; + cycles_delta = cycles_delta * NSEC_PER_MSEC; + + if (event == IRQ_UPDATE && is_idle_task(p)) + /* + * Time between mark_start of idle task and IRQ handler + * entry time is CPU cycle counter stall period. + * Upon IRQ handler entry walt_sched_account_irqstart() + * replenishes idle task's cpu cycle counter so + * cycles_delta now represents increased cycles during + * IRQ handler rather than time between idle entry and + * IRQ exit. Thus use irqtime as time delta. + */ + time_delta = irqtime; + else + time_delta = wallclock - p->wts.mark_start; + SCHED_BUG_ON((s64)time_delta < 0); + + rq->wrq.task_exec_scale = DIV64_U64_ROUNDUP(cycles_delta * + arch_scale_cpu_capacity(cpu), + time_delta * + rq->wrq.cluster->max_possible_freq); + + trace_sched_get_task_cpu_cycles(cpu, event, + cycles_delta, time_delta, p); + } + + p->wts.cpu_cycles = cur_cycles; +} + +static inline void run_walt_irq_work(u64 old_window_start, struct rq *rq) +{ + u64 result; + + if (old_window_start == rq->wrq.window_start) + return; + + result = atomic64_cmpxchg(&walt_irq_work_lastq_ws, old_window_start, + rq->wrq.window_start); + if (result == old_window_start) { + walt_irq_work_queue(&walt_cpufreq_irq_work); + trace_walt_window_rollover(rq->wrq.window_start); + } +} + +/* Reflect task activity on its demand and cpu's busy time statistics */ +void walt_update_task_ravg(struct task_struct *p, struct rq *rq, int event, + u64 wallclock, u64 irqtime) +{ + u64 old_window_start; + + if (!rq->wrq.window_start || p->wts.mark_start == wallclock) + return; + + lockdep_assert_held(&rq->lock); + + old_window_start = update_window_start(rq, wallclock, event); + + if (!p->wts.mark_start) { + update_task_cpu_cycles(p, cpu_of(rq), wallclock); + goto done; + } + + update_task_rq_cpu_cycles(p, rq, event, wallclock, irqtime); + update_task_demand(p, rq, event, wallclock); + update_cpu_busy_time(p, rq, event, wallclock, irqtime); + update_task_pred_demand(rq, p, event); + if (event == PUT_PREV_TASK && p->state) + p->wts.iowaited = p->in_iowait; + + trace_sched_update_task_ravg(p, rq, event, wallclock, irqtime, + &rq->wrq.grp_time); + trace_sched_update_task_ravg_mini(p, rq, event, wallclock, irqtime, + &rq->wrq.grp_time); + +done: + p->wts.mark_start = wallclock; + + run_walt_irq_work(old_window_start, rq); +} + +u32 sched_get_init_task_load(struct task_struct *p) +{ + return p->wts.init_load_pct; +} + +int sched_set_init_task_load(struct task_struct *p, int init_load_pct) +{ + if (init_load_pct < 0 || init_load_pct > 100) + return -EINVAL; + + p->wts.init_load_pct = init_load_pct; + + return 0; +} + +void init_new_task_load(struct task_struct *p) +{ + int i; + u32 init_load_windows = sched_init_task_load_windows; + u32 init_load_windows_scaled = sched_init_task_load_windows_scaled; + u32 init_load_pct = current->wts.init_load_pct; + + p->wts.init_load_pct = 0; + rcu_assign_pointer(p->wts.grp, NULL); + INIT_LIST_HEAD(&p->wts.grp_list); + + p->wts.mark_start = 0; + p->wts.sum = 0; + p->wts.curr_window = 0; + p->wts.prev_window = 0; + p->wts.active_time = 0; + for (i = 0; i < NUM_BUSY_BUCKETS; ++i) + p->wts.busy_buckets[i] = 0; + + p->wts.cpu_cycles = 0; + + p->wts.curr_window_cpu = kcalloc(nr_cpu_ids, sizeof(u32), + GFP_KERNEL | __GFP_NOFAIL); + p->wts.prev_window_cpu = kcalloc(nr_cpu_ids, sizeof(u32), + GFP_KERNEL | __GFP_NOFAIL); + + if (init_load_pct) { + init_load_windows = div64_u64((u64)init_load_pct * + (u64)sched_ravg_window, 100); + init_load_windows_scaled = scale_demand(init_load_windows); + } + + p->wts.demand = init_load_windows; + p->wts.demand_scaled = init_load_windows_scaled; + p->wts.coloc_demand = init_load_windows; + p->wts.pred_demand = 0; + p->wts.pred_demand_scaled = 0; + for (i = 0; i < RAVG_HIST_SIZE_MAX; ++i) + p->wts.sum_history[i] = init_load_windows; + p->wts.misfit = false; + p->wts.rtg_high_prio = false; + p->wts.unfilter = sysctl_sched_task_unfilter_period; +} + +/* + * kfree() may wakeup kswapd. So this function should NOT be called + * with any CPU's rq->lock acquired. + */ +void free_task_load_ptrs(struct task_struct *p) +{ + kfree(p->wts.curr_window_cpu); + kfree(p->wts.prev_window_cpu); + + /* + * walt_update_task_ravg() can be called for exiting tasks. While the + * function itself ensures correct behavior, the corresponding + * trace event requires that these pointers be NULL. + */ + p->wts.curr_window_cpu = NULL; + p->wts.prev_window_cpu = NULL; +} + +void walt_task_dead(struct task_struct *p) +{ + sched_set_group_id(p, 0); + free_task_load_ptrs(p); +} + +void reset_task_stats(struct task_struct *p) +{ + int i = 0; + u32 *curr_window_ptr; + u32 *prev_window_ptr; + + curr_window_ptr = p->wts.curr_window_cpu; + prev_window_ptr = p->wts.prev_window_cpu; + memset(curr_window_ptr, 0, sizeof(u32) * nr_cpu_ids); + memset(prev_window_ptr, 0, sizeof(u32) * nr_cpu_ids); + + p->wts.mark_start = 0; + p->wts.sum = 0; + p->wts.demand = 0; + p->wts.coloc_demand = 0; + for (i = 0; i < RAVG_HIST_SIZE_MAX; ++i) + p->wts.sum_history[i] = 0; + p->wts.curr_window = 0; + p->wts.prev_window = 0; + p->wts.pred_demand = 0; + for (i = 0; i < NUM_BUSY_BUCKETS; ++i) + p->wts.busy_buckets[i] = 0; + p->wts.demand_scaled = 0; + p->wts.pred_demand_scaled = 0; + p->wts.active_time = 0; + + p->wts.curr_window_cpu = curr_window_ptr; + p->wts.prev_window_cpu = prev_window_ptr; +} + +void mark_task_starting(struct task_struct *p) +{ + u64 wallclock; + struct rq *rq = task_rq(p); + + if (!rq->wrq.window_start) { + reset_task_stats(p); + return; + } + + wallclock = sched_ktime_clock(); + p->wts.mark_start = p->wts.last_wake_ts = wallclock; + p->wts.last_enqueued_ts = wallclock; + update_task_cpu_cycles(p, cpu_of(rq), wallclock); +} + +/* + * Task groups whose aggregate demand on a cpu is more than + * sched_group_upmigrate need to be up-migrated if possible. + */ +unsigned int __read_mostly sched_group_upmigrate = 20000000; +unsigned int __read_mostly sysctl_sched_group_upmigrate_pct = 100; + +/* + * Task groups, once up-migrated, will need to drop their aggregate + * demand to less than sched_group_downmigrate before they are "down" + * migrated. + */ +unsigned int __read_mostly sched_group_downmigrate = 19000000; +unsigned int __read_mostly sysctl_sched_group_downmigrate_pct = 95; + +static inline void walt_update_group_thresholds(void) +{ + unsigned int min_scale = arch_scale_cpu_capacity( + cluster_first_cpu(sched_cluster[0])); + u64 min_ms = min_scale * (sched_ravg_window >> SCHED_CAPACITY_SHIFT); + + sched_group_upmigrate = div64_ul(min_ms * + sysctl_sched_group_upmigrate_pct, 100); + sched_group_downmigrate = div64_ul(min_ms * + sysctl_sched_group_downmigrate_pct, 100); +} + +struct walt_sched_cluster *sched_cluster[NR_CPUS]; +__read_mostly int num_sched_clusters; + +struct list_head cluster_head; +cpumask_t asym_cap_sibling_cpus = CPU_MASK_NONE; + +static struct walt_sched_cluster init_cluster = { + .list = LIST_HEAD_INIT(init_cluster.list), + .id = 0, + .cur_freq = 1, + .max_possible_freq = 1, + .aggr_grp_load = 0, +}; + +void init_clusters(void) +{ + init_cluster.cpus = *cpu_possible_mask; + raw_spin_lock_init(&init_cluster.load_lock); + INIT_LIST_HEAD(&cluster_head); + list_add(&init_cluster.list, &cluster_head); +} + +static void +insert_cluster(struct walt_sched_cluster *cluster, struct list_head *head) +{ + struct walt_sched_cluster *tmp; + struct list_head *iter = head; + + list_for_each_entry(tmp, head, list) { + if (arch_scale_cpu_capacity(cluster_first_cpu(cluster)) + < arch_scale_cpu_capacity(cluster_first_cpu(tmp))) + break; + iter = &tmp->list; + } + + list_add(&cluster->list, iter); +} + +static struct walt_sched_cluster *alloc_new_cluster(const struct cpumask *cpus) +{ + struct walt_sched_cluster *cluster = NULL; + + cluster = kzalloc(sizeof(struct walt_sched_cluster), GFP_ATOMIC); + if (!cluster) { + pr_warn("Cluster allocation failed. Possible bad scheduling\n"); + return NULL; + } + + INIT_LIST_HEAD(&cluster->list); + cluster->cur_freq = 1; + cluster->max_possible_freq = 1; + + raw_spin_lock_init(&cluster->load_lock); + cluster->cpus = *cpus; + + return cluster; +} + +static void add_cluster(const struct cpumask *cpus, struct list_head *head) +{ + struct walt_sched_cluster *cluster = alloc_new_cluster(cpus); + int i; + + if (!cluster) + return; + + for_each_cpu(i, cpus) + cpu_rq(i)->wrq.cluster = cluster; + + insert_cluster(cluster, head); + num_sched_clusters++; +} + +static void cleanup_clusters(struct list_head *head) +{ + struct walt_sched_cluster *cluster, *tmp; + int i; + + list_for_each_entry_safe(cluster, tmp, head, list) { + for_each_cpu(i, &cluster->cpus) + cpu_rq(i)->wrq.cluster = &init_cluster; + + list_del(&cluster->list); + num_sched_clusters--; + kfree(cluster); + } +} + +static inline void assign_cluster_ids(struct list_head *head) +{ + struct walt_sched_cluster *cluster; + int pos = 0; + + list_for_each_entry(cluster, head, list) { + cluster->id = pos; + sched_cluster[pos++] = cluster; + } + + WARN_ON(pos > MAX_NR_CLUSTERS); +} + +static inline void +move_list(struct list_head *dst, struct list_head *src, bool sync_rcu) +{ + struct list_head *first, *last; + + first = src->next; + last = src->prev; + + if (sync_rcu) { + INIT_LIST_HEAD_RCU(src); + synchronize_rcu(); + } + + first->prev = dst; + dst->prev = last; + last->next = dst; + + /* Ensure list sanity before making the head visible to all CPUs. */ + smp_mb(); + dst->next = first; +} + +static void update_all_clusters_stats(void) +{ + struct walt_sched_cluster *cluster; + u64 highest_mpc = 0, lowest_mpc = U64_MAX; + unsigned long flags; + + acquire_rq_locks_irqsave(cpu_possible_mask, &flags); + + for_each_sched_cluster(cluster) { + u64 mpc = arch_scale_cpu_capacity( + cluster_first_cpu(cluster)); + + if (mpc > highest_mpc) + highest_mpc = mpc; + + if (mpc < lowest_mpc) + lowest_mpc = mpc; + } + + max_possible_capacity = highest_mpc; + min_max_possible_capacity = lowest_mpc; + walt_update_group_thresholds(); + + release_rq_locks_irqrestore(cpu_possible_mask, &flags); +} + +static bool walt_clusters_parsed; +__read_mostly cpumask_t **cpu_array; + +static cpumask_t **init_cpu_array(void) +{ + int i; + cpumask_t **tmp_array; + + tmp_array = kcalloc(num_sched_clusters, sizeof(cpumask_t *), + GFP_ATOMIC); + if (!tmp_array) + return NULL; + for (i = 0; i < num_sched_clusters; i++) { + tmp_array[i] = kcalloc(num_sched_clusters, sizeof(cpumask_t), + GFP_ATOMIC); + if (!tmp_array[i]) + return NULL; + } + + return tmp_array; +} + +static cpumask_t **build_cpu_array(void) +{ + int i; + cpumask_t **tmp_array = init_cpu_array(); + + if (!tmp_array) + return NULL; + + /*Construct cpu_array row by row*/ + for (i = 0; i < num_sched_clusters; i++) { + int j, k = 1; + + /* Fill out first column with appropriate cpu arrays*/ + cpumask_copy(&tmp_array[i][0], &sched_cluster[i]->cpus); + + /* + * k starts from column 1 because 0 is filled + * Fill clusters for the rest of the row, + * above i in ascending order + */ + for (j = i + 1; j < num_sched_clusters; j++) { + cpumask_copy(&tmp_array[i][k], + &sched_cluster[j]->cpus); + k++; + } + + /* + * k starts from where we left off above. + * Fill clusters below i in descending order. + */ + for (j = i - 1; j >= 0; j--) { + cpumask_copy(&tmp_array[i][k], + &sched_cluster[j]->cpus); + k++; + } + } + return tmp_array; +} + +static void walt_get_possible_siblings(int cpuid, struct cpumask *cluster_cpus) +{ + int cpu; + struct cpu_topology *cpu_topo, *cpuid_topo = &cpu_topology[cpuid]; + + if (cpuid_topo->package_id == -1) + return; + + for_each_possible_cpu(cpu) { + cpu_topo = &cpu_topology[cpu]; + + if (cpuid_topo->package_id != cpu_topo->package_id) + continue; + cpumask_set_cpu(cpu, cluster_cpus); + } +} + +void walt_update_cluster_topology(void) +{ + struct cpumask cpus = *cpu_possible_mask; + struct cpumask cluster_cpus; + struct walt_sched_cluster *cluster; + struct list_head new_head; + cpumask_t **tmp; + int i; + + INIT_LIST_HEAD(&new_head); + + for_each_cpu(i, &cpus) { + cpumask_clear(&cluster_cpus); + walt_get_possible_siblings(i, &cluster_cpus); + if (cpumask_empty(&cluster_cpus)) { + WARN(1, "WALT: Invalid cpu topology!!"); + cleanup_clusters(&new_head); + return; + } + cpumask_andnot(&cpus, &cpus, &cluster_cpus); + add_cluster(&cluster_cpus, &new_head); + } + + assign_cluster_ids(&new_head); + + list_for_each_entry(cluster, &new_head, list) { + struct cpufreq_policy *policy; + + policy = cpufreq_cpu_get_raw(cluster_first_cpu(cluster)); + /* + * walt_update_cluster_topology() must be called AFTER policies + * for all cpus are initialized. If not, simply BUG(). + */ + SCHED_BUG_ON(!policy); + + if (policy) { + cluster->max_possible_freq = policy->cpuinfo.max_freq; + + for_each_cpu(i, &cluster->cpus) + cpumask_copy(&cpu_rq(i)->wrq.freq_domain_cpumask, + policy->related_cpus); + } + } + + /* + * Ensure cluster ids are visible to all CPUs before making + * cluster_head visible. + */ + move_list(&cluster_head, &new_head, false); + update_all_clusters_stats(); + + for_each_sched_cluster(cluster) { + if (cpumask_weight(&cluster->cpus) == 1) + cpumask_or(&asym_cap_sibling_cpus, + &asym_cap_sibling_cpus, &cluster->cpus); + } + + if (cpumask_weight(&asym_cap_sibling_cpus) == 1) + cpumask_clear(&asym_cap_sibling_cpus); + + tmp = build_cpu_array(); + if (!tmp) { + BUG_ON(1); + return; + } + smp_store_release(&cpu_array, tmp); + walt_clusters_parsed = true; +} + +static int cpufreq_notifier_trans(struct notifier_block *nb, + unsigned long val, void *data) +{ + struct cpufreq_freqs *freq = (struct cpufreq_freqs *)data; + unsigned int cpu = freq->policy->cpu, new_freq = freq->new; + unsigned long flags; + struct walt_sched_cluster *cluster; + struct cpumask policy_cpus = cpu_rq(cpu)->wrq.freq_domain_cpumask; + int i, j; + + if (use_cycle_counter) + return NOTIFY_DONE; + + if (cpu_rq(cpumask_first(&policy_cpus))->wrq.cluster == &init_cluster) + return NOTIFY_DONE; + + if (val != CPUFREQ_POSTCHANGE) + return NOTIFY_DONE; + + if (cpu_cur_freq(cpu) == new_freq) + return NOTIFY_OK; + + for_each_cpu(i, &policy_cpus) { + cluster = cpu_rq(i)->wrq.cluster; + + for_each_cpu(j, &cluster->cpus) { + struct rq *rq = cpu_rq(j); + + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_task_ravg(rq->curr, rq, TASK_UPDATE, + sched_ktime_clock(), 0); + raw_spin_unlock_irqrestore(&rq->lock, flags); + } + + cluster->cur_freq = new_freq; + cpumask_andnot(&policy_cpus, &policy_cpus, &cluster->cpus); + } + + return NOTIFY_OK; +} + +static struct notifier_block notifier_trans_block = { + .notifier_call = cpufreq_notifier_trans +}; + +static int register_walt_callback(void) +{ + return cpufreq_register_notifier(¬ifier_trans_block, + CPUFREQ_TRANSITION_NOTIFIER); +} +/* + * cpufreq callbacks can be registered at core_initcall or later time. + * Any registration done prior to that is "forgotten" by cpufreq. See + * initialization of variable init_cpufreq_transition_notifier_list_called + * for further information. + */ +core_initcall(register_walt_callback); + +int register_cpu_cycle_counter_cb(struct cpu_cycle_counter_cb *cb) +{ + unsigned long flags; + + mutex_lock(&cluster_lock); + if (!cb->get_cpu_cycle_counter) { + mutex_unlock(&cluster_lock); + return -EINVAL; + } + + acquire_rq_locks_irqsave(cpu_possible_mask, &flags); + cpu_cycle_counter_cb = *cb; + use_cycle_counter = true; + release_rq_locks_irqrestore(cpu_possible_mask, &flags); + + mutex_unlock(&cluster_lock); + + cpufreq_unregister_notifier(¬ifier_trans_block, + CPUFREQ_TRANSITION_NOTIFIER); + return 0; +} + +static void transfer_busy_time(struct rq *rq, + struct walt_related_thread_group *grp, + struct task_struct *p, int event); + +/* + * Enable colocation and frequency aggregation for all threads in a process. + * The children inherits the group id from the parent. + */ +unsigned int __read_mostly sysctl_sched_coloc_downmigrate_ns; + +struct walt_related_thread_group + *related_thread_groups[MAX_NUM_CGROUP_COLOC_ID]; +static LIST_HEAD(active_related_thread_groups); +static DEFINE_RWLOCK(related_thread_group_lock); + +static inline +void update_best_cluster(struct walt_related_thread_group *grp, + u64 demand, bool boost) +{ + if (boost) { + /* + * since we are in boost, we can keep grp on min, the boosts + * will ensure tasks get to bigs + */ + grp->skip_min = false; + return; + } + + if (is_suh_max()) + demand = sched_group_upmigrate; + + if (!grp->skip_min) { + if (demand >= sched_group_upmigrate) { + grp->skip_min = true; + } + return; + } + if (demand < sched_group_downmigrate) { + if (!sysctl_sched_coloc_downmigrate_ns) { + grp->skip_min = false; + return; + } + if (!grp->downmigrate_ts) { + grp->downmigrate_ts = grp->last_update; + return; + } + if (grp->last_update - grp->downmigrate_ts > + sysctl_sched_coloc_downmigrate_ns) { + grp->downmigrate_ts = 0; + grp->skip_min = false; + } + } else if (grp->downmigrate_ts) + grp->downmigrate_ts = 0; +} + +int preferred_cluster(struct walt_sched_cluster *cluster, struct task_struct *p) +{ + struct walt_related_thread_group *grp; + int rc = -1; + + rcu_read_lock(); + + grp = task_related_thread_group(p); + if (grp) + rc = (sched_cluster[(int)grp->skip_min] == cluster || + cpumask_subset(&cluster->cpus, &asym_cap_sibling_cpus)); + + rcu_read_unlock(); + return rc; +} + +static void _set_preferred_cluster(struct walt_related_thread_group *grp) +{ + struct task_struct *p; + u64 combined_demand = 0; + bool group_boost = false; + u64 wallclock; + bool prev_skip_min = grp->skip_min; + + if (list_empty(&grp->tasks)) { + grp->skip_min = false; + goto out; + } + + if (!hmp_capable()) { + grp->skip_min = false; + goto out; + } + + wallclock = sched_ktime_clock(); + + /* + * wakeup of two or more related tasks could race with each other and + * could result in multiple calls to _set_preferred_cluster being issued + * at same time. Avoid overhead in such cases of rechecking preferred + * cluster + */ + if (wallclock - grp->last_update < sched_ravg_window / 10) + return; + + list_for_each_entry(p, &grp->tasks, wts.grp_list) { + if (task_boost_policy(p) == SCHED_BOOST_ON_BIG) { + group_boost = true; + break; + } + + if (p->wts.mark_start < wallclock - + (sched_ravg_window * sched_ravg_hist_size)) + continue; + + combined_demand += p->wts.coloc_demand; + if (!trace_sched_set_preferred_cluster_enabled()) { + if (combined_demand > sched_group_upmigrate) + break; + } + } + + grp->last_update = wallclock; + update_best_cluster(grp, combined_demand, group_boost); + trace_sched_set_preferred_cluster(grp, combined_demand); + +out: + if (grp->id == DEFAULT_CGROUP_COLOC_ID + && grp->skip_min != prev_skip_min) { + if (grp->skip_min) + grp->start_ts = sched_clock(); + sched_update_hyst_times(); + } +} + +void set_preferred_cluster(struct walt_related_thread_group *grp) +{ + raw_spin_lock(&grp->lock); + _set_preferred_cluster(grp); + raw_spin_unlock(&grp->lock); +} + +int update_preferred_cluster(struct walt_related_thread_group *grp, + struct task_struct *p, u32 old_load, bool from_tick) +{ + u32 new_load = task_load(p); + + if (!grp) + return 0; + + if (unlikely(from_tick && is_suh_max())) + return 1; + + /* + * Update if task's load has changed significantly or a complete window + * has passed since we last updated preference + */ + if (abs(new_load - old_load) > sched_ravg_window / 4 || + sched_ktime_clock() - grp->last_update > sched_ravg_window) + return 1; + + return 0; +} + +#define ADD_TASK 0 +#define REM_TASK 1 + +static inline struct walt_related_thread_group* +lookup_related_thread_group(unsigned int group_id) +{ + return related_thread_groups[group_id]; +} + +int alloc_related_thread_groups(void) +{ + int i, ret; + struct walt_related_thread_group *grp; + + /* groupd_id = 0 is invalid as it's special id to remove group. */ + for (i = 1; i < MAX_NUM_CGROUP_COLOC_ID; i++) { + grp = kzalloc(sizeof(*grp), GFP_NOWAIT); + if (!grp) { + ret = -ENOMEM; + goto err; + } + + grp->id = i; + INIT_LIST_HEAD(&grp->tasks); + INIT_LIST_HEAD(&grp->list); + raw_spin_lock_init(&grp->lock); + + related_thread_groups[i] = grp; + } + + return 0; + +err: + for (i = 1; i < MAX_NUM_CGROUP_COLOC_ID; i++) { + grp = lookup_related_thread_group(i); + if (grp) { + kfree(grp); + related_thread_groups[i] = NULL; + } else { + break; + } + } + + return ret; +} + +static void remove_task_from_group(struct task_struct *p) +{ + struct walt_related_thread_group *grp = p->wts.grp; + struct rq *rq; + int empty_group = 1; + struct rq_flags rf; + + raw_spin_lock(&grp->lock); + + rq = __task_rq_lock(p, &rf); + transfer_busy_time(rq, p->wts.grp, p, REM_TASK); + list_del_init(&p->wts.grp_list); + rcu_assign_pointer(p->wts.grp, NULL); + __task_rq_unlock(rq, &rf); + + + if (!list_empty(&grp->tasks)) { + empty_group = 0; + _set_preferred_cluster(grp); + } + + raw_spin_unlock(&grp->lock); + + /* Reserved groups cannot be destroyed */ + if (empty_group && grp->id != DEFAULT_CGROUP_COLOC_ID) + /* + * We test whether grp->list is attached with list_empty() + * hence re-init the list after deletion. + */ + list_del_init(&grp->list); +} + +static int +add_task_to_group(struct task_struct *p, struct walt_related_thread_group *grp) +{ + struct rq *rq; + struct rq_flags rf; + + raw_spin_lock(&grp->lock); + + /* + * Change p->wts.grp under rq->lock. Will prevent races with read-side + * reference of p->wts.grp in various hot-paths + */ + rq = __task_rq_lock(p, &rf); + transfer_busy_time(rq, grp, p, ADD_TASK); + list_add(&p->wts.grp_list, &grp->tasks); + rcu_assign_pointer(p->wts.grp, grp); + __task_rq_unlock(rq, &rf); + + _set_preferred_cluster(grp); + + raw_spin_unlock(&grp->lock); + + return 0; +} + +#ifdef CONFIG_UCLAMP_TASK_GROUP +static inline bool uclamp_task_colocated(struct task_struct *p) +{ + struct cgroup_subsys_state *css; + struct task_group *tg; + bool colocate; + + rcu_read_lock(); + css = task_css(p, cpu_cgrp_id); + if (!css) { + rcu_read_unlock(); + return false; + } + tg = container_of(css, struct task_group, css); + colocate = tg->wtg.colocate; + rcu_read_unlock(); + + return colocate; +} +#else +static inline bool uclamp_task_colocated(struct task_struct *p) +{ + return false; +} +#endif /* CONFIG_UCLAMP_TASK_GROUP */ + +void add_new_task_to_grp(struct task_struct *new) +{ + unsigned long flags; + struct walt_related_thread_group *grp; + + /* + * If the task does not belong to colocated schedtune + * cgroup, nothing to do. We are checking this without + * lock. Even if there is a race, it will be added + * to the co-located cgroup via cgroup attach. + */ + if (!uclamp_task_colocated(new)) + return; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + write_lock_irqsave(&related_thread_group_lock, flags); + + /* + * It's possible that someone already added the new task to the + * group. or it might have taken out from the colocated schedtune + * cgroup. check these conditions under lock. + */ + if (!uclamp_task_colocated(new) || new->wts.grp) { + write_unlock_irqrestore(&related_thread_group_lock, flags); + return; + } + + raw_spin_lock(&grp->lock); + + rcu_assign_pointer(new->wts.grp, grp); + list_add(&new->wts.grp_list, &grp->tasks); + + raw_spin_unlock(&grp->lock); + write_unlock_irqrestore(&related_thread_group_lock, flags); +} + +static int __sched_set_group_id(struct task_struct *p, unsigned int group_id) +{ + int rc = 0; + unsigned long flags; + struct walt_related_thread_group *grp = NULL; + + if (group_id >= MAX_NUM_CGROUP_COLOC_ID) + return -EINVAL; + + raw_spin_lock_irqsave(&p->pi_lock, flags); + write_lock(&related_thread_group_lock); + + /* Switching from one group to another directly is not permitted */ + if ((!p->wts.grp && !group_id) || (p->wts.grp && group_id)) + goto done; + + if (!group_id) { + remove_task_from_group(p); + goto done; + } + + grp = lookup_related_thread_group(group_id); + if (list_empty(&grp->list)) + list_add(&grp->list, &active_related_thread_groups); + + rc = add_task_to_group(p, grp); +done: + write_unlock(&related_thread_group_lock); + raw_spin_unlock_irqrestore(&p->pi_lock, flags); + + return rc; +} + +int sched_set_group_id(struct task_struct *p, unsigned int group_id) +{ + /* DEFAULT_CGROUP_COLOC_ID is a reserved id */ + if (group_id == DEFAULT_CGROUP_COLOC_ID) + return -EINVAL; + + return __sched_set_group_id(p, group_id); +} + +unsigned int sched_get_group_id(struct task_struct *p) +{ + unsigned int group_id; + struct walt_related_thread_group *grp; + + rcu_read_lock(); + grp = task_related_thread_group(p); + group_id = grp ? grp->id : 0; + rcu_read_unlock(); + + return group_id; +} + +#if defined(CONFIG_UCLAMP_TASK_GROUP) +/* + * We create a default colocation group at boot. There is no need to + * synchronize tasks between cgroups at creation time because the + * correct cgroup hierarchy is not available at boot. Therefore cgroup + * colocation is turned off by default even though the colocation group + * itself has been allocated. Furthermore this colocation group cannot + * be destroyted once it has been created. All of this has been as part + * of runtime optimizations. + * + * The job of synchronizing tasks to the colocation group is done when + * the colocation flag in the cgroup is turned on. + */ +static int __init create_default_coloc_group(void) +{ + struct walt_related_thread_group *grp = NULL; + unsigned long flags; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + write_lock_irqsave(&related_thread_group_lock, flags); + list_add(&grp->list, &active_related_thread_groups); + write_unlock_irqrestore(&related_thread_group_lock, flags); + + return 0; +} +late_initcall(create_default_coloc_group); + +int sync_cgroup_colocation(struct task_struct *p, bool insert) +{ + unsigned int grp_id = insert ? DEFAULT_CGROUP_COLOC_ID : 0; + + return __sched_set_group_id(p, grp_id); +} +#endif + +static bool is_cluster_hosting_top_app(struct walt_sched_cluster *cluster) +{ + struct walt_related_thread_group *grp; + bool grp_on_min; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + if (!grp) + return false; + + grp_on_min = !grp->skip_min && + (sched_boost_policy() != SCHED_BOOST_ON_BIG); + + return (is_min_capacity_cluster(cluster) == grp_on_min); +} + +static unsigned long thermal_cap_cpu[NR_CPUS]; + +unsigned long thermal_cap(int cpu) +{ + return thermal_cap_cpu[cpu] ?: SCHED_CAPACITY_SCALE; +} + +static inline unsigned long +do_thermal_cap(int cpu, unsigned long thermal_max_freq) +{ + if (unlikely(!walt_clusters_parsed)) + return capacity_orig_of(cpu); + + return mult_frac(arch_scale_cpu_capacity(cpu), thermal_max_freq, + cpu_max_possible_freq(cpu)); +} + +static DEFINE_SPINLOCK(cpu_freq_min_max_lock); +void sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, u32 fmax) +{ + struct cpumask cpumask; + int i; + unsigned long flags; + + spin_lock_irqsave(&cpu_freq_min_max_lock, flags); + cpumask_copy(&cpumask, cpus); + + for_each_cpu(i, &cpumask) + thermal_cap_cpu[i] = do_thermal_cap(i, fmax); + + spin_unlock_irqrestore(&cpu_freq_min_max_lock, flags); +} + +void note_task_waking(struct task_struct *p, u64 wallclock) +{ + p->wts.last_wake_ts = wallclock; +} + +/* + * Task's cpu usage is accounted in: + * rq->wrq.curr/prev_runnable_sum, when its ->grp is NULL + * grp->cpu_time[cpu]->curr/prev_runnable_sum, when its ->grp is !NULL + * + * Transfer task's cpu usage between those counters when transitioning between + * groups + */ +static void transfer_busy_time(struct rq *rq, + struct walt_related_thread_group *grp, + struct task_struct *p, int event) +{ + u64 wallclock; + struct group_cpu_time *cpu_time; + u64 *src_curr_runnable_sum, *dst_curr_runnable_sum; + u64 *src_prev_runnable_sum, *dst_prev_runnable_sum; + u64 *src_nt_curr_runnable_sum, *dst_nt_curr_runnable_sum; + u64 *src_nt_prev_runnable_sum, *dst_nt_prev_runnable_sum; + int migrate_type; + int cpu = cpu_of(rq); + bool new_task; + int i; + + wallclock = sched_ktime_clock(); + + walt_update_task_ravg(rq->curr, rq, TASK_UPDATE, wallclock, 0); + walt_update_task_ravg(p, rq, TASK_UPDATE, wallclock, 0); + new_task = is_new_task(p); + + cpu_time = &rq->wrq.grp_time; + if (event == ADD_TASK) { + migrate_type = RQ_TO_GROUP; + + src_curr_runnable_sum = &rq->wrq.curr_runnable_sum; + dst_curr_runnable_sum = &cpu_time->curr_runnable_sum; + src_prev_runnable_sum = &rq->wrq.prev_runnable_sum; + dst_prev_runnable_sum = &cpu_time->prev_runnable_sum; + + src_nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + dst_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + dst_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + + if (*src_curr_runnable_sum < p->wts.curr_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_curr_runnable_sum, + p->wts.curr_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_curr_runnable_sum -= p->wts.curr_window_cpu[cpu]; + + if (*src_prev_runnable_sum < p->wts.prev_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_prev_runnable_sum, + p->wts.prev_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_prev_runnable_sum -= p->wts.prev_window_cpu[cpu]; + + if (new_task) { + if (*src_nt_curr_runnable_sum < + p->wts.curr_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_curr_runnable_sum, + p->wts.curr_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_curr_runnable_sum -= + p->wts.curr_window_cpu[cpu]; + + if (*src_nt_prev_runnable_sum < + p->wts.prev_window_cpu[cpu]) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_prev_runnable_sum, + p->wts.prev_window_cpu[cpu]); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_prev_runnable_sum -= + p->wts.prev_window_cpu[cpu]; + } + + update_cluster_load_subtractions(p, cpu, + rq->wrq.window_start, new_task); + + } else { + migrate_type = GROUP_TO_RQ; + + src_curr_runnable_sum = &cpu_time->curr_runnable_sum; + dst_curr_runnable_sum = &rq->wrq.curr_runnable_sum; + src_prev_runnable_sum = &cpu_time->prev_runnable_sum; + dst_prev_runnable_sum = &rq->wrq.prev_runnable_sum; + + src_nt_curr_runnable_sum = &cpu_time->nt_curr_runnable_sum; + dst_nt_curr_runnable_sum = &rq->wrq.nt_curr_runnable_sum; + src_nt_prev_runnable_sum = &cpu_time->nt_prev_runnable_sum; + dst_nt_prev_runnable_sum = &rq->wrq.nt_prev_runnable_sum; + + if (*src_curr_runnable_sum < p->wts.curr_window) { + printk_deferred("WALT-UG pid=%u CPU=%d event=%d src_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_curr_runnable_sum, + p->wts.curr_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_curr_runnable_sum -= p->wts.curr_window; + + if (*src_prev_runnable_sum < p->wts.prev_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, *src_prev_runnable_sum, + p->wts.prev_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_prev_runnable_sum -= p->wts.prev_window; + + if (new_task) { + if (*src_nt_curr_runnable_sum < p->wts.curr_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_crs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_curr_runnable_sum, + p->wts.curr_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_curr_runnable_sum -= p->wts.curr_window; + + if (*src_nt_prev_runnable_sum < p->wts.prev_window) { + printk_deferred("WALT-BUG pid=%u CPU=%d event=%d src_nt_prs=%llu is lesser than task_contrib=%llu", + p->pid, cpu, event, + *src_nt_prev_runnable_sum, + p->wts.prev_window); + walt_task_dump(p); + SCHED_BUG_ON(1); + } + *src_nt_prev_runnable_sum -= p->wts.prev_window; + } + + /* + * Need to reset curr/prev windows for all CPUs, not just the + * ones in the same cluster. Since inter cluster migrations + * did not result in the appropriate book keeping, the values + * per CPU would be inaccurate. + */ + for_each_possible_cpu(i) { + p->wts.curr_window_cpu[i] = 0; + p->wts.prev_window_cpu[i] = 0; + } + } + + *dst_curr_runnable_sum += p->wts.curr_window; + *dst_prev_runnable_sum += p->wts.prev_window; + if (new_task) { + *dst_nt_curr_runnable_sum += p->wts.curr_window; + *dst_nt_prev_runnable_sum += p->wts.prev_window; + } + + /* + * When a task enter or exits a group, it's curr and prev windows are + * moved to a single CPU. This behavior might be sub-optimal in the + * exit case, however, it saves us the overhead of handling inter + * cluster migration fixups while the task is part of a related group. + */ + p->wts.curr_window_cpu[cpu] = p->wts.curr_window; + p->wts.prev_window_cpu[cpu] = p->wts.prev_window; + + trace_sched_migration_update_sum(p, migrate_type, rq); +} + +bool is_rtgb_active(void) +{ + struct walt_related_thread_group *grp; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + return grp && grp->skip_min; +} + +u64 get_rtgb_active_time(void) +{ + struct walt_related_thread_group *grp; + u64 now = sched_clock(); + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + if (grp && grp->skip_min && grp->start_ts) + return now - grp->start_ts; + + return 0; +} + +static void walt_init_window_dep(void); +static void walt_tunables_fixup(void) +{ + if (likely(num_sched_clusters > 0)) + walt_update_group_thresholds(); + walt_init_window_dep(); +} + +static void walt_update_irqload(struct rq *rq) +{ + u64 irq_delta = 0; + unsigned int nr_windows = 0; + u64 cur_irq_time; + u64 last_irq_window = READ_ONCE(rq->wrq.last_irq_window); + + if (rq->wrq.window_start > last_irq_window) + nr_windows = div64_u64(rq->wrq.window_start - last_irq_window, + sched_ravg_window); + + /* Decay CPU's irqload by 3/4 for each window. */ + if (nr_windows < 10) + rq->wrq.avg_irqload = mult_frac(rq->wrq.avg_irqload, 3, 4); + else + rq->wrq.avg_irqload = 0; + + cur_irq_time = irq_time_read(cpu_of(rq)); + if (cur_irq_time > rq->wrq.prev_irq_time) + irq_delta = cur_irq_time - rq->wrq.prev_irq_time; + + rq->wrq.avg_irqload += irq_delta; + rq->wrq.prev_irq_time = cur_irq_time; + + if (nr_windows < SCHED_HIGH_IRQ_TIMEOUT) + rq->wrq.high_irqload = (rq->wrq.avg_irqload >= + walt_cpu_high_irqload); + else + rq->wrq.high_irqload = 0; +} + +/* + * Runs in hard-irq context. This should ideally run just after the latest + * window roll-over. + */ +void walt_irq_work(struct irq_work *irq_work) +{ + struct walt_sched_cluster *cluster; + struct rq *rq; + int cpu; + u64 wc; + bool is_migration = false, is_asym_migration = false; + u64 total_grp_load = 0, min_cluster_grp_load = 0; + int level = 0; + u64 cur_jiffies_ts; + unsigned long flags; + + /* Am I the window rollover work or the migration work? */ + if (irq_work == &walt_migration_irq_work) + is_migration = true; + + for_each_cpu(cpu, cpu_possible_mask) { + if (level == 0) + raw_spin_lock(&cpu_rq(cpu)->lock); + else + raw_spin_lock_nested(&cpu_rq(cpu)->lock, level); + level++; + } + + wc = sched_ktime_clock(); + cur_jiffies_ts = get_jiffies_64(); + walt_load_reported_window = atomic64_read(&walt_irq_work_lastq_ws); + for_each_sched_cluster(cluster) { + u64 aggr_grp_load = 0; + + raw_spin_lock(&cluster->load_lock); + + for_each_cpu(cpu, &cluster->cpus) { + rq = cpu_rq(cpu); + if (rq->curr) { + walt_update_task_ravg(rq->curr, rq, + TASK_UPDATE, wc, 0); + account_load_subtractions(rq); + aggr_grp_load += + rq->wrq.grp_time.prev_runnable_sum; + } + if (is_migration && rq->wrq.notif_pending && + cpumask_test_cpu(cpu, &asym_cap_sibling_cpus)) { + is_asym_migration = true; + rq->wrq.notif_pending = false; + } + } + + cluster->aggr_grp_load = aggr_grp_load; + total_grp_load += aggr_grp_load; + + if (is_min_capacity_cluster(cluster)) + min_cluster_grp_load = aggr_grp_load; + raw_spin_unlock(&cluster->load_lock); + } + + if (total_grp_load) { + if (cpumask_weight(&asym_cap_sibling_cpus)) { + u64 big_grp_load = + total_grp_load - min_cluster_grp_load; + + for_each_cpu(cpu, &asym_cap_sibling_cpus) + cpu_cluster(cpu)->aggr_grp_load = big_grp_load; + } + rtgb_active = is_rtgb_active(); + } else { + rtgb_active = false; + } + + if (!is_migration && sysctl_sched_user_hint && time_after(jiffies, + sched_user_hint_reset_time)) + sysctl_sched_user_hint = 0; + + for_each_sched_cluster(cluster) { + cpumask_t cluster_online_cpus; + unsigned int num_cpus, i = 1; + + cpumask_and(&cluster_online_cpus, &cluster->cpus, + cpu_online_mask); + num_cpus = cpumask_weight(&cluster_online_cpus); + for_each_cpu(cpu, &cluster_online_cpus) { + int flag = SCHED_CPUFREQ_WALT; + + rq = cpu_rq(cpu); + + if (is_migration) { + if (rq->wrq.notif_pending) { + flag |= SCHED_CPUFREQ_INTERCLUSTER_MIG; + rq->wrq.notif_pending = false; + } + } + + if (is_asym_migration && cpumask_test_cpu(cpu, + &asym_cap_sibling_cpus)) + flag |= SCHED_CPUFREQ_INTERCLUSTER_MIG; + + if (i == num_cpus) + cpufreq_update_util(cpu_rq(cpu), flag); + else + cpufreq_update_util(cpu_rq(cpu), flag | + SCHED_CPUFREQ_CONTINUE); + i++; + + if (!is_migration) + walt_update_irqload(rq); + } + } + + /* + * If the window change request is in pending, good place to + * change sched_ravg_window since all rq locks are acquired. + * + * If the current window roll over is delayed such that the + * mark_start (current wallclock with which roll over is done) + * of the current task went past the window start with the + * updated new window size, delay the update to the next + * window roll over. Otherwise the CPU counters (prs and crs) are + * not rolled over properly as mark_start > window_start. + */ + if (!is_migration) { + spin_lock_irqsave(&sched_ravg_window_lock, flags); + + if ((sched_ravg_window != new_sched_ravg_window) && + (wc < this_rq()->wrq.window_start + new_sched_ravg_window)) { + sched_ravg_window_change_time = sched_ktime_clock(); + printk_deferred("ALERT: changing window size from %u to %u at %lu\n", + sched_ravg_window, + new_sched_ravg_window, + sched_ravg_window_change_time); + trace_sched_ravg_window_change(sched_ravg_window, + new_sched_ravg_window, + sched_ravg_window_change_time); + sched_ravg_window = new_sched_ravg_window; + walt_tunables_fixup(); + } + spin_unlock_irqrestore(&sched_ravg_window_lock, flags); + } + + for_each_cpu(cpu, cpu_possible_mask) + raw_spin_unlock(&cpu_rq(cpu)->lock); + + if (!is_migration) + core_ctl_check(this_rq()->wrq.window_start); +} + +void walt_rotation_checkpoint(int nr_big) +{ + if (!hmp_capable()) + return; + + if (!sysctl_sched_walt_rotate_big_tasks || sched_boost() != NO_BOOST) { + walt_rotation_enabled = 0; + return; + } + + walt_rotation_enabled = nr_big >= num_possible_cpus(); +} + +void walt_fill_ta_data(struct core_ctl_notif_data *data) +{ + struct walt_related_thread_group *grp; + unsigned long flags; + u64 total_demand = 0, wallclock; + struct task_struct *p; + int min_cap_cpu, scale = 1024; + struct walt_sched_cluster *cluster; + int i = 0; + + grp = lookup_related_thread_group(DEFAULT_CGROUP_COLOC_ID); + + raw_spin_lock_irqsave(&grp->lock, flags); + if (list_empty(&grp->tasks)) { + raw_spin_unlock_irqrestore(&grp->lock, flags); + goto fill_util; + } + + wallclock = sched_ktime_clock(); + + list_for_each_entry(p, &grp->tasks, wts.grp_list) { + if (p->wts.mark_start < wallclock - + (sched_ravg_window * sched_ravg_hist_size)) + continue; + + total_demand += p->wts.coloc_demand; + } + + raw_spin_unlock_irqrestore(&grp->lock, flags); + + /* + * Scale the total demand to the lowest capacity CPU and + * convert into percentage. + * + * P = total_demand/sched_ravg_window * 1024/scale * 100 + */ + + min_cap_cpu = this_rq()->rd->wrd.min_cap_orig_cpu; + if (min_cap_cpu != -1) + scale = arch_scale_cpu_capacity(min_cap_cpu); + + data->coloc_load_pct = div64_u64(total_demand * 1024 * 100, + (u64)sched_ravg_window * scale); + +fill_util: + for_each_sched_cluster(cluster) { + int fcpu = cluster_first_cpu(cluster); + + if (i == MAX_CLUSTERS) + break; + + scale = arch_scale_cpu_capacity(fcpu); + data->ta_util_pct[i] = div64_u64(cluster->aggr_grp_load * 1024 * + 100, (u64)sched_ravg_window * scale); + + scale = arch_scale_freq_capacity(fcpu); + data->cur_cap_pct[i] = (scale * 100)/1024; + i++; + } +} + +int walt_proc_group_thresholds_handler(struct ctl_table *table, int write, + void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + static DEFINE_MUTEX(mutex); + struct rq *rq = cpu_rq(cpumask_first(cpu_possible_mask)); + unsigned long flags; + + if (unlikely(num_sched_clusters <= 0)) + return -EPERM; + + mutex_lock(&mutex); + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (ret || !write) { + mutex_unlock(&mutex); + return ret; + } + + /* + * The load scale factor update happens with all + * rqs locked. so acquiring 1 CPU rq lock and + * updating the thresholds is sufficient for + * an atomic update. + */ + raw_spin_lock_irqsave(&rq->lock, flags); + walt_update_group_thresholds(); + raw_spin_unlock_irqrestore(&rq->lock, flags); + + mutex_unlock(&mutex); + + return ret; +} + +static void walt_init_window_dep(void) +{ + walt_cpu_util_freq_divisor = + (sched_ravg_window >> SCHED_CAPACITY_SHIFT) * 100; + walt_scale_demand_divisor = sched_ravg_window >> SCHED_CAPACITY_SHIFT; + + sched_init_task_load_windows = + div64_u64((u64)sysctl_sched_init_task_load_pct * + (u64)sched_ravg_window, 100); + sched_init_task_load_windows_scaled = + scale_demand(sched_init_task_load_windows); + + walt_cpu_high_irqload = div64_u64((u64)sched_ravg_window * 95, (u64) 100); +} + +static void walt_init_once(void) +{ + init_irq_work(&walt_migration_irq_work, walt_irq_work); + init_irq_work(&walt_cpufreq_irq_work, walt_irq_work); + walt_rotate_work_init(); + walt_init_window_dep(); +} + +void walt_sched_init_rq(struct rq *rq) +{ + int j; + + if (cpu_of(rq) == 0) + walt_init_once(); + + cpumask_set_cpu(cpu_of(rq), &rq->wrq.freq_domain_cpumask); + + rq->wrq.walt_stats.cumulative_runnable_avg_scaled = 0; + rq->wrq.prev_window_size = sched_ravg_window; + rq->wrq.window_start = 0; + rq->wrq.walt_stats.nr_big_tasks = 0; + rq->wrq.walt_flags = 0; + rq->wrq.avg_irqload = 0; + rq->wrq.prev_irq_time = 0; + rq->wrq.last_irq_window = 0; + rq->wrq.high_irqload = false; + rq->wrq.task_exec_scale = 1024; + rq->wrq.push_task = NULL; + + /* + * All cpus part of same cluster by default. This avoids the + * need to check for rq->wrq.cluster being non-NULL in hot-paths + * like select_best_cpu() + */ + rq->wrq.cluster = &init_cluster; + rq->wrq.curr_runnable_sum = rq->wrq.prev_runnable_sum = 0; + rq->wrq.nt_curr_runnable_sum = rq->wrq.nt_prev_runnable_sum = 0; + memset(&rq->wrq.grp_time, 0, sizeof(struct group_cpu_time)); + rq->wrq.old_busy_time = 0; + rq->wrq.old_estimated_time = 0; + rq->wrq.walt_stats.pred_demands_sum_scaled = 0; + rq->wrq.walt_stats.nr_rtg_high_prio_tasks = 0; + rq->wrq.ed_task = NULL; + rq->wrq.curr_table = 0; + rq->wrq.prev_top = 0; + rq->wrq.curr_top = 0; + rq->wrq.last_cc_update = 0; + rq->wrq.cycles = 0; + for (j = 0; j < NUM_TRACKED_WINDOWS; j++) { + memset(&rq->wrq.load_subs[j], 0, + sizeof(struct load_subtractions)); + rq->wrq.top_tasks[j] = kcalloc(NUM_LOAD_INDICES, + sizeof(u8), GFP_NOWAIT); + /* No other choice */ + BUG_ON(!rq->wrq.top_tasks[j]); + clear_top_tasks_bitmap(rq->wrq.top_tasks_bitmap[j]); + } + rq->wrq.cum_window_demand_scaled = 0; + rq->wrq.notif_pending = false; +} + +int walt_proc_user_hint_handler(struct ctl_table *table, + int write, void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret; + unsigned int old_value; + static DEFINE_MUTEX(mutex); + + mutex_lock(&mutex); + + sched_user_hint_reset_time = jiffies + HZ; + old_value = sysctl_sched_user_hint; + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (ret || !write || (old_value == sysctl_sched_user_hint)) + goto unlock; + + walt_irq_work_queue(&walt_migration_irq_work); + +unlock: + mutex_unlock(&mutex); + return ret; +} + +static inline void sched_window_nr_ticks_change(void) +{ + unsigned long flags; + + spin_lock_irqsave(&sched_ravg_window_lock, flags); + new_sched_ravg_window = mult_frac(sysctl_sched_ravg_window_nr_ticks, + NSEC_PER_SEC, HZ); + spin_unlock_irqrestore(&sched_ravg_window_lock, flags); +} + +int sched_ravg_window_handler(struct ctl_table *table, + int write, void __user *buffer, size_t *lenp, + loff_t *ppos) +{ + int ret = -EPERM; + static DEFINE_MUTEX(mutex); + unsigned int prev_value; + + mutex_lock(&mutex); + + if (write && (HZ != 250 || !sysctl_sched_dynamic_ravg_window_enable)) + goto unlock; + + prev_value = sysctl_sched_ravg_window_nr_ticks; + ret = proc_douintvec_ravg_window(table, write, buffer, lenp, ppos); + if (ret || !write || (prev_value == sysctl_sched_ravg_window_nr_ticks)) + goto unlock; + + sched_window_nr_ticks_change(); + +unlock: + mutex_unlock(&mutex); + return ret; +} diff --git a/kernel/sched/walt.h b/kernel/sched/walt/walt.h similarity index 99% rename from kernel/sched/walt.h rename to kernel/sched/walt/walt.h index e20992b70532..b5985cb63f27 100644 --- a/kernel/sched/walt.h +++ b/kernel/sched/walt/walt.h @@ -1,6 +1,6 @@ /* SPDX-License-Identifier: GPL-2.0-only */ /* - * Copyright (c) 2016-2020, The Linux Foundation. All rights reserved. + * Copyright (c) 2016-2021, The Linux Foundation. All rights reserved. */ #ifndef __WALT_H