From 0362218e2d0339cedec8d19101052507211afc53 Mon Sep 17 00:00:00 2001 From: Lingutla Chandrasekhar Date: Thu, 1 Mar 2018 18:36:36 +0530 Subject: [PATCH 1/4] Revert "softirq: Let ksoftirqd do its job" Ksfotirqd is a normal priority CFS task. It can experience higher scheduling latency under heavy load conditions. Currently once asynchronous softirq processing is deferred to ksoftirqd, softirqs are not processed further until ksoftirqd task gets a chance to run. High latencies for softirqs like TIMER, HI TASKLET is not acceptable. So revert 'commit 4cd13c21b207 ("softirq: Let ksoftirqd do its job")'. Change-Id: I38a1a88b5f42dd534c65d739dbb7e4321a7904db Signed-off-by: Lingutla Chandrasekhar [satyap@codeaurora.org: Fix trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- kernel/softirq.c | 17 +---------------- 1 file changed, 1 insertion(+), 16 deletions(-) diff --git a/kernel/softirq.c b/kernel/softirq.c index 53ca9850d9ef..13edfcbb71c8 100644 --- a/kernel/softirq.c +++ b/kernel/softirq.c @@ -83,18 +83,6 @@ static void wakeup_softirqd(void) wake_up_process(tsk); } -/* - * If ksoftirqd is scheduled, we do not want to process pending softirqs - * right now. Let ksoftirqd handle this at its own rate, to get fairness. - */ -static bool ksoftirqd_running(void) -{ - struct task_struct *tsk = __this_cpu_read(ksoftirqd); - - return tsk && (tsk->state == TASK_RUNNING) && - !__kthread_should_park(tsk); -} - /* * preempt_count and SOFTIRQ_OFFSET usage: * - preempt_count is changed by SOFTIRQ_OFFSET on entering or leaving @@ -351,7 +339,7 @@ asmlinkage __visible void do_softirq(void) pending = local_softirq_pending(); - if (pending && !ksoftirqd_running()) + if (pending) do_softirq_own_stack(); local_irq_restore(flags); @@ -378,9 +366,6 @@ void irq_enter(void) static inline void invoke_softirq(void) { - if (ksoftirqd_running()) - return; - if (!force_irqthreads) { #ifdef CONFIG_HAVE_IRQ_EXIT_ON_IRQ_STACK /* From 4e1224407af4f553a69e467559c2c342f1be39a4 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Mon, 25 Feb 2019 13:06:12 -0800 Subject: [PATCH 2/4] watchdog: use per_cpu_ptr() in watchdog_disable() Watchdog gets disabled from other CPUs when isolation is enabled. Using this_cpu_ptr() would lead to undesired issues, so, switch to per_cpu_ptr(). While at it, remove warning as it makes no sense when core isolation is enabled. Change-Id: Id1a59c4c88044c7570457fb55abb2dcc13de3c11 Signed-off-by: Satya Durga Srinivasu Prabhala --- kernel/watchdog.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/kernel/watchdog.c b/kernel/watchdog.c index 3154a257c2bd..378bbb91b52a 100644 --- a/kernel/watchdog.c +++ b/kernel/watchdog.c @@ -519,14 +519,12 @@ void watchdog_enable(unsigned int cpu) void watchdog_disable(unsigned int cpu) { - struct hrtimer *hrtimer = this_cpu_ptr(&watchdog_hrtimer); - unsigned int *enabled = this_cpu_ptr(&watchdog_en); + struct hrtimer *hrtimer = per_cpu_ptr(&watchdog_hrtimer, cpu); + unsigned int *enabled = per_cpu_ptr(&watchdog_en, cpu); if (!*enabled) return; - WARN_ON_ONCE(cpu != smp_processor_id()); - /* * Disable the perf event first. That prevents that a large delay * between disabling the timer and disabling the perf event causes @@ -534,7 +532,7 @@ void watchdog_disable(unsigned int cpu) */ watchdog_nmi_disable(cpu); hrtimer_cancel(hrtimer); - wait_for_completion(this_cpu_ptr(&softlockup_completion)); + wait_for_completion(per_cpu_ptr(&softlockup_completion, cpu)); /* * No need for barrier here since disabling the watchdog is From 7b456d0f2c0eba48c737a7fac1587ddf9cd72803 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Tue, 17 Sep 2019 09:23:19 -0700 Subject: [PATCH 3/4] sched: Add snapshot of task boost feature This snapshot is taken from msm-4.19 as of commit 5debecbe7195 ("trace: filter out spurious preemption and IRQs disable traces"). Change-Id: I3c9663da1fd89e9e942831fda00a47b4a29ea4e3 Signed-off-by: Satya Durga Srinivasu Prabhala --- fs/proc/base.c | 117 +++++++++++++++++++++++++++++++++++ include/linux/sched.h | 15 +++++ include/trace/events/sched.h | 10 ++- kernel/sched/core.c | 23 +++++++ kernel/sched/fair.c | 49 ++++++++++++--- kernel/sched/sched.h | 19 ++++++ 6 files changed, 222 insertions(+), 11 deletions(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index ed38e70f475a..b057ccdbd9f9 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -2874,6 +2874,121 @@ static int proc_tgid_io_accounting(struct seq_file *m, struct pid_namespace *ns, } #endif /* CONFIG_TASK_IO_ACCOUNTING */ +#ifdef CONFIG_SCHED_WALT +static ssize_t proc_sched_task_boost_read(struct file *file, + char __user *buf, size_t count, loff_t *ppos) +{ + struct task_struct *task = get_proc_task(file_inode(file)); + char buffer[PROC_NUMBUF]; + int sched_boost; + size_t len; + + if (!task) + return -ESRCH; + sched_boost = task->boost; + put_task_struct(task); + len = scnprintf(buffer, sizeof(buffer), "%d\n", sched_boost); + return simple_read_from_buffer(buf, count, ppos, buffer, len); +} + +static ssize_t proc_sched_task_boost_write(struct file *file, + const char __user *buf, size_t count, loff_t *ppos) +{ + struct task_struct *task = get_proc_task(file_inode(file)); + char buffer[PROC_NUMBUF]; + int sched_boost; + int err; + + if (!task) + return -ESRCH; + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtoint(strstrip(buffer), 0, &sched_boost); + if (err) + goto out; + if (sched_boost < TASK_BOOST_NONE || sched_boost >= TASK_BOOST_END) { + err = -EINVAL; + goto out; + } + + task->boost = sched_boost; + if (sched_boost == 0) + task->boost_period = 0; +out: + put_task_struct(task); + return err < 0 ? err : count; +} + +static ssize_t proc_sched_task_boost_period_read(struct file *file, + char __user *buf, size_t count, loff_t *ppos) +{ + struct task_struct *task = get_proc_task(file_inode(file)); + char buffer[PROC_NUMBUF]; + u64 sched_boost_period_ms = 0; + size_t len; + + if (!task) + return -ESRCH; + sched_boost_period_ms = div64_ul(task->boost_period, 1000000UL); + put_task_struct(task); + len = snprintf(buffer, sizeof(buffer), "%llu\n", sched_boost_period_ms); + return simple_read_from_buffer(buf, count, ppos, buffer, len); +} + +static ssize_t proc_sched_task_boost_period_write(struct file *file, + const char __user *buf, size_t count, loff_t *ppos) +{ + struct task_struct *task = get_proc_task(file_inode(file)); + char buffer[PROC_NUMBUF]; + unsigned int sched_boost_period; + int err; + + if (!task) + return -ESRCH; + + memset(buffer, 0, sizeof(buffer)); + if (count > sizeof(buffer) - 1) + count = sizeof(buffer) - 1; + if (copy_from_user(buffer, buf, count)) { + err = -EFAULT; + goto out; + } + + err = kstrtouint(strstrip(buffer), 0, &sched_boost_period); + if (err) + goto out; + if (task->boost == 0 && sched_boost_period) { + /* setting boost period without boost is invalid */ + err = -EINVAL; + goto out; + } + + task->boost_period = (u64)sched_boost_period * 1000 * 1000; + task->boost_expires = sched_clock() + task->boost_period; +out: + put_task_struct(task); + return err < 0 ? err : count; +} + +static const struct file_operations proc_task_boost_enabled_operations = { + .read = proc_sched_task_boost_read, + .write = proc_sched_task_boost_write, + .llseek = generic_file_llseek, +}; + +static const struct file_operations proc_task_boost_period_operations = { + .read = proc_sched_task_boost_period_read, + .write = proc_sched_task_boost_period_write, + .llseek = generic_file_llseek, +}; +#endif /* CONFIG_SCHED_WALT */ + #ifdef CONFIG_USER_NS static int proc_id_map_open(struct inode *inode, struct file *file, const struct seq_operations *seq_ops) @@ -3067,6 +3182,8 @@ static const struct pid_entry tgid_base_stuff[] = { REG("sched_init_task_load", 00644, proc_pid_sched_init_task_load_operations), REG("sched_group_id", 00666, proc_pid_sched_group_id_operations), + REG("sched_boost", 0666, proc_task_boost_enabled_operations), + REG("sched_boost_period_ms", 0666, proc_task_boost_period_operations), #endif #ifdef CONFIG_SCHED_DEBUG REG("sched", S_IRUGO|S_IWUSR, proc_pid_sched_operations), diff --git a/include/linux/sched.h b/include/linux/sched.h index 19a379613a94..0e92ee6464e5 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -129,6 +129,14 @@ enum fps { FPS120 = 120, }; +enum task_boost_type { + TASK_BOOST_NONE = 0, + TASK_BOOST_ON_MID, + TASK_BOOST_ON_MAX, + TASK_BOOST_STRICT_MAX, + TASK_BOOST_END, +}; + #ifdef CONFIG_DEBUG_ATOMIC_SLEEP /* @@ -540,6 +548,7 @@ extern void __weak sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, u32 fmax); extern void __weak free_task_load_ptrs(struct task_struct *p); extern void __weak sched_set_refresh_rate(enum fps fps); +extern int set_task_boost(int boost, u64 period); #define RAVG_HIST_SIZE_MAX 5 #define NUM_BUSY_BUCKETS 10 @@ -607,6 +616,8 @@ static inline void sched_update_cpu_freq_min_max(const cpumask_t *cpus, u32 fmin, u32 fmax) { } static inline void sched_set_refresh_rate(enum fps fps) { } + +static inline void set_task_boost(int boost, u64 period) { } #endif /* CONFIG_SCHED_WALT */ struct sched_rt_entity { @@ -806,7 +817,11 @@ struct task_struct { const struct sched_class *sched_class; struct sched_entity se; struct sched_rt_entity rt; + #ifdef CONFIG_SCHED_WALT + int boost; + u64 boost_period; + u64 boost_expires; u64 last_sleep_ts; bool wake_up_idle; struct ravg ravg; diff --git a/include/trace/events/sched.h b/include/trace/events/sched.h index 756a1efb89b7..0a1ec7cc06fd 100644 --- a/include/trace/events/sched.h +++ b/include/trace/events/sched.h @@ -1017,6 +1017,7 @@ TRACE_EVENT(sched_task_util, __field(int, start_cpu) __field(int, unfilter) __field(unsigned long, cpus_allowed) + __field(int, task_boost) ), TP_fast_assign( @@ -1042,15 +1043,20 @@ TRACE_EVENT(sched_task_util, #endif __entry->cpus_allowed = cpumask_bits(&p->cpus_mask)[0]; +#ifdef CONFIG_SCHED_WALT + __entry->task_boost = per_task_boost(p); +#else + __entry->task_boost = 0; +#endif ), - TP_printk("pid=%d comm=%s util=%lu prev_cpu=%d candidates=%#lx best_energy_cpu=%d sync=%d need_idle=%d fastpath=%d placement_boost=%d latency=%llu stune_boosted=%d is_rtg=%d rtg_skip_min=%d start_cpu=%d unfilter=%d affinity=%lx", + TP_printk("pid=%d comm=%s util=%lu prev_cpu=%d candidates=%#lx best_energy_cpu=%d sync=%d need_idle=%d fastpath=%d placement_boost=%d latency=%llu stune_boosted=%d is_rtg=%d rtg_skip_min=%d start_cpu=%d unfilter=%d affinity=%lx task_boost=%d", __entry->pid, __entry->comm, __entry->util, __entry->prev_cpu, __entry->candidates, __entry->best_energy_cpu, __entry->sync, __entry->need_idle, __entry->fastpath, __entry->placement_boost, __entry->latency, __entry->stune_boosted, __entry->is_rtg, __entry->rtg_skip_min, __entry->start_cpu, - __entry->unfilter, __entry->cpus_allowed) + __entry->unfilter, __entry->cpus_allowed, __entry->task_boost) ) /* diff --git a/kernel/sched/core.c b/kernel/sched/core.c index 2319d36a38e7..ba94863d96aa 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -2751,6 +2751,9 @@ static void __sched_fork(unsigned long clone_flags, struct task_struct *p) #ifdef CONFIG_SCHED_WALT p->last_sleep_ts = 0; p->wake_up_idle = false; + p->boost = 0; + p->boost_expires = 0; + p->boost_period = 0; #endif INIT_LIST_HEAD(&p->se.group_node); @@ -8392,6 +8395,26 @@ void dequeue_task_core(struct rq *rq, struct task_struct *p, int flags) } #ifdef CONFIG_SCHED_WALT +/* + *@boost:should be 0,1,2. + *@period:boost time based on ms units. + */ +int set_task_boost(int boost, u64 period) +{ + if (boost < TASK_BOOST_NONE || boost >= TASK_BOOST_END) + return -EINVAL; + if (boost) { + current->boost = boost; + current->boost_period = (u64)period * 1000 * 1000; + current->boost_expires = sched_clock() + current->boost_period; + } else { + current->boost = 0; + current->boost_expires = 0; + current->boost_period = 0; + } + return 0; +} + void sched_account_irqtime(int cpu, struct task_struct *curr, u64 delta, u64 wallclock) { diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 6cd50bb67b7d..6fe3dd16d027 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -3848,7 +3848,7 @@ static inline bool task_fits_max(struct task_struct *p, int cpu) { unsigned long capacity = capacity_orig_of(cpu); unsigned long max_capacity = cpu_rq(cpu)->rd->max_cpu_capacity.val; - unsigned long task_boost = 0; + unsigned long task_boost = per_task_boost(p); if (capacity == max_capacity) return true; @@ -3858,7 +3858,7 @@ static inline bool task_fits_max(struct task_struct *p, int cpu) task_boost > 0) return false; } else { /* mid cap cpu */ - if (task_boost > 1) + if (task_boost > TASK_BOOST_ON_MID) return false; } @@ -3883,6 +3883,7 @@ struct find_best_target_env { bool boosted; int fastpath; int start_cpu; + bool strict_max; }; static inline void adjust_cpus_for_packing(struct task_struct *p, @@ -6341,8 +6342,9 @@ static int get_start_cpu(struct task_struct *p) #ifdef CONFIG_SCHED_WALT struct root_domain *rd = cpu_rq(smp_processor_id())->rd; int start_cpu = rd->min_cap_orig_cpu; - int task_boost = 0; - bool boosted = task_boost_policy(p) == SCHED_BOOST_ON_BIG; + int task_boost = per_task_boost(p); + bool boosted = task_boost_policy(p) == SCHED_BOOST_ON_BIG || + task_boost == TASK_BOOST_ON_MID; bool task_skip_min = task_skip_min_cpu(p); /* @@ -6350,12 +6352,12 @@ static int get_start_cpu(struct task_struct *p) * or just mid will be -1, there never be any other combinations of -1s * beyond these */ - if (task_skip_min || boosted || task_boost == 1) { + if (task_skip_min || boosted) { start_cpu = rd->mid_cap_orig_cpu == -1 ? rd->max_cap_orig_cpu : rd->mid_cap_orig_cpu; } - if (task_boost == 2) { + if (task_boost > TASK_BOOST_ON_MID) { start_cpu = rd->max_cap_orig_cpu; return start_cpu; } @@ -6420,6 +6422,9 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, if (boosted) target_capacity = 0; + if (fbt_env->strict_max) + most_spare_wake_cap = LONG_MIN; + /* Find start CPU based on boost value */ start_cpu = fbt_env->start_cpu; /* Find SD for the start CPU */ @@ -6482,6 +6487,9 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, most_spare_cap_cpu = i; } + if (per_task_boost(cpu_rq(i)->curr) == + TASK_BOOST_STRICT_MAX) + continue; /* * Cumulative demand may already be accounting for the * task. If so, add just the boost-utilization to @@ -6627,7 +6635,8 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, * unless the task can't be accommodated in the higher * capacity CPUs. */ - if (boosted && (best_idle_cpu != -1 || target_cpu != -1)) { + if (boosted && (best_idle_cpu != -1 || target_cpu != -1 || + (fbt_env->strict_max && most_spare_cap_cpu != -1))) { if (boosted) { if (!next_group_higher_cap) break; @@ -6949,7 +6958,8 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, int sync) int placement_boost = task_boost_policy(p); u64 start_t = 0; int delta = 0; - bool boosted = uclamp_boosted(p); + int task_boost = per_task_boost(p); + bool boosted = uclamp_boosted(p) || (task_boost > 0); int start_cpu = get_start_cpu(p); if (start_cpu < 0) @@ -7000,6 +7010,8 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, int sync) fbt_env.need_idle = need_idle; fbt_env.start_cpu = start_cpu; fbt_env.boosted = boosted; + fbt_env.strict_max = is_rtg && + (task_boost == TASK_BOOST_STRICT_MAX); find_best_target(NULL, candidates, p, &fbt_env); @@ -7962,6 +7974,16 @@ static inline int migrate_degrades_locality(struct task_struct *p, } #endif +static inline bool can_migrate_boosted_task(struct task_struct *p, + int src_cpu, int dst_cpu) +{ + if (per_task_boost(p) == TASK_BOOST_STRICT_MAX && + task_in_related_thread_group(p) && + (capacity_orig_of(dst_cpu) < capacity_orig_of(src_cpu))) + return false; + return true; +} + /* * can_migrate_task - may task p from runqueue rq be migrated to this_cpu? */ @@ -7982,6 +8004,12 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env) if (throttled_lb_pair(task_group(p), env->src_cpu, env->dst_cpu)) return 0; + /* + * don't allow pull boost task to smaller cores. + */ + if (!can_migrate_boosted_task(p, env->src_cpu, env->dst_cpu)) + return 0; + if (!cpumask_test_cpu(env->dst_cpu, p->cpus_ptr)) { int cpu; @@ -10041,7 +10069,10 @@ no_move: * if the curr task on busiest CPU can't be * moved to this_cpu: */ - if (!cpumask_test_cpu(this_cpu, busiest->curr->cpus_ptr)) { + if (!cpumask_test_cpu(this_cpu, + busiest->curr->cpus_ptr) || + !can_migrate_boosted_task(busiest->curr, + cpu_of(busiest), this_cpu)) { raw_spin_unlock_irqrestore(&busiest->lock, flags); env.flags |= LBF_ALL_PINNED; diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index 4e3dd5190b26..a9a9223eee7d 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -2205,6 +2205,25 @@ static inline unsigned long capacity_orig_of(int cpu) return cpu_rq(cpu)->cpu_capacity_orig; } +#ifdef CONFIG_SCHED_WALT +static inline int per_task_boost(struct task_struct *p) +{ + if (p->boost_period) { + if (sched_clock() > p->boost_expires) { + p->boost_period = 0; + p->boost_expires = 0; + p->boost = 0; + } + } + return p->boost; +} +#else +static inline int per_task_boost(struct task_struct *p) +{ + return 0; +} +#endif + static inline unsigned long task_util(struct task_struct *p) { #ifdef CONFIG_SCHED_WALT From de785069ecc7ebd63bf1f3b9448409f4fc59adb4 Mon Sep 17 00:00:00 2001 From: Thara Gopinath Date: Fri, 23 Jun 2017 10:37:05 +0530 Subject: [PATCH 4/4] ANDROID: sched: Per-Sched-domain over utilization The current implementation of overutilization, aborts energy aware scheduling if any cpu in the system is over-utilized. This patch introduces over utilization flag per sched domain level instead of a single flag system wide. Load balancing is done at the sched domain where any of the cpu is over utilized. If energy aware scheduling is enabled and no cpu in a sched domain is overuttilized, load balancing is skipped for that sched domain and energy aware scheduling continues at that level. The implementation takes advantage of the shared sched_domain structure that is common across all the sched domains at a level. The new flag introduced is placed in this structure so that all the sched domains the same level share the flag. In case of an overutilized cpu, the flag gets set at level1 sched_domain. The flag at the parent sched_domain level gets set in either of the two following scenarios. 1. There is a misfit task in one of the cpu's in this sched_domain. 2. The total utilization of the domain is greater than the domain capacity The flag is cleared if no cpu in a sched domain is overutilized. This implementation still can have corner scenarios with respect to misfit tasks. For example consider a sched group with n cpus and n+1 70%utilized tasks. Ideally this is a case for load balance to happen in a parent sched domain. But neither the total group utilization is high enough for the load balance to be triggered in the parent domain nor there is a cpu with a single overutilized task so that aload balance is triggered in a parent domain. But again this could be a purely academic sceanrio, as during task wake up these tasks will be placed more appropriately. Signed-off-by: Thara Gopinath Change-Id: I3f327cff4080096a3e58208dd72c9b7f7913cdb2 Signed-off-by: Chris Redpath Git-commit: addef37808728c719d8c095a75bcf81befdacdaf Git-repo: https://android.googlesource.com/kernel/common/ [clingutla@codeaurora.org: Resolved trivial merge conflicts] Signed-off-by: Lingutla Chandrasekhar [satyap@codeaurora.org: port to 5.x and resolve trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/sched/topology.h | 4 ++ kernel/sched/fair.c | 92 +++++++++++++++++++++++++++++++++- kernel/sched/sched.h | 1 + kernel/sched/topology.c | 12 ++--- 4 files changed, 99 insertions(+), 10 deletions(-) diff --git a/include/linux/sched/topology.h b/include/linux/sched/topology.h index f341163fedc9..2fedf310a704 100644 --- a/include/linux/sched/topology.h +++ b/include/linux/sched/topology.h @@ -66,6 +66,10 @@ struct sched_domain_shared { atomic_t ref; atomic_t nr_busy_cpus; int has_idle_cores; + +#ifdef CONFIG_SCHED_WALT + bool overutilized; +#endif }; struct sched_domain { diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 6fe3dd16d027..297c4b0242d7 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -5297,12 +5297,40 @@ bool cpu_overutilized(int cpu) return __cpu_overutilized(cpu, 0); } +#ifdef CONFIG_SCHED_WALT +static bool sd_overutilized(struct sched_domain *sd) +{ + return sd->shared->overutilized; +} + +static void set_sd_overutilized(struct sched_domain *sd) +{ + sd->shared->overutilized = true; +} + +static void clear_sd_overutilized(struct sched_domain *sd) +{ + sd->shared->overutilized = false; +} +#endif + static inline void update_overutilized_status(struct rq *rq) { +#ifdef CONFIG_SCHED_WALT + struct sched_domain *sd; + + rcu_read_lock(); + sd = rcu_dereference(rq->sd); + if (sd && !sd_overutilized(sd) && + cpu_overutilized(rq->cpu)) + set_sd_overutilized(sd); + rcu_read_unlock(); +#else if (!READ_ONCE(rq->rd->overutilized) && cpu_overutilized(rq->cpu)) { WRITE_ONCE(rq->rd->overutilized, SG_OVERUTILIZED); trace_sched_overutilized_tp(rq->rd, SG_OVERUTILIZED); } +#endif } #else static inline void update_overutilized_status(struct rq *rq) { } @@ -8047,7 +8075,7 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env) if (static_branch_unlikely(&sched_energy_present)) { struct root_domain *rd = env->dst_rq->rd; - if (rcu_dereference(rd->pd) && !READ_ONCE(rd->overutilized) && + if ((rcu_dereference(rd->pd) && !sd_overutilized(env->sd)) && env->idle == CPU_NEWLY_IDLE && !task_in_related_thread_group(p)) { long util_cum_dst, util_cum_src; @@ -8549,6 +8577,7 @@ struct sd_lb_stats { unsigned long total_running; unsigned long total_load; /* Total load of all groups in sd */ unsigned long total_capacity; /* Total capacity of all groups in sd */ + unsigned long total_util; /* Total util of all groups in sd */ unsigned long avg_load; /* Average load across all groups in sd */ struct sg_lb_stats busiest_stat;/* Statistics of the busiest group */ @@ -8569,6 +8598,7 @@ static inline void init_sd_lb_stats(struct sd_lb_stats *sds) .total_running = 0UL, .total_load = 0UL, .total_capacity = 0UL, + .total_util = 0UL, .busiest_stat = { .avg_load = 0UL, .sum_nr_running = 0, @@ -8944,9 +8974,13 @@ static inline void update_sg_lb_stats(struct lb_env *env, if (nr_running > 1) *sg_status |= SG_OVERLOAD; - if (cpu_overutilized(i)) + if (cpu_overutilized(i)) { *sg_status |= SG_OVERUTILIZED; + if (rq->misfit_task_load) + *sg_status |= SG_HAS_MISFIT_TASK; + } + #ifdef CONFIG_NUMA_BALANCING sgs->nr_numa_running += rq->nr_numa_running; sgs->nr_preferred_running += rq->nr_preferred_running; @@ -9191,6 +9225,7 @@ next_group: sds->total_running += sgs->sum_nr_running; sds->total_load += sgs->group_load; sds->total_capacity += sgs->group_capacity; + sds->total_util += sgs->group_util; trace_sched_load_balance_sg_stats(sg->cpumask[0], sgs->group_type, sgs->idle_cpus, @@ -9223,6 +9258,7 @@ next_group: /* update overload indicator if we are at root domain */ WRITE_ONCE(rd->overload, sg_status & SG_OVERLOAD); +#ifndef CONFIG_SCHED_WALT /* Update over-utilization (tipping point, U >= 0) indicator */ WRITE_ONCE(rd->overutilized, sg_status & SG_OVERUTILIZED); trace_sched_overutilized_tp(rd, sg_status & SG_OVERUTILIZED); @@ -9231,7 +9267,50 @@ next_group: WRITE_ONCE(rd->overutilized, SG_OVERUTILIZED); trace_sched_overutilized_tp(rd, SG_OVERUTILIZED); +#endif } + +#ifdef CONFIG_SCHED_WALT + if (sg_status & SG_OVERUTILIZED) + set_sd_overutilized(env->sd); + else + clear_sd_overutilized(env->sd); + + /* + * If there is a misfit task in one cpu in this sched_domain + * it is likely that the imbalance cannot be sorted out among + * the cpu's in this sched_domain. In this case set the + * overutilized flag at the parent sched_domain. + */ + if (sg_status & SG_HAS_MISFIT_TASK) { + struct sched_domain *sd = env->sd->parent; + + /* + * In case of a misfit task, load balance at the parent + * sched domain level will make sense only if the the cpus + * have a different capacity. If cpus at a domain level have + * the same capacity, the misfit task cannot be well + * accomodated in any of the cpus and there in no point in + * trying a load balance at this level + */ + while (sd) { + if (sd->flags & SD_ASYM_CPUCAPACITY) { + set_sd_overutilized(sd); + break; + } + sd = sd->parent; + } + } + + /* + * If the domain util is greater that domain capacity, load balancing + * needs to be done at the next sched domain level as well. + */ + if (env->sd->parent && + sds->total_capacity * 1024 < sds->total_util * + sched_capacity_margin_up[group_first_cpu(sds->local)]) + set_sd_overutilized(env->sd->parent); +#endif } /** @@ -9523,7 +9602,11 @@ static struct sched_group *find_busiest_group(struct lb_env *env) if (sched_energy_enabled()) { struct root_domain *rd = env->dst_rq->rd; +#ifdef CONFIG_SCHED_WALT + if (rcu_dereference(rd->pd) && !sd_overutilized(env->sd)) { +#else if (rcu_dereference(rd->pd) && !READ_ONCE(rd->overutilized)) { +#endif int cpu_local, cpu_busiest; unsigned long capacity_local, capacity_busiest; @@ -10389,6 +10472,11 @@ static void rebalance_domains(struct rq *rq, enum cpu_idle_type idle) } max_cost += sd->max_newidle_lb_cost; +#ifdef CONFIG_SCHED_WALT + if (!sd_overutilized(sd)) + continue; +#endif + if (!(sd->flags & SD_LOAD_BALANCE)) continue; diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index a9a9223eee7d..ddea10411134 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -822,6 +822,7 @@ struct max_cpu_capacity { /* Scheduling group status flags */ #define SG_OVERLOAD 0x1 /* More than one runnable task on a CPU. */ #define SG_OVERUTILIZED 0x2 /* One or more CPUs are over-utilized. */ +#define SG_HAS_MISFIT_TASK 0x4 /* Group has misfit task. */ /* * We add the notion of a root-domain which will be used to define per-domain diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c index 54bb65d0f733..d686dd418194 100644 --- a/kernel/sched/topology.c +++ b/kernel/sched/topology.c @@ -1441,15 +1441,11 @@ sd_init(struct sched_domain_topology_level *tl, sd->cache_nice_tries = 1; } - /* - * For all levels sharing cache; connect a sched_domain_shared - * instance. - */ - if (sd->flags & SD_SHARE_PKG_RESOURCES) { - sd->shared = *per_cpu_ptr(sdd->sds, sd_id); - atomic_inc(&sd->shared->ref); + sd->shared = *per_cpu_ptr(sdd->sds, sd_id); + atomic_inc(&sd->shared->ref); + + if (sd->flags & SD_SHARE_PKG_RESOURCES) atomic_set(&sd->shared->nr_busy_cpus, sd_weight); - } sd->private = sdd;