From dc8050ca5dd073108fb052bfbf6c6862e148da91 Mon Sep 17 00:00:00 2001 From: Shaleen Agrawal Date: Fri, 14 Feb 2020 10:40:49 -0800 Subject: [PATCH] sched: fair: Improve the Scheduler This change is for general scheduler improvement. Change-Id: I92c13d8e6681adb2655a1dae1b5c92fd0fe32166 Signed-off-by: Shaleen Agrawal --- include/trace/events/sched.h | 41 +- kernel/sched/fair.c | 737 ++++++++++++++++------------------- kernel/sched/features.h | 5 - kernel/sched/walt.h | 3 +- 4 files changed, 356 insertions(+), 430 deletions(-) diff --git a/include/trace/events/sched.h b/include/trace/events/sched.h index 22b58c2a4ab1..e28fd7aaef26 100644 --- a/include/trace/events/sched.h +++ b/include/trace/events/sched.h @@ -998,11 +998,11 @@ TRACE_EVENT(sched_task_util, TP_PROTO(struct task_struct *p, unsigned long candidates, int best_energy_cpu, bool sync, int need_idle, int fastpath, bool placement_boost, u64 start_t, - bool stune_boosted, bool is_rtg, bool rtg_skip_min, + bool uclamp_boosted, bool is_rtg, bool rtg_skip_min, int start_cpu), TP_ARGS(p, candidates, best_energy_cpu, sync, need_idle, fastpath, - placement_boost, start_t, stune_boosted, is_rtg, rtg_skip_min, + placement_boost, start_t, uclamp_boosted, is_rtg, rtg_skip_min, start_cpu), TP_STRUCT__entry( @@ -1018,7 +1018,7 @@ TRACE_EVENT(sched_task_util, __field(int, placement_boost) __field(int, rtg_cpu) __field(u64, latency) - __field(bool, stune_boosted) + __field(bool, uclamp_boosted) __field(bool, is_rtg) __field(bool, rtg_skip_min) __field(int, start_cpu) @@ -1039,7 +1039,7 @@ TRACE_EVENT(sched_task_util, __entry->fastpath = fastpath; __entry->placement_boost = placement_boost; __entry->latency = (sched_clock() - start_t); - __entry->stune_boosted = stune_boosted; + __entry->uclamp_boosted = uclamp_boosted; __entry->is_rtg = is_rtg; __entry->rtg_skip_min = rtg_skip_min; __entry->start_cpu = start_cpu; @@ -1061,7 +1061,7 @@ TRACE_EVENT(sched_task_util, __entry->pid, __entry->comm, __entry->util, __entry->prev_cpu, __entry->candidates, __entry->best_energy_cpu, __entry->sync, __entry->need_idle, __entry->fastpath, __entry->placement_boost, - __entry->latency, __entry->stune_boosted, + __entry->latency, __entry->uclamp_boosted, __entry->is_rtg, __entry->rtg_skip_min, __entry->start_cpu, __entry->unfilter, __entry->cpus_allowed, __entry->task_boost) ) @@ -1073,12 +1073,13 @@ TRACE_EVENT(sched_find_best_target, TP_PROTO(struct task_struct *tsk, unsigned long min_util, int start_cpu, - int best_idle, int best_active, int most_spare_cap, - int target, int backup), + int best_idle, int most_spare_cap, int target, + int order_index, int end_index, + int skip, bool running), TP_ARGS(tsk, min_util, start_cpu, - best_idle, best_active, most_spare_cap, - target, backup), + best_idle, most_spare_cap, target, + order_index, end_index, skip, running), TP_STRUCT__entry( __array(char, comm, TASK_COMM_LEN) @@ -1086,10 +1087,12 @@ TRACE_EVENT(sched_find_best_target, __field(unsigned long, min_util) __field(int, start_cpu) __field(int, best_idle) - __field(int, best_active) __field(int, most_spare_cap) __field(int, target) - __field(int, backup) + __field(int, order_index) + __field(int, end_index) + __field(int, skip) + __field(bool, running) ), TP_fast_assign( @@ -1098,18 +1101,24 @@ TRACE_EVENT(sched_find_best_target, __entry->min_util = min_util; __entry->start_cpu = start_cpu; __entry->best_idle = best_idle; - __entry->best_active = best_active; __entry->most_spare_cap = most_spare_cap; __entry->target = target; - __entry->backup = backup; + __entry->order_index = order_index; + __entry->end_index = end_index; + __entry->skip = skip; + __entry->running = running; ), - TP_printk("pid=%d comm=%s start_cpu=%d best_idle=%d best_active=%d most_spare_cap=%d target=%d backup=%d", + TP_printk("pid=%d comm=%s start_cpu=%d best_idle=%d most_spare_cap=%d target=%d order_index=%d end_index=%d skip=%d running=%d", __entry->pid, __entry->comm, __entry->start_cpu, - __entry->best_idle, __entry->best_active, + __entry->best_idle, __entry->most_spare_cap, - __entry->target, __entry->backup) + __entry->target, + __entry->order_index, + __entry->end_index, + __entry->skip, + __entry->running) ); #endif /* CONFIG_SMP */ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 31faa0dfff3c..ffb00fe1a0ce 100755 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -3897,14 +3897,15 @@ static inline bool task_demand_fits(struct task_struct *p, int cpu) } struct find_best_target_env { - int placement_boost; + bool is_rtg; int need_idle; + bool boosted; int fastpath; int start_cpu; - int skip_cpu; - bool is_rtg; - bool boosted; + int order_index; + int end_index; bool strict_max; + int skip_cpu; }; static inline bool prefer_spread_on_idle(int cpu) @@ -3921,12 +3922,11 @@ static inline bool prefer_spread_on_idle(int cpu) return false; #endif } - -static inline void adjust_cpus_for_packing(struct task_struct *p, +#ifdef CONFIG_SCHED_WALT +static inline void walt_adjust_cpus_for_packing(struct task_struct *p, int *target_cpu, int *best_idle_cpu, int shallowest_idle_cstate, - struct find_best_target_env *fbt_env, - bool boosted) + struct find_best_target_env *fbt_env) { unsigned long tutil, estimated_capacity; @@ -3936,8 +3936,8 @@ static inline void adjust_cpus_for_packing(struct task_struct *p, if (prefer_spread_on_idle(*best_idle_cpu)) fbt_env->need_idle |= 2; - if (fbt_env->need_idle || task_placement_boost_enabled(p) || boosted || - shallowest_idle_cstate <= 0) { + if (fbt_env->need_idle || task_placement_boost_enabled(p) || + fbt_env->boosted || shallowest_idle_cstate <= 0) { *target_cpu = -1; return; } @@ -3963,6 +3963,7 @@ static inline void adjust_cpus_for_packing(struct task_struct *p, if (fbt_env->is_rtg) *best_idle_cpu = -1; } +#endif static inline void update_misfit_status(struct task_struct *p, struct rq *rq) { @@ -6424,7 +6425,7 @@ unsigned long capacity_curr_of(int cpu) } #ifdef CONFIG_SCHED_WALT -static inline bool get_rtg_status(struct task_struct *p) +static inline bool walt_get_rtg_status(struct task_struct *p) { struct related_thread_group *grp; bool ret = false; @@ -6440,77 +6441,59 @@ static inline bool get_rtg_status(struct task_struct *p) return ret; } -static inline bool task_skip_min_cpu(struct task_struct *p) +static inline bool walt_task_skip_min_cpu(struct task_struct *p) { return sched_boost() != CONSERVATIVE_BOOST && - get_rtg_status(p) && p->unfilter; + walt_get_rtg_status(p) && p->unfilter; } -static inline bool is_many_wakeup(int sibling_count_hint) +static inline bool walt_is_many_wakeup(int sibling_count_hint) { return sibling_count_hint >= sysctl_sched_many_wakeup_threshold; } -#else -static inline bool get_rtg_status(struct task_struct *p) +static inline bool walt_target_ok(int target_cpu, int order_index) { - return false; + return !((order_index != num_sched_clusters - 1) && + (cpumask_weight(&cpu_array[order_index][0]) == 1) && + (target_cpu == cpumask_first(&cpu_array[order_index][0]))); } -static inline bool task_skip_min_cpu(struct task_struct *p) +static void walt_get_indicies(struct task_struct *p, int *order_index, + int *end_index, int task_boost) { - return false; -} + int i = 0; -static inline bool is_many_wakeup(int sibling_count_hint) -{ - return false; -} -#endif + *order_index = 0; + *end_index = 0; -static int get_start_cpu(struct task_struct *p) -{ -#ifdef CONFIG_SCHED_WALT - struct root_domain *rd = cpu_rq(smp_processor_id())->rd; - int start_cpu = rd->min_cap_orig_cpu; - int task_boost = per_task_boost(p); - bool boosted = uclamp_boosted(p) || - task_boost_policy(p) == SCHED_BOOST_ON_BIG || - task_boost == TASK_BOOST_ON_MID; - bool task_skip_min = task_skip_min_cpu(p); - - /* - * note about min/mid/max_cap_orig_cpu - either all of them will be -ve - * or just mid will be -1, there never be any other combinations of -1s - * beyond these - */ - if (task_skip_min || boosted) { - start_cpu = rd->mid_cap_orig_cpu == -1 ? - rd->max_cap_orig_cpu : rd->mid_cap_orig_cpu; - } + if (num_sched_clusters <= 1) + return; if (task_boost > TASK_BOOST_ON_MID) { - start_cpu = rd->max_cap_orig_cpu; - return start_cpu; + *order_index = num_sched_clusters - 1; + return; } - if (start_cpu == -1 || start_cpu == rd->max_cap_orig_cpu) - return start_cpu; - - if (start_cpu == rd->min_cap_orig_cpu && - !task_demand_fits(p, start_cpu)) { - start_cpu = rd->mid_cap_orig_cpu == -1 ? - rd->max_cap_orig_cpu : rd->mid_cap_orig_cpu; + if (is_full_throttle_boost()) { + *order_index = num_sched_clusters - 1; + if ((*order_index > 1) && task_demand_fits(p, + cpumask_first(&cpu_array[*order_index][1]))) + *end_index = 1; + return; } - if (start_cpu == rd->mid_cap_orig_cpu && - !task_demand_fits(p, start_cpu)) - start_cpu = rd->max_cap_orig_cpu; + if (task_boost == TASK_BOOST_ON_MID || + task_boost_policy(p) == SCHED_BOOST_ON_BIG || + walt_task_skip_min_cpu(p)) + *order_index = 1; - return start_cpu; -#else - return smp_processor_id(); -#endif + for (i = *order_index ; i < num_sched_clusters - 1; i++) { + if (task_demand_fits(p, cpumask_first(&cpu_array[i][0]))) + break; + } + + *order_index = i; } enum fastpaths { @@ -6520,50 +6503,33 @@ enum fastpaths { MANY_WAKEUP, }; -static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, +static void walt_find_best_target(struct sched_domain *sd, cpumask_t *cpus, struct task_struct *p, struct find_best_target_env *fbt_env) { unsigned long min_util = uclamp_task_util(p); - unsigned long target_capacity = ULONG_MAX; unsigned long target_max_spare_cap = 0; unsigned long best_idle_cuml_util = ULONG_MAX; - bool boosted = fbt_env->boosted; /* Initialise with deepest possible cstate (INT_MAX) */ int shallowest_idle_cstate = INT_MAX; - struct sched_domain *start_sd; - struct sched_group *sg; - int best_active_cpu = -1; int best_idle_cpu = -1; int target_cpu = -1; - int backup_cpu = -1; int i, start_cpu; long spare_wake_cap, most_spare_wake_cap = 0; int most_spare_cap_cpu = -1; int prev_cpu = task_cpu(p); - bool next_group_higher_cap = false; - int isolated_candidate = -1; - - /* - * In most cases, target_capacity tracks capacity_orig of the most - * energy efficient CPU candidate, thus requiring to minimise - * target_capacity. For these cases target_capacity is already - * initialized to ULONG_MAX. However, for boosted tasks we look - * for a high performance CPU, thus requiring to maximise - * target_capacity. In this case we initialise target_capacity to 0. - */ - if (boosted) - target_capacity = 0; - - if (fbt_env->strict_max) - most_spare_wake_cap = LONG_MIN; + int unisolated_candidate = -1; + int order_index = fbt_env->order_index, end_index = fbt_env->end_index; + int cluster; /* Find start CPU based on boost value */ start_cpu = fbt_env->start_cpu; - /* Find SD for the start CPU */ - start_sd = rcu_dereference(per_cpu(sd_asym_cpucapacity, start_cpu)); - if (!start_sd) - goto out; + + if (fbt_env->strict_max) + target_max_spare_cap = LONG_MIN; + + if (p->state == TASK_RUNNING) + most_spare_wake_cap = ULONG_MAX; /* fast path for prev_cpu */ if (((capacity_orig_of(prev_cpu) == capacity_orig_of(start_cpu)) || @@ -6575,14 +6541,13 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, target_cpu = prev_cpu; fbt_env->fastpath = PREV_CPU_FASTPATH; - goto target; + cpumask_set_cpu(target_cpu, cpus); + goto out; } } + for (cluster = 0; cluster < num_sched_clusters; cluster++) { - /* Scan CPUs in all SDs */ - sg = start_sd->groups; - do { - for_each_cpu_and(i, &p->cpus_mask, sched_group_span(sg)) { + for_each_cpu(i, &cpu_array[order_index][cluster]) { unsigned long capacity_orig = capacity_orig_of(i); unsigned long wake_util, new_util, new_util_cuml; long spare_cap; @@ -6590,11 +6555,11 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, trace_sched_cpu_util(i); - if (!cpu_online(i) || cpu_isolated(i)) + if (!cpu_active(i) || cpu_isolated(i)) continue; - if (isolated_candidate == -1) - isolated_candidate = i; + if (unisolated_candidate == -1) + unisolated_candidate = i; /* * This CPU is the target of an active migration that's @@ -6644,7 +6609,7 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, * than the one required to boost the task. */ new_util = max(min_util, new_util); - if (new_util > capacity_orig) + if (!fbt_env->strict_max && new_util > capacity_orig) continue; /* @@ -6658,21 +6623,10 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, idle_idx = idle_get_state_idx(cpu_rq(i)); /* - * Skip processing placement further if we are visiting - * cpus with lower capacity than start cpu - */ - if (capacity_orig < capacity_orig_of(start_cpu)) - continue; - - /* - * Case B) Non latency sensitive tasks on IDLE CPUs. - * * Find an optimal backup IDLE CPU for non latency * sensitive tasks. * * Looking for: - * - minimizing the capacity_orig, - * i.e. preferring LITTLE CPUs * - favoring shallowest idle states * i.e. avoid to wakeup deep-idle CPUs * @@ -6693,17 +6647,14 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, * Prefer shallowest over deeper idle state cpu, * of same capacity cpus. */ - if (capacity_orig == target_capacity && - idle_idx > shallowest_idle_cstate) + if (idle_idx > shallowest_idle_cstate) continue; if (shallowest_idle_cstate == idle_idx && - target_capacity == capacity_orig && (best_idle_cpu == prev_cpu || (i != prev_cpu && new_util_cuml > best_idle_cuml_util))) continue; - target_capacity = capacity_orig; shallowest_idle_cstate = idle_idx; best_idle_cuml_util = new_util_cuml; best_idle_cpu = i; @@ -6716,160 +6667,53 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, if (p->state == TASK_RUNNING) continue; - /* - * Case C) Non latency sensitive tasks on ACTIVE CPUs. - * - * Pack tasks in the most energy efficient capacities. - * - * This task packing strategy prefers more energy - * efficient CPUs (i.e. pack on smaller maximum - * capacity CPUs) while also trying to spread tasks to - * run them all at the lower OPP. - * - * This assumes for example that it's more energy - * efficient to run two tasks on two CPUs at a lower - * OPP than packing both on a single CPU but running - * that CPU at an higher OPP. - * - * Thus, this case keep track of the CPU with the - * smallest maximum capacity and highest spare maximum - * capacity. - */ - /* Favor CPUs with maximum spare capacity */ if (spare_cap < target_max_spare_cap) continue; target_max_spare_cap = spare_cap; - target_capacity = capacity_orig; target_cpu = i; } - next_group_higher_cap = (capacity_orig_of(group_first_cpu(sg)) < - capacity_orig_of(group_first_cpu(sg->next))); - - /* - * If we've found a cpu, but the boost is ON_ALL we continue - * visiting other clusters. If the boost is ON_BIG we visit - * next cluster if they are higher in capacity. If we are - * not in any kind of boost, we break. - * - * And always visit higher capacity group, if solo cpu group - * is not in idle. - */ - if (!boosted && - ((target_cpu != -1 && (sg->group_weight > 1 || - !next_group_higher_cap)) || - best_idle_cpu != -1) && - (fbt_env->placement_boost == SCHED_BOOST_NONE || - !is_full_throttle_boost() || - (fbt_env->placement_boost == SCHED_BOOST_ON_BIG && - !next_group_higher_cap))) + if (best_idle_cpu != -1) break; - /* - * For boosted case, don't iterate lower capacity CPUs - * unless the task can't be accommodated in the higher - * capacity CPUs. - */ - if (boosted && (best_idle_cpu != -1 || target_cpu != -1 || - (fbt_env->strict_max && most_spare_cap_cpu != -1))) { - if (boosted) { - if (!next_group_higher_cap) - break; - } else { - if (next_group_higher_cap) - break; - } - } - - } while (sg = sg->next, sg != start_sd->groups); - - adjust_cpus_for_packing(p, &target_cpu, &best_idle_cpu, - shallowest_idle_cstate, fbt_env, boosted); - - /* - * For non latency sensitive tasks, cases B and C in the previous loop, - * we pick the best IDLE CPU only if we was not able to find a target - * ACTIVE CPU. - * - * Policies priorities: - * - * a) IDLE CPU available: best_idle_cpu - * b) ACTIVE CPU where task fits and has the bigger maximum spare - * capacity (i.e. target_cpu) - * c) ACTIVE CPU with less contention due to other tasks - * (i.e. best_active_cpu) - * - */ - - if (target_cpu == -1) - target_cpu = best_idle_cpu; - else - backup_cpu = best_idle_cpu; - - if (target_cpu == -1 && most_spare_cap_cpu != -1 && - /* ensure we use active cpu for active migration */ - !(p->state == TASK_RUNNING && !idle_cpu(most_spare_cap_cpu))) - target_cpu = most_spare_cap_cpu; - - if (target_cpu == -1 && isolated_candidate != -1 && - cpu_isolated(prev_cpu)) - target_cpu = isolated_candidate; - - if (backup_cpu >= 0) - cpumask_set_cpu(backup_cpu, cpus); - if (target_cpu >= 0) { -target: - cpumask_set_cpu(target_cpu, cpus); + if ((cluster >= end_index) && (target_cpu != -1) && + walt_target_ok(target_cpu, order_index)) + break; } + walt_adjust_cpus_for_packing(p, &target_cpu, &best_idle_cpu, + shallowest_idle_cstate, fbt_env); + + /* + * We set both idle and target as long as they are valid CPUs. + * If we don't find either, then we fallback to most_spare_cap, + * If we don't find most spare cap, we fallback to prev_cpu, + * provided that the prev_cpu is not isolated. + * If the prev_cpu is isolated, we fallback to unisolated_candidate. + */ + + if (unlikely(target_cpu == -1)) { + if (best_idle_cpu != -1) + target_cpu = best_idle_cpu; + else if (most_spare_cap_cpu != -1) + target_cpu = most_spare_cap_cpu; + else if (cpu_isolated(prev_cpu)) + target_cpu = unisolated_candidate; + } + + if (target_cpu != -1) + cpumask_set_cpu(target_cpu, cpus); + if (best_idle_cpu != -1 && target_cpu != best_idle_cpu) + cpumask_set_cpu(best_idle_cpu, cpus); out: trace_sched_find_best_target(p, min_util, start_cpu, - best_idle_cpu, best_active_cpu, - most_spare_cap_cpu, - target_cpu, backup_cpu); + best_idle_cpu, most_spare_cap_cpu, + target_cpu, order_index, end_index, + fbt_env->skip_cpu, p->state == TASK_RUNNING); } -/* - * Predicts what cpu_util(@cpu) would return if @p was migrated (and enqueued) - * to @dst_cpu. - */ -static unsigned long cpu_util_next(int cpu, struct task_struct *p, int dst_cpu) -{ - struct cfs_rq *cfs_rq = &cpu_rq(cpu)->cfs; - unsigned long util_est, util = READ_ONCE(cfs_rq->avg.util_avg); - - /* - * If @p migrates from @cpu to another, remove its contribution. Or, - * if @p migrates from another CPU to @cpu, add its contribution. In - * the other cases, @cpu is not impacted by the migration, so the - * util_avg should already be correct. - */ - if (task_cpu(p) == cpu && dst_cpu != cpu) - sub_positive(&util, task_util(p)); - else if (task_cpu(p) != cpu && dst_cpu == cpu) - util += task_util(p); - - if (sched_feat(UTIL_EST)) { - util_est = READ_ONCE(cfs_rq->avg.util_est.enqueued); - - /* - * During wake-up, the task isn't enqueued yet and doesn't - * appear in the cfs_rq->avg.util_est.enqueued of any rq, - * so just add it (if needed) to "simulate" what will be - * cpu_util() after the task has been enqueued. - */ - if (dst_cpu == cpu) - util_est += _task_util_est(p); - - util = max(util, util_est); - } - - return min(util, capacity_orig_of(cpu)); -} - -#ifdef CONFIG_SCHED_WALT static inline unsigned long cpu_util_next_walt(int cpu, struct task_struct *p, int dst_cpu) { @@ -6911,7 +6755,47 @@ cpu_util_next_walt(int cpu, struct task_struct *p, int dst_cpu) return min_t(unsigned long, util, capacity_orig_of(cpu)); } + +#else +/* + * Predicts what cpu_util(@cpu) would return if @p was migrated (and enqueued) + * to @dst_cpu. + */ +static unsigned long cpu_util_next(int cpu, struct task_struct *p, int dst_cpu) +{ + struct cfs_rq *cfs_rq = &cpu_rq(cpu)->cfs; + unsigned long util_est, util = READ_ONCE(cfs_rq->avg.util_avg); + + /* + * If @p migrates from @cpu to another, remove its contribution. Or, + * if @p migrates from another CPU to @cpu, add its contribution. In + * the other cases, @cpu is not impacted by the migration, so the + * util_avg should already be correct. + */ + if (task_cpu(p) == cpu && dst_cpu != cpu) + sub_positive(&util, task_util(p)); + else if (task_cpu(p) != cpu && dst_cpu == cpu) + util += task_util(p); + + if (sched_feat(UTIL_EST)) { + util_est = READ_ONCE(cfs_rq->avg.util_est.enqueued); + + /* + * During wake-up, the task isn't enqueued yet and doesn't + * appear in the cfs_rq->avg.util_est.enqueued of any rq, + * so just add it (if needed) to "simulate" what will be + * cpu_util() after the task has been enqueued. + */ + if (dst_cpu == cpu) + util_est += _task_util_est(p); + + util = max(util, util_est); + } + + return min(util, capacity_orig_of(cpu)); +} #endif + /* * compute_energy(): Estimates the energy that @pd would consume if @p was * migrated to @dst_cpu. compute_energy() predicts what will be the utilization @@ -7006,8 +6890,6 @@ static inline bool select_cpu_same_energy(int cpu, int best_cpu, int prev_cpu) return idle_cpu(best_cpu); } -static DEFINE_PER_CPU(cpumask_t, energy_cpus); - /* * find_energy_efficient_cpu(): Find most energy-efficient target CPU for the * waking task. find_energy_efficient_cpu() looks for the CPU with maximum @@ -7047,18 +6929,14 @@ static DEFINE_PER_CPU(cpumask_t, energy_cpus); * other use-cases too. So, until someone finds a better way to solve this, * let's keep things simple by re-using the existing slow path. */ +#ifdef CONFIG_SCHED_WALT +static DEFINE_PER_CPU(cpumask_t, energy_cpus); int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, int sync, int sibling_count_hint) { unsigned long prev_delta = ULONG_MAX, best_delta = ULONG_MAX; struct root_domain *rd = cpu_rq(smp_processor_id())->rd; - int max_spare_cap_cpu_ls = prev_cpu, best_idle_cpu = -1; - unsigned long max_spare_cap_ls = 0, target_cap; - unsigned long cpu_cap, util, base_energy = 0; - bool latency_sensitive = false; - unsigned int min_exit_lat = UINT_MAX; int weight, cpu = smp_processor_id(), best_energy_cpu = prev_cpu; - struct cpuidle_state *idle; struct sched_domain *sd; struct perf_domain *pd; unsigned long cur_energy; @@ -7066,16 +6944,20 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, bool is_rtg, curr_is_rtg; struct find_best_target_env fbt_env; bool need_idle = wake_to_idle(p); - int placement_boost = task_boost_policy(p); u64 start_t = 0; int delta = 0; int task_boost = per_task_boost(p); - bool boosted = uclamp_boosted(p) || (task_boost > 0); - int start_cpu = get_start_cpu(p); + bool is_uclamp_boosted = uclamp_boosted(p); + bool boosted = is_uclamp_boosted || (task_boost > 0); + int start_cpu, order_index, end_index; - if (start_cpu < 0) + + if (unlikely(!cpu_array)) goto eas_not_ready; + walt_get_indicies(p, &order_index, &end_index, task_boost); + start_cpu = cpumask_first(&cpu_array[order_index][0]); + is_rtg = task_in_related_thread_group(p); curr_is_rtg = task_in_related_thread_group(cpu_rq(cpu)->curr); @@ -7098,7 +6980,7 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, goto done; } - if (is_many_wakeup(sibling_count_hint) && prev_cpu != cpu && + if (walt_is_many_wakeup(sibling_count_hint) && prev_cpu != cpu && bias_to_this_cpu(p, prev_cpu, start_cpu)) { best_energy_cpu = prev_cpu; fbt_env.fastpath = MANY_WAKEUP; @@ -7124,150 +7006,63 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, if (!task_util_est(p)) goto unlock; - if (sched_feat(FIND_BEST_TARGET)) { - fbt_env.is_rtg = is_rtg; - fbt_env.placement_boost = placement_boost; - fbt_env.start_cpu = start_cpu; - fbt_env.boosted = boosted; - fbt_env.strict_max = is_rtg && - (task_boost == TASK_BOOST_STRICT_MAX); - fbt_env.skip_cpu = is_many_wakeup(sibling_count_hint) ? - cpu : -1; + fbt_env.is_rtg = is_rtg; + fbt_env.start_cpu = start_cpu; + fbt_env.order_index = order_index; + fbt_env.end_index = end_index; + fbt_env.boosted = boosted; + fbt_env.strict_max = is_rtg && + (task_boost == TASK_BOOST_STRICT_MAX); + fbt_env.skip_cpu = walt_is_many_wakeup(sibling_count_hint) ? + cpu : -1; - find_best_target(NULL, candidates, p, &fbt_env); + walt_find_best_target(NULL, candidates, p, &fbt_env); - /* Bail out if no candidate was found. */ - weight = cpumask_weight(candidates); - if (!weight) - goto unlock; + /* Bail out if no candidate was found. */ + weight = cpumask_weight(candidates); + if (!weight) + goto unlock; - /* If there is only one sensible candidate, select it now. */ - cpu = cpumask_first(candidates); - if (weight == 1 && (idle_cpu(cpu) || cpu == prev_cpu)) { + /* If there is only one sensible candidate, select it now. */ + cpu = cpumask_first(candidates); + if (weight == 1 && (idle_cpu(cpu) || cpu == prev_cpu)) { + best_energy_cpu = cpu; + goto unlock; + } + + if (p->state == TASK_WAKING) + delta = task_util(p); + + if (task_placement_boost_enabled(p) || fbt_env.need_idle || + boosted || is_rtg || __cpu_overutilized(prev_cpu, delta) || + !task_fits_max(p, prev_cpu) || cpu_isolated(prev_cpu)) { + best_energy_cpu = cpu; + goto unlock; + } + + if (cpumask_test_cpu(prev_cpu, &p->cpus_mask)) + prev_delta = best_delta = + compute_energy(p, prev_cpu, pd); + else + prev_delta = best_delta = ULONG_MAX; + + /* Select the best candidate energy-wise. */ + for_each_cpu(cpu, candidates) { + if (cpu == prev_cpu) + continue; + + cur_energy = compute_energy(p, cpu, pd); + trace_sched_compute_energy(p, cpu, cur_energy, + prev_delta, best_delta, best_energy_cpu); + + if (cur_energy < best_delta) { + best_delta = cur_energy; best_energy_cpu = cpu; - goto unlock; - } - -#ifdef CONFIG_SCHED_WALT - if (p->state == TASK_WAKING) - delta = task_util(p); -#endif - if (task_placement_boost_enabled(p) || fbt_env.need_idle || - boosted || is_rtg || __cpu_overutilized(prev_cpu, delta) || - !task_fits_max(p, prev_cpu) || cpu_isolated(prev_cpu)) { - best_energy_cpu = cpu; - goto unlock; - } - - if (cpumask_test_cpu(prev_cpu, &p->cpus_mask)) - prev_delta = best_delta = - compute_energy(p, prev_cpu, pd); - else - prev_delta = best_delta = ULONG_MAX; - - /* Select the best candidate energy-wise. */ - for_each_cpu(cpu, candidates) { - if (cpu == prev_cpu) - continue; - - cur_energy = compute_energy(p, cpu, pd); - trace_sched_compute_energy(p, cpu, cur_energy, - prev_delta, best_delta, best_energy_cpu); - - if (cur_energy < best_delta) { + } else if (cur_energy == best_delta) { + if (select_cpu_same_energy(cpu, best_energy_cpu, + prev_cpu)) { best_delta = cur_energy; best_energy_cpu = cpu; - } else if (cur_energy == best_delta) { - if (select_cpu_same_energy(cpu, best_energy_cpu, - prev_cpu)) { - best_delta = cur_energy; - best_energy_cpu = cpu; - } - } - } - - } else { - latency_sensitive = uclamp_latency_sensitive(p); - boosted = uclamp_boosted(p); - target_cap = boosted ? 0 : ULONG_MAX; - - for (; pd; pd = pd->next) { - unsigned long cur_delta, spare_cap, max_spare_cap = 0; - unsigned long base_energy_pd; - int max_spare_cap_cpu = -1; - - /* Compute the 'base' energy of the pd, without @p */ - base_energy_pd = compute_energy(p, -1, pd); - base_energy += base_energy_pd; - - for_each_cpu_and(cpu, perf_domain_span(pd), sched_domain_span(sd)) { - if (!cpumask_test_cpu(cpu, p->cpus_ptr)) - continue; - - util = cpu_util_next(cpu, p, cpu); - cpu_cap = capacity_of(cpu); - spare_cap = cpu_cap - util; - - /* - * Skip CPUs that cannot satisfy the capacity request. - * IOW, placing the task there would make the CPU - * overutilized. Take uclamp into account to see how - * much capacity we can get out of the CPU; this is - * aligned with schedutil_cpu_util(). - */ - util = uclamp_rq_util_with(cpu_rq(cpu), util, p); - if (!fits_capacity(util, cpu_cap)) - continue; - - /* Always use prev_cpu as a candidate. */ - if (!latency_sensitive && cpu == prev_cpu) { - prev_delta = compute_energy(p, prev_cpu, pd); - prev_delta -= base_energy_pd; - best_delta = min(best_delta, prev_delta); - } - - /* - * Find the CPU with the maximum spare capacity in - * the performance domain - */ - if (spare_cap > max_spare_cap) { - max_spare_cap = spare_cap; - max_spare_cap_cpu = cpu; - } - - if (!latency_sensitive) - continue; - - if (idle_cpu(cpu)) { - cpu_cap = capacity_orig_of(cpu); - if (boosted && cpu_cap < target_cap) - continue; - if (!boosted && cpu_cap > target_cap) - continue; - idle = idle_get_state(cpu_rq(cpu)); - if (idle && idle->exit_latency > min_exit_lat && - cpu_cap == target_cap) - continue; - - if (idle) - min_exit_lat = idle->exit_latency; - target_cap = cpu_cap; - best_idle_cpu = cpu; - } else if (spare_cap > max_spare_cap_ls) { - max_spare_cap_ls = spare_cap; - max_spare_cap_cpu_ls = cpu; - } - } - - /* Evaluate the energy impact of using this CPU. */ - if (!latency_sensitive && max_spare_cap_cpu >= 0 && - max_spare_cap_cpu != prev_cpu) { - cur_delta = compute_energy(p, max_spare_cap_cpu, pd); - cur_delta -= base_energy_pd; - if (cur_delta < best_delta) { - best_delta = cur_delta; - best_energy_cpu = max_spare_cap_cpu; - } } } } @@ -7275,9 +7070,6 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, unlock: rcu_read_unlock(); - if (latency_sensitive) - return best_idle_cpu >= 0 ? best_idle_cpu : max_spare_cap_cpu_ls; - /* * Pick the prev CPU, if best energy CPU can't saves at least 6% of * the energy used by prev_cpu. @@ -7291,9 +7083,9 @@ unlock: done: trace_sched_task_util(p, cpumask_bits(candidates)[0], best_energy_cpu, - sync, fbt_env.need_idle, fbt_env.fastpath, - placement_boost, start_t, boosted, is_rtg, - get_rtg_status(p), start_cpu); + sync, need_idle, fbt_env.fastpath, task_boost_policy(p), + start_t, boosted, is_rtg, walt_get_rtg_status(p), + start_cpu); return best_energy_cpu; @@ -7301,9 +7093,138 @@ fail: rcu_read_unlock(); eas_not_ready: - return -1; + return -EINVAL; } +#else +int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, + int sync, __attribute__((unused))int sibling_count_hint) +{ + unsigned long prev_delta = ULONG_MAX, best_delta = ULONG_MAX; + struct root_domain *rd = cpu_rq(smp_processor_id())->rd; + int max_spare_cap_cpu_ls = prev_cpu, best_idle_cpu = -1; + unsigned long max_spare_cap_ls = 0, target_cap; + unsigned long cpu_cap, util, base_energy = 0; + bool boosted, latency_sensitive = false; + unsigned int min_exit_lat = UINT_MAX; + int cpu, best_energy_cpu = prev_cpu; + struct cpuidle_state *idle; + struct sched_domain *sd; + struct perf_domain *pd; + rcu_read_lock(); + pd = rcu_dereference(rd->pd); + if (!pd || READ_ONCE(rd->overutilized)) + goto fail; + cpu = smp_processor_id(); + if (sync && cpu_rq(cpu)->nr_running == 1 && + cpumask_test_cpu(cpu, p->cpus_ptr)) { + rcu_read_unlock(); + return cpu; + } + /* + * Energy-aware wake-up happens on the lowest sched_domain starting + * from sd_asym_cpucapacity spanning over this_cpu and prev_cpu. + */ + sd = rcu_dereference(*this_cpu_ptr(&sd_asym_cpucapacity)); + while (sd && !cpumask_test_cpu(prev_cpu, sched_domain_span(sd))) + sd = sd->parent; + if (!sd) + goto fail; + sync_entity_load_avg(&p->se); + if (!task_util_est(p)) + goto unlock; + latency_sensitive = uclamp_latency_sensitive(p); + boosted = uclamp_boosted(p); + target_cap = boosted ? 0 : ULONG_MAX; + for (; pd; pd = pd->next) { + unsigned long cur_delta, spare_cap, max_spare_cap = 0; + unsigned long base_energy_pd; + int max_spare_cap_cpu = -1; + /* Compute the 'base' energy of the pd, without @p */ + base_energy_pd = compute_energy(p, -1, pd); + base_energy += base_energy_pd; + for_each_cpu_and(cpu, perf_domain_span(pd), + sched_domain_span(sd)) { + if (!cpumask_test_cpu(cpu, p->cpus_ptr)) + continue; + util = cpu_util_next(cpu, p, cpu); + cpu_cap = capacity_of(cpu); + spare_cap = cpu_cap - util; + /* + * Skip CPUs that cannot satisfy the capacity request. + * IOW, placing the task there would make the CPU + * overutilized. Take uclamp into account to see how + * much capacity we can get out of the CPU; this is + * aligned with schedutil_cpu_util(). + */ + util = uclamp_rq_util_with(cpu_rq(cpu), util, p); + if (!fits_capacity(util, cpu_cap)) + continue; + /* Always use prev_cpu as a candidate. */ + if (!latency_sensitive && cpu == prev_cpu) { + prev_delta = compute_energy(p, prev_cpu, pd); + prev_delta -= base_energy_pd; + best_delta = min(best_delta, prev_delta); + } + /* + * Find the CPU with the maximum spare capacity in + * the performance domain + */ + if (spare_cap > max_spare_cap) { + max_spare_cap = spare_cap; + max_spare_cap_cpu = cpu; + } + if (!latency_sensitive) + continue; + if (idle_cpu(cpu)) { + cpu_cap = capacity_orig_of(cpu); + if (boosted && cpu_cap < target_cap) + continue; + if (!boosted && cpu_cap > target_cap) + continue; + idle = idle_get_state(cpu_rq(cpu)); + if (idle && idle->exit_latency > min_exit_lat && + cpu_cap == target_cap) + continue; + if (idle) + min_exit_lat = idle->exit_latency; + target_cap = cpu_cap; + best_idle_cpu = cpu; + } else if (spare_cap > max_spare_cap_ls) { + max_spare_cap_ls = spare_cap; + max_spare_cap_cpu_ls = cpu; + } + } + /* Evaluate the energy impact of using this CPU. */ + if (!latency_sensitive && max_spare_cap_cpu >= 0 && + max_spare_cap_cpu != prev_cpu) { + cur_delta = compute_energy(p, max_spare_cap_cpu, pd); + cur_delta -= base_energy_pd; + if (cur_delta < best_delta) { + best_delta = cur_delta; + best_energy_cpu = max_spare_cap_cpu; + } + } + } +unlock: + rcu_read_unlock(); + if (latency_sensitive) + return best_idle_cpu >= 0 ? + best_idle_cpu : max_spare_cap_cpu_ls; + /* + * Pick the best CPU if prev_cpu cannot be used, or if it saves at + * least 6% of the energy used by prev_cpu. + */ + if (prev_delta == ULONG_MAX) + return best_energy_cpu; + if ((prev_delta - best_delta) > ((prev_delta + base_energy) >> 4)) + return best_energy_cpu; + return prev_cpu; +fail: + rcu_read_unlock(); + return -EINVAL; +} +#endif /* * select_task_rq_fair: Select target runqueue for the waking task in domains * that have the 'sd_flag' flag set. In practice, this is SD_BALANCE_WAKE, diff --git a/kernel/sched/features.h b/kernel/sched/features.h index 9fba55689636..16579ad03aae 100644 --- a/kernel/sched/features.h +++ b/kernel/sched/features.h @@ -90,11 +90,6 @@ SCHED_FEAT(WA_BIAS, true) */ SCHED_FEAT(UTIL_EST, true) -/* - * Fast pre-selection of CPU candidates for EAS. - */ -SCHED_FEAT(FIND_BEST_TARGET, true) - /* * Request max frequency from schedutil whenever a RT task is running. */ diff --git a/kernel/sched/walt.h b/kernel/sched/walt.h index 3fe0197a2672..437d73ee4950 100644 --- a/kernel/sched/walt.h +++ b/kernel/sched/walt.h @@ -14,7 +14,8 @@ #define EXITING_TASK_MARKER 0xdeaddead extern unsigned int __weak walt_rotation_enabled; - +extern int __read_mostly __weak num_sched_clusters; +extern cpumask_t __read_mostly __weak **cpu_array; extern void walt_update_task_ravg(struct task_struct *p, struct rq *rq, int event, u64 wallclock, u64 irqtime);