From 538a6d664f0f886c57e709919ecacbc44d7e3bb4 Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Fri, 19 Jun 2020 13:25:18 +0530 Subject: [PATCH 1/2] sched/fair: Tighten prefer_spread feature This patch tightens the prefer_spread feature by doing the following. (1) While picking the busiest group in update_sd_pick_busiest(), if the current group and busiest group are classified as same, use number of runnable tasks to break the ties. Use group load as the next tie breaker. Otherwise we may end up selecting the group with more utilization but with just 1 task. (2) Ignore average load checks when the load balancing CPU is idle and prefer_spread is set. (3) Allow no-hz idle balance CPUs to pull the tasks when the sched domain is not over-utilized but prefer_spread is set. (4) There are cases in calculate_imbalance() that skip imbalance override check due to which task are not getting pulled. Move this check to outside of calculate_imbalance() and set the imbalance to half of the group load. (5) when the weighted CPU load is 0, find_busiest_queue() can't find the busiest rq. Fix this as well. Change-Id: I93d1a62cbd4be34af993ae664a398aa868d29a0c Signed-off-by: Pavankumar Kondeti --- kernel/sched/fair.c | 51 ++++++++++++++++++++++++++------------------- 1 file changed, 30 insertions(+), 21 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index aa9c56bfc1e2..eb5cf294484a 100755 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -9147,10 +9147,19 @@ static bool update_sd_pick_busiest(struct lb_env *env, if (sgs->group_type < busiest->group_type) return false; - if (env->prefer_spread && env->idle != CPU_NOT_IDLE && - (sgs->sum_nr_running > busiest->sum_nr_running) && - (sgs->group_util > busiest->group_util)) - return true; + /* + * This sg and busiest are classified as same. when prefer_spread + * is true, we want to maximize the chance of pulling taks, so + * prefer to pick sg with more runnable tasks and break the ties + * with utilization. + */ + if (env->prefer_spread) { + if (sgs->sum_nr_running < busiest->sum_nr_running) + return false; + if (sgs->sum_nr_running > busiest->sum_nr_running) + return true; + return sgs->group_util > busiest->group_util; + } if (sgs->avg_load <= busiest->avg_load) return false; @@ -9186,10 +9195,6 @@ static bool update_sd_pick_busiest(struct lb_env *env, asym_packing: - if (env->prefer_spread && - (sgs->sum_nr_running < busiest->sum_nr_running)) - return false; - /* This is the busiest node in its class. */ if (!(env->sd->flags & SD_ASYM_PACKING)) return true; @@ -9670,15 +9675,6 @@ static inline void calculate_imbalance(struct lb_env *env, struct sd_lb_stats *s return fix_small_imbalance(env, sds); } - - /* - * If we couldn't find any imbalance, then boost the imbalance - * with the group util. - */ - if (env->prefer_spread && !env->imbalance && - env->idle != CPU_NOT_IDLE && - busiest->sum_nr_running > busiest->group_weight) - env->imbalance = busiest->group_util; } /******* find_busiest_group() helpers end here *********************/ @@ -9718,7 +9714,7 @@ static struct sched_group *find_busiest_group(struct lb_env *env) int cpu_local, cpu_busiest; unsigned long capacity_local, capacity_busiest; - if (env->idle != CPU_NEWLY_IDLE) + if (env->idle != CPU_NEWLY_IDLE && !env->prefer_spread) goto out_balanced; if (!sds.local || !sds.busiest) @@ -9767,9 +9763,13 @@ static struct sched_group *find_busiest_group(struct lb_env *env) /* * When dst_cpu is idle, prevent SMP nice and/or asymmetric group * capacities from resulting in underutilization due to avg_load. + * + * When prefer_spread is enabled, force the balance even when + * busiest group has some capacity but loaded with more than 1 + * task. */ if (env->idle != CPU_NOT_IDLE && group_has_capacity(env, local) && - busiest->group_no_capacity) + (busiest->group_no_capacity || env->prefer_spread)) goto force_balance; /* Misfit tasks should be dealt with regardless of the avg load */ @@ -9815,6 +9815,14 @@ force_balance: /* Looks like there is an imbalance. Compute it */ env->src_grp_type = busiest->group_type; calculate_imbalance(env, &sds); + + /* + * If we couldn't find any imbalance, then boost the imbalance + * based on the group util. + */ + if (!env->imbalance && env->prefer_spread) + env->imbalance = (busiest->group_util >> 1); + trace_sched_load_balance_stats(sds.busiest->cpumask[0], busiest->group_type, busiest->avg_load, busiest->load_per_task, sds.local->cpumask[0], @@ -9924,7 +9932,7 @@ static struct rq *find_busiest_queue(struct lb_env *env, * to: load_i * capacity_j > load_j * capacity_i; where j is * our previous maximum. */ - if (load * busiest_capacity > busiest_load * capacity) { + if (load * busiest_capacity >= busiest_load * capacity) { busiest_load = load; busiest_capacity = capacity; busiest = rq; @@ -10073,7 +10081,8 @@ static int load_balance(int this_cpu, struct rq *this_rq, }; #ifdef CONFIG_SCHED_WALT - env.prefer_spread = (prefer_spread_on_idle(this_cpu) && + env.prefer_spread = (idle != CPU_NOT_IDLE && + prefer_spread_on_idle(this_cpu) && !((sd->flags & SD_ASYM_CPUCAPACITY) && !cpumask_test_cpu(this_cpu, &asym_cap_sibling_cpus))); From 611e7f29a553582c07bb7830519542ffccd66e27 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Wed, 3 Jun 2020 20:23:34 -0700 Subject: [PATCH 2/2] sched/fair: Add policy for restricting prefer_spread to newly idle balance Add policy for restricting prefer_spread to newly idle load balance by expanding the tunable range. To allow lower capacity CPUs to do aggressive newly idle load balance: echo 3 > /proc/sys/kernel/sched_prefer_spread To allow bother lower capacity and higher capacity CPUs to do aggressive newly idle load balance: echo 4 > /proc/sys/kernel/sched_prefer_spread Change-Id: Ia62ddb29bdf592a956a9688f277178ef71dee1b3 Signed-off-by: Satya Durga Srinivasu Prabhala [pkondeti@codeaurora.org: The tunable range is expanded] Co-developed-by: Pavankumar Kondeti Signed-off-by: Pavankumar Kondeti --- kernel/sched/fair.c | 32 ++++++++++++++++++++------------ kernel/sysctl.c | 2 +- 2 files changed, 21 insertions(+), 13 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index eb5cf294484a..5b42f1edaf03 100755 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -3909,20 +3909,26 @@ struct find_best_target_env { int skip_cpu; }; -static inline bool prefer_spread_on_idle(int cpu) +static inline bool prefer_spread_on_idle(int cpu, bool new_ilb) { #ifdef CONFIG_SCHED_WALT - if (likely(!sysctl_sched_prefer_spread)) + switch (sysctl_sched_prefer_spread) { + case 1: + return is_min_capacity_cpu(cpu); + case 2: + return true; + case 3: + return (new_ilb && is_min_capacity_cpu(cpu)); + case 4: + return new_ilb; + default: return false; - - if (is_min_capacity_cpu(cpu)) - return sysctl_sched_prefer_spread >= 1; - - return sysctl_sched_prefer_spread > 1; + } #else return false; #endif } + #ifdef CONFIG_SCHED_WALT static inline void walt_adjust_cpus_for_packing(struct task_struct *p, int *target_cpu, int *best_idle_cpu, @@ -3934,7 +3940,7 @@ static inline void walt_adjust_cpus_for_packing(struct task_struct *p, if (*best_idle_cpu == -1 || *target_cpu == -1) return; - if (prefer_spread_on_idle(*best_idle_cpu)) + if (prefer_spread_on_idle(*best_idle_cpu, false)) fbt_env->need_idle |= 2; if (task_rtg_high_prio(p) && walt_nr_rtg_high_prio(*target_cpu) > 0) { @@ -10082,7 +10088,8 @@ static int load_balance(int this_cpu, struct rq *this_rq, #ifdef CONFIG_SCHED_WALT env.prefer_spread = (idle != CPU_NOT_IDLE && - prefer_spread_on_idle(this_cpu) && + prefer_spread_on_idle(this_cpu, + idle == CPU_NEWLY_IDLE) && !((sd->flags & SD_ASYM_CPUCAPACITY) && !cpumask_test_cpu(this_cpu, &asym_cap_sibling_cpus))); @@ -10632,7 +10639,8 @@ static void rebalance_domains(struct rq *rq, enum cpu_idle_type idle) max_cost += sd->max_newidle_lb_cost; #ifdef CONFIG_SCHED_WALT - if (!sd_overutilized(sd) && !prefer_spread_on_idle(cpu)) + if (!sd_overutilized(sd) && !prefer_spread_on_idle(cpu, + idle == CPU_NEWLY_IDLE)) continue; #endif @@ -10892,7 +10900,7 @@ static void nohz_balancer_kick(struct rq *rq) */ if (sched_energy_enabled()) { if (rq->nr_running >= 2 && (cpu_overutilized(cpu) || - prefer_spread_on_idle(cpu))) + prefer_spread_on_idle(cpu, false))) flags = NOHZ_KICK_MASK; goto out; } @@ -11301,7 +11309,7 @@ int newidle_balance(struct rq *this_rq, struct rq_flags *rf) int pulled_task = 0; u64 curr_cost = 0; u64 avg_idle = this_rq->avg_idle; - bool prefer_spread = prefer_spread_on_idle(this_cpu); + bool prefer_spread = prefer_spread_on_idle(this_cpu, true); bool force_lb = (!is_min_capacity_cpu(this_cpu) && silver_has_big_tasks() && (atomic_read(&this_rq->nr_iowait) == 0)); diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 559147b4e88d..162d0c946de7 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -552,7 +552,7 @@ static struct ctl_table kern_table[] = { .mode = 0644, .proc_handler = proc_dointvec_minmax, .extra1 = SYSCTL_ZERO, - .extra2 = &two, + .extra2 = &four, }, { .procname = "walt_rtg_cfs_boost_prio",