From eb329acdbb63b25a4a566695f99cd4085f644bfd Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Wed, 16 Oct 2019 12:26:48 +0530 Subject: [PATCH 1/4] sched/fair: reduce no-hz idle balance for energy aware systems We are kicking no-hz idle balance when the CPU has two tasks. It should be done only when the CPU is overutilized on energy aware systems. This approach seems to be causing more number of wakeups on HZ=250 systems compared to HZ=100. Revisit the no-hz idle scheme and only send the IPI if and when it is necessary. Change-Id: I82e58a2aa3b6508c86d05c93c9fb7fe6dfbde500 Signed-off-by: Pavankumar Kondeti [satyap@codeaurora.org: fix trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- kernel/sched/fair.c | 21 ++++++++++++++++----- 1 file changed, 16 insertions(+), 5 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 81a407729481..5ad74026f7a5 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -10615,7 +10615,7 @@ static inline int find_energy_aware_new_ilb(void) int ilb = nr_cpu_ids; struct sched_domain *sd; int cpu = raw_smp_processor_id(); - cpumask_t avail_cpus, tmp_cpus; + cpumask_t idle_cpus, tmp_cpus; struct sched_group *sg; unsigned long ref_cap = capacity_orig_of(cpu); unsigned long best_cap = 0, best_cap_cpu = -1; @@ -10625,16 +10625,16 @@ static inline int find_energy_aware_new_ilb(void) if (!sd) goto out; - cpumask_and(&avail_cpus, nohz.idle_cpus_mask, + cpumask_and(&idle_cpus, nohz.idle_cpus_mask, housekeeping_cpumask(HK_FLAG_MISC)); - cpumask_andnot(&avail_cpus, &avail_cpus, cpu_isolated_mask); + cpumask_andnot(&idle_cpus, &idle_cpus, cpu_isolated_mask); sg = sd->groups; do { int i; unsigned long cap; - cpumask_and(&tmp_cpus, &avail_cpus, sched_group_span(sg)); + cpumask_and(&tmp_cpus, &idle_cpus, sched_group_span(sg)); i = cpumask_first(&tmp_cpus); /* This sg did not have any idle CPUs */ @@ -10649,7 +10649,18 @@ static inline int find_energy_aware_new_ilb(void) break; } - /* The back up CPU is selected from the best capacity CPUs */ + /* + * When there are no idle CPUs in the same capacity group, + * we find the next best capacity CPU. + */ + if (best_cap > ref_cap) { + if (cap > ref_cap && cap < best_cap) { + best_cap = cap; + best_cap_cpu = i; + } + continue; + } + if (cap > best_cap) { best_cap = cap; best_cap_cpu = i; From a21fe9c3d8b49aa0faeefe88bf53408249d0d2ee Mon Sep 17 00:00:00 2001 From: Lingutla Chandrasekhar Date: Thu, 13 Feb 2020 11:38:47 +0530 Subject: [PATCH 2/4] sched: Improve the scheduler This change is for general scheduler improvements. Change-Id: I780f52ebc3cb69ee2c497d2d2f8e8fddd789421d Signed-off-by: Lingutla Chandrasekhar --- kernel/sched/core.c | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/kernel/sched/core.c b/kernel/sched/core.c index b3378fcbc66b..2c20bd870fb3 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -5627,6 +5627,7 @@ unsigned int sched_lib_mask_force; bool is_sched_lib_based_app(pid_t pid) { const char *name = NULL; + char *lib_list, *libname; struct vm_area_struct *vma; char path_buf[LIB_PATH_LENGTH]; bool found = false; @@ -5636,11 +5637,14 @@ bool is_sched_lib_based_app(pid_t pid) if (strnlen(sched_lib_name, LIB_PATH_LENGTH) == 0) return false; + lib_list = kstrdup(sched_lib_name, GFP_KERNEL); + rcu_read_lock(); p = find_process_by_pid(pid); if (!p) { rcu_read_unlock(); + kfree(lib_list); return false; } @@ -5660,10 +5664,12 @@ bool is_sched_lib_based_app(pid_t pid) if (IS_ERR(name)) goto release_sem; - if (strnstr(name, sched_lib_name, + while ((libname = strsep(&lib_list, ","))) { + if (strnstr(name, libname, strnlen(name, LIB_PATH_LENGTH))) { - found = true; - break; + found = true; + break; + } } } } @@ -5673,6 +5679,7 @@ release_sem: mmput(mm); put_task_struct: put_task_struct(p); + kfree(lib_list); return found; } From 2b0e26d3a83bccb797506a545004e85a22d58af5 Mon Sep 17 00:00:00 2001 From: Maria Yu Date: Mon, 15 Apr 2019 12:41:12 +0800 Subject: [PATCH 3/4] sched/fair: Allow load bigger task load balance when nr_running is 2 When there is only 2 tasks in 1 cpu and the other task is currently running, allow load bigger task to be balanced if the other task is currently running. Change-Id: I489e9624ba010f9293272a67585e8209a786b787 Signed-off-by: Maria Yu Signed-off-by: Cong Zhang --- kernel/sched/fair.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 5ad74026f7a5..453417cd450f 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -8282,7 +8282,17 @@ redo: if (sched_feat(LB_MIN) && load < 16 && !env->sd->nr_balance_failed) goto next; - if ((load / 2) > env->imbalance) + /* + * p is not running task when we goes until here, so if p is one + * of the 2 task in src cpu rq and not the running one, + * that means it is the only task that can be balanced. + * So only when there is other tasks can be balanced or + * there is situation to ignore big task, it is needed + * to skip the task load bigger than 2*imbalance. + */ + if (((cpu_rq(env->src_cpu)->nr_running > 2) || + (env->flags & LBF_IGNORE_BIG_TASKS)) && + ((load / 2) > env->imbalance)) goto next; detach_task(p, env); From 50a929516930463bbef4e4a3e4e7bc9a68dfb219 Mon Sep 17 00:00:00 2001 From: Lingutla Chandrasekhar Date: Wed, 8 Jan 2020 10:30:04 +0530 Subject: [PATCH 4/4] sched: Add support to spread tasks If sysctl_sched_prefer_spread is enabled, then tasks would be freely migrated to idle cpus within same cluster to reduce runnables. By default, the feature is disabled. User can trigger feature with: echo 1 > /proc/sys/kernel/sched_prefer_spread Aggressively spread tasks with in little cluster. echo 2 > /proc/sys/kernel/sched_prefer_spread Aggressively spread tasks with in little cluster as well as big cluster, but not between big and little. Change-Id: I0a4d87bd17de3525548765472e6f388a9970f13c Signed-off-by: Lingutla Chandrasekhar [satyap@codeaurora.org: fix trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/sched/sysctl.h | 1 + include/trace/events/sched.h | 14 +++-- kernel/sched/fair.c | 111 ++++++++++++++++++++++++++++------- kernel/sysctl.c | 9 +++ 4 files changed, 108 insertions(+), 27 deletions(-) diff --git a/include/linux/sched/sysctl.h b/include/linux/sched/sysctl.h index a5f514d5a891..c6b3e18273b1 100644 --- a/include/linux/sched/sysctl.h +++ b/include/linux/sched/sysctl.h @@ -53,6 +53,7 @@ extern unsigned int __weak sysctl_sched_window_stats_policy; extern unsigned int __weak sysctl_sched_ravg_window_nr_ticks; extern unsigned int __weak sysctl_sched_many_wakeup_threshold; extern unsigned int __weak sysctl_sched_dynamic_ravg_window_enable; +extern unsigned int sysctl_sched_prefer_spread; extern int walt_proc_group_thresholds_handler(struct ctl_table *table, int write, diff --git a/include/trace/events/sched.h b/include/trace/events/sched.h index 5f2ca2a7cbe3..22b58c2a4ab1 100644 --- a/include/trace/events/sched.h +++ b/include/trace/events/sched.h @@ -276,11 +276,11 @@ TRACE_EVENT(sched_load_balance, unsigned long group_mask, int busiest_nr_running, unsigned long imbalance, unsigned int env_flags, int ld_moved, unsigned int balance_interval, int active_balance, - int overutilized), + int overutilized, int prefer_spread), TP_ARGS(cpu, idle, balance, group_mask, busiest_nr_running, imbalance, env_flags, ld_moved, balance_interval, - active_balance, overutilized), + active_balance, overutilized, prefer_spread), TP_STRUCT__entry( __field(int, cpu) @@ -294,6 +294,7 @@ TRACE_EVENT(sched_load_balance, __field(unsigned int, balance_interval) __field(int, active_balance) __field(int, overutilized) + __field(int, prefer_spread) ), TP_fast_assign( @@ -308,9 +309,10 @@ TRACE_EVENT(sched_load_balance, __entry->balance_interval = balance_interval; __entry->active_balance = active_balance; __entry->overutilized = overutilized; + __entry->prefer_spread = prefer_spread; ), - TP_printk("cpu=%d state=%s balance=%d group=%#lx busy_nr=%d imbalance=%ld flags=%#x ld_moved=%d bal_int=%d active_balance=%d sd_overutilized=%d", + TP_printk("cpu=%d state=%s balance=%d group=%#lx busy_nr=%d imbalance=%ld flags=%#x ld_moved=%d bal_int=%d active_balance=%d sd_overutilized=%d prefer_spread=%d", __entry->cpu, __entry->idle == CPU_IDLE ? "idle" : (__entry->idle == CPU_NEWLY_IDLE ? "newly_idle" : "busy"), @@ -318,7 +320,7 @@ TRACE_EVENT(sched_load_balance, __entry->group_mask, __entry->busiest_nr_running, __entry->imbalance, __entry->env_flags, __entry->ld_moved, __entry->balance_interval, __entry->active_balance, - __entry->overutilized) + __entry->overutilized, __entry->prefer_spread) ); TRACE_EVENT(sched_load_balance_nohz_kick, @@ -994,7 +996,7 @@ TRACE_EVENT(sched_compute_energy, TRACE_EVENT(sched_task_util, TP_PROTO(struct task_struct *p, unsigned long candidates, - int best_energy_cpu, bool sync, bool need_idle, int fastpath, + int best_energy_cpu, bool sync, int need_idle, int fastpath, bool placement_boost, u64 start_t, bool stune_boosted, bool is_rtg, bool rtg_skip_min, int start_cpu), @@ -1011,7 +1013,7 @@ TRACE_EVENT(sched_task_util, __field(int, prev_cpu) __field(int, best_energy_cpu) __field(bool, sync) - __field(bool, need_idle) + __field(int, need_idle) __field(int, fastpath) __field(int, placement_boost) __field(int, rtg_cpu) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 453417cd450f..66cbb043c6fa 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -131,6 +131,9 @@ unsigned int sched_capacity_margin_up[NR_CPUS] = { unsigned int sched_capacity_margin_down[NR_CPUS] = { [0 ... NR_CPUS-1] = 1205}; /* ~15% margin */ +#ifdef CONFIG_SCHED_WALT +__read_mostly unsigned int sysctl_sched_prefer_spread; +#endif unsigned int sched_small_task_threshold = 102; static inline void update_load_add(struct load_weight *lw, unsigned long inc) @@ -3877,16 +3880,31 @@ static inline bool task_demand_fits(struct task_struct *p, int cpu) } struct find_best_target_env { - bool is_rtg; int placement_boost; - bool need_idle; - bool boosted; + int need_idle; int fastpath; int start_cpu; - bool strict_max; int skip_cpu; + bool is_rtg; + bool boosted; + bool strict_max; }; +static inline bool prefer_spread_on_idle(int cpu) +{ +#ifdef CONFIG_SCHED_WALT + if (likely(!sysctl_sched_prefer_spread)) + return false; + + if (is_min_capacity_cpu(cpu)) + return sysctl_sched_prefer_spread >= 1; + + return sysctl_sched_prefer_spread > 1; +#else + return false; +#endif +} + static inline void adjust_cpus_for_packing(struct task_struct *p, int *target_cpu, int *best_idle_cpu, int shallowest_idle_cstate, @@ -3898,8 +3916,11 @@ static inline void adjust_cpus_for_packing(struct task_struct *p, if (*best_idle_cpu == -1 || *target_cpu == -1) return; - if (task_placement_boost_enabled(p) || fbt_env->need_idle || boosted || - shallowest_idle_cstate <= 0) { + if (prefer_spread_on_idle(*best_idle_cpu)) + fbt_env->need_idle |= 2; + + if (fbt_env->need_idle || task_placement_boost_enabled(p) || boosted || + shallowest_idle_cstate <= 0) { *target_cpu = -1; return; } @@ -7024,6 +7045,7 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, curr_is_rtg = task_in_related_thread_group(cpu_rq(cpu)->curr); fbt_env.fastpath = 0; + fbt_env.need_idle = need_idle; if (trace_sched_task_util_enabled()) start_t = sched_clock(); @@ -7070,7 +7092,6 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, if (sched_feat(FIND_BEST_TARGET)) { fbt_env.is_rtg = is_rtg; fbt_env.placement_boost = placement_boost; - fbt_env.need_idle = need_idle; fbt_env.start_cpu = start_cpu; fbt_env.boosted = boosted; fbt_env.strict_max = is_rtg && @@ -7096,9 +7117,9 @@ int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu, if (p->state == TASK_WAKING) delta = task_util(p); #endif - if (task_placement_boost_enabled(p) || need_idle || boosted || - is_rtg || __cpu_overutilized(prev_cpu, delta) || - !task_fits_max(p, prev_cpu) || cpu_isolated(prev_cpu)) { + if (task_placement_boost_enabled(p) || fbt_env.need_idle || + boosted || is_rtg || __cpu_overutilized(prev_cpu, delta) || + !task_fits_max(p, prev_cpu) || cpu_isolated(prev_cpu)) { best_energy_cpu = cpu; goto unlock; } @@ -7231,8 +7252,9 @@ unlock: done: trace_sched_task_util(p, cpumask_bits(candidates)[0], best_energy_cpu, - sync, need_idle, fbt_env.fastpath, placement_boost, - start_t, boosted, is_rtg, get_rtg_status(p), start_cpu); + sync, fbt_env.need_idle, fbt_env.fastpath, + placement_boost, start_t, boosted, is_rtg, + get_rtg_status(p), start_cpu); return best_energy_cpu; @@ -7946,6 +7968,7 @@ struct lb_env { unsigned int loop; unsigned int loop_break; unsigned int loop_max; + bool prefer_spread; enum fbq_type fbq_type; enum group_type src_grp_type; @@ -8119,8 +8142,8 @@ int can_migrate_task(struct task_struct *p, struct lb_env *env) struct root_domain *rd = env->dst_rq->rd; if ((rcu_dereference(rd->pd) && !sd_overutilized(env->sd)) && - env->idle == CPU_NEWLY_IDLE && - !task_in_related_thread_group(p)) { + env->idle == CPU_NEWLY_IDLE && !env->prefer_spread && + !task_in_related_thread_group(p)) { long util_cum_dst, util_cum_src; unsigned long demand; @@ -8289,8 +8312,12 @@ redo: * So only when there is other tasks can be balanced or * there is situation to ignore big task, it is needed * to skip the task load bigger than 2*imbalance. + * + * And load based checks are skipped for prefer_spread in + * finding busiest group, ignore the task's h_load. */ - if (((cpu_rq(env->src_cpu)->nr_running > 2) || + if (!env->prefer_spread && + ((cpu_rq(env->src_cpu)->nr_running > 2) || (env->flags & LBF_IGNORE_BIG_TASKS)) && ((load / 2) > env->imbalance)) goto next; @@ -9092,6 +9119,11 @@ static bool update_sd_pick_busiest(struct lb_env *env, if (sgs->group_type < busiest->group_type) return false; + if (env->prefer_spread && env->idle != CPU_NOT_IDLE && + (sgs->sum_nr_running > busiest->sum_nr_running) && + (sgs->group_util > busiest->group_util)) + return true; + if (sgs->avg_load <= busiest->avg_load) return false; @@ -9125,6 +9157,11 @@ static bool update_sd_pick_busiest(struct lb_env *env, return false; asym_packing: + + if (env->prefer_spread && + (sgs->sum_nr_running < busiest->sum_nr_running)) + return false; + /* This is the busiest node in its class. */ if (!(env->sd->flags & SD_ASYM_PACKING)) return true; @@ -9605,6 +9642,15 @@ static inline void calculate_imbalance(struct lb_env *env, struct sd_lb_stats *s return fix_small_imbalance(env, sds); } + + /* + * If we couldn't find any imbalance, then boost the imbalance + * with the group util. + */ + if (env->prefer_spread && !env->imbalance && + env->idle != CPU_NOT_IDLE && + busiest->sum_nr_running > busiest->group_weight) + env->imbalance = busiest->group_util; } /******* find_busiest_group() helpers end here *********************/ @@ -9998,6 +10044,15 @@ static int load_balance(int this_cpu, struct rq *this_rq, .loop = 0, }; +#ifdef CONFIG_SCHED_WALT + env.prefer_spread = (prefer_spread_on_idle(this_cpu) && + !((sd->flags & SD_ASYM_CPUCAPACITY) && + !cpumask_test_cpu(this_cpu, + &asym_cap_sibling_cpus))); +#else + env.prefer_spread = false; +#endif + cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask); schedstat_inc(sd->lb_count[idle]); @@ -10287,10 +10342,11 @@ out: env.imbalance, env.flags, ld_moved, sd->balance_interval, active_balance, #ifdef CONFIG_SCHED_WALT - sd_overutilized(sd)); + sd_overutilized(sd), #else - READ_ONCE(this_rq->rd->overutilized)); + READ_ONCE(this_rq->rd->overutilized), #endif + env.prefer_spread); return ld_moved; } @@ -10535,7 +10591,7 @@ static void rebalance_domains(struct rq *rq, enum cpu_idle_type idle) max_cost += sd->max_newidle_lb_cost; #ifdef CONFIG_SCHED_WALT - if (!sd_overutilized(sd)) + if (!sd_overutilized(sd) && !prefer_spread_on_idle(cpu)) continue; #endif @@ -10782,7 +10838,8 @@ static void nohz_balancer_kick(struct rq *rq) * happens from the tickpath. */ if (sched_energy_enabled()) { - if (rq->nr_running >= 2 && cpu_overutilized(cpu)) + if (rq->nr_running >= 2 && (cpu_overutilized(cpu) || + prefer_spread_on_idle(cpu))) flags = NOHZ_KICK_MASK; goto out; } @@ -11187,6 +11244,11 @@ int newidle_balance(struct rq *this_rq, struct rq_flags *rf) int pulled_task = 0; u64 curr_cost = 0; u64 avg_idle = this_rq->avg_idle; + bool prefer_spread = prefer_spread_on_idle(this_cpu); + bool force_lb = (!is_min_capacity_cpu(this_cpu) && + silver_has_big_tasks() && + (atomic_read(&this_rq->nr_iowait) == 0)); + if (cpu_isolated(this_cpu)) return 0; @@ -11203,8 +11265,8 @@ int newidle_balance(struct rq *this_rq, struct rq_flags *rf) */ if (!cpu_active(this_cpu)) return 0; - if (!is_min_capacity_cpu(this_cpu) && silver_has_big_tasks() - && (atomic_read(&this_rq->nr_iowait) == 0)) + + if (force_lb || prefer_spread) avg_idle = ULLONG_MAX; /* * This is OK, because current is on_cpu, which avoids it being picked @@ -11239,6 +11301,13 @@ int newidle_balance(struct rq *this_rq, struct rq_flags *rf) if (!(sd->flags & SD_LOAD_BALANCE)) continue; +#ifdef CONFIG_SCHED_WALT + if (prefer_spread && !force_lb && + (sd->flags & SD_ASYM_CPUCAPACITY) && + !(cpumask_test_cpu(this_cpu, &asym_cap_sibling_cpus))) + avg_idle = this_rq->avg_idle; +#endif + if (avg_idle < curr_cost + sd->max_newidle_lb_cost) { update_next_balance(sd, &next_balance); break; diff --git a/kernel/sysctl.c b/kernel/sysctl.c index c42d51f6cb18..cb7689c85859 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -583,6 +583,15 @@ static struct ctl_table kern_table[] = { .mode = 0644, .proc_handler = sched_updown_migrate_handler, }, + { + .procname = "sched_prefer_spread", + .data = &sysctl_sched_prefer_spread, + .maxlen = sizeof(unsigned int), + .mode = 0644, + .proc_handler = proc_dointvec_minmax, + .extra1 = SYSCTL_ZERO, + .extra2 = &two, + }, #endif #ifdef CONFIG_SCHED_DEBUG {