From 311342d796822aead26a29d0ad69696818387a26 Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Thu, 28 Feb 2019 10:40:39 +0530 Subject: [PATCH 1/4] cpuset: Restore tasks affinity while moving across cpusets When tasks move across cpusets, the current affinity settings are lost. Cache the task affinity and restore it during cpuset migration. The restoring happens only when the cached affinity is subset of the current cpuset settings. Change-Id: I6c2ec1d5e3d994e176926d94b9e0cc92418020cc Signed-off-by: Pavankumar Kondeti [satyap@codeaurora.org: fix trivial merge conflicts and replace cs->cpus_requested with cs->cpus_allowed] Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/sched.h | 3 +++ init/init_task.c | 3 +++ kernel/cgroup/cpuset.c | 20 ++++++++++++++++++-- kernel/sched/core.c | 8 ++++++++ 4 files changed, 32 insertions(+), 2 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index 0e92ee6464e5..eb32de4fdda7 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -860,6 +860,9 @@ struct task_struct { int nr_cpus_allowed; const cpumask_t *cpus_ptr; cpumask_t cpus_mask; +#ifdef CONFIG_SCHED_WALT + cpumask_t cpus_requested; +#endif #ifdef CONFIG_PREEMPT_RCU int rcu_read_lock_nesting; diff --git a/init/init_task.c b/init/init_task.c index af4dc7dcf246..2ad8a7c21c77 100644 --- a/init/init_task.c +++ b/init/init_task.c @@ -73,6 +73,9 @@ struct task_struct init_task .cpus_ptr = &init_task.cpus_mask, .cpus_mask = CPU_MASK_ALL, .nr_cpus_allowed= NR_CPUS, +#ifdef CONFIG_SCHED_WALT + .cpus_requested = CPU_MASK_ALL, +#endif .mm = NULL, .active_mm = &init_mm, .restart_block = { diff --git a/kernel/cgroup/cpuset.c b/kernel/cgroup/cpuset.c index c87ee6412b36..9ea6269fff29 100644 --- a/kernel/cgroup/cpuset.c +++ b/kernel/cgroup/cpuset.c @@ -1022,6 +1022,22 @@ void rebuild_sched_domains(void) put_online_cpus(); } +static int update_cpus_allowed(struct cpuset *cs, struct task_struct *p, + const struct cpumask *new_mask) +{ +#ifdef CONFIG_SCHED_WALT + int ret; + + if (cpumask_subset(&p->cpus_requested, cs->cpus_allowed)) { + ret = set_cpus_allowed_ptr(p, &p->cpus_requested); + if (!ret) + return ret; + } +#endif + + return set_cpus_allowed_ptr(p, new_mask); +} + /** * update_tasks_cpumask - Update the cpumasks of tasks in the cpuset. * @cs: the cpuset in which each task's cpus_allowed mask needs to be changed @@ -1037,7 +1053,7 @@ static void update_tasks_cpumask(struct cpuset *cs) css_task_iter_start(&cs->css, 0, &it); while ((task = css_task_iter_next(&it))) - set_cpus_allowed_ptr(task, cs->effective_cpus); + update_cpus_allowed(cs, task, cs->effective_cpus); css_task_iter_end(&it); } @@ -2187,7 +2203,7 @@ static void cpuset_attach(struct cgroup_taskset *tset) * can_attach beforehand should guarantee that this doesn't * fail. TODO: have a better way to handle failure here */ - WARN_ON_ONCE(set_cpus_allowed_ptr(task, cpus_attach)); + WARN_ON_ONCE(update_cpus_allowed(cs, task, cpus_attach)); cpuset_change_task_nodemask(task, &cpuset_attach_nodemask_to); cpuset_update_task_spread_flag(cs, task); diff --git a/kernel/sched/core.c b/kernel/sched/core.c index ba94863d96aa..b10f9c2300ae 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -5634,6 +5634,11 @@ again: retval = -EINVAL; } +#ifdef CONFIG_SCHED_WALT + if (!retval && !(p->flags & PF_KTHREAD)) + cpumask_and(&p->cpus_requested, in_mask, cpu_possible_mask); +#endif + out_free_new_mask: free_cpumask_var(new_mask); out_free_cpus_allowed: @@ -6780,6 +6785,9 @@ void __init sched_init_smp(void) /* Move init over to a non-isolated CPU */ if (set_cpus_allowed_ptr(current, housekeeping_cpumask(HK_FLAG_DOMAIN)) < 0) BUG(); +#ifdef CONFIG_SCHED_WALT + cpumask_copy(¤t->cpus_requested, cpu_possible_mask); +#endif sched_init_granularity(); init_sched_rt_class(); From 408e7407731ce17ece079830b821fc6f1135812c Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Mon, 14 Oct 2019 16:47:10 +0530 Subject: [PATCH 2/4] sched: use the CPU true capacity while sorting the min/mid/max CPUs The current code uses the rq->cpu_capacity_orig while sorting the min/mid/max CPUs while rebuilding the scheduler domains. This rq->cpu_capacity_orig is subjected to changed when the scaling max frequency is clipped for that CPU's frequency domain. Since we don't recompute the min/mid/max CPUs when the capacity is changed, the sorting becomes incorrect later when the frequency limits are lifted. The task CPU selection algorithm depends heavily on min/mid/max CPUs and changing them on the fly results in incorrect task placement. Hence use the true capacity of CPUs while sorting the min/mid/max CPUs. This means that the sorting gets changed only when CPUs are hotplugged out, otherwise these reflect the correct topology all the time. The max_cpu_capacity struct in root domain also maintains the max CPU and its capacity. Since we now use true capacity, there is no need to compute this on the fly. So move this evaluation from the periodic load balancer to the scheduler domain rebuilding. The rq->cpu_capacity_orig is still subjected to change upon frequency or thermal limits. So we still identify the cases where tasks not fitting on the max/mid CPUs when the capacity is reduced. Change-Id: I42735accd079b2ece1eb58d1ebcf322d454e33a2 Signed-off-by: Pavankumar Kondeti [satyap@codeaurora.org: trivial changes to arch_scale_cpu_capacity()] Signed-off-by: Satya Durga Srinivasu Prabhala --- kernel/sched/fair.c | 24 ------------------------ kernel/sched/topology.c | 28 ++++++++++++++++++++-------- 2 files changed, 20 insertions(+), 32 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 3a5e8ba90f3c..150c3df8b7b0 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -8641,10 +8641,6 @@ static void update_cpu_capacity(struct sched_domain *sd, int cpu) { unsigned long capacity = arch_scale_cpu_capacity(cpu); struct sched_group *sdg = sd->groups; - struct max_cpu_capacity *mcc; - unsigned long max_capacity; - int max_cap_cpu; - unsigned long flags; capacity *= arch_scale_max_freq_capacity(sd, cpu); capacity >>= SCHED_CAPACITY_SHIFT; @@ -8652,26 +8648,6 @@ static void update_cpu_capacity(struct sched_domain *sd, int cpu) capacity = min(capacity, thermal_cap(cpu)); cpu_rq(cpu)->cpu_capacity_orig = capacity; - mcc = &cpu_rq(cpu)->rd->max_cpu_capacity; - - raw_spin_lock_irqsave(&mcc->lock, flags); - max_capacity = mcc->val; - max_cap_cpu = mcc->cpu; - - if ((max_capacity > capacity && max_cap_cpu == cpu) || - (max_capacity < capacity)) { - mcc->val = capacity; - mcc->cpu = cpu; -#ifdef CONFIG_SCHED_DEBUG - raw_spin_unlock_irqrestore(&mcc->lock, flags); - printk_deferred(KERN_INFO "CPU%d: update max cpu_capacity %lu\n", - cpu, capacity); - goto skip_unlock; -#endif - } - raw_spin_unlock_irqrestore(&mcc->lock, flags); - -skip_unlock: __attribute__ ((unused)); capacity = scale_rt_capacity(cpu, capacity); if (!capacity) diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c index d686dd418194..dd72f9740f14 100644 --- a/kernel/sched/topology.c +++ b/kernel/sched/topology.c @@ -2047,12 +2047,12 @@ build_sched_domains(const struct cpumask *cpu_map, struct sched_domain_attr *att sd = *per_cpu_ptr(d.sd, i); #ifdef CONFIG_SCHED_WALT - if ((max_cpu < 0) || (cpu_rq(i)->cpu_capacity_orig > - cpu_rq(max_cpu)->cpu_capacity_orig)) + if ((max_cpu < 0) || (arch_scale_cpu_capacity(i) > + arch_scale_cpu_capacity(max_cpu))) WRITE_ONCE(d.rd->max_cap_orig_cpu, i); - if ((min_cpu < 0) || (cpu_rq(i)->cpu_capacity_orig < - cpu_rq(min_cpu)->cpu_capacity_orig)) + if ((min_cpu < 0) || (arch_scale_cpu_capacity(i) < + arch_scale_cpu_capacity(min_cpu))) WRITE_ONCE(d.rd->min_cap_orig_cpu, i); #endif @@ -2065,15 +2065,27 @@ build_sched_domains(const struct cpumask *cpu_map, struct sched_domain_attr *att int max_cpu = READ_ONCE(d.rd->max_cap_orig_cpu); int min_cpu = READ_ONCE(d.rd->min_cap_orig_cpu); - if ((cpu_rq(i)->cpu_capacity_orig - != cpu_rq(min_cpu)->cpu_capacity_orig) && - (cpu_rq(i)->cpu_capacity_orig - != cpu_rq(max_cpu)->cpu_capacity_orig)) { + if ((arch_scale_cpu_capacity(i) + != arch_scale_cpu_capacity(min_cpu)) && + (arch_scale_cpu_capacity(i) + != arch_scale_cpu_capacity(max_cpu))) { WRITE_ONCE(d.rd->mid_cap_orig_cpu, i); break; } } + + /* + * The max_cpu_capacity reflect the original capacity which does not + * change dynamically. So update the max cap CPU and its capacity + * here. + */ + if (d.rd->max_cap_orig_cpu != -1) { + d.rd->max_cpu_capacity.cpu = d.rd->max_cap_orig_cpu; + d.rd->max_cpu_capacity.val = arch_scale_cpu_capacity( + d.rd->max_cap_orig_cpu); + } #endif + rcu_read_unlock(); if (has_asym) From cb858f11671673a6d6d972206d0b132fd209745b Mon Sep 17 00:00:00 2001 From: Pavankumar Kondeti Date: Wed, 11 Sep 2019 12:50:58 +0530 Subject: [PATCH 3/4] sched/fair: Improve the scheduler This change is for general scheduler improvement. Change-Id: I7a84c7f7444d57f31e0859c4d39450d23421eb73 Signed-off-by: Pavankumar Kondeti --- kernel/sched/fair.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 150c3df8b7b0..8e70e238e528 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -6463,9 +6463,10 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, goto out; /* fast path for prev_cpu */ - if ((capacity_orig_of(prev_cpu) == capacity_orig_of(start_cpu)) && - !cpu_isolated(prev_cpu) && cpu_online(prev_cpu) && - idle_cpu(prev_cpu)) { + if (((capacity_orig_of(prev_cpu) == capacity_orig_of(start_cpu)) || + asym_cap_siblings(prev_cpu, start_cpu)) && + !cpu_isolated(prev_cpu) && cpu_online(prev_cpu) && + idle_cpu(prev_cpu)) { if (idle_get_state_idx(cpu_rq(prev_cpu)) <= 1) { target_cpu = prev_cpu; From aa8b37baacdf6e3f407b551ac1598020895918d2 Mon Sep 17 00:00:00 2001 From: Brendan Jackman Date: Fri, 11 Aug 2017 10:45:58 +0100 Subject: [PATCH 4/4] FROMLIST: sched/fair: Use wake_q length as a hint for wake_wide This patch adds a parameter to select_task_rq, sibling_count_hint allowing the caller, where it has this information, to inform the sched_class the number of tasks that are being woken up as part of the same event. The wake_q mechanism is one case where this information is available. select_task_rq_fair can then use the information to detect that it needs to widen the search space for task placement in order to avoid overloading the last-level cache domain's CPUs. * * * The reason I am investigating this change is the following use case on ARM big.LITTLE (asymmetrical CPU capacity): 1 task per CPU, which all repeatedly do X amount of work then pthread_barrier_wait (i.e. sleep until the last task finishes its X and hits the barrier). On big.LITTLE, the tasks which get a "big" CPU finish faster, and then those CPUs pull over the tasks that are still running: v CPU v ->time-> ------------- 0 (big) 11111 /333 ------------- 1 (big) 22222 /444| ------------- 2 (LITTLE) 333333/ ------------- 3 (LITTLE) 444444/ ------------- Now when task 4 hits the barrier (at |) and wakes the others up, there are 4 tasks with prev_cpu= and 0 tasks with prev_cpu=. want_affine therefore means that we'll only look in CPUs 0 and 1 (sd_llc), so tasks will be unnecessarily coscheduled on the bigs until the next load balance, something like this: v CPU v ->time-> ------------------------ 0 (big) 11111 /333 31313\33333 ------------------------ 1 (big) 22222 /444|424\4444444 ------------------------ 2 (LITTLE) 333333/ \222222 ------------------------ 3 (LITTLE) 444444/ \1111 ------------------------ ^^^ underutilization So, I'm trying to get want_affine = 0 for these tasks. I don't _think_ any incarnation of the wakee_flips mechanism can help us here because which task is waker and which tasks are wakees generally changes with each iteration. However pthread_barrier_wait (or more accurately FUTEX_WAKE) has the nice property that we know exactly how many tasks are being woken, so we can cheat. It might be a disadvantage that we "widen" _every_ task that's woken in an event, while select_idle_sibling would work fine for the first sd_llc_size - 1 tasks. IIUC, if wake_affine() behaves correctly this trick wouldn't be necessary on SMP systems, so it might be best guarded by the presence of SD_ASYM_CPUCAPACITY? * * * Final note.. In order to observe "perfect" behaviour for this use case, I also had to disable the TTWU_QUEUE sched feature. Suppose during the wakeup above we are working through the work queue and have placed tasks 3 and 2, and are about to place task 1: v CPU v ->time-> -------------- 0 (big) 11111 /333 3 -------------- 1 (big) 22222 /444|4 -------------- 2 (LITTLE) 333333/ 2 -------------- 3 (LITTLE) 444444/ <- Task 1 should go here -------------- If TTWU_QUEUE is enabled, we will not yet have enqueued task 2 (having instead sent a reschedule IPI) or attached its load to CPU 2. So we are likely to also place task 1 on cpu 2. Disabling TTWU_QUEUE means that we enqueue task 2 before placing task 1, solving this issue. TTWU_QUEUE is there to minimise rq lock contention, and I guess that this contention is less of an issue on big.LITTLE systems since they have relatively few CPUs, which suggests the trade-off makes sense here. Signed-off-by: Brendan Jackman Cc: Ingo Molnar Cc: Peter Zijlstra Cc: Josef Bacik Cc: Joel Fernandes Cc: Mike Galbraith Cc: Matt Fleming ( - Applied from https://patchwork.kernel.org/patch/9895261/ - Fixed trivial conflict in kernel/sched/core.c - Fixed select_task_rq_idle, now in kernel/sched/idle.c - Fixed trivial conflict in select_task_rq_fair ) Signed-off-by: Quentin Perret Change-Id: I3cfc4bf48c3d7feef969db4d22449f4fbb4f795d [satyap@codeaurora.org: port to 5.4 and fix trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/sched/wake_q.h | 6 ++++++ kernel/sched/core.c | 39 ++++++++++++++++++++++++++---------- kernel/sched/deadline.c | 3 ++- kernel/sched/fair.c | 15 +++++++++----- kernel/sched/idle.c | 3 ++- kernel/sched/rt.c | 3 ++- kernel/sched/sched.h | 3 ++- kernel/sched/stop_task.c | 3 ++- 8 files changed, 54 insertions(+), 21 deletions(-) diff --git a/include/linux/sched/wake_q.h b/include/linux/sched/wake_q.h index 26a2013ac39c..3c04437a7a6f 100644 --- a/include/linux/sched/wake_q.h +++ b/include/linux/sched/wake_q.h @@ -38,6 +38,9 @@ struct wake_q_head { struct wake_q_node *first; struct wake_q_node **lastp; +#ifdef CONFIG_SCHED_WALT + int count; +#endif }; #define WAKE_Q_TAIL ((struct wake_q_node *) 0x01) @@ -49,6 +52,9 @@ static inline void wake_q_init(struct wake_q_head *head) { head->first = WAKE_Q_TAIL; head->lastp = &head->first; +#ifdef CONFIG_SCHED_WALT + head->count = 0; +#endif } static inline bool wake_q_empty(struct wake_q_head *head) diff --git a/kernel/sched/core.c b/kernel/sched/core.c index b10f9c2300ae..a1993537ea8e 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -431,6 +431,9 @@ static bool __wake_q_add(struct wake_q_head *head, struct task_struct *task) /* * The head is context local, there can be no concurrency. */ +#ifdef CONFIG_SCHED_WALT + head->count++; +#endif *head->lastp = node; head->lastp = &node->next; return true; @@ -477,6 +480,10 @@ void wake_q_add_safe(struct wake_q_head *head, struct task_struct *task) put_task_struct(task); } +static int +try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags, + int sibling_count_hint); + void wake_up_q(struct wake_q_head *head) { struct wake_q_node *node = head->first; @@ -491,10 +498,14 @@ void wake_up_q(struct wake_q_head *head) task->wake_q.next = NULL; /* - * wake_up_process() executes a full barrier, which pairs with + * try_to_wake_up() executes a full barrier, which pairs with * the queueing in wake_q_add() so as not to miss wakeups. */ - wake_up_process(task); +#ifdef CONFIG_SCHED_WALT + try_to_wake_up(task, TASK_NORMAL, 0, head->count); +#else + try_to_wake_up(task, TASK_NORMAL, 0, 1); +#endif put_task_struct(task); } } @@ -2142,14 +2153,16 @@ out: * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable. */ static inline -int select_task_rq(struct task_struct *p, int cpu, int sd_flags, int wake_flags) +int select_task_rq(struct task_struct *p, int cpu, int sd_flags, int wake_flags, + int sibling_count_hint) { bool allow_isolated = (p->flags & PF_KTHREAD); lockdep_assert_held(&p->pi_lock); if (p->nr_cpus_allowed > 1) - cpu = p->sched_class->select_task_rq(p, cpu, sd_flags, wake_flags); + cpu = p->sched_class->select_task_rq(p, cpu, sd_flags, wake_flags, + sibling_count_hint); else cpu = cpumask_any(p->cpus_ptr); @@ -2544,6 +2557,8 @@ static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags) * @p: the thread to be awakened * @state: the mask of task states that can be woken * @wake_flags: wake modifier flags (WF_*) + * @sibling_count_hint: A hint at the number of threads that are being woken up + * in this event. * * If (@state & @p->state) @p->state = TASK_RUNNING. * @@ -2559,7 +2574,8 @@ static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags) * %false otherwise. */ static int -try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags) +try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags, + int sibling_count_hint) { unsigned long flags; int cpu, success = 0; @@ -2672,7 +2688,8 @@ try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags) atomic_dec(&task_rq(p)->nr_iowait); } - cpu = select_task_rq(p, p->wake_cpu, SD_BALANCE_WAKE, wake_flags); + cpu = select_task_rq(p, p->wake_cpu, SD_BALANCE_WAKE, wake_flags, + sibling_count_hint); if (task_cpu(p) != cpu) { wake_flags |= WF_MIGRATED; psi_ttwu_dequeue(p); @@ -2723,13 +2740,13 @@ out: */ int wake_up_process(struct task_struct *p) { - return try_to_wake_up(p, TASK_NORMAL, 0); + return try_to_wake_up(p, TASK_NORMAL, 0, 1); } EXPORT_SYMBOL(wake_up_process); int wake_up_state(struct task_struct *p, unsigned int state) { - return try_to_wake_up(p, state, 0); + return try_to_wake_up(p, state, 0, 1); } /* @@ -3026,7 +3043,7 @@ void wake_up_new_task(struct task_struct *p) * as we're not fully set-up yet. */ p->recent_used_cpu = task_cpu(p); - __set_task_cpu(p, select_task_rq(p, task_cpu(p), SD_BALANCE_FORK, 0)); + __set_task_cpu(p, select_task_rq(p, task_cpu(p), SD_BALANCE_FORK, 0, 1)); #endif rq = __task_rq_lock(p, &rf); update_rq_clock(rq); @@ -3571,7 +3588,7 @@ void sched_exec(void) return; raw_spin_lock_irqsave(&p->pi_lock, flags); - dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), SD_BALANCE_EXEC, 0); + dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), SD_BALANCE_EXEC, 0, 1); if (dest_cpu == smp_processor_id()) goto unlock; @@ -4500,7 +4517,7 @@ asmlinkage __visible void __sched preempt_schedule_irq(void) int default_wake_function(wait_queue_entry_t *curr, unsigned mode, int wake_flags, void *key) { - return try_to_wake_up(curr->private, mode, wake_flags); + return try_to_wake_up(curr->private, mode, wake_flags, 1); } EXPORT_SYMBOL(default_wake_function); diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 0947361dc7d0..1b22dd85d15f 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -1602,7 +1602,8 @@ static void yield_task_dl(struct rq *rq) static int find_later_rq(struct task_struct *task); static int -select_task_rq_dl(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_dl(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { struct task_struct *curr; struct rq *rq; diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 8e70e238e528..f3d7ee959e66 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -5590,15 +5590,18 @@ static void record_wakee(struct task_struct *p) * whatever is irrelevant, spread criteria is apparent partner count exceeds * socket size. */ -static int wake_wide(struct task_struct *p) +static int wake_wide(struct task_struct *p, int sibling_count_hint) { unsigned int master = current->wakee_flips; unsigned int slave = p->wakee_flips; - int factor = this_cpu_read(sd_llc_size); + int llc_size = this_cpu_read(sd_llc_size); + + if (sibling_count_hint >= llc_size) + return 1; if (master < slave) swap(master, slave); - if (slave < factor || master < slave * factor) + if (slave < llc_size || master < slave * llc_size) return 0; return 1; } @@ -7220,7 +7223,8 @@ eas_not_ready: * preempt must be disabled. */ static int -select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_flags) +select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_flags, + int sibling_count_hint) { struct sched_domain *tmp, *sd = NULL; int cpu = smp_processor_id(); @@ -7247,7 +7251,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_f new_cpu = prev_cpu; } - want_affine = !wake_wide(p) && !wake_cap(p, cpu, prev_cpu) && + want_affine = !wake_wide(p, sibling_count_hint) && + !wake_cap(p, cpu, prev_cpu) && cpumask_test_cpu(cpu, p->cpus_ptr); } diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c index 27a5c1e62c72..e5dbe0da6f18 100644 --- a/kernel/sched/idle.c +++ b/kernel/sched/idle.c @@ -364,7 +364,8 @@ void cpu_startup_entry(enum cpuhp_state state) #ifdef CONFIG_SMP static int -select_task_rq_idle(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_idle(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { return task_cpu(p); /* IDLE tasks as never migrated */ } diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 3a9a757bfe03..1fd742ca6edb 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -1481,7 +1481,8 @@ task_may_not_preempt(struct task_struct *task, int cpu) } static int -select_task_rq_rt(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_rt(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { struct task_struct *curr; struct rq *rq; diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index ddea10411134..6387c70c252f 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -1883,7 +1883,8 @@ struct sched_class { #ifdef CONFIG_SMP int (*balance)(struct rq *rq, struct task_struct *prev, struct rq_flags *rf); - int (*select_task_rq)(struct task_struct *p, int task_cpu, int sd_flag, int flags); + int (*select_task_rq)(struct task_struct *p, int task_cpu, int sd_flag, int flags, + int subling_count_hint); void (*migrate_task_rq)(struct task_struct *p, int new_cpu); void (*task_woken)(struct rq *this_rq, struct task_struct *task); diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index e58457b4377a..7abbc22c2897 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -12,7 +12,8 @@ #ifdef CONFIG_SMP static int -select_task_rq_stop(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_stop(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { return task_cpu(p); /* stop tasks as never migrate */ }