diff --git a/include/linux/sched.h b/include/linux/sched.h index 0e92ee6464e5..eb32de4fdda7 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -860,6 +860,9 @@ struct task_struct { int nr_cpus_allowed; const cpumask_t *cpus_ptr; cpumask_t cpus_mask; +#ifdef CONFIG_SCHED_WALT + cpumask_t cpus_requested; +#endif #ifdef CONFIG_PREEMPT_RCU int rcu_read_lock_nesting; diff --git a/include/linux/sched/wake_q.h b/include/linux/sched/wake_q.h index 26a2013ac39c..3c04437a7a6f 100644 --- a/include/linux/sched/wake_q.h +++ b/include/linux/sched/wake_q.h @@ -38,6 +38,9 @@ struct wake_q_head { struct wake_q_node *first; struct wake_q_node **lastp; +#ifdef CONFIG_SCHED_WALT + int count; +#endif }; #define WAKE_Q_TAIL ((struct wake_q_node *) 0x01) @@ -49,6 +52,9 @@ static inline void wake_q_init(struct wake_q_head *head) { head->first = WAKE_Q_TAIL; head->lastp = &head->first; +#ifdef CONFIG_SCHED_WALT + head->count = 0; +#endif } static inline bool wake_q_empty(struct wake_q_head *head) diff --git a/init/init_task.c b/init/init_task.c index af4dc7dcf246..2ad8a7c21c77 100644 --- a/init/init_task.c +++ b/init/init_task.c @@ -73,6 +73,9 @@ struct task_struct init_task .cpus_ptr = &init_task.cpus_mask, .cpus_mask = CPU_MASK_ALL, .nr_cpus_allowed= NR_CPUS, +#ifdef CONFIG_SCHED_WALT + .cpus_requested = CPU_MASK_ALL, +#endif .mm = NULL, .active_mm = &init_mm, .restart_block = { diff --git a/kernel/cgroup/cpuset.c b/kernel/cgroup/cpuset.c index c87ee6412b36..9ea6269fff29 100644 --- a/kernel/cgroup/cpuset.c +++ b/kernel/cgroup/cpuset.c @@ -1022,6 +1022,22 @@ void rebuild_sched_domains(void) put_online_cpus(); } +static int update_cpus_allowed(struct cpuset *cs, struct task_struct *p, + const struct cpumask *new_mask) +{ +#ifdef CONFIG_SCHED_WALT + int ret; + + if (cpumask_subset(&p->cpus_requested, cs->cpus_allowed)) { + ret = set_cpus_allowed_ptr(p, &p->cpus_requested); + if (!ret) + return ret; + } +#endif + + return set_cpus_allowed_ptr(p, new_mask); +} + /** * update_tasks_cpumask - Update the cpumasks of tasks in the cpuset. * @cs: the cpuset in which each task's cpus_allowed mask needs to be changed @@ -1037,7 +1053,7 @@ static void update_tasks_cpumask(struct cpuset *cs) css_task_iter_start(&cs->css, 0, &it); while ((task = css_task_iter_next(&it))) - set_cpus_allowed_ptr(task, cs->effective_cpus); + update_cpus_allowed(cs, task, cs->effective_cpus); css_task_iter_end(&it); } @@ -2187,7 +2203,7 @@ static void cpuset_attach(struct cgroup_taskset *tset) * can_attach beforehand should guarantee that this doesn't * fail. TODO: have a better way to handle failure here */ - WARN_ON_ONCE(set_cpus_allowed_ptr(task, cpus_attach)); + WARN_ON_ONCE(update_cpus_allowed(cs, task, cpus_attach)); cpuset_change_task_nodemask(task, &cpuset_attach_nodemask_to); cpuset_update_task_spread_flag(cs, task); diff --git a/kernel/sched/core.c b/kernel/sched/core.c index ba94863d96aa..a1993537ea8e 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -431,6 +431,9 @@ static bool __wake_q_add(struct wake_q_head *head, struct task_struct *task) /* * The head is context local, there can be no concurrency. */ +#ifdef CONFIG_SCHED_WALT + head->count++; +#endif *head->lastp = node; head->lastp = &node->next; return true; @@ -477,6 +480,10 @@ void wake_q_add_safe(struct wake_q_head *head, struct task_struct *task) put_task_struct(task); } +static int +try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags, + int sibling_count_hint); + void wake_up_q(struct wake_q_head *head) { struct wake_q_node *node = head->first; @@ -491,10 +498,14 @@ void wake_up_q(struct wake_q_head *head) task->wake_q.next = NULL; /* - * wake_up_process() executes a full barrier, which pairs with + * try_to_wake_up() executes a full barrier, which pairs with * the queueing in wake_q_add() so as not to miss wakeups. */ - wake_up_process(task); +#ifdef CONFIG_SCHED_WALT + try_to_wake_up(task, TASK_NORMAL, 0, head->count); +#else + try_to_wake_up(task, TASK_NORMAL, 0, 1); +#endif put_task_struct(task); } } @@ -2142,14 +2153,16 @@ out: * The caller (fork, wakeup) owns p->pi_lock, ->cpus_ptr is stable. */ static inline -int select_task_rq(struct task_struct *p, int cpu, int sd_flags, int wake_flags) +int select_task_rq(struct task_struct *p, int cpu, int sd_flags, int wake_flags, + int sibling_count_hint) { bool allow_isolated = (p->flags & PF_KTHREAD); lockdep_assert_held(&p->pi_lock); if (p->nr_cpus_allowed > 1) - cpu = p->sched_class->select_task_rq(p, cpu, sd_flags, wake_flags); + cpu = p->sched_class->select_task_rq(p, cpu, sd_flags, wake_flags, + sibling_count_hint); else cpu = cpumask_any(p->cpus_ptr); @@ -2544,6 +2557,8 @@ static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags) * @p: the thread to be awakened * @state: the mask of task states that can be woken * @wake_flags: wake modifier flags (WF_*) + * @sibling_count_hint: A hint at the number of threads that are being woken up + * in this event. * * If (@state & @p->state) @p->state = TASK_RUNNING. * @@ -2559,7 +2574,8 @@ static void ttwu_queue(struct task_struct *p, int cpu, int wake_flags) * %false otherwise. */ static int -try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags) +try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags, + int sibling_count_hint) { unsigned long flags; int cpu, success = 0; @@ -2672,7 +2688,8 @@ try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags) atomic_dec(&task_rq(p)->nr_iowait); } - cpu = select_task_rq(p, p->wake_cpu, SD_BALANCE_WAKE, wake_flags); + cpu = select_task_rq(p, p->wake_cpu, SD_BALANCE_WAKE, wake_flags, + sibling_count_hint); if (task_cpu(p) != cpu) { wake_flags |= WF_MIGRATED; psi_ttwu_dequeue(p); @@ -2723,13 +2740,13 @@ out: */ int wake_up_process(struct task_struct *p) { - return try_to_wake_up(p, TASK_NORMAL, 0); + return try_to_wake_up(p, TASK_NORMAL, 0, 1); } EXPORT_SYMBOL(wake_up_process); int wake_up_state(struct task_struct *p, unsigned int state) { - return try_to_wake_up(p, state, 0); + return try_to_wake_up(p, state, 0, 1); } /* @@ -3026,7 +3043,7 @@ void wake_up_new_task(struct task_struct *p) * as we're not fully set-up yet. */ p->recent_used_cpu = task_cpu(p); - __set_task_cpu(p, select_task_rq(p, task_cpu(p), SD_BALANCE_FORK, 0)); + __set_task_cpu(p, select_task_rq(p, task_cpu(p), SD_BALANCE_FORK, 0, 1)); #endif rq = __task_rq_lock(p, &rf); update_rq_clock(rq); @@ -3571,7 +3588,7 @@ void sched_exec(void) return; raw_spin_lock_irqsave(&p->pi_lock, flags); - dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), SD_BALANCE_EXEC, 0); + dest_cpu = p->sched_class->select_task_rq(p, task_cpu(p), SD_BALANCE_EXEC, 0, 1); if (dest_cpu == smp_processor_id()) goto unlock; @@ -4500,7 +4517,7 @@ asmlinkage __visible void __sched preempt_schedule_irq(void) int default_wake_function(wait_queue_entry_t *curr, unsigned mode, int wake_flags, void *key) { - return try_to_wake_up(curr->private, mode, wake_flags); + return try_to_wake_up(curr->private, mode, wake_flags, 1); } EXPORT_SYMBOL(default_wake_function); @@ -5634,6 +5651,11 @@ again: retval = -EINVAL; } +#ifdef CONFIG_SCHED_WALT + if (!retval && !(p->flags & PF_KTHREAD)) + cpumask_and(&p->cpus_requested, in_mask, cpu_possible_mask); +#endif + out_free_new_mask: free_cpumask_var(new_mask); out_free_cpus_allowed: @@ -6780,6 +6802,9 @@ void __init sched_init_smp(void) /* Move init over to a non-isolated CPU */ if (set_cpus_allowed_ptr(current, housekeeping_cpumask(HK_FLAG_DOMAIN)) < 0) BUG(); +#ifdef CONFIG_SCHED_WALT + cpumask_copy(¤t->cpus_requested, cpu_possible_mask); +#endif sched_init_granularity(); init_sched_rt_class(); diff --git a/kernel/sched/deadline.c b/kernel/sched/deadline.c index 0947361dc7d0..1b22dd85d15f 100644 --- a/kernel/sched/deadline.c +++ b/kernel/sched/deadline.c @@ -1602,7 +1602,8 @@ static void yield_task_dl(struct rq *rq) static int find_later_rq(struct task_struct *task); static int -select_task_rq_dl(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_dl(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { struct task_struct *curr; struct rq *rq; diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 3a5e8ba90f3c..f3d7ee959e66 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -5590,15 +5590,18 @@ static void record_wakee(struct task_struct *p) * whatever is irrelevant, spread criteria is apparent partner count exceeds * socket size. */ -static int wake_wide(struct task_struct *p) +static int wake_wide(struct task_struct *p, int sibling_count_hint) { unsigned int master = current->wakee_flips; unsigned int slave = p->wakee_flips; - int factor = this_cpu_read(sd_llc_size); + int llc_size = this_cpu_read(sd_llc_size); + + if (sibling_count_hint >= llc_size) + return 1; if (master < slave) swap(master, slave); - if (slave < factor || master < slave * factor) + if (slave < llc_size || master < slave * llc_size) return 0; return 1; } @@ -6463,9 +6466,10 @@ static void find_best_target(struct sched_domain *sd, cpumask_t *cpus, goto out; /* fast path for prev_cpu */ - if ((capacity_orig_of(prev_cpu) == capacity_orig_of(start_cpu)) && - !cpu_isolated(prev_cpu) && cpu_online(prev_cpu) && - idle_cpu(prev_cpu)) { + if (((capacity_orig_of(prev_cpu) == capacity_orig_of(start_cpu)) || + asym_cap_siblings(prev_cpu, start_cpu)) && + !cpu_isolated(prev_cpu) && cpu_online(prev_cpu) && + idle_cpu(prev_cpu)) { if (idle_get_state_idx(cpu_rq(prev_cpu)) <= 1) { target_cpu = prev_cpu; @@ -7219,7 +7223,8 @@ eas_not_ready: * preempt must be disabled. */ static int -select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_flags) +select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_flags, + int sibling_count_hint) { struct sched_domain *tmp, *sd = NULL; int cpu = smp_processor_id(); @@ -7246,7 +7251,8 @@ select_task_rq_fair(struct task_struct *p, int prev_cpu, int sd_flag, int wake_f new_cpu = prev_cpu; } - want_affine = !wake_wide(p) && !wake_cap(p, cpu, prev_cpu) && + want_affine = !wake_wide(p, sibling_count_hint) && + !wake_cap(p, cpu, prev_cpu) && cpumask_test_cpu(cpu, p->cpus_ptr); } @@ -8641,10 +8647,6 @@ static void update_cpu_capacity(struct sched_domain *sd, int cpu) { unsigned long capacity = arch_scale_cpu_capacity(cpu); struct sched_group *sdg = sd->groups; - struct max_cpu_capacity *mcc; - unsigned long max_capacity; - int max_cap_cpu; - unsigned long flags; capacity *= arch_scale_max_freq_capacity(sd, cpu); capacity >>= SCHED_CAPACITY_SHIFT; @@ -8652,26 +8654,6 @@ static void update_cpu_capacity(struct sched_domain *sd, int cpu) capacity = min(capacity, thermal_cap(cpu)); cpu_rq(cpu)->cpu_capacity_orig = capacity; - mcc = &cpu_rq(cpu)->rd->max_cpu_capacity; - - raw_spin_lock_irqsave(&mcc->lock, flags); - max_capacity = mcc->val; - max_cap_cpu = mcc->cpu; - - if ((max_capacity > capacity && max_cap_cpu == cpu) || - (max_capacity < capacity)) { - mcc->val = capacity; - mcc->cpu = cpu; -#ifdef CONFIG_SCHED_DEBUG - raw_spin_unlock_irqrestore(&mcc->lock, flags); - printk_deferred(KERN_INFO "CPU%d: update max cpu_capacity %lu\n", - cpu, capacity); - goto skip_unlock; -#endif - } - raw_spin_unlock_irqrestore(&mcc->lock, flags); - -skip_unlock: __attribute__ ((unused)); capacity = scale_rt_capacity(cpu, capacity); if (!capacity) diff --git a/kernel/sched/idle.c b/kernel/sched/idle.c index 27a5c1e62c72..e5dbe0da6f18 100644 --- a/kernel/sched/idle.c +++ b/kernel/sched/idle.c @@ -364,7 +364,8 @@ void cpu_startup_entry(enum cpuhp_state state) #ifdef CONFIG_SMP static int -select_task_rq_idle(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_idle(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { return task_cpu(p); /* IDLE tasks as never migrated */ } diff --git a/kernel/sched/rt.c b/kernel/sched/rt.c index 3a9a757bfe03..1fd742ca6edb 100644 --- a/kernel/sched/rt.c +++ b/kernel/sched/rt.c @@ -1481,7 +1481,8 @@ task_may_not_preempt(struct task_struct *task, int cpu) } static int -select_task_rq_rt(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_rt(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { struct task_struct *curr; struct rq *rq; diff --git a/kernel/sched/sched.h b/kernel/sched/sched.h index ddea10411134..6387c70c252f 100644 --- a/kernel/sched/sched.h +++ b/kernel/sched/sched.h @@ -1883,7 +1883,8 @@ struct sched_class { #ifdef CONFIG_SMP int (*balance)(struct rq *rq, struct task_struct *prev, struct rq_flags *rf); - int (*select_task_rq)(struct task_struct *p, int task_cpu, int sd_flag, int flags); + int (*select_task_rq)(struct task_struct *p, int task_cpu, int sd_flag, int flags, + int subling_count_hint); void (*migrate_task_rq)(struct task_struct *p, int new_cpu); void (*task_woken)(struct rq *this_rq, struct task_struct *task); diff --git a/kernel/sched/stop_task.c b/kernel/sched/stop_task.c index e58457b4377a..7abbc22c2897 100644 --- a/kernel/sched/stop_task.c +++ b/kernel/sched/stop_task.c @@ -12,7 +12,8 @@ #ifdef CONFIG_SMP static int -select_task_rq_stop(struct task_struct *p, int cpu, int sd_flag, int flags) +select_task_rq_stop(struct task_struct *p, int cpu, int sd_flag, int flags, + int sibling_count_hint) { return task_cpu(p); /* stop tasks as never migrate */ } diff --git a/kernel/sched/topology.c b/kernel/sched/topology.c index d686dd418194..dd72f9740f14 100644 --- a/kernel/sched/topology.c +++ b/kernel/sched/topology.c @@ -2047,12 +2047,12 @@ build_sched_domains(const struct cpumask *cpu_map, struct sched_domain_attr *att sd = *per_cpu_ptr(d.sd, i); #ifdef CONFIG_SCHED_WALT - if ((max_cpu < 0) || (cpu_rq(i)->cpu_capacity_orig > - cpu_rq(max_cpu)->cpu_capacity_orig)) + if ((max_cpu < 0) || (arch_scale_cpu_capacity(i) > + arch_scale_cpu_capacity(max_cpu))) WRITE_ONCE(d.rd->max_cap_orig_cpu, i); - if ((min_cpu < 0) || (cpu_rq(i)->cpu_capacity_orig < - cpu_rq(min_cpu)->cpu_capacity_orig)) + if ((min_cpu < 0) || (arch_scale_cpu_capacity(i) < + arch_scale_cpu_capacity(min_cpu))) WRITE_ONCE(d.rd->min_cap_orig_cpu, i); #endif @@ -2065,15 +2065,27 @@ build_sched_domains(const struct cpumask *cpu_map, struct sched_domain_attr *att int max_cpu = READ_ONCE(d.rd->max_cap_orig_cpu); int min_cpu = READ_ONCE(d.rd->min_cap_orig_cpu); - if ((cpu_rq(i)->cpu_capacity_orig - != cpu_rq(min_cpu)->cpu_capacity_orig) && - (cpu_rq(i)->cpu_capacity_orig - != cpu_rq(max_cpu)->cpu_capacity_orig)) { + if ((arch_scale_cpu_capacity(i) + != arch_scale_cpu_capacity(min_cpu)) && + (arch_scale_cpu_capacity(i) + != arch_scale_cpu_capacity(max_cpu))) { WRITE_ONCE(d.rd->mid_cap_orig_cpu, i); break; } } + + /* + * The max_cpu_capacity reflect the original capacity which does not + * change dynamically. So update the max cap CPU and its capacity + * here. + */ + if (d.rd->max_cap_orig_cpu != -1) { + d.rd->max_cpu_capacity.cpu = d.rd->max_cap_orig_cpu; + d.rd->max_cpu_capacity.val = arch_scale_cpu_capacity( + d.rd->max_cap_orig_cpu); + } #endif + rcu_read_unlock(); if (has_asym)