From 44d528b138d9887931e045e8d867dbefbdfcf3d3 Mon Sep 17 00:00:00 2001 From: Puja Gupta Date: Wed, 13 Mar 2019 15:24:43 -0700 Subject: [PATCH 1/3] Revert "sched/deadline: Remove cpu_active_mask from cpudl_find()" This reverts commit 9659e1ee. The change leaves a possibility of scheduling dl task on offline cpu. The original commit text suggests that it should not be a problem since migrate_tasks() will take care of it, but there can be delay between marking cpu as offline and migrate_tasks() call since they are called on different hotplug notifications-offline happens on CPUHP_TEARDOWN_CPU while migrate_tasks() gets called on CPUHP_AP_SCHED_STARTING. And in between this duration there is a chance the dl task can get scheduled on this offline cpu and lead to BUG_ON. Hence revert the change. Change-Id: I6f4ea10d2ce4e83e12d36b9b90b1b630ffe4fc74 [pujag@codeaurora.org: fix minor conflicts due to reorg of the code] Signed-off-by: Puja Gupta [satyap@codeaurora.org: resolve trivial merge conflicts] Signed-off-by: Satya Durga Srinivasu Prabhala --- kernel/sched/cpudeadline.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/kernel/sched/cpudeadline.c b/kernel/sched/cpudeadline.c index 5cc4012572ec..0346837fac83 100644 --- a/kernel/sched/cpudeadline.c +++ b/kernel/sched/cpudeadline.c @@ -120,7 +120,8 @@ int cpudl_find(struct cpudl *cp, struct task_struct *p, const struct sched_dl_entity *dl_se = &p->dl; if (later_mask && - cpumask_and(later_mask, cp->free_cpus, p->cpus_ptr)) { + cpumask_and(later_mask, cp->free_cpus, p->cpus_ptr) && + cpumask_and(later_mask, later_mask, cpu_active_mask)) { return 1; } else { int best_cpu = cpudl_maximum(cp); @@ -128,6 +129,7 @@ int cpudl_find(struct cpudl *cp, struct task_struct *p, WARN_ON(best_cpu != -1 && !cpu_present(best_cpu)); if (cpumask_test_cpu(best_cpu, p->cpus_ptr) && + cpumask_test_cpu(best_cpu, cpu_active_mask) && dl_time_before(dl_se->deadline, cp->elements[0].dl)) { if (later_mask) cpumask_set_cpu(best_cpu, later_mask); From b53675c0d2192ce4b6f2d97ea59a5c21aa037cec Mon Sep 17 00:00:00 2001 From: Morten Rasmussen Date: Tue, 7 Mar 2017 16:41:26 +0000 Subject: [PATCH 2/3] ANDROID: sched/fair: Avoid unnecessary balancing of asymmetric capacity groups On systems with asymmetric cpu capacities, a skewed load distribution might yield better throughput than balancing load per group capacity. For example, running compute intensive tasks on high capacity cpus while leaving low capacity cpus idle. So we let load-balance back off if the busiest group isn't really overloaded. cc: Ingo Molnar cc: Peter Zijlstra Signed-off-by: Morten Rasmussen Change-Id: I8b08a0fa73f357a9972324bc76cec3912fe293cf Signed-off-by: Chris Redpath Git-commit: c42f9795e6a0c0e9d3cc6e1fe02fa60829ceb23a Git-repo: https://android.googlesource.com/kernel/common/ Signed-off-by: Satya Durga Srinivasu Prabhala [clingutla@codeaurora.org: Took partial commit '5494e2edf59c2 ("ANDROID: sched: Consider misfit tasks when load-balancing")', which defined function group_similar_cpu_capacity.] Signed-off-by: Lingutla Chandrasekhar --- kernel/sched/fair.c | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 2fe25a61ba76..dc342f8094c2 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -8061,6 +8061,19 @@ group_smaller_max_cpu_capacity(struct sched_group *sg, struct sched_group *ref) return fits_capacity(sg->sgc->max_capacity, ref->sgc->max_capacity); } +/* + * group_similar_cpu_capacity: Returns true if the minimum capacity of the + * compared groups differ by less than 12.5%. + */ +static inline bool +group_similar_cpu_capacity(struct sched_group *sg, struct sched_group *ref) +{ + long diff = sg->sgc->min_capacity - ref->sgc->min_capacity; + long max = max(sg->sgc->min_capacity, ref->sgc->min_capacity); + + return abs(diff) < max >> 3; +} + static inline enum group_type group_classify(struct sched_group *group, struct sg_lb_stats *sgs) @@ -8215,6 +8228,15 @@ static bool update_sd_pick_busiest(struct lb_env *env, group_smaller_min_cpu_capacity(sds->local, sg)) return false; + /* + * Candidate sg doesn't face any severe imbalance issues so + * don't disturb unless the groups are of similar capacity + * where balancing is more harmless. + */ + if (sgs->group_type == group_other && + !group_similar_cpu_capacity(sds->local, sg)) + return false; + /* * If we have more than one misfit sg go with the biggest misfit. */ From 79785b91ddb8d1cde9e2a72fec9456ec25184237 Mon Sep 17 00:00:00 2001 From: Abhijeet Dharmapurikar Date: Wed, 27 Mar 2019 09:51:24 -0700 Subject: [PATCH 3/3] power: em: correct increasing freq/power ratio As freq increases freq/power should decrease, indicating nonlinearly higher power is required for running at higher frequencies. If freq/power increases, the cost associated with that freq will be lower than its previous one, causing the energy calculations to choose a cpu with that frequency over a cpu with the previous lesser freq. For some frequencies, if their voltage increment from the previous ones is very small (or is the same), we could end up with higher freq/power ratio. This is primarily because of_dev_pm_opp_get_cpu_power() returns power in mW and looses precision. But instead of addressing it, enforce the same cost as the previous frequency. The energy evaluation code prefers the previous cpu when the energy costs are same as other candidates. By keeping the energy costs same in these situations we are increasing the likely hood of selecting prev cpu even if it results in a slight freq bump. In general, selecting prev cpu is beneficial because it avoids warming up the caches at a different cpu. Change-Id: Ic66ab18ba65f2917b73d9fbe92a39b743b9839a0 Signed-off-by: Abhijeet Dharmapurikar --- kernel/power/energy_model.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/kernel/power/energy_model.c b/kernel/power/energy_model.c index 0a9326f5f421..1ea7bfdd5784 100644 --- a/kernel/power/energy_model.c +++ b/kernel/power/energy_model.c @@ -141,7 +141,7 @@ static struct em_perf_domain *em_create_pd(cpumask_t *span, int nr_states, */ opp_eff = freq / power; if (opp_eff >= prev_opp_eff) - pr_warn("pd%d: hertz/watts ratio non-monotonically decreasing: em_cap_state %d >= em_cap_state%d\n", + pr_debug("pd%d: hertz/watts ratio non-monotonically decreasing: em_cap_state %d >= em_cap_state%d\n", cpu, i, i - 1); prev_opp_eff = opp_eff; } @@ -151,6 +151,10 @@ static struct em_perf_domain *em_create_pd(cpumask_t *span, int nr_states, for (i = 0; i < nr_states; i++) { table[i].cost = div64_u64(fmax * table[i].power, table[i].frequency); + if (i > 0 && (table[i].cost < table[i - 1].cost) && + (table[i].power > table[i - 1].power)) { + table[i].cost = table[i - 1].cost; + } } pd->table = table;