From 596006615b3b2de7c419cc80c4b396db16560658 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Tue, 22 Jan 2019 18:28:19 -0800 Subject: [PATCH 1/5] drivers: thermal: reintroduce notifier for max level transitions commit 547ecd3b0229362 ("drivers: thermal: cpu_cooling: Add a notifier for max level transitions") introduced notifier for max level transitions earlier, later, thermal drivers got refractored and removed chunk of code that was introduced to support notifier for max level transitions. Bringing back the code that was removed on top of refactored thermal code. Change-Id: Ifeb3a73a074d04f88e81f1d62dc4aecb51ef3cec Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/cpu_cooling.h | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/include/linux/cpu_cooling.h b/include/linux/cpu_cooling.h index bae54bb7c048..f54f92185cb9 100644 --- a/include/linux/cpu_cooling.h +++ b/include/linux/cpu_cooling.h @@ -62,4 +62,24 @@ of_cpufreq_cooling_register(struct cpufreq_policy *policy) } #endif /* defined(CONFIG_THERMAL_OF) && defined(CONFIG_CPU_THERMAL) */ +#ifdef CONFIG_QTI_CPU_ISOLATE_COOLING_DEVICE +extern void cpu_cooling_max_level_notifier_register(struct notifier_block *n); +extern void cpu_cooling_max_level_notifier_unregister(struct notifier_block *n); +extern const struct cpumask *cpu_cooling_get_max_level_cpumask(void); +#else +static inline +void cpu_cooling_max_level_notifier_register(struct notifier_block *n) +{ +} + +static inline +void cpu_cooling_max_level_notifier_unregister(struct notifier_block *n) +{ +} + +static inline const struct cpumask *cpu_cooling_get_max_level_cpumask(void) +{ + return cpu_none_mask; +} +#endif /* CONFIG_QTI_CPU_ISOLATE_COOLING_DEVICE */ #endif /* __CPU_COOLING_H__ */ From cff714d4234a4cb44397c6140a0e9887b9de1d74 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Fri, 6 Mar 2020 09:51:20 -0800 Subject: [PATCH 2/5] haven: hcall: Add vcpu affinity API vcpu affinity can be used to set or change the physical CPU affinity of a VCPU thread. Change-Id: I5f92205640ed1440e8ed0676e2bfcacbcef91f90 Signed-off-by: Satya Durga Srinivasu Prabhala --- include/linux/haven/hcall.h | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/include/linux/haven/hcall.h b/include/linux/haven/hcall.h index c39dd9d81893..a3d360406199 100644 --- a/include/linux/haven/hcall.h +++ b/include/linux/haven/hcall.h @@ -270,4 +270,17 @@ static inline int hh_hcall_msgq_configure_recv(hh_capid_t msgq_capid, return ret; } +static inline int hh_hcall_vcpu_affinity_set(hh_capid_t vcpu_capid, + uint32_t cpu_index) +{ + int ret; + struct hh_hcall_resp _resp = {0}; + + ret = _hh_hcall(0x603d, + (struct hh_hcall_args){ vcpu_capid, cpu_index, -1 }, + &_resp); + + return ret; +} + #endif From 8bb6422546d25b00e439c88fc869532d6f1b8800 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Fri, 6 Mar 2020 09:51:53 -0800 Subject: [PATCH 3/5] virt/haven: populate VCPU resources VCPU resources provides virtual CPU's affinity index which can be used to set or change the physical CPU affinity of a VCPU thread. Change-Id: I4889ed29ae71b828676e14d39a2e99eddd304add Signed-off-by: Satya Durga Srinivasu Prabhala --- drivers/virt/haven/hh_rm_core.c | 4 ++++ drivers/virt/haven/hh_rm_drv_private.h | 1 + include/linux/sched.h | 9 +++++++++ 3 files changed, 14 insertions(+) diff --git a/drivers/virt/haven/hh_rm_core.c b/drivers/virt/haven/hh_rm_core.c index 6be1b9e17924..25d0ff88dde5 100644 --- a/drivers/virt/haven/hh_rm_core.c +++ b/drivers/virt/haven/hh_rm_core.c @@ -16,6 +16,7 @@ #include #include #include +#include #include #include @@ -667,6 +668,9 @@ static int hh_rm_populate_hyp_res(void) ret = hh_msgq_populate_cap_info(label, cap_id, HH_MSGQ_DIRECTION_RX, linux_irq); break; + case HH_RM_RES_TYPE_VCPU: + ret = hh_vcpu_populate_affinity_info(label, cap_id); + break; case HH_RM_RES_TYPE_DB_TX: break; case HH_RM_RES_TYPE_DB_RX: diff --git a/drivers/virt/haven/hh_rm_drv_private.h b/drivers/virt/haven/hh_rm_drv_private.h index 4aa68ef98ddb..b71d59353559 100644 --- a/drivers/virt/haven/hh_rm_drv_private.h +++ b/drivers/virt/haven/hh_rm_drv_private.h @@ -139,6 +139,7 @@ struct hh_vm_console_write_resp_payload { #define HH_RM_RES_TYPE_DB_RX 1 #define HH_RM_RES_TYPE_MQ_TX 2 #define HH_RM_RES_TYPE_MQ_RX 3 +#define HH_RM_RES_TYPE_VCPU 4 struct hh_vm_get_hyp_res_req_payload { u16 vmid; diff --git a/include/linux/sched.h b/include/linux/sched.h index 5b68f8f56b60..c129efc6cc0f 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -540,6 +540,15 @@ struct cpu_cycle_counter_cb { DECLARE_PER_CPU_READ_MOSTLY(int, sched_load_boost); +#ifdef CONFIG_QCOM_HYP_CORE_CTL +extern int hh_vcpu_populate_affinity_info(u32 cpu_index, u64 cap_id); +#else +static inline int hh_vcpu_populate_affinity_info(u32 cpu_index, u64 cap_id) +{ + return 0; +} +#endif /* CONFIG_QCOM_HYP_CORE_CTL */ + #ifdef CONFIG_SCHED_WALT extern void sched_exit(struct task_struct *p); extern int __weak From 0d4657c6f48e3196234f130988b1269e794906e5 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Fri, 6 Mar 2020 10:27:57 -0800 Subject: [PATCH 4/5] soc: qcom: Add snapshot of hyp_core_ctl driver This snapshot is taken from msm-4.19 as of commit eee86f2a173b344 ("soc: qcom: Add snapshot of hyp_core_ctl driver") and updated driver as needed. Change-Id: I0e05043fec6dae76818720b5e542f85f701dae05 Signed-off-by: Satya Durga Srinivasu Prabhala --- drivers/soc/qcom/Kconfig | 9 + drivers/soc/qcom/Makefile | 1 + drivers/soc/qcom/hyp_core_ctl.c | 864 ++++++++++++++++++++++++++++ include/linux/cpuhotplug.h | 3 + include/trace/events/hyp_core_ctl.h | 72 +++ 5 files changed, 949 insertions(+) create mode 100644 drivers/soc/qcom/hyp_core_ctl.c create mode 100644 include/trace/events/hyp_core_ctl.h diff --git a/drivers/soc/qcom/Kconfig b/drivers/soc/qcom/Kconfig index 3906c604fe93..6bd031005283 100644 --- a/drivers/soc/qcom/Kconfig +++ b/drivers/soc/qcom/Kconfig @@ -763,4 +763,13 @@ config QCOM_GUESTVM Resource Manager driver to start the boot of VMs once it has successfully loaded the VM images in the designated memory. +config QCOM_HYP_CORE_CTL + bool "CPU reservation scheme for Hypervisor" + depends on QCOM_GUESTVM + help + This driver reserve the specified CPUS by isolating them. The reserved + CPUs can be assigned to the other guest OS by the hypervisor. + An offline CPU is considered as a reserved CPU since this OS can't use + it. + endmenu diff --git a/drivers/soc/qcom/Makefile b/drivers/soc/qcom/Makefile index b3d972094287..90dbe932a453 100644 --- a/drivers/soc/qcom/Makefile +++ b/drivers/soc/qcom/Makefile @@ -72,4 +72,5 @@ obj-$(CONFIG_MSM_SPCOM) += spcom.o obj-$(CONFIG_QCOM_FSA4480_I2C) += fsa4480-i2c.o obj-$(CONFIG_QCOM_EUD) += eud.o obj-$(CONFIG_QCOM_GUESTVM) += guestvm_loader.o +obj-$(CONFIG_QCOM_HYP_CORE_CTL) += hyp_core_ctl.o obj-$(CONFIG_MSM_QBT_HANDLER) += qbt_handler.o diff --git a/drivers/soc/qcom/hyp_core_ctl.c b/drivers/soc/qcom/hyp_core_ctl.c new file mode 100644 index 000000000000..8ef2dee5dbeb --- /dev/null +++ b/drivers/soc/qcom/hyp_core_ctl.c @@ -0,0 +1,864 @@ +// SPDX-License-Identifier: GPL-2.0-only +/* + * Copyright (c) 2018-2020, The Linux Foundation. All rights reserved. + */ + +#define pr_fmt(fmt) "hyp_core_ctl: " fmt + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include + +#define MAX_RESERVE_CPUS (num_possible_cpus()/2) + +/** + * struct hyp_core_ctl_cpumap - vcpu to pcpu mapping for the other guest + * @sid: System call id to be used while referring to this vcpu + * @pcpu: The physical CPU number corresponding to this vcpu + * @curr_pcpu: The current physical CPU number corresponding to this vcpu. + * The curr_pcu is set to another CPU when the original assigned + * CPU i.e pcpu can't be used due to thermal condition. + * + */ +struct hyp_core_ctl_cpu_map { + hh_capid_t sid; + hh_label_t pcpu; + hh_label_t curr_pcpu; +}; + +/** + * struct hyp_core_ctl_data - The private data structure of this driver + * @lock: spinlock to serialize task wakeup and enable/reserve_cpus + * @task: task_struct pointer to the thread running the state machine + * @pending: state machine work pending status + * @reservation_enabled: status of the reservation + * @reservation_mutex: synchronization between thermal handling and + * reservation. The physical CPUs are re-assigned + * during thermal conditions while reservation is + * not enabled. So this synchronization is needed. + * @reserve_cpus: The CPUs to be reserved. input. + * @our_isolated_cpus: The CPUs isolated by hyp_core_ctl driver. output. + * @final_reserved_cpus: The CPUs reserved for the Hypervisor. output. + * @cpumap: The vcpu to pcpu mapping table + */ +struct hyp_core_ctl_data { + spinlock_t lock; + struct task_struct *task; + bool pending; + bool reservation_enabled; + struct mutex reservation_mutex; + cpumask_t reserve_cpus; + cpumask_t our_isolated_cpus; + cpumask_t final_reserved_cpus; + struct hyp_core_ctl_cpu_map cpumap[NR_CPUS]; +}; + +#define CREATE_TRACE_POINTS +#include + +static struct hyp_core_ctl_data *the_hcd; +static struct hyp_core_ctl_cpu_map hh_cpumap[NR_CPUS]; +static bool populated_vcpu_info; +static bool init_done; +static int nr_vcpus; + +static inline void hyp_core_ctl_print_status(char *msg) +{ + trace_hyp_core_ctl_status(the_hcd, msg); + + pr_debug("%s: reserve=%*pbl reserved=%*pbl our_isolated=%*pbl online=%*pbl isolated=%*pbl thermal=%*pbl\n", + msg, cpumask_pr_args(&the_hcd->reserve_cpus), + cpumask_pr_args(&the_hcd->final_reserved_cpus), + cpumask_pr_args(&the_hcd->our_isolated_cpus), + cpumask_pr_args(cpu_online_mask), + cpumask_pr_args(cpu_isolated_mask), + cpumask_pr_args(cpu_cooling_get_max_level_cpumask())); +} + +static void hyp_core_ctl_undo_reservation(struct hyp_core_ctl_data *hcd) +{ + int cpu, ret; + + hyp_core_ctl_print_status("undo_reservation_start"); + + for_each_cpu(cpu, &hcd->our_isolated_cpus) { + ret = sched_unisolate_cpu(cpu); + if (ret < 0) { + pr_err("fail to un-isolate CPU%d. ret=%d\n", cpu, ret); + continue; + } + cpumask_clear_cpu(cpu, &hcd->our_isolated_cpus); + } + + hyp_core_ctl_print_status("undo_reservation_end"); +} + +static void finalize_reservation(struct hyp_core_ctl_data *hcd, cpumask_t *temp) +{ + cpumask_t vcpu_adjust_mask; + int i, orig_cpu, curr_cpu, replacement_cpu; + int err; + + /* + * When thermal conditions are not present, we return + * from here. + */ + if (cpumask_equal(temp, &hcd->final_reserved_cpus)) + return; + + /* + * When we can't match with the original reserve CPUs request, + * don't change the existing scheme. We can't assign the + * same physical CPU to multiple virtual CPUs. + * + * This may only happen when thermal isolate more CPUs. + */ + if (cpumask_weight(temp) < cpumask_weight(&hcd->reserve_cpus)) { + pr_debug("Fail to reserve some CPUs\n"); + return; + } + + cpumask_copy(&hcd->final_reserved_cpus, temp); + cpumask_clear(&vcpu_adjust_mask); + + /* + * In the first pass, we traverse all virtual CPUs and try + * to assign their original physical CPUs if they are + * reserved. if the original physical CPU is not reserved, + * then check the current physical CPU is reserved or not. + * so that we continue to use the current physical CPU. + * + * If both original CPU and the current CPU are not reserved, + * we have to find a replacement. These virtual CPUs are + * maintained in vcpu_adjust_mask and processed in the 2nd pass. + */ + for (i = 0; i < MAX_RESERVE_CPUS; i++) { + if (hcd->cpumap[i].sid == 0) + break; + + orig_cpu = hcd->cpumap[i].pcpu; + curr_cpu = hcd->cpumap[i].curr_pcpu; + + if (cpumask_test_cpu(orig_cpu, &hcd->final_reserved_cpus)) { + cpumask_clear_cpu(orig_cpu, temp); + + if (orig_cpu == curr_cpu) + continue; + + /* + * The original pcpu corresponding to this vcpu i.e i + * is available in final_reserved_cpus. so restore + * the assignment. + */ + err = hh_hcall_vcpu_affinity_set(hcd->cpumap[i].sid, + orig_cpu); + if (err != HH_ERROR_OK) { + pr_err("fail to assign pcpu for vcpu#%d\n", i); + continue; + } + + hcd->cpumap[i].curr_pcpu = orig_cpu; + pr_debug("err=%u vcpu=%d pcpu=%u curr_cpu=%u\n", + err, i, hcd->cpumap[i].pcpu, + hcd->cpumap[i].curr_pcpu); + continue; + } + + /* + * The original CPU is not available but the previously + * assigned CPU i.e curr_cpu is still available. so keep + * using it. + */ + if (cpumask_test_cpu(curr_cpu, &hcd->final_reserved_cpus)) { + cpumask_clear_cpu(curr_cpu, temp); + continue; + } + + /* + * A replacement CPU is found in the 2nd pass below. Make + * a note of this virtual CPU for which both original and + * current physical CPUs are not available in the + * final_reserved_cpus. + */ + cpumask_set_cpu(i, &vcpu_adjust_mask); + } + + /* + * The vcpu_adjust_mask contain the virtual CPUs that needs + * re-assignment. The temp CPU mask contains the remaining + * reserved CPUs. so we pick one by one from the remaining + * reserved CPUs and assign them to the pending virtual + * CPUs. + */ + for_each_cpu(i, &vcpu_adjust_mask) { + replacement_cpu = cpumask_any(temp); + cpumask_clear_cpu(replacement_cpu, temp); + + err = hh_hcall_vcpu_affinity_set(hcd->cpumap[i].sid, + replacement_cpu); + if (err != HH_ERROR_OK) { + pr_err("fail to assign pcpu for vcpu#%d\n", i); + continue; + } + + hcd->cpumap[i].curr_pcpu = replacement_cpu; + pr_debug("adjust err=%u vcpu=%d pcpu=%u curr_cpu=%u\n", + err, i, hcd->cpumap[i].pcpu, + hcd->cpumap[i].curr_pcpu); + + } + + /* Did we reserve more CPUs than needed? */ + WARN_ON(!cpumask_empty(temp)); +} + +static void hyp_core_ctl_do_reservation(struct hyp_core_ctl_data *hcd) +{ + cpumask_t offline_cpus, iter_cpus, temp_reserved_cpus; + int i, ret, iso_required, iso_done; + const cpumask_t *thermal_cpus = cpu_cooling_get_max_level_cpumask(); + + cpumask_clear(&offline_cpus); + cpumask_clear(&temp_reserved_cpus); + + hyp_core_ctl_print_status("reservation_start"); + + /* + * Iterate all reserve CPUs and isolate them if not done already. + * The offline CPUs can't be isolated but they are considered + * reserved. When an offline and reserved CPU comes online, it + * will be isolated to honor the reservation. + */ + cpumask_andnot(&iter_cpus, &hcd->reserve_cpus, &hcd->our_isolated_cpus); + cpumask_andnot(&iter_cpus, &iter_cpus, thermal_cpus); + + for_each_cpu(i, &iter_cpus) { + if (!cpu_online(i)) { + cpumask_set_cpu(i, &offline_cpus); + continue; + } + + ret = sched_isolate_cpu(i); + if (ret < 0) { + pr_debug("fail to isolate CPU%d. ret=%d\n", i, ret); + continue; + } + cpumask_set_cpu(i, &hcd->our_isolated_cpus); + } + + cpumask_andnot(&iter_cpus, &hcd->reserve_cpus, &offline_cpus); + iso_required = cpumask_weight(&iter_cpus); + iso_done = cpumask_weight(&hcd->our_isolated_cpus); + + if (iso_done < iso_required) { + int isolate_need; + + /* + * We have isolated fewer CPUs than required. This happens + * when some of the CPUs from the reserved_cpus mask + * are managed by thermal. Find the replacement CPUs and + * isolate them. + */ + isolate_need = iso_required - iso_done; + + /* + * Create a cpumask from which replacement CPUs can be + * picked. Exclude our isolated CPUs, thermal managed + * CPUs and offline CPUs, which are already considered + * as reserved. + */ + cpumask_andnot(&iter_cpus, cpu_possible_mask, + &hcd->our_isolated_cpus); + cpumask_andnot(&iter_cpus, &iter_cpus, thermal_cpus); + cpumask_andnot(&iter_cpus, &iter_cpus, &offline_cpus); + + /* + * Keep the replacement policy simple. The offline CPUs + * comes for free. so pick them first. + */ + for_each_cpu(i, &iter_cpus) { + if (!cpu_online(i)) { + cpumask_set_cpu(i, &offline_cpus); + if (--isolate_need == 0) + goto done; + } + } + + cpumask_andnot(&iter_cpus, &iter_cpus, &offline_cpus); + + for_each_cpu(i, &iter_cpus) { + ret = sched_isolate_cpu(i); + if (ret < 0) { + pr_debug("fail to isolate CPU%d. ret=%d\n", + i, ret); + continue; + } + cpumask_set_cpu(i, &hcd->our_isolated_cpus); + + if (--isolate_need == 0) + break; + } + } else if (iso_done > iso_required) { + int unisolate_need; + + /* + * We have isolated more CPUs than required. Un-isolate + * the additional CPUs which are not part of the + * reserve_cpus mask. + * + * This happens in the following scenario. + * + * - Lets say reserve CPUs are CPU4 and CPU5. They are + * isolated. + * - CPU4 is isolated by thermal. We found CPU0 as the + * replacement CPU. Now CPU0 and CPU5 are isolated by + * us. + * - CPU4 is un-isolated by thermal. We first isolate CPU4 + * since it is part of our reserve CPUs. Now CPU0, CPU4 + * and CPU5 are isolated by us. + * - Since iso_done (3) > iso_required (2), un-isolate + * a CPU which is not part of the reserve CPU. i.e CPU0. + */ + unisolate_need = iso_done - iso_required; + cpumask_andnot(&iter_cpus, &hcd->our_isolated_cpus, + &hcd->reserve_cpus); + for_each_cpu(i, &iter_cpus) { + ret = sched_unisolate_cpu(i); + if (ret < 0) { + pr_err("fail to unisolate CPU%d. ret=%d\n", + i, ret); + continue; + } + cpumask_clear_cpu(i, &hcd->our_isolated_cpus); + if (--unisolate_need == 0) + break; + } + } + +done: + cpumask_or(&temp_reserved_cpus, &hcd->our_isolated_cpus, &offline_cpus); + finalize_reservation(hcd, &temp_reserved_cpus); + + hyp_core_ctl_print_status("reservation_end"); +} + +static int hyp_core_ctl_thread(void *data) +{ + struct hyp_core_ctl_data *hcd = data; + + while (1) { + spin_lock(&hcd->lock); + if (!hcd->pending) { + set_current_state(TASK_INTERRUPTIBLE); + spin_unlock(&hcd->lock); + + schedule(); + + spin_lock(&hcd->lock); + set_current_state(TASK_RUNNING); + } + hcd->pending = false; + spin_unlock(&hcd->lock); + + if (kthread_should_stop()) + break; + + /* + * The reservation mutex synchronize the reservation + * happens in this thread against the thermal handling. + * The CPU re-assignment happens directly from the + * thermal callback context when the reservation is + * not enabled, since there is no need for isolating. + */ + mutex_lock(&hcd->reservation_mutex); + if (hcd->reservation_enabled) + hyp_core_ctl_do_reservation(hcd); + else + hyp_core_ctl_undo_reservation(hcd); + mutex_unlock(&hcd->reservation_mutex); + } + + return 0; +} + +static void hyp_core_ctl_handle_thermal(struct hyp_core_ctl_data *hcd, + int cpu, bool throttled) +{ + cpumask_t temp_mask, iter_cpus; + const cpumask_t *thermal_cpus = cpu_cooling_get_max_level_cpumask(); + bool notify = false; + int replacement_cpu; + + hyp_core_ctl_print_status("handle_thermal_start"); + + /* + * Take a copy of the final_reserved_cpus and adjust the mask + * based on the notified CPU's thermal state. + */ + cpumask_copy(&temp_mask, &hcd->final_reserved_cpus); + + if (throttled) { + /* + * Find a replacement CPU for this throttled CPU. Select + * any CPU that is not managed by thermal and not already + * part of the assigned CPUs. + */ + cpumask_andnot(&iter_cpus, cpu_possible_mask, thermal_cpus); + cpumask_andnot(&iter_cpus, &iter_cpus, + &hcd->final_reserved_cpus); + replacement_cpu = cpumask_any(&iter_cpus); + + if (replacement_cpu < nr_cpu_ids) { + cpumask_clear_cpu(cpu, &temp_mask); + cpumask_set_cpu(replacement_cpu, &temp_mask); + notify = true; + } + } else { + /* + * One of the original assigned CPU is unthrottled by thermal. + * Swap this CPU with any one of the replacement CPUs. + */ + cpumask_andnot(&iter_cpus, &hcd->final_reserved_cpus, + &hcd->reserve_cpus); + replacement_cpu = cpumask_any(&iter_cpus); + + if (replacement_cpu < nr_cpu_ids) { + cpumask_clear_cpu(replacement_cpu, &temp_mask); + cpumask_set_cpu(cpu, &temp_mask); + notify = true; + } + } + + if (notify) + finalize_reservation(hcd, &temp_mask); + + hyp_core_ctl_print_status("handle_thermal_end"); +} + +static int hyp_core_ctl_cpu_cooling_cb(struct notifier_block *nb, + unsigned long val, void *data) +{ + int cpu = (long) data; + const cpumask_t *thermal_cpus = cpu_cooling_get_max_level_cpumask(); + + if (!the_hcd) + return NOTIFY_DONE; + + mutex_lock(&the_hcd->reservation_mutex); + + pr_debug("CPU%d is %s by thermal\n", cpu, + val ? "throttled" : "unthrottled"); + + if (val) { + /* + * The thermal mitigated CPU is not part of our reserved + * CPUs. So nothing to do. + */ + if (!cpumask_test_cpu(cpu, &the_hcd->final_reserved_cpus)) + goto out; + + /* + * The thermal mitigated CPU is part of our reserved CPUs. + * + * If it is isolated by us, unisolate it. If it is not + * isolated, probably it is offline. In both cases, kick + * the state machine to find a replacement CPU. + */ + if (cpumask_test_cpu(cpu, &the_hcd->our_isolated_cpus)) { + sched_unisolate_cpu(cpu); + cpumask_clear_cpu(cpu, &the_hcd->our_isolated_cpus); + } + } else { + /* + * A CPU is unblocked by thermal. We are interested if + * + * (1) This CPU is part of the original reservation request + * In this case, this CPU should be swapped with one of + * the replacement CPU that is currently reserved. + * (2) When some of the thermal mitigated CPUs are currently + * reserved due to unavailability of CPUs. Now that + * thermal unblocked a CPU, swap this with one of the + * thermal mitigated CPU that is currently reserved. + */ + if (!cpumask_test_cpu(cpu, &the_hcd->reserve_cpus) && + !cpumask_intersects(&the_hcd->final_reserved_cpus, + thermal_cpus)) + goto out; + } + + if (the_hcd->reservation_enabled) { + spin_lock(&the_hcd->lock); + the_hcd->pending = true; + wake_up_process(the_hcd->task); + spin_unlock(&the_hcd->lock); + } else { + /* + * When the reservation is enabled, the state machine + * takes care of finding the new replacement CPU or + * isolating the unthrottled CPU. However when the + * reservation is not enabled, we still want to + * re-assign another CPU for a throttled CPU. + */ + hyp_core_ctl_handle_thermal(the_hcd, cpu, val); + } +out: + mutex_unlock(&the_hcd->reservation_mutex); + return NOTIFY_OK; +} + +static struct notifier_block hyp_core_ctl_nb = { + .notifier_call = hyp_core_ctl_cpu_cooling_cb, +}; + +static int hyp_core_ctl_hp_offline(unsigned int cpu) +{ + if (!the_hcd || !the_hcd->reservation_enabled) + return 0; + + /* + * A CPU can't be left in isolated state while it is + * going offline. So unisolate the CPU if it is + * isolated by us. An offline CPU is considered + * as reserved. So no further action is needed. + */ + if (cpumask_test_and_clear_cpu(cpu, &the_hcd->our_isolated_cpus)) + sched_unisolate_cpu_unlocked(cpu); + + return 0; +} + +static int hyp_core_ctl_hp_online(unsigned int cpu) +{ + if (!the_hcd || !the_hcd->reservation_enabled) + return 0; + + /* + * A reserved CPU is coming online. It should be isolated + * to honor the reservation. So kick the state machine. + */ + spin_lock(&the_hcd->lock); + if (cpumask_test_cpu(cpu, &the_hcd->final_reserved_cpus)) { + the_hcd->pending = true; + wake_up_process(the_hcd->task); + } + spin_unlock(&the_hcd->lock); + + return 0; +} + +static int hyp_core_ctl_init_reserve_cpus(struct hyp_core_ctl_data *hcd) +{ + int i, ret = 0; + + spin_lock(&hcd->lock); + cpumask_clear(&hcd->reserve_cpus); + + for (i = 0; i < MAX_RESERVE_CPUS; i++) { + if (hh_cpumap[i].sid == 0) + break; + + hcd->cpumap[i].sid = hh_cpumap[i].sid; + hcd->cpumap[i].pcpu = hh_cpumap[i].pcpu; + hcd->cpumap[i].curr_pcpu = hh_cpumap[i].curr_pcpu; + cpumask_set_cpu(hcd->cpumap[i].pcpu, &hcd->reserve_cpus); + pr_debug("vcpu%u map to pcpu%u\n", i, hcd->cpumap[i].pcpu); + } + + cpumask_copy(&hcd->final_reserved_cpus, &hcd->reserve_cpus); + spin_unlock(&hcd->lock); + pr_info("reserve_cpus=%*pbl ret=%d\n", + cpumask_pr_args(&hcd->reserve_cpus), ret); + + return ret; +} + +int hh_vcpu_populate_affinity_info(u32 cpu_idx, u64 cap_id) +{ + static struct hyp_core_ctl_cpu_map hh_cpumap[NR_CPUS]; + + hh_cpumap[nr_vcpus].sid = cap_id; + hh_cpumap[nr_vcpus].pcpu = cpu_idx; + hh_cpumap[nr_vcpus].curr_pcpu = cpu_idx; + + if (!populated_vcpu_info) + populated_vcpu_info = true; + + if (init_done) + hyp_core_ctl_init_reserve_cpus(the_hcd); + + nr_vcpus++; + pr_debug("cpu_index:%u vcpu_cap_id:%llu nr_vcpus:%d\n", + cpu_idx, cap_id, nr_vcpus); + return 0; +} + +static void hyp_core_ctl_enable(bool enable) +{ + spin_lock(&the_hcd->lock); + if (enable == the_hcd->reservation_enabled) + goto out; + + trace_hyp_core_ctl_enable(enable); + pr_debug("reservation %s\n", enable ? "enabled" : "disabled"); + + the_hcd->reservation_enabled = enable; + the_hcd->pending = true; + wake_up_process(the_hcd->task); +out: + spin_unlock(&the_hcd->lock); +} + +static ssize_t enable_store(struct device *dev, struct device_attribute *attr, + const char *buf, size_t count) +{ + bool enable; + int ret; + + ret = kstrtobool(buf, &enable); + if (ret < 0) + return -EINVAL; + + hyp_core_ctl_enable(enable); + + return count; +} + +static ssize_t enable_show(struct device *dev, struct device_attribute *attr, + char *buf) +{ + return scnprintf(buf, PAGE_SIZE, "%u\n", the_hcd->reservation_enabled); +} + +static DEVICE_ATTR_RW(enable); + +static ssize_t status_show(struct device *dev, struct device_attribute *attr, + char *buf) +{ + struct hyp_core_ctl_data *hcd = the_hcd; + ssize_t count; + int i; + + mutex_lock(&hcd->reservation_mutex); + + count = scnprintf(buf, PAGE_SIZE, "enabled=%d\n", + hcd->reservation_enabled); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "reserve_cpus=%*pbl\n", + cpumask_pr_args(&hcd->reserve_cpus)); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "reserved_cpus=%*pbl\n", + cpumask_pr_args(&hcd->final_reserved_cpus)); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "our_isolated_cpus=%*pbl\n", + cpumask_pr_args(&hcd->our_isolated_cpus)); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "online_cpus=%*pbl\n", + cpumask_pr_args(cpu_online_mask)); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "isolated_cpus=%*pbl\n", + cpumask_pr_args(cpu_isolated_mask)); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "thermal_cpus=%*pbl\n", + cpumask_pr_args(cpu_cooling_get_max_level_cpumask())); + + count += scnprintf(buf + count, PAGE_SIZE - count, + "Vcpu to Pcpu mappings:\n"); + + for (i = 0; i < MAX_RESERVE_CPUS; i++) { + if (hcd->cpumap[i].sid == 0) + break; + + count += scnprintf(buf + count, PAGE_SIZE - count, + "vcpu=%d pcpu=%u curr_pcpu=%u\n", + i, hcd->cpumap[i].pcpu, hcd->cpumap[i].curr_pcpu); + + } + + mutex_unlock(&hcd->reservation_mutex); + + return count; +} + +static DEVICE_ATTR_RO(status); + +static struct attribute *hyp_core_ctl_attrs[] = { + &dev_attr_enable.attr, + &dev_attr_status.attr, + NULL +}; + +static struct attribute_group hyp_core_ctl_attr_group = { + .attrs = hyp_core_ctl_attrs, + .name = "hyp_core_ctl", +}; + +#define CPULIST_SZ 32 +static ssize_t read_reserve_cpus(struct file *file, char __user *ubuf, + size_t count, loff_t *ppos) +{ + char kbuf[CPULIST_SZ]; + int ret; + + ret = scnprintf(kbuf, CPULIST_SZ, "%*pbl\n", + cpumask_pr_args(&the_hcd->reserve_cpus)); + + return simple_read_from_buffer(ubuf, count, ppos, kbuf, ret); +} + +static ssize_t write_reserve_cpus(struct file *file, const char __user *ubuf, + size_t count, loff_t *ppos) +{ + char kbuf[CPULIST_SZ]; + int ret; + cpumask_t temp_mask; + + ret = simple_write_to_buffer(kbuf, CPULIST_SZ - 1, ppos, ubuf, count); + if (ret < 0) + return ret; + + kbuf[ret] = '\0'; + ret = cpulist_parse(kbuf, &temp_mask); + if (ret < 0) + return ret; + + if (cpumask_weight(&temp_mask) != + cpumask_weight(&the_hcd->reserve_cpus)) { + pr_err("incorrect reserve CPU count. expected=%u\n", + cpumask_weight(&the_hcd->reserve_cpus)); + return -EINVAL; + } + + spin_lock(&the_hcd->lock); + if (the_hcd->reservation_enabled) { + count = -EPERM; + pr_err("reservation is enabled, can't change reserve_cpus\n"); + } else { + cpumask_copy(&the_hcd->reserve_cpus, &temp_mask); + } + spin_unlock(&the_hcd->lock); + + return count; +} + +static const struct file_operations debugfs_reserve_cpus_ops = { + .read = read_reserve_cpus, + .write = write_reserve_cpus, +}; + +static void hyp_core_ctl_debugfs_init(void) +{ + struct dentry *dir, *file; + + dir = debugfs_create_dir("hyp_core_ctl", NULL); + if (IS_ERR_OR_NULL(dir)) + return; + + file = debugfs_create_file("reserve_cpus", 0644, dir, NULL, + &debugfs_reserve_cpus_ops); + if (!file) + debugfs_remove(dir); +} + +static int hyp_core_ctl_probe(struct platform_device *pdev) +{ + int ret; + struct hyp_core_ctl_data *hcd; + struct sched_param param = { .sched_priority = MAX_RT_PRIO - 1 }; + + if (!populated_vcpu_info) { + pr_debug("VCPU info isn't populated, retry\n"); + ret = -EPROBE_DEFER; + goto out; + } + + hcd = kzalloc(sizeof(*hcd), GFP_KERNEL); + if (!hcd) { + ret = -ENOMEM; + goto out; + } + + ret = hyp_core_ctl_init_reserve_cpus(hcd); + if (ret < 0) { + pr_err("Fail to get reserve CPUs from Hyp. ret=%d\n", ret); + goto free_hcd; + } + + spin_lock_init(&hcd->lock); + mutex_init(&hcd->reservation_mutex); + hcd->task = kthread_run(hyp_core_ctl_thread, (void *) hcd, + "hyp_core_ctl"); + + if (IS_ERR(hcd->task)) { + ret = PTR_ERR(hcd->task); + goto free_hcd; + } + + sched_setscheduler_nocheck(hcd->task, SCHED_FIFO, ¶m); + + ret = sysfs_create_group(&cpu_subsys.dev_root->kobj, + &hyp_core_ctl_attr_group); + if (ret < 0) { + pr_err("Fail to create sysfs files. ret=%d\n", ret); + goto stop_task; + } + + cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN, + "qcom/hyp_core_ctl:online", + hyp_core_ctl_hp_online, NULL); + + cpuhp_setup_state_nocalls(CPUHP_HYP_CORE_CTL_ISOLATION_DEAD, + "qcom/hyp_core_ctl:dead", + NULL, hyp_core_ctl_hp_offline); + + cpu_cooling_max_level_notifier_register(&hyp_core_ctl_nb); + hyp_core_ctl_debugfs_init(); + + the_hcd = hcd; + init_done = true; + return 0; + +stop_task: + kthread_stop(hcd->task); +free_hcd: + kfree(hcd); +out: + return ret; +} + +static const struct of_device_id hyp_core_ctl_match_table[] = { + { .compatible = "qcom,hyp-core-ctl" }, + {}, +}; + +static struct platform_driver hyp_core_ctl_driver = { + .probe = hyp_core_ctl_probe, + .driver = { + .name = "hyp_core_ctl", + .owner = THIS_MODULE, + .of_match_table = hyp_core_ctl_match_table, + }, +}; + +builtin_platform_driver(hyp_core_ctl_driver); +MODULE_DESCRIPTION("Core Control for Hypervisor"); +MODULE_LICENSE("GPL v2"); diff --git a/include/linux/cpuhotplug.h b/include/linux/cpuhotplug.h index 206612b1f58e..67e4885df05d 100644 --- a/include/linux/cpuhotplug.h +++ b/include/linux/cpuhotplug.h @@ -71,6 +71,9 @@ enum cpuhp_state { CPUHP_RCUTREE_PREP, #ifdef CONFIG_SCHED_WALT CPUHP_CORE_CTL_ISOLATION_DEAD, +#endif +#ifdef CONFIG_QCOM_HYP_CORE_CTL + CPUHP_HYP_CORE_CTL_ISOLATION_DEAD, #endif CPUHP_CPUIDLE_COUPLED_PREPARE, CPUHP_POWERPC_PMAC_PREPARE, diff --git a/include/trace/events/hyp_core_ctl.h b/include/trace/events/hyp_core_ctl.h new file mode 100644 index 000000000000..6aafbdcfe2e6 --- /dev/null +++ b/include/trace/events/hyp_core_ctl.h @@ -0,0 +1,72 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + *Copyright (c) 2018 The Linux Foundation. All rights reserved. + */ + +#undef TRACE_SYSTEM +#define TRACE_SYSTEM hyp_core_ctl + +#if !defined(_TRACE_HYP_CORE_CTL_H) || defined(TRACE_HEADER_MULTI_READ) +#define _TRACE_HYP_CORE_CTL_H + +#include + +TRACE_EVENT(hyp_core_ctl_enable, + + TP_PROTO(bool enable), + + TP_ARGS(enable), + + TP_STRUCT__entry( + __field(bool, enable) + ), + + TP_fast_assign( + __entry->enable = enable; + ), + + TP_printk("enable=%d", __entry->enable) +); + +TRACE_EVENT(hyp_core_ctl_status, + + TP_PROTO(struct hyp_core_ctl_data *hcd, const char *event), + + TP_ARGS(hcd, event), + + TP_STRUCT__entry( + __string(event, event) + __array(char, reserve, 32) + __array(char, reserved, 32) + __array(char, our_isolated, 32) + __array(char, online, 32) + __array(char, isolated, 32) + __array(char, thermal, 32) + ), + + TP_fast_assign( + __assign_str(event, event); + scnprintf(__entry->reserve, sizeof(__entry->reserve), "%*pbl", + cpumask_pr_args(&hcd->reserve_cpus)); + scnprintf(__entry->reserved, sizeof(__entry->reserve), "%*pbl", + cpumask_pr_args(&hcd->final_reserved_cpus)); + scnprintf(__entry->our_isolated, sizeof(__entry->reserve), + "%*pbl", cpumask_pr_args(&hcd->our_isolated_cpus)); + scnprintf(__entry->online, sizeof(__entry->reserve), "%*pbl", + cpumask_pr_args(cpu_online_mask)); + scnprintf(__entry->isolated, sizeof(__entry->reserve), "%*pbl", + cpumask_pr_args(cpu_isolated_mask)); + scnprintf(__entry->thermal, sizeof(__entry->reserve), "%*pbl", + cpumask_pr_args(cpu_cooling_get_max_level_cpumask())); + ), + + TP_printk("event=%s reserve=%s reserved=%s our_isolated=%s online=%s isolated=%s thermal=%s", + __get_str(event), __entry->reserve, __entry->reserved, + __entry->our_isolated, __entry->online, __entry->isolated, + __entry->thermal) +); + +#endif /* _TRACE_HYP_CORE_CTL_H */ + +/* This part must be outside protection */ +#include From 3b62b33690734ea10deade271fcf3c4fdd33d422 Mon Sep 17 00:00:00 2001 From: Satya Durga Srinivasu Prabhala Date: Fri, 6 Mar 2020 10:30:11 -0800 Subject: [PATCH 5/5] defconfig: arm64: Enable hyp core control driver for Lahaiana hyp core control driver provides machanism to isolate/un-isolate CPUs as needed by other virtual machines in the system. Change-Id: I6ea459f20cc59ff82a87789ac389c4fab2bb44a9 Signed-off-by: Satya Durga Srinivasu Prabhala --- arch/arm64/configs/vendor/lahaina_QGKI.config | 1 + 1 file changed, 1 insertion(+) diff --git a/arch/arm64/configs/vendor/lahaina_QGKI.config b/arch/arm64/configs/vendor/lahaina_QGKI.config index 5e3c61b90f55..ca08e92081ee 100644 --- a/arch/arm64/configs/vendor/lahaina_QGKI.config +++ b/arch/arm64/configs/vendor/lahaina_QGKI.config @@ -15,6 +15,7 @@ CONFIG_IOMMU_IO_PGTABLE_FAST=y # CONFIG_IOMMU_IO_PGTABLE_FAST_PROVE_TLB is not set CONFIG_QCOM_LLCC_PERFMON=m CONFIG_SCHED_WALT=y +CONFIG_QCOM_HYP_CORE_CTL=y CONFIG_QGKI=y CONFIG_REGULATOR_QTI_DEBUG=y CONFIG_REGMAP_QTI_DEBUG=y