From 2f71f657c3bf66f1ee05cf2aba6509ee9a96071a Mon Sep 17 00:00:00 2001 From: Patrick Fay Date: Wed, 22 Jul 2020 15:19:29 +0530 Subject: [PATCH] Perf: core: create/delete shared kernel/user events This is a squash of the following two commits taken from msm-4.14 1) commit ("Perf: core: create/delete shared kernel events"). Frequently drivers want to monitor some event like L2 misses and cycles+instructions. If more than 1 driver wants to monitor the same event and the event attributes are the same then we can create just one instance of the event and let drivers share the event. Add shared event create and delete routines. 2) commit ("perf: Add support for user and kernel event sharing"). The ARM PMU counters are limited in number. Even for counting similar events, the PMU driver allocates a new counter. Hence, counters configured to count similar events are shared. This was only possible for the kernel clients, but not for user-space clients. Hence, as an extension to this, the kernel and the user-space are now able to share the similar events. The counters can be shared between user-space only clients, kernel only clients, and among user-space and kernel clients. The kernel and user's attr->type (hardware/raw) and attr->config should be same for them to share the same counter. Additionally, Current enablement of kernel event sharing is not GKI compliant. Make the sharing of kernel events configurable so it can be enabled only on non-GKI builds. Add dependency for enablement of sharing of user events on enablement of sharing of kernel events. Change-Id: I784db5a20968ba7fca20659186f23890bf6cc503 Signed-off-by: Patrick Fay Signed-off-by: Raghavendra Rao Ananta Signed-off-by: Rishabh Bhatnagar [kaushalk@codeaurora.org: Make event sharing configurable, add dependencies] Signed-off-by: Kaushal Kumar --- include/linux/perf_event.h | 4 + init/Kconfig | 20 +++ kernel/events/core.c | 325 ++++++++++++++++++++++++++++++++++--- 3 files changed, 330 insertions(+), 19 deletions(-) diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h index 48685d6ca501..27542442c0d0 100644 --- a/include/linux/perf_event.h +++ b/include/linux/perf_event.h @@ -726,6 +726,10 @@ struct perf_event { void *security; #endif struct list_head sb_list; +#ifdef CONFIG_PERF_KERNEL_SHARE + /* Is this event shared with other events */ + bool shared; +#endif #endif /* CONFIG_PERF_EVENTS */ }; diff --git a/init/Kconfig b/init/Kconfig index df4180822716..12c402302a55 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1731,6 +1731,26 @@ config PERF_EVENTS Say Y if unsure. +config PERF_KERNEL_SHARE + bool "Perf event sharing with kernel-space" + depends on QGKI + help + Say yes here to enable the kernel-space sharing of events. The events + can be shared among other kernel-space events or with kernel created + events that has the same config and type event attributes. + + Say N if unsure. + +config PERF_USER_SHARE + bool "Perf event sharing with user-space" + depends on PERF_KERNEL_SHARE + help + Say yes here to enable the user-space sharing of events. The events + can be shared among other user-space events or with kernel created + events that has the same config and type event attributes. + + Say N if unsure. + config DEBUG_PERF_USE_VMALLOC default n bool "Debug: use vmalloc to back perf mmap() buffers" diff --git a/kernel/events/core.c b/kernel/events/core.c index 4ebcaa758f6f..0ac39045d040 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -361,6 +361,34 @@ enum event_type_t { EVENT_ALL = EVENT_FLEXIBLE | EVENT_PINNED, }; +#ifdef CONFIG_PERF_KERNEL_SHARE +/* The shared events struct. */ +#define SHARED_EVENTS_MAX 7 + +struct shared_events_str { + /* + * Mutex to serialize access to shared list. Needed for the + * read/modify/write sequences. + */ + struct mutex list_mutex; + + /* + * A 1 bit for an index indicates that the slot is being used for + * an event. A 0 means that the slot can be used. + */ + DECLARE_BITMAP(used_mask, SHARED_EVENTS_MAX); + + /* + * The kernel events that are shared for a cpu; + */ + struct perf_event *events[SHARED_EVENTS_MAX]; + struct perf_event_attr attr[SHARED_EVENTS_MAX]; + atomic_t refcount[SHARED_EVENTS_MAX]; +}; + +static struct shared_events_str __percpu *shared_events; +#endif + /* * perf_sched_events : >0 events exist * perf_cgroup_events: >0 per-cpu cgroup events exist on this cpu @@ -1991,6 +2019,10 @@ static void perf_group_detach(struct perf_event *event) if (event->group_leader != event) { list_del_init(&event->sibling_list); event->group_leader->nr_siblings--; +#ifdef CONFIG_PERF_KERNEL_SHARE + if (event->shared) + event->group_leader = event; +#endif goto out; } @@ -4532,6 +4564,37 @@ static bool exclusive_event_installable(struct perf_event *event, static void perf_addr_filters_splice(struct perf_event *event, struct list_head *head); +#ifdef CONFIG_PERF_KERNEL_SHARE +static int +perf_event_delete_kernel_shared(struct perf_event *event) +{ + int rc = -1, cpu = event->cpu; + struct shared_events_str *shrd_events; + unsigned long idx; + + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return 0; + + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + for_each_set_bit(idx, shrd_events->used_mask, SHARED_EVENTS_MAX) { + if (shrd_events->events[idx] == event) { + if (atomic_dec_and_test(&shrd_events->refcount[idx])) { + clear_bit(idx, shrd_events->used_mask); + shrd_events->events[idx] = NULL; + } + rc = (int)atomic_read(&shrd_events->refcount[idx]); + break; + } + } + + mutex_unlock(&shrd_events->list_mutex); + return rc; +} +#endif + static void _free_event(struct perf_event *event) { irq_work_sync(&event->pending); @@ -4690,6 +4753,18 @@ int perf_event_release_kernel(struct perf_event *event) WARN_ON_ONCE(ctx->parent_ctx); perf_remove_from_context(event, DETACH_GROUP); +#ifdef CONFIG_PERF_KERNEL_SHARE + if (perf_event_delete_kernel_shared(event) > 0) { + perf_event__state_init(event); + perf_install_in_context(ctx, event, event->cpu); + + perf_event_ctx_unlock(event, ctx); + + perf_event_enable(event); + + return 0; + } +#endif raw_spin_lock_irq(&ctx->lock); /* * Mark this event as STATE_DEAD, there is no external reference to it @@ -10422,6 +10497,124 @@ enabled: account_pmu_sb_event(event); } +#ifdef CONFIG_PERF_KERNEL_SHARE +static struct perf_event * +perf_event_create_kernel_shared_check(struct perf_event_attr *attr, int cpu, + struct task_struct *task, + perf_overflow_handler_t overflow_handler, + struct perf_event *group_leader) +{ + unsigned long idx; + struct perf_event *event; + struct shared_events_str *shrd_events; + + /* + * Have to be per cpu events for sharing + */ + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return NULL; + + /* + * Can't handle these type requests for sharing right now. + */ + if (task || overflow_handler || attr->sample_period || + (attr->type != PERF_TYPE_HARDWARE && + attr->type != PERF_TYPE_RAW)) { + return NULL; + } + + /* + * Using per_cpu_ptr (or could do cross cpu call which is what most of + * perf does to access per cpu data structures + */ + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + event = NULL; + for_each_set_bit(idx, shrd_events->used_mask, SHARED_EVENTS_MAX) { + /* Do the comparisons field by field on the attr structure. + * This is because the user-space and kernel-space might + * be using different versions of perf. As a result, + * the fields' position in the memory and the size might not + * be the same. Hence memcmp() is not the best way to + * compare. + */ + if (attr->type == shrd_events->attr[idx].type && + attr->config == shrd_events->attr[idx].config) { + + event = shrd_events->events[idx]; + + /* Do not change the group for this shared event */ + if (group_leader && event->group_leader != event) { + event = NULL; + continue; + } + + event->shared = true; + atomic_inc(&shrd_events->refcount[idx]); + break; + } + } + mutex_unlock(&shrd_events->list_mutex); + + return event; +} + +static void +perf_event_create_kernel_shared_add(struct perf_event_attr *attr, int cpu, + struct task_struct *task, + perf_overflow_handler_t overflow_handler, + void *context, + struct perf_event *event) +{ + unsigned long idx; + struct shared_events_str *shrd_events; + + /* + * Have to be per cpu events for sharing + */ + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return; + + /* + * Can't handle these type requests for sharing right now. + */ + if (overflow_handler || attr->sample_period || + (attr->type != PERF_TYPE_HARDWARE && + attr->type != PERF_TYPE_RAW)) { + return; + } + + /* + * Using per_cpu_ptr (or could do cross cpu call which is what most of + * perf does to access per cpu data structures + */ + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + /* + * If we are in this routine, we know that this event isn't already in + * the shared list. Check if slot available in shared list + */ + idx = find_first_zero_bit(shrd_events->used_mask, SHARED_EVENTS_MAX); + + if (idx >= SHARED_EVENTS_MAX) + goto out; + + /* + * The event isn't in the list and there is an empty slot so add it. + */ + shrd_events->attr[idx] = *attr; + shrd_events->events[idx] = event; + set_bit(idx, shrd_events->used_mask); + atomic_set(&shrd_events->refcount[idx], 1); +out: + mutex_unlock(&shrd_events->list_mutex); +} +#endif + /* * Allocate and initialize an event structure */ @@ -10908,6 +11101,31 @@ again: return gctx; } +#ifdef CONFIG_PERF_USER_SHARE +static void perf_group_shared_event(struct perf_event *event, + struct perf_event *group_leader) +{ + if (!event->shared || !group_leader) + return; + + /* Do not attempt to change the group for this shared event */ + if (event->group_leader != event) + return; + + /* + * Single events have the group leaders as themselves. + * As we now have a new group to attach to, remove from + * the previous group and attach it to the new group. + */ + perf_remove_from_context(event, DETACH_GROUP); + + event->group_leader = group_leader; + perf_event__state_init(event); + + perf_install_in_context(group_leader->ctx, event, event->cpu); +} +#endif + /** * sys_perf_event_open - open a performance event, associate it to a task/cpu * @@ -10921,7 +11139,7 @@ SYSCALL_DEFINE5(perf_event_open, pid_t, pid, int, cpu, int, group_fd, unsigned long, flags) { struct perf_event *group_leader = NULL, *output_event = NULL; - struct perf_event *event, *sibling; + struct perf_event *event = NULL, *sibling; struct perf_event_attr attr; struct perf_event_context *ctx, *uninitialized_var(gctx); struct file *event_file = NULL; @@ -11042,11 +11260,17 @@ SYSCALL_DEFINE5(perf_event_open, if (flags & PERF_FLAG_PID_CGROUP) cgroup_fd = pid; - event = perf_event_alloc(&attr, cpu, task, group_leader, NULL, - NULL, NULL, cgroup_fd); - if (IS_ERR(event)) { - err = PTR_ERR(event); - goto err_cred; +#ifdef CONFIG_PERF_USER_SHARE + event = perf_event_create_kernel_shared_check(&attr, cpu, task, NULL, + group_leader); +#endif + if (!event) { + event = perf_event_alloc(&attr, cpu, task, group_leader, NULL, + NULL, NULL, cgroup_fd); + if (IS_ERR(event)) { + err = PTR_ERR(event); + goto err_cred; + } } if (is_sampling_event(event)) { @@ -11242,7 +11466,11 @@ SYSCALL_DEFINE5(perf_event_open, * Must be under the same ctx::mutex as perf_install_in_context(), * because we need to serialize with concurrent event creation. */ - if (!exclusive_event_installable(event, ctx)) { + if (!exclusive_event_installable(event, ctx) +#ifdef CONFIG_PERF_KERNEL_SHARE + && (!event->shared) +#endif + ) { err = -EBUSY; goto err_locked; } @@ -11308,10 +11536,20 @@ SYSCALL_DEFINE5(perf_event_open, perf_event__header_size(event); perf_event__id_header_size(event); - event->owner = current; +#ifdef CONFIG_PERF_USER_SHARE + if (event->shared && group_leader) + perf_group_shared_event(event, group_leader); +#endif +#ifdef CONFIG_PERF_KERNEL_SHARE + if (!event->shared) { +#endif + event->owner = current; - perf_install_in_context(ctx, event, event->cpu); - perf_unpin_context(ctx); + perf_install_in_context(ctx, event, event->cpu); + perf_unpin_context(ctx); +#ifdef CONFIG_PERF_KERNEL_SHARE + } +#endif if (move_group) perf_event_ctx_unlock(group_leader, gctx); @@ -11322,9 +11560,15 @@ SYSCALL_DEFINE5(perf_event_open, put_task_struct(task); } - mutex_lock(¤t->perf_event_mutex); - list_add_tail(&event->owner_entry, ¤t->perf_event_list); - mutex_unlock(¤t->perf_event_mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + if (!event->shared) { +#endif + mutex_lock(¤t->perf_event_mutex); + list_add_tail(&event->owner_entry, ¤t->perf_event_list); + mutex_unlock(¤t->perf_event_mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + } +#endif /* * Drop the reference on the group_event after placing the @@ -11334,6 +11578,14 @@ SYSCALL_DEFINE5(perf_event_open, */ fdput(group); fd_install(event_fd, event_file); + +#ifdef CONFIG_PERF_USER_SHARE + /* Add the event to the shared events list */ + if (!event->shared) + perf_event_create_kernel_shared_add(&attr, cpu, + task, NULL, ctx, event); +#endif + return event_fd; err_locked: @@ -11365,6 +11617,7 @@ err_fd: return err; } + /** * perf_event_create_kernel_counter * @@ -11379,7 +11632,7 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, void *context) { struct perf_event_context *ctx; - struct perf_event *event; + struct perf_event *event = NULL; int err; /* @@ -11389,16 +11642,27 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, if (attr->aux_output) return ERR_PTR(-EINVAL); - event = perf_event_alloc(attr, cpu, task, NULL, NULL, - overflow_handler, context, -1); - if (IS_ERR(event)) { - err = PTR_ERR(event); - goto err; +#ifdef CONFIG_PERF_KERNEL_SHARE + event = perf_event_create_kernel_shared_check(attr, cpu, task, + overflow_handler, NULL); +#endif + if (!event) { + event = perf_event_alloc(attr, cpu, task, NULL, NULL, + overflow_handler, context, -1); + if (IS_ERR(event)) { + err = PTR_ERR(event); + goto err; + } } /* Mark owner so we could distinguish it from user events. */ event->owner = TASK_TOMBSTONE; +#ifdef CONFIG_PERF_KERNEL_SHARE + if (event->shared) + return event; +#endif + /* * Get the target context (task or percpu): */ @@ -11439,6 +11703,13 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, perf_unpin_context(ctx); mutex_unlock(&ctx->mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + /* + * Check if can add event to shared list + */ + perf_event_create_kernel_shared_add(attr, cpu, + task, overflow_handler, context, event); +#endif return event; err_unlock: @@ -12260,9 +12531,25 @@ static struct notifier_block perf_reboot_notifier = { void __init perf_event_init(void) { int ret; +#ifdef CONFIG_PERF_KERNEL_SHARE + int cpu; +#endif idr_init(&pmu_idr); +#ifdef CONFIG_PERF_KERNEL_SHARE + shared_events = alloc_percpu(struct shared_events_str); + if (!shared_events) { + WARN(1, "alloc_percpu failed for shared_events struct"); + } else { + for_each_possible_cpu(cpu) { + struct shared_events_str *shrd_events = + per_cpu_ptr(shared_events, cpu); + + mutex_init(&shrd_events->list_mutex); + } + } +#endif perf_event_init_all_cpus(); init_srcu_struct(&pmus_srcu); perf_pmu_register(&perf_swevent, "software", PERF_TYPE_SOFTWARE);