diff --git a/include/linux/perf_event.h b/include/linux/perf_event.h index 48685d6ca501..27542442c0d0 100644 --- a/include/linux/perf_event.h +++ b/include/linux/perf_event.h @@ -726,6 +726,10 @@ struct perf_event { void *security; #endif struct list_head sb_list; +#ifdef CONFIG_PERF_KERNEL_SHARE + /* Is this event shared with other events */ + bool shared; +#endif #endif /* CONFIG_PERF_EVENTS */ }; diff --git a/init/Kconfig b/init/Kconfig index df4180822716..12c402302a55 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1731,6 +1731,26 @@ config PERF_EVENTS Say Y if unsure. +config PERF_KERNEL_SHARE + bool "Perf event sharing with kernel-space" + depends on QGKI + help + Say yes here to enable the kernel-space sharing of events. The events + can be shared among other kernel-space events or with kernel created + events that has the same config and type event attributes. + + Say N if unsure. + +config PERF_USER_SHARE + bool "Perf event sharing with user-space" + depends on PERF_KERNEL_SHARE + help + Say yes here to enable the user-space sharing of events. The events + can be shared among other user-space events or with kernel created + events that has the same config and type event attributes. + + Say N if unsure. + config DEBUG_PERF_USE_VMALLOC default n bool "Debug: use vmalloc to back perf mmap() buffers" diff --git a/kernel/events/core.c b/kernel/events/core.c index 4ebcaa758f6f..0ac39045d040 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -361,6 +361,34 @@ enum event_type_t { EVENT_ALL = EVENT_FLEXIBLE | EVENT_PINNED, }; +#ifdef CONFIG_PERF_KERNEL_SHARE +/* The shared events struct. */ +#define SHARED_EVENTS_MAX 7 + +struct shared_events_str { + /* + * Mutex to serialize access to shared list. Needed for the + * read/modify/write sequences. + */ + struct mutex list_mutex; + + /* + * A 1 bit for an index indicates that the slot is being used for + * an event. A 0 means that the slot can be used. + */ + DECLARE_BITMAP(used_mask, SHARED_EVENTS_MAX); + + /* + * The kernel events that are shared for a cpu; + */ + struct perf_event *events[SHARED_EVENTS_MAX]; + struct perf_event_attr attr[SHARED_EVENTS_MAX]; + atomic_t refcount[SHARED_EVENTS_MAX]; +}; + +static struct shared_events_str __percpu *shared_events; +#endif + /* * perf_sched_events : >0 events exist * perf_cgroup_events: >0 per-cpu cgroup events exist on this cpu @@ -1991,6 +2019,10 @@ static void perf_group_detach(struct perf_event *event) if (event->group_leader != event) { list_del_init(&event->sibling_list); event->group_leader->nr_siblings--; +#ifdef CONFIG_PERF_KERNEL_SHARE + if (event->shared) + event->group_leader = event; +#endif goto out; } @@ -4532,6 +4564,37 @@ static bool exclusive_event_installable(struct perf_event *event, static void perf_addr_filters_splice(struct perf_event *event, struct list_head *head); +#ifdef CONFIG_PERF_KERNEL_SHARE +static int +perf_event_delete_kernel_shared(struct perf_event *event) +{ + int rc = -1, cpu = event->cpu; + struct shared_events_str *shrd_events; + unsigned long idx; + + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return 0; + + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + for_each_set_bit(idx, shrd_events->used_mask, SHARED_EVENTS_MAX) { + if (shrd_events->events[idx] == event) { + if (atomic_dec_and_test(&shrd_events->refcount[idx])) { + clear_bit(idx, shrd_events->used_mask); + shrd_events->events[idx] = NULL; + } + rc = (int)atomic_read(&shrd_events->refcount[idx]); + break; + } + } + + mutex_unlock(&shrd_events->list_mutex); + return rc; +} +#endif + static void _free_event(struct perf_event *event) { irq_work_sync(&event->pending); @@ -4690,6 +4753,18 @@ int perf_event_release_kernel(struct perf_event *event) WARN_ON_ONCE(ctx->parent_ctx); perf_remove_from_context(event, DETACH_GROUP); +#ifdef CONFIG_PERF_KERNEL_SHARE + if (perf_event_delete_kernel_shared(event) > 0) { + perf_event__state_init(event); + perf_install_in_context(ctx, event, event->cpu); + + perf_event_ctx_unlock(event, ctx); + + perf_event_enable(event); + + return 0; + } +#endif raw_spin_lock_irq(&ctx->lock); /* * Mark this event as STATE_DEAD, there is no external reference to it @@ -10422,6 +10497,124 @@ enabled: account_pmu_sb_event(event); } +#ifdef CONFIG_PERF_KERNEL_SHARE +static struct perf_event * +perf_event_create_kernel_shared_check(struct perf_event_attr *attr, int cpu, + struct task_struct *task, + perf_overflow_handler_t overflow_handler, + struct perf_event *group_leader) +{ + unsigned long idx; + struct perf_event *event; + struct shared_events_str *shrd_events; + + /* + * Have to be per cpu events for sharing + */ + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return NULL; + + /* + * Can't handle these type requests for sharing right now. + */ + if (task || overflow_handler || attr->sample_period || + (attr->type != PERF_TYPE_HARDWARE && + attr->type != PERF_TYPE_RAW)) { + return NULL; + } + + /* + * Using per_cpu_ptr (or could do cross cpu call which is what most of + * perf does to access per cpu data structures + */ + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + event = NULL; + for_each_set_bit(idx, shrd_events->used_mask, SHARED_EVENTS_MAX) { + /* Do the comparisons field by field on the attr structure. + * This is because the user-space and kernel-space might + * be using different versions of perf. As a result, + * the fields' position in the memory and the size might not + * be the same. Hence memcmp() is not the best way to + * compare. + */ + if (attr->type == shrd_events->attr[idx].type && + attr->config == shrd_events->attr[idx].config) { + + event = shrd_events->events[idx]; + + /* Do not change the group for this shared event */ + if (group_leader && event->group_leader != event) { + event = NULL; + continue; + } + + event->shared = true; + atomic_inc(&shrd_events->refcount[idx]); + break; + } + } + mutex_unlock(&shrd_events->list_mutex); + + return event; +} + +static void +perf_event_create_kernel_shared_add(struct perf_event_attr *attr, int cpu, + struct task_struct *task, + perf_overflow_handler_t overflow_handler, + void *context, + struct perf_event *event) +{ + unsigned long idx; + struct shared_events_str *shrd_events; + + /* + * Have to be per cpu events for sharing + */ + if (!shared_events || (u32)cpu >= nr_cpu_ids) + return; + + /* + * Can't handle these type requests for sharing right now. + */ + if (overflow_handler || attr->sample_period || + (attr->type != PERF_TYPE_HARDWARE && + attr->type != PERF_TYPE_RAW)) { + return; + } + + /* + * Using per_cpu_ptr (or could do cross cpu call which is what most of + * perf does to access per cpu data structures + */ + shrd_events = per_cpu_ptr(shared_events, cpu); + + mutex_lock(&shrd_events->list_mutex); + + /* + * If we are in this routine, we know that this event isn't already in + * the shared list. Check if slot available in shared list + */ + idx = find_first_zero_bit(shrd_events->used_mask, SHARED_EVENTS_MAX); + + if (idx >= SHARED_EVENTS_MAX) + goto out; + + /* + * The event isn't in the list and there is an empty slot so add it. + */ + shrd_events->attr[idx] = *attr; + shrd_events->events[idx] = event; + set_bit(idx, shrd_events->used_mask); + atomic_set(&shrd_events->refcount[idx], 1); +out: + mutex_unlock(&shrd_events->list_mutex); +} +#endif + /* * Allocate and initialize an event structure */ @@ -10908,6 +11101,31 @@ again: return gctx; } +#ifdef CONFIG_PERF_USER_SHARE +static void perf_group_shared_event(struct perf_event *event, + struct perf_event *group_leader) +{ + if (!event->shared || !group_leader) + return; + + /* Do not attempt to change the group for this shared event */ + if (event->group_leader != event) + return; + + /* + * Single events have the group leaders as themselves. + * As we now have a new group to attach to, remove from + * the previous group and attach it to the new group. + */ + perf_remove_from_context(event, DETACH_GROUP); + + event->group_leader = group_leader; + perf_event__state_init(event); + + perf_install_in_context(group_leader->ctx, event, event->cpu); +} +#endif + /** * sys_perf_event_open - open a performance event, associate it to a task/cpu * @@ -10921,7 +11139,7 @@ SYSCALL_DEFINE5(perf_event_open, pid_t, pid, int, cpu, int, group_fd, unsigned long, flags) { struct perf_event *group_leader = NULL, *output_event = NULL; - struct perf_event *event, *sibling; + struct perf_event *event = NULL, *sibling; struct perf_event_attr attr; struct perf_event_context *ctx, *uninitialized_var(gctx); struct file *event_file = NULL; @@ -11042,11 +11260,17 @@ SYSCALL_DEFINE5(perf_event_open, if (flags & PERF_FLAG_PID_CGROUP) cgroup_fd = pid; - event = perf_event_alloc(&attr, cpu, task, group_leader, NULL, - NULL, NULL, cgroup_fd); - if (IS_ERR(event)) { - err = PTR_ERR(event); - goto err_cred; +#ifdef CONFIG_PERF_USER_SHARE + event = perf_event_create_kernel_shared_check(&attr, cpu, task, NULL, + group_leader); +#endif + if (!event) { + event = perf_event_alloc(&attr, cpu, task, group_leader, NULL, + NULL, NULL, cgroup_fd); + if (IS_ERR(event)) { + err = PTR_ERR(event); + goto err_cred; + } } if (is_sampling_event(event)) { @@ -11242,7 +11466,11 @@ SYSCALL_DEFINE5(perf_event_open, * Must be under the same ctx::mutex as perf_install_in_context(), * because we need to serialize with concurrent event creation. */ - if (!exclusive_event_installable(event, ctx)) { + if (!exclusive_event_installable(event, ctx) +#ifdef CONFIG_PERF_KERNEL_SHARE + && (!event->shared) +#endif + ) { err = -EBUSY; goto err_locked; } @@ -11308,10 +11536,20 @@ SYSCALL_DEFINE5(perf_event_open, perf_event__header_size(event); perf_event__id_header_size(event); - event->owner = current; +#ifdef CONFIG_PERF_USER_SHARE + if (event->shared && group_leader) + perf_group_shared_event(event, group_leader); +#endif +#ifdef CONFIG_PERF_KERNEL_SHARE + if (!event->shared) { +#endif + event->owner = current; - perf_install_in_context(ctx, event, event->cpu); - perf_unpin_context(ctx); + perf_install_in_context(ctx, event, event->cpu); + perf_unpin_context(ctx); +#ifdef CONFIG_PERF_KERNEL_SHARE + } +#endif if (move_group) perf_event_ctx_unlock(group_leader, gctx); @@ -11322,9 +11560,15 @@ SYSCALL_DEFINE5(perf_event_open, put_task_struct(task); } - mutex_lock(¤t->perf_event_mutex); - list_add_tail(&event->owner_entry, ¤t->perf_event_list); - mutex_unlock(¤t->perf_event_mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + if (!event->shared) { +#endif + mutex_lock(¤t->perf_event_mutex); + list_add_tail(&event->owner_entry, ¤t->perf_event_list); + mutex_unlock(¤t->perf_event_mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + } +#endif /* * Drop the reference on the group_event after placing the @@ -11334,6 +11578,14 @@ SYSCALL_DEFINE5(perf_event_open, */ fdput(group); fd_install(event_fd, event_file); + +#ifdef CONFIG_PERF_USER_SHARE + /* Add the event to the shared events list */ + if (!event->shared) + perf_event_create_kernel_shared_add(&attr, cpu, + task, NULL, ctx, event); +#endif + return event_fd; err_locked: @@ -11365,6 +11617,7 @@ err_fd: return err; } + /** * perf_event_create_kernel_counter * @@ -11379,7 +11632,7 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, void *context) { struct perf_event_context *ctx; - struct perf_event *event; + struct perf_event *event = NULL; int err; /* @@ -11389,16 +11642,27 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, if (attr->aux_output) return ERR_PTR(-EINVAL); - event = perf_event_alloc(attr, cpu, task, NULL, NULL, - overflow_handler, context, -1); - if (IS_ERR(event)) { - err = PTR_ERR(event); - goto err; +#ifdef CONFIG_PERF_KERNEL_SHARE + event = perf_event_create_kernel_shared_check(attr, cpu, task, + overflow_handler, NULL); +#endif + if (!event) { + event = perf_event_alloc(attr, cpu, task, NULL, NULL, + overflow_handler, context, -1); + if (IS_ERR(event)) { + err = PTR_ERR(event); + goto err; + } } /* Mark owner so we could distinguish it from user events. */ event->owner = TASK_TOMBSTONE; +#ifdef CONFIG_PERF_KERNEL_SHARE + if (event->shared) + return event; +#endif + /* * Get the target context (task or percpu): */ @@ -11439,6 +11703,13 @@ perf_event_create_kernel_counter(struct perf_event_attr *attr, int cpu, perf_unpin_context(ctx); mutex_unlock(&ctx->mutex); +#ifdef CONFIG_PERF_KERNEL_SHARE + /* + * Check if can add event to shared list + */ + perf_event_create_kernel_shared_add(attr, cpu, + task, overflow_handler, context, event); +#endif return event; err_unlock: @@ -12260,9 +12531,25 @@ static struct notifier_block perf_reboot_notifier = { void __init perf_event_init(void) { int ret; +#ifdef CONFIG_PERF_KERNEL_SHARE + int cpu; +#endif idr_init(&pmu_idr); +#ifdef CONFIG_PERF_KERNEL_SHARE + shared_events = alloc_percpu(struct shared_events_str); + if (!shared_events) { + WARN(1, "alloc_percpu failed for shared_events struct"); + } else { + for_each_possible_cpu(cpu) { + struct shared_events_str *shrd_events = + per_cpu_ptr(shared_events, cpu); + + mutex_init(&shrd_events->list_mutex); + } + } +#endif perf_event_init_all_cpus(); init_srcu_struct(&pmus_srcu); perf_pmu_register(&perf_swevent, "software", PERF_TYPE_SOFTWARE);