mirror of
https://github.com/BobTheBlinker/android_kernel_motorola_sm6375.git
synced 2026-10-05 19:31:57 -04:00
Merge remote-tracking branch 'sm8350/lineage-20' into lineage-23.2
* sm8350/lineage-20: UPSTREAM: tty: allow TIOCSLCKTRMIOS with CAP_CHECKPOINT_RESTORE UPSTREAM: selftests: add clone3() CAP_CHECKPOINT_RESTORE test UPSTREAM: prctl: exe link permission error changed from -EINVAL to -EPERM UPSTREAM: prctl: Allow local CAP_CHECKPOINT_RESTORE to change /proc/self/exe UPSTREAM: proc: allow access in init userns for map_files with CAP_CHECKPOINT_RESTORE UPSTREAM: pid_namespace: use checkpoint_restore_ns_capable() for ns_last_pid UPSTREAM: pid: use checkpoint_restore_ns_capable() for set_tid UPSTREAM: capabilities: Introduce CAP_CHECKPOINT_RESTORE UPSTREAM: selftests: add tests for clone3() with *set_tid UPSTREAM: fork: extend clone3() to support setting a PID UPSTREAM: selftests: add tests for clone3() UPSTREAM: tests: test CLONE_CLEAR_SIGHAND UPSTREAM: clone3: add CLONE_CLEAR_SIGHAND Change-Id: I7890fb550e85b9fd4c0c3a27144d5bfd8963d496
This commit is contained in:
commit
5048089773
22 changed files with 1139 additions and 61 deletions
|
|
@ -12851,6 +12851,7 @@ S: Maintained
|
|||
T: git git://git.kernel.org/pub/scm/linux/kernel/git/brauner/linux.git
|
||||
F: samples/pidfd/
|
||||
F: tools/testing/selftests/pidfd/
|
||||
F: tools/testing/selftests/clone3/
|
||||
K: (?i)pidfd
|
||||
K: (?i)clone3
|
||||
K: \b(clone_args|kernel_clone_args)\b
|
||||
|
|
|
|||
|
|
@ -807,7 +807,7 @@ int tty_mode_ioctl(struct tty_struct *tty, struct file *file,
|
|||
ret = -EFAULT;
|
||||
return ret;
|
||||
case TIOCSLCKTRMIOS:
|
||||
if (!capable(CAP_SYS_ADMIN))
|
||||
if (!checkpoint_restore_ns_capable(&init_user_ns))
|
||||
return -EPERM;
|
||||
copy_termios_locked(real_tty, &kterm);
|
||||
if (user_termios_to_kernel_termios(&kterm,
|
||||
|
|
@ -824,7 +824,7 @@ int tty_mode_ioctl(struct tty_struct *tty, struct file *file,
|
|||
ret = -EFAULT;
|
||||
return ret;
|
||||
case TIOCSLCKTRMIOS:
|
||||
if (!capable(CAP_SYS_ADMIN))
|
||||
if (!checkpoint_restore_ns_capable(&init_user_ns))
|
||||
return -EPERM;
|
||||
copy_termios_locked(real_tty, &kterm);
|
||||
if (user_termios_to_kernel_termios_1(&kterm,
|
||||
|
|
|
|||
|
|
@ -2211,16 +2211,16 @@ struct map_files_info {
|
|||
};
|
||||
|
||||
/*
|
||||
* Only allow CAP_SYS_ADMIN to follow the links, due to concerns about how the
|
||||
* symlinks may be used to bypass permissions on ancestor directories in the
|
||||
* path to the file in question.
|
||||
* Only allow CAP_SYS_ADMIN and CAP_CHECKPOINT_RESTORE to follow the links, due
|
||||
* to concerns about how the symlinks may be used to bypass permissions on
|
||||
* ancestor directories in the path to the file in question.
|
||||
*/
|
||||
static const char *
|
||||
proc_map_files_get_link(struct dentry *dentry,
|
||||
struct inode *inode,
|
||||
struct delayed_call *done)
|
||||
{
|
||||
if (!capable(CAP_SYS_ADMIN))
|
||||
if (!checkpoint_restore_ns_capable(&init_user_ns))
|
||||
return ERR_PTR(-EPERM);
|
||||
|
||||
return proc_pid_get_link(dentry, inode, done);
|
||||
|
|
|
|||
|
|
@ -261,6 +261,12 @@ static inline bool bpf_capable(void)
|
|||
return capable(CAP_BPF) || capable(CAP_SYS_ADMIN);
|
||||
}
|
||||
|
||||
static inline bool checkpoint_restore_ns_capable(struct user_namespace *ns)
|
||||
{
|
||||
return ns_capable(ns, CAP_CHECKPOINT_RESTORE) ||
|
||||
ns_capable(ns, CAP_SYS_ADMIN);
|
||||
}
|
||||
|
||||
/* audit system wants to get cap info from files as well */
|
||||
extern int get_vfs_caps_from_disk(const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps);
|
||||
|
||||
|
|
|
|||
|
|
@ -126,7 +126,8 @@ extern struct pid *find_vpid(int nr);
|
|||
extern struct pid *find_get_pid(int nr);
|
||||
extern struct pid *find_ge_pid(int nr, struct pid_namespace *);
|
||||
|
||||
extern struct pid *alloc_pid(struct pid_namespace *ns);
|
||||
extern struct pid *alloc_pid(struct pid_namespace *ns, pid_t *set_tid,
|
||||
size_t set_tid_size);
|
||||
extern void free_pid(struct pid *pid);
|
||||
extern void disable_pid_allocation(struct pid_namespace *ns);
|
||||
|
||||
|
|
|
|||
|
|
@ -12,6 +12,8 @@
|
|||
#include <linux/ns_common.h>
|
||||
#include <linux/idr.h>
|
||||
|
||||
/* MAX_PID_NS_LEVEL is needed for limiting size of 'struct pid' */
|
||||
#define MAX_PID_NS_LEVEL 32
|
||||
|
||||
struct fs_pin;
|
||||
|
||||
|
|
|
|||
|
|
@ -26,6 +26,9 @@ struct kernel_clone_args {
|
|||
unsigned long stack;
|
||||
unsigned long stack_size;
|
||||
unsigned long tls;
|
||||
pid_t *set_tid;
|
||||
/* Number of elements in *set_tid */
|
||||
size_t set_tid_size;
|
||||
};
|
||||
|
||||
/*
|
||||
|
|
|
|||
|
|
@ -405,7 +405,14 @@ struct vfs_ns_cap_data {
|
|||
*/
|
||||
#define CAP_BPF 39
|
||||
|
||||
#define CAP_LAST_CAP CAP_BPF
|
||||
|
||||
/* Allow checkpoint/restore related operations */
|
||||
/* Allow PID selection during clone3() */
|
||||
/* Allow writing to ns_last_pid */
|
||||
|
||||
#define CAP_CHECKPOINT_RESTORE 40
|
||||
|
||||
#define CAP_LAST_CAP CAP_CHECKPOINT_RESTORE
|
||||
|
||||
#define cap_valid(x) ((x) >= 0 && (x) <= CAP_LAST_CAP)
|
||||
|
||||
|
|
|
|||
|
|
@ -33,31 +33,48 @@
|
|||
#define CLONE_NEWNET 0x40000000 /* New network namespace */
|
||||
#define CLONE_IO 0x80000000 /* Clone io context */
|
||||
|
||||
/* Flags for the clone3() syscall. */
|
||||
#define CLONE_CLEAR_SIGHAND 0x100000000ULL /* Clear any signal handler and reset to SIG_DFL. */
|
||||
|
||||
#ifndef __ASSEMBLY__
|
||||
/**
|
||||
* struct clone_args - arguments for the clone3 syscall
|
||||
* @flags: Flags for the new process as listed above.
|
||||
* All flags are valid except for CSIGNAL and
|
||||
* CLONE_DETACHED.
|
||||
* @pidfd: If CLONE_PIDFD is set, a pidfd will be
|
||||
* returned in this argument.
|
||||
* @child_tid: If CLONE_CHILD_SETTID is set, the TID of the
|
||||
* child process will be returned in the child's
|
||||
* memory.
|
||||
* @parent_tid: If CLONE_PARENT_SETTID is set, the TID of
|
||||
* the child process will be returned in the
|
||||
* parent's memory.
|
||||
* @exit_signal: The exit_signal the parent process will be
|
||||
* sent when the child exits.
|
||||
* @stack: Specify the location of the stack for the
|
||||
* child process.
|
||||
* Note, @stack is expected to point to the
|
||||
* lowest address. The stack direction will be
|
||||
* determined by the kernel and set up
|
||||
* appropriately based on @stack_size.
|
||||
* @stack_size: The size of the stack for the child process.
|
||||
* @tls: If CLONE_SETTLS is set, the tls descriptor
|
||||
* is set to tls.
|
||||
* @flags: Flags for the new process as listed above.
|
||||
* All flags are valid except for CSIGNAL and
|
||||
* CLONE_DETACHED.
|
||||
* @pidfd: If CLONE_PIDFD is set, a pidfd will be
|
||||
* returned in this argument.
|
||||
* @child_tid: If CLONE_CHILD_SETTID is set, the TID of the
|
||||
* child process will be returned in the child's
|
||||
* memory.
|
||||
* @parent_tid: If CLONE_PARENT_SETTID is set, the TID of
|
||||
* the child process will be returned in the
|
||||
* parent's memory.
|
||||
* @exit_signal: The exit_signal the parent process will be
|
||||
* sent when the child exits.
|
||||
* @stack: Specify the location of the stack for the
|
||||
* child process.
|
||||
* Note, @stack is expected to point to the
|
||||
* lowest address. The stack direction will be
|
||||
* determined by the kernel and set up
|
||||
* appropriately based on @stack_size.
|
||||
* @stack_size: The size of the stack for the child process.
|
||||
* @tls: If CLONE_SETTLS is set, the tls descriptor
|
||||
* is set to tls.
|
||||
* @set_tid: Pointer to an array of type *pid_t. The size
|
||||
* of the array is defined using @set_tid_size.
|
||||
* This array is used to select PIDs/TIDs for
|
||||
* newly created processes. The first element in
|
||||
* this defines the PID in the most nested PID
|
||||
* namespace. Each additional element in the array
|
||||
* defines the PID in the parent PID namespace of
|
||||
* the original PID namespace. If the array has
|
||||
* less entries than the number of currently
|
||||
* nested PID namespaces only the PIDs in the
|
||||
* corresponding namespaces are set.
|
||||
* @set_tid_size: This defines the size of the array referenced
|
||||
* in @set_tid. This cannot be larger than the
|
||||
* kernel's limit of nested PID namespaces.
|
||||
*
|
||||
* The structure is versioned by size and thus extensible.
|
||||
* New struct members must go at the end of the struct and
|
||||
|
|
@ -72,10 +89,13 @@ struct clone_args {
|
|||
__aligned_u64 stack;
|
||||
__aligned_u64 stack_size;
|
||||
__aligned_u64 tls;
|
||||
__aligned_u64 set_tid;
|
||||
__aligned_u64 set_tid_size;
|
||||
};
|
||||
#endif
|
||||
|
||||
#define CLONE_ARGS_SIZE_VER0 64 /* sizeof first published struct */
|
||||
#define CLONE_ARGS_SIZE_VER1 80 /* sizeof second published struct */
|
||||
|
||||
/*
|
||||
* Scheduling policies
|
||||
|
|
|
|||
|
|
@ -1540,6 +1540,11 @@ static int copy_sighand(unsigned long clone_flags, struct task_struct *tsk)
|
|||
spin_lock_irq(¤t->sighand->siglock);
|
||||
memcpy(sig->action, current->sighand->action, sizeof(sig->action));
|
||||
spin_unlock_irq(¤t->sighand->siglock);
|
||||
|
||||
/* Reset all signal handler not set to SIG_IGN to SIG_DFL. */
|
||||
if (clone_flags & CLONE_CLEAR_SIGHAND)
|
||||
flush_signal_handlers(tsk, 0);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
|
@ -2161,7 +2166,8 @@ static __latent_entropy struct task_struct *copy_process(
|
|||
stackleak_task_init(p);
|
||||
|
||||
if (pid != &init_struct_pid) {
|
||||
pid = alloc_pid(p->nsproxy->pid_ns_for_children);
|
||||
pid = alloc_pid(p->nsproxy->pid_ns_for_children, args->set_tid,
|
||||
args->set_tid_size);
|
||||
if (IS_ERR(pid)) {
|
||||
retval = PTR_ERR(pid);
|
||||
goto bad_fork_cleanup_thread;
|
||||
|
|
@ -2659,6 +2665,7 @@ noinline static int copy_clone_args_from_user(struct kernel_clone_args *kargs,
|
|||
{
|
||||
int err;
|
||||
struct clone_args args;
|
||||
pid_t *kset_tid = kargs->set_tid;
|
||||
|
||||
if (unlikely(usize > PAGE_SIZE))
|
||||
return -E2BIG;
|
||||
|
|
@ -2669,6 +2676,15 @@ noinline static int copy_clone_args_from_user(struct kernel_clone_args *kargs,
|
|||
if (err)
|
||||
return err;
|
||||
|
||||
if (unlikely(args.set_tid_size > MAX_PID_NS_LEVEL))
|
||||
return -EINVAL;
|
||||
|
||||
if (unlikely(!args.set_tid && args.set_tid_size > 0))
|
||||
return -EINVAL;
|
||||
|
||||
if (unlikely(args.set_tid && args.set_tid_size == 0))
|
||||
return -EINVAL;
|
||||
|
||||
/*
|
||||
* Verify that higher 32bits of exit_signal are unset and that
|
||||
* it is a valid signal
|
||||
|
|
@ -2686,8 +2702,16 @@ noinline static int copy_clone_args_from_user(struct kernel_clone_args *kargs,
|
|||
.stack = args.stack,
|
||||
.stack_size = args.stack_size,
|
||||
.tls = args.tls,
|
||||
.set_tid_size = args.set_tid_size,
|
||||
};
|
||||
|
||||
if (args.set_tid &&
|
||||
copy_from_user(kset_tid, u64_to_user_ptr(args.set_tid),
|
||||
(kargs->set_tid_size * sizeof(pid_t))))
|
||||
return -EFAULT;
|
||||
|
||||
kargs->set_tid = kset_tid;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
|
@ -2721,11 +2745,8 @@ static inline bool clone3_stack_valid(struct kernel_clone_args *kargs)
|
|||
|
||||
static bool clone3_args_valid(struct kernel_clone_args *kargs)
|
||||
{
|
||||
/*
|
||||
* All lower bits of the flag word are taken.
|
||||
* Verify that no other unknown flags are passed along.
|
||||
*/
|
||||
if (kargs->flags & ~CLONE_LEGACY_FLAGS)
|
||||
/* Verify that no unknown flags are passed along. */
|
||||
if (kargs->flags & ~(CLONE_LEGACY_FLAGS | CLONE_CLEAR_SIGHAND))
|
||||
return false;
|
||||
|
||||
/*
|
||||
|
|
@ -2735,6 +2756,10 @@ static bool clone3_args_valid(struct kernel_clone_args *kargs)
|
|||
if (kargs->flags & (CLONE_DETACHED | CSIGNAL))
|
||||
return false;
|
||||
|
||||
if ((kargs->flags & (CLONE_SIGHAND | CLONE_CLEAR_SIGHAND)) ==
|
||||
(CLONE_SIGHAND | CLONE_CLEAR_SIGHAND))
|
||||
return false;
|
||||
|
||||
if ((kargs->flags & (CLONE_THREAD | CLONE_PARENT)) &&
|
||||
kargs->exit_signal)
|
||||
return false;
|
||||
|
|
@ -2761,6 +2786,9 @@ SYSCALL_DEFINE2(clone3, struct clone_args __user *, uargs, size_t, size)
|
|||
int err;
|
||||
|
||||
struct kernel_clone_args kargs;
|
||||
pid_t set_tid[MAX_PID_NS_LEVEL];
|
||||
|
||||
kargs.set_tid = set_tid;
|
||||
|
||||
err = copy_clone_args_from_user(&kargs, uargs, size);
|
||||
if (err)
|
||||
|
|
|
|||
70
kernel/pid.c
70
kernel/pid.c
|
|
@ -157,7 +157,8 @@ void free_pid(struct pid *pid)
|
|||
call_rcu(&pid->rcu, delayed_put_pid);
|
||||
}
|
||||
|
||||
struct pid *alloc_pid(struct pid_namespace *ns)
|
||||
struct pid *alloc_pid(struct pid_namespace *ns, pid_t *set_tid,
|
||||
size_t set_tid_size)
|
||||
{
|
||||
struct pid *pid;
|
||||
enum pid_type type;
|
||||
|
|
@ -166,6 +167,17 @@ struct pid *alloc_pid(struct pid_namespace *ns)
|
|||
struct upid *upid;
|
||||
int retval = -ENOMEM;
|
||||
|
||||
/*
|
||||
* set_tid_size contains the size of the set_tid array. Starting at
|
||||
* the most nested currently active PID namespace it tells alloc_pid()
|
||||
* which PID to set for a process in that most nested PID namespace
|
||||
* up to set_tid_size PID namespaces. It does not have to set the PID
|
||||
* for a process in all nested PID namespaces but set_tid_size must
|
||||
* never be greater than the current ns->level + 1.
|
||||
*/
|
||||
if (set_tid_size > ns->level + 1)
|
||||
return ERR_PTR(-EINVAL);
|
||||
|
||||
pid = kmem_cache_alloc(ns->pid_cachep, GFP_KERNEL);
|
||||
if (!pid)
|
||||
return ERR_PTR(retval);
|
||||
|
|
@ -174,24 +186,54 @@ struct pid *alloc_pid(struct pid_namespace *ns)
|
|||
pid->level = ns->level;
|
||||
|
||||
for (i = ns->level; i >= 0; i--) {
|
||||
int pid_min = 1;
|
||||
int tid = 0;
|
||||
|
||||
if (set_tid_size) {
|
||||
tid = set_tid[ns->level - i];
|
||||
|
||||
retval = -EINVAL;
|
||||
if (tid < 1 || tid >= pid_max)
|
||||
goto out_free;
|
||||
/*
|
||||
* Also fail if a PID != 1 is requested and
|
||||
* no PID 1 exists.
|
||||
*/
|
||||
if (tid != 1 && !tmp->child_reaper)
|
||||
goto out_free;
|
||||
retval = -EPERM;
|
||||
if (!checkpoint_restore_ns_capable(tmp->user_ns))
|
||||
goto out_free;
|
||||
set_tid_size--;
|
||||
}
|
||||
|
||||
idr_preload(GFP_KERNEL);
|
||||
spin_lock_irq(&pidmap_lock);
|
||||
|
||||
/*
|
||||
* init really needs pid 1, but after reaching the maximum
|
||||
* wrap back to RESERVED_PIDS
|
||||
*/
|
||||
if (idr_get_cursor(&tmp->idr) > RESERVED_PIDS)
|
||||
pid_min = RESERVED_PIDS;
|
||||
if (tid) {
|
||||
nr = idr_alloc(&tmp->idr, NULL, tid,
|
||||
tid + 1, GFP_ATOMIC);
|
||||
/*
|
||||
* If ENOSPC is returned it means that the PID is
|
||||
* alreay in use. Return EEXIST in that case.
|
||||
*/
|
||||
if (nr == -ENOSPC)
|
||||
nr = -EEXIST;
|
||||
} else {
|
||||
int pid_min = 1;
|
||||
/*
|
||||
* init really needs pid 1, but after reaching the
|
||||
* maximum wrap back to RESERVED_PIDS
|
||||
*/
|
||||
if (idr_get_cursor(&tmp->idr) > RESERVED_PIDS)
|
||||
pid_min = RESERVED_PIDS;
|
||||
|
||||
/*
|
||||
* Store a null pointer so find_pid_ns does not find
|
||||
* a partially initialized PID (see below).
|
||||
*/
|
||||
nr = idr_alloc_cyclic(&tmp->idr, NULL, pid_min,
|
||||
pid_max, GFP_ATOMIC);
|
||||
/*
|
||||
* Store a null pointer so find_pid_ns does not find
|
||||
* a partially initialized PID (see below).
|
||||
*/
|
||||
nr = idr_alloc_cyclic(&tmp->idr, NULL, pid_min,
|
||||
pid_max, GFP_ATOMIC);
|
||||
}
|
||||
spin_unlock_irq(&pidmap_lock);
|
||||
idr_preload_end();
|
||||
|
||||
|
|
|
|||
|
|
@ -26,8 +26,6 @@
|
|||
|
||||
static DEFINE_MUTEX(pid_caches_mutex);
|
||||
static struct kmem_cache *pid_ns_cachep;
|
||||
/* MAX_PID_NS_LEVEL is needed for limiting size of 'struct pid' */
|
||||
#define MAX_PID_NS_LEVEL 32
|
||||
/* Write once array, filled from the beginning. */
|
||||
static struct kmem_cache *pid_cache[MAX_PID_NS_LEVEL];
|
||||
|
||||
|
|
@ -272,7 +270,7 @@ static int pid_ns_ctl_handler(struct ctl_table *table, int write,
|
|||
struct ctl_table tmp = *table;
|
||||
int ret, next;
|
||||
|
||||
if (write && !ns_capable(pid_ns->user_ns, CAP_SYS_ADMIN))
|
||||
if (write && !checkpoint_restore_ns_capable(pid_ns->user_ns))
|
||||
return -EPERM;
|
||||
|
||||
/*
|
||||
|
|
|
|||
13
kernel/sys.c
13
kernel/sys.c
|
|
@ -2005,12 +2005,15 @@ static int prctl_set_mm_map(int opt, const void __user *addr, unsigned long data
|
|||
|
||||
if (prctl_map.exe_fd != (u32)-1) {
|
||||
/*
|
||||
* Make sure the caller has the rights to
|
||||
* change /proc/pid/exe link: only local sys admin should
|
||||
* be allowed to.
|
||||
* Check if the current user is checkpoint/restore capable.
|
||||
* At the time of this writing, it checks for CAP_SYS_ADMIN
|
||||
* or CAP_CHECKPOINT_RESTORE.
|
||||
* Note that a user with access to ptrace can masquerade an
|
||||
* arbitrary program as any executable, even setuid ones.
|
||||
* This may have implications in the tomoyo subsystem.
|
||||
*/
|
||||
if (!ns_capable(current_user_ns(), CAP_SYS_ADMIN))
|
||||
return -EINVAL;
|
||||
if (!checkpoint_restore_ns_capable(current_user_ns()))
|
||||
return -EPERM;
|
||||
|
||||
error = prctl_set_mm_exe_file(mm, prctl_map.exe_fd);
|
||||
if (error)
|
||||
|
|
|
|||
|
|
@ -27,9 +27,10 @@
|
|||
"audit_control", "setfcap"
|
||||
|
||||
#define COMMON_CAP2_PERMS "mac_override", "mac_admin", "syslog", \
|
||||
"wake_alarm", "block_suspend", "audit_read", "perfmon", "bpf"
|
||||
"wake_alarm", "block_suspend", "audit_read", "perfmon", "bpf", \
|
||||
"checkpoint_restore"
|
||||
|
||||
#if CAP_LAST_CAP > CAP_BPF
|
||||
#if CAP_LAST_CAP > CAP_CHECKPOINT_RESTORE
|
||||
#error New capability defined, please update COMMON_CAP2_PERMS.
|
||||
#endif
|
||||
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ TARGETS += bpf
|
|||
TARGETS += breakpoints
|
||||
TARGETS += capabilities
|
||||
TARGETS += cgroup
|
||||
TARGETS += clone3
|
||||
TARGETS += cpufreq
|
||||
TARGETS += cpu-hotplug
|
||||
TARGETS += drivers/dma-buf
|
||||
|
|
|
|||
4
tools/testing/selftests/clone3/.gitignore
vendored
Normal file
4
tools/testing/selftests/clone3/.gitignore
vendored
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
clone3
|
||||
clone3_clear_sighand
|
||||
clone3_set_tid
|
||||
clone3_cap_checkpoint_restore
|
||||
8
tools/testing/selftests/clone3/Makefile
Normal file
8
tools/testing/selftests/clone3/Makefile
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
# SPDX-License-Identifier: GPL-2.0
|
||||
CFLAGS += -g -I../../../../usr/include/
|
||||
LDLIBS += -lcap
|
||||
|
||||
TEST_GEN_PROGS := clone3 clone3_clear_sighand clone3_set_tid \
|
||||
clone3_cap_checkpoint_restore
|
||||
|
||||
include ../lib.mk
|
||||
201
tools/testing/selftests/clone3/clone3.c
Normal file
201
tools/testing/selftests/clone3/clone3.c
Normal file
|
|
@ -0,0 +1,201 @@
|
|||
// SPDX-License-Identifier: GPL-2.0
|
||||
|
||||
/* Based on Christian Brauner's clone3() example */
|
||||
|
||||
#define _GNU_SOURCE
|
||||
#include <errno.h>
|
||||
#include <inttypes.h>
|
||||
#include <linux/types.h>
|
||||
#include <linux/sched.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/un.h>
|
||||
#include <sys/wait.h>
|
||||
#include <unistd.h>
|
||||
#include <sched.h>
|
||||
|
||||
#include "../kselftest.h"
|
||||
#include "clone3_selftests.h"
|
||||
|
||||
/*
|
||||
* Different sizes of struct clone_args
|
||||
*/
|
||||
#ifndef CLONE3_ARGS_SIZE_V0
|
||||
#define CLONE3_ARGS_SIZE_V0 64
|
||||
#endif
|
||||
|
||||
enum test_mode {
|
||||
CLONE3_ARGS_NO_TEST,
|
||||
CLONE3_ARGS_ALL_0,
|
||||
CLONE3_ARGS_INVAL_EXIT_SIGNAL_BIG,
|
||||
CLONE3_ARGS_INVAL_EXIT_SIGNAL_NEG,
|
||||
CLONE3_ARGS_INVAL_EXIT_SIGNAL_CSIG,
|
||||
CLONE3_ARGS_INVAL_EXIT_SIGNAL_NSIG,
|
||||
};
|
||||
|
||||
static int call_clone3(uint64_t flags, size_t size, enum test_mode test_mode)
|
||||
{
|
||||
struct clone_args args = {
|
||||
.flags = flags,
|
||||
.exit_signal = SIGCHLD,
|
||||
};
|
||||
|
||||
struct clone_args_extended {
|
||||
struct clone_args args;
|
||||
__aligned_u64 excess_space[2];
|
||||
} args_ext;
|
||||
|
||||
pid_t pid = -1;
|
||||
int status;
|
||||
|
||||
memset(&args_ext, 0, sizeof(args_ext));
|
||||
if (size > sizeof(struct clone_args))
|
||||
args_ext.excess_space[1] = 1;
|
||||
|
||||
if (size == 0)
|
||||
size = sizeof(struct clone_args);
|
||||
|
||||
switch (test_mode) {
|
||||
case CLONE3_ARGS_ALL_0:
|
||||
args.flags = 0;
|
||||
args.exit_signal = 0;
|
||||
break;
|
||||
case CLONE3_ARGS_INVAL_EXIT_SIGNAL_BIG:
|
||||
args.exit_signal = 0xbadc0ded00000000ULL;
|
||||
break;
|
||||
case CLONE3_ARGS_INVAL_EXIT_SIGNAL_NEG:
|
||||
args.exit_signal = 0x0000000080000000ULL;
|
||||
break;
|
||||
case CLONE3_ARGS_INVAL_EXIT_SIGNAL_CSIG:
|
||||
args.exit_signal = 0x0000000000000100ULL;
|
||||
break;
|
||||
case CLONE3_ARGS_INVAL_EXIT_SIGNAL_NSIG:
|
||||
args.exit_signal = 0x00000000000000f0ULL;
|
||||
break;
|
||||
}
|
||||
|
||||
memcpy(&args_ext.args, &args, sizeof(struct clone_args));
|
||||
|
||||
pid = sys_clone3((struct clone_args *)&args_ext, size);
|
||||
if (pid < 0) {
|
||||
ksft_print_msg("%s - Failed to create new process\n",
|
||||
strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (pid == 0) {
|
||||
ksft_print_msg("I am the child, my PID is %d\n", getpid());
|
||||
_exit(EXIT_SUCCESS);
|
||||
}
|
||||
|
||||
ksft_print_msg("I am the parent (%d). My child's pid is %d\n",
|
||||
getpid(), pid);
|
||||
|
||||
if (waitpid(-1, &status, __WALL) < 0) {
|
||||
ksft_print_msg("Child returned %s\n", strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
if (WEXITSTATUS(status))
|
||||
return WEXITSTATUS(status);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void test_clone3(uint64_t flags, size_t size, int expected,
|
||||
enum test_mode test_mode)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ksft_print_msg(
|
||||
"[%d] Trying clone3() with flags %#" PRIx64 " (size %zu)\n",
|
||||
getpid(), flags, size);
|
||||
ret = call_clone3(flags, size, test_mode);
|
||||
ksft_print_msg("[%d] clone3() with flags says: %d expected %d\n",
|
||||
getpid(), ret, expected);
|
||||
if (ret != expected)
|
||||
ksft_test_result_fail(
|
||||
"[%d] Result (%d) is different than expected (%d)\n",
|
||||
getpid(), ret, expected);
|
||||
else
|
||||
ksft_test_result_pass(
|
||||
"[%d] Result (%d) matches expectation (%d)\n",
|
||||
getpid(), ret, expected);
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
pid_t pid;
|
||||
|
||||
uid_t uid = getuid();
|
||||
|
||||
ksft_print_header();
|
||||
ksft_set_plan(17);
|
||||
|
||||
/* Just a simple clone3() should return 0.*/
|
||||
test_clone3(0, 0, 0, CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() in a new PID NS.*/
|
||||
if (uid == 0)
|
||||
test_clone3(CLONE_NEWPID, 0, 0, CLONE3_ARGS_NO_TEST);
|
||||
else
|
||||
ksft_test_result_skip("Skipping clone3() with CLONE_NEWPID\n");
|
||||
|
||||
/* Do a clone3() with CLONE3_ARGS_SIZE_V0. */
|
||||
test_clone3(0, CLONE3_ARGS_SIZE_V0, 0, CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() with CLONE3_ARGS_SIZE_V0 - 8 */
|
||||
test_clone3(0, CLONE3_ARGS_SIZE_V0 - 8, -EINVAL, CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() with sizeof(struct clone_args) + 8 */
|
||||
test_clone3(0, sizeof(struct clone_args) + 8, 0, CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() with exit_signal having highest 32 bits non-zero */
|
||||
test_clone3(0, 0, -EINVAL, CLONE3_ARGS_INVAL_EXIT_SIGNAL_BIG);
|
||||
|
||||
/* Do a clone3() with negative 32-bit exit_signal */
|
||||
test_clone3(0, 0, -EINVAL, CLONE3_ARGS_INVAL_EXIT_SIGNAL_NEG);
|
||||
|
||||
/* Do a clone3() with exit_signal not fitting into CSIGNAL mask */
|
||||
test_clone3(0, 0, -EINVAL, CLONE3_ARGS_INVAL_EXIT_SIGNAL_CSIG);
|
||||
|
||||
/* Do a clone3() with NSIG < exit_signal < CSIG */
|
||||
test_clone3(0, 0, -EINVAL, CLONE3_ARGS_INVAL_EXIT_SIGNAL_NSIG);
|
||||
|
||||
test_clone3(0, sizeof(struct clone_args) + 8, 0, CLONE3_ARGS_ALL_0);
|
||||
|
||||
test_clone3(0, sizeof(struct clone_args) + 16, -E2BIG,
|
||||
CLONE3_ARGS_ALL_0);
|
||||
|
||||
test_clone3(0, sizeof(struct clone_args) * 2, -E2BIG,
|
||||
CLONE3_ARGS_ALL_0);
|
||||
|
||||
/* Do a clone3() with > page size */
|
||||
test_clone3(0, getpagesize() + 8, -E2BIG, CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() with CLONE3_ARGS_SIZE_V0 in a new PID NS. */
|
||||
if (uid == 0)
|
||||
test_clone3(CLONE_NEWPID, CLONE3_ARGS_SIZE_V0, 0,
|
||||
CLONE3_ARGS_NO_TEST);
|
||||
else
|
||||
ksft_test_result_skip("Skipping clone3() with CLONE_NEWPID\n");
|
||||
|
||||
/* Do a clone3() with CLONE3_ARGS_SIZE_V0 - 8 in a new PID NS */
|
||||
test_clone3(CLONE_NEWPID, CLONE3_ARGS_SIZE_V0 - 8, -EINVAL,
|
||||
CLONE3_ARGS_NO_TEST);
|
||||
|
||||
/* Do a clone3() with sizeof(struct clone_args) + 8 in a new PID NS */
|
||||
if (uid == 0)
|
||||
test_clone3(CLONE_NEWPID, sizeof(struct clone_args) + 8, 0,
|
||||
CLONE3_ARGS_NO_TEST);
|
||||
else
|
||||
ksft_test_result_skip("Skipping clone3() with CLONE_NEWPID\n");
|
||||
|
||||
/* Do a clone3() with > page size in a new PID NS */
|
||||
test_clone3(CLONE_NEWPID, getpagesize() + 8, -E2BIG,
|
||||
CLONE3_ARGS_NO_TEST);
|
||||
|
||||
return !ksft_get_fail_cnt() ? ksft_exit_pass() : ksft_exit_fail();
|
||||
}
|
||||
182
tools/testing/selftests/clone3/clone3_cap_checkpoint_restore.c
Normal file
182
tools/testing/selftests/clone3/clone3_cap_checkpoint_restore.c
Normal file
|
|
@ -0,0 +1,182 @@
|
|||
// SPDX-License-Identifier: GPL-2.0
|
||||
|
||||
/*
|
||||
* Based on Christian Brauner's clone3() example.
|
||||
* These tests are assuming to be running in the host's
|
||||
* PID namespace.
|
||||
*/
|
||||
|
||||
/* capabilities related code based on selftests/bpf/test_verifier.c */
|
||||
|
||||
#define _GNU_SOURCE
|
||||
#include <errno.h>
|
||||
#include <linux/types.h>
|
||||
#include <linux/sched.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdbool.h>
|
||||
#include <sys/capability.h>
|
||||
#include <sys/prctl.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/un.h>
|
||||
#include <sys/wait.h>
|
||||
#include <unistd.h>
|
||||
#include <sched.h>
|
||||
|
||||
#include "../kselftest_harness.h"
|
||||
#include "clone3_selftests.h"
|
||||
|
||||
#ifndef MAX_PID_NS_LEVEL
|
||||
#define MAX_PID_NS_LEVEL 32
|
||||
#endif
|
||||
|
||||
static void child_exit(int ret)
|
||||
{
|
||||
fflush(stdout);
|
||||
fflush(stderr);
|
||||
_exit(ret);
|
||||
}
|
||||
|
||||
static int call_clone3_set_tid(struct __test_metadata *_metadata,
|
||||
pid_t *set_tid, size_t set_tid_size)
|
||||
{
|
||||
int status;
|
||||
pid_t pid = -1;
|
||||
|
||||
struct clone_args args = {
|
||||
.exit_signal = SIGCHLD,
|
||||
.set_tid = ptr_to_u64(set_tid),
|
||||
.set_tid_size = set_tid_size,
|
||||
};
|
||||
|
||||
pid = sys_clone3(&args, sizeof(struct clone_args));
|
||||
if (pid < 0) {
|
||||
TH_LOG("%s - Failed to create new process", strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (pid == 0) {
|
||||
int ret;
|
||||
char tmp = 0;
|
||||
|
||||
TH_LOG("I am the child, my PID is %d (expected %d)", getpid(), set_tid[0]);
|
||||
|
||||
if (set_tid[0] != getpid())
|
||||
child_exit(EXIT_FAILURE);
|
||||
child_exit(EXIT_SUCCESS);
|
||||
}
|
||||
|
||||
TH_LOG("I am the parent (%d). My child's pid is %d", getpid(), pid);
|
||||
|
||||
if (waitpid(pid, &status, 0) < 0) {
|
||||
TH_LOG("Child returned %s", strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (!WIFEXITED(status))
|
||||
return -1;
|
||||
|
||||
return WEXITSTATUS(status);
|
||||
}
|
||||
|
||||
static int test_clone3_set_tid(struct __test_metadata *_metadata,
|
||||
pid_t *set_tid, size_t set_tid_size)
|
||||
{
|
||||
int ret;
|
||||
|
||||
TH_LOG("[%d] Trying clone3() with CLONE_SET_TID to %d", getpid(), set_tid[0]);
|
||||
ret = call_clone3_set_tid(_metadata, set_tid, set_tid_size);
|
||||
TH_LOG("[%d] clone3() with CLONE_SET_TID %d says:%d", getpid(), set_tid[0], ret);
|
||||
return ret;
|
||||
}
|
||||
|
||||
struct libcap {
|
||||
struct __user_cap_header_struct hdr;
|
||||
struct __user_cap_data_struct data[2];
|
||||
};
|
||||
|
||||
static int set_capability(void)
|
||||
{
|
||||
cap_value_t cap_values[] = { CAP_SETUID, CAP_SETGID };
|
||||
struct libcap *cap;
|
||||
int ret = -1;
|
||||
cap_t caps;
|
||||
|
||||
caps = cap_get_proc();
|
||||
if (!caps) {
|
||||
perror("cap_get_proc");
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Drop all capabilities */
|
||||
if (cap_clear(caps)) {
|
||||
perror("cap_clear");
|
||||
goto out;
|
||||
}
|
||||
|
||||
cap_set_flag(caps, CAP_EFFECTIVE, 2, cap_values, CAP_SET);
|
||||
cap_set_flag(caps, CAP_PERMITTED, 2, cap_values, CAP_SET);
|
||||
|
||||
cap = (struct libcap *) caps;
|
||||
|
||||
/* 40 -> CAP_CHECKPOINT_RESTORE */
|
||||
cap->data[1].effective |= 1 << (40 - 32);
|
||||
cap->data[1].permitted |= 1 << (40 - 32);
|
||||
|
||||
if (cap_set_proc(caps)) {
|
||||
perror("cap_set_proc");
|
||||
goto out;
|
||||
}
|
||||
ret = 0;
|
||||
out:
|
||||
if (cap_free(caps))
|
||||
perror("cap_free");
|
||||
return ret;
|
||||
}
|
||||
|
||||
TEST(clone3_cap_checkpoint_restore)
|
||||
{
|
||||
pid_t pid;
|
||||
int status;
|
||||
int ret = 0;
|
||||
pid_t set_tid[1];
|
||||
|
||||
test_clone3_supported();
|
||||
|
||||
EXPECT_EQ(getuid(), 0)
|
||||
XFAIL(return, "Skipping all tests as non-root\n");
|
||||
|
||||
memset(&set_tid, 0, sizeof(set_tid));
|
||||
|
||||
/* Find the current active PID */
|
||||
pid = fork();
|
||||
if (pid == 0) {
|
||||
TH_LOG("Child has PID %d", getpid());
|
||||
child_exit(EXIT_SUCCESS);
|
||||
}
|
||||
ASSERT_GT(waitpid(pid, &status, 0), 0)
|
||||
TH_LOG("Waiting for child %d failed", pid);
|
||||
|
||||
/* After the child has finished, its PID should be free. */
|
||||
set_tid[0] = pid;
|
||||
|
||||
ASSERT_EQ(set_capability(), 0)
|
||||
TH_LOG("Could not set CAP_CHECKPOINT_RESTORE");
|
||||
|
||||
ASSERT_EQ(prctl(PR_SET_KEEPCAPS, 1, 0, 0, 0), 0);
|
||||
|
||||
EXPECT_EQ(setgid(65534), 0)
|
||||
TH_LOG("Failed to setgid(65534)");
|
||||
ASSERT_EQ(setuid(65534), 0);
|
||||
|
||||
set_tid[0] = pid;
|
||||
/* This would fail without CAP_CHECKPOINT_RESTORE */
|
||||
ASSERT_EQ(test_clone3_set_tid(_metadata, set_tid, 1), -EPERM);
|
||||
ASSERT_EQ(set_capability(), 0)
|
||||
TH_LOG("Could not set CAP_CHECKPOINT_RESTORE");
|
||||
/* This should work as we have CAP_CHECKPOINT_RESTORE as non-root */
|
||||
ASSERT_EQ(test_clone3_set_tid(_metadata, set_tid, 1), 0);
|
||||
}
|
||||
|
||||
TEST_HARNESS_MAIN
|
||||
154
tools/testing/selftests/clone3/clone3_clear_sighand.c
Normal file
154
tools/testing/selftests/clone3/clone3_clear_sighand.c
Normal file
|
|
@ -0,0 +1,154 @@
|
|||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
|
||||
#define _GNU_SOURCE
|
||||
#include <errno.h>
|
||||
#include <sched.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
#include <linux/sched.h>
|
||||
#include <linux/types.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/wait.h>
|
||||
|
||||
#include "../kselftest.h"
|
||||
#include "clone3_selftests.h"
|
||||
|
||||
#ifndef CLONE_CLEAR_SIGHAND
|
||||
#define CLONE_CLEAR_SIGHAND 0x100000000ULL
|
||||
#endif
|
||||
|
||||
static void test_clone3_supported(void)
|
||||
{
|
||||
pid_t pid;
|
||||
struct clone_args args = {};
|
||||
|
||||
if (__NR_clone3 < 0)
|
||||
ksft_exit_skip("clone3() syscall is not supported\n");
|
||||
|
||||
/* Set to something that will always cause EINVAL. */
|
||||
args.exit_signal = -1;
|
||||
pid = sys_clone3(&args, sizeof(args));
|
||||
if (!pid)
|
||||
exit(EXIT_SUCCESS);
|
||||
|
||||
if (pid > 0) {
|
||||
wait(NULL);
|
||||
ksft_exit_fail_msg(
|
||||
"Managed to create child process with invalid exit_signal\n");
|
||||
}
|
||||
|
||||
if (errno == ENOSYS)
|
||||
ksft_exit_skip("clone3() syscall is not supported\n");
|
||||
|
||||
ksft_print_msg("clone3() syscall supported\n");
|
||||
}
|
||||
|
||||
static void nop_handler(int signo)
|
||||
{
|
||||
}
|
||||
|
||||
static int wait_for_pid(pid_t pid)
|
||||
{
|
||||
int status, ret;
|
||||
|
||||
again:
|
||||
ret = waitpid(pid, &status, 0);
|
||||
if (ret == -1) {
|
||||
if (errno == EINTR)
|
||||
goto again;
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (!WIFEXITED(status))
|
||||
return -1;
|
||||
|
||||
return WEXITSTATUS(status);
|
||||
}
|
||||
|
||||
static void test_clone3_clear_sighand(void)
|
||||
{
|
||||
int ret;
|
||||
pid_t pid;
|
||||
struct clone_args args = {};
|
||||
struct sigaction act;
|
||||
|
||||
/*
|
||||
* Check that CLONE_CLEAR_SIGHAND and CLONE_SIGHAND are mutually
|
||||
* exclusive.
|
||||
*/
|
||||
args.flags |= CLONE_CLEAR_SIGHAND | CLONE_SIGHAND;
|
||||
args.exit_signal = SIGCHLD;
|
||||
pid = sys_clone3(&args, sizeof(args));
|
||||
if (pid > 0)
|
||||
ksft_exit_fail_msg(
|
||||
"clone3(CLONE_CLEAR_SIGHAND | CLONE_SIGHAND) succeeded\n");
|
||||
|
||||
act.sa_handler = nop_handler;
|
||||
ret = sigemptyset(&act.sa_mask);
|
||||
if (ret < 0)
|
||||
ksft_exit_fail_msg("%s - sigemptyset() failed\n",
|
||||
strerror(errno));
|
||||
|
||||
act.sa_flags = 0;
|
||||
|
||||
/* Register signal handler for SIGUSR1 */
|
||||
ret = sigaction(SIGUSR1, &act, NULL);
|
||||
if (ret < 0)
|
||||
ksft_exit_fail_msg(
|
||||
"%s - sigaction(SIGUSR1, &act, NULL) failed\n",
|
||||
strerror(errno));
|
||||
|
||||
/* Register signal handler for SIGUSR2 */
|
||||
ret = sigaction(SIGUSR2, &act, NULL);
|
||||
if (ret < 0)
|
||||
ksft_exit_fail_msg(
|
||||
"%s - sigaction(SIGUSR2, &act, NULL) failed\n",
|
||||
strerror(errno));
|
||||
|
||||
/* Check that CLONE_CLEAR_SIGHAND works. */
|
||||
args.flags = CLONE_CLEAR_SIGHAND;
|
||||
pid = sys_clone3(&args, sizeof(args));
|
||||
if (pid < 0)
|
||||
ksft_exit_fail_msg("%s - clone3(CLONE_CLEAR_SIGHAND) failed\n",
|
||||
strerror(errno));
|
||||
|
||||
if (pid == 0) {
|
||||
ret = sigaction(SIGUSR1, NULL, &act);
|
||||
if (ret < 0)
|
||||
exit(EXIT_FAILURE);
|
||||
|
||||
if (act.sa_handler != SIG_DFL)
|
||||
exit(EXIT_FAILURE);
|
||||
|
||||
ret = sigaction(SIGUSR2, NULL, &act);
|
||||
if (ret < 0)
|
||||
exit(EXIT_FAILURE);
|
||||
|
||||
if (act.sa_handler != SIG_DFL)
|
||||
exit(EXIT_FAILURE);
|
||||
|
||||
exit(EXIT_SUCCESS);
|
||||
}
|
||||
|
||||
ret = wait_for_pid(pid);
|
||||
if (ret)
|
||||
ksft_exit_fail_msg(
|
||||
"Failed to clear signal handler for child process\n");
|
||||
|
||||
ksft_test_result_pass("Cleared signal handlers for child process\n");
|
||||
}
|
||||
|
||||
int main(int argc, char **argv)
|
||||
{
|
||||
ksft_print_header();
|
||||
ksft_set_plan(1);
|
||||
|
||||
test_clone3_supported();
|
||||
test_clone3_clear_sighand();
|
||||
|
||||
return ksft_exit_pass();
|
||||
}
|
||||
35
tools/testing/selftests/clone3/clone3_selftests.h
Normal file
35
tools/testing/selftests/clone3/clone3_selftests.h
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
/* SPDX-License-Identifier: GPL-2.0 */
|
||||
|
||||
#ifndef _CLONE3_SELFTESTS_H
|
||||
#define _CLONE3_SELFTESTS_H
|
||||
|
||||
#define _GNU_SOURCE
|
||||
#include <sched.h>
|
||||
#include <stdint.h>
|
||||
#include <syscall.h>
|
||||
#include <linux/types.h>
|
||||
|
||||
#define ptr_to_u64(ptr) ((__u64)((uintptr_t)(ptr)))
|
||||
|
||||
#ifndef __NR_clone3
|
||||
#define __NR_clone3 -1
|
||||
struct clone_args {
|
||||
__aligned_u64 flags;
|
||||
__aligned_u64 pidfd;
|
||||
__aligned_u64 child_tid;
|
||||
__aligned_u64 parent_tid;
|
||||
__aligned_u64 exit_signal;
|
||||
__aligned_u64 stack;
|
||||
__aligned_u64 stack_size;
|
||||
__aligned_u64 tls;
|
||||
__aligned_u64 set_tid;
|
||||
__aligned_u64 set_tid_size;
|
||||
};
|
||||
#endif
|
||||
|
||||
static pid_t sys_clone3(struct clone_args *args, size_t size)
|
||||
{
|
||||
return syscall(__NR_clone3, args, size);
|
||||
}
|
||||
|
||||
#endif /* _CLONE3_SELFTESTS_H */
|
||||
381
tools/testing/selftests/clone3/clone3_set_tid.c
Normal file
381
tools/testing/selftests/clone3/clone3_set_tid.c
Normal file
|
|
@ -0,0 +1,381 @@
|
|||
// SPDX-License-Identifier: GPL-2.0
|
||||
|
||||
/*
|
||||
* Based on Christian Brauner's clone3() example.
|
||||
* These tests are assuming to be running in the host's
|
||||
* PID namespace.
|
||||
*/
|
||||
|
||||
#define _GNU_SOURCE
|
||||
#include <errno.h>
|
||||
#include <linux/types.h>
|
||||
#include <linux/sched.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdbool.h>
|
||||
#include <sys/syscall.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/un.h>
|
||||
#include <sys/wait.h>
|
||||
#include <unistd.h>
|
||||
#include <sched.h>
|
||||
|
||||
#include "../kselftest.h"
|
||||
#include "clone3_selftests.h"
|
||||
|
||||
#ifndef MAX_PID_NS_LEVEL
|
||||
#define MAX_PID_NS_LEVEL 32
|
||||
#endif
|
||||
|
||||
static int pipe_1[2];
|
||||
static int pipe_2[2];
|
||||
|
||||
static int call_clone3_set_tid(pid_t *set_tid,
|
||||
size_t set_tid_size,
|
||||
int flags,
|
||||
int expected_pid,
|
||||
bool wait_for_it)
|
||||
{
|
||||
int status;
|
||||
pid_t pid = -1;
|
||||
|
||||
struct clone_args args = {
|
||||
.flags = flags,
|
||||
.exit_signal = SIGCHLD,
|
||||
.set_tid = ptr_to_u64(set_tid),
|
||||
.set_tid_size = set_tid_size,
|
||||
};
|
||||
|
||||
pid = sys_clone3(&args, sizeof(struct clone_args));
|
||||
if (pid < 0) {
|
||||
ksft_print_msg("%s - Failed to create new process\n",
|
||||
strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (pid == 0) {
|
||||
int ret;
|
||||
char tmp = 0;
|
||||
int exit_code = EXIT_SUCCESS;
|
||||
|
||||
ksft_print_msg("I am the child, my PID is %d (expected %d)\n",
|
||||
getpid(), set_tid[0]);
|
||||
if (wait_for_it) {
|
||||
ksft_print_msg("[%d] Child is ready and waiting\n",
|
||||
getpid());
|
||||
|
||||
/* Signal the parent that the child is ready */
|
||||
close(pipe_1[0]);
|
||||
ret = write(pipe_1[1], &tmp, 1);
|
||||
if (ret != 1) {
|
||||
ksft_print_msg(
|
||||
"Writing to pipe returned %d", ret);
|
||||
exit_code = EXIT_FAILURE;
|
||||
}
|
||||
close(pipe_1[1]);
|
||||
close(pipe_2[1]);
|
||||
ret = read(pipe_2[0], &tmp, 1);
|
||||
if (ret != 1) {
|
||||
ksft_print_msg(
|
||||
"Reading from pipe returned %d", ret);
|
||||
exit_code = EXIT_FAILURE;
|
||||
}
|
||||
close(pipe_2[0]);
|
||||
}
|
||||
|
||||
if (set_tid[0] != getpid())
|
||||
_exit(EXIT_FAILURE);
|
||||
_exit(exit_code);
|
||||
}
|
||||
|
||||
if (expected_pid == 0 || expected_pid == pid) {
|
||||
ksft_print_msg("I am the parent (%d). My child's pid is %d\n",
|
||||
getpid(), pid);
|
||||
} else {
|
||||
ksft_print_msg(
|
||||
"Expected child pid %d does not match actual pid %d\n",
|
||||
expected_pid, pid);
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (waitpid(pid, &status, 0) < 0) {
|
||||
ksft_print_msg("Child returned %s\n", strerror(errno));
|
||||
return -errno;
|
||||
}
|
||||
|
||||
if (!WIFEXITED(status))
|
||||
return -1;
|
||||
|
||||
return WEXITSTATUS(status);
|
||||
}
|
||||
|
||||
static void test_clone3_set_tid(pid_t *set_tid,
|
||||
size_t set_tid_size,
|
||||
int flags,
|
||||
int expected,
|
||||
int expected_pid,
|
||||
bool wait_for_it)
|
||||
{
|
||||
int ret;
|
||||
|
||||
ksft_print_msg(
|
||||
"[%d] Trying clone3() with CLONE_SET_TID to %d and 0x%x\n",
|
||||
getpid(), set_tid[0], flags);
|
||||
ret = call_clone3_set_tid(set_tid, set_tid_size, flags, expected_pid,
|
||||
wait_for_it);
|
||||
ksft_print_msg(
|
||||
"[%d] clone3() with CLONE_SET_TID %d says :%d - expected %d\n",
|
||||
getpid(), set_tid[0], ret, expected);
|
||||
if (ret != expected)
|
||||
ksft_test_result_fail(
|
||||
"[%d] Result (%d) is different than expected (%d)\n",
|
||||
getpid(), ret, expected);
|
||||
else
|
||||
ksft_test_result_pass(
|
||||
"[%d] Result (%d) matches expectation (%d)\n",
|
||||
getpid(), ret, expected);
|
||||
}
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
FILE *f;
|
||||
char buf;
|
||||
char *line;
|
||||
int status;
|
||||
int ret = -1;
|
||||
size_t len = 0;
|
||||
int pid_max = 0;
|
||||
uid_t uid = getuid();
|
||||
char proc_path[100] = {0};
|
||||
pid_t pid, ns1, ns2, ns3, ns_pid;
|
||||
pid_t set_tid[MAX_PID_NS_LEVEL * 2];
|
||||
|
||||
if (pipe(pipe_1) < 0 || pipe(pipe_2) < 0)
|
||||
ksft_exit_fail_msg("pipe() failed\n");
|
||||
|
||||
ksft_print_header();
|
||||
ksft_set_plan(27);
|
||||
|
||||
f = fopen("/proc/sys/kernel/pid_max", "r");
|
||||
if (f == NULL)
|
||||
ksft_exit_fail_msg(
|
||||
"%s - Could not open /proc/sys/kernel/pid_max\n",
|
||||
strerror(errno));
|
||||
fscanf(f, "%d", &pid_max);
|
||||
fclose(f);
|
||||
ksft_print_msg("/proc/sys/kernel/pid_max %d\n", pid_max);
|
||||
|
||||
/* Try invalid settings */
|
||||
memset(&set_tid, 0, sizeof(set_tid));
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL + 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 2, 0, -EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 2 + 1, 0,
|
||||
-EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 42, 0, -EINVAL, 0, 0);
|
||||
|
||||
/*
|
||||
* This can actually work if this test running in a MAX_PID_NS_LEVEL - 1
|
||||
* nested PID namespace.
|
||||
*/
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL - 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
memset(&set_tid, 0xff, sizeof(set_tid));
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL + 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 2, 0, -EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 2 + 1, 0,
|
||||
-EINVAL, 0, 0);
|
||||
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL * 42, 0, -EINVAL, 0, 0);
|
||||
|
||||
/*
|
||||
* This can actually work if this test running in a MAX_PID_NS_LEVEL - 1
|
||||
* nested PID namespace.
|
||||
*/
|
||||
test_clone3_set_tid(set_tid, MAX_PID_NS_LEVEL - 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
memset(&set_tid, 0, sizeof(set_tid));
|
||||
/* Try with an invalid PID */
|
||||
set_tid[0] = 0;
|
||||
test_clone3_set_tid(set_tid, 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
set_tid[0] = -1;
|
||||
test_clone3_set_tid(set_tid, 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
/* Claim that the set_tid array actually contains 2 elements. */
|
||||
test_clone3_set_tid(set_tid, 2, 0, -EINVAL, 0, 0);
|
||||
|
||||
/* Try it in a new PID namespace */
|
||||
if (uid == 0)
|
||||
test_clone3_set_tid(set_tid, 1, CLONE_NEWPID, -EINVAL, 0, 0);
|
||||
else
|
||||
ksft_test_result_skip("Clone3() with set_tid requires root\n");
|
||||
|
||||
/* Try with a valid PID (1) this should return -EEXIST. */
|
||||
set_tid[0] = 1;
|
||||
if (uid == 0)
|
||||
test_clone3_set_tid(set_tid, 1, 0, -EEXIST, 0, 0);
|
||||
else
|
||||
ksft_test_result_skip("Clone3() with set_tid requires root\n");
|
||||
|
||||
/* Try it in a new PID namespace */
|
||||
if (uid == 0)
|
||||
test_clone3_set_tid(set_tid, 1, CLONE_NEWPID, 0, 0, 0);
|
||||
else
|
||||
ksft_test_result_skip("Clone3() with set_tid requires root\n");
|
||||
|
||||
/* pid_max should fail everywhere */
|
||||
set_tid[0] = pid_max;
|
||||
test_clone3_set_tid(set_tid, 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
if (uid == 0)
|
||||
test_clone3_set_tid(set_tid, 1, CLONE_NEWPID, -EINVAL, 0, 0);
|
||||
else
|
||||
ksft_test_result_skip("Clone3() with set_tid requires root\n");
|
||||
|
||||
if (uid != 0) {
|
||||
/*
|
||||
* All remaining tests require root. Tell the framework
|
||||
* that all those tests are skipped as non-root.
|
||||
*/
|
||||
ksft_cnt.ksft_xskip += ksft_plan - ksft_test_num();
|
||||
goto out;
|
||||
}
|
||||
|
||||
/* Find the current active PID */
|
||||
pid = fork();
|
||||
if (pid == 0) {
|
||||
ksft_print_msg("Child has PID %d\n", getpid());
|
||||
_exit(EXIT_SUCCESS);
|
||||
}
|
||||
if (waitpid(pid, &status, 0) < 0)
|
||||
ksft_exit_fail_msg("Waiting for child %d failed", pid);
|
||||
|
||||
/* After the child has finished, its PID should be free. */
|
||||
set_tid[0] = pid;
|
||||
test_clone3_set_tid(set_tid, 1, 0, 0, 0, 0);
|
||||
|
||||
/* This should fail as there is no PID 1 in that namespace */
|
||||
test_clone3_set_tid(set_tid, 1, CLONE_NEWPID, -EINVAL, 0, 0);
|
||||
|
||||
/*
|
||||
* Creating a process with PID 1 in the newly created most nested
|
||||
* PID namespace and PID 'pid' in the parent PID namespace. This
|
||||
* needs to work.
|
||||
*/
|
||||
set_tid[0] = 1;
|
||||
set_tid[1] = pid;
|
||||
test_clone3_set_tid(set_tid, 2, CLONE_NEWPID, 0, pid, 0);
|
||||
|
||||
ksft_print_msg("unshare PID namespace\n");
|
||||
if (unshare(CLONE_NEWPID) == -1)
|
||||
ksft_exit_fail_msg("unshare(CLONE_NEWPID) failed: %s\n",
|
||||
strerror(errno));
|
||||
|
||||
set_tid[0] = pid;
|
||||
|
||||
/* This should fail as there is no PID 1 in that namespace */
|
||||
test_clone3_set_tid(set_tid, 1, 0, -EINVAL, 0, 0);
|
||||
|
||||
/* Let's create a PID 1 */
|
||||
ns_pid = fork();
|
||||
if (ns_pid == 0) {
|
||||
ksft_print_msg("Child in PID namespace has PID %d\n", getpid());
|
||||
set_tid[0] = 2;
|
||||
test_clone3_set_tid(set_tid, 1, 0, 0, 2, 0);
|
||||
|
||||
set_tid[0] = 1;
|
||||
set_tid[1] = -1;
|
||||
set_tid[2] = pid;
|
||||
/* This should fail as there is invalid PID at level '1'. */
|
||||
test_clone3_set_tid(set_tid, 3, CLONE_NEWPID, -EINVAL, 0, 0);
|
||||
|
||||
set_tid[0] = 1;
|
||||
set_tid[1] = 42;
|
||||
set_tid[2] = pid;
|
||||
/*
|
||||
* This should fail as there are not enough active PID
|
||||
* namespaces. Again assuming this is running in the host's
|
||||
* PID namespace. Not yet nested.
|
||||
*/
|
||||
test_clone3_set_tid(set_tid, 4, CLONE_NEWPID, -EINVAL, 0, 0);
|
||||
|
||||
/*
|
||||
* This should work and from the parent we should see
|
||||
* something like 'NSpid: pid 42 1'.
|
||||
*/
|
||||
test_clone3_set_tid(set_tid, 3, CLONE_NEWPID, 0, 42, true);
|
||||
|
||||
_exit(ksft_cnt.ksft_pass);
|
||||
}
|
||||
|
||||
close(pipe_1[1]);
|
||||
close(pipe_2[0]);
|
||||
while (read(pipe_1[0], &buf, 1) > 0) {
|
||||
ksft_print_msg("[%d] Child is ready and waiting\n", getpid());
|
||||
break;
|
||||
}
|
||||
|
||||
snprintf(proc_path, sizeof(proc_path), "/proc/%d/status", pid);
|
||||
f = fopen(proc_path, "r");
|
||||
if (f == NULL)
|
||||
ksft_exit_fail_msg(
|
||||
"%s - Could not open %s\n",
|
||||
strerror(errno), proc_path);
|
||||
|
||||
while (getline(&line, &len, f) != -1) {
|
||||
if (strstr(line, "NSpid")) {
|
||||
int i;
|
||||
|
||||
/* Verify that all generated PIDs are as expected. */
|
||||
i = sscanf(line, "NSpid:\t%d\t%d\t%d",
|
||||
&ns3, &ns2, &ns1);
|
||||
if (i != 3) {
|
||||
ksft_print_msg(
|
||||
"Unexpected 'NSPid:' entry: %s",
|
||||
line);
|
||||
ns1 = ns2 = ns3 = 0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
fclose(f);
|
||||
free(line);
|
||||
close(pipe_2[0]);
|
||||
|
||||
/* Tell the clone3()'d child to finish. */
|
||||
write(pipe_2[1], &buf, 1);
|
||||
close(pipe_2[1]);
|
||||
|
||||
if (waitpid(ns_pid, &status, 0) < 0) {
|
||||
ksft_print_msg("Child returned %s\n", strerror(errno));
|
||||
ret = -errno;
|
||||
goto out;
|
||||
}
|
||||
|
||||
if (!WIFEXITED(status))
|
||||
ksft_test_result_fail("Child error\n");
|
||||
|
||||
if (WEXITSTATUS(status))
|
||||
/*
|
||||
* Update the number of total tests with the tests from the
|
||||
* child processes.
|
||||
*/
|
||||
ksft_cnt.ksft_pass = WEXITSTATUS(status);
|
||||
|
||||
if (ns3 == pid && ns2 == 42 && ns1 == 1)
|
||||
ksft_test_result_pass(
|
||||
"PIDs in all namespaces as expected (%d,%d,%d)\n",
|
||||
ns3, ns2, ns1);
|
||||
else
|
||||
ksft_test_result_fail(
|
||||
"PIDs in all namespaces not as expected (%d,%d,%d)\n",
|
||||
ns3, ns2, ns1);
|
||||
out:
|
||||
ret = 0;
|
||||
|
||||
return !ret ? ksft_exit_pass() : ksft_exit_fail();
|
||||
}
|
||||
Loading…
Reference in a new issue