[RFC PATCH v2 2/3] security: Add PR_CAPBSET_DROP_MASK for process-wide bounding-set drops

Jinjie Ruan ruanjinjie at huawei.com
Tue Sep 29 13:01:59 UTC 2026


PR_CAPBSET_DROP only affects the calling thread. Process-wide capability
dropping requires user space to loop over every thread and every
capability, which is highly expensive. For instance, long-lived
multi-threaded processes (like gVisor's sentry) spend milliseconds
trimming the bounding set because the runtime has to coordinate
and signal every thread.

Add PR_CAPBSET_DROP_MASK, an opt-in prctl that removes a set of
capabilities from the entire thread group's bounding set in a single call.
The capabilities are specified via a 64-bit mask across arg2 (low 32 bits)
and arg3 (high 32 bits).

The drop is recorded in a per-thread-group mask,
signal_struct::cap_bset_pending, under sighand->siglock. This pending drop
is dynamically folded into the bounding set in critical paths:
cap_capset(), PR_CAPBSET_READ(), and cap_bprm_creds_from_file().
Running, concurrently created, or future threads within the group will
all immediately observe the drop. A forked child inherits this mask
in cap_bset_drop_fork() under siglock to close the race window
against clone().

The operation is drop-only, so concurrent callers commute and a single
atomic OR is the linearization point.  It performs no per-thread allocation
and only the caller's replacement cred can fail, reported synchronously as
-ENOMEM before anything changes.

In an arm64 KVM guest, trimming 41 capabilities of a multi-threaded Go
process takes ~8.5-23.6ms with the per-thread PR_CAPBSET_DROP loop and
~11-14us with PR_CAPBSET_DROP_MASK, independent of the thread count.

Signed-off-by: Jinjie Ruan <ruanjinjie at huawei.com>
---
 fs/proc/array.c              |  3 +-
 include/linux/capability.h   |  5 ++
 include/linux/sched/signal.h |  8 ++++
 include/uapi/linux/prctl.h   |  1 +
 kernel/fork.c                |  1 +
 security/commoncap.c         | 91 ++++++++++++++++++++++++++++++++++--
 6 files changed, 105 insertions(+), 4 deletions(-)

diff --git a/fs/proc/array.c b/fs/proc/array.c
index f6f75d206762..f1cde26d079c 100644
--- a/fs/proc/array.c
+++ b/fs/proc/array.c
@@ -63,6 +63,7 @@
 #include <linux/tty.h>
 #include <linux/string.h>
 #include <linux/mman.h>
+#include <linux/capability.h>
 #include <linux/sched/mm.h>
 #include <linux/sched/numa_balancing.h>
 #include <linux/sched/task_stack.h>
@@ -318,7 +319,7 @@ static inline void task_cap(struct seq_file *m, struct task_struct *p)
 	cap_inheritable	= cred->cap_inheritable;
 	cap_permitted	= cred->cap_permitted;
 	cap_effective	= cred->cap_effective;
-	cap_bset	= cred->cap_bset;
+	cap_bset	= cap_bset_effective(p, cred);
 	cap_ambient	= cred->cap_ambient;
 	rcu_read_unlock();
 
diff --git a/include/linux/capability.h b/include/linux/capability.h
index 7921a0b3b04a..6413c7fedb69 100644
--- a/include/linux/capability.h
+++ b/include/linux/capability.h
@@ -38,6 +38,7 @@ struct file;
 struct inode;
 struct dentry;
 struct task_struct;
+struct cred;
 struct user_namespace;
 struct mnt_idmap;
 
@@ -197,6 +198,10 @@ bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap,
 			      const struct inode *inode, int cap);
 extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap);
 extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns);
+extern kernel_cap_t cap_bset_effective(const struct task_struct *task,
+				       const struct cred *cred);
+extern void cap_bset_drop_fork(struct task_struct *p);
+
 static inline bool perfmon_capable(void)
 {
 	return capable(CAP_PERFMON) || capable(CAP_SYS_ADMIN);
diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h
index d45a5476b97d..bda6f18b65b9 100644
--- a/include/linux/sched/signal.h
+++ b/include/linux/sched/signal.h
@@ -98,6 +98,14 @@ struct signal_struct {
 	int			quick_threads;
 	struct list_head	thread_head;
 
+	/*
+	 * Capabilities being removed from the bounding set of every thread in
+	 * this group by PR_CAPBSET_DROP_MASK.  Read on capability-transition
+	 * paths without sighand->siglock, hence atomic64 (a 64-bit value would
+	 * otherwise tear on 32-bit).
+	 */
+	atomic64_t		cap_bset_pending;
+
 	wait_queue_head_t	wait_chldexit;	/* for wait4() */
 
 	/* current thread group signal load-balancing target: */
diff --git a/include/uapi/linux/prctl.h b/include/uapi/linux/prctl.h
index b6ec6f693719..85ac71897070 100644
--- a/include/uapi/linux/prctl.h
+++ b/include/uapi/linux/prctl.h
@@ -70,6 +70,7 @@
 /* Get/set the capability bounding set (as per security/commoncap.c) */
 #define PR_CAPBSET_READ 23
 #define PR_CAPBSET_DROP 24
+#define PR_CAPBSET_DROP_MASK 82
 
 /* Get/set the process' ability to use the timestamp counter instruction */
 #define PR_GET_TSC 25
diff --git a/kernel/fork.c b/kernel/fork.c
index 5ef413368912..bf671a85d39d 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -2504,6 +2504,7 @@ __latent_entropy struct task_struct *copy_process(
 	 * before holding sighand lock.
 	 */
 	copy_seccomp(p);
+	cap_bset_drop_fork(p);
 
 	if (clone_flags & CLONE_NNP)
 		task_set_no_new_privs(p);
diff --git a/security/commoncap.c b/security/commoncap.c
index 3399535808fe..3bec15d43731 100644
--- a/security/commoncap.c
+++ b/security/commoncap.c
@@ -3,6 +3,7 @@
  */
 
 #include <linux/capability.h>
+#include <linux/cred.h>
 #include <linux/audit.h>
 #include <linux/init.h>
 #include <linux/kernel.h>
@@ -19,6 +20,7 @@
 #include <linux/hugetlb.h>
 #include <linux/mount.h>
 #include <linux/sched.h>
+#include <linux/sched/signal.h>
 #include <linux/prctl.h>
 #include <linux/securebits.h>
 #include <linux/user_namespace.h>
@@ -30,6 +32,25 @@
 #define CREATE_TRACE_POINTS
 #include <trace/events/capability.h>
 
+/**
+ * Effective bounding set of @cred in @task's thread group
+ * @task: task whose thread group's pending drop applies
+ * @cred: credentials to read the bounding set from
+ *
+ * A drop recorded by PR_CAPBSET_DROP_MASK is authoritative on the thread group
+ * and may not have been materialized into every thread's cred yet, so the
+ * effective bounding set is the cred's own set minus the group's pending drop.
+ */
+kernel_cap_t cap_bset_effective(const struct task_struct *task,
+				const struct cred *cred)
+{
+	kernel_cap_t pending = {
+		.val = atomic64_read(&task->signal->cap_bset_pending),
+	};
+
+	return cap_drop(cred->cap_bset, pending);
+}
+
 /*
  * If a non-root user executes a setuid-root binary in
  * !secure(SECURE_NOROOT) mode, then we raise capabilities.
@@ -284,8 +305,8 @@ int cap_capset(struct cred *new,
 
 	if (!cap_issubset(*inheritable,
 			  cap_combine(old->cap_inheritable,
-				      old->cap_bset)))
 		/* no new pI capabilities outside bounding set */
+				      cap_bset_effective(current, old))))
 		return -EPERM;
 
 	/* verify restrictions on target's new Permitted set */
@@ -849,7 +870,7 @@ static void handle_privileged_root(struct linux_binprm *bprm, bool has_fcap,
 	 */
 	if (__is_eff(root_uid, new) || __is_real(root_uid, new)) {
 		/* pP' = (cap_bset & ~0) | (pI & ~0) */
-		new->cap_permitted = cap_combine(old->cap_bset,
+		new->cap_permitted = cap_combine(new->cap_bset,
 						 old->cap_inheritable);
 	}
 	/*
@@ -925,6 +946,8 @@ int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file)
 	int ret;
 	kuid_t root_uid;
 
+	new->cap_bset = cap_bset_effective(current, new);
+
 	if (WARN_ON(!cap_ambient_invariant_ok(old)))
 		return -EPERM;
 
@@ -1283,6 +1306,63 @@ static int cap_prctl_drop(unsigned long cap)
 	return commit_creds(new);
 }
 
+static int cap_bset_drop_process(kernel_cap_t mask)
+{
+	kernel_cap_t pending;
+	struct cred *new;
+
+	new = prepare_creds();
+	if (!new)
+		return -ENOMEM;
+
+	/*
+	 * Record the drop before committing the caller's cred, so that a
+	 * thread created from now on is guaranteed to observe it.  Apply the
+	 * current union, not just this call's mask, to the caller's cred.
+	 */
+	spin_lock_irq(&current->sighand->siglock);
+	atomic64_or(mask.val, &current->signal->cap_bset_pending);
+	pending.val = atomic64_read(&current->signal->cap_bset_pending);
+	spin_unlock_irq(&current->sighand->siglock);
+
+	new->cap_bset = cap_drop(new->cap_bset, pending);
+	commit_creds(new);
+
+	return 0;
+}
+
+/*
+ * Propagate a pending process-wide bounding-set drop to @p, a task being
+ * created by the current thread.  Threads sharing the group read
+ * current->signal->cap_bset_pending directly; a forked child gets its own
+ * signal_struct and must carry the mask itself.  Called under
+ * current->sighand->siglock, which serializes it with cap_bset_drop_process().
+ */
+void cap_bset_drop_fork(struct task_struct *p)
+{
+	kernel_cap_t mask = {
+		.val = atomic64_read(&current->signal->cap_bset_pending),
+	};
+
+	if (cap_isclear(mask) || p->signal == current->signal)
+		return;
+
+	atomic64_or(mask.val, &p->signal->cap_bset_pending);
+}
+
+static int cap_prctl_drop_mask(unsigned long low, unsigned long high)
+{
+	kernel_cap_t mask = mk_kernel_cap((u32)low, (u32)high);
+
+	if (cap_isclear(mask))
+		return 0;
+
+	if (!ns_capable(current_user_ns(), CAP_SETPCAP))
+		return -EPERM;
+
+	return cap_bset_drop_process(mask);
+}
+
 /**
  * cap_task_prctl - Implement process control functions for this security module
  * @option: The process control function requested
@@ -1308,11 +1388,16 @@ int cap_task_prctl(int option, unsigned long arg2, unsigned long arg3,
 	case PR_CAPBSET_READ:
 		if (!cap_valid(arg2))
 			return -EINVAL;
-		return !!cap_raised(old->cap_bset, arg2);
+		return !!cap_raised(cap_bset_effective(current, old), arg2);
 
 	case PR_CAPBSET_DROP:
 		return cap_prctl_drop(arg2);
 
+	case PR_CAPBSET_DROP_MASK:
+		if (arg4 || arg5)
+			return -EINVAL;
+		return cap_prctl_drop_mask(arg2, arg3);
+
 	/*
 	 * The next four prctl's remain to assist with transitioning a
 	 * system from legacy UID=0 based privilege (when filesystem
-- 
2.34.1




More information about the Linux-security-module-archive mailing list