PR_CAPBSET_DROP only affects the calling thread. Process-wide capability
dropping requires user space to loop over every thread and every
capability, which is highly expensive. For instance, long-lived
multi-threaded processes (like gVisor's sentry) spend milliseconds
trimming the bounding set because the runtime has to coordinate
and signal every thread.

Add PR_CAPBSET_DROP_MASK, an opt-in prctl that removes a set of
capabilities from the entire thread group's bounding set in a single call.
The capabilities are specified via a 64-bit mask across arg2 (low 32 bits)
and arg3 (high 32 bits).

The drop is recorded in a per-thread-group mask,
signal_struct::cap_bset_pending, under sighand->siglock. This pending drop
is dynamically folded into the bounding set in critical paths:
cap_capset(), PR_CAPBSET_READ(), and cap_bprm_creds_from_file().
Running, concurrently created, or future threads within the group will
all immediately observe the drop. A forked child inherits this mask
in cap_bset_drop_fork() under siglock to close the race window
against clone().

The operation is drop-only, so concurrent callers commute and a single
atomic OR is the linearization point.  It performs no per-thread allocation
and only the caller's replacement cred can fail, reported synchronously as
-ENOMEM before anything changes.

In an arm64 KVM guest, trimming 41 capabilities of a multi-threaded Go
process takes ~8.5-23.6ms with the per-thread PR_CAPBSET_DROP loop and
~11-14us with PR_CAPBSET_DROP_MASK, independent of the thread count.

Signed-off-by: Jinjie Ruan <[email protected]>
---
 fs/proc/array.c              |  3 +-
 include/linux/capability.h   |  5 ++
 include/linux/sched/signal.h |  8 ++++
 include/uapi/linux/prctl.h   |  1 +
 kernel/fork.c                |  1 +
 security/commoncap.c         | 91 ++++++++++++++++++++++++++++++++++--
 6 files changed, 105 insertions(+), 4 deletions(-)

diff --git a/fs/proc/array.c b/fs/proc/array.c
index f6f75d206762..f1cde26d079c 100644
--- a/fs/proc/array.c
+++ b/fs/proc/array.c
@@ -63,6 +63,7 @@
 #include <linux/tty.h>
 #include <linux/string.h>
 #include <linux/mman.h>
+#include <linux/capability.h>
 #include <linux/sched/mm.h>
 #include <linux/sched/numa_balancing.h>
 #include <linux/sched/task_stack.h>
@@ -318,7 +319,7 @@ static inline void task_cap(struct seq_file *m, struct 
task_struct *p)
        cap_inheritable = cred->cap_inheritable;
        cap_permitted   = cred->cap_permitted;
        cap_effective   = cred->cap_effective;
-       cap_bset        = cred->cap_bset;
+       cap_bset        = cap_bset_effective(p, cred);
        cap_ambient     = cred->cap_ambient;
        rcu_read_unlock();
 
diff --git a/include/linux/capability.h b/include/linux/capability.h
index 7921a0b3b04a..6413c7fedb69 100644
--- a/include/linux/capability.h
+++ b/include/linux/capability.h
@@ -38,6 +38,7 @@ struct file;
 struct inode;
 struct dentry;
 struct task_struct;
+struct cred;
 struct user_namespace;
 struct mnt_idmap;
 
@@ -197,6 +198,10 @@ bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap,
                              const struct inode *inode, int cap);
 extern bool file_ns_capable(const struct file *file, struct user_namespace 
*ns, int cap);
 extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace 
*ns);
+extern kernel_cap_t cap_bset_effective(const struct task_struct *task,
+                                      const struct cred *cred);
+extern void cap_bset_drop_fork(struct task_struct *p);
+
 static inline bool perfmon_capable(void)
 {
        return capable(CAP_PERFMON) || capable(CAP_SYS_ADMIN);
diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h
index d45a5476b97d..bda6f18b65b9 100644
--- a/include/linux/sched/signal.h
+++ b/include/linux/sched/signal.h
@@ -98,6 +98,14 @@ struct signal_struct {
        int                     quick_threads;
        struct list_head        thread_head;
 
+       /*
+        * Capabilities being removed from the bounding set of every thread in
+        * this group by PR_CAPBSET_DROP_MASK.  Read on capability-transition
+        * paths without sighand->siglock, hence atomic64 (a 64-bit value would
+        * otherwise tear on 32-bit).
+        */
+       atomic64_t              cap_bset_pending;
+
        wait_queue_head_t       wait_chldexit;  /* for wait4() */
 
        /* current thread group signal load-balancing target: */
diff --git a/include/uapi/linux/prctl.h b/include/uapi/linux/prctl.h
index b6ec6f693719..85ac71897070 100644
--- a/include/uapi/linux/prctl.h
+++ b/include/uapi/linux/prctl.h
@@ -70,6 +70,7 @@
 /* Get/set the capability bounding set (as per security/commoncap.c) */
 #define PR_CAPBSET_READ 23
 #define PR_CAPBSET_DROP 24
+#define PR_CAPBSET_DROP_MASK 82
 
 /* Get/set the process' ability to use the timestamp counter instruction */
 #define PR_GET_TSC 25
diff --git a/kernel/fork.c b/kernel/fork.c
index 5ef413368912..bf671a85d39d 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -2504,6 +2504,7 @@ __latent_entropy struct task_struct *copy_process(
         * before holding sighand lock.
         */
        copy_seccomp(p);
+       cap_bset_drop_fork(p);
 
        if (clone_flags & CLONE_NNP)
                task_set_no_new_privs(p);
diff --git a/security/commoncap.c b/security/commoncap.c
index 3399535808fe..3bec15d43731 100644
--- a/security/commoncap.c
+++ b/security/commoncap.c
@@ -3,6 +3,7 @@
  */
 
 #include <linux/capability.h>
+#include <linux/cred.h>
 #include <linux/audit.h>
 #include <linux/init.h>
 #include <linux/kernel.h>
@@ -19,6 +20,7 @@
 #include <linux/hugetlb.h>
 #include <linux/mount.h>
 #include <linux/sched.h>
+#include <linux/sched/signal.h>
 #include <linux/prctl.h>
 #include <linux/securebits.h>
 #include <linux/user_namespace.h>
@@ -30,6 +32,25 @@
 #define CREATE_TRACE_POINTS
 #include <trace/events/capability.h>
 
+/**
+ * Effective bounding set of @cred in @task's thread group
+ * @task: task whose thread group's pending drop applies
+ * @cred: credentials to read the bounding set from
+ *
+ * A drop recorded by PR_CAPBSET_DROP_MASK is authoritative on the thread group
+ * and may not have been materialized into every thread's cred yet, so the
+ * effective bounding set is the cred's own set minus the group's pending drop.
+ */
+kernel_cap_t cap_bset_effective(const struct task_struct *task,
+                               const struct cred *cred)
+{
+       kernel_cap_t pending = {
+               .val = atomic64_read(&task->signal->cap_bset_pending),
+       };
+
+       return cap_drop(cred->cap_bset, pending);
+}
+
 /*
  * If a non-root user executes a setuid-root binary in
  * !secure(SECURE_NOROOT) mode, then we raise capabilities.
@@ -284,8 +305,8 @@ int cap_capset(struct cred *new,
 
        if (!cap_issubset(*inheritable,
                          cap_combine(old->cap_inheritable,
-                                     old->cap_bset)))
                /* no new pI capabilities outside bounding set */
+                                     cap_bset_effective(current, old))))
                return -EPERM;
 
        /* verify restrictions on target's new Permitted set */
@@ -849,7 +870,7 @@ static void handle_privileged_root(struct linux_binprm 
*bprm, bool has_fcap,
         */
        if (__is_eff(root_uid, new) || __is_real(root_uid, new)) {
                /* pP' = (cap_bset & ~0) | (pI & ~0) */
-               new->cap_permitted = cap_combine(old->cap_bset,
+               new->cap_permitted = cap_combine(new->cap_bset,
                                                 old->cap_inheritable);
        }
        /*
@@ -925,6 +946,8 @@ int cap_bprm_creds_from_file(struct linux_binprm *bprm, 
const struct file *file)
        int ret;
        kuid_t root_uid;
 
+       new->cap_bset = cap_bset_effective(current, new);
+
        if (WARN_ON(!cap_ambient_invariant_ok(old)))
                return -EPERM;
 
@@ -1283,6 +1306,63 @@ static int cap_prctl_drop(unsigned long cap)
        return commit_creds(new);
 }
 
+static int cap_bset_drop_process(kernel_cap_t mask)
+{
+       kernel_cap_t pending;
+       struct cred *new;
+
+       new = prepare_creds();
+       if (!new)
+               return -ENOMEM;
+
+       /*
+        * Record the drop before committing the caller's cred, so that a
+        * thread created from now on is guaranteed to observe it.  Apply the
+        * current union, not just this call's mask, to the caller's cred.
+        */
+       spin_lock_irq(&current->sighand->siglock);
+       atomic64_or(mask.val, &current->signal->cap_bset_pending);
+       pending.val = atomic64_read(&current->signal->cap_bset_pending);
+       spin_unlock_irq(&current->sighand->siglock);
+
+       new->cap_bset = cap_drop(new->cap_bset, pending);
+       commit_creds(new);
+
+       return 0;
+}
+
+/*
+ * Propagate a pending process-wide bounding-set drop to @p, a task being
+ * created by the current thread.  Threads sharing the group read
+ * current->signal->cap_bset_pending directly; a forked child gets its own
+ * signal_struct and must carry the mask itself.  Called under
+ * current->sighand->siglock, which serializes it with cap_bset_drop_process().
+ */
+void cap_bset_drop_fork(struct task_struct *p)
+{
+       kernel_cap_t mask = {
+               .val = atomic64_read(&current->signal->cap_bset_pending),
+       };
+
+       if (cap_isclear(mask) || p->signal == current->signal)
+               return;
+
+       atomic64_or(mask.val, &p->signal->cap_bset_pending);
+}
+
+static int cap_prctl_drop_mask(unsigned long low, unsigned long high)
+{
+       kernel_cap_t mask = mk_kernel_cap((u32)low, (u32)high);
+
+       if (cap_isclear(mask))
+               return 0;
+
+       if (!ns_capable(current_user_ns(), CAP_SETPCAP))
+               return -EPERM;
+
+       return cap_bset_drop_process(mask);
+}
+
 /**
  * cap_task_prctl - Implement process control functions for this security 
module
  * @option: The process control function requested
@@ -1308,11 +1388,16 @@ int cap_task_prctl(int option, unsigned long arg2, 
unsigned long arg3,
        case PR_CAPBSET_READ:
                if (!cap_valid(arg2))
                        return -EINVAL;
-               return !!cap_raised(old->cap_bset, arg2);
+               return !!cap_raised(cap_bset_effective(current, old), arg2);
 
        case PR_CAPBSET_DROP:
                return cap_prctl_drop(arg2);
 
+       case PR_CAPBSET_DROP_MASK:
+               if (arg4 || arg5)
+                       return -EINVAL;
+               return cap_prctl_drop_mask(arg2, arg3);
+
        /*
         * The next four prctl's remain to assist with transitioning a
         * system from legacy UID=0 based privilege (when filesystem
-- 
2.34.1


Reply via email to