From b030ffaedc1d638f232f6992406cdcdcc82d5602 Mon Sep 17 00:00:00 2001 From: Jann Horn Date: Thu, 6 Aug 2026 22:15:54 +0200 Subject: cred: clarify that task_struct::cred is only for the current task The `cred` field in task_struct is currently marked as __rcu, but that's not true: It can point to credentials from access_override_creds(), which do not get freed with RCU delay. What actually protects task_struct::cred is that accessing it is only permitted for the current task (except for setting up a child during fork() or tearing down a dead process). (There is currently code in Smack that violates this rule, but that's a bug and causes UAF, I have sent a separate fix for that.) Clarify this, remove the __rcu marker, and remove RCU helpers from all accesses to this field. Signed-off-by: Jann Horn Reviewed-by: Serge Hallyn [PM: style tweak in security_init(), applied fixup from JH] Signed-off-by: Paul Moore --- include/linux/cred.h | 17 +++++++++++------ include/linux/sched.h | 8 ++++++-- 2 files changed, 17 insertions(+), 8 deletions(-) (limited to 'include/linux') diff --git a/include/linux/cred.h b/include/linux/cred.h index 6ef1750c93e2..49c26af37349 100644 --- a/include/linux/cred.h +++ b/include/linux/cred.h @@ -180,12 +180,18 @@ static inline bool cap_ambient_invariant_ok(const struct cred *cred) static inline const struct cred *override_creds(const struct cred *override_cred) { - return rcu_replace_pointer(current->cred, override_cred, 1); + const struct cred *old = current->cred; + + current->cred = override_cred; + return old; } static inline const struct cred *revert_creds(const struct cred *revert_cred) { - return rcu_replace_pointer(current->cred, revert_cred, 1); + const struct cred *override_cred = current->cred; + + current->cred = revert_cred; + return override_cred; } DEFINE_CLASS(override_creds, @@ -293,11 +299,10 @@ DEFINE_FREE(put_cred, struct cred *, if (!IS_ERR_OR_NULL(_T)) put_cred(_T)) /** * current_cred - Access the current task's subjective credentials * - * Access the subjective credentials of the current task. RCU-safe, - * since nobody else can modify it. + * Access the subjective credentials of the current task. + * Nobody else can modify it. */ -#define current_cred() \ - rcu_dereference_protected(current->cred, 1) +#define current_cred() (current->cred) /** * current_real_cred - Access the current task's objective credentials diff --git a/include/linux/sched.h b/include/linux/sched.h index 8b3d47a325cc..50c7157fee82 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1172,8 +1172,12 @@ struct task_struct { /* Objective and real subjective task credentials (COW): */ const struct cred __rcu *real_cred; - /* Effective (overridable) subjective task credentials (COW): */ - const struct cred __rcu *cred; + /* + * Effective (overridable) subjective task credentials (COW). + * Only accessible for the current task and during task creation/freeing. + * This pointer is not managed by RCU! + */ + const struct cred *cred; #ifdef CONFIG_KEYS /* Cached requested key. */ -- cgit v1.2.3 From f675d2e9556963133b71df68739b771bf51c3a67 Mon Sep 17 00:00:00 2001 From: Christian Brauner Date: Sun, 26 Jul 2026 18:13:47 +0200 Subject: lsm: add LSM blob and hooks for namespaces MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All namespace types now share the same ns_common infrastructure. Extend this to include a security blob so LSMs can start managing namespaces uniformly without having to add one-off hooks or security fields to every individual namespace type. Add a ns_security pointer to ns_common and the corresponding lbs_ns blob size to lsm_blob_sizes. Allocation and freeing hooks are called from the common __ns_common_init() and __ns_common_free() paths so every namespace type gets covered in one go. All information about the namespace type and the appropriate casting helpers to get at the containing namespace are available via ns_common making it straightforward for LSMs to differentiate when they need to. A namespace_install hook is called from validate_ns() during setns(2) giving LSMs a chance to enforce policy on namespace transitions. The LSM check runs before ns->ops->install() so the security module can deny the operation before any type-specific installation effects. Individual namespace types can still have their own specialized security hooks when needed. This is just the common baseline that makes it easy to track and manage namespaces from the security side without requiring every namespace type to reinvent the wheel. Cc: Günther Noack Cc: Paul Moore Cc: Serge E. Hallyn Signed-off-by: Christian Brauner Link: https://lore.kernel.org/r/20260216-work-security-namespace-v1-1-075c28758e1f@kernel.org Signed-off-by: Mickaël Salaün [PM: subject tweak] Signed-off-by: Paul Moore --- include/linux/lsm_hook_defs.h | 3 ++ include/linux/lsm_hooks.h | 1 + include/linux/ns/ns_common_types.h | 3 ++ include/linux/security.h | 20 +++++++++++ kernel/nscommon.c | 14 ++++++++ kernel/nsproxy.c | 6 ++++ security/lsm_init.c | 2 ++ security/security.c | 72 ++++++++++++++++++++++++++++++++++++++ 8 files changed, 121 insertions(+) (limited to 'include/linux') diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h index 65c9609ec207..0da6c7e8f659 100644 --- a/include/linux/lsm_hook_defs.h +++ b/include/linux/lsm_hook_defs.h @@ -265,6 +265,9 @@ LSM_HOOK(int, -ENOSYS, task_prctl, int option, unsigned long arg2, LSM_HOOK(void, LSM_RET_VOID, task_to_inode, struct task_struct *p, struct inode *inode) LSM_HOOK(int, 0, userns_create, const struct cred *cred) +LSM_HOOK(int, 0, namespace_init, struct ns_common *ns) +LSM_HOOK(void, LSM_RET_VOID, namespace_free, struct ns_common *ns) +LSM_HOOK(int, 0, namespace_install, const struct nsset *nsset, struct ns_common *ns) LSM_HOOK(int, 0, ipc_permission, struct kern_ipc_perm *ipcp, short flag) LSM_HOOK(void, LSM_RET_VOID, ipc_getlsmprop, struct kern_ipc_perm *ipcp, struct lsm_prop *prop) diff --git a/include/linux/lsm_hooks.h b/include/linux/lsm_hooks.h index c4488c4a6d8a..13621e9e233e 100644 --- a/include/linux/lsm_hooks.h +++ b/include/linux/lsm_hooks.h @@ -112,6 +112,7 @@ struct lsm_blob_sizes { unsigned int lbs_ipc; unsigned int lbs_key; unsigned int lbs_msg_msg; + unsigned int lbs_ns; unsigned int lbs_perf_event; unsigned int lbs_task; unsigned int lbs_xattr_count; /* num xattr slots in new_xattrs array */ diff --git a/include/linux/ns/ns_common_types.h b/include/linux/ns/ns_common_types.h index ea45c54e4435..5cfe0ce3c881 100644 --- a/include/linux/ns/ns_common_types.h +++ b/include/linux/ns/ns_common_types.h @@ -116,6 +116,9 @@ struct ns_common { struct dentry *stashed; const struct proc_ns_operations *ops; unsigned int inum; +#ifdef CONFIG_SECURITY + void *ns_security; +#endif union { struct ns_tree; struct rcu_head ns_rcu; diff --git a/include/linux/security.h b/include/linux/security.h index 153e9043058f..bf002ed14ac8 100644 --- a/include/linux/security.h +++ b/include/linux/security.h @@ -67,6 +67,7 @@ enum fs_value_type; struct watch; struct watch_notification; struct lsm_ctx; +struct nsset; /* Default (no) options for the capable function */ #define CAP_OPT_NONE 0x0 @@ -80,6 +81,7 @@ struct lsm_ctx; struct ctl_table; struct audit_krule; +struct ns_common; struct user_namespace; struct timezone; @@ -540,6 +542,9 @@ int security_task_prctl(int option, unsigned long arg2, unsigned long arg3, unsigned long arg4, unsigned long arg5); void security_task_to_inode(struct task_struct *p, struct inode *inode); int security_create_user_ns(const struct cred *cred); +int security_namespace_init(struct ns_common *ns); +void security_namespace_free(struct ns_common *ns); +int security_namespace_install(const struct nsset *nsset, struct ns_common *ns); int security_ipc_permission(struct kern_ipc_perm *ipcp, short flag); void security_ipc_getlsmprop(struct kern_ipc_perm *ipcp, struct lsm_prop *prop); int security_msg_msg_alloc(struct msg_msg *msg); @@ -1431,6 +1436,21 @@ static inline int security_create_user_ns(const struct cred *cred) return 0; } +static inline int security_namespace_init(struct ns_common *ns) +{ + return 0; +} + +static inline void security_namespace_free(struct ns_common *ns) +{ +} + +static inline int security_namespace_install(const struct nsset *nsset, + struct ns_common *ns) +{ + return 0; +} + static inline int security_ipc_permission(struct kern_ipc_perm *ipcp, short flag) { diff --git a/kernel/nscommon.c b/kernel/nscommon.c index e6f623e1bc37..e72426bba29a 100644 --- a/kernel/nscommon.c +++ b/kernel/nscommon.c @@ -4,6 +4,7 @@ #include #include #include +#include #include #include @@ -59,6 +60,9 @@ int __ns_common_init(struct ns_common *ns, u32 ns_type, const struct proc_ns_ope refcount_set(&ns->__ns_ref, 1); ns->stashed = NULL; +#ifdef CONFIG_SECURITY + ns->ns_security = NULL; +#endif ns->ops = ops; ns->ns_id = 0; ns->ns_type = ns_type; @@ -77,6 +81,14 @@ int __ns_common_init(struct ns_common *ns, u32 ns_type, const struct proc_ns_ope ret = proc_alloc_inum(&ns->inum); if (ret) return ret; + + ret = security_namespace_init(ns); + if (ret) { + if (!inum) + proc_free_inum(ns->inum); + return ret; + } + /* * Tree ref starts at 0. It's incremented when namespace enters * active use (installed in nsproxy) and decremented when all @@ -91,6 +103,8 @@ int __ns_common_init(struct ns_common *ns, u32 ns_type, const struct proc_ns_ope void __ns_common_free(struct ns_common *ns) { + security_namespace_free(ns); + if (ns->inum > MNT_NS_INO_SPECIAL_MAX) proc_free_inum(ns->inum); } diff --git a/kernel/nsproxy.c b/kernel/nsproxy.c index d9d3d5973bf5..0f1b208d8eef 100644 --- a/kernel/nsproxy.c +++ b/kernel/nsproxy.c @@ -385,6 +385,12 @@ out: static inline int validate_ns(struct nsset *nsset, struct ns_common *ns) { + int ret; + + ret = security_namespace_install(nsset, ns); + if (ret) + return ret; + return ns->ops->install(nsset, ns); } diff --git a/security/lsm_init.c b/security/lsm_init.c index 04b18d06ba59..751da523bb42 100644 --- a/security/lsm_init.c +++ b/security/lsm_init.c @@ -303,6 +303,7 @@ static void __init lsm_prepare(struct lsm_info *lsm) lsm_blob_size_update(&blobs->lbs_ipc, &blob_sizes.lbs_ipc); lsm_blob_size_update(&blobs->lbs_key, &blob_sizes.lbs_key); lsm_blob_size_update(&blobs->lbs_msg_msg, &blob_sizes.lbs_msg_msg); + lsm_blob_size_update(&blobs->lbs_ns, &blob_sizes.lbs_ns); lsm_blob_size_update(&blobs->lbs_perf_event, &blob_sizes.lbs_perf_event); lsm_blob_size_update(&blobs->lbs_sock, &blob_sizes.lbs_sock); @@ -450,6 +451,7 @@ int __init security_init(void) lsm_pr("blob(ipc) size %d\n", blob_sizes.lbs_ipc); lsm_pr("blob(key) size %d\n", blob_sizes.lbs_key); lsm_pr("blob(msg_msg)_size %d\n", blob_sizes.lbs_msg_msg); + lsm_pr("blob(ns) size %d\n", blob_sizes.lbs_ns); lsm_pr("blob(sock) size %d\n", blob_sizes.lbs_sock); lsm_pr("blob(superblock) size %d\n", blob_sizes.lbs_superblock); lsm_pr("blob(perf_event) size %d\n", blob_sizes.lbs_perf_event); diff --git a/security/security.c b/security/security.c index 2ee276ab15c5..b284e24e3771 100644 --- a/security/security.c +++ b/security/security.c @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -381,6 +382,19 @@ static int lsm_superblock_alloc(struct super_block *sb) GFP_KERNEL); } +/** + * lsm_ns_alloc - allocate a composite namespace blob + * @ns: the namespace that needs a blob + * + * Allocate the namespace blob for all the modules + * + * Returns 0, or -ENOMEM if memory can't be allocated. + */ +static int lsm_ns_alloc(struct ns_common *ns) +{ + return lsm_blob_alloc(&ns->ns_security, blob_sizes.lbs_ns, GFP_KERNEL); +} + /** * lsm_fill_user_ctx - Fill a user space lsm_ctx structure * @uctx: a userspace LSM context to be filled @@ -3357,6 +3371,64 @@ int security_create_user_ns(const struct cred *cred) return call_int_hook(userns_create, cred); } +/** + * security_namespace_init() - Initialize LSM security data for a namespace + * @ns: the namespace being initialized + * + * Initialize the LSM security blob attached to the namespace. The namespace type + * is available via ns->ns_type, and the owning user namespace (if any) + * via ns->ops->owner(ns). + * + * Return: Returns 0 if successful, otherwise < 0 error code. + */ +int security_namespace_init(struct ns_common *ns) +{ + int rc; + + rc = lsm_ns_alloc(ns); + if (unlikely(rc)) + return rc; + + rc = call_int_hook(namespace_init, ns); + if (unlikely(rc)) + security_namespace_free(ns); + + return rc; +} + +/** + * security_namespace_free() - Release LSM security data from a namespace + * @ns: the namespace being freed + * + * Release security data attached to the namespace. Called before the + * namespace structure is freed. + */ +void security_namespace_free(struct ns_common *ns) +{ + if (!ns->ns_security) + return; + + call_void_hook(namespace_free, ns); + + kfree(ns->ns_security); + ns->ns_security = NULL; +} + +/** + * security_namespace_install() - Check permission to install a namespace + * @nsset: the target nsset being configured + * @ns: the namespace being installed + * + * Check permission before allowing a namespace to be installed into the + * process's set of namespaces via setns(2). + * + * Return: Returns 0 if permission is granted, otherwise < 0 error code. + */ +int security_namespace_install(const struct nsset *nsset, struct ns_common *ns) +{ + return call_int_hook(namespace_install, nsset, ns); +} + /** * security_ipc_permission() - Check if sysv ipc access is allowed * @ipcp: ipc permission structure -- cgit v1.2.3 From e26307d83048803274837abf33f65396a88e21ac Mon Sep 17 00:00:00 2001 From: Mickaël Salaün Date: Sun, 26 Jul 2026 18:13:48 +0200 Subject: lsm: add LSM_AUDIT_DATA_NS for namespace audit records MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a new LSM audit data type LSM_AUDIT_DATA_NS that logs namespace information in audit records. Two fields are provided: - namespace_type: the CLONE_NEW* flag identifying the namespace type, logged in hexadecimal. - namespace_id: the unique 64-bit namespace identifier, retrievable from userspace via NS_GET_ID or listns(2). Unlike the proc inode number (inum), namespace_id is never recycled. For namespace creation denials, namespace_id is 0 because the namespace does not exist yet. A new audit data type is needed because no existing LSM_AUDIT_DATA_* type carries namespace information. The closest alternatives (e.g. LSM_AUDIT_DATA_TASK or LSM_AUDIT_DATA_NONE with custom strings) would either lose the namespace type or require ad-hoc formatting that bypasses the structured audit data union. Cc: Günther Noack Cc: Paul Moore Reviewed-by: Christian Brauner Reviewed-by: Günther Noack Signed-off-by: Mickaël Salaün [PM: corrected audit fields in the description, subject tweaks] Signed-off-by: Paul Moore --- include/linux/lsm_audit.h | 5 +++++ security/lsm_audit.c | 4 ++++ 2 files changed, 9 insertions(+) (limited to 'include/linux') diff --git a/include/linux/lsm_audit.h b/include/linux/lsm_audit.h index 584db296e43b..526a8e7471c8 100644 --- a/include/linux/lsm_audit.h +++ b/include/linux/lsm_audit.h @@ -78,6 +78,7 @@ struct common_audit_data { #define LSM_AUDIT_DATA_NOTIFICATION 16 #define LSM_AUDIT_DATA_ANONINODE 17 #define LSM_AUDIT_DATA_NLMSGTYPE 18 +#define LSM_AUDIT_DATA_NS 19 union { struct path path; struct dentry *dentry; @@ -100,6 +101,10 @@ struct common_audit_data { int reason; const char *anonclass; u16 nlmsg_type; + struct { + u32 ns_type; + u64 ns_id; + } ns; } u; /* this union contains LSM specific data */ union { diff --git a/security/lsm_audit.c b/security/lsm_audit.c index 737f5a263a8f..404ccbbbf94c 100644 --- a/security/lsm_audit.c +++ b/security/lsm_audit.c @@ -403,6 +403,10 @@ void audit_log_lsm_data(struct audit_buffer *ab, case LSM_AUDIT_DATA_NLMSGTYPE: audit_log_format(ab, " nl-msgtype=%hu", a->u.nlmsg_type); break; + case LSM_AUDIT_DATA_NS: + audit_log_format(ab, " namespace_type=0x%x namespace_id=%llu", + a->u.ns.ns_type, a->u.ns.ns_id); + break; } /* switch (a->type) */ } -- cgit v1.2.3