diff options
519 files changed, 12572 insertions, 5091 deletions
@@ -57,7 +57,7 @@ S: Longford, Ireland S: Sydney, Australia N: Tigran A. Aivazian -E: tigran@aivazian.fsnet.co.uk +E: aivazian.tigran@gmail.com W: http://www.moses.uklinux.net/patches D: BFS filesystem D: Intel IA32 CPU microcode update support diff --git a/Documentation/admin-guide/binfmt-misc.rst b/Documentation/admin-guide/binfmt-misc.rst index d26b63a27c25..9e84b877d06d 100644 --- a/Documentation/admin-guide/binfmt-misc.rst +++ b/Documentation/admin-guide/binfmt-misc.rst @@ -19,6 +19,9 @@ To actually register a new binary type, you have to set up a string looking like ``:name:type:offset:magic:mask:interpreter:flags`` (where you can choose the ``:`` upon your needs) and echo it to ``/proc/sys/fs/binfmt_misc/register``. +The first character of the string is its field delimiter and can be any +ASCII punctuation character other than the backslash ``\``. + Here is what the fields mean: - ``name`` diff --git a/Documentation/filesystems/befs.rst b/Documentation/filesystems/befs.rst index a22f603b2938..c1dbfe9c95d2 100644 --- a/Documentation/filesystems/befs.rst +++ b/Documentation/filesystems/befs.rst @@ -44,8 +44,8 @@ implementation. Which is it, BFS or BEFS? ========================= Be, Inc said, "BeOS Filesystem is officially called BFS, not BeFS". -But Unixware Boot Filesystem is called bfs, too. And they are already in -the kernel. Because of this naming conflict, on Linux the BeOS +But the UnixWare Boot Filesystem is called bfs, too, and it was already +in the kernel. Because of this naming conflict, on Linux the BeOS filesystem is called befs. How to Install diff --git a/Documentation/filesystems/bfs.rst b/Documentation/filesystems/bfs.rst deleted file mode 100644 index ce14b9018807..000000000000 --- a/Documentation/filesystems/bfs.rst +++ /dev/null @@ -1,60 +0,0 @@ -.. SPDX-License-Identifier: GPL-2.0 - -======================== -BFS Filesystem for Linux -======================== - -The BFS filesystem is used by SCO UnixWare OS for the /stand slice, which -usually contains the kernel image and a few other files required for the -boot process. - -In order to access /stand partition under Linux you obviously need to -know the partition number and the kernel must support UnixWare disk slices -(CONFIG_UNIXWARE_DISKLABEL config option). However BFS support does not -depend on having UnixWare disklabel support because one can also mount -BFS filesystem via loopback:: - - # losetup /dev/loop0 stand.img - # mount -t bfs /dev/loop0 /mnt/stand - -where stand.img is a file containing the image of BFS filesystem. -When you have finished using it and umounted you need to also deallocate -/dev/loop0 device by:: - - # losetup -d /dev/loop0 - -You can simplify mounting by just typing:: - - # mount -t bfs -o loop stand.img /mnt/stand - -this will allocate the first available loopback device (and load loop.o -kernel module if necessary) automatically. If the loopback driver is not -loaded automatically, make sure that you have compiled the module and -that modprobe is functioning. Beware that umount will not deallocate -/dev/loopN device if /etc/mtab file on your system is a symbolic link to -/proc/mounts. You will need to do it manually using "-d" switch of -losetup(8). Read losetup(8) manpage for more info. - -To create the BFS image under UnixWare you need to find out first which -slice contains it. The command prtvtoc(1M) is your friend:: - - # prtvtoc /dev/rdsk/c0b0t0d0s0 - -(assuming your root disk is on target=0, lun=0, bus=0, controller=0). Then you -look for the slice with tag "STAND", which is usually slice 10. With this -information you can use dd(1) to create the BFS image:: - - # umount /stand - # dd if=/dev/rdsk/c0b0t0d0sa of=stand.img bs=512 - -Just in case, you can verify that you have done the right thing by checking -the magic number:: - - # od -Ad -tx4 stand.img | more - -The first 4 bytes should be 0x1badface. - -If you have any patches, questions or suggestions regarding this BFS -implementation please contact the author: - -Tigran Aivazian <aivazian.tigran@gmail.com> diff --git a/Documentation/filesystems/index.rst b/Documentation/filesystems/index.rst index fbd55915a318..1100130ccf0a 100644 --- a/Documentation/filesystems/index.rst +++ b/Documentation/filesystems/index.rst @@ -75,7 +75,6 @@ Documentation for filesystem implementations. autofs autofs-mount-control befs - bfs btrfs ceph coda diff --git a/Documentation/filesystems/locking.rst b/Documentation/filesystems/locking.rst index 844d65eb47a5..6330653287d5 100644 --- a/Documentation/filesystems/locking.rst +++ b/Documentation/filesystems/locking.rst @@ -61,23 +61,23 @@ inode_operations prototypes:: - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t); + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *,umode_t); struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t); + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *,const char *); + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *,struct dentry *,umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *,umode_t,dev_t); + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); int (*readlink) (struct dentry *, char __user *,int); const char *(*get_link) (struct dentry *, struct inode *, struct delayed_call *); void (*truncate) (struct inode *); - int (*permission) (struct mnt_idmap *, struct inode *, int, unsigned int); + int (*permission) (const struct mnt_idmap *, struct inode *, int, unsigned int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, u64 len); void (*update_time)(struct inode *inode, enum fs_update_time type, @@ -86,12 +86,12 @@ prototypes:: int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); - struct posix_acl * (*get_acl)(struct mnt_idmap *, struct dentry *, int); + struct posix_acl * (*get_acl)(const struct mnt_idmap *, struct dentry *, int); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); locking rules: @@ -148,7 +148,7 @@ prototypes:: struct inode *inode, const char *name, void *buffer, size_t size); int (*set)(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags); diff --git a/Documentation/filesystems/netfs_library.rst b/Documentation/filesystems/netfs_library.rst index ddd799df6ce3..0c9786ffe192 100644 --- a/Documentation/filesystems/netfs_library.rst +++ b/Documentation/filesystems/netfs_library.rst @@ -195,7 +195,7 @@ structure is defined:: struct inode inode; const struct netfs_request_ops *ops; struct fscache_cookie * cache; - loff_t remote_i_size; + loff_t _remote_i_size; unsigned long flags; ... }; @@ -229,11 +229,14 @@ filesystem: Local caching cookie, or NULL if no caching is enabled. This field does not exist if fscache is disabled. - * ``remote_i_size`` + * ``_remote_i_size`` The size of the file on the server. This differs from inode->i_size if local modifications have been made but not yet written back. + Use netfs_read_remote_i_size() and netfs_write_remote_i_size() to access + this field. Hold inode->i_lock when writing it. + * ``flags`` A set of flags, some of which the filesystem might be interested in: diff --git a/Documentation/filesystems/porting.rst b/Documentation/filesystems/porting.rst index 60880eb0c49d..e666edab789f 100644 --- a/Documentation/filesystems/porting.rst +++ b/Documentation/filesystems/porting.rst @@ -348,7 +348,7 @@ simply of return 1. Note that all actual eviction work is done by caller after As before, clear_inode() must be called exactly once on each call of ->evict_inode() (as it used to be for each call of ->delete_inode()). Unlike before, if you are using inode-associated metadata buffers (i.e. -mark_buffer_dirty_inode()), it's your responsibility to call +mmb_mark_buffer_dirty()), it's your responsibility to call invalidate_inode_buffers() before clear_inode(). NOTE: checking i_nlink in the beginning of ->write_inode() and bailing out @@ -1203,16 +1203,16 @@ will fail-safe. --- -** mandatory** +**mandatory** lookup_one(), lookup_one_unlocked(), lookup_one_positive_unlocked() now take a qstr instead of a name and len. These, not the "one_len" versions, should be used whenever accessing a filesystem from outside -that filesysmtem, through a mount point - which will have a mnt_idmap. +that filesystem, through a mount point - which will have a mnt_idmap. --- -** mandatory** +**mandatory** Functions try_lookup_one_len(), lookup_one_len(), lookup_one_len_unlocked() and lookup_positive_unlocked() have been @@ -1229,7 +1229,7 @@ already been performed such as after vfs_path_parent_lookup() --- -** mandatory** +**mandatory** d_hash_and_lookup() is no longer exported or available outside the VFS. Use try_lookup_noperm() instead. This adds name validation and takes @@ -1370,7 +1370,7 @@ similar. --- -** mandatory** +**mandatory** lock_rename(), lock_rename_child(), unlock_rename() are no longer available. Use start_renaming() or similar. @@ -1409,3 +1409,16 @@ use only if you have no alternative. The .create inode_operation no longer receives the 'excl' arg. It must always assume the file does not already exist. If the filesystem needs to be involved in non-exclusive create, it should provide atomic_open. + +--- + +**mandatory** + +All struct mnt_idmap pointers handed to filesystems are const now. +->create(), ->mkdir(), ->mknod(), ->symlink(), ->rename(), ->setattr(), +->getattr(), ->permission(), ->tmpfile(), ->get_acl(), ->set_acl() and +->fileattr_set() as well as the xattr ->set() handler and the vfs_*() +helpers take a const struct mnt_idmap *. mnt_idmap() and file_mnt_idmap() +return one. The idmapping is immutable so nothing should have modified it +anyway. References are taken and dropped via mnt_idmap_get() and +mnt_idmap_put() as before, both accept a const pointer. diff --git a/Documentation/filesystems/proc.rst b/Documentation/filesystems/proc.rst index c102b62023cd..fc59c98acca1 100644 --- a/Documentation/filesystems/proc.rst +++ b/Documentation/filesystems/proc.rst @@ -1963,6 +1963,10 @@ For example:: $ echo 0x7 > /proc/self/coredump_filter $ ./some_program +If the coredump socket protocol is used a coredump server can select memory +types to include dynamically. See COREDUMP_MEMORY_TYPES in +include/uapi/linux/coredump.h. + 3.5 /proc/<pid>/mountinfo - Information about mounts -------------------------------------------------------- diff --git a/Documentation/filesystems/sharedsubtree.rst b/Documentation/filesystems/sharedsubtree.rst index 8b7dc9159083..8bae6d9a7e04 100644 --- a/Documentation/filesystems/sharedsubtree.rst +++ b/Documentation/filesystems/sharedsubtree.rst @@ -564,8 +564,8 @@ f) Unmount semantics where 'A' is a mount mounted on mount 'B' at dentry 'b'. If mount 'B' is shared, then all most-recently-mounted mounts at dentry - 'b' on mounts that receive propagation from mount 'B' and does not have - sub-mounts within them are unmounted. + 'b' on mounts that receive propagation from mount 'B' are unmounted as + well, if every mount below them is also unmounted. Example: Let's say 'B1', 'B2', 'B3' are shared mounts that propagate to each other. @@ -584,10 +584,18 @@ f) Unmount semantics So all 'C1', 'C2' and 'C3' should be unmounted. - If any of 'C2' or 'C3' has some child mounts, then that mount is not - unmounted, but all other mounts are unmounted. However if 'C1' is told - to be unmounted and 'C1' has some sub-mounts, the umount operation is - failed entirely. + If any of 'C2' or 'C3' has a child mount that cannot be unmounted + then that mount is not unmounted. But all other mounts are unmounted. + A child mount that is itself unmounted by the same unmount + propagation does not keep its parent mounted. However if 'C1' is + supposed to be unmounted and 'C1' has some sub-mounts, the unmount + fails. + + A lazy umount (MNT_DETACH) takes a whole tree. Every mount of the + tree then propagates its unmount from its own parent as described + above, so the mounts that receive propagation lose the corresponding + trees as well. Documentation/filesystems/propagate_umount.txt has the + precise rules, including the ones for locked mounts. g) Clone Namespace diff --git a/Documentation/filesystems/squashfs.rst b/Documentation/filesystems/squashfs.rst index 45653b3228f9..d6397961d16b 100644 --- a/Documentation/filesystems/squashfs.rst +++ b/Documentation/filesystems/squashfs.rst @@ -177,9 +177,9 @@ or if the compressed block was larger than the uncompressed block. Inodes are packed into the metadata blocks, and are not aligned to block boundaries, therefore inodes overlap compressed blocks. Inodes are identified -by a 48-bit number which encodes the location of the compressed metadata block -containing the inode, and the byte offset into that block where the inode is -placed (<block, offset>). +by a 64-bit number: the upper 48 bits encode the location of the compressed +metadata block containing the inode, and the lower 16 bits give the byte offset +into that block where the inode is placed (<block, offset>). To maximise compression there are different inodes for each file type (regular file, directory, device, etc.), the inode contents and length diff --git a/Documentation/filesystems/vfs.rst b/Documentation/filesystems/vfs.rst index d3a93eec3945..99495497ea6d 100644 --- a/Documentation/filesystems/vfs.rst +++ b/Documentation/filesystems/vfs.rst @@ -117,7 +117,7 @@ members are defined: const struct fs_parameter_spec *parameters; void (*kill_sb) (struct super_block *); struct module *owner; - struct file_system_type * next; + struct hlist_node list; struct hlist_head fs_supers; struct lock_class_key s_lock_key; @@ -415,33 +415,33 @@ As of kernel 2.6.22, the following members are defined: .. code-block:: c struct inode_operations { - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, umode_t); + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t); + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *,const char *); + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *,struct dentry *,umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *,umode_t,dev_t); + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); int (*readlink) (struct dentry *, char __user *,int); const char *(*get_link) (struct dentry *, struct inode *, struct delayed_call *); - int (*permission) (struct mnt_idmap *, struct inode *, int); + int (*permission) (const struct mnt_idmap *, struct inode *, int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); void (*update_time)(struct inode *inode, enum fs_update_time type, int flags); void (*sync_lazytime)(struct inode *inode); int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, struct file *, umode_t); - struct posix_acl * (*get_acl)(struct mnt_idmap *, struct dentry *, int); - int (*set_acl)(struct mnt_idmap *, struct dentry *, struct posix_acl *, int); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); + struct posix_acl * (*get_acl)(const struct mnt_idmap *, struct dentry *, int); + int (*set_acl)(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); @@ -507,8 +507,8 @@ otherwise noted. dentry before the first mkdir returns. If there is any chance this could happen, then the new inode - should be d_drop()ed and attached with d_splice_alias(). The - returned dentry (if any) should be returned by ->mkdir(). + should be attached with d_splice_alias(). The returned + dentry (if any) should be returned by ->mkdir(). ``rmdir`` called by the rmdir(2) system call. Only required if you want diff --git a/MAINTAINERS b/MAINTAINERS index 55c3aeac289d..b57bad613497 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -4627,13 +4627,6 @@ S: Odd Fixes F: Documentation/block/bfq-iosched.rst F: block/bfq-* -BFS FILE SYSTEM -M: "Tigran A. Aivazian" <aivazian.tigran@gmail.com> -S: Maintained -F: Documentation/filesystems/bfs.rst -F: fs/bfs/ -F: include/uapi/linux/bfs_fs.h - BITMAP API M: Yury Norov <yury.norov@gmail.com> R: Rasmus Villemoes <linux@rasmusvillemoes.dk> @@ -14420,6 +14413,7 @@ S: Supported T: git git://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git F: fs/kernfs/ F: include/linux/kernfs.h +F: tools/testing/selftests/filesystems/kernfs_test.c KEXEC M: Andrew Morton <akpm@linux-foundation.org> diff --git a/arch/mips/configs/malta_defconfig b/arch/mips/configs/malta_defconfig index 56a8f76dab41..4fb4d45f3cdc 100644 --- a/arch/mips/configs/malta_defconfig +++ b/arch/mips/configs/malta_defconfig @@ -333,7 +333,6 @@ CONFIG_AFFS_FS=m CONFIG_HFS_FS=m CONFIG_HFSPLUS_FS=m CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_EFS_FS=m CONFIG_JFFS2_FS=m CONFIG_JFFS2_FS_XATTR=y diff --git a/arch/mips/configs/malta_kvm_defconfig b/arch/mips/configs/malta_kvm_defconfig index 85e95c0ae410..3d1c32d3475e 100644 --- a/arch/mips/configs/malta_kvm_defconfig +++ b/arch/mips/configs/malta_kvm_defconfig @@ -340,7 +340,6 @@ CONFIG_AFFS_FS=m CONFIG_HFS_FS=m CONFIG_HFSPLUS_FS=m CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_EFS_FS=m CONFIG_JFFS2_FS=m CONFIG_JFFS2_FS_XATTR=y diff --git a/arch/mips/configs/maltaup_xpa_defconfig b/arch/mips/configs/maltaup_xpa_defconfig index 0498a0115349..2545168b75bb 100644 --- a/arch/mips/configs/maltaup_xpa_defconfig +++ b/arch/mips/configs/maltaup_xpa_defconfig @@ -339,7 +339,6 @@ CONFIG_AFFS_FS=m CONFIG_HFS_FS=m CONFIG_HFSPLUS_FS=m CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_EFS_FS=m CONFIG_JFFS2_FS=m CONFIG_JFFS2_FS_XATTR=y diff --git a/arch/mips/configs/rm200_defconfig b/arch/mips/configs/rm200_defconfig index 09e006c7cd7c..cd2a831614c8 100644 --- a/arch/mips/configs/rm200_defconfig +++ b/arch/mips/configs/rm200_defconfig @@ -318,7 +318,6 @@ CONFIG_ADFS_FS=m CONFIG_AFFS_FS=m CONFIG_HFS_FS=m CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_EFS_FS=m CONFIG_CRAMFS=m CONFIG_VXFS_FS=m diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 2580e27e4328..40c874fe2f53 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -160,7 +160,6 @@ config PPC select ARCH_HAS_UBSAN select ARCH_HAS_VDSO_ARCH_DATA select ARCH_HAVE_NMI_SAFE_CMPXCHG - select ARCH_HAVE_EXTRA_ELF_NOTES if SPU_BASE select ARCH_KEEP_MEMBLOCK select ARCH_MHP_MEMMAP_ON_MEMORY_ENABLE if PPC_RADIX_MMU select ARCH_MIGHT_HAVE_PC_PARPORT diff --git a/arch/powerpc/configs/fsl-emb-nonhw.config b/arch/powerpc/configs/fsl-emb-nonhw.config index 391c99117ee0..688051b11ce1 100644 --- a/arch/powerpc/configs/fsl-emb-nonhw.config +++ b/arch/powerpc/configs/fsl-emb-nonhw.config @@ -2,7 +2,6 @@ CONFIG_ADFS_FS=m CONFIG_AFFS_FS=m CONFIG_AUDIT=y CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_BINFMT_MISC=m # CONFIG_BLK_DEV_BSG is not set CONFIG_BLK_DEV_INITRD=y diff --git a/arch/powerpc/configs/ppc6xx_defconfig b/arch/powerpc/configs/ppc6xx_defconfig index acffc5c17f92..8aab229fb909 100644 --- a/arch/powerpc/configs/ppc6xx_defconfig +++ b/arch/powerpc/configs/ppc6xx_defconfig @@ -945,7 +945,6 @@ CONFIG_ECRYPT_FS=m CONFIG_HFS_FS=m CONFIG_HFSPLUS_FS=m CONFIG_BEFS_FS=m -CONFIG_BFS_FS=m CONFIG_EFS_FS=m CONFIG_CRAMFS=m CONFIG_VXFS_FS=m diff --git a/arch/powerpc/include/asm/elf.h b/arch/powerpc/include/asm/elf.h index bb4b94444d3e..5dc8c4923eb1 100644 --- a/arch/powerpc/include/asm/elf.h +++ b/arch/powerpc/include/asm/elf.h @@ -123,12 +123,6 @@ extern int arch_setup_additional_pages(struct linux_binprm *bprm, (0x7ff >> (PAGE_SHIFT - 12)) : \ (0x3ffff >> (PAGE_SHIFT - 12))) -#ifdef CONFIG_SPU_BASE -/* Notes used in ET_CORE. Note name is "SPU/<fd>/<filename>". */ -#define NT_SPU 1 - -#endif /* CONFIG_SPU_BASE */ - #ifdef CONFIG_PPC64 #define get_cache_geometry(level) \ diff --git a/arch/powerpc/include/asm/spu.h b/arch/powerpc/include/asm/spu.h index 96ad4510c895..7152285b6268 100644 --- a/arch/powerpc/include/asm/spu.h +++ b/arch/powerpc/include/asm/spu.h @@ -210,15 +210,12 @@ extern long spu_sys_callback(struct spu_syscall_block *s); /* syscalls implemented in spufs */ struct file; -struct coredump_params; struct spufs_calls { long (*create_thread)(const char __user *name, unsigned int flags, umode_t mode, struct file *neighbor); long (*spu_run)(struct file *filp, __u32 __user *unpc, __u32 __user *ustatus); - int (*coredump_extra_notes_size)(void); - int (*coredump_extra_notes_write)(struct coredump_params *cprm); void (*notify_spus_active)(void); struct module *owner; }; diff --git a/arch/powerpc/platforms/cell/Kconfig b/arch/powerpc/platforms/cell/Kconfig index db65bfcd1e74..6bd26815c331 100644 --- a/arch/powerpc/platforms/cell/Kconfig +++ b/arch/powerpc/platforms/cell/Kconfig @@ -10,7 +10,6 @@ config SPU_FS tristate "SPU file system" default m depends on PPC_CELL - depends on COREDUMP select SPU_BASE help The SPU file system is used to access Synergistic Processing diff --git a/arch/powerpc/platforms/cell/spu_syscalls.c b/arch/powerpc/platforms/cell/spu_syscalls.c index 000894e07b02..8be81207e886 100644 --- a/arch/powerpc/platforms/cell/spu_syscalls.c +++ b/arch/powerpc/platforms/cell/spu_syscalls.c @@ -88,26 +88,6 @@ SYSCALL_DEFINE3(spu_run,int, fd, __u32 __user *, unpc, __u32 __user *, ustatus) return calls->spu_run(fd_file(arg), unpc, ustatus); } -#ifdef CONFIG_COREDUMP -int elf_coredump_extra_notes_size(void) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_size(); -} - -int elf_coredump_extra_notes_write(struct coredump_params *cprm) -{ - CLASS(spufs_calls, calls)(); - if (!calls) - return 0; - - return calls->coredump_extra_notes_write(cprm); -} -#endif - void notify_spus_active(void) { struct spufs_calls *calls; diff --git a/arch/powerpc/platforms/cell/spufs/Makefile b/arch/powerpc/platforms/cell/spufs/Makefile index 52e4c80ec8d0..60319d4ff25a 100644 --- a/arch/powerpc/platforms/cell/spufs/Makefile +++ b/arch/powerpc/platforms/cell/spufs/Makefile @@ -4,7 +4,6 @@ obj-$(CONFIG_SPU_FS) += spufs.o spufs-y += inode.o file.o context.o syscalls.o spufs-y += sched.o backing_ops.o hw_ops.o run.o gang.o spufs-y += switch.o fault.o lscsa_alloc.o -spufs-$(CONFIG_COREDUMP) += coredump.o # magic for the trace events CFLAGS_sched.o := -I$(src) diff --git a/arch/powerpc/platforms/cell/spufs/coredump.c b/arch/powerpc/platforms/cell/spufs/coredump.c deleted file mode 100644 index 301ee7d8b7df..000000000000 --- a/arch/powerpc/platforms/cell/spufs/coredump.c +++ /dev/null @@ -1,183 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-or-later -/* - * SPU core dump code - * - * (C) Copyright 2006 IBM Corp. - * - * Author: Dwayne Grant McConnell <decimal@us.ibm.com> - */ - -#include <linux/elf.h> -#include <linux/file.h> -#include <linux/fdtable.h> -#include <linux/fs.h> -#include <linux/gfp.h> -#include <linux/list.h> -#include <linux/syscalls.h> -#include <linux/coredump.h> -#include <linux/binfmts.h> - -#include <linux/uaccess.h> - -#include "spufs.h" - -static int spufs_ctx_note_size(struct spu_context *ctx, int dfd) -{ - int i, sz, total = 0; - char *name; - char fullname[80]; - - for (i = 0; spufs_coredump_read[i].name != NULL; i++) { - name = spufs_coredump_read[i].name; - sz = spufs_coredump_read[i].size; - - sprintf(fullname, "SPU/%d/%s", dfd, name); - - total += sizeof(struct elf_note); - total += roundup(strlen(fullname) + 1, 4); - total += roundup(sz, 4); - } - - return total; -} - -static int match_context(const void *v, struct file *file, unsigned fd) -{ - struct spu_context *ctx; - if (file->f_op != &spufs_context_fops) - return 0; - ctx = SPUFS_I(file_inode(file))->i_ctx; - if (ctx->flags & SPU_CREATE_NOSCHED) - return 0; - return fd + 1; -} - -/* - * The additional architecture-specific notes for Cell are various - * context files in the spu context. - * - * This function iterates over all open file descriptors and sees - * if they are a directory in spufs. In that case we use spufs - * internal functionality to dump them without needing to actually - * open the files. - */ -/* - * descriptor table is not shared, so files can't change or go away. - */ -static struct spu_context *coredump_next_context(int *fd) -{ - struct spu_context *ctx = NULL; - struct file *file; - int n = iterate_fd(current->files, *fd, match_context, NULL); - if (!n) - return NULL; - *fd = n - 1; - - file = fget_raw(*fd); - if (file) { - ctx = SPUFS_I(file_inode(file))->i_ctx; - get_spu_context(ctx); - fput(file); - } - - return ctx; -} - -int spufs_coredump_extra_notes_size(void) -{ - struct spu_context *ctx; - int size = 0, rc, fd; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) { - put_spu_context(ctx); - break; - } - - rc = spufs_ctx_note_size(ctx, fd); - spu_release_saved(ctx); - if (rc < 0) { - put_spu_context(ctx); - break; - } - - size += rc; - - /* start searching the next fd next time */ - fd++; - put_spu_context(ctx); - } - - return size; -} - -static int spufs_arch_write_note(struct spu_context *ctx, int i, - struct coredump_params *cprm, int dfd) -{ - size_t sz = spufs_coredump_read[i].size; - char fullname[80]; - struct elf_note en; - int ret; - - sprintf(fullname, "SPU/%d/%s", dfd, spufs_coredump_read[i].name); - en.n_namesz = strlen(fullname) + 1; - en.n_descsz = sz; - en.n_type = NT_SPU; - - if (!dump_emit(cprm, &en, sizeof(en))) - return -EIO; - if (!dump_emit(cprm, fullname, en.n_namesz)) - return -EIO; - if (!dump_align(cprm, 4)) - return -EIO; - - if (spufs_coredump_read[i].dump) { - ret = spufs_coredump_read[i].dump(ctx, cprm); - if (ret < 0) - return ret; - } else { - char buf[32]; - - ret = snprintf(buf, sizeof(buf), "0x%.16llx", - spufs_coredump_read[i].get(ctx)); - if (ret >= sizeof(buf)) - return sizeof(buf); - - /* count trailing the NULL: */ - if (!dump_emit(cprm, buf, ret + 1)) - return -EIO; - } - - dump_skip_to(cprm, roundup(cprm->pos - ret + sz, 4)); - return 0; -} - -int spufs_coredump_extra_notes_write(struct coredump_params *cprm) -{ - struct spu_context *ctx; - int fd, j, rc; - - fd = 0; - while ((ctx = coredump_next_context(&fd)) != NULL) { - rc = spu_acquire_saved(ctx); - if (rc) - return rc; - - for (j = 0; spufs_coredump_read[j].name != NULL; j++) { - rc = spufs_arch_write_note(ctx, j, cprm, fd); - if (rc) { - spu_release_saved(ctx); - return rc; - } - } - - spu_release_saved(ctx); - - /* start searching the next fd next time */ - fd++; - } - - return 0; -} diff --git a/arch/powerpc/platforms/cell/spufs/file.c b/arch/powerpc/platforms/cell/spufs/file.c index de7494748fec..98c47bafaf67 100644 --- a/arch/powerpc/platforms/cell/spufs/file.c +++ b/arch/powerpc/platforms/cell/spufs/file.c @@ -9,7 +9,6 @@ #undef DEBUG -#include <linux/coredump.h> #include <linux/fs.h> #include <linux/ioctl.h> #include <linux/export.h> @@ -130,14 +129,6 @@ out: return ret; } -static ssize_t spufs_dump_emit(struct coredump_params *cprm, void *buf, - size_t size) -{ - if (!dump_emit(cprm, buf, size)) - return -EIO; - return size; -} - #define DEFINE_SPUFS_SIMPLE_ATTRIBUTE(__fops, __get, __set, __fmt) \ static int __fops ## _open(struct inode *inode, struct file *file) \ { \ @@ -181,12 +172,6 @@ spufs_mem_release(struct inode *inode, struct file *file) } static ssize_t -spufs_mem_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->ops->get_ls(ctx), LS_SIZE); -} - -static ssize_t spufs_mem_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) { @@ -467,13 +452,6 @@ spufs_regs_open(struct inode *inode, struct file *file) } static ssize_t -spufs_regs_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, ctx->csa.lscsa->gprs, - sizeof(ctx->csa.lscsa->gprs)); -} - -static ssize_t spufs_regs_read(struct file *file, char __user *buffer, size_t size, loff_t *pos) { @@ -524,13 +502,6 @@ static const struct file_operations spufs_regs_fops = { }; static ssize_t -spufs_fpcr_dump(struct spu_context *ctx, struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.lscsa->fpcr, - sizeof(ctx->csa.lscsa->fpcr)); -} - -static ssize_t spufs_fpcr_read(struct file *file, char __user * buffer, size_t size, loff_t * pos) { @@ -953,15 +924,6 @@ spufs_signal1_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal1_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[3]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[3], - sizeof(ctx->csa.spu_chnldata_RW[3])); -} - static ssize_t __spufs_signal1_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1086,15 +1048,6 @@ spufs_signal2_release(struct inode *inode, struct file *file) return 0; } -static ssize_t spufs_signal2_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!ctx->csa.spu_chnlcnt_RW[4]) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.spu_chnldata_RW[4], - sizeof(ctx->csa.spu_chnldata_RW[4])); -} - static ssize_t __spufs_signal2_read(struct spu_context *ctx, char __user *buf, size_t len) { @@ -1924,15 +1877,6 @@ static const struct file_operations spufs_caps_fops = { .release = single_release, }; -static ssize_t spufs_mbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0x0000ff)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.prob.pu_mb_R, - sizeof(ctx->csa.prob.pu_mb_R)); -} - static ssize_t spufs_mbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -1962,15 +1906,6 @@ static const struct file_operations spufs_mbox_info_fops = { .llseek = generic_file_llseek, }; -static ssize_t spufs_ibox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - if (!(ctx->csa.prob.mb_stat_R & 0xff0000)) - return 0; - return spufs_dump_emit(cprm, &ctx->csa.priv2.puint_mb_R, - sizeof(ctx->csa.priv2.puint_mb_R)); -} - static ssize_t spufs_ibox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2005,13 +1940,6 @@ static size_t spufs_wbox_info_cnt(struct spu_context *ctx) return (4 - ((ctx->csa.prob.mb_stat_R & 0x00ff00) >> 8)) * sizeof(u32); } -static ssize_t spufs_wbox_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - return spufs_dump_emit(cprm, &ctx->csa.spu_mailbox_data, - spufs_wbox_info_cnt(ctx)); -} - static ssize_t spufs_wbox_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2059,15 +1987,6 @@ static void spufs_get_dma_info(struct spu_context *ctx, } } -static ssize_t spufs_dma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_dma_info info; - - spufs_get_dma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_dma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2112,15 +2031,6 @@ static void spufs_get_proxydma_info(struct spu_context *ctx, } } -static ssize_t spufs_proxydma_info_dump(struct spu_context *ctx, - struct coredump_params *cprm) -{ - struct spu_proxydma_info info; - - spufs_get_proxydma_info(ctx, &info); - return spufs_dump_emit(cprm, &info, sizeof(info)); -} - static ssize_t spufs_proxydma_info_read(struct file *file, char __user *buf, size_t len, loff_t *pos) { @@ -2580,27 +2490,3 @@ const struct spufs_tree_descr spufs_dir_debug_contents[] = { { ".ctx", &spufs_ctx_fops, 0444, }, {}, }; - -const struct spufs_coredump_reader spufs_coredump_read[] = { - { "regs", spufs_regs_dump, NULL, sizeof(struct spu_reg128[128])}, - { "fpcr", spufs_fpcr_dump, NULL, sizeof(struct spu_reg128) }, - { "lslr", NULL, spufs_lslr_get, 19 }, - { "decr", NULL, spufs_decr_get, 19 }, - { "decr_status", NULL, spufs_decr_status_get, 19 }, - { "mem", spufs_mem_dump, NULL, LS_SIZE, }, - { "signal1", spufs_signal1_dump, NULL, sizeof(u32) }, - { "signal1_type", NULL, spufs_signal1_type_get, 19 }, - { "signal2", spufs_signal2_dump, NULL, sizeof(u32) }, - { "signal2_type", NULL, spufs_signal2_type_get, 19 }, - { "event_mask", NULL, spufs_event_mask_get, 19 }, - { "event_status", NULL, spufs_event_status_get, 19 }, - { "mbox_info", spufs_mbox_info_dump, NULL, sizeof(u32) }, - { "ibox_info", spufs_ibox_info_dump, NULL, sizeof(u32) }, - { "wbox_info", spufs_wbox_info_dump, NULL, 4 * sizeof(u32)}, - { "dma_info", spufs_dma_info_dump, NULL, sizeof(struct spu_dma_info)}, - { "proxydma_info", spufs_proxydma_info_dump, - NULL, sizeof(struct spu_proxydma_info)}, - { "object-id", NULL, spufs_object_id_get, 19 }, - { "npc", NULL, spufs_npc_get, 19 }, - { NULL }, -}; diff --git a/arch/powerpc/platforms/cell/spufs/inode.c b/arch/powerpc/platforms/cell/spufs/inode.c index 2b54afb31529..f066d9c7344d 100644 --- a/arch/powerpc/platforms/cell/spufs/inode.c +++ b/arch/powerpc/platforms/cell/spufs/inode.c @@ -92,7 +92,7 @@ out: } static int -spufs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +spufs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -266,9 +266,9 @@ spufs_mkdir(struct inode *dir, struct dentry *dentry, unsigned int flags, static int spufs_context_open(const struct path *path) { FD_PREPARE(fdf, 0, dentry_open(path, O_RDONLY, current_cred())); - if (fdf.err) - return fdf.err; - fd_prepare_file(fdf)->f_op = &spufs_context_fops; + if (fdf->fd < 0) + return fdf->fd; + fdf->file->f_op = &spufs_context_fops; return fd_publish(fdf); } @@ -499,9 +499,9 @@ static int spufs_gang_open(const struct path *path) * in error path of *_open(). */ FD_PREPARE(fdf, 0, dentry_open(path, O_RDONLY, current_cred())); - if (fdf.err) - return fdf.err; - fd_prepare_file(fdf)->f_op = &spufs_gang_fops; + if (fdf->fd < 0) + return fdf->fd; + fdf->file->f_op = &spufs_gang_fops; return fd_publish(fdf); } diff --git a/arch/powerpc/platforms/cell/spufs/spufs.h b/arch/powerpc/platforms/cell/spufs/spufs.h index d33787c57c39..612b5075d0ec 100644 --- a/arch/powerpc/platforms/cell/spufs/spufs.h +++ b/arch/powerpc/platforms/cell/spufs/spufs.h @@ -232,13 +232,9 @@ extern const struct spufs_tree_descr spufs_dir_debug_contents[]; /* system call implementation */ extern struct spufs_calls spufs_calls; -struct coredump_params; long spufs_run_spu(struct spu_context *ctx, u32 *npc, u32 *status); long spufs_create(const struct path *nd, struct dentry *dentry, unsigned int flags, umode_t mode, struct file *filp); -/* ELF coredump callbacks for writing SPU ELF notes */ -extern int spufs_coredump_extra_notes_size(void); -extern int spufs_coredump_extra_notes_write(struct coredump_params *cprm); extern const struct file_operations spufs_context_fops; @@ -335,14 +331,6 @@ void spufs_stop_callback(struct spu *spu, int irq); void spufs_mfc_callback(struct spu *spu); void spufs_dma_callback(struct spu *spu, int type); -struct spufs_coredump_reader { - char *name; - ssize_t (*dump)(struct spu_context *ctx, struct coredump_params *cprm); - u64 (*get)(struct spu_context *ctx); - size_t size; -}; -extern const struct spufs_coredump_reader spufs_coredump_read[]; - extern int spu_init_csa(struct spu_state *csa); extern void spu_fini_csa(struct spu_state *csa); extern int spu_save(struct spu_state *prev, struct spu *spu); diff --git a/arch/powerpc/platforms/cell/spufs/syscalls.c b/arch/powerpc/platforms/cell/spufs/syscalls.c index ea4ba1b6ce6a..b6de37150e73 100644 --- a/arch/powerpc/platforms/cell/spufs/syscalls.c +++ b/arch/powerpc/platforms/cell/spufs/syscalls.c @@ -82,8 +82,4 @@ struct spufs_calls spufs_calls = { .spu_run = do_spu_run, .notify_spus_active = do_notify_spus_active, .owner = THIS_MODULE, -#ifdef CONFIG_COREDUMP - .coredump_extra_notes_size = spufs_coredump_extra_notes_size, - .coredump_extra_notes_write = spufs_coredump_extra_notes_write, -#endif }; diff --git a/block/bio-integrity-fs.c b/block/bio-integrity-fs.c index 692403dfa047..c8e91ada8ca6 100644 --- a/block/bio-integrity-fs.c +++ b/block/bio-integrity-fs.c @@ -31,6 +31,7 @@ unsigned int fs_bio_integrity_alloc(struct bio *bio) bio_integrity_setup_default(bio); return action; } +EXPORT_SYMBOL_GPL(fs_bio_integrity_alloc); void fs_bio_integrity_free(struct bio *bio) { @@ -43,6 +44,7 @@ void fs_bio_integrity_free(struct bio *bio) bio->bi_integrity = NULL; bio->bi_opf &= ~REQ_INTEGRITY; } +EXPORT_SYMBOL_GPL(fs_bio_integrity_free); void fs_bio_integrity_generate(struct bio *bio) { @@ -52,14 +54,10 @@ void fs_bio_integrity_generate(struct bio *bio) } EXPORT_SYMBOL_GPL(fs_bio_integrity_generate); -int fs_bio_integrity_verify(struct bio *bio, sector_t sector, unsigned int size) +int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter) { struct blk_integrity *bi = blk_get_integrity(bio->bi_bdev->bd_disk); struct bio_integrity_payload *bip = bio_integrity(bio); - struct bvec_iter data_iter = { - .bi_sector = sector, - .bi_size = size, - }; if (!bip || !(bip->bip_flags & BIP_CHECK_FLAGS)) return 0; @@ -71,9 +69,10 @@ int fs_bio_integrity_verify(struct bio *bio, sector_t sector, unsigned int size) * bio. Requires the submitter to remember the sector and the size. */ memset(&bip->bip_iter, 0, sizeof(bip->bip_iter)); - bip->bip_iter.bi_sector = sector; - bip->bip_iter.bi_size = bio_integrity_bytes(bi, size >> SECTOR_SHIFT); - return blk_status_to_errno(bio_integrity_verify(bio, &data_iter)); + bip->bip_iter.bi_sector = data_iter->bi_sector; + bip->bip_iter.bi_size = + bio_integrity_bytes(bi, data_iter->bi_size >> SECTOR_SHIFT); + return blk_status_to_errno(bio_integrity_verify(bio, data_iter)); } static int __init fs_bio_integrity_init(void) diff --git a/block/bio-integrity.c b/block/bio-integrity.c index b23e2434d80c..d3df726e0f08 100644 --- a/block/bio-integrity.c +++ b/block/bio-integrity.c @@ -72,6 +72,7 @@ void bio_integrity_alloc_buf(struct bio *bio, gfp_t gfp, bool zero_buffer) unsigned int len = bio_integrity_bytes(bi, bio_sectors(bio)); void *buf; + WARN_ON_ONCE(len > BLK_INTEGRITY_MAX_SIZE); buf = kmalloc(len, gfp | __GFP_NOWARN | (zero_buffer ? __GFP_ZERO : 0)); if (unlikely(!buf)) { struct page *page; diff --git a/block/bio.c b/block/bio.c index f95b63c0604a..b48091c7663f 100644 --- a/block/bio.c +++ b/block/bio.c @@ -320,6 +320,26 @@ void bio_reuse(struct bio *bio, blk_opf_t opf) } EXPORT_SYMBOL_GPL(bio_reuse); +/** + * bio_prepare_reissue - prepare a bio for reuissing the original I/O + * @bio: bio to reuse + * @bdev: block device to use the bio for + * + * Prepare @bio to be resubmitted to retry the original operation. + * The caller must reset bio->bi_iter to the original state. + */ +void bio_prepare_reissue(struct bio *bio, struct block_device *bdev) +{ + bio->bi_bdev = bdev; + bio_associate_blkg(bio); + bio->bi_flags &= + (BIO_PAGE_PINNED | BIO_CLONED | BIO_QUIET | BIO_REFFED); + bio->bi_status = BLK_STS_OK; + bio->bi_bvec_gap_bit = 0; + atomic_set(&bio->__bi_remaining, 1); +} +EXPORT_SYMBOL_GPL(bio_prepare_reissue); + static struct bio *__bio_chain_endio(struct bio *bio) { struct bio *parent = bio->bi_private; @@ -1203,8 +1223,9 @@ bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter) * for the next iteration. */ static int bio_iov_iter_align_down(struct bio *bio, struct iov_iter *iter, - struct bio_vec *bv, unsigned len_align_mask) + unsigned len_align_mask) { + struct bio_vec *bv = &bio->bi_io_vec[bio->bi_vcnt - 1]; size_t nbytes = bio->bi_iter.bi_size & len_align_mask; if (!nbytes) @@ -1262,6 +1283,7 @@ static inline bool bio_iov_bvec_aligned(const struct bio *bio, * bio_iov_iter_get_pages - add user or kernel pages to a bio * @bio: bio to add pages to * @iter: iov iterator describing the region to be added + * @maxlen: maximum size to consume from @iter * @mem_align_mask: the mask the source address and length must be aligned to, * 0 for no requirement * @len_align_mask: the mask to align the total size to, 0 for any length @@ -1282,7 +1304,8 @@ static inline bool bio_iov_bvec_aligned(const struct bio *bio, * is returned only if 0 pages could be pinned. */ int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, - unsigned mem_align_mask, unsigned len_align_mask) + unsigned maxlen, unsigned mem_align_mask, + unsigned len_align_mask) { iov_iter_extraction_t flags = 0; @@ -1294,6 +1317,8 @@ int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, !bio_iov_bvec_aligned(bio, mem_align_mask)) return -EINVAL; + /* Truncate to the maximum size that the caller can handle */ + bio->bi_iter.bi_size = min(bio->bi_iter.bi_size, maxlen); iov_iter_advance(iter, bio->bi_iter.bi_size); return 0; } @@ -1307,7 +1332,7 @@ int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, ssize_t ret; ret = iov_iter_extract_bvecs(iter, bio->bi_io_vec, - BIO_MAX_SIZE - bio->bi_iter.bi_size, + maxlen - bio->bi_iter.bi_size, &bio->bi_vcnt, bio->bi_max_vecs, mem_align_mask, flags); if (ret <= 0) { @@ -1330,8 +1355,7 @@ int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, if (is_pci_p2pdma_page(bio->bi_io_vec->bv_page)) bio->bi_opf |= REQ_NOMERGE; - return bio_iov_iter_align_down(bio, iter, - &bio->bi_io_vec[bio->bi_vcnt - 1], len_align_mask); + return bio_iov_iter_align_down(bio, iter, len_align_mask); } static struct folio *folio_alloc_greedy(gfp_t gfp, size_t *size, @@ -1350,7 +1374,7 @@ static struct folio *folio_alloc_greedy(gfp_t gfp, size_t *size, return folio_alloc(gfp, get_order(*size)); } -static void bio_free_folios(struct bio *bio) +void bio_free_folios(struct bio *bio) { struct bio_vec *bv; int i; @@ -1363,11 +1387,8 @@ static void bio_free_folios(struct bio *bio) } } -static int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, - size_t maxlen, size_t minsize) +int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize) { - size_t total_len = min(maxlen, iov_iter_count(iter)); - if (WARN_ON_ONCE(bio_flagged(bio, BIO_CLONED))) return -EINVAL; if (WARN_ON_ONCE(bio->bi_iter.bi_size)) @@ -1377,7 +1398,6 @@ static int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, do { size_t this_len = min(total_len, SZ_1M); - size_t copied; struct folio *folio; if (this_len > minsize * 2) @@ -1389,164 +1409,69 @@ static int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, folio = folio_alloc_greedy(GFP_KERNEL, &this_len, minsize); if (!folio) break; - bio_add_folio_nofail(bio, folio, this_len, 0); - if (iter->nofault) - copied = copy_folio_from_iter_atomic(folio, 0, this_len, - iter); - else - copied = copy_folio_from_iter(folio, 0, this_len, iter); - if (copied < this_len) { - /* - * Need to revert the iov iter for all bytes we have - * copied. - * - * However the bio size differs from the real copied - * bytes as @this_len is queued but only advanced - * less than that. - * Need to compensate that for the revert. - */ - iov_iter_revert(iter, bio->bi_iter.bi_size - this_len + - copied); - bio_free_folios(bio); - return -EFAULT; - } + /* + * Align down the size to the minimum alignment. In practice + * this should not happen as minsize is expected to be a power + * of two, as is the allocation size, but it offers us a cheap + * extra safety belt. + */ + this_len &= ~(minsize - 1); + bio_add_folio_nofail(bio, folio, this_len, 0); total_len -= this_len; } while (total_len && bio->bi_vcnt < bio->bi_max_vecs); if (!bio->bi_iter.bi_size) return -ENOMEM; - return bio_iov_iter_align_down(bio, iter, - &bio->bi_io_vec[bio->bi_vcnt - 1], minsize - 1); -} - -static int bio_iov_iter_bounce_read(struct bio *bio, struct iov_iter *iter, - size_t maxlen, size_t minsize) -{ - size_t len = min3(iov_iter_count(iter), maxlen, SZ_1M); - struct folio *folio; - ssize_t ret; - - folio = folio_alloc_greedy(GFP_KERNEL, &len, minsize); - if (!folio) - return -ENOMEM; - - do { - ret = iov_iter_extract_bvecs(iter, bio->bi_io_vec + 1, len, - &bio->bi_vcnt, bio->bi_max_vecs - 1, 0, 0); - if (ret <= 0) { - if (!bio->bi_vcnt) - goto out_folio_put; - break; - } - len -= ret; - bio->bi_iter.bi_size += ret; - } while (len && bio->bi_vcnt < bio->bi_max_vecs - 1); - - /* - * Set the folio directly here. The above loop has already calculated - * the correct bi_size, and we use bi_vcnt for the user buffers. That - * is safe as bi_vcnt is only used by the submitter and not the actual - * I/O path. - */ - bvec_set_folio(&bio->bi_io_vec[0], folio, bio->bi_iter.bi_size, 0); - if (iov_iter_extract_will_pin(iter)) - bio_set_flag(bio, BIO_PAGE_PINNED); - - /* The first vec stores the bounce buffer, so do not subtract 1 here. */ - ret = bio_iov_iter_align_down(bio, iter, - &bio->bi_io_vec[bio->bi_vcnt], minsize - 1); - if (ret) - goto out_folio_put; - - /* Update the bounc buffer bv_len to the aligned down size. */ - bio->bi_io_vec[0].bv_len = bio->bi_iter.bi_size; return 0; - -out_folio_put: - folio_put(folio); - return ret; } /** - * bio_iov_iter_bounce - bounce buffer data from an iter into a bio + * bio_iov_iter_bounce_write - bounce buffer data from an iter into a bio * @bio: bio to send - * @iter: iter to read from / write into + * @iter: iter to read from * @maxlen: maximum size to bounce * @minsize: minimum folio allocation size * - * Helper for direct I/O implementations that need to bounce buffer because - * we need to checksum the data or perform other operations that require - * consistency. Allocates folios to back the bounce buffer, and for writes - * copies the data into it. Needs to be paired with bio_iov_iter_unbounce() - * called on completion. + * Helper for direct I/O write implementations that need to bounce buffer + * because they need need to checksum the data or perform other operations that + * require consistency. Allocates folios to back the bounce buffer, and copies + * the data into it. Needs to be paired with bio_free_folios() called on + * completion. */ -int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen, - size_t minsize) -{ - if (op_is_write(bio_op(bio))) - return bio_iov_iter_bounce_write(bio, iter, maxlen, minsize); - return bio_iov_iter_bounce_read(bio, iter, maxlen, minsize); -} - -static void bvec_unpin(struct bio_vec *bv, bool mark_dirty) -{ - struct folio *folio = bvec_folio(bv); - size_t nr_pages = (bv->bv_offset + bv->bv_len - 1) / PAGE_SIZE - - bv->bv_offset / PAGE_SIZE + 1; - - if (mark_dirty) - folio_mark_dirty_lock(folio); - unpin_user_folio(folio, nr_pages); -} - -static void bio_iov_iter_unbounce_read(struct bio *bio, bool is_error, - bool mark_dirty) +int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, + size_t maxlen, size_t minsize) { - unsigned int len = bio->bi_io_vec[0].bv_len; - - if (likely(!is_error)) { - void *buf = bvec_virt(&bio->bi_io_vec[0]); - struct iov_iter to; + size_t total_len = min(maxlen, iov_iter_count(iter)); + size_t total_copied = 0; + struct bio_vec *bv; + int i, error; - iov_iter_bvec(&to, ITER_DEST, bio->bi_io_vec + 1, bio->bi_vcnt, - len); - /* copying to pinned pages should always work */ - WARN_ON_ONCE(copy_to_iter(buf, len, &to) != len); - } else { - /* No need to mark folios dirty if never copied to them */ - mark_dirty = false; - } + error = bio_alloc_bounce_folios(bio, total_len, minsize); + if (error) + return error; - if (bio_flagged(bio, BIO_PAGE_PINNED)) { - int i; + bio_for_each_bvec_all(bv, bio, i) { + struct folio *folio = page_folio(bv->bv_page); + size_t copied; - for (i = 0; i < bio->bi_vcnt; i++) - bvec_unpin(&bio->bi_io_vec[1 + i], mark_dirty); + if (iter->nofault) + copied = copy_folio_from_iter_atomic(folio, 0, + bv->bv_len, iter); + else + copied = copy_folio_from_iter(folio, 0, bv->bv_len, + iter); + total_copied += copied; + if (copied < bv->bv_len) { + iov_iter_revert(iter, total_copied); + bio_free_folios(bio); + return -EFAULT; + } } - folio_put(bvec_folio(&bio->bi_io_vec[0])); -} - -/** - * bio_iov_iter_unbounce - finish a bounce buffer operation - * @bio: completed bio - * @is_error: %true if an I/O error occurred and data should not be copied - * @mark_dirty: If %true, folios will be marked dirty. - * - * Helper for direct I/O implementations that need to bounce buffer because - * we need to checksum the data or perform other operations that require - * consistency. Called to complete a bio set up by bio_iov_iter_bounce(). - * Copies data back for reads, and marks the original folios dirty if - * requested and then frees the bounce buffer. - */ -void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty) -{ - if (op_is_write(bio_op(bio))) - bio_free_folios(bio); - else - bio_iov_iter_unbounce_read(bio, is_error, mark_dirty); + return 0; } +EXPORT_SYMBOL_GPL(bio_iov_iter_bounce_write); static void bio_wait_end_io(struct bio *bio) { diff --git a/block/blk-map.c b/block/blk-map.c index 9cb9605d1f62..81cba3af4e9c 100644 --- a/block/blk-map.c +++ b/block/blk-map.c @@ -274,7 +274,7 @@ static int bio_map_user_iov(struct request *rq, struct iov_iter *iter, * No alignment requirements on our part to support arbitrary * passthrough commands. */ - ret = bio_iov_iter_get_pages(bio, iter, 0, 0); + ret = bio_iov_iter_get_pages(bio, iter, BIO_MAX_SIZE, 0, 0); if (ret) goto out_put; ret = blk_rq_append_bio(rq, bio); diff --git a/block/blk-settings.c b/block/blk-settings.c index 8274631290db..e469baa1f08b 100644 --- a/block/blk-settings.c +++ b/block/blk-settings.c @@ -206,6 +206,12 @@ static int blk_validate_integrity_limits(struct queue_limits *lim) lim->max_sectors = min(lim->max_sectors, max_integrity_io_size(lim) >> SECTOR_SHIFT); + if (lim->features & BLK_FEAT_ATOMIC_WRITES) { + lim->atomic_write_max_sectors = + min(lim->atomic_write_max_sectors, + max_integrity_io_size(lim) >> SECTOR_SHIFT); + } + return 0; } diff --git a/block/fops.c b/block/fops.c index 2ce7c6c4714e..a83df69b175a 100644 --- a/block/fops.c +++ b/block/fops.c @@ -46,7 +46,8 @@ static bool blkdev_dio_invalid(struct block_device *bdev, struct kiocb *iocb, static inline int blkdev_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, struct block_device *bdev) { - return bio_iov_iter_get_pages(bio, iter, bdev_dma_alignment(bdev), + return bio_iov_iter_get_pages(bio, iter, BIO_MAX_SIZE, + bdev_dma_alignment(bdev), bdev_logical_block_size(bdev) - 1); } diff --git a/block/partitions/efi.h b/block/partitions/efi.h index 84b9f36b9e47..1f56f93b2804 100644 --- a/block/partitions/efi.h +++ b/block/partitions/efi.h @@ -75,18 +75,12 @@ typedef struct _gpt_header { */ } __packed gpt_header; -typedef struct _gpt_entry_attributes { - u64 required_to_function:1; - u64 reserved:47; - u64 type_guid_specific:16; -} __packed gpt_entry_attributes; - typedef struct _gpt_entry { efi_guid_t partition_type_guid; efi_guid_t unique_partition_guid; __le64 starting_lba; __le64 ending_lba; - gpt_entry_attributes attributes; + __le64 attributes; __le16 partition_name[72/sizeof(__le16)]; } __packed gpt_entry; diff --git a/drivers/android/binder/rust_binderfs.c b/drivers/android/binder/rust_binderfs.c index 300cc65562d1..e98e6ca163df 100644 --- a/drivers/android/binder/rust_binderfs.c +++ b/drivers/android/binder/rust_binderfs.c @@ -338,7 +338,7 @@ static inline bool is_binderfs_control_device(const struct dentry *dentry) return info->control_dentry == dentry; } -static int binderfs_rename(struct mnt_idmap *idmap, +static int binderfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/drivers/android/binderfs.c b/drivers/android/binderfs.c index 361d69f756f5..087eae6ca8ec 100644 --- a/drivers/android/binderfs.c +++ b/drivers/android/binderfs.c @@ -344,7 +344,7 @@ static inline bool is_binderfs_control_device(const struct dentry *dentry) return info->control_dentry == dentry; } -static int binderfs_rename(struct mnt_idmap *idmap, +static int binderfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/drivers/base/devtmpfs.c b/drivers/base/devtmpfs.c index aef0fcc6aba1..11c70888f38b 100644 --- a/drivers/base/devtmpfs.c +++ b/drivers/base/devtmpfs.c @@ -72,39 +72,90 @@ static struct file_system_type internal_fs_type = { .kill_sb = kill_anon_super, }; -/* Simply take a ref on the existing mount */ +struct devtmpfs_context { + struct fs_context *fc; +}; + +static void devtmpfs_free(struct fs_context *fc) +{ + struct devtmpfs_context *ctx = fc->fs_private; + + if (ctx) { + put_fs_context(ctx->fc); + kfree(ctx); + } +} + +static int devtmpfs_parse_param(struct fs_context *fc, struct fs_parameter *param) +{ + struct devtmpfs_context *ctx = fc->fs_private; + + return ctx->fc->ops->parse_param(ctx->fc, param); +} + +static int devtmpfs_parse_monolithic(struct fs_context *fc, void *data) +{ + struct devtmpfs_context *ctx = fc->fs_private; + + if (ctx->fc->ops->parse_monolithic) + return ctx->fc->ops->parse_monolithic(ctx->fc, data); + return generic_parse_monolithic(ctx->fc, data); +} + static int devtmpfs_get_tree(struct fs_context *fc) { + struct devtmpfs_context *ctx = fc->fs_private; struct super_block *sb = mnt->mnt_sb; + int err; atomic_inc(&sb->s_active); down_write(&sb->s_umount); + + if (ctx->fc->ops->reconfigure) { + err = ctx->fc->ops->reconfigure(ctx->fc); + if (err) { + deactivate_locked_super(sb); + return err; + } + } + fc->root = dget(sb->s_root); return 0; } -/* Ops are filled in during init depending on underlying shmem or ramfs type */ -static struct fs_context_operations devtmpfs_context_ops = {}; +static const struct fs_context_operations devtmpfs_context_ops = { + .free = devtmpfs_free, + .parse_param = devtmpfs_parse_param, + .parse_monolithic = devtmpfs_parse_monolithic, + .get_tree = devtmpfs_get_tree, +}; -/* Call the underlying initialization and set to our ops */ static int devtmpfs_init_fs_context(struct fs_context *fc) { - int ret; -#ifdef CONFIG_TMPFS - ret = shmem_init_fs_context(fc); -#else - ret = ramfs_init_fs_context(fc); -#endif - if (ret < 0) - return ret; + struct devtmpfs_context *ctx; + int err; + + ctx = kzalloc_obj(struct devtmpfs_context); + if (!ctx) + return -ENOMEM; + + /* Each mount will reconfigure the shared superblock w/ new options */ + ctx->fc = fs_context_for_reconfigure(mnt->mnt_root, + mnt->mnt_sb->s_flags, MS_RMT_MASK); + if (IS_ERR(ctx->fc)) { + err = PTR_ERR(ctx->fc); + kfree(ctx); + return err; + } + fc->fs_private = ctx; fc->ops = &devtmpfs_context_ops; return 0; } static struct file_system_type dev_fs_type = { - .name = "devtmpfs", + .name = "devtmpfs", .init_fs_context = devtmpfs_init_fs_context, }; @@ -443,31 +494,6 @@ static int __ref devtmpfsd(void *p) } /* - * Get the underlying (shmem/ramfs) context ops to build ours - */ -static int devtmpfs_configure_context(void) -{ - struct fs_context *fc; - - fc = fs_context_for_reconfigure(mnt->mnt_root, mnt->mnt_sb->s_flags, - MS_RMT_MASK); - if (IS_ERR(fc)) - return PTR_ERR(fc); - - /* Set up devtmpfs_context_ops based on underlying type */ - devtmpfs_context_ops.free = fc->ops->free; - devtmpfs_context_ops.dup = fc->ops->dup; - devtmpfs_context_ops.parse_param = fc->ops->parse_param; - devtmpfs_context_ops.parse_monolithic = fc->ops->parse_monolithic; - devtmpfs_context_ops.get_tree = &devtmpfs_get_tree; - devtmpfs_context_ops.reconfigure = fc->ops->reconfigure; - - put_fs_context(fc); - - return 0; -} - -/* * Create devtmpfs instance, driver-core devices will add their device * nodes here. */ @@ -482,12 +508,6 @@ int __init devtmpfs_init(void) return PTR_ERR(mnt); } - err = devtmpfs_configure_context(); - if (err) { - pr_err("unable to configure devtmpfs type %d\n", err); - return err; - } - err = register_filesystem(&dev_fs_type); if (err) { pr_err("unable to register devtmpfs type %d\n", err); diff --git a/drivers/gpio/gpiolib-cdev.c b/drivers/gpio/gpiolib-cdev.c index 5d53bfcdf726..a127e6efd7a1 100644 --- a/drivers/gpio/gpiolib-cdev.c +++ b/drivers/gpio/gpiolib-cdev.c @@ -377,11 +377,11 @@ static int linehandle_create(struct gpio_device *gdev, void __user *ip) FD_PREPARE(fdf, O_RDONLY | O_CLOEXEC, anon_inode_getfile("gpio-linehandle", &linehandle_fileops, lh, O_RDONLY | O_CLOEXEC)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; retain_and_null_ptr(lh); - handlereq.fd = fd_prepare_fd(fdf); + handlereq.fd = fdf->fd; if (copy_to_user(ip, &handlereq, sizeof(handlereq))) return -EFAULT; @@ -1715,11 +1715,11 @@ static int linereq_create(struct gpio_device *gdev, void __user *ip) FD_PREPARE(fdf, O_RDONLY | O_CLOEXEC, anon_inode_getfile("gpio-line", &line_fileops, lr, O_RDONLY | O_CLOEXEC)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; retain_and_null_ptr(lr); - ulr.fd = fd_prepare_fd(fdf); + ulr.fd = fdf->fd; if (copy_to_user(ip, &ulr, sizeof(ulr))) return -EFAULT; @@ -2115,11 +2115,11 @@ static int lineevent_create(struct gpio_device *gdev, void __user *ip) FD_PREPARE(fdf, O_RDONLY | O_CLOEXEC, anon_inode_getfile("gpio-event", &lineevent_fileops, le, O_RDONLY | O_CLOEXEC)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; retain_and_null_ptr(le); - eventreq.fd = fd_prepare_fd(fdf); + eventreq.fd = fdf->fd; if (copy_to_user(ip, &eventreq, sizeof(eventreq))) return -EFAULT; diff --git a/drivers/gpu/drm/msm/msm_perfcntr.c b/drivers/gpu/drm/msm/msm_perfcntr.c index ce65b1160955..7fa2e858bd08 100644 --- a/drivers/gpu/drm/msm/msm_perfcntr.c +++ b/drivers/gpu/drm/msm/msm_perfcntr.c @@ -543,8 +543,8 @@ msm_ioctl_perfcntr_config(struct drm_device *dev, void *data, struct drm_file *f FD_PREPARE(fdf, O_CLOEXEC, anon_inode_getfile("[msm_perfcntrs]", &stream_fops, stream, 0)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; INIT_WORK(&stream->sel_work, sel_worker); kthread_init_work(&stream->sample_work, sample_worker); diff --git a/drivers/media/mc/mc-request.c b/drivers/media/mc/mc-request.c index 13e77648807c..e1387f039780 100644 --- a/drivers/media/mc/mc-request.c +++ b/drivers/media/mc/mc-request.c @@ -316,15 +316,15 @@ int media_request_alloc(struct media_device *mdev, int *alloc_fd) FD_PREPARE(fdf, O_CLOEXEC, anon_inode_getfile("request", &request_fops, NULL, O_CLOEXEC)); - if (fdf.err) { - ret = fdf.err; + if (fdf->fd < 0) { + ret = fdf->fd; goto err_free_req; } - fd_prepare_file(fdf)->private_data = req; + fdf->file->private_data = req; snprintf(req->debug_str, sizeof(req->debug_str), "%u:%d", - atomic_inc_return(&mdev->request_id), fd_prepare_fd(fdf)); + atomic_inc_return(&mdev->request_id), fdf->fd); atomic_inc(&mdev->num_requests); dev_dbg(mdev->dev, "request: allocated %s\n", req->debug_str); diff --git a/drivers/misc/ntsync.c b/drivers/misc/ntsync.c index 4a805919bb0c..2857ae37d3c8 100644 --- a/drivers/misc/ntsync.c +++ b/drivers/misc/ntsync.c @@ -724,9 +724,9 @@ static int ntsync_obj_get_fd(struct ntsync_obj *obj) { FD_PREPARE(fdf, O_CLOEXEC, anon_inode_getfile("ntsync", &ntsync_obj_fops, obj, O_RDWR)); - if (fdf.err) - return fdf.err; - obj->file = fd_prepare_file(fdf); + if (fdf->fd < 0) + return fdf->fd; + obj->file = fdf->file; return fd_publish(fdf); } diff --git a/fs/9p/acl.c b/fs/9p/acl.c index ae7e7cf7523a..c6c7c47d32b9 100644 --- a/fs/9p/acl.c +++ b/fs/9p/acl.c @@ -140,7 +140,7 @@ struct posix_acl *v9fs_iop_get_inode_acl(struct inode *inode, int type, bool rcu } -struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, +struct posix_acl *v9fs_iop_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct v9fs_session_info *v9ses; @@ -152,7 +152,7 @@ struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, return v9fs_get_cached_acl(d_inode(dentry), type); } -int v9fs_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int v9fs_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int retval; diff --git a/fs/9p/acl.h b/fs/9p/acl.h index 333cfcc281da..2d1b24abcd3f 100644 --- a/fs/9p/acl.h +++ b/fs/9p/acl.h @@ -10,9 +10,9 @@ int v9fs_get_acl(struct inode *inode, struct p9_fid *fid); struct posix_acl *v9fs_iop_get_inode_acl(struct inode *inode, int type, bool rcu); -struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, +struct posix_acl *v9fs_iop_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int v9fs_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int v9fs_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int v9fs_acl_chmod(struct inode *inode, struct p9_fid *fid); int v9fs_set_create_acl(struct inode *inode, struct p9_fid *fid, diff --git a/fs/9p/v9fs.h b/fs/9p/v9fs.h index a462bcbfc7da..54a4a4ec5c15 100644 --- a/fs/9p/v9fs.h +++ b/fs/9p/v9fs.h @@ -188,7 +188,7 @@ extern struct dentry *v9fs_vfs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags); extern int v9fs_vfs_unlink(struct inode *i, struct dentry *d); extern int v9fs_vfs_rmdir(struct inode *i, struct dentry *d); -extern int v9fs_vfs_rename(struct mnt_idmap *idmap, +extern int v9fs_vfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); diff --git a/fs/9p/v9fs_vfs.h b/fs/9p/v9fs_vfs.h index 1856d91f8703..5e7b60042498 100644 --- a/fs/9p/v9fs_vfs.h +++ b/fs/9p/v9fs_vfs.h @@ -74,7 +74,7 @@ int v9fs_file_open(struct inode *inode, struct file *file); int v9fs_uflags2omode(int uflags, int extended); void v9fs_blank_wstat(struct p9_wstat *wstat); -int v9fs_vfs_setattr_dotl(struct mnt_idmap *idmap, +int v9fs_vfs_setattr_dotl(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); int v9fs_file_fsync_dotl(struct file *filp, loff_t start, loff_t end, int datasync); diff --git a/fs/9p/vfs_addr.c b/fs/9p/vfs_addr.c index 13cf87a5f90c..170a2b91c5f0 100644 --- a/fs/9p/vfs_addr.c +++ b/fs/9p/vfs_addr.c @@ -150,7 +150,6 @@ static int v9fs_init_request(struct netfs_io_request *rreq, struct file *file) struct p9_fid *fid; struct dentry *dentry; bool writing = (rreq->origin == NETFS_READ_FOR_WRITE || - rreq->origin == NETFS_WRITETHROUGH || rreq->origin == NETFS_UNBUFFERED_WRITE || rreq->origin == NETFS_DIO_WRITE); diff --git a/fs/9p/vfs_inode.c b/fs/9p/vfs_inode.c index 3829554ca369..c95e653344f6 100644 --- a/fs/9p/vfs_inode.c +++ b/fs/9p/vfs_inode.c @@ -652,7 +652,7 @@ error: */ static int -v9fs_vfs_create(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct v9fs_session_info *v9ses = v9fs_inode2v9ses(dir); @@ -679,7 +679,7 @@ v9fs_vfs_create(struct mnt_idmap *idmap, struct inode *dir, * */ -static struct dentry *v9fs_vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *v9fs_vfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { u32 perm; @@ -858,7 +858,7 @@ int v9fs_vfs_rmdir(struct inode *i, struct dentry *d) */ int -v9fs_vfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +v9fs_vfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -966,7 +966,7 @@ error: */ static int -v9fs_vfs_getattr(struct mnt_idmap *idmap, const struct path *path, +v9fs_vfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct dentry *dentry = path->dentry; @@ -1014,7 +1014,7 @@ v9fs_vfs_getattr(struct mnt_idmap *idmap, const struct path *path, * */ -static int v9fs_vfs_setattr(struct mnt_idmap *idmap, +static int v9fs_vfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int retval, use_dentry = 0; @@ -1249,7 +1249,7 @@ static int v9fs_vfs_mkspecial(struct inode *dir, struct dentry *dentry, */ static int -v9fs_vfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { p9_debug(P9_DEBUG_VFS, " %llu,%pd,%s\n", @@ -1304,7 +1304,7 @@ v9fs_vfs_link(struct dentry *old_dentry, struct inode *dir, */ static int -v9fs_vfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct v9fs_session_info *v9ses = v9fs_inode2v9ses(dir); diff --git a/fs/9p/vfs_inode_dotl.c b/fs/9p/vfs_inode_dotl.c index 116b29e95f21..2cd2898580a9 100644 --- a/fs/9p/vfs_inode_dotl.c +++ b/fs/9p/vfs_inode_dotl.c @@ -29,7 +29,7 @@ #include "acl.h" static int -v9fs_vfs_mknod_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode, dev_t rdev); /** @@ -216,7 +216,7 @@ int v9fs_open_to_dotl_flags(int flags) * */ static int -v9fs_vfs_create_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_create_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode) { return v9fs_vfs_mknod_dotl(idmap, dir, dentry, omode, 0); @@ -344,7 +344,7 @@ out: * */ -static struct dentry *v9fs_vfs_mkdir_dotl(struct mnt_idmap *idmap, +static struct dentry *v9fs_vfs_mkdir_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode) { @@ -414,7 +414,7 @@ error: } static int -v9fs_vfs_getattr_dotl(struct mnt_idmap *idmap, +v9fs_vfs_getattr_dotl(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -508,7 +508,7 @@ static int v9fs_mapped_iattr_valid(int iattr_valid) * */ -int v9fs_vfs_setattr_dotl(struct mnt_idmap *idmap, +int v9fs_vfs_setattr_dotl(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int retval, use_dentry = 0; @@ -682,7 +682,7 @@ v9fs_stat2inode_dotl(struct p9_stat_dotl *stat, struct inode *inode, } static int -v9fs_vfs_symlink_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_symlink_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int err; @@ -809,7 +809,7 @@ v9fs_vfs_link_dotl(struct dentry *old_dentry, struct inode *dir, * */ static int -v9fs_vfs_mknod_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode, dev_t rdev) { int err; diff --git a/fs/9p/xattr.c b/fs/9p/xattr.c index 8604e3377ee7..dac06587f67a 100644 --- a/fs/9p/xattr.c +++ b/fs/9p/xattr.c @@ -153,7 +153,7 @@ static int v9fs_xattr_handler_get(const struct xattr_handler *handler, } static int v9fs_xattr_handler_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/Kconfig b/fs/Kconfig index e05917adcd60..46313f65ee54 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -314,7 +314,6 @@ source "fs/ecryptfs/Kconfig" source "fs/hfs/Kconfig" source "fs/hfsplus/Kconfig" source "fs/befs/Kconfig" -source "fs/bfs/Kconfig" source "fs/jffs2/Kconfig" # UBIFS File system configuration source "fs/ubifs/Kconfig" @@ -421,4 +420,12 @@ source "fs/unicode/Kconfig" config IO_WQ bool +config FDTABLE_KUNIT_TEST + bool "KUnit test for fdtable" if !KUNIT_ALL_TESTS + depends on KUNIT=y + default KUNIT_ALL_TESTS + help + This builds the fdtable KUnit tests, which tests various aspects + of the fdtable structure and allocation. + endmenu diff --git a/fs/Makefile b/fs/Makefile index 055dfc23d82b..16f1108b64e1 100644 --- a/fs/Makefile +++ b/fs/Makefile @@ -76,7 +76,6 @@ obj-$(CONFIG_CODA_FS) += coda/ obj-$(CONFIG_MINIX_FS) += minix/ obj-$(CONFIG_FAT_FS) += fat/ obj-$(CONFIG_EXFAT_FS) += exfat/ -obj-$(CONFIG_BFS_FS) += bfs/ obj-$(CONFIG_ISO9660_FS) += isofs/ obj-$(CONFIG_HFSPLUS_FS) += hfsplus/ # Before hfs to find wrapped HFS+ obj-$(CONFIG_HFS_FS) += hfs/ diff --git a/fs/adfs/adfs.h b/fs/adfs/adfs.h index 0d32b7cd99b4..6003832277f8 100644 --- a/fs/adfs/adfs.h +++ b/fs/adfs/adfs.h @@ -144,7 +144,7 @@ struct adfs_discmap { /* Inode stuff */ struct inode *adfs_iget(struct super_block *sb, struct object_info *obj); int adfs_write_inode(struct inode *inode, struct writeback_control *wbc); -int adfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int adfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); /* map.c */ diff --git a/fs/adfs/dir.c b/fs/adfs/dir.c index 11afa9e157aa..b8cc6a697a05 100644 --- a/fs/adfs/dir.c +++ b/fs/adfs/dir.c @@ -191,7 +191,7 @@ static int adfs_dir_sync(struct adfs_dir *dir) for (i = dir->nr_buffers - 1; i >= 0; i--) { struct buffer_head *bh = dir->bhs[i]; sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; } diff --git a/fs/adfs/inode.c b/fs/adfs/inode.c index 4ac442d0a8c0..34598b499372 100644 --- a/fs/adfs/inode.c +++ b/fs/adfs/inode.c @@ -299,7 +299,7 @@ out: * later. */ int -adfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) +adfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); struct super_block *sb = inode->i_sb; diff --git a/fs/affs/affs.h b/fs/affs/affs.h index d1c506c1f310..2518de96a0c8 100644 --- a/fs/affs/affs.h +++ b/fs/affs/affs.h @@ -166,17 +166,17 @@ extern const struct export_operations affs_export_ops; extern int affs_hash_name(struct super_block *sb, const u8 *name, unsigned int len); extern struct dentry *affs_lookup(struct inode *dir, struct dentry *dentry, unsigned int); extern int affs_unlink(struct inode *dir, struct dentry *dentry); -extern int affs_create(struct mnt_idmap *idmap, struct inode *dir, +extern int affs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); -extern struct dentry *affs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +extern struct dentry *affs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); extern int affs_rmdir(struct inode *dir, struct dentry *dentry); extern int affs_link(struct dentry *olddentry, struct inode *dir, struct dentry *dentry); -extern int affs_symlink(struct mnt_idmap *idmap, +extern int affs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname); -extern int affs_rename2(struct mnt_idmap *idmap, +extern int affs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); @@ -184,7 +184,7 @@ extern int affs_rename2(struct mnt_idmap *idmap, /* inode.c */ extern struct inode *affs_new_inode(struct inode *dir); -extern int affs_setattr(struct mnt_idmap *idmap, +extern int affs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void affs_evict_inode(struct inode *inode); extern struct inode *affs_iget(struct super_block *sb, diff --git a/fs/affs/inode.c b/fs/affs/inode.c index d4a3f381c4bc..2a48d7422091 100644 --- a/fs/affs/inode.c +++ b/fs/affs/inode.c @@ -213,7 +213,7 @@ affs_write_inode(struct inode *inode, struct writeback_control *wbc) } int -affs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) +affs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); int error; diff --git a/fs/affs/namei.c b/fs/affs/namei.c index 6cb52efafe5f..2e32899a32d5 100644 --- a/fs/affs/namei.c +++ b/fs/affs/namei.c @@ -242,7 +242,7 @@ affs_unlink(struct inode *dir, struct dentry *dentry) } int -affs_create(struct mnt_idmap *idmap, struct inode *dir, +affs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -274,7 +274,7 @@ affs_create(struct mnt_idmap *idmap, struct inode *dir, } struct dentry * -affs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +affs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -313,7 +313,7 @@ affs_rmdir(struct inode *dir, struct dentry *dentry) } int -affs_symlink(struct mnt_idmap *idmap, struct inode *dir, +affs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct super_block *sb = dir->i_sb; @@ -503,7 +503,7 @@ done: return retval; } -int affs_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +int affs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/afs/dir.c b/fs/afs/dir.c index 2db534a2c7cc..75f8f70cfeff 100644 --- a/fs/afs/dir.c +++ b/fs/afs/dir.c @@ -33,17 +33,17 @@ static bool afs_lookup_one_filldir(struct dir_context *ctx, const char *name, in static bool afs_lookup_filldir(struct dir_context *ctx, const char *name, int nlen, u64 ino, u32 uniquifier); #define AFS_LOOKUP ((filldir_t)0x137UL) -static int afs_create(struct mnt_idmap *idmap, struct inode *dir, +static int afs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); -static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *afs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); static int afs_rmdir(struct inode *dir, struct dentry *dentry); static int afs_unlink(struct inode *dir, struct dentry *dentry); static int afs_link(struct dentry *from, struct inode *dir, struct dentry *dentry); -static int afs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int afs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *content); -static int afs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int afs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); static int afs_dir_writepages(struct address_space *mapping, @@ -1310,7 +1310,7 @@ static const struct afs_operation_ops afs_mkdir_operation = { /* * create a directory on an AFS filesystem */ -static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *afs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct afs_operation *op; @@ -1632,7 +1632,7 @@ static const struct afs_operation_ops afs_create_operation = { /* * create a regular file on an AFS filesystem */ -static int afs_create(struct mnt_idmap *idmap, struct inode *dir, +static int afs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct afs_operation *op; @@ -1779,7 +1779,7 @@ static const struct afs_operation_ops afs_symlink_operation = { /* * create a symlink in an AFS filesystem */ -static int afs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int afs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *content) { struct afs_operation *op; @@ -2067,7 +2067,7 @@ static const struct afs_operation_ops afs_rename_exchange_operation = { /* * rename a file in an AFS filesystem and/or move it between directories */ -static int afs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int afs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/afs/file.c b/fs/afs/file.c index 0467742bfeee..4c78d3441785 100644 --- a/fs/afs/file.c +++ b/fs/afs/file.c @@ -400,7 +400,6 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) } break; case NETFS_WRITEBACK: - case NETFS_WRITETHROUGH: case NETFS_UNBUFFERED_WRITE: case NETFS_DIO_WRITE: if (S_ISREG(rreq->inode->i_mode)) @@ -413,7 +412,7 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) return 0; } -static int afs_check_write_begin(struct file *file, loff_t pos, unsigned len, +static int afs_check_write_begin(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata) { struct afs_vnode *vnode = AFS_FS_I(file_inode(file)); @@ -434,7 +433,7 @@ static void afs_free_request(struct netfs_io_request *rreq) * Also, estimate the number of 512 bytes blocks used, rounded up to nearest 1K * for consistency with other AFS clients. */ -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size) { struct inode *inode = &vnode->netfs.inode; loff_t i_size; @@ -448,10 +447,9 @@ void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) } spin_unlock(&inode->i_lock); write_sequnlock(&vnode->cb_lock); - fscache_update_cookie(afs_vnode_cache(vnode), NULL, &new_i_size); } -static void afs_update_i_size(struct inode *inode, loff_t new_i_size) +static void afs_update_i_size(struct inode *inode, uoff_t new_i_size) { afs_set_i_size(AFS_FS_I(inode), new_i_size); } diff --git a/fs/afs/inode.c b/fs/afs/inode.c index 14f39a9bea6c..8e6ca6b45c6b 100644 --- a/fs/afs/inode.c +++ b/fs/afs/inode.c @@ -596,7 +596,7 @@ error: /* * read the attributes of an inode */ -int afs_getattr(struct mnt_idmap *idmap, const struct path *path, +int afs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -759,7 +759,7 @@ static const struct afs_operation_ops afs_setattr_operation = { /* * set the attributes of an inode */ -int afs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int afs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { const unsigned int supported = diff --git a/fs/afs/internal.h b/fs/afs/internal.h index 330654ed16ec..5744a347ce2e 100644 --- a/fs/afs/internal.h +++ b/fs/afs/internal.h @@ -1171,7 +1171,7 @@ extern int afs_open(struct inode *, struct file *); extern int afs_release(struct inode *, struct file *); void afs_fetch_data_async_rx(struct work_struct *work); void afs_fetch_data_immediate_cancel(struct afs_call *call); -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size); +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size); /* * flock.c @@ -1266,9 +1266,9 @@ extern int afs_fetch_status(struct afs_vnode *, struct key *, bool, afs_access_t extern int afs_ilookup5_test_by_fid(struct inode *, void *); extern struct inode *afs_iget(struct afs_operation *, struct afs_vnode_param *); extern struct inode *afs_root_iget(struct super_block *, struct key *); -extern int afs_getattr(struct mnt_idmap *idmap, const struct path *, +extern int afs_getattr(const struct mnt_idmap *idmap, const struct path *, struct kstat *, u32, unsigned int); -extern int afs_setattr(struct mnt_idmap *idmap, struct dentry *, struct iattr *); +extern int afs_setattr(const struct mnt_idmap *idmap, struct dentry *, struct iattr *); extern void afs_evict_inode(struct inode *); extern int afs_drop_inode(struct inode *); @@ -1538,7 +1538,7 @@ extern void afs_cache_permit(struct afs_vnode *, struct key *, unsigned int, extern struct key *afs_request_key(struct afs_cell *); extern struct key *afs_request_key_rcu(struct afs_cell *); extern int afs_check_permit(struct afs_vnode *, struct key *, afs_access_t *); -extern int afs_permission(struct mnt_idmap *, struct inode *, int); +extern int afs_permission(const struct mnt_idmap *, struct inode *, int); extern void __exit afs_clean_up_permit_cache(void); /* diff --git a/fs/afs/security.c b/fs/afs/security.c index 6d00d62a65ed..fd040f7c6766 100644 --- a/fs/afs/security.c +++ b/fs/afs/security.c @@ -428,7 +428,7 @@ int afs_check_permit(struct afs_vnode *vnode, struct key *key, * - AFS ACLs are attached to directories only, and a file is controlled by its * parent directory's ACL */ -int afs_permission(struct mnt_idmap *idmap, struct inode *inode, +int afs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct afs_vnode *vnode = AFS_FS_I(inode); diff --git a/fs/afs/xattr.c b/fs/afs/xattr.c index 3770ed236f67..bcffd7236cc9 100644 --- a/fs/afs/xattr.c +++ b/fs/afs/xattr.c @@ -97,7 +97,7 @@ static const struct afs_operation_ops afs_store_acl_operation = { * Set a file's AFS3 ACL. */ static int afs_xattr_set_acl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -228,7 +228,7 @@ static const struct afs_operation_ops yfs_store_opaque_acl2_operation = { * Set a file's YFS ACL. */ static int afs_xattr_set_yfs(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -936,7 +936,7 @@ static int kill_ioctx(struct mm_struct *mm, struct kioctx *ctx, /* * exit_aio: called when the last user of mm goes away. At this point, there is - * no way for any new requests to be submited or any of the io_* syscalls to be + * no way for any new requests to be submitted or any of the io_* syscalls to be * called on the context. * * There may be outstanding kiocbs, but free_ioctx() will explicitly wait on @@ -1280,7 +1280,7 @@ static long aio_read_events_ring(struct kioctx *ctx, * The mutex can block and wake us up and that will cause * wait_event_interruptible_hrtimeout() to schedule without sleeping * and repeat. This should be rare enough that it doesn't cause - * peformance issues. See the comment in read_events() for more detail. + * performance issues. See the comment in read_events() for more detail. */ sched_annotate_sleep(); mutex_lock(&ctx->ring_lock); @@ -1869,7 +1869,12 @@ static int aio_poll_wake(struct wait_queue_entry *wait, unsigned mode, int sync, list_del_init(&req->wait.entry); list_del(&iocb->ki_list); iocb->ki_res.res = mangle_poll(mask); - if (iocb->ki_eventfd && !eventfd_signal_allowed()) { + /* + * We hold an arbitrary provider waitqueue lock here. Signaling a + * result eventfd can feed back through epoll and try to take the same + * lock again. Defer all eventfd-backed poll completions. + */ + if (iocb->ki_eventfd) { iocb = NULL; INIT_WORK(&req->work, aio_poll_put_work); schedule_work(&req->work); diff --git a/fs/anon_inodes.c b/fs/anon_inodes.c index a7b9b948e33d..8f07c8d9bda0 100644 --- a/fs/anon_inodes.c +++ b/fs/anon_inodes.c @@ -46,7 +46,7 @@ static struct inode *anon_inode_inode __ro_after_init; * Rather than mess with our internal sane inode data, just fix it * up here in getattr() by masking off the format bits. */ -int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, +int anon_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -57,7 +57,7 @@ int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int anon_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int anon_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return -EOPNOTSUPP; diff --git a/fs/attr.c b/fs/attr.c index 71888ac903c2..9706d370f34a 100644 --- a/fs/attr.c +++ b/fs/attr.c @@ -30,7 +30,7 @@ * * Return: ATTR_KILL_SGID if setgid bit needs to be removed, 0 otherwise. */ -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode) { umode_t mode = inode->i_mode; @@ -60,7 +60,7 @@ EXPORT_SYMBOL(setattr_should_drop_sgid); * Return: A mask of ATTR_KILL_S{G,U}ID indicating which - if any - setid bits * to remove, 0 otherwise. */ -int setattr_should_drop_suidgid(struct mnt_idmap *idmap, +int setattr_should_drop_suidgid(const struct mnt_idmap *idmap, struct inode *inode) { umode_t mode = inode->i_mode; @@ -91,7 +91,7 @@ EXPORT_SYMBOL(setattr_should_drop_suidgid); * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -static bool chown_ok(struct mnt_idmap *idmap, +static bool chown_ok(const struct mnt_idmap *idmap, const struct inode *inode, vfsuid_t ia_vfsuid) { vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); @@ -118,7 +118,7 @@ static bool chown_ok(struct mnt_idmap *idmap, * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -static bool chgrp_ok(struct mnt_idmap *idmap, +static bool chgrp_ok(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t ia_vfsgid) { vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); @@ -158,7 +158,7 @@ static bool chgrp_ok(struct mnt_idmap *idmap, * Should be called as the first thing in ->setattr implementations, * possibly after taking additional locks. */ -int setattr_prepare(struct mnt_idmap *idmap, struct dentry *dentry, +int setattr_prepare(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -339,7 +339,7 @@ static void setattr_copy_mgtime(struct inode *inode, const struct iattr *attr) * that for "simple" filesystems, the struct inode is the inode storage. * The caller is free to mark the inode dirty afterwards if needed. */ -void setattr_copy(struct mnt_idmap *idmap, struct inode *inode, +void setattr_copy(const struct mnt_idmap *idmap, struct inode *inode, const struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; @@ -369,7 +369,7 @@ void setattr_copy(struct mnt_idmap *idmap, struct inode *inode, } EXPORT_SYMBOL(setattr_copy); -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid) { int error; @@ -424,7 +424,7 @@ EXPORT_SYMBOL(may_setattr); * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -int notify_change(struct mnt_idmap *idmap, struct dentry *dentry, +int notify_change(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct delegated_inode *delegated_inode) { struct inode *inode = dentry->d_inode; diff --git a/fs/autofs/root.c b/fs/autofs/root.c index b36439f4521e..28f38f5d0236 100644 --- a/fs/autofs/root.c +++ b/fs/autofs/root.c @@ -11,12 +11,12 @@ #include "autofs_i.h" -static int autofs_dir_permission(struct mnt_idmap *, struct inode *, int); -static int autofs_dir_symlink(struct mnt_idmap *, struct inode *, +static int autofs_dir_permission(const struct mnt_idmap *, struct inode *, int); +static int autofs_dir_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *); static int autofs_dir_unlink(struct inode *, struct dentry *); static int autofs_dir_rmdir(struct inode *, struct dentry *); -static struct dentry *autofs_dir_mkdir(struct mnt_idmap *, struct inode *, +static struct dentry *autofs_dir_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); static long autofs_root_ioctl(struct file *, unsigned int, unsigned long); #ifdef CONFIG_COMPAT @@ -552,7 +552,7 @@ static struct dentry *autofs_lookup(struct inode *dir, return NULL; } -static int autofs_dir_permission(struct mnt_idmap *idmap, +static int autofs_dir_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { if (mask & MAY_WRITE) { @@ -572,7 +572,7 @@ static int autofs_dir_permission(struct mnt_idmap *idmap, return generic_permission(idmap, inode, mask); } -static int autofs_dir_symlink(struct mnt_idmap *idmap, +static int autofs_dir_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { @@ -724,7 +724,7 @@ static int autofs_dir_rmdir(struct inode *dir, struct dentry *dentry) return 0; } -static struct dentry *autofs_dir_mkdir(struct mnt_idmap *idmap, +static struct dentry *autofs_dir_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { diff --git a/fs/backing-file.c b/fs/backing-file.c index cc101143f921..5614cb7801e1 100644 --- a/fs/backing-file.c +++ b/fs/backing-file.c @@ -59,7 +59,7 @@ struct file *backing_tmpfile_open(const struct file *user_file, int flags, const struct path *real_parentpath, umode_t mode, const struct cred *cred) { - struct mnt_idmap *real_idmap = mnt_idmap(real_parentpath->mnt); + const struct mnt_idmap *real_idmap = mnt_idmap(real_parentpath->mnt); const struct path *user_path = &user_file->f_path; struct file *f; int error; diff --git a/fs/bad_inode.c b/fs/bad_inode.c index 486c40f73e51..bea9f4876ee0 100644 --- a/fs/bad_inode.c +++ b/fs/bad_inode.c @@ -27,7 +27,7 @@ static const struct file_operations bad_file_ops = .open = bad_file_open, }; -static int bad_inode_create(struct mnt_idmap *idmap, +static int bad_inode_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { @@ -51,14 +51,14 @@ static int bad_inode_unlink(struct inode *dir, struct dentry *dentry) return -EIO; } -static int bad_inode_symlink(struct mnt_idmap *idmap, +static int bad_inode_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { return -EIO; } -static struct dentry *bad_inode_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *bad_inode_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(-EIO); @@ -69,13 +69,13 @@ static int bad_inode_rmdir (struct inode *dir, struct dentry *dentry) return -EIO; } -static int bad_inode_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int bad_inode_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { return -EIO; } -static int bad_inode_rename2(struct mnt_idmap *idmap, +static int bad_inode_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -89,20 +89,20 @@ static int bad_inode_readlink(struct dentry *dentry, char __user *buffer, return -EIO; } -static int bad_inode_permission(struct mnt_idmap *idmap, +static int bad_inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return -EIO; } -static int bad_inode_getattr(struct mnt_idmap *idmap, +static int bad_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { return -EIO; } -static int bad_inode_setattr(struct mnt_idmap *idmap, +static int bad_inode_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs) { return -EIO; @@ -146,14 +146,14 @@ static int bad_inode_atomic_open(struct inode *inode, struct dentry *dentry, return -EIO; } -static int bad_inode_tmpfile(struct mnt_idmap *idmap, +static int bad_inode_tmpfile(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, umode_t mode) { return -EIO; } -static int bad_inode_set_acl(struct mnt_idmap *idmap, +static int bad_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { diff --git a/fs/bfs/Kconfig b/fs/bfs/Kconfig deleted file mode 100644 index 8e7ef866b62a..000000000000 --- a/fs/bfs/Kconfig +++ /dev/null @@ -1,21 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -config BFS_FS - tristate "BFS file system support" - depends on BLOCK - select BUFFER_HEAD - help - Boot File System (BFS) is a file system used under SCO UnixWare to - allow the bootloader access to the kernel image and other important - files during the boot process. It is usually mounted under /stand - and corresponds to the slice marked as "STAND" in the UnixWare - partition. You should say Y if you want to read or write the files - on your /stand slice from within Linux. You then also need to say Y - to "UnixWare slices support", below. More information about the BFS - file system is contained in the file - <file:Documentation/filesystems/bfs.rst>. - - If you don't know what this is about, say N. - - To compile this as a module, choose M here: the module will be called - bfs. Note that the file system of your root partition (the one - containing the directory /) cannot be compiled as a module. diff --git a/fs/bfs/Makefile b/fs/bfs/Makefile deleted file mode 100644 index 2b6bc5eb4de9..000000000000 --- a/fs/bfs/Makefile +++ /dev/null @@ -1,8 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -# -# Makefile for BFS filesystem. -# - -obj-$(CONFIG_BFS_FS) += bfs.o - -bfs-objs := inode.o file.o dir.o diff --git a/fs/bfs/bfs.h b/fs/bfs/bfs.h deleted file mode 100644 index b08afe733e63..000000000000 --- a/fs/bfs/bfs.h +++ /dev/null @@ -1,69 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -/* - * fs/bfs/bfs.h - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - */ -#ifndef _FS_BFS_BFS_H -#define _FS_BFS_BFS_H - -#include <linux/bfs_fs.h> - -/* In theory BFS supports up to 512 inodes, numbered from 2 (for /) up to 513 inclusive. - In actual fact, attempting to create the 512th inode (i.e. inode No. 513 or file No. 511) - will fail with ENOSPC in bfs_add_entry(): the root directory cannot contain so many entries, counting '..'. - So, mkfs.bfs(8) should really limit its -N option to 511 and not 512. For now, we just print a warning - if a filesystem is mounted with such "impossible to fill up" number of inodes */ -#define BFS_MAX_LASTI 513 - -/* - * BFS file system in-core superblock info - */ -struct bfs_sb_info { - unsigned long si_blocks; - unsigned long si_freeb; - unsigned long si_freei; - unsigned long si_lf_eblk; - unsigned long si_lasti; - DECLARE_BITMAP(si_imap, BFS_MAX_LASTI+1); - struct mutex bfs_lock; -}; - -/* - * BFS file system in-core inode info - */ -struct bfs_inode_info { - unsigned long i_dsk_ino; /* inode number from the disk, can be 0 */ - unsigned long i_sblock; - unsigned long i_eblock; - struct mapping_metadata_bhs i_metadata_bhs; - struct inode vfs_inode; -}; - -static inline struct bfs_sb_info *BFS_SB(struct super_block *sb) -{ - return sb->s_fs_info; -} - -static inline struct bfs_inode_info *BFS_I(struct inode *inode) -{ - return container_of(inode, struct bfs_inode_info, vfs_inode); -} - - -#define printf(format, args...) \ - printk(KERN_ERR "BFS-fs: %s(): " format, __func__, ## args) - -/* inode.c */ -extern struct inode *bfs_iget(struct super_block *sb, unsigned long ino); -extern void bfs_dump_imap(const char *, struct super_block *); - -/* file.c */ -extern const struct inode_operations bfs_file_inops; -extern const struct file_operations bfs_file_operations; -extern const struct address_space_operations bfs_aops; - -/* dir.c */ -extern const struct inode_operations bfs_dir_inops; -extern const struct file_operations bfs_dir_operations; - -#endif /* _FS_BFS_BFS_H */ diff --git a/fs/bfs/dir.c b/fs/bfs/dir.c index 91a4871fa051..b944bd62f5d0 100644 --- a/fs/bfs/dir.c +++ b/fs/bfs/dir.c @@ -75,7 +75,7 @@ const struct file_operations bfs_dir_operations = { .llseek = generic_file_llseek, }; -static int bfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int bfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int err; @@ -199,7 +199,7 @@ out_brelse: return error; } -static int bfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int bfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/bfs/file.c b/fs/bfs/file.c deleted file mode 100644 index d33d6bde992b..000000000000 --- a/fs/bfs/file.c +++ /dev/null @@ -1,203 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0 -/* - * fs/bfs/file.c - * BFS file operations. - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - * - * Make the file block allocation algorithm understand the size - * of the underlying block device. - * Copyright (C) 2007 Dmitri Vorobiev <dmitri.vorobiev@gmail.com> - * - */ - -#include <linux/fs.h> -#include <linux/mpage.h> -#include <linux/buffer_head.h> -#include "bfs.h" - -#undef DEBUG - -#ifdef DEBUG -#define dprintf(x...) printf(x) -#else -#define dprintf(x...) -#endif - -const struct file_operations bfs_file_operations = { - .llseek = generic_file_llseek, - .read_iter = generic_file_read_iter, - .write_iter = generic_file_write_iter, - .mmap_prepare = generic_file_mmap_prepare, - .splice_read = filemap_splice_read, -}; - -static int bfs_move_block(unsigned long from, unsigned long to, - struct super_block *sb) -{ - struct buffer_head *bh, *new; - - bh = sb_bread(sb, from); - if (!bh) - return -EIO; - new = sb_getblk(sb, to); - memcpy(new->b_data, bh->b_data, bh->b_size); - mark_buffer_dirty(new); - bforget(bh); - brelse(new); - return 0; -} - -static int bfs_move_blocks(struct super_block *sb, unsigned long start, - unsigned long end, unsigned long where) -{ - unsigned long i; - - dprintf("%08lx-%08lx->%08lx\n", start, end, where); - for (i = start; i <= end; i++) - if(bfs_move_block(i, where + i, sb)) { - dprintf("failed to move block %08lx -> %08lx\n", i, - where + i); - return -EIO; - } - return 0; -} - -static int bfs_get_block(struct inode *inode, sector_t block, - struct buffer_head *bh_result, int create) -{ - unsigned long phys; - int err; - struct super_block *sb = inode->i_sb; - struct bfs_sb_info *info = BFS_SB(sb); - struct bfs_inode_info *bi = BFS_I(inode); - - phys = bi->i_sblock + block; - if (!create) { - if (phys <= bi->i_eblock) { - dprintf("c=%d, b=%08lx, phys=%09lx (granted)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - } - return 0; - } - - /* - * If the file is not empty and the requested block is within the - * range of blocks allocated for this file, we can grant it. - */ - if (bi->i_sblock && (phys <= bi->i_eblock)) { - dprintf("c=%d, b=%08lx, phys=%08lx (interim block granted)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - return 0; - } - - /* The file will be extended, so let's see if there is enough space. */ - if (phys >= info->si_blocks) - return -ENOSPC; - - /* The rest has to be protected against itself. */ - mutex_lock(&info->bfs_lock); - - /* - * If the last data block for this file is the last allocated - * block, we can extend the file trivially, without moving it - * anywhere. - */ - if (bi->i_eblock == info->si_lf_eblk) { - dprintf("c=%d, b=%08lx, phys=%08lx (simple extension)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - info->si_freeb -= phys - bi->i_eblock; - info->si_lf_eblk = bi->i_eblock = phys; - mark_inode_dirty(inode); - err = 0; - goto out; - } - - /* Ok, we have to move this entire file to the next free block. */ - phys = info->si_lf_eblk + 1; - if (phys + block >= info->si_blocks) { - err = -ENOSPC; - goto out; - } - - if (bi->i_sblock) { - err = bfs_move_blocks(inode->i_sb, bi->i_sblock, - bi->i_eblock, phys); - if (err) { - dprintf("failed to move ino=%08lx -> fs corruption\n", - inode->i_ino); - goto out; - } - } else - err = 0; - - dprintf("c=%d, b=%08lx, phys=%08lx (moved)\n", - create, (unsigned long)block, phys); - bi->i_sblock = phys; - phys += block; - info->si_lf_eblk = bi->i_eblock = phys; - - /* - * This assumes nothing can write the inode back while we are here - * and thus update inode->i_blocks! (XXX) - */ - info->si_freeb -= bi->i_eblock - bi->i_sblock + 1 - inode->i_blocks; - mark_inode_dirty(inode); - map_bh(bh_result, sb, phys); -out: - mutex_unlock(&info->bfs_lock); - return err; -} - -static int bfs_writepages(struct address_space *mapping, - struct writeback_control *wbc) -{ - return mpage_writepages(mapping, wbc, bfs_get_block); -} - -static int bfs_read_folio(struct file *file, struct folio *folio) -{ - return block_read_full_folio(folio, bfs_get_block); -} - -static void bfs_write_failed(struct address_space *mapping, loff_t to) -{ - struct inode *inode = mapping->host; - - if (to > inode->i_size) - truncate_pagecache(inode, inode->i_size); -} - -static int bfs_write_begin(const struct kiocb *iocb, - struct address_space *mapping, - loff_t pos, unsigned len, - struct folio **foliop, void **fsdata) -{ - int ret; - - ret = block_write_begin(mapping, pos, len, foliop, bfs_get_block); - if (unlikely(ret)) - bfs_write_failed(mapping, pos + len); - - return ret; -} - -static sector_t bfs_bmap(struct address_space *mapping, sector_t block) -{ - return generic_block_bmap(mapping, block, bfs_get_block); -} - -const struct address_space_operations bfs_aops = { - .dirty_folio = block_dirty_folio, - .invalidate_folio = block_invalidate_folio, - .read_folio = bfs_read_folio, - .writepages = bfs_writepages, - .write_begin = bfs_write_begin, - .write_end = generic_write_end, - .migrate_folio = buffer_migrate_folio, - .bmap = bfs_bmap, -}; - -const struct inode_operations bfs_file_inops; diff --git a/fs/bfs/inode.c b/fs/bfs/inode.c deleted file mode 100644 index 06e3a848b4ef..000000000000 --- a/fs/bfs/inode.c +++ /dev/null @@ -1,538 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -/* - * fs/bfs/inode.c - * BFS superblock and inode operations. - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - * From fs/minix, Copyright (C) 1991, 1992 Linus Torvalds. - * Made endianness-clean by Andrew Stribblehill <ads@wompom.org>, 2005. - */ - -#include <linux/module.h> -#include <linux/mm.h> -#include <linux/slab.h> -#include <linux/init.h> -#include <linux/fs.h> -#include <linux/buffer_head.h> -#include <linux/vfs.h> -#include <linux/writeback.h> -#include <linux/uio.h> -#include <linux/uaccess.h> -#include <linux/fs_context.h> -#include "bfs.h" - -MODULE_AUTHOR("Tigran Aivazian <aivazian.tigran@gmail.com>"); -MODULE_DESCRIPTION("SCO UnixWare BFS filesystem for Linux"); -MODULE_LICENSE("GPL"); - -#undef DEBUG - -#ifdef DEBUG -#define dprintf(x...) printf(x) -#else -#define dprintf(x...) -#endif - -struct inode *bfs_iget(struct super_block *sb, unsigned long ino) -{ - struct bfs_inode *di; - struct inode *inode; - struct buffer_head *bh; - int block, off; - - inode = iget_locked(sb, ino); - if (!inode) - return ERR_PTR(-ENOMEM); - if (!(inode_state_read_once(inode) & I_NEW)) - return inode; - - if ((ino < BFS_ROOT_INO) || (ino > BFS_SB(inode->i_sb)->si_lasti)) { - printf("Bad inode number %s:%08lx\n", inode->i_sb->s_id, ino); - goto error; - } - - block = (ino - BFS_ROOT_INO) / BFS_INODES_PER_BLOCK + 1; - bh = sb_bread(inode->i_sb, block); - if (!bh) { - printf("Unable to read inode %s:%08lx\n", inode->i_sb->s_id, - ino); - goto error; - } - - off = (ino - BFS_ROOT_INO) % BFS_INODES_PER_BLOCK; - di = (struct bfs_inode *)bh->b_data + off; - - /* - * https://martin.hinner.info/fs/bfs/bfs-structure.html explains that - * BFS in SCO UnixWare environment used only lower 9 bits of di->i_mode - * value. This means that, although bfs_write_inode() saves whole - * inode->i_mode bits (which include S_IFMT bits and S_IS{UID,GID,VTX} - * bits), middle 7 bits of di->i_mode value can be garbage when these - * bits were not saved by bfs_write_inode(). - * Since we can't tell whether middle 7 bits are garbage, use only - * lower 12 bits (i.e. tolerate S_IS{UID,GID,VTX} bits possibly being - * garbage) and reconstruct S_IFMT bits for Linux environment from - * di->i_vtype value. - */ - inode->i_mode = 0x00000FFF & le32_to_cpu(di->i_mode); - if (le32_to_cpu(di->i_vtype) == BFS_VDIR) { - inode->i_mode |= S_IFDIR; - inode->i_op = &bfs_dir_inops; - inode->i_fop = &bfs_dir_operations; - } else if (le32_to_cpu(di->i_vtype) == BFS_VREG) { - inode->i_mode |= S_IFREG; - inode->i_op = &bfs_file_inops; - inode->i_fop = &bfs_file_operations; - inode->i_mapping->a_ops = &bfs_aops; - } else { - brelse(bh); - printf("Unknown vtype=%u %s:%08lx\n", - le32_to_cpu(di->i_vtype), inode->i_sb->s_id, ino); - goto error; - } - - BFS_I(inode)->i_sblock = le32_to_cpu(di->i_sblock); - BFS_I(inode)->i_eblock = le32_to_cpu(di->i_eblock); - BFS_I(inode)->i_dsk_ino = le16_to_cpu(di->i_ino); - i_uid_write(inode, le32_to_cpu(di->i_uid)); - i_gid_write(inode, le32_to_cpu(di->i_gid)); - set_nlink(inode, le32_to_cpu(di->i_nlink)); - inode->i_size = BFS_FILESIZE(di); - inode->i_blocks = BFS_FILEBLOCKS(di); - inode_set_atime(inode, le32_to_cpu(di->i_atime), 0); - inode_set_mtime(inode, le32_to_cpu(di->i_mtime), 0); - inode_set_ctime(inode, le32_to_cpu(di->i_ctime), 0); - - brelse(bh); - unlock_new_inode(inode); - return inode; - -error: - iget_failed(inode); - return ERR_PTR(-EIO); -} - -static struct bfs_inode *find_inode(struct super_block *sb, u16 ino, struct buffer_head **p) -{ - if ((ino < BFS_ROOT_INO) || (ino > BFS_SB(sb)->si_lasti)) { - printf("Bad inode number %s:%08x\n", sb->s_id, ino); - return ERR_PTR(-EIO); - } - - ino -= BFS_ROOT_INO; - - *p = sb_bread(sb, 1 + ino / BFS_INODES_PER_BLOCK); - if (!*p) { - printf("Unable to read inode %s:%08x\n", sb->s_id, ino); - return ERR_PTR(-EIO); - } - - return (struct bfs_inode *)(*p)->b_data + ino % BFS_INODES_PER_BLOCK; -} - -static int bfs_write_inode(struct inode *inode, struct writeback_control *wbc) -{ - struct bfs_sb_info *info = BFS_SB(inode->i_sb); - unsigned int ino = (u16)inode->i_ino; - unsigned long i_sblock; - struct bfs_inode *di; - struct buffer_head *bh; - - dprintf("ino=%08x\n", ino); - - di = find_inode(inode->i_sb, ino, &bh); - if (IS_ERR(di)) - return PTR_ERR(di); - - mutex_lock(&info->bfs_lock); - - if (ino == BFS_ROOT_INO) - di->i_vtype = cpu_to_le32(BFS_VDIR); - else - di->i_vtype = cpu_to_le32(BFS_VREG); - - di->i_ino = cpu_to_le16(ino); - di->i_mode = cpu_to_le32(inode->i_mode); - di->i_uid = cpu_to_le32(i_uid_read(inode)); - di->i_gid = cpu_to_le32(i_gid_read(inode)); - di->i_nlink = cpu_to_le32(inode->i_nlink); - di->i_atime = cpu_to_le32(inode_get_atime_sec(inode)); - di->i_mtime = cpu_to_le32(inode_get_mtime_sec(inode)); - di->i_ctime = cpu_to_le32(inode_get_ctime_sec(inode)); - i_sblock = BFS_I(inode)->i_sblock; - di->i_sblock = cpu_to_le32(i_sblock); - di->i_eblock = cpu_to_le32(BFS_I(inode)->i_eblock); - di->i_eoffset = cpu_to_le32(i_sblock * BFS_BSIZE + inode->i_size - 1); - - mark_buffer_dirty(bh); - brelse(bh); - mutex_unlock(&info->bfs_lock); - set_inode_metadata_writeback(inode); - return 0; -} - -static int bfs_sync_inode_metadata(struct inode *inode, - struct writeback_control *wbc) -{ - int err = 0; - struct bfs_inode *di; - struct buffer_head *bh; - - di = find_inode(inode->i_sb, (u16)inode->i_ino, &bh); - if (IS_ERR(di)) - return PTR_ERR(di); - - sync_dirty_buffer(bh); - if (buffer_write_io_error(bh)) { - err = -EIO; - goto out; - } - err = mmb_sync(&BFS_I(inode)->i_metadata_bhs); -out: - brelse(bh); - return err; -} - -static void bfs_evict_inode(struct inode *inode) -{ - unsigned long ino = inode->i_ino; - struct bfs_inode *di; - struct buffer_head *bh; - struct super_block *s = inode->i_sb; - struct bfs_sb_info *info = BFS_SB(s); - struct bfs_inode_info *bi = BFS_I(inode); - - dprintf("ino=%08lx\n", ino); - - truncate_inode_pages_final(&inode->i_data); - if (inode->i_nlink) - mmb_sync(&BFS_I(inode)->i_metadata_bhs); - mmb_invalidate(&BFS_I(inode)->i_metadata_bhs); - clear_inode(inode); - - if (inode->i_nlink) - return; - - di = find_inode(s, inode->i_ino, &bh); - if (IS_ERR(di)) - return; - - mutex_lock(&info->bfs_lock); - /* clear on-disk inode */ - memset(di, 0, sizeof(struct bfs_inode)); - mark_buffer_dirty(bh); - brelse(bh); - - if (bi->i_dsk_ino) { - if (bi->i_sblock) - info->si_freeb += bi->i_eblock + 1 - bi->i_sblock; - info->si_freei++; - clear_bit(ino, info->si_imap); - bfs_dump_imap("evict_inode", s); - } - - /* - * If this was the last file, make the previous block - * "last block of the last file" even if there is no - * real file there, saves us 1 gap. - */ - if (info->si_lf_eblk == bi->i_eblock) - info->si_lf_eblk = bi->i_sblock - 1; - mutex_unlock(&info->bfs_lock); -} - -static void bfs_put_super(struct super_block *s) -{ - struct bfs_sb_info *info = BFS_SB(s); - - if (!info) - return; - - mutex_destroy(&info->bfs_lock); - kfree(info); - s->s_fs_info = NULL; -} - -static int bfs_statfs(struct dentry *dentry, struct kstatfs *buf) -{ - struct super_block *s = dentry->d_sb; - struct bfs_sb_info *info = BFS_SB(s); - u64 id = huge_encode_dev(s->s_bdev->bd_dev); - buf->f_type = BFS_MAGIC; - buf->f_bsize = s->s_blocksize; - buf->f_blocks = info->si_blocks; - buf->f_bfree = buf->f_bavail = info->si_freeb; - buf->f_files = info->si_lasti + 1 - BFS_ROOT_INO; - buf->f_ffree = info->si_freei; - buf->f_fsid = u64_to_fsid(id); - buf->f_namelen = BFS_NAMELEN; - return 0; -} - -static struct kmem_cache *bfs_inode_cachep; - -static struct inode *bfs_alloc_inode(struct super_block *sb) -{ - struct bfs_inode_info *bi; - bi = alloc_inode_sb(sb, bfs_inode_cachep, GFP_KERNEL); - if (!bi) - return NULL; - mmb_init(&bi->i_metadata_bhs, &bi->vfs_inode.i_data); - - return &bi->vfs_inode; -} - -static void bfs_free_inode(struct inode *inode) -{ - kmem_cache_free(bfs_inode_cachep, BFS_I(inode)); -} - -static void init_once(void *foo) -{ - struct bfs_inode_info *bi = foo; - - inode_init_once(&bi->vfs_inode); -} - -static int __init init_inodecache(void) -{ - bfs_inode_cachep = kmem_cache_create("bfs_inode_cache", - sizeof(struct bfs_inode_info), - 0, (SLAB_RECLAIM_ACCOUNT| - SLAB_ACCOUNT), - init_once); - if (bfs_inode_cachep == NULL) - return -ENOMEM; - return 0; -} - -static void destroy_inodecache(void) -{ - /* - * Make sure all delayed rcu free inodes are flushed before we - * destroy cache. - */ - rcu_barrier(); - kmem_cache_destroy(bfs_inode_cachep); -} - -static const struct super_operations bfs_sops = { - .alloc_inode = bfs_alloc_inode, - .free_inode = bfs_free_inode, - .write_inode = bfs_write_inode, - .sync_inode_metadata = bfs_sync_inode_metadata, - .evict_inode = bfs_evict_inode, - .put_super = bfs_put_super, - .statfs = bfs_statfs, -}; - -void bfs_dump_imap(const char *prefix, struct super_block *s) -{ -#ifdef DEBUG - int i; - char *tmpbuf = kzalloc(PAGE_SIZE, GFP_KERNEL); - - if (!tmpbuf) - return; - for (i = BFS_SB(s)->si_lasti; i >= 0; i--) { - if (i > PAGE_SIZE - 100) break; - if (test_bit(i, BFS_SB(s)->si_imap)) - strcat(tmpbuf, "1"); - else - strcat(tmpbuf, "0"); - } - printf("%s: lasti=%08lx <%s>\n", prefix, BFS_SB(s)->si_lasti, tmpbuf); - kfree(tmpbuf); -#endif -} - -static int bfs_fill_super(struct super_block *s, struct fs_context *fc) -{ - struct buffer_head *bh, *sbh; - struct bfs_super_block *bfs_sb; - struct inode *inode; - unsigned i; - struct bfs_sb_info *info; - int ret = -EINVAL; - unsigned long i_sblock, i_eblock, i_eoff, s_size; - int silent = fc->sb_flags & SB_SILENT; - - info = kzalloc_obj(*info); - if (!info) - return -ENOMEM; - mutex_init(&info->bfs_lock); - s->s_fs_info = info; - s->s_time_min = 0; - s->s_time_max = U32_MAX; - - if (!sb_set_blocksize(s, BFS_BSIZE)) - goto out; - - sbh = sb_bread(s, 0); - if (!sbh) - goto out; - bfs_sb = (struct bfs_super_block *)sbh->b_data; - if (le32_to_cpu(bfs_sb->s_magic) != BFS_MAGIC) { - if (!silent) - printf("No BFS filesystem on %s (magic=%08x)\n", s->s_id, le32_to_cpu(bfs_sb->s_magic)); - goto out1; - } - if (BFS_UNCLEAN(bfs_sb, s) && !silent) - printf("%s is unclean, continuing\n", s->s_id); - - s->s_magic = BFS_MAGIC; - - if (le32_to_cpu(bfs_sb->s_start) > le32_to_cpu(bfs_sb->s_end) || - le32_to_cpu(bfs_sb->s_start) < sizeof(struct bfs_super_block) + sizeof(struct bfs_dirent)) { - printf("Superblock is corrupted on %s\n", s->s_id); - goto out1; - } - - info->si_lasti = (le32_to_cpu(bfs_sb->s_start) - BFS_BSIZE) / sizeof(struct bfs_inode) + BFS_ROOT_INO - 1; - if (info->si_lasti == BFS_MAX_LASTI) - printf("NOTE: filesystem %s was created with 512 inodes, the real maximum is 511, mounting anyway\n", s->s_id); - else if (info->si_lasti > BFS_MAX_LASTI) { - printf("Impossible last inode number %lu > %d on %s\n", info->si_lasti, BFS_MAX_LASTI, s->s_id); - goto out1; - } - for (i = 0; i < BFS_ROOT_INO; i++) - set_bit(i, info->si_imap); - - s->s_op = &bfs_sops; - inode = bfs_iget(s, BFS_ROOT_INO); - if (IS_ERR(inode)) { - ret = PTR_ERR(inode); - goto out1; - } - s->s_root = d_make_root(inode); - if (!s->s_root) { - ret = -ENOMEM; - goto out1; - } - - info->si_blocks = (le32_to_cpu(bfs_sb->s_end) + 1) >> BFS_BSIZE_BITS; - info->si_freeb = (le32_to_cpu(bfs_sb->s_end) + 1 - le32_to_cpu(bfs_sb->s_start)) >> BFS_BSIZE_BITS; - info->si_freei = 0; - info->si_lf_eblk = 0; - - /* can we read the last block? */ - bh = sb_bread(s, info->si_blocks - 1); - if (!bh) { - printf("Last block not available on %s: %lu\n", s->s_id, info->si_blocks - 1); - ret = -EIO; - goto out2; - } - brelse(bh); - - bh = NULL; - for (i = BFS_ROOT_INO; i <= info->si_lasti; i++) { - struct bfs_inode *di; - int block = (i - BFS_ROOT_INO) / BFS_INODES_PER_BLOCK + 1; - int off = (i - BFS_ROOT_INO) % BFS_INODES_PER_BLOCK; - unsigned long eblock; - - if (!off) { - brelse(bh); - bh = sb_bread(s, block); - } - - if (!bh) - continue; - - di = (struct bfs_inode *)bh->b_data + off; - - /* test if filesystem is not corrupted */ - - i_eoff = le32_to_cpu(di->i_eoffset); - i_sblock = le32_to_cpu(di->i_sblock); - i_eblock = le32_to_cpu(di->i_eblock); - s_size = le32_to_cpu(bfs_sb->s_end); - - if (i_sblock > info->si_blocks || - i_eblock > info->si_blocks || - i_sblock > i_eblock || - (i_eoff != le32_to_cpu(-1) && i_eoff > s_size) || - i_sblock * BFS_BSIZE > i_eoff) { - - printf("Inode 0x%08x corrupted on %s\n", i, s->s_id); - - brelse(bh); - ret = -EIO; - goto out2; - } - - if (!di->i_ino) { - info->si_freei++; - continue; - } - set_bit(i, info->si_imap); - info->si_freeb -= BFS_FILEBLOCKS(di); - - eblock = le32_to_cpu(di->i_eblock); - if (eblock > info->si_lf_eblk) - info->si_lf_eblk = eblock; - } - brelse(bh); - brelse(sbh); - bfs_dump_imap("fill_super", s); - return 0; - -out2: - dput(s->s_root); - s->s_root = NULL; -out1: - brelse(sbh); -out: - mutex_destroy(&info->bfs_lock); - kfree(info); - s->s_fs_info = NULL; - return ret; -} - -static int bfs_get_tree(struct fs_context *fc) -{ - return get_tree_bdev(fc, bfs_fill_super); -} - -static const struct fs_context_operations bfs_context_ops = { - .get_tree = bfs_get_tree, -}; - -static int bfs_init_fs_context(struct fs_context *fc) -{ - fc->ops = &bfs_context_ops; - - return 0; -} - -static struct file_system_type bfs_fs_type = { - .owner = THIS_MODULE, - .name = "bfs", - .init_fs_context = bfs_init_fs_context, - .kill_sb = kill_block_super, - .fs_flags = FS_REQUIRES_DEV, -}; -MODULE_ALIAS_FS("bfs"); - -static int __init init_bfs_fs(void) -{ - int err = init_inodecache(); - if (err) - goto out1; - err = register_filesystem(&bfs_fs_type); - if (err) - goto out; - return 0; -out: - destroy_inodecache(); -out1: - return err; -} - -static void __exit exit_bfs_fs(void) -{ - unregister_filesystem(&bfs_fs_type); - destroy_inodecache(); -} - -module_init(init_bfs_fs) -module_exit(exit_bfs_fs) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index 06d0df105382..bf7f8f47548d 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -74,7 +74,7 @@ static int load_elf_binary(struct linux_binprm *bprm); * don't even try. */ #ifdef CONFIG_ELF_CORE -static int elf_core_dump(struct coredump_params *cprm); +static bool elf_core_dump(struct coredump_params *cprm); #else #define elf_core_dump NULL #endif @@ -1875,7 +1875,7 @@ static int fill_note_info(struct elfhdr *elf, int phdrs, return 0; info->thread->task = dump_task; - for (ct = dump_task->signal->core_state->dumper.next; ct; ct = ct->next) { + for (ct = dump_task->signal->core_state->tasks; ct; ct = ct->next) { t = kzalloc_flex(*t, notes, info->thread_notes); if (unlikely(!t)) return 0; @@ -1987,9 +1987,9 @@ static void fill_extnum_info(struct elfhdr *elf, struct elf_shdr *shdr4extnum, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_core_dump(struct coredump_params *cprm) +static bool elf_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs, i; struct elfhdr elf; loff_t offset = 0, dataoff; @@ -2020,7 +2020,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!fill_note_info(&elf, e_phnum, &info, cprm)) goto end_coredump; - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; offset += sizeof(elf); /* ELF header */ offset += segs * sizeof(struct elf_phdr); /* Program headers */ @@ -2029,7 +2029,7 @@ static int elf_core_dump(struct coredump_params *cprm) { size_t sz = info.size; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ sz += elf_coredump_extra_notes_size(); phdr4note = kmalloc_obj(*phdr4note); @@ -2093,7 +2093,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!write_note_info(&info, cprm)) goto end_coredump; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ if (elf_coredump_extra_notes_write(cprm)) goto end_coredump; @@ -2115,11 +2115,13 @@ static int elf_core_dump(struct coredump_params *cprm) goto end_coredump; } + ret = true; + end_coredump: free_note_info(&info); kfree(shdr4extnum); kfree(phdr4note); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/binfmt_elf_fdpic.c b/fs/binfmt_elf_fdpic.c index 068c46875c74..d3872169f55e 100644 --- a/fs/binfmt_elf_fdpic.c +++ b/fs/binfmt_elf_fdpic.c @@ -75,7 +75,7 @@ static int elf_fdpic_map_file_by_direct_mmap(struct elf_fdpic_params *, struct file *, struct mm_struct *); #ifdef CONFIG_ELF_CORE -static int elf_fdpic_core_dump(struct coredump_params *cprm); +static bool elf_fdpic_core_dump(struct coredump_params *cprm); #endif static struct linux_binfmt elf_fdpic_format = { @@ -1477,9 +1477,9 @@ static bool elf_fdpic_dump_segments(struct coredump_params *cprm, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_fdpic_core_dump(struct coredump_params *cprm) +static bool elf_fdpic_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs; int i; struct elfhdr *elf = NULL; @@ -1504,7 +1504,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) if (!psinfo) goto end_coredump; - for (ct = current->signal->core_state->dumper.next; + for (ct = current->signal->core_state->tasks; ct; ct = ct->next) { tmp = elf_dump_thread_status(cprm->siginfo->si_signo, ct->task, &thread_status_size); @@ -1536,7 +1536,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) /* Set up header */ fill_elf_fdpic_header(elf, e_phnum); - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; /* * Set up the notes in similar form to SVR4 core dumps made * with info from their /proc. @@ -1656,6 +1656,8 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) cprm->file->f_pos, offset); } + ret = true; + end_coredump: while (thread_list) { tmp = thread_list; @@ -1666,7 +1668,7 @@ end_coredump: kfree(elf); kfree(psinfo); kfree(shdr4extnum); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/binfmt_misc.c b/fs/binfmt_misc.c index 620da85948b4..d945b4f6e158 100644 --- a/fs/binfmt_misc.c +++ b/fs/binfmt_misc.c @@ -100,6 +100,13 @@ static const struct binfmt_misc_flag *misc_flag_by_char(const char c) return NULL; } +static bool misc_valid_delim(const char c) +{ + if (!isascii(c) || !ispunct(c)) + return false; + return c != '\\'; +} + struct binfmt_misc_entry { struct hlist_node node; unsigned long flags; /* type, status, etc. */ @@ -871,10 +878,9 @@ static struct binfmt_misc_entry *create_entry(const char __user *buffer, del = *p++; /* delimiter */ - pr_debug("register: delim: %#x {%c}\n", del, del); + pr_debug("register: delim: %#x\n", del); - /* A flag-char delimiter runs the flag scan off the buffer. */ - if (misc_flag_by_char(del)) + if (!misc_valid_delim(del)) return ERR_PTR(-EINVAL); /* Pad the buffer with the delim to simplify parsing below. */ diff --git a/fs/bpf_fs_kfuncs.c b/fs/bpf_fs_kfuncs.c index 357a379ef92a..abdfbd83dc57 100644 --- a/fs/bpf_fs_kfuncs.c +++ b/fs/bpf_fs_kfuncs.c @@ -237,7 +237,7 @@ int bpf_set_dentry_xattr_locked(struct dentry *dentry, const char *name__str, * @dentry: dentry to get xattr from * @name__str: name of the xattr * - * Rmove xattr *name__str* of *dentry*. + * Remove xattr *name__str* of *dentry*. * * For security reasons, only *name__str* with prefix "security.bpf." * is allowed. @@ -305,7 +305,7 @@ __bpf_kfunc int bpf_set_dentry_xattr(struct dentry *dentry, const char *name__st * @dentry: dentry to get xattr from * @name__str: name of the xattr * - * Rmove xattr *name__str* of *dentry*. + * Remove xattr *name__str* of *dentry*. * * For security reasons, only *name__str* with prefix "security.bpf." * is allowed. diff --git a/fs/btrfs/acl.c b/fs/btrfs/acl.c index 662cdd1cbdef..10a0d733bfd1 100644 --- a/fs/btrfs/acl.c +++ b/fs/btrfs/acl.c @@ -101,7 +101,7 @@ int __btrfs_set_acl(struct btrfs_trans_handle *trans, struct inode *inode, return 0; } -int btrfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int btrfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int ret; diff --git a/fs/btrfs/acl.h b/fs/btrfs/acl.h index 0458cd51ed48..6eae2db3654d 100644 --- a/fs/btrfs/acl.h +++ b/fs/btrfs/acl.h @@ -15,7 +15,7 @@ struct mnt_idmap; struct dentry; struct posix_acl *btrfs_get_acl(struct inode *inode, int type, bool rcu); -int btrfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int btrfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int __btrfs_set_acl(struct btrfs_trans_handle *trans, struct inode *inode, struct posix_acl *acl, int type); diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 114a5c38afd3..b673851d8d2e 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -555,7 +555,7 @@ int btrfs_new_inode_prepare(struct btrfs_new_inode_args *args, int btrfs_create_new_inode(struct btrfs_trans_handle *trans, struct btrfs_new_inode_args *args); void btrfs_new_inode_args_destroy(struct btrfs_new_inode_args *args); -struct inode *btrfs_new_subvol_inode(struct mnt_idmap *idmap, +struct inode *btrfs_new_subvol_inode(const struct mnt_idmap *idmap, struct inode *dir); void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *state, u32 bits); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index f4b68205f621..1d79f5263de3 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -5428,7 +5428,7 @@ static int btrfs_setsize(struct inode *inode, struct iattr *attr) return ret; } -static int btrfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int btrfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -6976,7 +6976,7 @@ out_inode: return ret; } -static int btrfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -6990,7 +6990,7 @@ static int btrfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return btrfs_create_common(dir, dentry, inode); } -static int btrfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -7087,7 +7087,7 @@ fail: return ret; } -static struct dentry *btrfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *btrfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -7984,7 +7984,7 @@ out: return ret; } -struct inode *btrfs_new_subvol_inode(struct mnt_idmap *idmap, +struct inode *btrfs_new_subvol_inode(const struct mnt_idmap *idmap, struct inode *dir) { struct inode *inode; @@ -8178,7 +8178,7 @@ int __init btrfs_init_cachep(void) return 0; } -static int btrfs_getattr(struct mnt_idmap *idmap, +static int btrfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -8494,7 +8494,7 @@ out_notrans: return ret; } -static struct inode *new_whiteout_inode(struct mnt_idmap *idmap, +static struct inode *new_whiteout_inode(const struct mnt_idmap *idmap, struct inode *dir) { struct inode *inode; @@ -8509,7 +8509,7 @@ static struct inode *new_whiteout_inode(struct mnt_idmap *idmap, return inode; } -static int btrfs_rename(struct mnt_idmap *idmap, +static int btrfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -8809,7 +8809,7 @@ out_fscrypt_names: return ret; } -static int btrfs_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +static int btrfs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -8997,7 +8997,7 @@ out: return ret; } -static int btrfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct btrfs_fs_info *fs_info = inode_to_fs_info(dir); @@ -9313,7 +9313,7 @@ next: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to * elide calls here. */ -static int btrfs_permission(struct mnt_idmap *idmap, +static int btrfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct btrfs_root *root = BTRFS_I(inode)->root; @@ -9329,7 +9329,7 @@ static int btrfs_permission(struct mnt_idmap *idmap, return generic_permission(idmap, inode, mask); } -static int btrfs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct btrfs_fs_info *fs_info = inode_to_fs_info(dir); diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 52aab510aea0..f3e2afe221be 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -278,7 +278,7 @@ int btrfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int btrfs_fileattr_set(struct mnt_idmap *idmap, +int btrfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct btrfs_inode *inode = BTRFS_I(d_inode(dentry)); @@ -549,7 +549,7 @@ static unsigned int create_subvol_num_items(const struct btrfs_qgroup_inherit *i return num_items; } -static noinline int create_subvol(struct mnt_idmap *idmap, +static noinline int create_subvol(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct btrfs_qgroup_inherit *inherit) { @@ -879,7 +879,7 @@ free_pending: * inside this filesystem so it's quite a bit simpler. */ static noinline int btrfs_mksubvol(struct dentry *parent, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct qstr *qname, struct btrfs_root *snap_src, bool readonly, struct btrfs_qgroup_inherit *inherit) @@ -926,7 +926,7 @@ out_dput: } static noinline int btrfs_mksnapshot(struct dentry *parent, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct qstr *qname, struct btrfs_root *root, bool readonly, @@ -1164,7 +1164,7 @@ static noinline int __btrfs_ioctl_snap_create(struct file *file, { int ret; struct qstr qname = QSTR(name); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); if (!S_ISDIR(file_inode(file)->i_mode)) return -ENOTDIR; @@ -1741,7 +1741,7 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_root *root, u64 dirid return 0; } -static int btrfs_search_path_in_tree_user(struct mnt_idmap *idmap, +static int btrfs_search_path_in_tree_user(const struct mnt_idmap *idmap, struct inode *inode, struct btrfs_ioctl_ino_lookup_user_args *args) { @@ -2241,7 +2241,7 @@ static noinline int btrfs_ioctl_snap_destroy(struct file *file, struct btrfs_root *dest = NULL; struct btrfs_ioctl_vol_args AUTO_KFREE(vol_args); struct btrfs_ioctl_vol_args_v2 AUTO_KFREE(vol_args2); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); char *subvol_name, *subvol_name_ptr = NULL; int ret = 0; bool destroy_parent = false; @@ -3901,7 +3901,7 @@ static long btrfs_ioctl_quota_rescan_wait(struct btrfs_fs_info *fs_info) } static long _btrfs_ioctl_set_received_subvol(struct file *file, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct btrfs_ioctl_received_subvol_args *sa) { struct inode *inode = file_inode(file); diff --git a/fs/btrfs/ioctl.h b/fs/btrfs/ioctl.h index ccf6bed9cc24..55f86aeb3500 100644 --- a/fs/btrfs/ioctl.h +++ b/fs/btrfs/ioctl.h @@ -17,7 +17,7 @@ struct btrfs_ioctl_balance_args; long btrfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg); long btrfs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); int btrfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int btrfs_fileattr_set(struct mnt_idmap *idmap, +int btrfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int btrfs_ioctl_get_supported_features(void __user *arg); void btrfs_sync_inode_flags_to_i_flags(struct btrfs_inode *inode); diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index ab55d10bd71f..a06420b9c662 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -353,7 +353,7 @@ static int btrfs_xattr_handler_get(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -395,7 +395,7 @@ static int btrfs_xattr_handler_get_security(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set_security(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, @@ -413,7 +413,7 @@ static int btrfs_xattr_handler_set_security(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set_prop(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/buffer.c b/fs/buffer.c index ed966fa73b1b..1dc933ba6925 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -203,11 +203,10 @@ void bh_end_write(struct bio *bio) bool success = bio_endio_bh(bio, &bh); if (success) { - set_buffer_uptodate(bh); + clear_buffer_write_io_error(bh); } else { buffer_io_error(bh, ", lost sync page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } unlock_buffer(bh); } @@ -265,7 +264,7 @@ __find_get_block_slow(struct block_device *bdev, sector_t block, bool atomic) bh = bh->b_this_page; } while (bh != head); - /* we might be here because some of the buffers on this page are + /* we might be here because some of the buffers on this folio are * not mapped. This is due to various races between * file io on the block device and getblk. It gets dealt with * elsewhere, don't buffer_error if we had some unmapped buffers @@ -311,7 +310,7 @@ static void end_buffer_async_read(struct buffer_head *bh, int uptodate) /* * Be _very_ careful from here on. Bad things can happen if * two buffer heads end IO at almost the same time and both - * decide that the page is now completely done. + * decide that the folio is now completely done. */ first = folio_buffers(folio); spin_lock_irqsave(&first->b_uptodate_lock, flags); @@ -408,11 +407,10 @@ void bh_end_async_write(struct bio *bio) folio = bh->b_folio; if (success) { - set_buffer_uptodate(bh); + clear_buffer_write_io_error(bh); } else { buffer_io_error(bh, ", lost async page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } first = folio_buffers(folio); @@ -520,8 +518,8 @@ EXPORT_SYMBOL_GPL(mmb_has_buffers); * * Do this in two main stages: first we copy dirty buffers to a * temporary inode list, queueing the writes as we go. Then we clean - * up, waiting for those writes to complete. mark_buffer_dirty_inode() - * doesn't touch b_assoc_buffers list if b_mmb is not NULL so we are sure the + * up, waiting for those writes to complete. mmb_mark_buffer_dirty() + * doesn't touch b_assoc_buffers list if b_mmb is set so we are sure the * buffer stays on our list until IO completes (at which point it can be * reaped). */ @@ -542,7 +540,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) bh = BH_ENTRY(mmb->list.next); WARN_ON_ONCE(bh->b_mmb != mmb); __remove_assoc_queue(mmb, bh); - /* Avoid race with mark_buffer_dirty_inode() which does + /* Avoid race with mmb_mark_buffer_dirty() which does * a lockless check and we rely on seeing the dirty bit */ smp_mb(); if (buffer_dirty(bh) || buffer_locked(bh)) { @@ -580,7 +578,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) bh = BH_ENTRY(tmp.prev); get_bh(bh); __remove_assoc_queue(mmb, bh); - /* Avoid race with mark_buffer_dirty_inode() which does + /* Avoid race with mmb_mark_buffer_dirty() which does * a lockless check and we rely on seeing the dirty bit */ smp_mb(); if (buffer_dirty(bh)) { @@ -589,7 +587,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) } spin_unlock(&mmb->lock); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; brelse(bh); spin_lock(&mmb->lock); @@ -618,6 +616,14 @@ void write_boundary_block(struct block_device *bdev, } } +/** + * mmb_mark_buffer_dirty - Mark a metadata buffer dirty. + * @bh: The buffer to mark dirty. + * @mmb: The list of buffers to add the buffer to. + * + * Mark the buffer dirty and add it to the list if it is not already on + * a list. + */ void mmb_mark_buffer_dirty(struct buffer_head *bh, struct mapping_metadata_bhs *mmb) { @@ -686,7 +692,7 @@ bool block_dirty_folio(struct address_space *mapping, struct folio *folio) } while (bh != head); } /* - * Lock out page's memcg migration to keep PageDirty + * Lock out folio's memcg migration to keep folio dirty flag * synchronized with per-memcg dirty page counters. */ newly_dirty = !folio_test_set_dirty(folio); @@ -952,23 +958,23 @@ __getblk_slow(struct block_device *bdev, sector_t block, } /* - * The relationship between dirty buffers and dirty pages: + * The relationship between dirty buffers and dirty folios: * - * Whenever a page has any dirty buffers, the page's dirty bit is set, and - * the page is tagged dirty in the page cache. + * Whenever a folio has any dirty buffers, the folio's dirty flag is set, and + * the folio is tagged dirty in the page cache. * * At all times, the dirtiness of the buffers represents the dirtiness of - * subsections of the page. If the page has buffers, the page dirty bit is + * subsections of the folio. If the folio has buffers, the folio dirty flag is * merely a hint about the true dirty state. * - * When a page is set dirty in its entirety, all its buffers are marked dirty - * (if the page has buffers). + * When a folio is set dirty in its entirety, all its buffers are marked dirty + * (if the folio has buffers). * - * When a buffer is marked dirty, its page is dirtied, but the page's other + * When a buffer is marked dirty, its folio is dirtied, but the folio's other * buffers are not. * * Also. When blockdev buffers are explicitly read with bread(), they - * individually become uptodate. But their backing page remains not + * individually become uptodate. But their backing folio remains not * uptodate - even if all of its buffers are uptodate. A subsequent * block_read_full_folio() against that folio will discover all the uptodate * buffers, will set the folio uptodate and will perform no I/O. @@ -979,7 +985,7 @@ __getblk_slow(struct block_device *bdev, sector_t block, * @bh: the buffer_head to mark dirty * * mark_buffer_dirty() will set the dirty bit against the buffer, then set - * its backing page dirty, then tag the page as dirty in the page cache + * its backing folio dirty, then tag the folio as dirty in the page cache * and then attach the address_space's inode to its superblock's dirty * inode list. * @@ -1062,6 +1068,7 @@ EXPORT_SYMBOL(__brelse); void __bforget(struct buffer_head *bh) { clear_buffer_dirty(bh); + clear_buffer_write_io_error(bh); remove_assoc_queue(bh); __brelse(bh); } @@ -1070,12 +1077,16 @@ EXPORT_SYMBOL(__bforget); static void buffer_set_crypto_ctx(struct bio *bio, const struct buffer_head *bh, gfp_t gfp_mask) { - const struct address_space *mapping = folio_mapping(bh->b_folio); + const struct address_space *mapping; /* * The ext4 journal (jbd2) can submit a buffer_head it directly created - * for a non-pagecache page. fscrypt doesn't care about these. + * for memory that is not in the page cache at all. fscrypt doesn't + * care about these. */ + if (!bh->b_folio) + return; + mapping = bh->b_folio->mapping; if (!mapping) return; fscrypt_set_bio_crypt_ctx(bio, mapping->host, @@ -1086,7 +1097,6 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, enum rw_hint write_hint, struct writeback_control *wbc, bio_end_io_t end_bio) { - const enum req_op op = opf & REQ_OP_MASK; struct bio *bio; BUG_ON(!buffer_locked(bh)); @@ -1094,11 +1104,7 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, BUG_ON(buffer_delay(bh)); BUG_ON(buffer_unwritten(bh)); - /* - * Only clear out a write error when rewriting - */ - if (test_set_buffer_req(bh) && (op == REQ_OP_WRITE)) - clear_buffer_write_io_error(bh); + set_buffer_req(bh); if (buffer_meta(bh)) opf |= REQ_META; @@ -1107,7 +1113,8 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, bio = bio_alloc(bh->b_bdev, 1, opf, GFP_NOIO); - if (folio_test_dropbehind(bh->b_folio) && op_is_write(opf)) + if (bh->b_folio && folio_test_dropbehind(bh->b_folio) && + op_is_write(opf)) bio_set_flag(bio, BIO_COMPLETE_IN_TASK); if (IS_ENABLED(CONFIG_FS_ENCRYPTION)) @@ -1116,7 +1123,11 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, bio->bi_iter.bi_sector = bh->b_blocknr * (bh->b_size >> 9); bio->bi_write_hint = write_hint; - bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, bh_offset(bh)); + if (bh->b_folio) + bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, + bh_offset(bh)); + else + bio_add_virt_nofail(bio, bh->b_data, bh->b_size); bio->bi_end_io = end_bio; bio->bi_private = bh; @@ -1126,7 +1137,8 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, if (wbc) { wbc_init_bio(wbc, bio); - wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); + if (bh->b_folio) + wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); } blk_crypto_submit_bio(bio); @@ -1216,7 +1228,7 @@ static void bh_lru_install(struct buffer_head *bh) /* * the refcount of buffer_head in bh_lru prevents dropping the - * attached page(i.e., try_to_free_buffers) so it could cause + * attached folio (i.e., try_to_free_buffers) so it could cause * failing page migration. * Skip putting upcoming bh into bh_lru until migration is done. */ @@ -1280,7 +1292,7 @@ lookup_bh_lru(struct block_device *bdev, sector_t block, unsigned size) * Perform a pagecache lookup for the matching buffer. If it's there, refresh * it in the LRU and mark it as accessed. If it is not present then return * NULL. Atomic context callers may also return NULL if the buffer is being - * migrated; similarly the page is not marked accessed either. + * migrated; similarly the folio is not marked accessed either. */ static struct buffer_head * find_get_block_common(struct block_device *bdev, sector_t block, @@ -1289,7 +1301,7 @@ find_get_block_common(struct block_device *bdev, sector_t block, struct buffer_head *bh = lookup_bh_lru(bdev, block, size); if (bh == NULL) { - /* __find_get_block_slow will mark the page accessed */ + /* __find_get_block_slow will mark the folio accessed */ bh = __find_get_block_slow(bdev, block, atomic); if (bh) bh_lru_install(bh); @@ -1475,15 +1487,14 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio, } EXPORT_SYMBOL(folio_set_bh); -/* - * Called when truncating a buffer on a page completely. - */ - /* Bits that are cleared during an invalidate */ #define BUFFER_FLAGS_DISCARD \ (1 << BH_Mapped | 1 << BH_New | 1 << BH_Req | \ - 1 << BH_Delay | 1 << BH_Unwritten) + 1 << BH_Delay | 1 << BH_Unwritten | 1 << BH_Write_EIO) +/* + * Called when truncating a buffer on a folio completely. + */ static void discard_buffer(struct buffer_head * bh) { unsigned long b_state; @@ -1611,9 +1622,7 @@ EXPORT_SYMBOL(create_empty_buffers); * moment when something will explicitly mark the buffer dirty (hopefully that * will not happen until we will free that block ;-) We don't even need to mark * it not-uptodate - nobody can expect anything from a newly allocated buffer - * anyway. We used to use unmap_buffer() for such invalidation, but that was - * wrong. We definitely don't want to mark the alias unmapped, for example - it - * would confuse anyone who might pick it with bread() afterwards... + * anyway. * * Also.. Note that bforget() doesn't lock the buffer. So there can be * writeout I/O going on against recently-freed buffers. We don't wait on that @@ -1649,7 +1658,7 @@ void clean_bdev_aliases(struct block_device *bdev, sector_t block, sector_t len) /* Recheck when the folio is locked which pins bhs */ head = folio_buffers(folio); if (!head) - goto unlock_page; + goto unlock_folio; bh = head; do { if (!buffer_mapped(bh) || (bh->b_blocknr < block)) @@ -1662,7 +1671,7 @@ void clean_bdev_aliases(struct block_device *bdev, sector_t block, sector_t len) next: bh = bh->b_this_page; } while (bh != head); -unlock_page: +unlock_folio: folio_unlock(folio); } folio_batch_release(&fbatch); @@ -1710,7 +1719,7 @@ static struct buffer_head *folio_create_buffers(struct folio *folio, * * If block_write_full_folio() is called for regular writeback * (wbc->sync_mode == WB_SYNC_NONE) then it will redirty a folio which - * has a locked buffer. This only can happen if someone has written + * has a locked buffer. This can only happen if someone has written * the buffer directly, with bh_submit(). At the address_space level * the folio writeback flag prevents this contention from occurring. * @@ -2213,9 +2222,9 @@ int generic_write_end(const struct kiocb *iocb, struct address_space *mapping, if (old_size < pos) pagecache_isize_extended(inode, old_size, pos); /* - * Don't mark the inode dirty under page lock. First, it unnecessarily - * makes the holding time of page lock longer. Second, it forces lock - * ordering of page lock and transaction start for journaling + * Don't mark the inode dirty under folio lock. First, it unnecessarily + * makes the holding time of folio lock longer. Second, it forces lock + * ordering of folio lock and transaction start for journaling * filesystems. */ if (i_size_changed) @@ -2341,7 +2350,7 @@ int block_read_full_folio(struct folio *folio, get_block_t *get_block) * BH_Async_Read tells end_buffer_async_read() that this * buffer is not under async I/O. * - * The folio comes unlocked when it has no locked + * The folio is unlocked when it has no locked * buffer_async buffers left. * * The folio lock prevents anyone starting new async @@ -2451,7 +2460,7 @@ static int cont_expand_zero(const struct kiocb *iocb, } } - /* page covers the boundary, find the boundary offset */ + /* folio crosses the boundary, find the boundary offset */ if (index == curidx) { zerofrom = curpos & ~PAGE_MASK; /* if we will expand the thing last block will be filled */ @@ -2509,18 +2518,18 @@ EXPORT_SYMBOL(cont_write_begin); /* * block_page_mkwrite() is not allowed to change the file size as it gets - * called from a page fault handler when a page is first dirtied. Hence we must - * be careful to check for EOF conditions here. We set the page up correctly - * for a written page which means we get ENOSPC checking when writing into + * called from a page fault handler when a folio is first dirtied. Hence we must + * be careful to check for EOF conditions here. We set the folio up correctly + * for a written folio which means we get ENOSPC checking when writing into * holes and correct delalloc and unwritten extent mapping on filesystems that * support these features. * * We are not allowed to take the i_rwsem here so we have to play games to - * protect against truncate races as the page could now be beyond EOF. Because - * truncate writes the inode size before removing pages, once we have the - * page lock we can determine safely if the page is beyond EOF. If it is not - * beyond EOF, then the page is guaranteed safe against truncation until we - * unlock the page. + * protect against truncate races as the folio could now be beyond EOF. Because + * truncate writes the inode size before removing folios, once we have the + * folio lock we can determine safely if the folio is beyond EOF. If it is not + * beyond EOF, then the folio is guaranteed safe against truncation until we + * unlock the folio. * * Direct callers of this function should protect against filesystem freezing * using sb_start_pagefault() - sb_end_pagefault() functions. @@ -2538,7 +2547,7 @@ int block_page_mkwrite(struct vm_area_struct *vma, struct vm_fault *vmf, size = i_size_read(inode); if ((folio->mapping != inode->i_mapping) || (folio_pos(folio) >= size)) { - /* We overload EFAULT to mean page got truncated */ + /* We overload EFAULT to mean folio got truncated */ ret = -EFAULT; goto out_unlock; } @@ -2710,7 +2719,7 @@ int __sync_dirty_buffer(struct buffer_head *bh, blk_opf_t op_flags) bh_submit(bh, REQ_OP_WRITE | op_flags, bh_end_write); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) return -EIO; } else { unlock_buffer(bh); diff --git a/fs/cachefiles/Kconfig b/fs/cachefiles/Kconfig index afb25b6af5aa..c9c168c7e072 100644 --- a/fs/cachefiles/Kconfig +++ b/fs/cachefiles/Kconfig @@ -17,7 +17,7 @@ config CACHEFILES_DEBUG help This permits debugging to be dynamically enabled in the filesystem caching on files module. If this is set, the debugging output may be - enabled by setting bits in /sys/modules/cachefiles/parameter/debug or + enabled by setting bits in /sys/module/cachefiles/parameters/debug or by including a debugging specifier in /etc/cachefilesd.conf. config CACHEFILES_ERROR_INJECTION diff --git a/fs/cachefiles/interface.c b/fs/cachefiles/interface.c index 50a000310a8c..789ff6abe926 100644 --- a/fs/cachefiles/interface.c +++ b/fs/cachefiles/interface.c @@ -100,73 +100,6 @@ void cachefiles_put_object(struct cachefiles_object *object, } /* - * Adjust the size of a cache file if necessary to match the DIO size. We keep - * the EOF marker a multiple of DIO blocks so that we don't fall back to doing - * non-DIO for a partial block straddling the EOF, but we also have to be - * careful of someone expanding the file and accidentally accreting the - * padding. - */ -static int cachefiles_adjust_size(struct cachefiles_object *object) -{ - struct iattr newattrs; - struct file *file = object->file; - uint64_t ni_size; - loff_t oi_size; - int ret; - - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - _enter("{OBJ%x},[%llu]", - object->debug_id, (unsigned long long) ni_size); - - if (!file) - return -ENOBUFS; - - oi_size = i_size_read(file_inode(file)); - if (oi_size == ni_size) - return 0; - - inode_lock(file_inode(file)); - - /* if there's an extension to a partial page at the end of the backing - * file, we need to discard the partial page so that we pick up new - * data after it */ - if (oi_size & ~PAGE_MASK && ni_size > oi_size) { - _debug("discard tail %llx", oi_size); - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = oi_size & PAGE_MASK; - ret = cachefiles_inject_remove_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - if (ret < 0) - goto truncate_failed; - } - - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = ni_size; - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - -truncate_failed: - inode_unlock(file_inode(file)); - - if (ret < 0) - trace_cachefiles_io_error(NULL, file_inode(file), ret, - cachefiles_trace_notify_change_error); - if (ret == -EIO) { - cachefiles_io_error_obj(object, "Size set failed"); - ret = -ENOBUFS; - } - - _leave(" = %d", ret); - return ret; -} - -/* * Attempt to look up the nominated node in this cache */ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) @@ -198,7 +131,6 @@ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) spin_lock(&cache->object_list_lock); list_add(&object->cache_link, &cache->object_list); spin_unlock(&cache->object_list_lock); - cachefiles_adjust_size(object); cachefiles_end_secure(cache, saved_cred); _leave(" = t"); @@ -225,14 +157,14 @@ fail: * any unused granules. */ static bool cachefiles_shorten_object(struct cachefiles_object *object, - struct file *file, loff_t new_size) + struct file *file, uoff_t new_size) { struct cachefiles_cache *cache = object->volume->cache; struct inode *inode = file_inode(file); - loff_t i_size, dio_size; + uoff_t i_size, dio_size; int ret; - dio_size = round_up(new_size, CACHEFILES_DIO_BLOCK_SIZE); + dio_size = round_up(new_size, cache->bsize); i_size = i_size_read(inode); trace_cachefiles_trunc(object, inode, i_size, dio_size, @@ -264,6 +196,7 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, } } + object->object_size = new_size; return true; } @@ -271,29 +204,38 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, * Resize the backing object. */ static void cachefiles_resize_cookie(struct netfs_cache_resources *cres, - loff_t new_size) + uoff_t new_size) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; struct fscache_cookie *cookie = object->cookie; const struct cred *saved_cred; struct file *file = cachefiles_cres_file(cres); - loff_t old_size = cookie->object_size; + uoff_t i_size = i_size_read(file_inode(file)); - _enter("%llu->%llu", old_size, new_size); + _enter("%llu->%llu", object->object_size, new_size); - if (new_size < old_size) { + /* If the file is being shrunk, we need to downsize the backing file + * and clear the end of the final block. + */ + if (new_size < object->object_size) { + if (new_size >= i_size) + goto out; cachefiles_begin_secure(cache, &saved_cred); cachefiles_shorten_object(object, file, new_size); cachefiles_end_secure(cache, saved_cred); object->cookie->object_size = new_size; + if (new_size == 0) + object->content_info = CACHEFILES_CONTENT_NO_DATA; return; } /* The file is being expanded. We don't need to do anything - * particularly. cookie->initial_size doesn't change and so the point - * at which we have to download before doesn't change. + * particularly. The tail of the last block should have been cleared + * both when it is written and when it is shrunk. */ +out: + object->object_size = new_size; cookie->object_size = new_size; } diff --git a/fs/cachefiles/internal.h b/fs/cachefiles/internal.h index c93324e0f98c..664be64ab538 100644 --- a/fs/cachefiles/internal.h +++ b/fs/cachefiles/internal.h @@ -16,8 +16,6 @@ #include <linux/cred.h> #include <linux/security.h> -#define CACHEFILES_DIO_BLOCK_SIZE 4096 - struct cachefiles_cache; struct cachefiles_object; @@ -51,12 +49,17 @@ struct cachefiles_object { struct list_head cache_link; /* Link in cache->*_list */ struct file *file; /* The file representing this object */ char *d_name; /* Backing file name */ + unsigned long flags; +#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + uoff_t object_size; /* Size of the object stored + * (independent of cookie->object_size for + * coherency reasons) + */ + atomic64_t read_limit; /* Point beyond which uncommitted writes */ int debug_id; spinlock_t lock; refcount_t ref; - enum cachefiles_content content_info:8; /* Info about content presence */ - unsigned long flags; -#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + enum cachefiles_content content_info; /* Info about content presence */ }; /* @@ -203,11 +206,11 @@ extern bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state); extern int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet); extern int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -280,6 +283,7 @@ void cachefiles_withdraw_volume(struct cachefiles_volume *volume); /* * xattr.c */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file); extern int cachefiles_set_object_xattr(struct cachefiles_object *object); extern int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file); diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 9540ec25b3cb..4f547d97356e 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -19,7 +19,7 @@ struct cachefiles_kiocb { struct kiocb iocb; refcount_t ki_refcnt; - loff_t start; + uoff_t start; union { size_t skipped; size_t len; @@ -32,6 +32,8 @@ struct cachefiles_kiocb { u64 b_writing; }; +#define IS_ERR_VALUE_LL(x) unlikely((x) >= (unsigned long long)-MAX_ERRNO) + static inline void cachefiles_put_kiocb(struct cachefiles_kiocb *ki) { if (refcount_dec_and_test(&ki->ki_refcnt)) { @@ -73,7 +75,7 @@ static void cachefiles_read_complete(struct kiocb *iocb, long ret) * Initiate a read from the cache. */ static int cachefiles_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -193,60 +195,81 @@ presubmission_error: } /* - * Query the occupancy of the cache in a region, returning where the next chunk - * of data starts and how long it is. + * Query the occupancy of the cache in a region, returning the extent of the + * next two chunks of cached data and the next hole. */ static int cachefiles_query_occupancy(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len) + struct fscache_occupancy *occ) { struct cachefiles_object *object; + struct inode *inode; struct file *file; - loff_t off, off2; - - *_data_start = -1; - *_data_len = 0; + uoff_t read_limit; + loff_t ret; + int i; if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) return -ENOBUFS; object = cachefiles_cres_object(cres); file = cachefiles_cres_file(cres); - granularity = max_t(size_t, object->volume->cache->bsize, granularity); + inode = file_inode(file); + occ->granularity = object->volume->cache->bsize; + /* Read read_limit before content_info. */ + read_limit = atomic64_read_acquire(&object->read_limit); + + _enter("%pD,%llu,%llx-%llx/%llx", + file, inode->i_ino, occ->query_from, occ->query_to, read_limit); + + if (read_limit == 0) + goto done; + + switch (READ_ONCE(object->content_info)) { + case CACHEFILES_CONTENT_ALL: + case CACHEFILES_CONTENT_SINGLE: + if (read_limit > occ->query_from) { + occ->cached_from[0] = 0; + occ->cached_to[0] = read_limit; + occ->cached_type[0] = FSCACHE_EXTENT_DATA; + occ->query_from = ULLONG_MAX; + } + goto done; + default: + break; + } - _enter("%pD,%llu,%llx,%zx/%llx", - file, file_inode(file)->i_ino, start, len, - i_size_read(file_inode(file))); + for (i = 0; i < ARRAY_SIZE(occ->cached_from); i++) { + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_DATA); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_type[i] = FSCACHE_EXTENT_DATA; + occ->cached_from[i] = ret; + occ->query_from = ret; + + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_HOLE); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_to[i] = ret; + occ->query_from = ret; + if (occ->query_from >= occ->query_to) + break; + } - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off < 0 && off >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - if (round_up(off, granularity) >= start + len) - return -ENODATA; /* No data in range */ - - off2 = cachefiles_inject_read_error(); - if (off2 == 0) - off2 = vfs_llseek(file, off, SEEK_HOLE); - if (off2 == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off2 < 0 && off2 >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - - /* Round away partial blocks */ - off = round_up(off, granularity); - off2 = round_down(off2, granularity); - if (off2 <= off) - return -ENODATA; - - *_data_start = off; - if (off2 > start + len) - *_data_len = len; - else - *_data_len = off2 - off; +done: + _debug("query[0] %llx-%llx", occ->cached_from[0], occ->cached_to[0]); + _debug("query[1] %llx-%llx", occ->cached_from[1], occ->cached_to[1]); return 0; } @@ -280,7 +303,7 @@ static void cachefiles_write_complete(struct kiocb *iocb, long ret) */ int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -357,7 +380,7 @@ in_progress: } static int cachefiles_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -375,127 +398,12 @@ static int cachefiles_write(struct netfs_cache_resources *cres, term_func, term_func_priv); } -static inline enum netfs_io_source -cachefiles_do_prepare_read(struct netfs_cache_resources *cres, - loff_t start, size_t *_len, loff_t i_size, - unsigned long *_flags, ino_t netfs_ino) -{ - enum cachefiles_prepare_read_trace why; - struct cachefiles_object *object = NULL; - struct cachefiles_cache *cache; - struct fscache_cookie *cookie = fscache_cres_cookie(cres); - const struct cred *saved_cred; - struct file *file = cachefiles_cres_file(cres); - enum netfs_io_source ret = NETFS_DOWNLOAD_FROM_SERVER; - size_t len = *_len; - loff_t off, to; - ino_t ino = file ? file_inode(file)->i_ino : 0; - - _enter("%zx @%llx/%llx", len, start, i_size); - - if (start >= i_size) { - ret = NETFS_FILL_WITH_ZEROES; - why = cachefiles_trace_read_after_eof; - goto out_no_object; - } - - if (test_bit(FSCACHE_COOKIE_NO_DATA_TO_READ, &cookie->flags)) { - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); - why = cachefiles_trace_read_no_data; - goto out_no_object; - } - - /* The object and the file may be being created in the background. */ - if (!file) { - why = cachefiles_trace_read_no_file; - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) - goto out_no_object; - file = cachefiles_cres_file(cres); - if (!file) - goto out_no_object; - ino = file_inode(file)->i_ino; - } - - object = cachefiles_cres_object(cres); - cache = object->volume->cache; - cachefiles_begin_secure(cache, &saved_cred); - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off < 0 && off >= (loff_t)-MAX_ERRNO) { - if (off == (loff_t)-ENXIO) { - why = cachefiles_trace_read_seek_nxio; - goto download_and_store; - } - trace_cachefiles_io_error(object, file_inode(file), off, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (off >= start + len) { - why = cachefiles_trace_read_found_hole; - goto download_and_store; - } - - if (off > start) { - off = round_up(off, cache->bsize); - len = off - start; - *_len = len; - why = cachefiles_trace_read_found_part; - goto download_and_store; - } - - to = cachefiles_inject_read_error(); - if (to == 0) - to = vfs_llseek(file, start, SEEK_HOLE); - if (to < 0 && to >= (loff_t)-MAX_ERRNO) { - trace_cachefiles_io_error(object, file_inode(file), to, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (to < start + len) { - if (start + len >= i_size) - to = round_up(to, cache->bsize); - else - to = round_down(to, cache->bsize); - len = to - start; - *_len = len; - } - - why = cachefiles_trace_read_have_data; - ret = NETFS_READ_FROM_CACHE; - goto out; - -download_and_store: - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); -out: - cachefiles_end_secure(cache, saved_cred); -out_no_object: - trace_cachefiles_prep_read(object, start, len, *_flags, ret, why, ino, netfs_ino); - return ret; -} - -/* - * Prepare a read operation, shortening it to a cached/uncached - * boundary as appropriate. - */ -static enum netfs_io_source cachefiles_prepare_read(struct netfs_io_subrequest *subreq, - unsigned long long i_size) -{ - return cachefiles_do_prepare_read(&subreq->rreq->cache_resources, - subreq->start, &subreq->len, i_size, - &subreq->flags, subreq->rreq->inode->i_ino); -} - /* * Prepare for a write to occur. */ int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet) { struct cachefiles_cache *cache = object->volume->cache; @@ -504,7 +412,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, int ret; /* Round to DIO size */ - start = round_down(*_start, PAGE_SIZE); + start = round_down(*_start, cache->bsize); if (start != *_start || *_len > upper_len) { /* Probably asked to cache a streaming write written into the * pagecache when the cookie was temporarily out of service to @@ -514,7 +422,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return -ENOBUFS; } - *_len = round_up(len, PAGE_SIZE); + *_len = round_up(len, cache->bsize); /* We need to work out whether there's sufficient disk space to perform * the write - but we can skip that check if we have space already @@ -540,10 +448,14 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, * space, we need to see if it's fully allocated. If it's not, we may * want to cull it. */ - if (cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_check) == 0) + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, + cachefiles_has_space_check); + if (ret == 0) return 0; /* Enough space to simply overwrite the whole block */ + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace_2); + pos = cachefiles_inject_read_error(); if (pos == 0) pos = vfs_llseek(file, start, SEEK_HOLE); @@ -572,13 +484,16 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return ret; check_space: - return cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_for_write); + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, + cachefiles_has_space_for_write); + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace); + return ret; } static int cachefiles_prepare_write(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet) + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; @@ -612,10 +527,14 @@ static void cachefiles_prepare_write_subreq(struct netfs_io_subrequest *subreq) stream->sreq_max_segs = BIO_MAX_VECS; if (!cachefiles_cres_file(cres)) { - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) + if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_waitfail); return netfs_prepare_write_failed(subreq); - if (!cachefiles_cres_file(cres)) + } + if (!cachefiles_cres_file(cres)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_nofile); return netfs_prepare_write_failed(subreq); + } } } @@ -628,16 +547,16 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) struct netfs_io_stream *stream = &wreq->io_streams[subreq->stream_nr]; const struct cred *saved_cred; size_t off, pre, post, len = subreq->len; - loff_t start = subreq->start; + uoff_t start = subreq->start; int ret; _enter("W=%x[%x] %llx-%llx", wreq->debug_id, subreq->debug_index, start, start + len - 1); /* We need to start on the cache granularity boundary */ - off = start & (CACHEFILES_DIO_BLOCK_SIZE - 1); + off = start & (cache->bsize - 1); if (off) { - pre = CACHEFILES_DIO_BLOCK_SIZE - off; + pre = cache->bsize - off; if (pre >= len) { fscache_count_dio_misfit(); netfs_write_subrequest_terminated(subreq, len); @@ -651,8 +570,8 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) /* We also need to end on the cache granularity boundary */ if (start + len == wreq->i_size) { - size_t part = len % CACHEFILES_DIO_BLOCK_SIZE; - size_t need = CACHEFILES_DIO_BLOCK_SIZE - part; + size_t part = len & (cache->bsize - 1); + size_t need = cache->bsize - part; if (part && stream->submit_extendable_to >= need) { len += need; @@ -661,7 +580,7 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) } } - post = len & (CACHEFILES_DIO_BLOCK_SIZE - 1); + post = len & (cache->bsize - 1); if (post) { len -= post; if (len == 0) { @@ -689,6 +608,198 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) } /* + * Collect the result of buffered writeback to the cache. This includes + * copying a read to the cache. Netfslib collates the results, which might + * occur out of order, and delivers them to the cache so that it can update its + * content record. + * + * block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a contiguous block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a hole due to a failed/cancelled write + * + * The writes we made are all rounded out at both sides to the nearest DIO + * block boundary, so if the final block contains the EOF in the middle of it + * (rather than at the end), padding will have been written to the file. The + * backing file's filesize will have been updated if the write extended the + * file; the filesize may still change due to outstanding subreqs. + * + * The metadata in the cache file xattr records the size of the object we have + * stored, but the cache file EOF only goes up to where we've cached data to + * and, furthermore, is rounded up to the nearest DIO block boundary. + * + * Concurrent updates should be protected against by the caller. Netfslib + * holds NETFS_ICTX_WB_LOCK as a lock on writeback requests. DIO writes + * invalidate the cookie and caching is kept disabled until all users have + * unused the cookie. + */ +static void cachefiles_collect_write(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + struct cachefiles_object *object = cachefiles_cres_object(cres); + struct cachefiles_cache *cache = object->volume->cache; + struct inode *inode; + struct file *file = cachefiles_cres_file(cres); + uoff_t read_limit; + uoff_t old_size = cres->cache_i_size; + uoff_t new_size; + uoff_t data_to = object->object_size; + uoff_t end = start + len; + int ret; + + if (!file) + return; + + inode = file_inode(file); + new_size = i_size_read(inode); + + _enter("%llx,%zx,%x", start, len, cache->bsize); + + if (WARN_ON(old_size & (cache->bsize - 1)) || + WARN_ON(new_size & (cache->bsize - 1)) || + WARN_ON(start & (cache->bsize - 1)) || + WARN_ON(len & (cache->bsize - 1))) { + trace_cachefiles_io_error(object, inode, -EIO, + cachefiles_trace_alignment_error); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + /* If this is recording a gap, due to discontiguous writes or lack of + * cache space, then a hole may have been introduced into the backing + * file. Treat it as a zero-length data block. + */ + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + start = end; + len = 0; + } + + /* Zeroth case: Single monolithic files are handled specially. + */ + if (wreq->origin == NETFS_WRITEBACK_SINGLE) { + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + trace_cachefiles_trunc(object, inode, data_to, 0, + cachefiles_trunc_zap); + ret = cachefiles_inject_remove_error(); + if (ret == 0) + ret = vfs_truncate(&file->f_path, 0); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_trunc_error); + cachefiles_io_error_obj(object, "truncate failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + object->content_info = CACHEFILES_CONTENT_NO_DATA; + read_limit = 0; + } else { + object->content_info = CACHEFILES_CONTENT_SINGLE; + read_limit = len; + } + goto update_sizes_2; + } + + /* First case: The backing file was empty. */ + if (old_size == 0) { + if (start == 0) + object->content_info = CACHEFILES_CONTENT_ALL; + else + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + goto update_sizes; + } + + /* Second case: The backing file is entirely within the old object size + * and thus there can be no partial tail block to deal with in the + * cache file. + */ + if (old_size <= data_to) { + if (start > old_size) + goto discontiguous; + goto update_sizes; + } + + /* Third case: The write happened entirely within the bounds of the + * current cache file's size. + */ + if (end <= old_size) + goto update_sizes; + + /* Fourth case: The write overwrote the partial tail block and extended + * the file. We only need to update the object size because netfslib + * rounds out/pads cache writes to whole disk blocks. + */ + if (start < old_size) + goto update_sizes; + + /* Fifth case: The write started from the end of the whole tail block + * and extended the file. Just extend our notion of the filesize. + */ + if (start == old_size && old_size == data_to) + goto update_sizes; + + /* Sixth case: The write continued on from the partial tail block and + * extended the file. Need to clear the gap. + */ + if (start == old_size && old_size > data_to) + goto clear_gap; + +discontiguous: + /* Seventh case: The write was beyond the EOF on the cache file, so now + * there's a hole in the file and we can no longer say in the metadata + * that we can assume we have it all. We may also need to clear the + * end of the partial tail block. + */ + /* TODO: For the moment, we will have to use SEEK_HOLE/SEEK_DATA. */ + if (object->content_info != CACHEFILES_CONTENT_BACKFS_MAP) { + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + trace_cachefiles_coherency(object, inode->i_ino, data_to, NULL, + CACHEFILES_CONTENT_BACKFS_MAP, + cachefiles_coherency_discontiguous); + } + +clear_gap: + /* We need to clear any partial padding that got jumped over. It + * *should* be all zeros, but shared-writable mmap exists... + */ + if (old_size > data_to) { + trace_cachefiles_trunc(object, inode, data_to, old_size, + cachefiles_trunc_clear_padding); + ret = cachefiles_inject_write_error(); + if (ret == 0) + ret = vfs_fallocate(file, FALLOC_FL_ZERO_RANGE, + data_to, old_size - data_to); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_fallocate_error); + cachefiles_io_error_obj(object, "fallocate zero pad failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + } + +update_sizes: + read_limit = umax(old_size, end); +update_sizes_2: + cres->cache_i_size = read_limit; + + /* We need to be careful setting the object_size: we may have written + * more to the cache than to the server (due to cache DIO rounding) and + * the i_size set on the netfs inode may include unwritten data that + * the server doesn't know about yet. + */ + object->object_size = umin(read_limit, wreq->i_size); + + /* Raise the limit at which reads can access the file. */ + /* Update read_limit after content_info */ + atomic64_set_release(&object->read_limit, read_limit); +} + +/* * Clean up an operation. */ static void cachefiles_end_operation(struct netfs_cache_resources *cres) @@ -705,10 +816,10 @@ static const struct netfs_cache_ops cachefiles_netfs_cache_ops = { .read = cachefiles_read, .write = cachefiles_write, .issue_write = cachefiles_issue_write, - .prepare_read = cachefiles_prepare_read, .prepare_write = cachefiles_prepare_write, .prepare_write_subreq = cachefiles_prepare_write_subreq, .query_occupancy = cachefiles_query_occupancy, + .collect_write = cachefiles_collect_write, }; /* @@ -718,13 +829,20 @@ bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state) { struct cachefiles_object *object = cachefiles_cres_object(cres); + struct file *file; + + cres->dio_size = object->volume->cache->bsize; if (!cachefiles_cres_file(cres)) { cres->ops = &cachefiles_netfs_cache_ops; + cres->object_id = object->debug_id; if (object->file) { spin_lock(&object->lock); - if (!cres->cache_priv2 && object->file) - cres->cache_priv2 = get_file(object->file); + file = object->file; + if (!cres->cache_priv2 && file) { + cres->cache_priv2 = get_file(file); + cres->cache_i_size = i_size_read(file_inode(file)); + } spin_unlock(&object->lock); } } diff --git a/fs/cachefiles/namei.c b/fs/cachefiles/namei.c index 88955249a1a6..ef656a319ede 100644 --- a/fs/cachefiles/namei.c +++ b/fs/cachefiles/namei.c @@ -117,8 +117,11 @@ retry: if (d_is_negative(subdir)) { ret = cachefiles_has_space(cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(NULL, cachefiles_trace_mkdir_nospace); goto mkdir_error; + } _debug("attempt mkdir"); @@ -414,7 +417,6 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) struct dentry *fan = volume->fanout[(u8)object->cookie->key_hash]; struct file *file; const struct path parentpath = { .mnt = cache->mnt, .dentry = fan }; - uint64_t ni_size; long ret; @@ -442,31 +444,20 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) if (!cachefiles_mark_inode_in_use(object, file_inode(file))) WARN_ON(1); - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - if (ni_size > 0) { - trace_cachefiles_trunc(object, file_inode(file), 0, ni_size, - cachefiles_trunc_expand_tmpfile); - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = vfs_truncate(&file->f_path, ni_size); - if (ret < 0) { - trace_cachefiles_vfs_error( - object, file_inode(file), ret, - cachefiles_trace_trunc_error); - goto err_unuse; - } - } - ret = -EINVAL; if (unlikely(!file->f_op->read_iter) || unlikely(!file->f_op->write_iter)) { pr_notice("Cache does not support read_iter and write_iter\n"); goto err_unuse; } + + /* Preallocate space for the xattr. */ + ret = cachefiles_preset_object_xattr(object, file); + if (ret < 0) + goto err_unuse; out: cachefiles_end_secure(cache, saved_cred); + object->content_info = CACHEFILES_CONTENT_ALL; return file; err_unuse: @@ -487,8 +478,11 @@ static bool cachefiles_create_file(struct cachefiles_object *object) ret = cachefiles_has_space(object->volume->cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_create_nospace); return false; + } file = cachefiles_create_tmpfile(object); if (IS_ERR(file)) diff --git a/fs/cachefiles/xattr.c b/fs/cachefiles/xattr.c index c70bf67e52b0..8ebb713482e3 100644 --- a/fs/cachefiles/xattr.c +++ b/fs/cachefiles/xattr.c @@ -35,6 +35,57 @@ struct cachefiles_vol_xattr { } __packed; /* + * Preset the state xattr on a cache file to allocate space for it. + */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file) +{ + struct cachefiles_xattr *buf; + struct dentry *dentry = file->f_path.dentry; + unsigned int len = object->cookie->aux_len; + int ret; + + buf = kzalloc(sizeof(struct cachefiles_xattr) + min(len, sizeof(__be64)), GFP_KERNEL); + if (!buf) + return -ENOMEM; + + buf->type = CACHEFILES_COOKIE_TYPE_DATA; + buf->content = CACHEFILES_CONTENT_DIRTY; + + ret = cachefiles_inject_write_error(); + if (ret == 0) { + ret = mnt_want_write_file(file); + if (ret == 0) { + ret = vfs_setxattr(&nop_mnt_idmap, dentry, + cachefiles_xattr_cache, buf, + sizeof(struct cachefiles_xattr) + len, 0); + mnt_drop_write_file(file); + } + } + if (ret < 0) { + trace_cachefiles_vfs_error(object, file_inode(file), ret, + cachefiles_trace_setxattr_error); + trace_cachefiles_coherency(object, file_inode(file)->i_ino, + object->object_size, + buf->data, buf->content, + cachefiles_coherency_set_fail); + switch (ret) { + case -ENOMEM: + case -ENOSPC: + break; + default: + cachefiles_io_error_obj( + object, + "Failed to set xattr with error %d", ret); + break; + } + } + + kfree(buf); + _leave(" = %d", ret); + return ret; +} + +/* * set the state xattr on a cache file */ int cachefiles_set_object_xattr(struct cachefiles_object *object) @@ -43,6 +94,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) struct dentry *dentry; struct file *file = object->file; unsigned int len = object->cookie->aux_len; + uoff_t object_size = object->cookie->object_size; int ret; if (!file) @@ -55,7 +107,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) if (!buf) return -ENOMEM; - buf->object_size = cpu_to_be64(object->cookie->object_size); + buf->object_size = cpu_to_be64(object_size); buf->zero_point = 0; buf->type = CACHEFILES_COOKIE_TYPE_DATA; buf->content = object->content_info; @@ -79,15 +131,21 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) trace_cachefiles_vfs_error(object, file_inode(file), ret, cachefiles_trace_setxattr_error); trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_fail); - if (ret != -ENOMEM) + switch (ret) { + case -ENOMEM: + break; + case -ENOSPC: + default: cachefiles_io_error_obj( object, "Failed to set xattr with error %d", ret); + break; + } } else { trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_ok); } @@ -103,10 +161,12 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file { struct cachefiles_xattr *buf; struct dentry *dentry = file->f_path.dentry; + struct inode *inode = file_inode(file); unsigned int len = object->cookie->aux_len, tlen; const void *p = fscache_get_aux(object->cookie); enum cachefiles_coherency_trace why; ssize_t xlen; + uoff_t obj_size; int ret = -ESTALE; tlen = sizeof(struct cachefiles_xattr) + len; @@ -121,34 +181,39 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file if (xlen != tlen) { if (xlen < 0) { ret = xlen; - trace_cachefiles_vfs_error(object, file_inode(file), xlen, + trace_cachefiles_vfs_error(object, inode, xlen, cachefiles_trace_getxattr_error); } if (xlen == -EIO) cachefiles_io_error_obj( object, "Failed to read aux with error %zd", xlen); + obj_size = 0; why = cachefiles_coherency_check_xattr; goto out; } + obj_size = be64_to_cpu(buf->object_size); if (buf->type != CACHEFILES_COOKIE_TYPE_DATA) { why = cachefiles_coherency_check_type; } else if (memcmp(buf->data, p, len) != 0) { why = cachefiles_coherency_check_aux; - } else if (be64_to_cpu(buf->object_size) != object->cookie->object_size) { + } else if (obj_size != object->cookie->object_size) { why = cachefiles_coherency_check_objsize; } else if (buf->content == CACHEFILES_CONTENT_DIRTY) { // TODO: Begin conflict resolution pr_warn("Dirty object in cache\n"); why = cachefiles_coherency_check_dirty; } else { + object->content_info = buf->content; + object->object_size = obj_size; + atomic64_set(&object->read_limit, i_size_read(inode)); why = cachefiles_coherency_check_ok; ret = 0; } out: - trace_cachefiles_coherency(object, file_inode(file)->i_ino, + trace_cachefiles_coherency(object, inode->i_ino, obj_size, buf->data, buf->content, why); kfree(buf); return ret; @@ -163,6 +228,9 @@ int cachefiles_remove_object_xattr(struct cachefiles_cache *cache, { int ret; + trace_cachefiles_coherency(object, d_inode(dentry)->i_ino, 0, NULL, 0, + cachefiles_coherency_remove); + ret = cachefiles_inject_remove_error(); if (ret == 0) { ret = mnt_want_write(cache->mnt); diff --git a/fs/ceph/Kconfig b/fs/ceph/Kconfig index 3d64a316ca31..aa6ccd7794d2 100644 --- a/fs/ceph/Kconfig +++ b/fs/ceph/Kconfig @@ -4,6 +4,7 @@ config CEPH_FS depends on INET select CEPH_LIB select NETFS_SUPPORT + select NETFS_PGPRIV2 select FS_ENCRYPTION_ALGS if FS_ENCRYPTION default n help diff --git a/fs/ceph/acl.c b/fs/ceph/acl.c index 85d3dd48b167..124f07ae5b2d 100644 --- a/fs/ceph/acl.c +++ b/fs/ceph/acl.c @@ -87,7 +87,7 @@ retry: return acl; } -int ceph_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ceph_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int ret = 0; diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c index e598b2d424ec..42b55ce30a32 100644 --- a/fs/ceph/addr.c +++ b/fs/ceph/addr.c @@ -65,7 +65,7 @@ (CONGESTION_ON_THRESH(congestion_kb) - \ (CONGESTION_ON_THRESH(congestion_kb) >> 2)) -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata); static inline struct ceph_snap_context *page_snap_context(struct page *page) @@ -1868,7 +1868,7 @@ ceph_find_incompatible(struct folio *folio) return NULL; } -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata) { struct inode *inode = file_inode(file); diff --git a/fs/ceph/dir.c b/fs/ceph/dir.c index 2e5c0ccb1b34..d9615d67bf1c 100644 --- a/fs/ceph/dir.c +++ b/fs/ceph/dir.c @@ -921,7 +921,7 @@ int ceph_handle_notrace_create(struct inode *dir, struct dentry *dentry) return PTR_ERR(result); } -static int ceph_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -988,7 +988,7 @@ out: return err; } -static int ceph_create(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ceph_mknod(idmap, dir, dentry, mode, 0); @@ -1032,7 +1032,7 @@ static int prep_encrypted_symlink_target(struct ceph_mds_request *req, } #endif -static int ceph_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *dest) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -1106,7 +1106,7 @@ out: return err; } -static struct dentry *ceph_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ceph_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -1478,7 +1478,7 @@ out: return err; } -static int ceph_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ceph_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/ceph/file.c b/fs/ceph/file.c index bd3e3f5c269e..2c994c08ed4b 100644 --- a/fs/ceph/file.c +++ b/fs/ceph/file.c @@ -795,7 +795,7 @@ static int ceph_finish_async_create(struct inode *dir, struct inode *inode, int ceph_atomic_open(struct inode *dir, struct dentry *dentry, struct file *file, unsigned flags, umode_t mode) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct ceph_fs_client *fsc = ceph_sb_to_fs_client(dir->i_sb); struct ceph_client *cl = fsc->client; struct ceph_mds_client *mdsc = fsc->mdsc; diff --git a/fs/ceph/inode.c b/fs/ceph/inode.c index d52e2b389e0b..a695dba82554 100644 --- a/fs/ceph/inode.c +++ b/fs/ceph/inode.c @@ -2398,7 +2398,7 @@ static const char *ceph_encrypted_get_link(struct dentry *dentry, done); } -static int ceph_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int ceph_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) @@ -2568,7 +2568,7 @@ out: return ret; } -int __ceph_setattr(struct mnt_idmap *idmap, struct inode *inode, +int __ceph_setattr(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *attr, struct ceph_iattr *cia) { struct ceph_inode_info *ci = ceph_inode(inode); @@ -2921,7 +2921,7 @@ out: /* * setattr */ -int ceph_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ceph_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -3098,7 +3098,7 @@ out: * Check inode permissions. We verify we have a valid value for * the AUTH cap, then call the generic handler. */ -int ceph_permission(struct mnt_idmap *idmap, struct inode *inode, +int ceph_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int err; @@ -3145,7 +3145,7 @@ static int statx_to_caps(u32 want, umode_t mode) * Get all the attributes. If we have sufficient caps for the requested attrs, * then we can avoid talking to the MDS at all. */ -int ceph_getattr(struct mnt_idmap *idmap, const struct path *path, +int ceph_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); diff --git a/fs/ceph/mds_client.h b/fs/ceph/mds_client.h index e7a262c9c2ab..ea48ec5383ef 100644 --- a/fs/ceph/mds_client.h +++ b/fs/ceph/mds_client.h @@ -375,7 +375,7 @@ struct ceph_mds_request { int r_fmode; /* file mode, if expecting cap */ int r_request_release_offset; const struct cred *r_cred; - struct mnt_idmap *r_mnt_idmap; + const struct mnt_idmap *r_mnt_idmap; struct timespec64 r_stamp; /* for choosing which mds to send this request to */ diff --git a/fs/ceph/super.h b/fs/ceph/super.h index 72d4e30304dc..a033331bb151 100644 --- a/fs/ceph/super.h +++ b/fs/ceph/super.h @@ -1166,18 +1166,18 @@ static inline int ceph_do_getattr(struct inode *inode, int mask, bool force) { return __ceph_do_getattr(inode, NULL, mask, force); } -extern int ceph_permission(struct mnt_idmap *idmap, +extern int ceph_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); struct ceph_iattr { struct ceph_fscrypt_auth *fscrypt_auth; }; -extern int __ceph_setattr(struct mnt_idmap *idmap, struct inode *inode, +extern int __ceph_setattr(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *attr, struct ceph_iattr *cia); -extern int ceph_setattr(struct mnt_idmap *idmap, +extern int ceph_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -extern int ceph_getattr(struct mnt_idmap *idmap, +extern int ceph_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); void ceph_inode_shutdown(struct inode *inode); @@ -1252,7 +1252,7 @@ void ceph_release_acl_sec_ctx(struct ceph_acl_sec_ctx *as_ctx); #ifdef CONFIG_CEPH_FS_POSIX_ACL struct posix_acl *ceph_get_acl(struct inode *, int, bool); -int ceph_set_acl(struct mnt_idmap *idmap, +int ceph_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int ceph_pre_init_acls(struct inode *dir, umode_t *mode, struct ceph_acl_sec_ctx *as_ctx); diff --git a/fs/ceph/xattr.c b/fs/ceph/xattr.c index cc4ffbbcb719..7d77214c76c6 100644 --- a/fs/ceph/xattr.c +++ b/fs/ceph/xattr.c @@ -1352,7 +1352,7 @@ static int ceph_get_xattr_handler(const struct xattr_handler *handler, } static int ceph_set_xattr_handler(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/char_dev.c b/fs/char_dev.c index 00229e25c10f..5ce5423f6c99 100644 --- a/fs/char_dev.c +++ b/fs/char_dev.c @@ -280,7 +280,9 @@ int __register_chrdev(unsigned int major, unsigned int baseminor, cdev->owner = fops->owner; cdev->ops = fops; - kobject_set_name(&cdev->kobj, "%s", name); + err = kobject_set_name(&cdev->kobj, "%s", name); + if (err) + goto out; err = cdev_add(cdev, MKDEV(cd->major, baseminor), count); if (err) diff --git a/fs/coda/coda_linux.h b/fs/coda/coda_linux.h index dd6277d87afb..0c0d5f81653c 100644 --- a/fs/coda/coda_linux.h +++ b/fs/coda/coda_linux.h @@ -46,12 +46,12 @@ extern const struct file_operations coda_ioctl_operations; /* operations shared over more than one file */ int coda_open(struct inode *i, struct file *f); int coda_release(struct inode *i, struct file *f); -int coda_permission(struct mnt_idmap *idmap, struct inode *inode, +int coda_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int coda_revalidate_inode(struct inode *); -int coda_getattr(struct mnt_idmap *, const struct path *, struct kstat *, +int coda_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); -int coda_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int coda_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); /* this file: helpers */ char *coda_f2s(struct CodaFid *f); diff --git a/fs/coda/dir.c b/fs/coda/dir.c index 67148edfadee..a85be5962e62 100644 --- a/fs/coda/dir.c +++ b/fs/coda/dir.c @@ -73,7 +73,7 @@ static struct dentry *coda_lookup(struct inode *dir, struct dentry *entry, unsig } -int coda_permission(struct mnt_idmap *idmap, struct inode *inode, +int coda_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int error; @@ -133,7 +133,7 @@ static inline void coda_dir_drop_nlink(struct inode *dir) } /* creation routines: create, mknod, mkdir, link, symlink */ -static int coda_create(struct mnt_idmap *idmap, struct inode *dir, +static int coda_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *de, umode_t mode) { int error; @@ -166,7 +166,7 @@ err_out: return error; } -static struct dentry *coda_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *coda_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *de, umode_t mode) { struct inode *inode; @@ -233,7 +233,7 @@ static int coda_link(struct dentry *source_de, struct inode *dir_inode, } -static int coda_symlink(struct mnt_idmap *idmap, +static int coda_symlink(const struct mnt_idmap *idmap, struct inode *dir_inode, struct dentry *de, const char *symname) { @@ -300,7 +300,7 @@ static int coda_rmdir(struct inode *dir, struct dentry *de) } /* rename */ -static int coda_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int coda_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/coda/inode.c b/fs/coda/inode.c index 40b43866e6a5..c449954e23c2 100644 --- a/fs/coda/inode.c +++ b/fs/coda/inode.c @@ -294,7 +294,7 @@ static void coda_evict_inode(struct inode *inode) coda_cache_clear_inode(inode); } -int coda_getattr(struct mnt_idmap *idmap, const struct path *path, +int coda_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { int err = coda_revalidate_inode(d_inode(path->dentry)); @@ -304,7 +304,7 @@ int coda_getattr(struct mnt_idmap *idmap, const struct path *path, return err; } -int coda_setattr(struct mnt_idmap *idmap, struct dentry *de, +int coda_setattr(const struct mnt_idmap *idmap, struct dentry *de, struct iattr *iattr) { struct inode *inode = d_inode(de); diff --git a/fs/coda/pioctl.c b/fs/coda/pioctl.c index 36e35c15561a..c457e9bab94b 100644 --- a/fs/coda/pioctl.c +++ b/fs/coda/pioctl.c @@ -24,7 +24,7 @@ #include "coda_linux.h" /* pioctl ops */ -static int coda_ioctl_permission(struct mnt_idmap *idmap, +static int coda_ioctl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); static long coda_pioctl(struct file *filp, unsigned int cmd, unsigned long user_data); @@ -41,7 +41,7 @@ const struct file_operations coda_ioctl_operations = { }; /* the coda pioctl inode ops */ -static int coda_ioctl_permission(struct mnt_idmap *idmap, +static int coda_ioctl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return (mask & MAY_EXEC) ? -EACCES : 0; diff --git a/fs/configfs/configfs_internal.h b/fs/configfs/configfs_internal.h index 4bc19cd8d666..5f627e58f135 100644 --- a/fs/configfs/configfs_internal.h +++ b/fs/configfs/configfs_internal.h @@ -76,7 +76,7 @@ extern int configfs_make_dirent(struct configfs_dirent *, struct dentry *, extern int configfs_dirent_is_ready(struct configfs_dirent *); extern const unsigned char * configfs_get_name(struct configfs_dirent *sd); -extern int configfs_setattr(struct mnt_idmap *idmap, +extern int configfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); extern struct dentry *configfs_pin_fs(void); @@ -90,7 +90,7 @@ extern const struct inode_operations configfs_root_inode_operations; extern const struct inode_operations configfs_symlink_inode_operations; extern const struct dentry_operations configfs_dentry_ops; -extern int configfs_symlink(struct mnt_idmap *idmap, +extern int configfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname); extern int configfs_unlink(struct inode *dir, struct dentry *dentry); diff --git a/fs/configfs/dir.c b/fs/configfs/dir.c index cb45e151d852..0c80feec5926 100644 --- a/fs/configfs/dir.c +++ b/fs/configfs/dir.c @@ -1295,7 +1295,7 @@ out_root_unlock: } EXPORT_SYMBOL(configfs_depend_item_unlocked); -static struct dentry *configfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *configfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int ret = 0; diff --git a/fs/configfs/inode.c b/fs/configfs/inode.c index 69f1f24e890f..c92a05251a47 100644 --- a/fs/configfs/inode.c +++ b/fs/configfs/inode.c @@ -32,7 +32,7 @@ static const struct inode_operations configfs_inode_operations ={ .setattr = configfs_setattr, }; -int configfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int configfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode * inode = d_inode(dentry); diff --git a/fs/configfs/symlink.c b/fs/configfs/symlink.c index 3b31c714400f..89178f5d371a 100644 --- a/fs/configfs/symlink.c +++ b/fs/configfs/symlink.c @@ -146,7 +146,7 @@ static int get_target(const char *symname, struct config_item **target, } -int configfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int configfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int ret; diff --git a/fs/coredump.c b/fs/coredump.c index 6114839f5178..f33ece836b10 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -39,6 +39,7 @@ #include <linux/oom.h> #include <linux/compat.h> #include <linux/fs.h> +#include <linux/wait_bit.h> #include <linux/path.h> #include <linux/timekeeping.h> #include <linux/sysctl.h> @@ -51,7 +52,6 @@ #include <net/sock.h> #include <uapi/linux/pidfd.h> #include <uapi/linux/un.h> -#include <uapi/linux/coredump.h> #include <linux/uaccess.h> #include <asm/mmu_context.h> @@ -68,6 +68,8 @@ static bool dump_vma_snapshot(struct coredump_params *cprm); static void free_vma_snapshot(struct coredump_params *cprm); +static void dump_end_record(struct coredump_params *cprm); +static bool dump_flush_skip(struct coredump_params *cprm); #define CORE_FILE_NOTE_SIZE_DEFAULT (4*1024*1024) /* Define a reasonable max cap */ @@ -83,6 +85,8 @@ static int core_uses_pid; static unsigned int core_pipe_limit; static unsigned int core_sort_vma; static char core_pattern[CORENAME_MAX_SIZE] = "core"; +/* Taken around every copy in and out of core_pattern. */ +static DEFINE_SPINLOCK(core_pattern_lock); static int core_name_size = CORENAME_MAX_SIZE; unsigned int core_file_note_size_limit = CORE_FILE_NOTE_SIZE_DEFAULT; static atomic_t core_pipe_count = ATOMIC_INIT(0); @@ -98,9 +102,7 @@ struct core_name { char *corename __counted_by_ptr(size); int used, size; unsigned int core_pipe_limit; - bool core_dumped; enum coredump_type_t core_type; - u64 mask; }; static int expand_corename(struct core_name *cn, int size) @@ -240,18 +242,22 @@ static bool coredump_parse(struct core_name *cn, struct coredump_params *cprm, size_t **argv, int *argc) { const struct cred *cred = current_cred(); - const char *pat_ptr = core_pattern; + char pattern[CORENAME_MAX_SIZE]; + const char *pat_ptr = pattern; bool was_space = false; int pid_in_pattern = 0; int err = 0; - cn->mask = COREDUMP_KERNEL; + /* The sysctl handler may be publishing a new pattern. */ + scoped_guard(spinlock, &core_pattern_lock) + strscpy(pattern, core_pattern); + + cprm->mask = COREDUMP_KERNEL; if (core_pipe_limit) - cn->mask |= COREDUMP_WAIT; + cprm->mask |= COREDUMP_WAIT; cn->used = 0; cn->corename = NULL; cn->core_pipe_limit = 0; - cn->core_dumped = false; if (*pat_ptr == '|') cn->core_type = COREDUMP_PIPE; else if (*pat_ptr == '@') @@ -508,60 +514,64 @@ static int zap_threads(struct task_struct *tsk, int nr = -EAGAIN; spin_lock_irq(&tsk->sighand->siglock); - if (!(signal->flags & SIGNAL_GROUP_EXIT) && !signal->group_exec_task) { + /* A freeze requested before the dump would be lost with TIF_SIGPENDING. */ + if (!(signal->flags & SIGNAL_GROUP_EXIT) && !signal->group_exec_task && + !freezing(tsk) && !(tsk->jobctl & JOBCTL_TRAP_FREEZE)) { /* Allow SIGKILL, see prepare_signal() */ signal->core_state = core_state; nr = zap_process(signal, exit_code); clear_tsk_thread_flag(tsk, TIF_SIGPENDING); tsk->flags |= PF_DUMPCORE; - atomic_set(&core_state->nr_threads, nr); + atomic_set(&core_state->threads_remaining, nr); } spin_unlock_irq(&tsk->sighand->siglock); return nr; } +static void coredump_wait_inactive(struct core_state *core_state) +{ + struct core_thread *ptr; + + wait_var_event_state(&core_state->threads_remaining, + !atomic_read_acquire(&core_state->threads_remaining), + TASK_UNINTERRUPTIBLE | TASK_FREEZABLE); + /* + * Wait for all the threads to become inactive, so that + * all the thread context (extended register state, like + * fpu etc) gets copied to the memory. + */ + for (ptr = core_state->tasks; ptr; ptr = ptr->next) + wait_task_inactive(ptr->task, TASK_ANY); +} + static int coredump_wait(int exit_code, struct core_state *core_state) { struct task_struct *tsk = current; int core_waiters = -EBUSY; - init_completion(&core_state->startup); - core_state->dumper.task = tsk; - core_state->dumper.next = NULL; + core_state->tasks = NULL; core_waiters = zap_threads(tsk, core_state, exit_code); - if (core_waiters > 0) { - struct core_thread *ptr; - - wait_for_completion_state(&core_state->startup, - TASK_UNINTERRUPTIBLE|TASK_FREEZABLE); - /* - * Wait for all the threads to become inactive, so that - * all the thread context (extended register state, like - * fpu etc) gets copied to the memory. - */ - ptr = core_state->dumper.next; - while (ptr != NULL) { - wait_task_inactive(ptr->task, TASK_ANY); - ptr = ptr->next; - } - } + if (core_waiters > 0) + coredump_wait_inactive(core_state); return core_waiters; } -static void coredump_finish(bool core_dumped) +static void coredump_finish(enum coredump_state state) { struct core_thread *curr, *next; struct task_struct *task; spin_lock_irq(¤t->sighand->siglock); - if (core_dumped && !__fatal_signal_pending(current)) + if ((state & COREDUMP_STATE_STARTED) && !__fatal_signal_pending(current)) current->signal->group_exit_code |= 0x80; - next = current->signal->core_state->dumper.next; + next = current->signal->core_state->tasks; current->signal->core_state = NULL; spin_unlock_irq(¤t->sighand->siglock); + /* A released thread may exit and be freed before it is woken. */ + guard(rcu)(); while ((curr = next) != NULL) { next = curr->next; task = curr->task; @@ -570,6 +580,7 @@ static void coredump_finish(bool core_dumped) * ->task == NULL before we read ->next. */ smp_mb(); + /* Any wakeup now lets the thread exit, rcu keeps it alive. */ curr->task = NULL; wake_up_process(task); } @@ -577,13 +588,8 @@ static void coredump_finish(bool core_dumped) static bool dump_interrupted(void) { - /* - * SIGKILL or freezing() interrupt the coredumping. Perhaps we - * can do try_to_freeze() and check __fatal_signal_pending(), - * but then we need to teach dump_write() to restart and clear - * TIF_SIGPENDING. - */ - return fatal_signal_pending(current) || freezing(current); + /* Only SIGKILL and the freezers set it after zap_threads(). */ + return task_sigpending(current); } static void wait_for_dump_helpers(struct file *file) @@ -664,7 +670,12 @@ static int umh_coredump_setup(struct subprocess_info *info, struct cred *new) return 0; } +static_assert(sizeof(struct coredump_record_header) == COREDUMP_RECORD_HEADER_SIZE_VER0); + #ifdef CONFIG_UNIX +/* af_unix halves the send buffer to size a single skb. */ +#define COREDUMP_SOCK_SNDBUF_MIN (3 * PAGE_SIZE) + static bool coredump_sock_connect(struct core_name *cn, struct coredump_params *cprm) { struct file *file __free(fput) = NULL; @@ -690,6 +701,10 @@ static bool coredump_sock_connect(struct core_name *cn, struct coredump_params * if (retval < 0) return false; + /* Don't let a page-sized write split into several skbs. */ + socket->sk->sk_sndbuf = max_t(int, socket->sk->sk_sndbuf, + COREDUMP_SOCK_SNDBUF_MIN); + file = sock_alloc_file(socket, 0, NULL); if (IS_ERR(file)) return false; @@ -752,8 +767,39 @@ static inline bool coredump_sock_send(struct file *file, struct coredump_req *re return ret == sizeof(*req); } +static_assert(sizeof(struct coredump_req) == COREDUMP_REQ_SIZE_VER1); +static_assert(sizeof(struct coredump_ack) == COREDUMP_ACK_SIZE_VER1); static_assert(sizeof(enum coredump_mark) == sizeof(__u32)); +/* Every memory type this kernel knows. */ +#define COREDUMP_MEMORY_ALL \ + (COREDUMP_MEMORY_ANON_PRIVATE | COREDUMP_MEMORY_ANON_SHARED | \ + COREDUMP_MEMORY_FILE_PRIVATE | COREDUMP_MEMORY_FILE_SHARED | \ + COREDUMP_MEMORY_ELF_HEADERS | \ + COREDUMP_MEMORY_HUGETLB_PRIVATE | COREDUMP_MEMORY_HUGETLB_SHARED | \ + COREDUMP_MEMORY_DAX_PRIVATE | COREDUMP_MEMORY_DAX_SHARED) + +#define COREDUMP_MEMORY_TYPE_BIT(mmf) BIT((mmf) - MMF_DUMP_FILTER_SHIFT) +static_assert(COREDUMP_MEMORY_ALL == (MMF_DUMP_FILTER_MASK >> MMF_DUMP_FILTER_SHIFT)); +static_assert(COREDUMP_MEMORY_ANON_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_PRIVATE)); +static_assert(COREDUMP_MEMORY_ANON_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_SHARED)); +static_assert(COREDUMP_MEMORY_FILE_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_PRIVATE)); +static_assert(COREDUMP_MEMORY_FILE_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_SHARED)); +static_assert(COREDUMP_MEMORY_ELF_HEADERS == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ELF_HEADERS)); +static_assert(COREDUMP_MEMORY_HUGETLB_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_PRIVATE)); +static_assert(COREDUMP_MEMORY_HUGETLB_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_SHARED)); +static_assert(COREDUMP_MEMORY_DAX_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_PRIVATE)); +static_assert(COREDUMP_MEMORY_DAX_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_SHARED)); + static inline bool coredump_sock_mark(struct file *file, enum coredump_mark mark) { struct msghdr msg = { .msg_flags = MSG_NOSIGNAL }; @@ -795,10 +841,14 @@ static inline void coredump_sock_shutdown(struct file *file) static bool coredump_sock_request(struct core_name *cn, struct coredump_params *cprm) { struct coredump_req req = { - .size = sizeof(struct coredump_req), - .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT, - .size_ack = sizeof(struct coredump_ack), + .size = sizeof(struct coredump_req), + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .size_ack = sizeof(struct coredump_ack), + .memory_types = cprm->memory_types, + .memory_types_mask = COREDUMP_MEMORY_ALL, }; struct coredump_ack ack = {}; ssize_t usize; @@ -851,7 +901,54 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } - cn->mask = ack.mask; + /* Records only describe a coredump the kernel writes. */ + if ((ack.mask & COREDUMP_RECORDS) && !(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Zero records only exist inside a record stream. */ + if ((ack.mask & COREDUMP_SPARSE) && !(ack.mask & COREDUMP_RECORDS)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + if (ack.mask & COREDUMP_MEMORY_TYPES) { + /* The memory types need the whole field. */ + if (usize < COREDUMP_ACK_SIZE_VER1) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_MINSIZE); + return false; + } + + /* The memory types only select what the kernel writes. */ + if (!(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Refuse unknown memory types. */ + if (ack.memory_types & ~req.memory_types_mask) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + } else if (ack.memory_types) { + /* Like @spare the field must be zero when it isn't used. */ + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + + /* Record header scratch; a bvec can't point at the stack. */ + if (ack.mask & COREDUMP_RECORDS) { + cprm->record_hdr = kmalloc_obj(*cprm->record_hdr); + if (!cprm->record_hdr) + return false; + } + + /* The server's selection replaces the task's entirely. */ + if (ack.mask & COREDUMP_MEMORY_TYPES) + cprm->memory_types = ack.memory_types; + + cprm->mask = ack.mask; return coredump_sock_mark(cprm->file, COREDUMP_MARK_REQACK); } @@ -878,7 +975,7 @@ static inline bool coredump_force_suid_safe(const struct coredump_params *cprm) static bool coredump_file(struct core_name *cn, struct coredump_params *cprm, const struct linux_binfmt *binfmt) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode; struct file *file __free(fput) = NULL; int open_flags = O_CREAT | O_WRONLY | O_NOFOLLOW | O_LARGEFILE | O_EXCL; @@ -1032,29 +1129,41 @@ static bool coredump_pipe(struct core_name *cn, struct coredump_params *cprm, return true; } -static bool coredump_write(struct core_name *cn, - struct coredump_params *cprm, - const struct linux_binfmt *binfmt) +static bool coredump_write(struct coredump_params *cprm, + const struct linux_binfmt *binfmt) { - - if (dump_interrupted()) + if (dump_interrupted()) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return true; + } - if (!dump_vma_snapshot(cprm)) + if (!dump_vma_snapshot(cprm)) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return false; + } file_start_write(cprm->file); - cn->core_dumped = binfmt->core_dump(cprm); + if (!binfmt->core_dump(cprm)) + cprm->state |= COREDUMP_STATE_TRUNCATED; /* - * Ensures that file size is big enough to contain the current - * file postion. This prevents gdb from complaining about - * a truncated file if the last "write" to the file was - * dump_skip. + * A trailing hole still has to land in the coredump. Seeking over + * it doesn't grow the file, so the last byte of it is written + * instead and gdb doesn't see a truncated file. Everything else + * puts the hole on the wire as it flushes it. */ if (cprm->to_skip) { - cprm->to_skip--; - dump_emit(cprm, "", 1); + bool flushed; + + if (cprm->file->f_mode & FMODE_LSEEK) { + cprm->to_skip--; + flushed = dump_emit(cprm, "", 1); + } else { + flushed = dump_flush_skip(cprm); + } + if (!flushed) + cprm->state |= COREDUMP_STATE_TRUNCATED; } + dump_end_record(cprm); file_end_write(cprm->file); free_vma_snapshot(cprm); return true; @@ -1069,7 +1178,8 @@ static void coredump_cleanup(struct core_name *cn, struct coredump_params *cprm) atomic_dec(&core_pipe_count); } kfree(cn->corename); - coredump_finish(cn->core_dumped); + kfree(cprm->record_hdr); + coredump_finish(cprm->state); } static inline bool coredump_skip(const struct coredump_params *cprm, @@ -1115,29 +1225,24 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } /* Don't even generate the coredump. */ - if (cn->mask & COREDUMP_REJECT) - return; - - /* get us an unshared descriptor table; almost always a no-op */ - /* The cell spufs coredump code reads the file descriptor tables */ - if (unshare_files()) + if (cprm->mask & COREDUMP_REJECT) return; - if ((cn->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) + if ((cprm->mask & COREDUMP_KERNEL) && !coredump_write(cprm, binfmt)) return; coredump_sock_shutdown(cprm->file); /* Let the parent know that a coredump was generated. */ - if (cn->mask & COREDUMP_USERSPACE) - cn->core_dumped = true; + if (cprm->mask & COREDUMP_USERSPACE) + cprm->state |= COREDUMP_STATE_STARTED; /* * When core_pipe_limit is set we wait for the coredump server * or usermodehelper to finish before exiting so it can e.g., * inspect /proc/<pid>. */ - if (cn->mask & COREDUMP_WAIT) { + if (cprm->mask & COREDUMP_WAIT) { switch (cn->core_type) { case COREDUMP_PIPE: wait_for_dump_helpers(cprm->file); @@ -1153,6 +1258,10 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } } +#define COREDUMP_TASK_MEMORY_TYPES(mm) \ + ((__mm_flags_get_word((mm)) & MMF_DUMP_FILTER_MASK) >> \ + MMF_DUMP_FILTER_SHIFT) + void vfs_coredump(const kernel_siginfo_t *siginfo) { size_t *argv __free(kfree) = NULL; @@ -1164,8 +1273,8 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) struct coredump_params cprm = { .siginfo = siginfo, .limit = rlimit(RLIMIT_CORE), - /* Snapshot MMF_DUMP_FILTER_* (unlocked) and dumpable for the dump. */ - .mm_flags = __mm_flags_get_word(mm), + /* Snapshot the memory types (unlocked) and dumpable for the dump. */ + .memory_types = COREDUMP_TASK_MEMORY_TYPES(mm), .dumpable = task_exec_state_get_dumpable(current), .vma_meta = NULL, .cpu = raw_smp_processor_id(), @@ -1191,6 +1300,8 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) if (coredump_wait(siginfo->si_signo, &core_state) < 0) return; + /* Task work must not cut the dump short, see signal_pending(). */ + guard(no_notify_signal)(); scoped_with_creds(cred) do_coredump(&cn, &cprm, &argv, &argc, binfmt); coredump_cleanup(&cn, &cprm); @@ -1202,60 +1313,181 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) * do on a core-file: use only these functions to write out all the * necessary info. */ -static int __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +static bool dump_records(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_RECORDS; +} + +static bool dump_sparse(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_SPARSE; +} + +/* Describe the next @len bytes of the coredump. Returns the header size. */ +static size_t dump_record_init(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, + u64 len) +{ + if (!dump_records(cprm)) + return 0; + + *cprm->record_hdr = (struct coredump_record_header) { + .size = sizeof(*cprm->record_hdr), + .type = type, + .flags = flags, + .offset = cprm->pos, + .len = len, + }; + + return sizeof(*cprm->record_hdr); +} + +/* Write @iter whole or fail. @len is what it advances the coredump by. */ +static bool dump_write_iter(struct coredump_params *cprm, struct iov_iter *iter, + size_t len) { struct file *file = cprm->file; + size_t count = iov_iter_count(iter); loff_t pos = file->f_pos; ssize_t n; - if (cprm->written + nr > cprm->limit) - return 0; - if (dump_interrupted()) - return 0; - n = __kernel_write(file, addr, nr, &pos); - if (n != nr) - return 0; + n = __kernel_write_iter(file, iter, &pos); + if (n != (ssize_t)count) + return false; file->f_pos = pos; - cprm->written += n; - cprm->pos += n; + cprm->written += count; + cprm->pos += len; + + return true; +} + +/* One record, never more than a page. See __dump_emit(). */ +static bool dump_emit_chunk(struct coredump_params *cprm, const void *addr, + int nr) +{ + struct kvec kvec[2]; + struct iov_iter iter; + unsigned int nseg = 0; + size_t hdrlen; - return 1; + if (dump_interrupted()) + return false; + + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, nr); + if (hdrlen) { + kvec[nseg].iov_base = cprm->record_hdr; + kvec[nseg].iov_len = hdrlen; + nseg++; + } + kvec[nseg].iov_base = (void *)addr; + kvec[nseg].iov_len = nr; + nseg++; + + iov_iter_kvec(&iter, ITER_SOURCE, kvec, nseg, hdrlen + nr); + + return dump_write_iter(cprm, &iter, nr); +} + +static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (cprm->written + nr > cprm->limit) + return false; + + while (nr) { + int chunk = min_t(int, nr, PAGE_SIZE); + + if (!dump_emit_chunk(cprm, addr, chunk)) + return false; + + addr += chunk; + nr -= chunk; + } + + return true; } -static int __dump_skip(struct coredump_params *cprm, size_t nr) +/* Send a record that stands on its own: a header and nothing else. */ +static bool dump_emit_record(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, u64 len) +{ + struct kvec kvec; + struct iov_iter iter; + size_t hdrlen; + + hdrlen = dump_record_init(cprm, type, flags, len); + if (!hdrlen) + return false; + + kvec.iov_base = cprm->record_hdr; + kvec.iov_len = hdrlen; + iov_iter_kvec(&iter, ITER_SOURCE, &kvec, 1, hdrlen); + + return dump_write_iter(cprm, &iter, len); +} + +/* Close the record stream. Only a whole coredump gets an end record. */ +static void dump_end_record(struct coredump_params *cprm) +{ + if (cprm->state & COREDUMP_STATE_TRUNCATED) + return; + + dump_emit_record(cprm, COREDUMP_RECORD_END, 0, 0); +} + +static bool __dump_skip(struct coredump_params *cprm, size_t nr) { static char zeroes[PAGE_SIZE]; struct file *file = cprm->file; + if (dump_sparse(cprm)) { + /* Hand the server the length of the hole instead of the hole itself. */ + if (dump_interrupted()) + return false; + return dump_emit_record(cprm, COREDUMP_RECORD_ZERO, 0, nr); + } + if (file->f_mode & FMODE_LSEEK) { if (dump_interrupted() || vfs_llseek(file, nr, SEEK_CUR) < 0) - return 0; + return false; cprm->pos += nr; - return 1; + return true; } - while (nr > PAGE_SIZE) { - if (!__dump_emit(cprm, zeroes, PAGE_SIZE)) - return 0; - nr -= PAGE_SIZE; + while (nr) { + size_t chunk = min_t(size_t, nr, PAGE_SIZE); + + if (!__dump_emit(cprm, zeroes, chunk)) + return false; + + nr -= chunk; } - return __dump_emit(cprm, zeroes, nr); + return true; } -int dump_emit(struct coredump_params *cprm, const void *addr, int nr) +/* Flush the accumulated hole before writing data. */ +static bool dump_flush_skip(struct coredump_params *cprm) { if (cprm->to_skip) { if (!__dump_skip(cprm, cprm->to_skip)) - return 0; + return false; cprm->to_skip = 0; } + return true; +} + +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (!dump_flush_skip(cprm)) + return false; return __dump_emit(cprm, addr, nr); } EXPORT_SYMBOL(dump_emit); void dump_skip_to(struct coredump_params *cprm, unsigned long pos) { + if (WARN_ON_ONCE(pos < cprm->pos)) + return; cprm->to_skip = pos - cprm->pos; } EXPORT_SYMBOL(dump_skip_to); @@ -1267,37 +1499,32 @@ void dump_skip(struct coredump_params *cprm, size_t nr) EXPORT_SYMBOL(dump_skip); #ifdef CONFIG_ELF_CORE -static int dump_emit_page(struct coredump_params *cprm, struct page *page) +static bool dump_emit_page(struct coredump_params *cprm, struct page *page) { - struct bio_vec bvec; + struct bio_vec bvec[2]; struct iov_iter iter; - struct file *file = cprm->file; - loff_t pos; - ssize_t n; + unsigned int nseg = 0; + size_t hdrlen; if (!page) - return 0; + return false; - if (cprm->to_skip) { - if (!__dump_skip(cprm, cprm->to_skip)) - return 0; - cprm->to_skip = 0; - } + if (!dump_flush_skip(cprm)) + return false; if (cprm->written + PAGE_SIZE > cprm->limit) - return 0; + return false; if (dump_interrupted()) - return 0; - pos = file->f_pos; - bvec_set_page(&bvec, page, PAGE_SIZE, 0); - iov_iter_bvec(&iter, ITER_SOURCE, &bvec, 1, PAGE_SIZE); - n = __kernel_write_iter(cprm->file, &iter, &pos); - if (n != PAGE_SIZE) - return 0; - file->f_pos = pos; - cprm->written += PAGE_SIZE; - cprm->pos += PAGE_SIZE; + return false; - return 1; + /* Hand the record header to the same write as the page it describes. */ + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, PAGE_SIZE); + if (hdrlen) + bvec_set_virt(&bvec[nseg++], cprm->record_hdr, hdrlen); + bvec_set_page(&bvec[nseg++], page, PAGE_SIZE, 0); + + iov_iter_bvec(&iter, ITER_SOURCE, bvec, nseg, hdrlen + PAGE_SIZE); + + return dump_write_iter(cprm, &iter, PAGE_SIZE); } /* @@ -1329,18 +1556,19 @@ static inline struct page *dump_page_copy(struct page *src, struct page *dst) } #endif -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len) +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len) { unsigned long addr; struct page *dump_page; - int locked, ret; + int locked; + bool ret; dump_page = dump_page_alloc(); if (!dump_page) - return 0; + return false; - ret = 0; + ret = false; locked = 0; for (addr = start; addr < start + len; addr += PAGE_SIZE) { struct page *page; @@ -1364,7 +1592,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, mmap_read_unlock(current->mm); locked = 0; } - int stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); + bool stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); put_page(page); if (stop) goto out; @@ -1383,7 +1611,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, } cond_resched(); } - ret = 1; + ret = true; out: if (locked) mmap_read_unlock(current->mm); @@ -1393,14 +1621,14 @@ out: } #endif -int dump_align(struct coredump_params *cprm, int align) +bool dump_align(struct coredump_params *cprm, int align) { unsigned mod = (cprm->pos + cprm->to_skip) & (align - 1); if (align & (align - 1)) - return 0; + return false; if (mod) cprm->to_skip += align - mod; - return 1; + return true; } EXPORT_SYMBOL(dump_align); @@ -1417,11 +1645,11 @@ void validate_coredump_safety(void) } } -static inline bool check_coredump_socket(void) +static inline bool check_coredump_socket(const char *pattern) { const char *p; - if (core_pattern[0] != '@') + if (pattern[0] != '@') return true; /* @@ -1433,16 +1661,16 @@ static inline bool check_coredump_socket(void) return false; /* Must be an absolute path... */ - if (core_pattern[1] != '/') { + if (pattern[1] != '/') { /* ... or the socket request protocol... */ - if (core_pattern[1] != '@') + if (pattern[1] != '@') return false; /* ... and if so must be an absolute path. */ - if (core_pattern[2] != '/') + if (pattern[2] != '/') return false; - p = &core_pattern[2]; + p = &pattern[2]; } else { - p = &core_pattern[1]; + p = &pattern[1]; } /* The path obviously cannot exceed UNIX_PATH_MAX. */ @@ -1450,7 +1678,7 @@ static inline bool check_coredump_socket(void) return false; /* Must not contain ".." in the path. */ - if (name_contains_dotdot(core_pattern)) + if (name_contains_dotdot(pattern)) return false; return true; @@ -1459,27 +1687,35 @@ static inline bool check_coredump_socket(void) static int proc_dostring_coredump(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { + char pattern[CORENAME_MAX_SIZE]; + const struct ctl_table tmp = { + .procname = table->procname, + .data = pattern, + .maxlen = sizeof(pattern), + }; + bool changed = false; int error; - ssize_t retval; - char old_core_pattern[CORENAME_MAX_SIZE]; - if (!write) - return proc_dostring(table, write, buffer, lenp, ppos); + /* Work on a copy, proc_dostring() appends at *ppos. */ + scoped_guard(spinlock, &core_pattern_lock) + strscpy(pattern, core_pattern); - retval = strscpy(old_core_pattern, core_pattern, CORENAME_MAX_SIZE); - - error = proc_dostring(table, write, buffer, lenp, ppos); - if (error) + error = proc_dostring(&tmp, write, buffer, lenp, ppos); + if (error || !write) return error; - if (!check_coredump_socket()) { - strscpy(core_pattern, old_core_pattern, retval + 1); + if (!check_coredump_socket(pattern)) return -EINVAL; - } - if (strncmp(old_core_pattern, core_pattern, CORENAME_MAX_SIZE)) + /* Publish the validated pattern whole. */ + scoped_guard(spinlock, &core_pattern_lock) { + changed = strncmp(pattern, core_pattern, CORENAME_MAX_SIZE); + if (changed) + strscpy(core_pattern, pattern); + } + if (changed) validate_coredump_safety(); - return error; + return 0; } static const unsigned int core_file_note_size_min = CORE_FILE_NOTE_SIZE_DEFAULT; @@ -1582,15 +1818,15 @@ static bool always_dump_vma(struct vm_area_struct *vma) } #define DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER 1 +#define COREDUMP_MEMORY_TYPE_INCLUDE(types, type) \ + ((types) & COREDUMP_MEMORY_##type) /* * Decide how much of @vma's contents should be included in a core dump. */ static unsigned long vma_dump_size(struct vm_area_struct *vma, - unsigned long mm_flags) + u64 memory_types) { -#define FILTER(type) (mm_flags & (1UL << MMF_DUMP_##type)) - /* always dump the vdso and vsyscall sections */ if (always_dump_vma(vma)) goto whole; @@ -1600,18 +1836,22 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* support for DAX */ if (vma_is_dax(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(DAX_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(DAX_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_PRIVATE)) goto whole; return 0; } /* Hugetlb memory check */ if (is_vm_hugetlb_page(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_PRIVATE)) goto whole; return 0; } @@ -1623,25 +1863,27 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* By default, dump shared memory if mapped from an anonymous file. */ if (vma->vm_flags & VM_SHARED) { if (file_inode(vma->vm_file)->i_nlink == 0 ? - FILTER(ANON_SHARED) : FILTER(MAPPED_SHARED)) + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_SHARED) : + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_SHARED)) goto whole; return 0; } /* Dump segments that have been written to. */ - if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && FILTER(ANON_PRIVATE)) + if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_PRIVATE)) goto whole; if (vma->vm_file == NULL) return 0; - if (FILTER(MAPPED_PRIVATE)) + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_PRIVATE)) goto whole; /* * If this is the beginning of an executable file mapping, * dump the first page to aid in determining what was mapped here. */ - if (FILTER(ELF_HEADERS) && + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ELF_HEADERS) && vma->vm_pgoff == 0 && (vma->vm_flags & VM_READ)) { if ((READ_ONCE(file_inode(vma->vm_file)->i_mode) & 0111) != 0) return PAGE_SIZE; @@ -1657,8 +1899,6 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, return DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER; } -#undef FILTER - return 0; whole: @@ -1743,7 +1983,7 @@ static bool dump_vma_snapshot(struct coredump_params *cprm) m->start = vma->vm_start; m->end = vma->vm_end; m->flags = vma->vm_flags; - m->dump_size = vma_dump_size(vma, cprm->mm_flags); + m->dump_size = vma_dump_size(vma, cprm->memory_types); m->pgoff = vma->vm_pgoff; m->file = vma->vm_file; if (m->file) @@ -469,8 +469,6 @@ static void dax_folio_init(void *entry) if (order > 0) { prep_compound_page(&folio->page, order); - if (order > 1) - INIT_LIST_HEAD(&folio->_deferred_list); WARN_ON_ONCE(folio_ref_count(folio)); } } @@ -775,24 +773,23 @@ fallback: /** * dax_layout_busy_page_range - find first pinned page in @mapping - * @mapping: address space to scan for a page with ref count > 1 + * @mapping: address space to scan for a pinned page * @start: Starting offset. Page containing 'start' is included. * @end: End offset. Page containing 'end' is included. If 'end' is LLONG_MAX, * pages from 'start' till the end of file are included. * - * DAX requires ZONE_DEVICE mapped pages. These pages are never - * 'onlined' to the page allocator so they are considered idle when - * page->count == 1. A filesystem uses this interface to determine if - * any page in the mapping is busy, i.e. for DMA, or other - * get_user_pages() usages. + * DAX requires ZONE_DEVICE mapped pages. A page is considered busy when + * folio_ref_count(folio) exceeds folio_mapcount(folio). This helper is + * used to determine if any page in the mapping is busy, i.e. for DMA, + * or other get_user_pages() usages. * * It is expected that the filesystem is holding locks to block the * establishment of new mappings in this address_space. I.e. it expects - * to be able to run unmap_mapping_range() and subsequently not race + * to be able to run unmap_mapping_pages() and subsequently not race * mapping_mapped() becoming true. */ -struct page *dax_layout_busy_page_range(struct address_space *mapping, - loff_t start, loff_t end) +static struct page *dax_layout_busy_page_range(struct address_space *mapping, + loff_t start, loff_t end) { void *entry; unsigned int scanned = 0; @@ -844,13 +841,6 @@ struct page *dax_layout_busy_page_range(struct address_space *mapping, xas_unlock_irq(&xas); return page; } -EXPORT_SYMBOL_GPL(dax_layout_busy_page_range); - -struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return dax_layout_busy_page_range(mapping, 0, LLONG_MAX); -} -EXPORT_SYMBOL_GPL(dax_layout_busy_page); static int __dax_invalidate_entry(struct address_space *mapping, pgoff_t index, bool trunc) diff --git a/fs/dcache.c b/fs/dcache.c index a66be85f9d01..7a9346c4f2e4 100644 --- a/fs/dcache.c +++ b/fs/dcache.c @@ -32,6 +32,7 @@ #include <linux/bit_spinlock.h> #include <linux/rculist_bl.h> #include <linux/list_lru.h> +#include <linux/namei.h> #include "internal.h" #include "mount.h" @@ -451,6 +452,17 @@ static void dentry_free(struct dentry *dentry) } /* + * If inode is unlinked and doesn't have any aliases (i.e., all fds pointing to + * it are closed), it is pretty much dead. Except that file handle lookup could + * still revive it which causes issues to fsnotify. So once inode reaches this + * state we make sure to block creating any new aliases. + */ +static bool inode_notify_dead(struct inode *inode) +{ + return !inode->i_nlink && hlist_empty(&inode->i_dentry); +} + +/* * Release the dentry's inode, using the filesystem * d_iput() operation if defined. */ @@ -459,6 +471,7 @@ static void dentry_unlink_inode(struct dentry * dentry) __releases(dentry->d_inode->i_lock) { struct inode *inode = dentry->d_inode; + bool notify_dead; raw_write_seqcount_begin(&dentry->d_seq); __d_clear_type_and_inode(dentry); @@ -469,9 +482,10 @@ static void dentry_unlink_inode(struct dentry * dentry) */ dentry->waiters = NULL; raw_write_seqcount_end(&dentry->d_seq); + notify_dead = inode_notify_dead(inode); spin_unlock(&dentry->d_lock); spin_unlock(&inode->i_lock); - if (!inode->i_nlink) + if (notify_dead) fsnotify_inoderemove(inode); if (dentry->d_op && dentry->d_op->d_iput) dentry->d_op->d_iput(dentry, inode); @@ -830,7 +844,7 @@ static struct dentry *dentry_kill(struct dentry *dentry) if (dentry->d_op && dentry->d_op->d_release) dentry->d_op->d_release(dentry); - cond_resched(); + cond_resched_tasks_rcu_qs(); /* now that it's negative, ->d_parent is stable */ if (!IS_ROOT(dentry)) { parent = dentry->d_parent; @@ -1900,6 +1914,7 @@ EXPORT_SYMBOL(d_invalidate); static struct dentry *__d_alloc(struct super_block *sb, const struct qstr *name) { + static struct lock_class_key __lookup_key; struct dentry *dentry; char *dname; int err; @@ -1961,6 +1976,8 @@ static struct dentry *__d_alloc(struct super_block *sb, const struct qstr *name) dentry->waiters = NULL; INIT_HLIST_NODE(&dentry->d_sib); + lockdep_init_map(&dentry->lookup_map, "DCACHE_PAR_LOOKUP", &__lookup_key, 0); + if (dentry->d_op && dentry->d_op->d_init) { err = dentry->d_op->d_init(dentry); if (err) { @@ -2003,6 +2020,58 @@ struct dentry *d_alloc(struct dentry * parent, const struct qstr *name) } EXPORT_SYMBOL(d_alloc); +/** + * d_duplicate - duplicate a dentry for combined atomic operation + * @dentry: the dentry to duplicate + * + * Some rename operations need to be combined with another operation + * inside the filesystem. + * 1/ A cluster filesystem when renaming to an in-use file might need to + * first "silly-rename" that target out of the way before the main rename + * 2/ A filesystem that supports white-out might want to create a whiteout + * in place of the file being moved. + * + * For this they need two dentries which temporarily have the same name, + * before one is renamed. d_duplicate() provides for this. Given a + * positive hashed dentry, it creates a second in-lookup dentry. + * Because the original dentry exists, no other thread will try to + * create an in-lookup dentry, so there can be no race in this create. + * + * The caller should d_move() the original to a new name, often via a + * rename request, and should call d_lookup_done() on the newly created + * dentry. If the new is instantiated then the old MUST either be moved + * or dropped. + * + * Parent must be locked. + * + * Returns: an in-lookup dentry, or -ENOMEM. + */ +struct dentry *d_duplicate(struct dentry *dentry) +{ + unsigned int hash = dentry->d_name.hash; + struct dentry *parent = dentry->d_parent; + struct hlist_bl_head *b = in_lookup_hash(parent, hash); + struct dentry *new = __d_alloc(parent->d_sb, &dentry->d_name); + + if (unlikely(!new)) + return ERR_PTR(-ENOMEM); + + new->d_flags |= DCACHE_PAR_LOOKUP; + lock_map_acquire_try(&new->lookup_map); + spin_lock(&parent->d_lock); + new->d_parent = dget_dlock(parent); + hlist_add_head(&new->d_sib, &parent->d_children); + if (parent->d_flags & DCACHE_DISCONNECTED) + new->d_flags |= DCACHE_DISCONNECTED; + spin_unlock(&parent->d_lock); + + hlist_bl_lock(b); + hlist_bl_add_head(&new->d_in_lookup_hash, b); + hlist_bl_unlock(b); + return new; +} +EXPORT_SYMBOL(d_duplicate); + struct dentry *d_alloc_anon(struct super_block *sb) { return __d_alloc(sb, NULL); @@ -2172,7 +2241,6 @@ static void __d_instantiate(struct dentry *dentry, struct inode *inode) * (or otherwise set) by the caller to indicate that it is now * in use by the dcache. */ - void d_instantiate(struct dentry *entry, struct inode * inode) { BUG_ON(d_really_is_positive(entry)); @@ -2241,7 +2309,12 @@ static struct dentry *__d_obtain_alias(struct inode *inode, bool disconnected) sb = inode->i_sb; - res = d_find_any_alias(inode); /* existing alias? */ + spin_lock(&inode->i_lock); + if (!inode_notify_dead(inode)) + res = __d_find_any_alias(inode); /* existing alias? */ + else + res = ERR_PTR(-ESTALE); + spin_unlock(&inode->i_lock); if (res) goto out; @@ -2253,7 +2326,10 @@ static struct dentry *__d_obtain_alias(struct inode *inode, bool disconnected) security_d_instantiate(new, inode); spin_lock(&inode->i_lock); - res = __d_find_any_alias(inode); /* recheck under lock */ + if (!inode_notify_dead(inode)) + res = __d_find_any_alias(inode); /* recheck under lock */ + else + res = ERR_PTR(-ESTALE); if (likely(!res)) { /* still no alias, attach a disconnected dentry */ unsigned add_flags = d_flags_for_inode(inode); @@ -2754,6 +2830,15 @@ static inline void end_dir_add(struct inode *dir, unsigned int n) static void d_wait_lookup(struct dentry *dentry) { if (likely(d_in_lookup(dentry))) { + /* + * Tell lockdep we will wait for the lookup lock, after + * dropping ->d_lock, but won't actually take it. + */ + spin_release(&dentry->d_lock.dep_map, _THIS_IP_); + lock_map_acquire(&dentry->lookup_map); + lock_map_release(&dentry->lookup_map); + spin_acquire(&dentry->d_lock.dep_map, 0, 1, _THIS_IP_); + dentry->d_flags |= DCACHE_LOOKUP_WAITERS; wait_var_event_spinlock(&dentry->d_flags, !d_in_lookup(dentry), @@ -2761,8 +2846,16 @@ static void d_wait_lookup(struct dentry *dentry) } } -struct dentry *d_alloc_parallel(struct dentry *parent, - const struct qstr *name) +/* What to do when __d_alloc_parallel finds a d_in_lookup dentry */ +enum alloc_para { + ALLOC_PARA_WAIT, + ALLOC_PARA_FAIL, +}; + +static inline +struct dentry *__d_alloc_parallel(struct dentry *parent, + const struct qstr *name, + enum alloc_para how) { unsigned int hash = name->hash; struct hlist_bl_head *b = in_lookup_hash(parent, hash); @@ -2835,6 +2928,12 @@ retry: spin_unlock(&dentry->d_lock); goto retry; } + if (unlikely(how == ALLOC_PARA_FAIL)) { + /* mustn't wait for concurrent lookup to complete */ + spin_unlock(&dentry->d_lock); + dput(new); + return ERR_PTR(-EWOULDBLOCK); + } /* * somebody is likely to be still doing lookup for it; * pin it and wait for them to finish @@ -2862,14 +2961,77 @@ retry: } hlist_bl_add_head(&new->d_in_lookup_hash, b); hlist_bl_unlock(b); + lock_map_acquire_try(&new->lookup_map); return new; mismatch: spin_unlock(&dentry->d_lock); dput(dentry); goto retry; } + +/** + * d_alloc_parallel() - allocate a new dentry and ensure uniqueness + * @parent: dentry of the parent + * @name: name of the dentry within that parent. + * + * A new dentry is allocated and, providing it is unique, added to the + * relevant index. + * If an existing dentry is found with the same parent/name that is + * not d_in_lookup(), then that is returned instead. + * If the existing dentry is d_in_lookup(), d_alloc_parallel() waits for + * that lookup to complete before returning the dentry and then ensures the + * match is still valid. + * Thus if the returned dentry is d_in_lookup() then the caller has + * exclusive access until it completes the lookup. + * If the returned dentry is not d_in_lookup() then a lookup has + * already completed. + * + * The @name must already have ->hash set, as can be achieved + * by e.g. try_lookup_noperm(). + * + * Returns: the dentry, whether found or allocated, or an error %-ENOMEM. + */ +struct dentry *d_alloc_parallel(struct dentry *parent, + const struct qstr *name) +{ + return __d_alloc_parallel(parent, name, ALLOC_PARA_WAIT); +} EXPORT_SYMBOL(d_alloc_parallel); +/** + * d_alloc_trylock() - find or allocate a new dentry + * @parent: dentry of the parent + * @name: name of the dentry within that parent. + * + * A new dentry is allocated and, providing it is unique, added to the + * relevant index. + * If an existing dentry is found with the same parent/name that is + * not d_in_lookup() then that is returned instead. + * If the existing dentry is d_in_lookup(), d_alloc_trylock() + * returns with error %-EWOULDBLOCK. + * Thus if the returned dentry is d_in_lookup() then the caller has + * exclusive access until it completes the lookup. + * If the returned dentry is not d_in_lookup() then a lookup has + * already completed. + * + * The @name need not already have ->hash set. + * + * Returns: the dentry, whether found or allocated, or an error + * %-ENOMEM, %-EWOULDBLOCK, %-EACCES (for a bad name) or + * anything returned by ->d_hash(). + */ +struct dentry *d_alloc_trylock(struct dentry *parent, + struct qstr *name) +{ + struct dentry *de; + + de = try_lookup_noperm(name, parent); + if (!de) + de = __d_alloc_parallel(parent, name, ALLOC_PARA_FAIL); + return de; +} +EXPORT_SYMBOL(d_alloc_trylock); + /* * Move dentry from in-lookup state to busy-negative one. * @@ -2898,6 +3060,7 @@ static void __d_lookup_unhash(struct dentry *dentry) b = in_lookup_hash(dentry->d_parent, dentry->d_name.hash); hlist_bl_lock(b); dentry->d_flags &= ~DCACHE_PAR_LOOKUP; + lock_map_release(&dentry->lookup_map); __hlist_bl_del(&dentry->d_in_lookup_hash); hlist_bl_unlock(b); dentry->waiters = NULL; @@ -2935,15 +3098,10 @@ static inline void __d_add(struct dentry *dentry, struct inode *inode, } if (unlikely(ops)) d_set_d_op(dentry, ops); - if (inode) { - unsigned add_flags = d_flags_for_inode(inode); - hlist_add_head(&dentry->d_alias, &inode->i_dentry); - raw_write_seqcount_begin(&dentry->d_seq); - __d_set_inode_and_type(dentry, inode, add_flags); - raw_write_seqcount_end(&dentry->d_seq); - fsnotify_update_flags(dentry); - } - __d_rehash(dentry); + if (inode) + __d_instantiate(dentry, inode); + if (d_unhashed(dentry)) + __d_rehash(dentry); if (dir) { end_dir_add(dir, n); __d_wake_in_lookup_waiters(dentry); @@ -3245,7 +3403,7 @@ struct dentry *d_splice_alias_ops(struct inode *inode, struct dentry *dentry, if (IS_ERR(inode)) return ERR_CAST(inode); - BUG_ON(!d_unhashed(dentry)); + BUG_ON(d_really_is_positive(dentry)); if (!inode) goto out; @@ -3301,6 +3459,8 @@ out: * @inode: the inode which may have a disconnected dentry * @dentry: a negative dentry which we want to point to the inode. * + * @dentry must be negative and may be in-lookup or unhashed or hashed. + * * If inode is a directory and has an IS_ROOT alias, then d_move that in * place of the given dentry and return it, else simply d_add the inode * to the dentry and return NULL. @@ -3308,16 +3468,14 @@ out: * If a non-IS_ROOT directory is found, the filesystem is corrupt, and * we should error out: directories can't have multiple aliases. * - * This is needed in the lookup routine of any filesystem that is exportable - * (via knfsd) so that we can build dcache paths to directories effectively. + * This should be used to return the result of ->lookup() and to + * instantiate the result of ->mkdir(), is often useful for + * ->atomic_open, and may be used to instantiate other objects. * * If a dentry was found and moved, then it is returned. Otherwise NULL - * is returned. This matches the expected return value of ->lookup. + * is returned. This matches the expected return value of ->lookup and + * ->mkdir. * - * Cluster filesystems may call this function with a negative, hashed dentry. - * In that case, we know that the inode will be a regular file, and also this - * will only occur during atomic_open. So we need to check for the dentry - * being already hashed only in the final case. */ struct dentry *d_splice_alias(struct inode *inode, struct dentry *dentry) { diff --git a/fs/debugfs/inode.c b/fs/debugfs/inode.c index a4d08bd3743b..b4915551ad31 100644 --- a/fs/debugfs/inode.c +++ b/fs/debugfs/inode.c @@ -42,7 +42,7 @@ static bool debugfs_enabled __ro_after_init = IS_ENABLED(CONFIG_DEBUG_FS_ALLOW_A * so that we can use the file mode as part of a heuristic to determine whether * to lock down individual files. */ -static int debugfs_setattr(struct mnt_idmap *idmap, +static int debugfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { int ret; diff --git a/fs/devpts/inode.c b/fs/devpts/inode.c index 9844dcf354ee..bd1e8eb26edb 100644 --- a/fs/devpts/inode.c +++ b/fs/devpts/inode.c @@ -249,6 +249,8 @@ static int devpts_parse_param(struct fs_context *fc, struct fs_parameter *param) case Opt_max: if (result.uint_32 > NR_UNIX98_PTY_MAX) return invalf(fc, "max out of range"); + if (result.uint_32 == 0) + return invalf(fc, "max must be greater than 0"); opts->max = result.uint_32; break; } diff --git a/fs/ecryptfs/inode.c b/fs/ecryptfs/inode.c index 525297c7ebd8..48e520960d66 100644 --- a/fs/ecryptfs/inode.c +++ b/fs/ecryptfs/inode.c @@ -266,7 +266,7 @@ out: * Returns zero on success; non-zero on error condition */ static int -ecryptfs_create(struct mnt_idmap *idmap, +ecryptfs_create(const struct mnt_idmap *idmap, struct inode *directory_inode, struct dentry *ecryptfs_dentry, umode_t mode) { @@ -462,7 +462,7 @@ static int ecryptfs_unlink(struct inode *dir, struct dentry *dentry) return ecryptfs_do_unlink(dir, dentry, d_inode(dentry)); } -static int ecryptfs_symlink(struct mnt_idmap *idmap, +static int ecryptfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { @@ -503,7 +503,7 @@ out_lock: return rc; } -static struct dentry *ecryptfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ecryptfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int rc; @@ -562,7 +562,7 @@ static int ecryptfs_rmdir(struct inode *dir, struct dentry *dentry) } static int -ecryptfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +ecryptfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { int rc; @@ -590,7 +590,7 @@ out: } static int -ecryptfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +ecryptfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -849,7 +849,7 @@ int ecryptfs_truncate(struct dentry *dentry, loff_t new_length) } static int -ecryptfs_permission(struct mnt_idmap *idmap, struct inode *inode, +ecryptfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return inode_permission(&nop_mnt_idmap, @@ -869,7 +869,7 @@ ecryptfs_permission(struct mnt_idmap *idmap, struct inode *inode, * All other metadata changes will be passed right to the lower filesystem, * and we will just update our inode to look like the lower. */ -static int ecryptfs_setattr(struct mnt_idmap *idmap, +static int ecryptfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { struct inode *inode = d_inode(dentry); @@ -939,7 +939,7 @@ out: return rc; } -static int ecryptfs_getattr_link(struct mnt_idmap *idmap, +static int ecryptfs_getattr_link(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -965,7 +965,7 @@ static int ecryptfs_getattr_link(struct mnt_idmap *idmap, return rc; } -static int ecryptfs_getattr(struct mnt_idmap *idmap, +static int ecryptfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1078,7 +1078,7 @@ static int ecryptfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return vfs_fileattr_get(ecryptfs_dentry_to_lower(dentry), fa); } -static int ecryptfs_fileattr_set(struct mnt_idmap *idmap, +static int ecryptfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct dentry *lower_dentry = ecryptfs_dentry_to_lower(dentry); @@ -1090,14 +1090,14 @@ static int ecryptfs_fileattr_set(struct mnt_idmap *idmap, return rc; } -static struct posix_acl *ecryptfs_get_acl(struct mnt_idmap *idmap, +static struct posix_acl *ecryptfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { return vfs_get_acl(idmap, ecryptfs_dentry_to_lower(dentry), posix_acl_xattr_name(type)); } -static int ecryptfs_set_acl(struct mnt_idmap *idmap, +static int ecryptfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { @@ -1158,7 +1158,7 @@ static int ecryptfs_xattr_get(const struct xattr_handler *handler, } static int ecryptfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/efivarfs/inode.c b/fs/efivarfs/inode.c index f0d009555fc6..07602cd5d33c 100644 --- a/fs/efivarfs/inode.c +++ b/fs/efivarfs/inode.c @@ -74,7 +74,7 @@ static bool efivarfs_valid_name(const char *str, int len) return uuid_is_valid(s); } -static int efivarfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int efivarfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode = NULL; @@ -150,7 +150,7 @@ efivarfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) } static int -efivarfs_fileattr_set(struct mnt_idmap *idmap, +efivarfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { unsigned int i_flags = 0; @@ -170,7 +170,7 @@ efivarfs_fileattr_set(struct mnt_idmap *idmap, } /* copy of simple_setattr except that it doesn't do i_size updates */ -static int efivarfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int efivarfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/erofs/inode.c b/fs/erofs/inode.c index 45afe5c50de8..26ea3790ff21 100644 --- a/fs/erofs/inode.c +++ b/fs/erofs/inode.c @@ -311,7 +311,7 @@ struct inode *erofs_iget(struct super_block *sb, erofs_nid_t nid) return inode; } -int erofs_getattr(struct mnt_idmap *idmap, const struct path *path, +int erofs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/erofs/internal.h b/fs/erofs/internal.h index 12e3a5b80a5a..ab817091bd29 100644 --- a/fs/erofs/internal.h +++ b/fs/erofs/internal.h @@ -417,7 +417,7 @@ void erofs_onlinefolio_init(struct folio *folio); void erofs_onlinefolio_split(struct folio *folio); void erofs_onlinefolio_end(struct folio *folio, int err, bool dirty); struct inode *erofs_iget(struct super_block *sb, erofs_nid_t nid); -int erofs_getattr(struct mnt_idmap *idmap, const struct path *path, +int erofs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); int erofs_namei(struct inode *dir, const struct qstr *name, diff --git a/fs/eventfd.c b/fs/eventfd.c index 9d33a02757d5..52426795752e 100644 --- a/fs/eventfd.c +++ b/fs/eventfd.c @@ -403,8 +403,8 @@ static int do_eventfd(unsigned int count, int flags) FD_PREPARE(fdf, flags, anon_inode_getfile_fmode("[eventfd]", &eventfd_fops, ctx, flags, FMODE_NOWAIT)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; ctx->id = ida_alloc(&eventfd_ida, GFP_KERNEL); retain_and_null_ptr(ctx); diff --git a/fs/eventpoll.c b/fs/eventpoll.c index e0c4bf88a838..f48b829a710f 100644 --- a/fs/eventpoll.c +++ b/fs/eventpoll.c @@ -2514,11 +2514,11 @@ static int do_epoll_create(int flags) FD_PREPARE(fdf, O_RDWR | (flags & O_CLOEXEC), anon_inode_getfile("[eventpoll]", &eventpoll_fops, ep, O_RDWR | (flags & O_CLOEXEC))); - if (fdf.err) { + if (fdf->fd < 0) { ep_clear_and_put(ep); - return fdf.err; + return fdf->fd; } - ep->file = fd_prepare_file(fdf); + ep->file = fdf->file; return fd_publish(fdf); } diff --git a/fs/exec.c b/fs/exec.c index a5269b5e00df..33a1e4689e49 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1136,6 +1136,7 @@ static void posixtimer_exec(struct task_struct *me) int begin_new_exec(struct linux_binprm * bprm) { struct task_struct *me = current; + struct files_struct *files = NULL; int retval; /* A pending PT_INTERP substitution this format cannot consume. */ @@ -1160,6 +1161,13 @@ int begin_new_exec(struct linux_binprm * bprm) */ bprm->point_of_no_return = true; + /* + * Cancel any io_uring activity across execve. This runs task work + * that may still create an io-wq worker, so do it while de_thread() + * can still zap it. + */ + io_uring_task_cancel(); + /* Make this the only thread in the thread group */ retval = de_thread(me); if (retval) @@ -1176,15 +1184,13 @@ int begin_new_exec(struct linux_binprm * bprm) /* see the comment in check_unsafe_exec() */ current->fs->in_exec = 0; - /* - * Cancel any io_uring activity across execve - */ - io_uring_task_cancel(); /* Ensure the files table is not shared. */ - retval = unshare_files(); + retval = unshare_fd(CLONE_FILES, &files); if (retval) goto out; + if (files) + switch_files_struct(me, files); /* * We have to apply CLOEXEC before we change whether the process is @@ -1192,13 +1198,13 @@ int begin_new_exec(struct linux_binprm * bprm) * trying to access the should-be-closed file descriptors of a process * undergoing exec(2). * - * This can block on filesystem ->flush() handlers, including waiting - * for FUSE daemons, so do it before exec_mmap takes the - * exec_update_lock. + * This can block on filesystem ->flush() and ->release() handlers, + * including waiting for FUSE daemons, so do it before exec_mmap + * takes the exec_update_lock. * This must happen after the point of no return, and after unsharing * the FD table. */ - do_close_on_exec(me->files); + close_cloexec_files(me->files); /* * Must be called _before_ exec_mmap() as bprm->mm is @@ -1359,7 +1365,7 @@ EXPORT_SYMBOL(begin_new_exec); void would_dump(struct linux_binprm *bprm, struct file *file) { struct inode *inode = file_inode(file); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); if (inode_permission(idmap, inode, MAY_READ) < 0) { struct user_namespace *old, *user_ns; bprm->interp_flags |= BINPRM_FLAGS_ENFORCE_NONDUMP; @@ -1643,7 +1649,7 @@ static void check_unsafe_exec(struct linux_binprm *bprm) static void bprm_fill_uid(struct linux_binprm *bprm, struct file *file) { /* Handle suid and sgid on files */ - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode = file_inode(file); unsigned int mode; vfsuid_t vfsuid; diff --git a/fs/exfat/exfat_fs.h b/fs/exfat/exfat_fs.h index 41a2c7dfc479..5f258e96fce9 100644 --- a/fs/exfat/exfat_fs.h +++ b/fs/exfat/exfat_fs.h @@ -556,9 +556,9 @@ int exfat_trim_fs(struct inode *inode, struct fstrim_range *range); /* file.c */ extern const struct file_operations exfat_file_operations; int __exfat_truncate(struct inode *inode); -int exfat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int exfat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int exfat_getattr(struct mnt_idmap *idmap, const struct path *path, +int exfat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags); struct file_kattr; diff --git a/fs/exfat/file.c b/fs/exfat/file.c index a2a9ee1a2004..3867e78c2312 100644 --- a/fs/exfat/file.c +++ b/fs/exfat/file.c @@ -143,7 +143,7 @@ error: return err; } -static bool exfat_allow_set_time(struct mnt_idmap *idmap, +static bool exfat_allow_set_time(const struct mnt_idmap *idmap, struct exfat_sb_info *sbi, struct inode *inode) { mode_t allow_utime = sbi->options.allow_utime; @@ -319,7 +319,7 @@ write_size: mutex_unlock(&sbi->s_lock); } -int exfat_getattr(struct mnt_idmap *idmap, const struct path *path, +int exfat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags) { @@ -347,7 +347,7 @@ int exfat_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int exfat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int exfat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct exfat_sb_info *sbi = EXFAT_SB(dentry->d_sb); diff --git a/fs/exfat/misc.c b/fs/exfat/misc.c index 6f11a96a4ffa..dfd0bbf31c94 100644 --- a/fs/exfat/misc.c +++ b/fs/exfat/misc.c @@ -187,7 +187,7 @@ int exfat_update_bhs(struct buffer_head **bhs, int nr_bhs, int sync) for (i = 0; i < nr_bhs && sync; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; diff --git a/fs/exfat/namei.c b/fs/exfat/namei.c index 3c5746fc57d9..d116c89d724e 100644 --- a/fs/exfat/namei.c +++ b/fs/exfat/namei.c @@ -552,7 +552,7 @@ out: return ret; } -static int exfat_create(struct mnt_idmap *idmap, struct inode *dir, +static int exfat_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -826,7 +826,7 @@ unlock: return err; } -static struct dentry *exfat_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *exfat_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1264,7 +1264,7 @@ out: return ret; } -static int exfat_rename(struct mnt_idmap *idmap, +static int exfat_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/ext2/acl.c b/fs/ext2/acl.c index 7e54c31589c7..b2746657fc53 100644 --- a/fs/ext2/acl.c +++ b/fs/ext2/acl.c @@ -219,7 +219,7 @@ __ext2_set_acl(struct inode *inode, struct posix_acl *acl, int type) * inode->i_mutex: down */ int -ext2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ext2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; diff --git a/fs/ext2/acl.h b/fs/ext2/acl.h index 4a8443a2b8ec..e68bc3545608 100644 --- a/fs/ext2/acl.h +++ b/fs/ext2/acl.h @@ -56,7 +56,7 @@ static inline int ext2_acl_count(size_t size) /* acl.c */ extern struct posix_acl *ext2_get_acl(struct inode *inode, int type, bool rcu); -extern int ext2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int ext2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ext2_init_acl (struct inode *, struct inode *); diff --git a/fs/ext2/ext2.h b/fs/ext2/ext2.h index 7aeb7cfb0ceb..7bdada93dd06 100644 --- a/fs/ext2/ext2.h +++ b/fs/ext2/ext2.h @@ -741,8 +741,8 @@ extern int ext2_sync_inode_metadata(struct inode *, struct writeback_control *); extern void ext2_evict_inode(struct inode *); void ext2_write_failed(struct address_space *mapping, loff_t to); extern int ext2_get_block(struct inode *, sector_t, struct buffer_head *, int); -extern int ext2_setattr (struct mnt_idmap *, struct dentry *, struct iattr *); -extern int ext2_getattr (struct mnt_idmap *, const struct path *, +extern int ext2_setattr (const struct mnt_idmap *, struct dentry *, struct iattr *); +extern int ext2_getattr (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext2_set_inode_flags(struct inode *inode); extern int ext2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, @@ -750,7 +750,7 @@ extern int ext2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, /* ioctl.c */ extern int ext2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -extern int ext2_fileattr_set(struct mnt_idmap *idmap, +extern int ext2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); extern long ext2_ioctl(struct file *, unsigned int, unsigned long); extern long ext2_compat_ioctl(struct file *, unsigned int, unsigned long); diff --git a/fs/ext2/inode.c b/fs/ext2/inode.c index 1a1ea1fd485b..12ac4cfe1500 100644 --- a/fs/ext2/inode.c +++ b/fs/ext2/inode.c @@ -1598,7 +1598,7 @@ out: return err; } -int ext2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ext2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -1624,7 +1624,7 @@ int ext2_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int ext2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ext2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ext2/ioctl.c b/fs/ext2/ioctl.c index c3fea55b8efa..f2218455fa47 100644 --- a/fs/ext2/ioctl.c +++ b/fs/ext2/ioctl.c @@ -27,7 +27,7 @@ int ext2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ext2_fileattr_set(struct mnt_idmap *idmap, +int ext2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ext2/namei.c b/fs/ext2/namei.c index 8666233ec63b..bfb6a463a95e 100644 --- a/fs/ext2/namei.c +++ b/fs/ext2/namei.c @@ -97,7 +97,7 @@ struct dentry *ext2_get_parent(struct dentry *child) * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ext2_create (struct mnt_idmap * idmap, +static int ext2_create (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { @@ -117,7 +117,7 @@ static int ext2_create (struct mnt_idmap * idmap, return ext2_add_nondir(dentry, inode); } -static int ext2_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ext2_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = ext2_new_inode(dir, mode, NULL); @@ -131,7 +131,7 @@ static int ext2_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int ext2_mknod (struct mnt_idmap * idmap, struct inode * dir, +static int ext2_mknod (const struct mnt_idmap * idmap, struct inode * dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode * inode; @@ -152,7 +152,7 @@ static int ext2_mknod (struct mnt_idmap * idmap, struct inode * dir, return err; } -static int ext2_symlink (struct mnt_idmap * idmap, struct inode * dir, +static int ext2_symlink (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, const char * symname) { struct super_block * sb = dir->i_sb; @@ -223,7 +223,7 @@ static int ext2_link (struct dentry * old_dentry, struct inode * dir, return err; } -static struct dentry *ext2_mkdir(struct mnt_idmap * idmap, +static struct dentry *ext2_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { @@ -316,7 +316,7 @@ static int ext2_rmdir (struct inode * dir, struct dentry *dentry) return err; } -static int ext2_rename (struct mnt_idmap * idmap, +static int ext2_rename (const struct mnt_idmap * idmap, struct inode * old_dir, struct dentry * old_dentry, struct inode * new_dir, struct dentry * new_dentry, unsigned int flags) diff --git a/fs/ext2/xattr.c b/fs/ext2/xattr.c index 9b68c490ab26..8f608930a48c 100644 --- a/fs/ext2/xattr.c +++ b/fs/ext2/xattr.c @@ -769,7 +769,7 @@ ext2_xattr_set2(struct inode *inode, struct buffer_head *old_bh, if (IS_SYNC(inode)) { sync_dirty_buffer(new_bh); error = -EIO; - if (buffer_req(new_bh) && !buffer_uptodate(new_bh)) + if (buffer_write_io_error(new_bh)) goto cleanup; } } diff --git a/fs/ext2/xattr_security.c b/fs/ext2/xattr_security.c index db47b8ab153e..ade074354258 100644 --- a/fs/ext2/xattr_security.c +++ b/fs/ext2/xattr_security.c @@ -19,7 +19,7 @@ ext2_xattr_security_get(const struct xattr_handler *handler, static int ext2_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext2/xattr_trusted.c b/fs/ext2/xattr_trusted.c index 995f931228ce..0f12d634d6d0 100644 --- a/fs/ext2/xattr_trusted.c +++ b/fs/ext2/xattr_trusted.c @@ -26,7 +26,7 @@ ext2_xattr_trusted_get(const struct xattr_handler *handler, static int ext2_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext2/xattr_user.c b/fs/ext2/xattr_user.c index dd1507231081..48002c033e9c 100644 --- a/fs/ext2/xattr_user.c +++ b/fs/ext2/xattr_user.c @@ -30,7 +30,7 @@ ext2_xattr_user_get(const struct xattr_handler *handler, static int ext2_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/acl.c b/fs/ext4/acl.c index 3bffe862f954..59fac55a2426 100644 --- a/fs/ext4/acl.c +++ b/fs/ext4/acl.c @@ -225,7 +225,7 @@ __ext4_set_acl(handle_t *handle, struct inode *inode, int type, } int -ext4_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ext4_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { handle_t *handle; diff --git a/fs/ext4/acl.h b/fs/ext4/acl.h index 0c5a79c3b5d4..a14838c5bc42 100644 --- a/fs/ext4/acl.h +++ b/fs/ext4/acl.h @@ -56,7 +56,7 @@ static inline int ext4_acl_count(size_t size) /* acl.c */ struct posix_acl *ext4_get_acl(struct inode *inode, int type, bool rcu); -int ext4_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ext4_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ext4_init_acl(handle_t *, struct inode *, struct inode *); diff --git a/fs/ext4/ext4.h b/fs/ext4/ext4.h index 724a27e8be61..cbc59d03ca81 100644 --- a/fs/ext4/ext4.h +++ b/fs/ext4/ext4.h @@ -3044,7 +3044,7 @@ extern int ext4fs_dirhash(const struct inode *dir, const char *name, int len, /* ialloc.c */ extern int ext4_mark_inode_used(struct super_block *sb, int ino); -extern struct inode *__ext4_new_inode(struct mnt_idmap *, handle_t *, +extern struct inode *__ext4_new_inode(const struct mnt_idmap *, handle_t *, struct inode *, umode_t, const struct qstr *qstr, __u32 goal, uid_t *owner, __u32 i_flags, @@ -3179,14 +3179,14 @@ extern struct inode *__ext4_iget(struct super_block *sb, unsigned long ino, extern int ext4_write_inode(struct inode *, struct writeback_control *); extern int ext4_sync_inode_metadata(struct inode *, struct writeback_control *); -extern int ext4_setattr(struct mnt_idmap *, struct dentry *, +extern int ext4_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern u32 ext4_dio_alignment(struct inode *inode); -extern int ext4_getattr(struct mnt_idmap *, const struct path *, +extern int ext4_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext4_evict_inode(struct inode *); extern void ext4_clear_inode(struct inode *); -extern int ext4_file_getattr(struct mnt_idmap *, const struct path *, +extern int ext4_file_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext4_dirty_inode(struct inode *, int); extern int ext4_change_inode_journal_flag(struct inode *, int); @@ -3246,7 +3246,7 @@ extern int ext4_ind_remove_space(handle_t *handle, struct inode *inode, /* ioctl.c */ extern long ext4_ioctl(struct file *, unsigned int, unsigned long); extern long ext4_compat_ioctl(struct file *, unsigned int, unsigned long); -int ext4_fileattr_set(struct mnt_idmap *idmap, +int ext4_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ext4_fileattr_get(struct dentry *dentry, struct file_kattr *fa); extern void ext4_reset_inode_seed(struct inode *inode); diff --git a/fs/ext4/ext4_jbd2.c b/fs/ext4/ext4_jbd2.c index 53ddedb52a6f..c241f50b97bc 100644 --- a/fs/ext4/ext4_jbd2.c +++ b/fs/ext4/ext4_jbd2.c @@ -421,7 +421,7 @@ int __ext4_handle_dirty_metadata(const char *where, unsigned int line, } if (inode && inode_needs_sync(inode)) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ext4_error_inode_err(inode, where, line, bh->b_blocknr, EIO, "IO error syncing itable block"); diff --git a/fs/ext4/ialloc.c b/fs/ext4/ialloc.c index a5831fc536db..529623103ae7 100644 --- a/fs/ext4/ialloc.c +++ b/fs/ext4/ialloc.c @@ -930,7 +930,7 @@ static int ext4_xattr_credits_for_new_inode(struct inode *dir, mode_t mode, * For other inodes, search forward from the parent directory's block * group to find a free inode. */ -struct inode *__ext4_new_inode(struct mnt_idmap *idmap, +struct inode *__ext4_new_inode(const struct mnt_idmap *idmap, handle_t *handle, struct inode *dir, umode_t mode, const struct qstr *qstr, __u32 goal, uid_t *owner, __u32 i_flags, diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c index 26f0f9714f03..cb68bf50a3d6 100644 --- a/fs/ext4/inode.c +++ b/fs/ext4/inode.c @@ -6006,7 +6006,7 @@ static void ext4_wait_for_tail_page_commit(struct inode *inode) * * Called with inode->i_rwsem down. */ -int ext4_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ext4_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -6263,7 +6263,7 @@ u32 ext4_dio_alignment(struct inode *inode) return 1; /* use the iomap defaults */ } -int ext4_getattr(struct mnt_idmap *idmap, const struct path *path, +int ext4_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -6332,7 +6332,7 @@ int ext4_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int ext4_file_getattr(struct mnt_idmap *idmap, +int ext4_file_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/ext4/ioctl.c b/fs/ext4/ioctl.c index c8387e6a2c6e..0a54b00e5be5 100644 --- a/fs/ext4/ioctl.c +++ b/fs/ext4/ioctl.c @@ -373,7 +373,7 @@ void ext4_reset_inode_seed(struct inode *inode) * */ static long swap_inode_boot_loader(struct super_block *sb, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode) { handle_t *handle; @@ -1008,7 +1008,7 @@ int ext4_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ext4_fileattr_set(struct mnt_idmap *idmap, +int ext4_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -1539,7 +1539,7 @@ static long __ext4_ioctl(struct file *filp, unsigned int cmd, unsigned long arg) { struct inode *inode = file_inode(filp); struct super_block *sb = inode->i_sb; - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); ext4_debug("cmd = %u, arg = %lu\n", cmd, arg); diff --git a/fs/ext4/mmp.c b/fs/ext4/mmp.c index 7ce361484b38..4b18ddef468d 100644 --- a/fs/ext4/mmp.c +++ b/fs/ext4/mmp.c @@ -49,7 +49,7 @@ static int write_mmp_block_thawed(struct super_block *sb, bh_submit(bh, REQ_OP_WRITE | REQ_SYNC | REQ_META | REQ_PRIO, bh_end_write); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) return -EIO; return 0; } diff --git a/fs/ext4/namei.c b/fs/ext4/namei.c index a6386c1d237f..6e0630a49e48 100644 --- a/fs/ext4/namei.c +++ b/fs/ext4/namei.c @@ -2812,7 +2812,7 @@ static int ext4_add_nondir(handle_t *handle, * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ext4_create(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { handle_t *handle; @@ -2847,7 +2847,7 @@ retry: return err; } -static int ext4_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { handle_t *handle; @@ -2881,7 +2881,7 @@ retry: return err; } -static int ext4_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { handle_t *handle; @@ -2994,7 +2994,7 @@ out: return err; } -static struct dentry *ext4_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ext4_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { handle_t *handle; @@ -3360,7 +3360,7 @@ out: return err; } -static int ext4_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { handle_t *handle; @@ -3753,7 +3753,7 @@ static void ext4_update_dir_count(handle_t *handle, struct ext4_renament *ent) } } -static struct inode *ext4_whiteout_for_rename(struct mnt_idmap *idmap, +static struct inode *ext4_whiteout_for_rename(const struct mnt_idmap *idmap, struct ext4_renament *ent, int credits, handle_t **h) { @@ -3796,7 +3796,7 @@ retry: * while new_{dentry,inode) refers to the destination dentry/inode * This comes from rename(const char *oldpath, const char *newpath) */ -static int ext4_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ext4_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -4191,7 +4191,7 @@ end_rename: return retval; } -static int ext4_rename2(struct mnt_idmap *idmap, +static int ext4_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/ext4/symlink.c b/fs/ext4/symlink.c index b612262719ed..e680d1e45b47 100644 --- a/fs/ext4/symlink.c +++ b/fs/ext4/symlink.c @@ -55,7 +55,7 @@ static const char *ext4_encrypted_get_link(struct dentry *dentry, return paddr; } -static int ext4_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int ext4_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) diff --git a/fs/ext4/xattr_hurd.c b/fs/ext4/xattr_hurd.c index 8a5842e4cd95..a3ecbff72b10 100644 --- a/fs/ext4/xattr_hurd.c +++ b/fs/ext4/xattr_hurd.c @@ -32,7 +32,7 @@ ext4_xattr_hurd_get(const struct xattr_handler *handler, static int ext4_xattr_hurd_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_security.c b/fs/ext4/xattr_security.c index 776cf11d24ca..af5b8a93fed1 100644 --- a/fs/ext4/xattr_security.c +++ b/fs/ext4/xattr_security.c @@ -23,7 +23,7 @@ ext4_xattr_security_get(const struct xattr_handler *handler, static int ext4_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_trusted.c b/fs/ext4/xattr_trusted.c index 9811eb0ab276..458e1982ef83 100644 --- a/fs/ext4/xattr_trusted.c +++ b/fs/ext4/xattr_trusted.c @@ -30,7 +30,7 @@ ext4_xattr_trusted_get(const struct xattr_handler *handler, static int ext4_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_user.c b/fs/ext4/xattr_user.c index 4b70bf4e7626..ad35215f6610 100644 --- a/fs/ext4/xattr_user.c +++ b/fs/ext4/xattr_user.c @@ -31,7 +31,7 @@ ext4_xattr_user_get(const struct xattr_handler *handler, static int ext4_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/f2fs/acl.c b/fs/f2fs/acl.c index 34c9aa279040..a7485bc38252 100644 --- a/fs/f2fs/acl.c +++ b/fs/f2fs/acl.c @@ -219,7 +219,7 @@ struct posix_acl *f2fs_get_acl(struct inode *inode, int type, bool rcu) return __f2fs_get_acl(inode, type, NULL); } -static int f2fs_acl_update_mode(struct mnt_idmap *idmap, +static int f2fs_acl_update_mode(const struct mnt_idmap *idmap, struct inode *inode, umode_t *mode_p, struct posix_acl **acl) { @@ -240,7 +240,7 @@ static int f2fs_acl_update_mode(struct mnt_idmap *idmap, return 0; } -static int __f2fs_set_acl(struct mnt_idmap *idmap, +static int __f2fs_set_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, struct posix_acl *acl, struct f2fs_cached_block *ientry) { @@ -289,7 +289,7 @@ static int __f2fs_set_acl(struct mnt_idmap *idmap, return error; } -int f2fs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/f2fs/acl.h b/fs/f2fs/acl.h index 0f639367a0ab..b1085efcc05e 100644 --- a/fs/f2fs/acl.h +++ b/fs/f2fs/acl.h @@ -34,7 +34,7 @@ struct f2fs_acl_header { #ifdef CONFIG_F2FS_FS_POSIX_ACL struct posix_acl *f2fs_get_acl(struct inode *, int, bool); -int f2fs_set_acl(struct mnt_idmap *, struct dentry *, +int f2fs_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); int f2fs_init_acl(struct inode *inode, struct inode *dir, struct f2fs_cached_block *ientry, diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 089a62c054ea..c1f1e339f085 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -3972,9 +3972,9 @@ int f2fs_sync_file(struct file *file, loff_t start, loff_t end, int datasync); int f2fs_do_truncate_blocks(struct inode *inode, u64 from, bool lock); int f2fs_truncate_blocks(struct inode *inode, u64 from, bool lock); int f2fs_truncate(struct inode *inode); -int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, +int f2fs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int f2fs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int f2fs_truncate_hole(struct inode *inode, pgoff_t pg_start, pgoff_t pg_end); void f2fs_truncate_data_blocks_range(struct dnode_of_data *dn, int count); @@ -3982,7 +3982,7 @@ int f2fs_do_shutdown(struct f2fs_sb_info *sbi, unsigned int flag, bool readonly, bool need_lock); int f2fs_precache_extents(struct inode *inode); int f2fs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int f2fs_fileattr_set(struct mnt_idmap *idmap, +int f2fs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long f2fs_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); long f2fs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); @@ -4014,7 +4014,7 @@ void f2fs_destroy_evict_inode_work(void); int f2fs_update_extension_list(struct f2fs_sb_info *sbi, const char *name, bool hot, bool set); struct dentry *f2fs_get_parent(struct dentry *child); -int f2fs_get_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int f2fs_get_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct inode **new_inode); /* diff --git a/fs/f2fs/file.c b/fs/f2fs/file.c index ef4d218e694b..626c6f97b6f8 100644 --- a/fs/f2fs/file.c +++ b/fs/f2fs/file.c @@ -1033,7 +1033,7 @@ static bool f2fs_force_buffered_io(struct inode *inode, int rw) return false; } -int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, +int f2fs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -1097,7 +1097,7 @@ int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, } #ifdef CONFIG_F2FS_FS_POSIX_ACL -static void __setattr_copy(struct mnt_idmap *idmap, +static void __setattr_copy(const struct mnt_idmap *idmap, struct inode *inode, const struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; @@ -1122,7 +1122,7 @@ static void __setattr_copy(struct mnt_idmap *idmap, #define __setattr_copy setattr_copy #endif -int f2fs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -2364,7 +2364,7 @@ static int f2fs_ioc_getversion(struct file *filp, unsigned long arg) static int f2fs_ioc_start_atomic_write(struct file *filp, bool truncate) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); struct f2fs_inode_info *fi = F2FS_I(inode); struct f2fs_sb_info *sbi = F2FS_I_SB(inode); loff_t isize; @@ -2476,7 +2476,7 @@ out: static int f2fs_ioc_commit_atomic_write(struct file *filp) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); int ret; if (!(filp->f_mode & FMODE_WRITE)) @@ -2511,7 +2511,7 @@ static int f2fs_ioc_commit_atomic_write(struct file *filp) static int f2fs_ioc_abort_atomic_write(struct file *filp) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); int ret; if (!(filp->f_mode & FMODE_WRITE)) @@ -3585,7 +3585,7 @@ int f2fs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int f2fs_fileattr_set(struct mnt_idmap *idmap, +int f2fs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/f2fs/namei.c b/fs/f2fs/namei.c index 38fcea8b72bf..ce5d536d892b 100644 --- a/fs/f2fs/namei.c +++ b/fs/f2fs/namei.c @@ -231,7 +231,7 @@ static void set_file_temperature(struct f2fs_sb_info *sbi, struct inode *inode, file_set_hot(inode); } -static struct inode *f2fs_new_inode(struct mnt_idmap *idmap, +static struct inode *f2fs_new_inode(const struct mnt_idmap *idmap, struct inode *dir, umode_t mode, const char *name) { @@ -365,7 +365,7 @@ fail_drop: return ERR_PTR(err); } -static int f2fs_create(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -663,7 +663,7 @@ static const char *f2fs_get_link(struct dentry *dentry, return link; } -static int f2fs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -752,7 +752,7 @@ free_inode: goto out; } -static struct dentry *f2fs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *f2fs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -811,7 +811,7 @@ static int f2fs_rmdir(struct inode *dir, struct dentry *dentry) return -ENOTEMPTY; } -static int f2fs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -858,7 +858,7 @@ out: return err; } -static int __f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int __f2fs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode, bool is_whiteout, struct inode **new_inode, struct f2fs_filename *fname) { @@ -929,7 +929,7 @@ out: return err; } -static int f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -945,7 +945,7 @@ static int f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, err); } -static int f2fs_create_whiteout(struct mnt_idmap *idmap, +static int f2fs_create_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct inode **whiteout, struct f2fs_filename *fname) { @@ -953,14 +953,14 @@ static int f2fs_create_whiteout(struct mnt_idmap *idmap, true, whiteout, fname); } -int f2fs_get_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int f2fs_get_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct inode **new_inode) { return __f2fs_tmpfile(idmap, dir, NULL, S_IFREG, false, new_inode, NULL); } -static int f2fs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int f2fs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1345,7 +1345,7 @@ out: return err; } -static int f2fs_rename2(struct mnt_idmap *idmap, +static int f2fs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1398,7 +1398,7 @@ static const char *f2fs_encrypted_get_link(struct dentry *dentry, return target; } -static int f2fs_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int f2fs_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) diff --git a/fs/f2fs/xattr.c b/fs/f2fs/xattr.c index 4328c9d9de45..0ed879fc3076 100644 --- a/fs/f2fs/xattr.c +++ b/fs/f2fs/xattr.c @@ -67,7 +67,7 @@ static int f2fs_xattr_generic_get(const struct xattr_handler *handler, } static int f2fs_xattr_generic_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -111,7 +111,7 @@ static int f2fs_xattr_advise_get(const struct xattr_handler *handler, } static int f2fs_xattr_advise_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/failfs.c b/fs/failfs.c index 66a36da3d236..437cdc981c2d 100644 --- a/fs/failfs.c +++ b/fs/failfs.c @@ -22,7 +22,7 @@ bool failfs_mnt(const struct vfsmount *mnt) return mnt->mnt_sb == failfs_root_path.mnt->mnt_sb; } -static int failfs_permission(struct mnt_idmap *idmap, struct inode *inode, +static int failfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return -EOPNOTSUPP; @@ -35,7 +35,7 @@ static struct dentry *failfs_lookup(struct inode *dir, struct dentry *dentry, return ERR_PTR(-EOPNOTSUPP); } -static int failfs_getattr(struct mnt_idmap *idmap, const struct path *path, +static int failfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/fat/fat.h b/fs/fat/fat.h index 61338413d9f3..dbbcfc90a9c2 100644 --- a/fs/fat/fat.h +++ b/fs/fat/fat.h @@ -404,10 +404,10 @@ extern long fat_generic_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); extern const struct file_operations fat_file_operations; extern const struct inode_operations fat_file_inode_operations; -extern int fat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int fat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void fat_truncate_blocks(struct inode *inode, loff_t offset); -extern int fat_getattr(struct mnt_idmap *idmap, +extern int fat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); int fat_fileattr_get(struct dentry *dentry, struct file_kattr *fa); diff --git a/fs/fat/file.c b/fs/fat/file.c index 1c835ca5f21a..2c6aee9f0305 100644 --- a/fs/fat/file.c +++ b/fs/fat/file.c @@ -432,7 +432,7 @@ int fat_fileattr_get(struct dentry *dentry, struct file_kattr *fa) } EXPORT_SYMBOL_GPL(fat_fileattr_get); -int fat_getattr(struct mnt_idmap *idmap, const struct path *path, +int fat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); @@ -493,7 +493,7 @@ static int fat_sanitize_mode(const struct msdos_sb_info *sbi, return 0; } -static int fat_allow_set_time(struct mnt_idmap *idmap, +static int fat_allow_set_time(const struct mnt_idmap *idmap, struct msdos_sb_info *sbi, struct inode *inode) { umode_t allow_utime = sbi->options.allow_utime; @@ -514,7 +514,7 @@ static int fat_allow_set_time(struct mnt_idmap *idmap, /* valid file mode bits */ #define FAT_VALID_MODE (S_IFREG | S_IFDIR | S_IRWXUGO) -int fat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct msdos_sb_info *sbi = MSDOS_SB(dentry->d_sb); diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf1975..0d04228f916e 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -360,7 +360,7 @@ int fat_sync_bhs(struct buffer_head **bhs, int nr_bhs) for (i = 0; i < nr_bhs; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; diff --git a/fs/fat/namei_msdos.c b/fs/fat/namei_msdos.c index d46d1a3851f2..dde4215616f9 100644 --- a/fs/fat/namei_msdos.c +++ b/fs/fat/namei_msdos.c @@ -263,7 +263,7 @@ static int msdos_add_entry(struct inode *dir, const unsigned char *name, } /***** Create a file */ -static int msdos_create(struct mnt_idmap *idmap, struct inode *dir, +static int msdos_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -345,7 +345,7 @@ out: } /***** Make a directory */ -static struct dentry *msdos_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *msdos_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -600,7 +600,7 @@ error_inode: } /***** Rename, a wrapper for rename_same_dir & rename_diff_dir */ -static int msdos_rename(struct mnt_idmap *idmap, +static int msdos_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/fat/namei_vfat.c b/fs/fat/namei_vfat.c index da3e89c0b16a..3dc063ba0a73 100644 --- a/fs/fat/namei_vfat.c +++ b/fs/fat/namei_vfat.c @@ -753,7 +753,7 @@ error: return ERR_PTR(err); } -static int vfat_create(struct mnt_idmap *idmap, struct inode *dir, +static int vfat_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -846,7 +846,7 @@ out: return err; } -static struct dentry *vfat_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *vfat_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1160,7 +1160,7 @@ error_exchange: goto out; } -static int vfat_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +static int vfat_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/fhandle.c b/fs/fhandle.c index f8829231e3d7..2aa55b8a878a 100644 --- a/fs/fhandle.c +++ b/fs/fhandle.c @@ -201,7 +201,7 @@ static int vfs_dentry_acceptable(void *context, struct dentry *dentry) struct handle_to_path_ctx *ctx = context; struct user_namespace *user_ns = current_user_ns(); struct dentry *d, *root = ctx->root.dentry; - struct mnt_idmap *idmap = mnt_idmap(ctx->root.mnt); + const struct mnt_idmap *idmap = mnt_idmap(ctx->root.mnt); int retval = 0; if (!root) diff --git a/fs/file.c b/fs/file.c index 628ca07dc4b1..88b7a5340815 100644 --- a/fs/file.c +++ b/fs/file.c @@ -352,24 +352,78 @@ static inline bool fd_is_open(unsigned int fd, const struct fdtable *fdt) return test_bit(fd, fdt->open_fds); } +/* Bits of [range->from, range->to] that fall into word @i of a bitmap. */ +static unsigned long fd_range_word(struct fd_range *range, unsigned int i) +{ + unsigned int first = i * BITS_PER_LONG; + unsigned int last = first + BITS_PER_LONG - 1; + + if (range->to < first || range->from > last) + return 0; + return GENMASK(min(range->to, last) - first, + max(range->from, first) - first); +} + +/* Bits of word @i that dup_fd() leaves behind and __range_close() closes. */ +static unsigned long dup_fd_dropped_word(struct fdtable *fdt, unsigned int i, + struct fd_range *range) +{ + unsigned long dropped; + + if (!range) + return 0; + dropped = fd_range_word(range, i); + if (range->flags & FD_RANGE_EXCEPT) + dropped = ~dropped; + if (range->flags & FD_RANGE_CLOEXEC_ONLY) + dropped &= fdt->close_on_exec[i]; + return dropped; +} + /* * Note that a sane fdtable size always has to be a multiple of * BITS_PER_LONG, since we have bitmaps that are sized by this. * - * punch_hole is optional - when close_range() is asked to unshare - * and close, we don't need to copy descriptors in that range, so - * a smaller cloned descriptor table might suffice if the last - * currently opened descriptor falls into that range. + * range is optional. When close_range() is asked to unshare dup_fd() + * will leave any files behind according to the range and its flags. The + * cloned table only has to reach the last open descriptor that is + * carried over. */ -static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punch_hole) +static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *range) { unsigned int last = find_last_bit(fdt->open_fds, fdt->max_fds); + unsigned int i; if (last == fdt->max_fds) return NR_OPEN_DEFAULT; - if (punch_hole && punch_hole->to >= last && punch_hole->from <= last) { - last = find_last_bit(fdt->open_fds, punch_hole->from); - if (last == punch_hole->from) + if (!range) + return ALIGN(last + 1, BITS_PER_LONG); + + if (range->flags & FD_RANGE_CLOEXEC_ONLY) { + /* The close-on-exec bits decide what is dropped, walk the words. */ + i = last / BITS_PER_LONG + 1; + while (i--) { + unsigned long dropped = dup_fd_dropped_word(fdt, i, range); + + if (fdt->open_fds[i] & ~dropped) + return (i + 1) * BITS_PER_LONG; + } + return NR_OPEN_DEFAULT; + } + + if (range->flags & FD_RANGE_EXCEPT) { + /* Only the range is carried over. */ + if (last > range->to) { + last = find_last_bit(fdt->open_fds, range->to + 1); + if (last > range->to) + return NR_OPEN_DEFAULT; + } + if (last < range->from) + return NR_OPEN_DEFAULT; + } else if (last >= range->from && last <= range->to) { + /* The last open descriptor goes, the kept ones sit below the range. */ + last = find_last_bit(fdt->open_fds, range->from); + if (last == range->from) return NR_OPEN_DEFAULT; } return ALIGN(last + 1, BITS_PER_LONG); @@ -378,13 +432,14 @@ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punc /* * Allocate a new descriptor table and copy contents from the passed in * instance. Returns a pointer to cloned table on success, ERR_PTR() - * on failure. For 'punch_hole' see sane_fdtable_size(). + * on failure. For 'range' see sane_fdtable_size(). */ -struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_hole) +struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *range) { struct files_struct *newf; struct file **old_fds, **new_fds; - unsigned int open_files, i; + unsigned int open_files, fd; + unsigned long dropped = 0; struct fdtable *old_fdt, *new_fdt; newf = kmem_cache_alloc(files_cachep, GFP_KERNEL); @@ -406,7 +461,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); /* * Check whether we need to allocate a larger fd array and fd set. @@ -430,7 +485,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho */ spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); } copy_fd_bitmaps(new_fdt, old_fdt, open_files / BITS_PER_LONG); @@ -451,13 +506,18 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho * * Instead of trying to placate userspace racing with itself, we * ref the file if we see it and mark the fd slot as unused otherwise. + * Descriptors dup_fd() is asked to leave behind get the same treatment. */ - for (i = open_files; i != 0; i--) { + for (fd = 0; fd < open_files; fd++) { struct file *f = rcu_dereference_raw(*old_fds++); - if (f) { + + if (!(fd % BITS_PER_LONG)) + dropped = dup_fd_dropped_word(old_fdt, fd / BITS_PER_LONG, range); + if (f && !(dropped & BIT_MASK(fd))) { get_file(f); } else { - __clear_open_fd(open_files - i, new_fdt); + f = NULL; + __clear_open_fd(fd, new_fdt); } rcu_assign_pointer(*new_fds++, f); } @@ -471,7 +531,25 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho return newf; } -static struct fdtable *close_files(struct files_struct * files) +/* + * Unshare file descriptor table if it is being shared + */ +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) +{ + struct files_struct *fd = current->files; + + if ((unshare_flags & CLONE_FILES) && + (fd && atomic_read(&fd->count) > 1)) { + fd = dup_fd(fd, NULL); + if (IS_ERR(fd)) + return PTR_ERR(fd); + *new_fdp = fd; + } + + return 0; +} + +static struct fdtable *close_files(struct files_struct *files) { /* * It is safe to dereference the fd table without RCU or @@ -479,24 +557,21 @@ static struct fdtable *close_files(struct files_struct * files) * files structure. */ struct fdtable *fdt = rcu_dereference_raw(files->fdt); - unsigned int i, j = 0; + unsigned int j = fdt->max_fds / BITS_PER_LONG; + + /* Highest fd first, the order the deferred puts ran in. */ + while (j--) { + unsigned long set = fdt->open_fds[j]; - for (;;) { - unsigned long set; - i = j * BITS_PER_LONG; - if (i >= fdt->max_fds) - break; - set = fdt->open_fds[j++]; while (set) { - if (set & 1) { - struct file *file = fdt->fd[i]; - if (file) { - filp_close(file, files); - cond_resched(); - } + unsigned int bit = __fls(set); + struct file *file = fdt->fd[j * BITS_PER_LONG + bit]; + + set ^= 1UL << bit; + if (file) { + filp_close_sync(file, files); + cond_resched(); } - i++; - set >>= 1; } } @@ -515,16 +590,18 @@ void put_files_struct(struct files_struct *files) } } -void exit_files(struct task_struct *tsk) +/* Install @files on @tsk, consuming the reference, and put the old table. */ +void switch_files_struct(struct task_struct *tsk, struct files_struct *files) { - struct files_struct * files = tsk->files; + scoped_guard(task_lock, tsk) + swap(tsk->files, files); + put_files_struct(files); +} - if (files) { - task_lock(tsk); - tsk->files = NULL; - task_unlock(tsk); - put_files_struct(files); - } +void exit_files(struct task_struct *tsk) +{ + if (tsk->files) + switch_files_struct(tsk, NULL); } struct files_struct init_files = { @@ -732,16 +809,13 @@ struct file *file_close_fd_locked(struct files_struct *files, unsigned fd) int close_fd(unsigned fd) { - struct files_struct *files = current->files; struct file *file; - spin_lock(&files->file_lock); - file = file_close_fd_locked(files, fd); - spin_unlock(&files->file_lock); + file = file_close_fd(fd); if (!file) return -EBADF; - return filp_close(file, files); + return filp_close(file, current->files); } EXPORT_SYMBOL(close_fd); @@ -759,38 +833,90 @@ static inline unsigned last_fd(struct fdtable *fdt) } static inline void __range_cloexec(struct files_struct *cur_fds, - unsigned int fd, unsigned int max_fd) + struct fd_range *range) { struct fdtable *fdt; + unsigned int last; - /* make sure we're using the correct maximum value */ spin_lock(&cur_fds->file_lock); fdt = files_fdtable(cur_fds); - max_fd = min(last_fd(fdt), max_fd); - if (fd <= max_fd) - bitmap_set(fdt->close_on_exec, fd, max_fd - fd + 1); + /* make sure we're using the correct maximum value */ + last = last_fd(fdt); + if (!(range->flags & FD_RANGE_EXCEPT)) { + if (range->from <= last) + bitmap_set(fdt->close_on_exec, range->from, + min(range->to, last) - range->from + 1); + } else { + if (range->from > 0) + bitmap_set(fdt->close_on_exec, 0, + min(range->from - 1, last) + 1); + if (range->to < last) + bitmap_set(fdt->close_on_exec, range->to + 1, + last - range->to); + } spin_unlock(&cur_fds->file_lock); } -static inline void __range_close(struct files_struct *files, unsigned int fd, - unsigned int max_fd) +/* Highest open descriptor below @n that @range selects, or @n. */ +static inline unsigned int last_fd_to_close(struct fdtable *fdt, unsigned int n, + struct fd_range *range) +{ + unsigned int i, lo = 0; + + if (!(range->flags & FD_RANGE_EXCEPT)) + lo = range->from / BITS_PER_LONG; + for (i = n ? (n - 1) / BITS_PER_LONG + 1 : 0; i-- > lo; ) { + unsigned long set = fdt->open_fds[i]; + + if (!set) { + /* Skip the empty stretch at find_last_bit() speed. */ + unsigned int last = find_last_bit(fdt->open_fds, i * BITS_PER_LONG); + + if (last >= i * BITS_PER_LONG) + break; + i = last / BITS_PER_LONG + 1; + continue; + } + /* Hop below the kept window in one step. */ + if ((range->flags & FD_RANGE_EXCEPT) && + i * BITS_PER_LONG >= range->from && + i * BITS_PER_LONG + BITS_PER_LONG - 1 <= range->to) { + if (!range->from) + break; + i = (range->from - 1) / BITS_PER_LONG + 1; + continue; + } + set &= dup_fd_dropped_word(fdt, i, range); + if (i == (n - 1) / BITS_PER_LONG) + set &= BITMAP_LAST_WORD_MASK(n); + if (set) + return i * BITS_PER_LONG + __fls(set); + } + return n; +} + +static inline void __range_close(struct files_struct *files, + struct fd_range *range) { struct file *file; struct fdtable *fdt; - unsigned n; + unsigned int fd, n; spin_lock(&files->file_lock); fdt = files_fdtable(files); - n = last_fd(fdt); - max_fd = min(max_fd, n); + if (range->flags & FD_RANGE_EXCEPT) + /* Outside of the range means the whole table. */ + n = fdt->max_fds; + else + n = min(range->to, last_fd(fdt)) + 1; - for (fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); - fd <= max_fd; - fd = find_next_bit(fdt->open_fds, max_fd + 1, fd + 1)) { + /* Highest fd first, see close_files(). */ + while ((fd = last_fd_to_close(fdt, n, range)) < n) { + n = fd; file = file_close_fd_locked(files, fd); if (file) { spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); fdt = files_fdtable(files); @@ -814,21 +940,43 @@ static inline void __range_close(struct files_struct *files, unsigned int fd, * This closes a range of file descriptors. All file descriptors * from @fd up to and including @max_fd are closed. * Currently, errors to close a given file descriptor are ignored. + * + * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead: + * every open file descriptor outside of [@fd, @max_fd] is closed, or + * marked close-on-exec with CLOSE_RANGE_CLOEXEC. + * + * With CLOSE_RANGE_CLOEXEC_ONLY only file descriptors that have + * close-on-exec set are closed. Together with CLOSE_RANGE_EXCEPT the + * range names the close-on-exec file descriptors to keep. To keep none + * of them, name a range that cannot hold an open file descriptor, e.g. + * close_range(~0U, ~0U, ...). */ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, unsigned int, flags) { struct task_struct *me = current; struct files_struct *cur_fds = me->files, *fds = NULL; + struct fd_range range = {fd, max_fd}; + + if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_EXCEPT | CLOSE_RANGE_CLOEXEC_ONLY)) + return -EINVAL; - if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC)) + /* One marks close-on-exec, the other closes what is marked. */ + if (hweight32(flags & (CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY)) > 1) return -EINVAL; if (fd > max_fd) return -EINVAL; + if (flags & CLOSE_RANGE_EXCEPT) + range.flags |= FD_RANGE_EXCEPT; + if (flags & CLOSE_RANGE_CLOEXEC_ONLY) + range.flags |= FD_RANGE_CLOEXEC_ONLY; + if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) { - struct fd_range range = {fd, max_fd}, *punch_hole = ⦥ + struct fd_range *drop = ⦥ /* * If the caller requested all fds to be made cloexec we always @@ -836,9 +984,9 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, * use them. */ if (flags & CLOSE_RANGE_CLOEXEC) - punch_hole = NULL; + drop = NULL; - fds = dup_fd(cur_fds, punch_hole); + fds = dup_fd(cur_fds, drop); if (IS_ERR(fds)) return PTR_ERR(fds); /* @@ -848,20 +996,19 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, swap(cur_fds, fds); } - if (flags & CLOSE_RANGE_CLOEXEC) - __range_cloexec(cur_fds, fd, max_fd); - else - __range_close(cur_fds, fd, max_fd); + if (flags & CLOSE_RANGE_CLOEXEC) { + __range_cloexec(cur_fds, &range); + } else if (!fds) { + /* If we unshared, dup_fd() already left behind what we'd close. */ + __range_close(cur_fds, &range); + } if (fds) { /* * We're done closing the files we were supposed to. Time to install * the new file descriptor table and drop the old one. */ - task_lock(me); - me->files = cur_fds; - task_unlock(me); - put_files_struct(fds); + switch_files_struct(me, cur_fds); } return 0; @@ -887,34 +1034,36 @@ struct file *file_close_fd(unsigned int fd) return file; } -void do_close_on_exec(struct files_struct *files) +void close_cloexec_files(struct files_struct *files) { unsigned i; struct fdtable *fdt; /* exec unshares first */ spin_lock(&files->file_lock); - for (i = 0; ; i++) { + fdt = files_fdtable(files); + /* Highest fd first, see close_files(). */ + for (i = fdt->max_fds / BITS_PER_LONG; i--; ) { unsigned long set; - unsigned fd = i * BITS_PER_LONG; + fdt = files_fdtable(files); - if (fd >= fdt->max_fds) - break; set = fdt->close_on_exec[i]; if (!set) continue; fdt->close_on_exec[i] = 0; - for ( ; set ; fd++, set >>= 1) { + while (set) { + unsigned int bit = __fls(set); + unsigned fd = i * BITS_PER_LONG + bit; struct file *file; - if (!(set & 1)) - continue; + + set ^= 1UL << bit; file = fdt->fd[fd]; if (!file) continue; rcu_assign_pointer(fdt->fd[fd], NULL); __put_unused_fd(files, fd); spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); } @@ -1391,17 +1540,17 @@ int receive_fd(struct file *file, int __user *ufd, unsigned int o_flags) return error; FD_PREPARE(fdf, o_flags, file); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; get_file(file); if (ufd) { - error = put_user(fd_prepare_fd(fdf), ufd); + error = put_user(fdf->fd, ufd); if (error) return error; } - __receive_sock(fd_prepare_file(fdf)); + __receive_sock(fdf->file); return fd_publish(fdf); } EXPORT_SYMBOL_GPL(receive_fd); @@ -1529,3 +1678,7 @@ int iterate_fd(struct files_struct *files, unsigned n, return res; } EXPORT_SYMBOL(iterate_fd); + +#ifdef CONFIG_FDTABLE_KUNIT_TEST +#include "tests/fdtable_kunit.c" +#endif diff --git a/fs/file_attr.c b/fs/file_attr.c index bfb00d256dd5..81af4364e33a 100644 --- a/fs/file_attr.c +++ b/fs/file_attr.c @@ -265,7 +265,7 @@ static int fileattr_set_prepare(struct inode *inode, * * Return: 0 on success, or a negative error on failure. */ -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -323,7 +323,7 @@ int ioctl_getflags(struct file *file, unsigned int __user *argp) int ioctl_setflags(struct file *file, unsigned int __user *argp) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct dentry *dentry = file->f_path.dentry; struct file_kattr fa = {}; unsigned int flags; @@ -355,7 +355,7 @@ int ioctl_fsgetxattr(struct file *file, void __user *argp) int ioctl_fssetxattr(struct file *file, void __user *argp) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct dentry *dentry = file->f_path.dentry; struct file_kattr fa = {}; int err; diff --git a/fs/fs-writeback.c b/fs/fs-writeback.c index ea3eb40bf828..58a6403780aa 100644 --- a/fs/fs-writeback.c +++ b/fs/fs-writeback.c @@ -233,7 +233,7 @@ void wb_wait_for_completion(struct wb_completion *done) * Parameters for foreign inode detection, see wbc_detach_inode() to see * how they're used. * - * These paramters are inherently heuristical as the detection target + * These parameters are inherently heuristical as the detection target * itself is fuzzy. All we want to do is detaching an inode from the * current owner if it's being written to by some other cgroups too much. * @@ -248,7 +248,7 @@ void wb_wait_for_completion(struct wb_completion *done) * to 16 slots. To avoid tiny writes from swinging the decision too much, * writes smaller than 1/8 of avg size are ignored. */ -#define WB_FRN_TIME_SHIFT 13 /* 1s = 2^13, upto 8 secs w/ 16bit */ +#define WB_FRN_TIME_SHIFT 13 /* 1s = 2^13, up to 8 secs w/ 16bit */ #define WB_FRN_TIME_AVG_SHIFT 3 /* avg = avg * 7/8 + new * 1/8 */ #define WB_FRN_TIME_CUT_DIV 8 /* ignore rounds < avg / 8 */ #define WB_FRN_TIME_PERIOD (2 * (1 << WB_FRN_TIME_SHIFT)) /* 2s */ @@ -259,7 +259,7 @@ void wb_wait_for_completion(struct wb_completion *done) #define WB_FRN_HIST_THR_SLOTS (WB_FRN_HIST_SLOTS / 2) /* if foreign slots >= 8, switch */ #define WB_FRN_HIST_MAX_SLOTS (WB_FRN_HIST_THR_SLOTS / 2 + 1) - /* one round can affect upto 5 slots */ + /* one round can affect up to 5 slots */ #define WB_FRN_MAX_IN_FLIGHT 1024 /* don't queue too many concurrently */ /* @@ -1181,7 +1181,7 @@ int cgroup_writeback_by_id(u64 bdi_id, int memcg_id, struct cgroup_subsys_state *memcg_css; struct bdi_writeback *wb; struct wb_writeback_work *work; - unsigned long dirty; + long dirty; int ret; /* lookup bdi and memcg */ @@ -1210,16 +1210,13 @@ int cgroup_writeback_by_id(u64 bdi_id, int memcg_id, } /* - * The caller is attempting to write out most of - * the currently dirty pages. Let's take the current dirty page - * count and inflate it by 25% which should be large enough to - * flush out most dirty pages while avoiding getting livelocked by - * concurrent dirtiers. - * - * BTW the memcg stats are flushed periodically and this is best-effort - * estimation, so some potential error is ok. + * The caller is attempting to write out most of the target wb's + * currently dirty pages. Size the work from the wb's reclaimable pages + * and inflate the count by 25%, which should be large enough to flush + * out most dirty pages while avoiding getting livelocked by concurrent + * dirtiers. */ - dirty = memcg_page_state(mem_cgroup_from_css(memcg_css), NR_FILE_DIRTY); + dirty = wb_stat_sum(wb, WB_RECLAIMABLE); dirty = dirty * 10 / 8; /* issue the writeback work */ diff --git a/fs/fs_pin.c b/fs/fs_pin.c index 47ef3c71ce90..1a508f2167e0 100644 --- a/fs/fs_pin.c +++ b/fs/fs_pin.c @@ -1,5 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 #include <linux/fs.h> +#include <linux/rculist.h> #include <linux/sched.h> #include <linux/slab.h> #include "internal.h" @@ -22,8 +23,8 @@ void pin_remove(struct fs_pin *pin) void pin_insert(struct fs_pin *pin, struct vfsmount *m) { spin_lock(&pin_lock); - hlist_add_head(&pin->s_list, &m->mnt_sb->s_pins); - hlist_add_head(&pin->m_list, &real_mount(m)->mnt_pins); + hlist_add_head_rcu(&pin->s_list, &m->mnt_sb->s_pins); + hlist_add_head_rcu(&pin->m_list, &real_mount(m)->mnt_pins); spin_unlock(&pin_lock); } @@ -73,7 +74,7 @@ void mnt_pin_kill(struct mount *m) while (1) { struct hlist_node *p; rcu_read_lock(); - p = READ_ONCE(m->mnt_pins.first); + p = rcu_dereference(hlist_first_rcu(&m->mnt_pins)); if (!p) { rcu_read_unlock(); break; @@ -87,7 +88,7 @@ void group_pin_kill(struct hlist_head *p) while (1) { struct hlist_node *q; rcu_read_lock(); - q = READ_ONCE(p->first); + q = rcu_dereference(hlist_first_rcu(p)); if (!q) { rcu_read_unlock(); break; diff --git a/fs/fuse/acl.c b/fs/fuse/acl.c index 31fb50e16aed..738abed9a816 100644 --- a/fs/fuse/acl.c +++ b/fs/fuse/acl.c @@ -62,7 +62,7 @@ static inline bool fuse_no_acl(const struct fuse_conn *fc, return !fc->posix_acl && (i_user_ns(inode) != &init_user_ns); } -struct posix_acl *fuse_get_acl(struct mnt_idmap *idmap, +struct posix_acl *fuse_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -90,7 +90,7 @@ struct posix_acl *fuse_get_inode_acl(struct inode *inode, int type, bool rcu) return __fuse_get_acl(fc, inode, type, rcu); } -int fuse_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index f48fafccce4b..d11e2aabbfd5 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -751,7 +751,7 @@ static u32 fuse_ext_size(size_t size) /* * This adds just a single supplementary group that matches the parent's group. */ -static int get_create_supp_group(struct mnt_idmap *idmap, +static int get_create_supp_group(const struct mnt_idmap *idmap, struct inode *dir, struct fuse_in_arg *ext) { @@ -782,7 +782,7 @@ static int get_create_supp_group(struct mnt_idmap *idmap, return 0; } -static int get_create_ext(struct mnt_idmap *idmap, +static int get_create_ext(const struct mnt_idmap *idmap, struct fuse_args *args, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -820,7 +820,7 @@ static void free_ext_value(struct fuse_args *args) * If the filesystem doesn't support this, then fall back to separate * 'mknod' + 'open' requests. */ -static int fuse_create_open(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_create_open(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode, u32 opcode) { @@ -934,14 +934,14 @@ out_err: return err; } -static int fuse_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +static int fuse_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t); static int fuse_atomic_open(struct inode *dir, struct dentry *entry, struct file *file, unsigned flags, umode_t mode) { int err; - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct fuse_conn *fc = get_fuse_conn(dir); if (fuse_is_bad(dir)) @@ -980,7 +980,7 @@ mknod: /* * Code shared between mknod, mkdir, symlink and link */ -static struct dentry *create_new_entry(struct mnt_idmap *idmap, struct fuse_mount *fm, +static struct dentry *create_new_entry(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args, struct inode *dir, struct dentry *entry, umode_t mode) { @@ -1053,7 +1053,7 @@ static struct dentry *create_new_entry(struct mnt_idmap *idmap, struct fuse_moun return ERR_PTR(err); } -static int create_new_nondir(struct mnt_idmap *idmap, struct fuse_mount *fm, +static int create_new_nondir(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args, struct inode *dir, struct dentry *entry, umode_t mode) { @@ -1069,7 +1069,7 @@ static int create_new_nondir(struct mnt_idmap *idmap, struct fuse_mount *fm, return PTR_ERR(create_new_entry(idmap, fm, args, dir, entry, mode)); } -static int fuse_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { struct fuse_mknod_in inarg; @@ -1092,13 +1092,13 @@ static int fuse_mknod(struct mnt_idmap *idmap, struct inode *dir, return create_new_nondir(idmap, fm, &args, dir, entry, mode); } -static int fuse_create(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode) { return fuse_mknod(idmap, dir, entry, mode, 0); } -static int fuse_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct fuse_conn *fc = get_fuse_conn(dir); @@ -1116,7 +1116,7 @@ static int fuse_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return err; } -static struct dentry *fuse_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *fuse_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode) { struct fuse_mkdir_in inarg; @@ -1146,7 +1146,7 @@ static struct dentry *fuse_mkdir(struct mnt_idmap *idmap, struct inode *dir, return create_new_entry(idmap, fm, &args, dir, entry, S_IFDIR); } -static int fuse_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, const char *link) { struct fuse_mount *fm = get_fuse_mount(dir); @@ -1256,9 +1256,10 @@ static int fuse_rmdir(struct inode *dir, struct dentry *entry) return err; } -static int fuse_rename_common(struct mnt_idmap *idmap, struct inode *olddir, struct dentry *oldent, - struct inode *newdir, struct dentry *newent, - unsigned int flags, int opcode, size_t argsize) +static int fuse_rename_common(const struct mnt_idmap *idmap, struct inode *olddir, + struct dentry *oldent, struct inode *newdir, + struct dentry *newent, unsigned int flags, + int opcode, size_t argsize) { int err; struct fuse_rename2_in inarg; @@ -1306,7 +1307,7 @@ static int fuse_rename_common(struct mnt_idmap *idmap, struct inode *olddir, str return err; } -static int fuse_rename2(struct mnt_idmap *idmap, struct inode *olddir, +static int fuse_rename2(const struct mnt_idmap *idmap, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags) { @@ -1375,7 +1376,7 @@ out: return err; } -static void fuse_fillattr(struct mnt_idmap *idmap, struct inode *inode, +static void fuse_fillattr(const struct mnt_idmap *idmap, struct inode *inode, struct fuse_attr *attr, struct kstat *stat) { unsigned int blkbits; @@ -1429,7 +1430,7 @@ static void fuse_statx_to_attr(struct fuse_statx *sx, struct fuse_attr *attr) attr->blksize = sx->blksize; } -static int fuse_do_statx(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_do_statx(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, struct kstat *stat) { int err; @@ -1490,7 +1491,7 @@ static int fuse_do_statx(struct mnt_idmap *idmap, struct inode *inode, return 0; } -static int fuse_do_getattr(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_do_getattr(const struct mnt_idmap *idmap, struct inode *inode, struct kstat *stat, struct file *file) { int err; @@ -1536,7 +1537,7 @@ static int fuse_do_getattr(struct mnt_idmap *idmap, struct inode *inode, return err; } -static int fuse_update_get_attr(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_update_get_attr(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1762,7 +1763,7 @@ static int fuse_perm_getattr(struct inode *inode, int mask) * access request is sent. Execute permission is still checked * locally based on file mode. */ -static int fuse_permission(struct mnt_idmap *idmap, +static int fuse_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct fuse_conn *fc = get_fuse_conn(inode); @@ -2000,7 +2001,7 @@ static bool update_mtime(unsigned ivalid, bool trust_local_mtime) return true; } -static void iattr_to_fattr(struct mnt_idmap *idmap, struct fuse_conn *fc, +static void iattr_to_fattr(const struct mnt_idmap *idmap, struct fuse_conn *fc, struct iattr *iattr, struct fuse_setattr_in *arg, bool trust_local_cmtime) { @@ -2142,7 +2143,7 @@ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) * vmtruncate() doesn't allow for this case, so do the rlimit checking * and the actual truncation by hand. */ -int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_do_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct file *file) { struct inode *inode = d_inode(dentry); @@ -2323,7 +2324,7 @@ unlock: return err; } -static int fuse_setattr(struct mnt_idmap *idmap, struct dentry *entry, +static int fuse_setattr(const struct mnt_idmap *idmap, struct dentry *entry, struct iattr *attr) { struct inode *inode = d_inode(entry); @@ -2386,7 +2387,7 @@ static int fuse_setattr(struct mnt_idmap *idmap, struct dentry *entry, return ret; } -static int fuse_getattr(struct mnt_idmap *idmap, +static int fuse_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/fuse/file.c b/fs/fuse/file.c index d73afcbc1eb2..3d209e2b71ba 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1507,7 +1507,7 @@ static const struct iomap_write_ops fuse_iomap_write_ops = { static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct address_space *mapping = file->f_mapping; ssize_t written = 0; struct inode *inode = mapping->host; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 8546855386b5..87e2bd9d4bb1 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1012,7 +1012,7 @@ void __exit fuse_ctl_cleanup(void); /* * Simple request sending that does request allocation and freeing */ -ssize_t __fuse_simple_request(struct mnt_idmap *idmap, +ssize_t __fuse_simple_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args); @@ -1021,7 +1021,7 @@ static inline ssize_t fuse_simple_request(struct fuse_mount *fm, struct fuse_arg return __fuse_simple_request(&invalid_mnt_idmap, fm, args); } -static inline ssize_t fuse_simple_idmap_request(struct mnt_idmap *idmap, +static inline ssize_t fuse_simple_idmap_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args) { @@ -1198,7 +1198,7 @@ bool fuse_write_update_attr(struct inode *inode, loff_t pos, ssize_t written); int fuse_flush_times(struct inode *inode, struct fuse_file *ff); int fuse_write_inode(struct inode *inode, struct writeback_control *wbc); -int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_do_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct file *file); void fuse_unlock_inode(struct inode *inode, bool locked); @@ -1214,9 +1214,9 @@ extern const struct xattr_handler * const fuse_xattr_handlers[]; struct posix_acl; struct posix_acl *fuse_get_inode_acl(struct inode *inode, int type, bool rcu); -struct posix_acl *fuse_get_acl(struct mnt_idmap *idmap, +struct posix_acl *fuse_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int fuse_set_acl(struct mnt_idmap *, struct dentry *dentry, +int fuse_set_acl(const struct mnt_idmap *, struct dentry *dentry, struct posix_acl *acl, int type); /* readdir.c */ @@ -1247,7 +1247,7 @@ long fuse_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg); long fuse_file_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); int fuse_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int fuse_fileattr_set(struct mnt_idmap *idmap, +int fuse_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); /* iomode.c */ diff --git a/fs/fuse/ioctl.c b/fs/fuse/ioctl.c index dc3a188f5d72..ce1807704da6 100644 --- a/fs/fuse/ioctl.c +++ b/fs/fuse/ioctl.c @@ -537,7 +537,7 @@ cleanup: return err; } -int fuse_fileattr_set(struct mnt_idmap *idmap, +int fuse_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/fuse/req.c b/fs/fuse/req.c index 51cfd64ea1ba..6274a23af445 100644 --- a/fs/fuse/req.c +++ b/fs/fuse/req.c @@ -3,7 +3,8 @@ #include "dev.h" #include "fuse_i.h" -static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, struct mnt_idmap *idmap) +static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, + const struct mnt_idmap *idmap) { struct fuse_conn *fc = fm->fc; bool no_idmap = !fm->sb || (fm->sb->s_iflags & SB_I_NOIDMAP); @@ -48,7 +49,8 @@ static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, struct return 0; } -static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, struct mnt_idmap *idmap) +static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, + const struct mnt_idmap *idmap) { if (!args->force && fm->fc->conn_error) return -ECONNREFUSED; @@ -56,7 +58,7 @@ static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, struct m return fuse_fill_creds(fm, args, idmap); } -ssize_t __fuse_simple_request(struct mnt_idmap *idmap, struct fuse_mount *fm, +ssize_t __fuse_simple_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args) { struct fuse_conn *fc = fm->fc; diff --git a/fs/fuse/xattr.c b/fs/fuse/xattr.c index cab2685acc65..53e3c5e6fff0 100644 --- a/fs/fuse/xattr.c +++ b/fs/fuse/xattr.c @@ -188,7 +188,7 @@ static int fuse_xattr_get(const struct xattr_handler *handler, } static int fuse_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/gfs2/acl.c b/fs/gfs2/acl.c index 49e489fe27ef..f1c6a5e392b4 100644 --- a/fs/gfs2/acl.c +++ b/fs/gfs2/acl.c @@ -102,7 +102,7 @@ out: return error; } -int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int gfs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/gfs2/acl.h b/fs/gfs2/acl.h index 82f5b09c04e6..d19d41755936 100644 --- a/fs/gfs2/acl.h +++ b/fs/gfs2/acl.h @@ -13,7 +13,7 @@ struct posix_acl *gfs2_get_acl(struct inode *inode, int type, bool rcu); int __gfs2_set_acl(struct inode *inode, struct posix_acl *acl, int type); -int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int gfs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); #endif /* __ACL_DOT_H__ */ diff --git a/fs/gfs2/file.c b/fs/gfs2/file.c index 6fb2adeef274..1efd0679badd 100644 --- a/fs/gfs2/file.c +++ b/fs/gfs2/file.c @@ -277,7 +277,7 @@ out: return error; } -int gfs2_fileattr_set(struct mnt_idmap *idmap, +int gfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index f748f8b3c73a..2c2dc459c037 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -973,7 +973,7 @@ fail: * Returns: errno */ -static int gfs2_create(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return gfs2_create_inode(dir, dentry, NULL, S_IFREG | mode, 0, NULL, 0, 1); @@ -1329,7 +1329,7 @@ out_inodes: * Returns: errno */ -static int gfs2_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { unsigned int size; @@ -1351,7 +1351,7 @@ static int gfs2_symlink(struct mnt_idmap *idmap, struct inode *dir, * Returns: the dentry, or ERR_PTR(errno) */ -static struct dentry *gfs2_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *gfs2_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { unsigned dsize = gfs2_max_stuffed_size(GFS2_I(dir)); @@ -1369,7 +1369,7 @@ static struct dentry *gfs2_mkdir(struct mnt_idmap *idmap, struct inode *dir, * */ -static int gfs2_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { return gfs2_create_inode(dir, dentry, NULL, mode, dev, NULL, 0, 1); @@ -1887,7 +1887,7 @@ out: return error; } -static int gfs2_rename2(struct mnt_idmap *idmap, struct inode *odir, +static int gfs2_rename2(const struct mnt_idmap *idmap, struct inode *odir, struct dentry *odentry, struct inode *ndir, struct dentry *ndentry, unsigned int flags) { @@ -1974,7 +1974,7 @@ out: * Returns: errno */ -int gfs2_permission(struct mnt_idmap *idmap, struct inode *inode, +int gfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int may_not_block = mask & MAY_NOT_BLOCK; @@ -2106,7 +2106,7 @@ out: * Returns: errno */ -static int gfs2_setattr(struct mnt_idmap *idmap, +static int gfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -2168,7 +2168,7 @@ out: * Returns: errno */ -static int gfs2_getattr(struct mnt_idmap *idmap, +static int gfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/gfs2/inode.h b/fs/gfs2/inode.h index 2fcd96dd1361..99196d3114e4 100644 --- a/fs/gfs2/inode.h +++ b/fs/gfs2/inode.h @@ -97,7 +97,7 @@ int gfs2_dinode_dealloc(struct gfs2_inode *ip); struct inode *gfs2_lookupi(struct inode *dir, const struct qstr *name, int is_root); -int gfs2_permission(struct mnt_idmap *idmap, +int gfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); struct inode *gfs2_lookup_meta(struct inode *dip, const char *name); void gfs2_dinode_out(const struct gfs2_inode *ip, void *buf); @@ -109,7 +109,7 @@ extern const struct file_operations gfs2_file_fops_nolock; extern const struct file_operations gfs2_dir_fops_nolock; int gfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int gfs2_fileattr_set(struct mnt_idmap *idmap, +int gfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); void gfs2_set_inode_flags(struct inode *inode); diff --git a/fs/gfs2/log.c b/fs/gfs2/log.c index a92c84146de9..b55daf1e6381 100644 --- a/fs/gfs2/log.c +++ b/fs/gfs2/log.c @@ -107,7 +107,7 @@ __acquires(&sdp->sd_ail_lock) gfs2_assert(sdp, bd->bd_tr == tr); if (!buffer_busy(bh)) { - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { list_move(&bd->bd_ail_st_list, &tr->tr_ail2_list); continue; @@ -321,7 +321,7 @@ static int gfs2_ail1_empty_one(struct gfs2_sbd *sdp, struct gfs2_trans *tr, active_count++; continue; } - if (!buffer_uptodate(bh) && + if (buffer_write_io_error(bh) && !cmpxchg(&sdp->sd_log_error, 0, -EIO)) gfs2_io_error_bh(sdp, bh); /* diff --git a/fs/gfs2/lops.c b/fs/gfs2/lops.c index 77ef22eab368..88c84895ec6f 100644 --- a/fs/gfs2/lops.c +++ b/fs/gfs2/lops.c @@ -48,7 +48,7 @@ void gfs2_pin(struct gfs2_sbd *sdp, struct buffer_head *bh) clear_buffer_dirty(bh); if (test_set_buffer_pinned(bh)) gfs2_assert_withdraw(sdp, 0); - if (!buffer_uptodate(bh)) + if (!buffer_uptodate(bh) || buffer_write_io_error(bh)) gfs2_io_error_bh(sdp, bh); bd = bh->b_private; /* If this buffer is in the AIL and it has already been written @@ -179,6 +179,8 @@ static void gfs2_end_log_write_bh(struct gfs2_sbd *sdp, struct folio *folio, do { if (error) mark_buffer_write_io_error(bh); + else + clear_buffer_write_io_error(bh); unlock_buffer(bh); next = bh->b_this_page; size -= bh->b_size; diff --git a/fs/gfs2/xattr.c b/fs/gfs2/xattr.c index db38d972debd..c26f180419d9 100644 --- a/fs/gfs2/xattr.c +++ b/fs/gfs2/xattr.c @@ -1239,7 +1239,7 @@ int __gfs2_xattr_set(struct inode *inode, const char *name, } static int gfs2_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/hfs/attr.c b/fs/hfs/attr.c index f8395cdd1adf..6d737a085461 100644 --- a/fs/hfs/attr.c +++ b/fs/hfs/attr.c @@ -121,7 +121,7 @@ static int hfs_xattr_get(const struct xattr_handler *handler, } static int hfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/hfs/dir.c b/fs/hfs/dir.c index e1f1fb351464..f6b97da19788 100644 --- a/fs/hfs/dir.c +++ b/fs/hfs/dir.c @@ -183,7 +183,7 @@ static int hfs_dir_release(struct inode *inode, struct file *file) * a directory and return a corresponding inode, given the inode for * the directory and the name (and its length) of the new file. */ -static int hfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -213,7 +213,7 @@ static int hfs_create(struct mnt_idmap *idmap, struct inode *dir, * in a directory, given the inode for the parent directory and the * name (and its length) of the new directory. */ -static struct dentry *hfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -281,7 +281,7 @@ static int hfs_remove(struct inode *dir, struct dentry *dentry) * new file/directory. * XXX: how do you handle must_be dir? */ -static int hfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int hfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/hfs/hfs_fs.h b/fs/hfs/hfs_fs.h index e250f87a5e33..fdfa5d303d5e 100644 --- a/fs/hfs/hfs_fs.h +++ b/fs/hfs/hfs_fs.h @@ -212,7 +212,7 @@ extern struct inode *hfs_new_inode(struct inode *dir, const struct qstr *name, extern void hfs_inode_write_fork(struct inode *inode, struct hfs_extent *ext, __be32 *log_size, __be32 *phys_size); extern int hfs_write_inode(struct inode *inode, struct writeback_control *wbc); -extern int hfs_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int hfs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void hfs_inode_read_fork(struct inode *inode, struct hfs_extent *ext, __be32 __log_size, __be32 phys_size, diff --git a/fs/hfs/inode.c b/fs/hfs/inode.c index 2aef3c36a150..c81314f668ac 100644 --- a/fs/hfs/inode.c +++ b/fs/hfs/inode.c @@ -643,7 +643,7 @@ static int hfs_file_release(struct inode *inode, struct file *file) return 0; } -int hfs_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int hfs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hfsplus/dir.c b/fs/hfsplus/dir.c index 51fcba2e6d40..b3a1193a491f 100644 --- a/fs/hfsplus/dir.c +++ b/fs/hfsplus/dir.c @@ -460,7 +460,7 @@ out: return res; } -static int hfsplus_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct hfsplus_sb_info *sbi = HFSPLUS_SB(dir->i_sb); @@ -511,7 +511,7 @@ out: return res; } -static int hfsplus_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct hfsplus_sb_info *sbi = HFSPLUS_SB(dir->i_sb); @@ -561,19 +561,19 @@ out: return res; } -static int hfsplus_create(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); } -static struct dentry *hfsplus_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hfsplus_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0)); } -static int hfsplus_rename(struct mnt_idmap *idmap, +static int hfsplus_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/hfsplus/hfsplus_fs.h b/fs/hfsplus/hfsplus_fs.h index 1e5b58e6a13f..d55cb16e3897 100644 --- a/fs/hfsplus/hfsplus_fs.h +++ b/fs/hfsplus/hfsplus_fs.h @@ -459,13 +459,13 @@ void hfsplus_inode_write_fork(struct inode *inode, struct hfsplus_fork_raw *fork); int hfsplus_cat_read_inode(struct inode *inode, struct hfs_find_data *fd); int hfsplus_cat_write_inode(struct inode *inode); -int hfsplus_getattr(struct mnt_idmap *idmap, const struct path *path, +int hfsplus_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); int hfsplus_file_fsync(struct file *file, loff_t start, loff_t end, int datasync); int hfsplus_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int hfsplus_fileattr_set(struct mnt_idmap *idmap, +int hfsplus_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); /* ioctl.c */ diff --git a/fs/hfsplus/inode.c b/fs/hfsplus/inode.c index 2ce6de574fa6..aed0866499b2 100644 --- a/fs/hfsplus/inode.c +++ b/fs/hfsplus/inode.c @@ -305,7 +305,7 @@ static int hfsplus_file_release(struct inode *inode, struct file *file) return 0; } -static int hfsplus_setattr(struct mnt_idmap *idmap, +static int hfsplus_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -335,7 +335,7 @@ static int hfsplus_setattr(struct mnt_idmap *idmap, return 0; } -int hfsplus_getattr(struct mnt_idmap *idmap, const struct path *path, +int hfsplus_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -797,7 +797,7 @@ int hfsplus_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int hfsplus_fileattr_set(struct mnt_idmap *idmap, +int hfsplus_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/hfsplus/xattr.c b/fs/hfsplus/xattr.c index 21a1c196c71f..71364e093fa7 100644 --- a/fs/hfsplus/xattr.c +++ b/fs/hfsplus/xattr.c @@ -1008,7 +1008,7 @@ static int hfsplus_osx_getxattr(const struct xattr_handler *handler, } static int hfsplus_osx_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_security.c b/fs/hfsplus/xattr_security.c index 90f68ec119cd..1969919c12cb 100644 --- a/fs/hfsplus/xattr_security.c +++ b/fs/hfsplus/xattr_security.c @@ -23,7 +23,7 @@ static int hfsplus_security_getxattr(const struct xattr_handler *handler, } static int hfsplus_security_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_trusted.c b/fs/hfsplus/xattr_trusted.c index fdbaebc1c49a..c140a95ab3f0 100644 --- a/fs/hfsplus/xattr_trusted.c +++ b/fs/hfsplus/xattr_trusted.c @@ -22,7 +22,7 @@ static int hfsplus_trusted_getxattr(const struct xattr_handler *handler, } static int hfsplus_trusted_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_user.c b/fs/hfsplus/xattr_user.c index 6464b6c3d58d..7e5da15f9937 100644 --- a/fs/hfsplus/xattr_user.c +++ b/fs/hfsplus/xattr_user.c @@ -22,7 +22,7 @@ static int hfsplus_user_getxattr(const struct xattr_handler *handler, } static int hfsplus_user_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hostfs/hostfs_kern.c b/fs/hostfs/hostfs_kern.c index 7add056d47d8..613146e76dec 100644 --- a/fs/hostfs/hostfs_kern.c +++ b/fs/hostfs/hostfs_kern.c @@ -592,7 +592,7 @@ static struct inode *hostfs_iget(struct super_block *sb, char *name) return inode; } -static int hostfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hostfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -673,7 +673,7 @@ static int hostfs_unlink(struct inode *ino, struct dentry *dentry) return err; } -static int hostfs_symlink(struct mnt_idmap *idmap, struct inode *ino, +static int hostfs_symlink(const struct mnt_idmap *idmap, struct inode *ino, struct dentry *dentry, const char *to) { char *file; @@ -686,7 +686,7 @@ static int hostfs_symlink(struct mnt_idmap *idmap, struct inode *ino, return err; } -static struct dentry *hostfs_mkdir(struct mnt_idmap *idmap, struct inode *ino, +static struct dentry *hostfs_mkdir(const struct mnt_idmap *idmap, struct inode *ino, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -719,7 +719,7 @@ static int hostfs_rmdir(struct inode *ino, struct dentry *dentry) return err; } -static int hostfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hostfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode *inode; @@ -745,7 +745,7 @@ static int hostfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return 0; } -static int hostfs_rename2(struct mnt_idmap *idmap, +static int hostfs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -774,7 +774,7 @@ static int hostfs_rename2(struct mnt_idmap *idmap, return err; } -static int hostfs_permission(struct mnt_idmap *idmap, +static int hostfs_permission(const struct mnt_idmap *idmap, struct inode *ino, int desired) { char *name; @@ -801,7 +801,7 @@ static int hostfs_permission(struct mnt_idmap *idmap, return err; } -static int hostfs_setattr(struct mnt_idmap *idmap, +static int hostfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hpfs/hpfs_fn.h b/fs/hpfs/hpfs_fn.h index 237c1c23e855..a398dd8bdf30 100644 --- a/fs/hpfs/hpfs_fn.h +++ b/fs/hpfs/hpfs_fn.h @@ -280,7 +280,7 @@ void hpfs_init_inode(struct inode *); void hpfs_read_inode(struct inode *); void hpfs_write_inode(struct inode *); void hpfs_write_inode_nolock(struct inode *); -int hpfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int hpfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); void hpfs_write_if_changed(struct inode *); void hpfs_evict_inode(struct inode *); diff --git a/fs/hpfs/inode.c b/fs/hpfs/inode.c index 1b4fcf760aad..396773d0b669 100644 --- a/fs/hpfs/inode.c +++ b/fs/hpfs/inode.c @@ -257,7 +257,7 @@ void hpfs_write_inode_nolock(struct inode *i) brelse(bh); } -int hpfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int hpfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hpfs/namei.c b/fs/hpfs/namei.c index 9446f4038874..ac9b5e3e83fa 100644 --- a/fs/hpfs/namei.c +++ b/fs/hpfs/namei.c @@ -19,7 +19,7 @@ static void hpfs_update_directory_times(struct inode *dir) hpfs_write_inode_nolock(dir); } -static struct dentry *hpfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hpfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { const unsigned char *name = dentry->d_name.name; @@ -128,7 +128,7 @@ bail: return ERR_PTR(err); } -static int hpfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { const unsigned char *name = dentry->d_name.name; @@ -215,7 +215,7 @@ bail: return err; } -static int hpfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { const unsigned char *name = dentry->d_name.name; @@ -289,7 +289,7 @@ bail: return err; } -static int hpfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symlink) { const unsigned char *name = dentry->d_name.name; @@ -500,7 +500,7 @@ const struct address_space_operations hpfs_symlink_aops = { .read_folio = hpfs_symlink_read_folio }; -static int hpfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int hpfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index 7611a8470ea2..6656807ce437 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -828,7 +828,7 @@ out_nolock: return error; } -static int hugetlbfs_setattr(struct mnt_idmap *idmap, +static int hugetlbfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -892,7 +892,7 @@ static struct inode *hugetlbfs_get_root(struct super_block *sb, static struct lock_class_key hugetlbfs_i_mmap_rwsem_key; static struct inode *hugetlbfs_get_inode(struct super_block *sb, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, umode_t mode, dev_t dev) { @@ -954,7 +954,7 @@ static struct inode *hugetlbfs_get_inode(struct super_block *sb, /* * File creation. Allocate an inode, and we're done.. */ -static int hugetlbfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hugetlbfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode *inode; @@ -967,7 +967,7 @@ static int hugetlbfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return 0; } -static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hugetlbfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int retval = hugetlbfs_mknod(idmap, dir, dentry, @@ -977,14 +977,14 @@ static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir return ERR_PTR(retval); } -static int hugetlbfs_create(struct mnt_idmap *idmap, +static int hugetlbfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return hugetlbfs_mknod(idmap, dir, dentry, mode | S_IFREG, 0); } -static int hugetlbfs_tmpfile(struct mnt_idmap *idmap, +static int hugetlbfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { @@ -998,7 +998,7 @@ static int hugetlbfs_tmpfile(struct mnt_idmap *idmap, return finish_open_simple(file, 0); } -static int hugetlbfs_symlink(struct mnt_idmap *idmap, +static int hugetlbfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { diff --git a/fs/inode.c b/fs/inode.c index a9d37be390a1..1cb6293b237a 100644 --- a/fs/inode.c +++ b/fs/inode.c @@ -1770,7 +1770,7 @@ EXPORT_SYMBOL(ilookup); * function must never block --- find_inode() can block in * __wait_on_freeing_inode() --- or when the caller can not increment * the reference count because the resulting iput() might cause an - * inode eviction. The tradeoff is that the @match funtion must be + * inode eviction. The tradeoff is that the @match function must be * very carefully implemented. */ struct inode *find_inode_nowait(struct super_block *sb, @@ -2336,7 +2336,7 @@ EXPORT_SYMBOL(touch_atime); * response to write or truncate. Return 0 if nothing has to be changed. * Negative value on error (change should be denied). */ -int dentry_needs_remove_privs(struct mnt_idmap *idmap, +int dentry_needs_remove_privs(const struct mnt_idmap *idmap, struct dentry *dentry) { struct inode *inode = d_inode(dentry); @@ -2355,7 +2355,7 @@ int dentry_needs_remove_privs(struct mnt_idmap *idmap, return mask; } -static int __remove_privs(struct mnt_idmap *idmap, +static int __remove_privs(const struct mnt_idmap *idmap, struct dentry *dentry, int kill) { struct iattr newattrs; @@ -2715,7 +2715,7 @@ EXPORT_SYMBOL(init_special_inode); * and initializing i_uid and i_gid. On non-idmapped mounts or if permission * checking is to be performed on the raw inode simply pass @nop_mnt_idmap. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode) { inode_fsuid_set(inode, idmap); @@ -2745,7 +2745,7 @@ EXPORT_SYMBOL(inode_init_owner); * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode) { vfsuid_t vfsuid; @@ -3032,7 +3032,7 @@ EXPORT_SYMBOL(inode_set_ctime_deleg); * * Return: true if the caller is sufficiently privileged, false if not. */ -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid) { if (vfsgid_in_group_p(vfsgid)) @@ -3057,7 +3057,7 @@ EXPORT_SYMBOL(in_group_or_capable); * * Return: the new mode to use for the file */ -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode) { if ((mode & (S_ISGID | S_IXGRP)) != (S_ISGID | S_IXGRP)) diff --git a/fs/internal.h b/fs/internal.h index c658c8a5ebd5..e833c7e6e14f 100644 --- a/fs/internal.h +++ b/fs/internal.h @@ -55,7 +55,7 @@ extern int filename_lookup(int dfd, struct filename *name, unsigned flags, struct path *path, const struct path *root); int filename_rmdir(int dfd, struct filename *name); int filename_unlinkat(int dfd, struct filename *name); -int may_linkat(struct mnt_idmap *idmap, const struct path *link); +int may_linkat(const struct mnt_idmap *idmap, const struct path *link); int filename_renameat2(int olddfd, struct filename *oldname, int newdfd, struct filename *newname, unsigned int flags); int filename_mkdirat(int dfd, struct filename *name, umode_t mode); @@ -63,7 +63,7 @@ int filename_mknodat(int dfd, struct filename *name, umode_t mode, unsigned int int filename_symlinkat(struct filename *from, int newdfd, struct filename *to); int filename_linkat(int olddfd, struct filename *old, int newdfd, struct filename *new, int flags); -int vfs_tmpfile(struct mnt_idmap *idmap, +int vfs_tmpfile(const struct mnt_idmap *idmap, const struct path *parentpath, struct file *file, umode_t mode); struct dentry *d_hash_and_lookup(struct dentry *, struct qstr *); @@ -198,6 +198,7 @@ extern struct file *do_file_open_root(const struct path *, extern struct open_how build_open_how(int flags, umode_t mode); extern int build_open_flags(const struct open_how *how, struct open_flags *op); struct file *file_close_fd_locked(struct files_struct *files, unsigned fd); +int filp_close_sync(struct file *filp, fl_owner_t id); int do_ftruncate(struct file *file, loff_t length, unsigned int flags); int chmod_common(const struct path *path, umode_t mode); @@ -205,13 +206,14 @@ int do_fchownat(int dfd, const char __user *filename, uid_t user, gid_t group, int flag); int chown_common(const struct path *path, uid_t user, gid_t group); extern int vfs_open(const struct path *, struct file *); +int vfs_open_consume(struct path *, struct file *); /* * inode.c */ extern long prune_icache_sb(struct super_block *sb, struct shrink_control *sc); -int dentry_needs_remove_privs(struct mnt_idmap *, struct dentry *dentry); -bool in_group_or_capable(struct mnt_idmap *idmap, +int dentry_needs_remove_privs(const struct mnt_idmap *, struct dentry *dentry); +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -299,21 +301,21 @@ int filename_setxattr(int dfd, struct filename *filename, int setxattr_copy(const char __user *name, struct kernel_xattr_ctx *ctx); int import_xattr_name(struct xattr_name *kname, const char __user *name); -int may_write_xattr(struct mnt_idmap *idmap, struct inode *inode); +int may_write_xattr(const struct mnt_idmap *idmap, struct inode *inode); #ifdef CONFIG_FS_POSIX_ACL -int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size); -ssize_t do_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size); #else -static inline int do_set_acl(struct mnt_idmap *idmap, +static inline int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size) { return -EOPNOTSUPP; } -static inline ssize_t do_get_acl(struct mnt_idmap *idmap, +static inline ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size) { @@ -327,8 +329,8 @@ ssize_t __kernel_write_iter(struct file *file, struct iov_iter *from, loff_t *po * fs/attr.c */ struct mnt_idmap *alloc_mnt_idmap(struct user_namespace *mnt_userns); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); struct stashed_operations { struct dentry *(*stash_dentry)(struct dentry **stashed, struct dentry *dentry); @@ -354,12 +356,12 @@ static inline bool path_mounted(const struct path *path) } void file_f_owner_release(struct file *file); bool file_seek_cur_needs_f_lock(struct file *file); -int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map); +int statmount_mnt_idmap(const struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map); struct dentry *find_next_child(struct dentry *parent, struct dentry *prev); -int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, +int anon_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); -int anon_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int anon_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); void pidfs_get_root(struct path *path); void nsfs_get_root(struct path *path); diff --git a/fs/iomap/bio.c b/fs/iomap/bio.c index 48100c614431..d46c2f8ea18c 100644 --- a/fs/iomap/bio.c +++ b/fs/iomap/bio.c @@ -169,6 +169,7 @@ int iomap_bio_read_folio_range_sync(const struct iomap_iter *iter, { const struct iomap *srcmap = iomap_iter_srcmap(iter); sector_t sector = iomap_sector(srcmap, pos); + struct bvec_iter saved_iter; struct bio_vec bvec; struct bio bio; int error; @@ -178,10 +179,11 @@ int iomap_bio_read_folio_range_sync(const struct iomap_iter *iter, bio_add_folio_nofail(&bio, folio, len, offset_in_folio(folio, pos)); if (srcmap->flags & IOMAP_F_INTEGRITY) fs_bio_integrity_alloc(&bio); + saved_iter = bio.bi_iter; error = submit_bio_wait(&bio); if (bio_integrity(&bio)) { if (!error) - error = fs_bio_integrity_verify(&bio, sector, len); + error = fs_bio_integrity_verify(&bio, &saved_iter); fs_bio_integrity_free(&bio); } bio_uninit(&bio); diff --git a/fs/iomap/buffered-io.c b/fs/iomap/buffered-io.c index 0a5ebfda90f1..678f4c5329e7 100644 --- a/fs/iomap/buffered-io.c +++ b/fs/iomap/buffered-io.c @@ -143,8 +143,8 @@ static unsigned ifs_next_clean_block(struct folio *folio, blks + start_blk) - blks; } -static unsigned ifs_find_dirty_range(struct folio *folio, - struct iomap_folio_state *ifs, u64 *range_start, u64 range_end) +static unsigned ifs_find_dirty_range(struct folio *folio, u64 *range_start, + u64 range_end) { struct inode *inode = folio->mapping->host; unsigned start_blk = @@ -176,7 +176,7 @@ static unsigned iomap_find_dirty_range(struct folio *folio, u64 *range_start, return 0; if (ifs) - return ifs_find_dirty_range(folio, ifs, range_start, range_end); + return ifs_find_dirty_range(folio, range_start, range_end); return range_end - *range_start; } @@ -1708,7 +1708,7 @@ static int iomap_zero_iter(struct iomap_iter *iter, bool *did_zero, * @iomap_flags: Flags to set on the associated iomap to track the batch. * * Returns the folio count directly. Also returns the associated control flag if - * the the batch lookup is performed and the expected offset of a subsequent + * the batch lookup is performed and the expected offset of a subsequent * lookup via out params. The caller is responsible to set the flag on the * associated iomap. */ diff --git a/fs/iomap/direct-io.c b/fs/iomap/direct-io.c index 8b4039d16ce8..a431ceda9ebd 100644 --- a/fs/iomap/direct-io.c +++ b/fs/iomap/direct-io.c @@ -76,10 +76,19 @@ static void iomap_dio_submit_bio(const struct iomap_iter *iter, if (dio->dops && dio->dops->submit_io) { dio->dops->submit_io(iter, bio, pos); - } else { - WARN_ON_ONCE(iter->iomap.flags & IOMAP_F_ANON_WRITE); - blk_crypto_submit_bio(bio); + return; + } + + WARN_ON_ONCE(iter->iomap.flags & IOMAP_F_ANON_WRITE); + + if (iter->iomap.flags & IOMAP_F_INTEGRITY) { + if (dio->flags & IOMAP_DIO_WRITE) + fs_bio_integrity_generate(bio); + else + fs_bio_integrity_alloc(bio); } + + blk_crypto_submit_bio(bio); } static inline enum fserror_type iomap_dio_err_type(const struct iomap_dio *dio) @@ -246,8 +255,7 @@ static void __iomap_dio_bio_end_io(struct bio *bio, bool inline_completion) fs_bio_integrity_free(bio); if (dio->flags & IOMAP_DIO_BOUNCE) { - bio_iov_iter_unbounce(bio, !!dio->error, - dio->flags & IOMAP_DIO_USER_BACKED); + bio_free_folios(bio); bio_put(bio); } else if (dio->flags & IOMAP_DIO_USER_BACKED) { bio_check_pages_dirty(bio); @@ -336,6 +344,7 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, struct iomap_dio *dio, loff_t pos, unsigned int alignment, blk_opf_t op) { + unsigned int maxsize = iomap_max_bio_size(&iter->iomap); unsigned int nr_vecs; struct bio *bio; ssize_t ret; @@ -353,14 +362,12 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, bio->bi_private = dio; bio->bi_end_io = iomap_dio_bio_end_io; - if (dio->flags & IOMAP_DIO_BOUNCE) - ret = bio_iov_iter_bounce(bio, dio->submit.iter, - iomap_max_bio_size(&iter->iomap), alignment); + ret = bio_iov_iter_bounce_write(bio, dio->submit.iter, maxsize, + alignment); else - ret = bio_iov_iter_get_pages(bio, dio->submit.iter, - bdev_dma_alignment(bio->bi_bdev), - alignment - 1); + ret = bio_iov_iter_get_pages(bio, dio->submit.iter, maxsize, + bdev_dma_alignment(bio->bi_bdev), alignment - 1); if (unlikely(ret)) goto out_put_bio; ret = bio->bi_iter.bi_size; @@ -374,13 +381,6 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, goto out_bio_release_pages; } - if (iter->iomap.flags & IOMAP_F_INTEGRITY) { - if (dio->flags & IOMAP_DIO_WRITE) - fs_bio_integrity_generate(bio); - else - fs_bio_integrity_alloc(bio); - } - if (dio->flags & IOMAP_DIO_WRITE) task_io_account_write(ret); else if ((dio->flags & IOMAP_DIO_USER_BACKED) && @@ -397,7 +397,7 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, out_bio_release_pages: if (dio->flags & IOMAP_DIO_BOUNCE) - bio_iov_iter_unbounce(bio, true, false); + bio_free_folios(bio); else bio_release_pages(bio, false); out_put_bio: @@ -505,7 +505,7 @@ static int iomap_dio_bio_iter(struct iomap_iter *iter, struct iomap_dio *dio) * We can only do inline completion for pure overwrites that * don't require additional I/O at completion time. * - * This rules out writes that need zeroing or metdata updates to + * This rules out writes that need zeroing or metadata updates to * convert unwritten or shared extents. * * Writes that extend i_size are also not supported, but this is @@ -1034,9 +1034,9 @@ ssize_t __iomap_dio_read_simple(struct kiocb *iocb, struct iov_iter *iter, bio->bi_iter.bi_sector = iomap_sector(&iomi->iomap, iomi->pos); bio->bi_ioprio = iocb->ki_ioprio; - ret = bio_iov_iter_get_pages(bio, iter, - bdev_dma_alignment(bio->bi_bdev), - alignment - 1); + ret = bio_iov_iter_get_pages(bio, iter, BIO_MAX_SIZE, + bdev_dma_alignment(bio->bi_bdev), + alignment - 1); if (unlikely(ret)) goto out_bio_put; diff --git a/fs/iomap/ioend.c b/fs/iomap/ioend.c index 7bbbb417f915..bbebecc31670 100644 --- a/fs/iomap/ioend.c +++ b/fs/iomap/ioend.c @@ -25,6 +25,7 @@ struct iomap_ioend *iomap_init_ioend(struct inode *inode, ioend->io_parent = NULL; INIT_LIST_HEAD(&ioend->io_list); ioend->io_flags = ioend_flags; + ioend->io_bvec_offset = bio->bi_iter.bi_offset; ioend->io_inode = inode; ioend->io_offset = file_offset; ioend->io_size = bio->bi_iter.bi_size; @@ -149,7 +150,7 @@ int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error) return error; } - if (wpc->iomap.flags & IOMAP_F_INTEGRITY) + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) fs_bio_integrity_generate(&ioend->io_bio); submit_bio(&ioend->io_bio); return 0; @@ -215,7 +216,7 @@ ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, { struct iomap_ioend *ioend = wpc->wb_ctx; size_t poff = offset_in_folio(folio, pos); - unsigned int ioend_flags = 0; + unsigned int ioend_flags = iomap_ioend_flags(&wpc->iomap); unsigned int map_len = min_t(u64, dirty_len, wpc->iomap.offset + wpc->iomap.length - pos); int error; @@ -225,20 +226,16 @@ ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, WARN_ON_ONCE(!folio->private && map_len < dirty_len); switch (wpc->iomap.type) { + case IOMAP_HOLE: + return map_len; case IOMAP_UNWRITTEN: - ioend_flags |= IOMAP_IOEND_UNWRITTEN; - break; case IOMAP_MAPPED: break; - case IOMAP_HOLE: - return map_len; default: WARN_ON_ONCE(1); return -EIO; } - if (wpc->iomap.flags & IOMAP_F_SHARED) - ioend_flags |= IOMAP_IOEND_SHARED; if (pos == wpc->iomap.offset && (wpc->iomap.flags & IOMAP_F_BOUNDARY)) ioend_flags |= IOMAP_IOEND_BOUNDARY; @@ -312,6 +309,16 @@ new_ioend: } EXPORT_SYMBOL_GPL(iomap_add_to_ioend); +#ifdef CONFIG_BLK_DEV_INTEGRITY +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend) +{ + struct bvec_iter data_iter = BVEC_ITER_IOEND(ioend); + + return fs_bio_integrity_verify(&ioend->io_bio, &data_iter); +} +EXPORT_SYMBOL_GPL(iomap_ioend_integrity_verify); +#endif /* CONFIG_BLK_DEV_INTEGRITY */ + static u32 iomap_finish_ioend(struct iomap_ioend *ioend, int error) { if (ioend->io_parent) { @@ -327,13 +334,6 @@ static u32 iomap_finish_ioend(struct iomap_ioend *ioend, int error) if (!atomic_dec_and_test(&ioend->io_remaining)) return 0; - if (!ioend->io_error && - bio_integrity(&ioend->io_bio) && - bio_op(&ioend->io_bio) == REQ_OP_READ) { - ioend->io_error = fs_bio_integrity_verify(&ioend->io_bio, - ioend->io_sector, ioend->io_size); - } - if (ioend->io_flags & IOMAP_IOEND_DIRECT) return iomap_finish_ioend_direct(ioend); if (bio_op(&ioend->io_bio) == REQ_OP_READ) @@ -512,6 +512,96 @@ struct iomap_ioend *iomap_split_ioend(struct iomap_ioend *ioend, } EXPORT_SYMBOL_GPL(iomap_split_ioend); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)) +{ + struct inode *inode = orig_ioend->io_inode; + struct bio *orig_bio = &orig_ioend->io_bio; + loff_t file_offset = orig_ioend->io_offset; + sector_t sector = orig_ioend->io_sector; + size_t total_len = round_up(orig_ioend->io_size, minsize); + + WARN_ON_ONCE(!(orig_ioend->io_flags & IOMAP_IOEND_DIRECT)); + + /* We can't poll a bio that is not passed on to hardware */ + orig_bio->bi_opf &= ~REQ_POLLED; + + do { + struct iomap_ioend *ioend; + struct bio *bio; + int error; + + bio = bio_alloc_bioset(orig_bio->bi_bdev, + min(total_len / minsize, BIO_MAX_VECS), + orig_bio->bi_opf, GFP_KERNEL, + &iomap_ioend_split_bioset); + error = bio_alloc_bounce_folios(bio, total_len, minsize); + if (error) { + bio_put(bio); + orig_bio->bi_status = errno_to_blk_status(error); + break; + } + bio->bi_ioprio = orig_bio->bi_ioprio; + bio->bi_write_hint = orig_bio->bi_write_hint; + bio->bi_write_stream = orig_bio->bi_write_stream; + bio->bi_iter.bi_sector = sector; + + ioend = iomap_init_ioend(inode, bio, file_offset, + orig_ioend->io_flags); + + total_len -= bio->bi_iter.bi_size; + file_offset += bio->bi_iter.bi_size; + sector += (bio->bi_iter.bi_size >> SECTOR_SHIFT); + + bio->bi_private = orig_bio; + bio_inc_remaining(orig_bio); + submit_ioend(ioend); + } while (total_len > 0); + + bio_endio(&orig_ioend->io_bio); +} +EXPORT_SYMBOL_GPL(iomap_bounce_read); + +static void iomap_ioend_unbounce(struct iomap_ioend *orig_ioend, + struct iomap_ioend *ioend) +{ + struct bio *orig_bio = &orig_ioend->io_bio; + struct iov_iter to; + struct bio_vec *bv; + int i; + + iov_iter_bvec(&to, ITER_DEST, orig_bio->bi_io_vec, orig_bio->bi_vcnt, + orig_ioend->io_size); + to.iov_offset = orig_ioend->io_bvec_offset; + + if (ioend->io_offset != orig_ioend->io_offset) { + WARN_ON_ONCE(ioend->io_offset < orig_ioend->io_offset); + iov_iter_advance(&to, ioend->io_offset - orig_ioend->io_offset); + } + + /* copying to pinned pages should always work */ + bio_for_each_bvec_all(bv, &ioend->io_bio, i) + WARN_ON_ONCE(copy_to_iter(bvec_virt(bv), bv->bv_len, &to) != + bv->bv_len); +} + +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error) +{ + if (error) + orig_bio->bi_status = errno_to_blk_status(error); + else + iomap_ioend_unbounce(iomap_ioend_from_bio(orig_bio), ioend); + + bio_free_folios(&ioend->io_bio); + if (bio_integrity(&ioend->io_bio)) + fs_bio_integrity_free(&ioend->io_bio); + bio_put(&ioend->io_bio); + + bio_endio(orig_bio); +} +EXPORT_SYMBOL_GPL(iomap_bounce_read_end_io); + static int __init iomap_ioend_init(void) { const unsigned int nr_mempool_entries = 4 * (PAGE_SIZE / SECTOR_SIZE); diff --git a/fs/jbd2/commit.c b/fs/jbd2/commit.c index 3029cb6f6d64..ebf6ba58ff4d 100644 --- a/fs/jbd2/commit.c +++ b/fs/jbd2/commit.c @@ -32,14 +32,14 @@ static void journal_end_buffer_io_sync(struct bio *bio) { struct buffer_head *bh; - bool uptodate = bio_endio_bh(bio, &bh); + bool success = bio_endio_bh(bio, &bh); struct buffer_head *orig_bh = bh->b_private; BUFFER_TRACE(bh, ""); - if (uptodate) - set_buffer_uptodate(bh); + if (success) + clear_buffer_write_io_error(bh); else - clear_buffer_uptodate(bh); + mark_buffer_write_io_error(bh); if (orig_bh) { clear_and_wake_up_bit(BH_Shadow, &orig_bh->b_state); } @@ -169,7 +169,7 @@ static int journal_wait_on_commit_record(journal_t *journal, clear_buffer_dirty(bh); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) ret = -EIO; put_bh(bh); /* One for getblk() */ @@ -330,9 +330,9 @@ static __u32 jbd2_checksum_data(__u32 crc32_sum, struct buffer_head *bh) char *addr; __u32 checksum; - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); checksum = crc32_be(crc32_sum, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); return checksum; } @@ -357,10 +357,10 @@ static void jbd2_block_tag_csum_set(journal_t *j, journal_block_tag_t *tag, return; seq = cpu_to_be32(sequence); - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); csum32 = jbd2_chksum(j->j_csum_seed, (__u8 *)&seq, sizeof(seq)); csum32 = jbd2_chksum(csum32, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); if (jbd2_has_feature_csum3(j)) tag3->t_checksum = cpu_to_be32(csum32); @@ -834,7 +834,7 @@ start_journal_io: wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; jbd2_unfile_log_bh(bh); stats.run.rs_blocks_logged++; @@ -877,7 +877,7 @@ start_journal_io: wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; BUFFER_TRACE(bh, "ph5: control buffer writeout done: unfile"); diff --git a/fs/jbd2/journal.c b/fs/jbd2/journal.c index 00f5a98f3d4f..cda1ff8851dc 100644 --- a/fs/jbd2/journal.c +++ b/fs/jbd2/journal.c @@ -328,8 +328,6 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, { int do_escape = 0; struct buffer_head *new_bh; - struct folio *new_folio; - unsigned int new_offset; struct buffer_head *bh_in = jh2bh(jh_in); journal_t *journal = transaction->t_journal; @@ -349,24 +347,31 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* keep subsequent assertions sane */ atomic_set(&new_bh->b_count, 1); + /* + * b_frozen_data is slab memory, not page cache, so when we use it the + * shadow buffer gets no folio at all: b_folio stays NULL from the + * allocation and b_data points straight at the copy. Pointing it at + * the slab folio instead would hand its overloaded ->mapping to + * anything that goes looking for an address_space. + */ + spin_lock(&jh_in->b_state_lock); /* * If a new transaction has already done a buffer copy-out, then * we use that version of the data for the commit. */ if (jh_in->b_frozen_data) { - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); do_escape = jbd2_data_needs_escaping(jh_in->b_frozen_data); if (do_escape) jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } else { + struct folio *folio = bh_in->b_folio; + unsigned int offset = offset_in_folio(folio, bh_in->b_data); char *tmp; char *mapped_data; - new_folio = bh_in->b_folio; - new_offset = offset_in_folio(new_folio, bh_in->b_data); - mapped_data = kmap_local_folio(new_folio, new_offset); + mapped_data = kmap_local_folio(folio, offset); /* * Fire data frozen trigger if data already wasn't frozen. Do * this before checking for escaping, as the trigger may modify @@ -380,8 +385,10 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* * Do we need to do a data copy? */ - if (!do_escape) + if (!do_escape) { + folio_set_bh(new_bh, folio, offset); goto escape_done; + } spin_unlock(&jh_in->b_state_lock); tmp = kmalloc(bh_in->b_size, GFP_NOFS | __GFP_NOFAIL); @@ -392,7 +399,7 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, } jh_in->b_frozen_data = tmp; - memcpy_from_folio(tmp, new_folio, new_offset, bh_in->b_size); + memcpy_from_folio(tmp, folio, offset, bh_in->b_size); /* * This isn't strictly necessary, as we're using frozen * data for the escaping, but it keeps consistency with @@ -401,13 +408,11 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, jh_in->b_frozen_triggers = jh_in->b_triggers; copy_done: - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } escape_done: - folio_set_bh(new_bh, new_folio, new_offset); new_bh->b_size = bh_in->b_size; new_bh->b_bdev = journal->j_dev; new_bh->b_blocknr = blocknr; @@ -882,7 +887,7 @@ int jbd2_fc_wait_bufs(journal_t *journal, int num_blks) * Update j_fc_off so jbd2_fc_release_bufs can release remain * buffer head. */ - if (unlikely(!buffer_uptodate(bh))) { + if (unlikely(buffer_write_io_error(bh))) { journal->j_fc_off = i + 1; return -EIO; } diff --git a/fs/jbd2/transaction.c b/fs/jbd2/transaction.c index 5cc7d097b2ac..85d84d909f78 100644 --- a/fs/jbd2/transaction.c +++ b/fs/jbd2/transaction.c @@ -920,7 +920,7 @@ static void jbd2_freeze_jh_data(struct journal_head *jh) char *source; struct buffer_head *bh = jh2bh(jh); - J_EXPECT_JH(jh, buffer_uptodate(bh), "Possible IO failure.\n"); + J_EXPECT_JH(jh, buffer_uptodate(bh), "Buffer not uptodate!\n"); source = kmap_local_folio(bh->b_folio, bh_offset(bh)); /* Fire data frozen trigger just before we copy the data */ jbd2_buffer_frozen_trigger(jh, source, jh->b_triggers); diff --git a/fs/jffs2/acl.c b/fs/jffs2/acl.c index f0f8a4f57add..7548f44bf327 100644 --- a/fs/jffs2/acl.c +++ b/fs/jffs2/acl.c @@ -228,7 +228,7 @@ static int __jffs2_set_acl(struct inode *inode, int xprefix, struct posix_acl *a return rc; } -int jffs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int rc, xprefix; diff --git a/fs/jffs2/acl.h b/fs/jffs2/acl.h index e976b8cb82cf..bc5df521633f 100644 --- a/fs/jffs2/acl.h +++ b/fs/jffs2/acl.h @@ -28,7 +28,7 @@ struct jffs2_acl_header { #ifdef CONFIG_JFFS2_FS_POSIX_ACL struct posix_acl *jffs2_get_acl(struct inode *inode, int type, bool rcu); -int jffs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int jffs2_init_acl_pre(struct inode *, struct inode *, umode_t *); extern int jffs2_init_acl_post(struct inode *); diff --git a/fs/jffs2/dir.c b/fs/jffs2/dir.c index 656c920864c5..23813f191281 100644 --- a/fs/jffs2/dir.c +++ b/fs/jffs2/dir.c @@ -25,20 +25,20 @@ static int jffs2_readdir (struct file *, struct dir_context *); -static int jffs2_create (struct mnt_idmap *, struct inode *, +static int jffs2_create (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); static struct dentry *jffs2_lookup (struct inode *,struct dentry *, unsigned int); static int jffs2_link (struct dentry *,struct inode *,struct dentry *); static int jffs2_unlink (struct inode *,struct dentry *); -static int jffs2_symlink (struct mnt_idmap *, struct inode *, +static int jffs2_symlink (const struct mnt_idmap *, struct inode *, struct dentry *, const char *); -static struct dentry *jffs2_mkdir (struct mnt_idmap *, struct inode *,struct dentry *, +static struct dentry *jffs2_mkdir (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); static int jffs2_rmdir (struct inode *,struct dentry *); -static int jffs2_mknod (struct mnt_idmap *, struct inode *,struct dentry *, +static int jffs2_mknod (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); -static int jffs2_rename (struct mnt_idmap *, struct inode *, +static int jffs2_rename (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); @@ -162,7 +162,7 @@ static int jffs2_readdir(struct file *file, struct dir_context *ctx) /***********************************************************************/ -static int jffs2_create(struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_create(const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode) { struct jffs2_raw_inode *ri; @@ -284,7 +284,7 @@ static int jffs2_link (struct dentry *old_dentry, struct inode *dir_i, struct de /***********************************************************************/ -static int jffs2_symlink (struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_symlink (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, const char *target) { struct jffs2_inode_info *f, *dir_f; @@ -448,7 +448,7 @@ static int jffs2_symlink (struct mnt_idmap *idmap, struct inode *dir_i, } -static struct dentry *jffs2_mkdir (struct mnt_idmap *idmap, struct inode *dir_i, +static struct dentry *jffs2_mkdir (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode) { struct jffs2_inode_info *f, *dir_f; @@ -620,7 +620,7 @@ static int jffs2_rmdir (struct inode *dir_i, struct dentry *dentry) return ret; } -static int jffs2_mknod (struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_mknod (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode, dev_t rdev) { struct jffs2_inode_info *f, *dir_f; @@ -769,7 +769,7 @@ static int jffs2_mknod (struct mnt_idmap *idmap, struct inode *dir_i, return ret; } -static int jffs2_rename (struct mnt_idmap *idmap, +static int jffs2_rename (const struct mnt_idmap *idmap, struct inode *old_dir_i, struct dentry *old_dentry, struct inode *new_dir_i, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/jffs2/fs.c b/fs/jffs2/fs.c index 6ada8369a762..05cf860307c7 100644 --- a/fs/jffs2/fs.c +++ b/fs/jffs2/fs.c @@ -190,7 +190,7 @@ int jffs2_do_setattr (struct inode *inode, struct iattr *iattr) return 0; } -int jffs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/jffs2/os-linux.h b/fs/jffs2/os-linux.h index 86ab014a349c..bff2134d771d 100644 --- a/fs/jffs2/os-linux.h +++ b/fs/jffs2/os-linux.h @@ -164,7 +164,7 @@ long jffs2_ioctl(struct file *, unsigned int, unsigned long); extern const struct inode_operations jffs2_symlink_inode_operations; /* fs.c */ -int jffs2_setattr (struct mnt_idmap *, struct dentry *, struct iattr *); +int jffs2_setattr (const struct mnt_idmap *, struct dentry *, struct iattr *); int jffs2_do_setattr (struct inode *, struct iattr *); struct inode *jffs2_iget(struct super_block *, unsigned long); void jffs2_evict_inode (struct inode *); diff --git a/fs/jffs2/security.c b/fs/jffs2/security.c index 437f3a2c1b54..67330aeb8ae8 100644 --- a/fs/jffs2/security.c +++ b/fs/jffs2/security.c @@ -57,7 +57,7 @@ static int jffs2_security_getxattr(const struct xattr_handler *handler, } static int jffs2_security_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jffs2/xattr_trusted.c b/fs/jffs2/xattr_trusted.c index b7c5da2d89bd..85133ad8b449 100644 --- a/fs/jffs2/xattr_trusted.c +++ b/fs/jffs2/xattr_trusted.c @@ -25,7 +25,7 @@ static int jffs2_trusted_getxattr(const struct xattr_handler *handler, } static int jffs2_trusted_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jffs2/xattr_user.c b/fs/jffs2/xattr_user.c index f64edce4927b..dcfd3caf1d8b 100644 --- a/fs/jffs2/xattr_user.c +++ b/fs/jffs2/xattr_user.c @@ -25,7 +25,7 @@ static int jffs2_user_getxattr(const struct xattr_handler *handler, } static int jffs2_user_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jfs/acl.c b/fs/jfs/acl.c index 16b71a23ff1e..6e0a7feb6c80 100644 --- a/fs/jfs/acl.c +++ b/fs/jfs/acl.c @@ -89,7 +89,7 @@ static int __jfs_set_acl(tid_t tid, struct inode *inode, int type, return rc; } -int jfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int rc; diff --git a/fs/jfs/file.c b/fs/jfs/file.c index 246568cb9a6e..2f5bb79c0591 100644 --- a/fs/jfs/file.c +++ b/fs/jfs/file.c @@ -89,7 +89,7 @@ static int jfs_release(struct inode *inode, struct file *file) return 0; } -int jfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/jfs/ioctl.c b/fs/jfs/ioctl.c index 563f148be8af..27d39cddaca6 100644 --- a/fs/jfs/ioctl.c +++ b/fs/jfs/ioctl.c @@ -70,7 +70,7 @@ int jfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int jfs_fileattr_set(struct mnt_idmap *idmap, +int jfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/jfs/jfs_acl.h b/fs/jfs/jfs_acl.h index f892e54d0fcd..bda26b333519 100644 --- a/fs/jfs/jfs_acl.h +++ b/fs/jfs/jfs_acl.h @@ -8,7 +8,7 @@ #ifdef CONFIG_JFS_POSIX_ACL struct posix_acl *jfs_get_acl(struct inode *inode, int type, bool rcu); -int jfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int jfs_init_acl(tid_t, struct inode *, struct inode *); diff --git a/fs/jfs/jfs_inode.h b/fs/jfs/jfs_inode.h index 2c6c81c8cb9f..5a118b07fbff 100644 --- a/fs/jfs/jfs_inode.h +++ b/fs/jfs/jfs_inode.h @@ -10,7 +10,7 @@ struct fid; extern struct inode *ialloc(struct inode *, umode_t); extern int jfs_fsync(struct file *, loff_t, loff_t, int); extern int jfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -extern int jfs_fileattr_set(struct mnt_idmap *idmap, +extern int jfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); extern long jfs_ioctl(struct file *, unsigned int, unsigned long); extern struct inode *jfs_iget(struct super_block *, unsigned long); @@ -28,7 +28,7 @@ extern struct dentry *jfs_fh_to_parent(struct super_block *sb, struct fid *fid, int fh_len, int fh_type); extern void jfs_set_inode_flags(struct inode *); extern int jfs_get_block(struct inode *, sector_t, struct buffer_head *, int); -extern int jfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int jfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern const struct address_space_operations jfs_aops; extern const struct inode_operations jfs_dir_inode_operations; diff --git a/fs/jfs/namei.c b/fs/jfs/namei.c index 8a36c218f0f7..8ab2e952ce16 100644 --- a/fs/jfs/namei.c +++ b/fs/jfs/namei.c @@ -60,7 +60,7 @@ static inline void free_ea_wmap(struct inode *inode) * RETURN: Errors from subroutines * */ -static int jfs_create(struct mnt_idmap *idmap, struct inode *dip, +static int jfs_create(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, umode_t mode) { int rc = 0; @@ -193,7 +193,7 @@ static int jfs_create(struct mnt_idmap *idmap, struct inode *dip, * note: * EACCES: user needs search+write permission on the parent directory */ -static struct dentry *jfs_mkdir(struct mnt_idmap *idmap, struct inode *dip, +static struct dentry *jfs_mkdir(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, umode_t mode) { int rc = 0; @@ -876,7 +876,7 @@ static int jfs_link(struct dentry *old_dentry, * an intermediate result whose length exceeds PATH_MAX [XPG4.2] */ -static int jfs_symlink(struct mnt_idmap *idmap, struct inode *dip, +static int jfs_symlink(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, const char *name) { int rc; @@ -1066,7 +1066,7 @@ static int jfs_symlink(struct mnt_idmap *idmap, struct inode *dip, * * FUNCTION: rename a file or directory */ -static int jfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int jfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1355,7 +1355,7 @@ static int jfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, * * FUNCTION: Create a special file (device) */ -static int jfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int jfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct jfs_inode_info *jfs_ip; diff --git a/fs/jfs/xattr.c b/fs/jfs/xattr.c index 11d7f74d207b..dcc4a69d44fe 100644 --- a/fs/jfs/xattr.c +++ b/fs/jfs/xattr.c @@ -956,7 +956,7 @@ static int jfs_xattr_get(const struct xattr_handler *handler, } static int jfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -975,7 +975,7 @@ static int jfs_xattr_get_os2(const struct xattr_handler *handler, } static int jfs_xattr_set_os2(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/kernfs/dir.c b/fs/kernfs/dir.c index 82bbaeb326aa..324d61a00545 100644 --- a/fs/kernfs/dir.c +++ b/fs/kernfs/dir.c @@ -30,6 +30,8 @@ static char kernfs_pr_cont_buf[PATH_MAX]; /* protected by pr_cont_lock */ #define rb_to_kn(X) rb_entry((X), struct kernfs_node, rb) +static void kernfs_activate_one(struct kernfs_node *kn); + static bool __kernfs_active(struct kernfs_node *kn) { return atomic_read(&kn->active) >= 0; @@ -736,13 +738,19 @@ struct kernfs_node *kernfs_new_node(struct kernfs_node *parent, { struct kernfs_node *kn; - if (parent->mode & S_ISGID) { + /* + * The mode and the gid below are read unlocked on purpose: they feed + * a node that does not exist yet, so nothing orders a racing chmod or + * chown against this creation. + */ + if (READ_ONCE(parent->mode) & S_ISGID) { /* this code block imitates inode_init_owner() for * kernfs */ + struct kernfs_iattrs *attrs = READ_ONCE(parent->iattr); - if (parent->iattr) - gid = parent->iattr->ia_gid; + if (attrs) + gid = READ_ONCE(attrs->ia_gid); if (flags & KERNFS_DIR) mode |= S_ISGID; @@ -855,7 +863,6 @@ int kernfs_add_one(struct kernfs_node *kn) } up_write(&root->kernfs_iattr_rwsem); - up_write(&root->kernfs_rwsem); /* * Activate the new node unless CREATE_DEACTIVATED is requested. @@ -863,9 +870,15 @@ int kernfs_add_one(struct kernfs_node *kn) * activating the node with kernfs_activate(). A node which hasn't * been activated is not visible to userland and its removal won't * trigger deactivation. + * + * @kn has no children yet, so kernfs_activate() would walk only @kn. + * Do it here rather than dropping the write lock and taking it again + * for every new node. */ - if (!(kernfs_root(kn)->flags & KERNFS_ROOT_CREATE_DEACTIVATED)) - kernfs_activate(kn); + if (!(root->flags & KERNFS_ROOT_CREATE_DEACTIVATED)) + kernfs_activate_one(kn); + + up_write(&root->kernfs_rwsem); return 0; out_unlock: @@ -1171,23 +1184,18 @@ struct kernfs_node *kernfs_create_empty_dir(struct kernfs_node *parent, static int kernfs_dop_revalidate(struct inode *dir, const struct qstr *name, struct dentry *dentry, unsigned int flags) { - struct kernfs_node *kn, *parent; - struct kernfs_root *root; + struct kernfs_node *parent = dir->i_private; + struct kernfs_node *kn; + const char *kn_name; if (flags & LOOKUP_RCU) return -ECHILD; /* Negative hashed dentry? */ if (d_really_is_negative(dentry)) { - /* If the kernfs parent node has changed discard and - * proceed to ->lookup. - * - * There's nothing special needed here when getting the - * dentry parent, even if a concurrent rename is in - * progress. That's because the dentry is negative so - * it can only be the target of the rename and it will - * be doing a d_move() not a replace. Consequently the - * dentry d_parent won't change over the d_move(). + /* + * If the kernfs parent node has changed discard and proceed to + * ->lookup. * * Also kernfs negative dentries transitioning from * negative to positive during revalidate won't happen @@ -1195,50 +1203,41 @@ static int kernfs_dop_revalidate(struct inode *dir, const struct qstr *name, * changes and the lookup re-done so that a new positive * dentry can be properly created. */ - root = kernfs_root_from_sb(dentry->d_sb); - down_read(&root->kernfs_rwsem); - parent = kernfs_dentry_node(dentry->d_parent); - if (parent) { - if (kernfs_dir_changed(parent, dentry)) { - up_read(&root->kernfs_rwsem); - return 0; - } - } - up_read(&root->kernfs_rwsem); - - /* The kernfs parent node hasn't changed, leave the - * dentry negative and return success. - */ - return 1; + return !kernfs_dir_changed(parent, dentry); } kn = kernfs_dentry_node(dentry); - root = kernfs_root(kn); - down_read(&root->kernfs_rwsem); + + guard(rcu)(); /* The kernfs node has been deactivated */ - if (!kernfs_active(kn)) - goto out_bad; + if (!__kernfs_active(kn)) + return 0; - parent = kernfs_parent(kn); /* The kernfs node has been moved? */ - if (kernfs_dentry_node(dentry->d_parent) != parent) - goto out_bad; + if (kernfs_parent(kn) != parent) + return 0; /* The kernfs node has been renamed */ - if (strcmp(dentry->d_name.name, kernfs_rcu_name(kn)) != 0) - goto out_bad; + kn_name = kernfs_rcu_name(kn); + if (name->len != strlen(kn_name) || + memcmp(name->name, kn_name, name->len)) + return 0; - /* The kernfs node has been moved to a different namespace */ - if (parent && kernfs_ns_enabled(parent) && - kernfs_ns_id(kernfs_info(dentry->d_sb)->ns) != kernfs_ns_id(kn->ns)) - goto out_bad; + /* + * The kernfs node has been moved to a different namespace. + * + * KERNFS_NS is set by kernfs_enable_ns() while @parent still has no + * children, so it cannot change while a child of @parent is being + * revalidated. The other bits in that word, KERNFS_ACTIVATED and + * KERNFS_REMOVING, are updated under kernfs_rwsem and are not read + * here, so racing with them is intentional and harmless. + */ + if (data_race(kernfs_ns_enabled(parent)) && + kernfs_info(dir->i_sb)->ns != READ_ONCE(kn->ns)) + return 0; - up_read(&root->kernfs_rwsem); return 1; -out_bad: - up_read(&root->kernfs_rwsem); - return 0; } const struct dentry_operations kernfs_dops = { @@ -1288,7 +1287,7 @@ static struct dentry *kernfs_iop_lookup(struct inode *dir, return d_splice_alias(inode, dentry); } -static struct dentry *kernfs_iop_mkdir(struct mnt_idmap *idmap, +static struct dentry *kernfs_iop_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { @@ -1326,7 +1325,7 @@ static int kernfs_iop_rmdir(struct inode *dir, struct dentry *dentry) return ret; } -static int kernfs_iop_rename(struct mnt_idmap *idmap, +static int kernfs_iop_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1820,14 +1819,20 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, const char *new_name, const struct ns_common *new_ns) { struct kernfs_node *old_parent; + const char *dup_name = NULL; + const char *put_name = NULL; struct kernfs_root *root; const char *old_name; + bool reparent; int error; /* can't move or rename root */ if (!rcu_access_pointer(kn->__parent)) return -EINVAL; + if (new_name) + dup_name = kstrdup_const(new_name, GFP_KERNEL); + root = kernfs_root(kn); down_write(&root->kernfs_rwsem); @@ -1859,9 +1864,10 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, /* rename kernfs_node */ if (strcmp(old_name, new_name) != 0) { error = -ENOMEM; - new_name = kstrdup_const(new_name, GFP_KERNEL); - if (!new_name) + if (!dup_name) goto out; + new_name = dup_name; + dup_name = NULL; } else { new_name = NULL; } @@ -1871,35 +1877,39 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, */ kernfs_unlink_sibling(kn); - /* rename_lock protects ->parent accessors */ - if (old_parent != new_parent) { + reparent = old_parent != new_parent; + if (reparent) kernfs_get(new_parent); - write_lock_irq(&root->kernfs_rename_lock); + /* + * kernfs_rename_lock protects ->__parent, ->ns and ->name, so take it + * even when the parent does not change. + */ + write_lock_irq(&root->kernfs_rename_lock); + + if (reparent) rcu_assign_pointer(kn->__parent, new_parent); + WRITE_ONCE(kn->ns, new_ns); + if (new_name) + rcu_assign_pointer(kn->name, new_name); - kn->ns = new_ns; - if (new_name) - rcu_assign_pointer(kn->name, new_name); + write_unlock_irq(&root->kernfs_rename_lock); - write_unlock_irq(&root->kernfs_rename_lock); + if (reparent) kernfs_put(old_parent); - } else { - /* name assignment is RCU protected, parent is the same */ - kn->ns = new_ns; - if (new_name) - rcu_assign_pointer(kn->name, new_name); - } kn->hash = kernfs_name_hash(new_name ?: old_name, kn->ns); kernfs_link_sibling(kn); if (new_name && !is_kernel_rodata((unsigned long)old_name)) - kfree_rcu_mightsleep(old_name); + put_name = old_name; error = 0; out: up_write(&root->kernfs_rwsem); + kfree_const(dup_name); + if (put_name) + kfree_rcu_mightsleep(put_name); return error; } @@ -1909,33 +1919,49 @@ static int kernfs_dir_fop_release(struct inode *inode, struct file *filp) return 0; } +/* + * Find where a listing left off. @resumed says whether @pos is still that + * entry; if not, the search falls back to @hash, keyed by @name if given. + */ static struct kernfs_node *kernfs_dir_pos(const struct ns_common *ns, - struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos) + struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos, + const char *name, bool *resumed) { + if (resumed) + *resumed = false; if (pos) { + /* + * A rename keeps the hash if the new name hashes the same, so + * check @name too. Otherwise the caller would step over the + * entry now sitting where @pos used to be. + */ int valid = kernfs_active(pos) && rcu_access_pointer(pos->__parent) == parent && - hash == pos->hash; + hash == pos->hash && + (!name || !strcmp(name, kernfs_rcu_name(pos))); kernfs_put(pos); if (!valid) pos = NULL; + else if (resumed) + *resumed = true; } if (!pos && (hash > 1) && (hash < INT_MAX)) { struct rb_node *node = parent->dir.children.rb_node; - u64 ns_id = kernfs_ns_id(ns); + + /* + * Keep a node only on the way left, so the search ends on the + * first entry after the key. An empty @name sorts before all + * entries sharing the hash, so it lands on the first of them. + */ while (node) { - pos = rb_to_kn(node); + struct kernfs_node *kn = rb_to_kn(node); - if (hash < pos->hash) + if (kernfs_name_compare(hash, name ?: "", ns, kn) < 0) { + pos = kn; node = node->rb_left; - else if (hash > pos->hash) + } else { node = node->rb_right; - else if (ns_id < kernfs_ns_id(pos->ns)) - node = node->rb_left; - else if (ns_id > kernfs_ns_id(pos->ns)) - node = node->rb_right; - else - break; + } } } /* Skip over entries which are dying/dead or in the wrong namespace */ @@ -1951,10 +1977,14 @@ static struct kernfs_node *kernfs_dir_pos(const struct ns_common *ns, } static struct kernfs_node *kernfs_dir_next_pos(const struct ns_common *ns, - struct kernfs_node *parent, ino_t ino, struct kernfs_node *pos) + struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos, + const char *name) { - pos = kernfs_dir_pos(ns, parent, ino, pos); - if (pos) { + bool resumed; + + pos = kernfs_dir_pos(ns, parent, hash, pos, name, &resumed); + /* Step over @pos only if it survived; @name finds the spot if not. */ + if (pos && resumed) { do { struct rb_node *node = rb_next(&pos->rb); if (!node) @@ -1972,34 +2002,55 @@ static int kernfs_fop_readdir(struct file *file, struct dir_context *ctx) struct dentry *dentry = file->f_path.dentry; struct kernfs_node *parent = kernfs_dentry_node(dentry); struct kernfs_node *pos = file->private_data; + char *name __free(kfree) = NULL; struct kernfs_root *root; const struct ns_common *ns = NULL; if (!dir_emit_dots(file, ctx)) return 0; + /* + * One buffer for the call, holding the name of the entry the listing + * is on. PATH_MAX: kernfs bounds no single name. + */ + name = kmalloc(PATH_MAX, GFP_KERNEL); + if (!name) + return -ENOMEM; + root = kernfs_root(parent); down_read(&root->kernfs_rwsem); if (kernfs_ns_enabled(parent)) ns = kernfs_info(dentry->d_sb)->ns; - for (pos = kernfs_dir_pos(ns, parent, ctx->pos, pos); + for (pos = kernfs_dir_pos(ns, parent, ctx->pos, pos, NULL, NULL); pos; - pos = kernfs_dir_next_pos(ns, parent, ctx->pos, pos)) { - const char *name = kernfs_rcu_name(pos); + pos = kernfs_dir_next_pos(ns, parent, ctx->pos, pos, name)) { unsigned int type = fs_umode_to_dtype(pos->mode); - int len = strlen(name); ino_t ino = kernfs_ino(pos); + int len; + + /* + * The copy is also the resume key, so a truncated name would + * resume here again. getname() caps a path, so only an + * in-kernel caller can get here; end the listing instead. + */ + len = strscpy(name, kernfs_rcu_name(pos), PATH_MAX); + if (WARN_ON_ONCE(len < 0)) + break; ctx->pos = pos->hash; file->private_data = pos; kernfs_get(pos); - if (!dir_emit(ctx, name, len, ino, type)) { - up_read(&root->kernfs_rwsem); + /* + * dir_emit() can fault, so run it unlocked. @pos is pinned + * above and kernfs_dir_pos() rechecks it on the way back. + */ + up_read(&root->kernfs_rwsem); + if (!dir_emit(ctx, name, len, ino, type)) return 0; - } + down_read(&root->kernfs_rwsem); } up_read(&root->kernfs_rwsem); file->private_data = NULL; diff --git a/fs/kernfs/file.c b/fs/kernfs/file.c index 8e0e90c93372..cca9f83fc9b5 100644 --- a/fs/kernfs/file.c +++ b/fs/kernfs/file.c @@ -525,18 +525,31 @@ out_unlock: static int kernfs_get_open_node(struct kernfs_node *kn, struct kernfs_open_file *of) { - struct kernfs_open_node *on; + struct kernfs_open_node *on, *new_on = NULL; struct mutex *mutex; + /* + * Peek without the mutex: if nothing has this open, we will need a + * node and can allocate before taking a mutex shared by every node + * hashing to it. + */ + if (!rcu_access_pointer(kn->attr.open)) + new_on = kzalloc_obj(*new_on); + mutex = kernfs_open_file_mutex_lock(kn); on = kernfs_deref_open_node_locked(kn); if (!on) { /* not there, initialize a new one */ - on = kzalloc_obj(*on); + on = new_on; + new_on = NULL; if (!on) { - mutex_unlock(mutex); - return -ENOMEM; + /* the peek raced; rare, so allocate here */ + on = kzalloc_obj(*on); + if (!on) { + mutex_unlock(mutex); + return -ENOMEM; + } } atomic_set(&on->event, 1); init_waitqueue_head(&on->poll); @@ -549,6 +562,7 @@ static int kernfs_get_open_node(struct kernfs_node *kn, on->nr_to_release++; mutex_unlock(mutex); + kfree(new_on); return 0; } @@ -904,9 +918,12 @@ static loff_t kernfs_fop_llseek(struct file *file, loff_t offset, int whence) static void kernfs_notify_workfn(struct work_struct *work) { - struct kernfs_node *kn; + char name_buf[NAME_MAX + 1]; struct kernfs_super_info *info; + struct kernfs_node *kn; struct kernfs_root *root; + struct qstr name; + bool have_name; repeat: /* pop one off the notify_list */ spin_lock_irq(&kernfs_notify_lock); @@ -922,14 +939,20 @@ repeat: root = kernfs_root(kn); /* kick fsnotify */ + /* + * Sample the name once so kernfs_rwsem need not be held across the + * loop. A name that does not fit is reported without one; fsnotify() + * takes the name as optional, so a watcher loses the name and not the + * event. + */ + have_name = kernfs_name(kn, name_buf, sizeof(name_buf)) >= 0; + name = QSTR(name_buf); + down_read(&root->kernfs_supers_rwsem); - down_read(&root->kernfs_rwsem); - list_for_each_entry(info, &kernfs_root(kn)->supers, node) { + list_for_each_entry(info, &root->supers, node) { struct kernfs_node *parent; struct inode *p_inode = NULL; - const char *kn_name; struct inode *inode; - struct qstr name; /* * We want fsnotify_modify() on @kn but as the @@ -941,15 +964,14 @@ repeat: if (!inode) continue; - kn_name = kernfs_rcu_name(kn); - name = QSTR(kn_name); parent = kernfs_get_parent(kn); if (parent) { p_inode = ilookup(info->sb, kernfs_ino(parent)); if (p_inode) { fsnotify(FS_MODIFY | FS_EVENT_ON_CHILD, inode, FSNOTIFY_EVENT_INODE, - p_inode, &name, inode, 0); + p_inode, have_name ? &name : NULL, + inode, 0); iput(p_inode); } @@ -962,7 +984,6 @@ repeat: iput(inode); } - up_read(&root->kernfs_rwsem); up_read(&root->kernfs_supers_rwsem); kernfs_put(kn); goto repeat; diff --git a/fs/kernfs/inode.c b/fs/kernfs/inode.c index abb286bc3474..6630b29d7c07 100644 --- a/fs/kernfs/inode.c +++ b/fs/kernfs/inode.c @@ -107,7 +107,7 @@ int kernfs_setattr(struct kernfs_node *kn, const struct iattr *iattr) return ret; } -int kernfs_iop_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int kernfs_iop_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -179,7 +179,7 @@ static void kernfs_refresh_inode(struct kernfs_node *kn, struct inode *inode) set_nlink(inode, kn->dir.subdirs + 2); } -int kernfs_iop_getattr(struct mnt_idmap *idmap, +int kernfs_iop_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -270,7 +270,7 @@ void kernfs_evict_inode(struct inode *inode) kernfs_put(kn); } -int kernfs_iop_permission(struct mnt_idmap *idmap, +int kernfs_iop_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct kernfs_node *kn; @@ -342,7 +342,7 @@ static int kernfs_vfs_xattr_get(const struct xattr_handler *handler, } static int kernfs_vfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) @@ -354,7 +354,7 @@ static int kernfs_vfs_xattr_set(const struct xattr_handler *handler, } static int kernfs_vfs_user_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) diff --git a/fs/kernfs/kernfs-internal.h b/fs/kernfs/kernfs-internal.h index aa784b540b36..f1e93b09e27d 100644 --- a/fs/kernfs/kernfs-internal.h +++ b/fs/kernfs/kernfs-internal.h @@ -117,7 +117,14 @@ static inline bool kernfs_rename_is_locked(const struct kernfs_node *kn) static inline const char *kernfs_rcu_name(const struct kernfs_node *kn) { - return rcu_dereference_check(kn->name, kernfs_root_is_locked(kn)); + /* + * Like kernfs_node::__parent below, the name is only replaced under + * both kernfs_root::kernfs_rwsem and kernfs_root::kernfs_rename_lock, + * so either one keeps it, and the string it points at, stable. + */ + return rcu_dereference_check(kn->name, + kernfs_root_is_locked(kn) || + kernfs_rename_is_locked(kn)); } static inline struct kernfs_node *kernfs_parent(const struct kernfs_node *kn) @@ -147,20 +154,19 @@ static inline struct kernfs_node *kernfs_dentry_node(struct dentry *dentry) static inline void kernfs_set_rev(struct kernfs_node *parent, struct dentry *dentry) { - dentry->d_time = parent->dir.rev; + WRITE_ONCE(dentry->d_time, READ_ONCE(parent->dir.rev)); } static inline void kernfs_inc_rev(struct kernfs_node *parent) { - parent->dir.rev++; + lockdep_assert_held_write(&parent->dir.root->kernfs_rwsem); + WRITE_ONCE(parent->dir.rev, parent->dir.rev + 1); } static inline bool kernfs_dir_changed(struct kernfs_node *parent, struct dentry *dentry) { - if (parent->dir.rev != dentry->d_time) - return true; - return false; + return READ_ONCE(parent->dir.rev) != READ_ONCE(dentry->d_time); } extern const struct super_operations kernfs_sops; @@ -171,11 +177,11 @@ extern struct kmem_cache *kernfs_node_cache, *kernfs_iattrs_cache; */ extern const struct xattr_handler * const kernfs_xattr_handlers[]; void kernfs_evict_inode(struct inode *inode); -int kernfs_iop_permission(struct mnt_idmap *idmap, +int kernfs_iop_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); -int kernfs_iop_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int kernfs_iop_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); -int kernfs_iop_getattr(struct mnt_idmap *idmap, +int kernfs_iop_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); ssize_t kernfs_iop_listxattr(struct dentry *dentry, char *buf, size_t size); diff --git a/fs/kernfs/mount.c b/fs/kernfs/mount.c index a57399021c8b..a0be784bfb06 100644 --- a/fs/kernfs/mount.c +++ b/fs/kernfs/mount.c @@ -124,22 +124,32 @@ static struct dentry *__kernfs_fh_to_dentry(struct super_block *sb, return NULL; } - kn = kernfs_find_and_get_node_by_id(info->root, id); - if (!kn) - return ERR_PTR(-ESTALE); + /* + * Hold kernfs_rwsem across the lookup as well as kernfs_get_inode(). + * __kernfs_remove() deactivates the subtree and clears i_nlink on its + * inodes under the write lock, so under the read lock either + * kernfs_find_and_get_node_by_id() refuses the node, or the inode is + * in the inode hash before the ilookup() pass goes looking for it. + */ + scoped_guard(rwsem_read, &info->root->kernfs_rwsem) { + kn = kernfs_find_and_get_node_by_id(info->root, id); + if (!kn) + return ERR_PTR(-ESTALE); - if (get_parent) { - struct kernfs_node *parent; + if (get_parent) { + struct kernfs_node *parent; - parent = kernfs_get_parent(kn); + parent = kernfs_get_parent(kn); + kernfs_put(kn); + kn = parent; + if (!kn) + return ERR_PTR(-ESTALE); + } + + inode = kernfs_get_inode(sb, kn); kernfs_put(kn); - kn = parent; - if (!kn) - return ERR_PTR(-ESTALE); } - inode = kernfs_get_inode(sb, kn); - kernfs_put(kn); return d_obtain_alias(inode); } diff --git a/fs/kernfs/symlink.c b/fs/kernfs/symlink.c index 90e2b3221b83..3e53105d3abf 100644 --- a/fs/kernfs/symlink.c +++ b/fs/kernfs/symlink.c @@ -31,9 +31,20 @@ struct kernfs_node *kernfs_create_link(struct kernfs_node *parent, kuid_t uid = GLOBAL_ROOT_UID; kgid_t gid = GLOBAL_ROOT_GID; - if (target->iattr) { - uid = target->iattr->ia_uid; - gid = target->iattr->ia_gid; + /* + * A symlink takes its owner from its target, so both fields have to + * come from the same moment: read them under kernfs_iattr_rwsem, or + * a chown of the target racing this could leave the link with the + * old uid and the new gid. The section ends before kernfs_add_one() + * takes kernfs_rwsem. + */ + scoped_guard(rwsem_read, &kernfs_root(target)->kernfs_iattr_rwsem) { + struct kernfs_iattrs *attrs = READ_ONCE(target->iattr); + + if (attrs) { + uid = attrs->ia_uid; + gid = attrs->ia_gid; + } } kn = kernfs_new_node(parent, name, S_IFLNK|0777, uid, gid, KERNFS_LINK); diff --git a/fs/libfs.c b/fs/libfs.c index 27d7dc16fcb0..8e2cc627bc7f 100644 --- a/fs/libfs.c +++ b/fs/libfs.c @@ -29,7 +29,7 @@ #include "internal.h" -int simple_getattr(struct mnt_idmap *idmap, const struct path *path, +int simple_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -867,7 +867,7 @@ int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, } EXPORT_SYMBOL_GPL(simple_rename_exchange); -int simple_rename(struct mnt_idmap *idmap, struct inode *old_dir, +int simple_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -913,7 +913,7 @@ EXPORT_SYMBOL(simple_rename); * on simple regular filesystems. Anything that needs to change on-disk * or wire state on size changes needs its own setattr method. */ -int simple_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int simple_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -1727,7 +1727,7 @@ static struct dentry *empty_dir_lookup(struct inode *dir, struct dentry *dentry, return ERR_PTR(-ENOENT); } -static int empty_dir_setattr(struct mnt_idmap *idmap, +static int empty_dir_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return -EPERM; diff --git a/fs/minix/file.c b/fs/minix/file.c index 02aabbdb5dea..c0929fc38fbc 100644 --- a/fs/minix/file.c +++ b/fs/minix/file.c @@ -23,7 +23,7 @@ const struct file_operations minix_file_operations = { .splice_read = filemap_splice_read, }; -static int minix_setattr(struct mnt_idmap *idmap, +static int minix_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/minix/inode.c b/fs/minix/inode.c index daf83e4ff25c..670179173645 100644 --- a/fs/minix/inode.c +++ b/fs/minix/inode.c @@ -724,7 +724,7 @@ out: return err; } -int minix_getattr(struct mnt_idmap *idmap, const struct path *path, +int minix_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct super_block *sb = path->dentry->d_sb; diff --git a/fs/minix/minix.h b/fs/minix/minix.h index 78722ce22e1e..db92cf9e0e1b 100644 --- a/fs/minix/minix.h +++ b/fs/minix/minix.h @@ -55,7 +55,7 @@ unsigned long minix_count_free_inodes(struct super_block *sb); int minix_new_block(struct inode *inode); void minix_free_block(struct inode *inode, unsigned long block); unsigned long minix_count_free_blocks(struct super_block *sb); -int minix_getattr(struct mnt_idmap *, const struct path *, +int minix_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); int minix_prepare_chunk(struct folio *folio, loff_t pos, unsigned len); struct mapping_metadata_bhs *minix_get_metadata_bhs(struct inode *inode); diff --git a/fs/minix/namei.c b/fs/minix/namei.c index 5525ba367ed7..f450b11b9860 100644 --- a/fs/minix/namei.c +++ b/fs/minix/namei.c @@ -33,7 +33,7 @@ static struct dentry *minix_lookup(struct inode * dir, struct dentry *dentry, un return d_splice_alias(inode, dentry); } -static int minix_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int minix_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -50,7 +50,7 @@ static int minix_mknod(struct mnt_idmap *idmap, struct inode *dir, return add_nondir(dentry, inode); } -static int minix_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int minix_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = minix_new_inode(dir, mode); @@ -63,13 +63,13 @@ static int minix_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int minix_create(struct mnt_idmap *idmap, struct inode *dir, +static int minix_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return minix_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); } -static int minix_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int minix_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int i = strlen(symname)+1; @@ -104,7 +104,7 @@ static int minix_link(struct dentry * old_dentry, struct inode * dir, return add_nondir(dentry, inode); } -static struct dentry *minix_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *minix_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode * inode; @@ -187,7 +187,7 @@ out: return err; } -static int minix_rename(struct mnt_idmap *idmap, +static int minix_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/mnt_idmapping.c b/fs/mnt_idmapping.c index cb61fbdb52e9..bed57094cef0 100644 --- a/fs/mnt_idmapping.c +++ b/fs/mnt_idmapping.c @@ -28,7 +28,7 @@ struct mnt_idmap { * mapping. This means that {g,u}id 0 is mapped to {g,u}id 0, {g,u}id 1 is * mapped to {g,u}id 1, [...], {g,u}id 1000 to {g,u}id 1000, [...]. */ -struct mnt_idmap nop_mnt_idmap = { +const struct mnt_idmap nop_mnt_idmap = { .count = REFCOUNT_INIT(1), }; EXPORT_SYMBOL_GPL(nop_mnt_idmap); @@ -37,7 +37,7 @@ EXPORT_SYMBOL_GPL(nop_mnt_idmap); * Carries the invalid idmapping of a full 0-4294967295 {g,u}id range. * This means that all {g,u}ids are mapped to INVALID_VFS{G,U}ID. */ -struct mnt_idmap invalid_mnt_idmap = { +const struct mnt_idmap invalid_mnt_idmap = { .count = REFCOUNT_INIT(1), }; EXPORT_SYMBOL_GPL(invalid_mnt_idmap); @@ -77,7 +77,7 @@ static inline bool initial_idmapping(const struct user_namespace *ns) * returned. */ -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid) { @@ -117,7 +117,7 @@ EXPORT_SYMBOL_GPL(make_vfsuid); * If @kgid has no mapping in either @idmap or @fs_userns INVALID_GID is * returned. */ -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid) { gid_t gid; @@ -147,7 +147,7 @@ EXPORT_SYMBOL_GPL(make_vfsgid); * * Return: @vfsuid mapped into the filesystem idmapping */ -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { uid_t uid; @@ -176,7 +176,7 @@ EXPORT_SYMBOL_GPL(from_vfsuid); * * Return: @vfsgid mapped into the filesystem idmapping */ -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { gid_t gid; @@ -312,10 +312,12 @@ struct mnt_idmap *alloc_mnt_idmap(struct user_namespace *mnt_userns) * * Return: @idmap with reference count bumped if @not_mnt_idmap isn't passed. */ -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap) +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap) { + struct mnt_idmap *nonconst_idmap = (struct mnt_idmap *)idmap; + if (idmap != &nop_mnt_idmap && idmap != &invalid_mnt_idmap) - refcount_inc(&idmap->count); + refcount_inc(&nonconst_idmap->count); return idmap; } @@ -328,17 +330,20 @@ EXPORT_SYMBOL_GPL(mnt_idmap_get); * If this is a non-initial idmapping, put the reference count when a mount is * released and free it if we're the last user. */ -void mnt_idmap_put(struct mnt_idmap *idmap) +void mnt_idmap_put(const struct mnt_idmap *idmap) { + struct mnt_idmap *nonconst_idmap = (struct mnt_idmap *)idmap; + if (idmap != &nop_mnt_idmap && idmap != &invalid_mnt_idmap && - refcount_dec_and_test(&idmap->count)) - free_mnt_idmap(idmap); + refcount_dec_and_test(&nonconst_idmap->count)) + free_mnt_idmap(nonconst_idmap); } EXPORT_SYMBOL_GPL(mnt_idmap_put); -int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map) +int statmount_mnt_idmap(const struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map) { - struct uid_gid_map *map, *map_up; + const struct uid_gid_map *map; + struct uid_gid_map *map_up; u32 idx, nr_mappings; if (!is_valid_mnt_idmap(idmap)) @@ -358,7 +363,7 @@ int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_ for (idx = 0, nr_mappings = 0; idx < map->nr_extents; idx++) { uid_t lower; - struct uid_gid_extent *extent; + const struct uid_gid_extent *extent; if (map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS) extent = &map->extent[idx]; diff --git a/fs/mount.h b/fs/mount.h index 94fcc306d21e..85f136786bbc 100644 --- a/fs/mount.h +++ b/fs/mount.h @@ -33,7 +33,8 @@ struct mnt_namespace { } __randomize_layout; struct mnt_pcp { - int mnt_count; + unsigned int mnt_gets; + unsigned int mnt_puts; int mnt_writers; }; diff --git a/fs/namei.c b/fs/namei.c index d95249dd527c..59c8a669081a 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -371,7 +371,7 @@ struct filename *complete_getname(struct delayed_filename *v) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -static int check_acl(struct mnt_idmap *idmap, +static int check_acl(const struct mnt_idmap *idmap, struct inode *inode, int mask) { #ifdef CONFIG_FS_POSIX_ACL @@ -435,7 +435,7 @@ static inline bool no_acl_inode(struct inode *inode) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -static int acl_permission_check(struct mnt_idmap *idmap, +static int acl_permission_check(const struct mnt_idmap *idmap, struct inode *inode, int mask) { unsigned int mode = inode->i_mode; @@ -518,7 +518,7 @@ static int acl_permission_check(struct mnt_idmap *idmap, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int generic_permission(struct mnt_idmap *idmap, struct inode *inode, +int generic_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret; @@ -575,7 +575,7 @@ EXPORT_SYMBOL(generic_permission); * flag in inode->i_opflags, that says "this has not special * permission function, use the fast case". */ -static inline int do_inode_permission(struct mnt_idmap *idmap, +static inline int do_inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { if (unlikely(!(inode->i_opflags & IOP_FASTPERM))) { @@ -625,7 +625,7 @@ static int sb_permission(struct super_block *sb, struct inode *inode, int mask) * * When checking for MAY_APPEND, MAY_WRITE must also be set in @mask. */ -int inode_permission(struct mnt_idmap *idmap, +int inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int retval; @@ -680,7 +680,7 @@ EXPORT_SYMBOL(inode_permission); * on IOP_FASTPERM can still get the optimization if they set IOP_FASTPERM_MAY_EXEC * on their directory inodes. */ -static __always_inline int lookup_inode_permission_may_exec(struct mnt_idmap *idmap, +static __always_inline int lookup_inode_permission_may_exec(const struct mnt_idmap *idmap, struct inode *inode, int mask) { /* Lookup already checked this to return -ENOTDIR */ @@ -1273,7 +1273,7 @@ fs_initcall(init_fs_namei_sysctls); */ static inline int may_follow_link(struct nameidata *nd, const struct inode *inode) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; vfsuid_t vfsuid; if (!sysctl_protected_symlinks) @@ -1314,7 +1314,7 @@ static inline int may_follow_link(struct nameidata *nd, const struct inode *inod * * Otherwise returns true. */ -static bool safe_hardlink_source(struct mnt_idmap *idmap, +static bool safe_hardlink_source(const struct mnt_idmap *idmap, struct inode *inode) { umode_t mode = inode->i_mode; @@ -1357,7 +1357,7 @@ static bool safe_hardlink_source(struct mnt_idmap *idmap, * * Returns 0 if successful, -ve on error. */ -int may_linkat(struct mnt_idmap *idmap, const struct path *link) +int may_linkat(const struct mnt_idmap *idmap, const struct path *link) { struct inode *inode = link->dentry->d_inode; @@ -1407,7 +1407,7 @@ int may_linkat(struct mnt_idmap *idmap, const struct path *link) * * Returns 0 if the open is allowed, -ve on error. */ -static int may_create_in_sticky(struct mnt_idmap *idmap, struct nameidata *nd, +static int may_create_in_sticky(const struct mnt_idmap *idmap, struct nameidata *nd, struct inode *const inode) { umode_t dir_mode = nd->dir_mode; @@ -1933,7 +1933,7 @@ static noinline struct dentry *lookup_slow(const struct qstr *name, struct inode *inode = dir->d_inode; struct dentry *res; inode_lock_shared(inode); - res = __lookup_slow(name, dir, flags); + res = __lookup_slow(name, dir, flags | LOOKUP_SHARED); inode_unlock_shared(inode); return res; } @@ -1947,12 +1947,12 @@ static struct dentry *lookup_slow_killable(const struct qstr *name, if (inode_lock_shared_killable(inode)) return ERR_PTR(-EINTR); - res = __lookup_slow(name, dir, flags); + res = __lookup_slow(name, dir, flags | LOOKUP_SHARED); inode_unlock_shared(inode); return res; } -static inline int may_lookup(struct mnt_idmap *idmap, +static inline int may_lookup(const struct mnt_idmap *idmap, struct nameidata *restrict nd) { int err, mask; @@ -2596,7 +2596,7 @@ static int link_path_walk(const char *name, struct nameidata *nd) /* At this point we know we have a real path component. */ for(;;) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; const char *link; unsigned long lastword; @@ -2946,8 +2946,8 @@ struct dentry *start_dirop(struct dentry *parent, struct qstr *name, * end_dirop - signal completion of a dirop * @de: the dentry which was returned by start_dirop or similar. * - * If the de is an error, nothing happens. Otherwise any lock taken to - * protect the dentry is dropped and the dentry itself is release (dput()). + * If the @de is an error, nothing happens. Otherwise any lock taken to + * protect the dentry is dropped and the dentry itself is released (dput()). */ void end_dirop(struct dentry *de) { @@ -3111,7 +3111,7 @@ int lookup_noperm_common(struct qstr *qname, struct dentry *base) return 0; } -static int lookup_one_common(struct mnt_idmap *idmap, +static int lookup_one_common(const struct mnt_idmap *idmap, struct qstr *qname, struct dentry *base) { int err; @@ -3190,7 +3190,7 @@ EXPORT_SYMBOL(lookup_noperm); * * The caller must hold base->i_rwsem. */ -struct dentry *lookup_one(struct mnt_idmap *idmap, struct qstr *name, +struct dentry *lookup_one(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { struct dentry *dentry; @@ -3210,7 +3210,7 @@ EXPORT_SYMBOL(lookup_one); /** * lookup_one_unlocked - lookup single pathname component * @idmap: idmap of the mount the lookup is performed from - * @name: qstr olding pathname component to lookup + * @name: qstr holding pathname component to lookup * @base: base directory to lookup from * * This can be used for in-kernel filesystem clients such as file servers. @@ -3223,7 +3223,7 @@ EXPORT_SYMBOL(lookup_one); * - ERR_PTR(-ENOENT) if parent has been removed, or * - ERR_PTR(-EACCES) if parent directory is not searchable. */ -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, struct qstr *name, +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { int err; @@ -3243,7 +3243,7 @@ EXPORT_SYMBOL(lookup_one_unlocked); /** * lookup_one_positive_killable - lookup single pathname component * @idmap: idmap of the mount the lookup is performed from - * @name: qstr olding pathname component to lookup + * @name: qstr holding pathname component to lookup * @base: base directory to lookup from * * This helper will yield ERR_PTR(-ENOENT) on negatives. The helper returns @@ -3259,11 +3259,11 @@ EXPORT_SYMBOL(lookup_one_unlocked); * the i_rwsem itself if necessary. If a fatal signal is pending or * delivered, it will return %-EINTR if the lock is needed. * - * Returns: A dentry, possibly negative, or + * Returns: A positive dentry, or * - same errors as lookup_one_unlocked() or * - ERR_PTR(-EINTR) if a fatal signal is pending. */ -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { @@ -3306,7 +3306,7 @@ EXPORT_SYMBOL(lookup_one_positive_killable); * - ERR_PTR(-ENOENT) if the name could not be found, or * - same errors as lookup_one_unlocked(). */ -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { @@ -3381,7 +3381,7 @@ struct dentry *lookup_noperm_positive_unlocked(struct qstr *name, EXPORT_SYMBOL(lookup_noperm_positive_unlocked); /** - * start_creating - prepare to create a given name with permission checking + * start_creating - prepare to access or create a given name with permission checking * @idmap: idmap of the mount * @parent: directory in which to prepare to create the name * @name: the name to be created @@ -3396,7 +3396,7 @@ EXPORT_SYMBOL(lookup_noperm_positive_unlocked); * * Returns: a negative or positive dentry, or an error. */ -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { int err = lookup_one_common(idmap, name, parent); @@ -3413,8 +3413,8 @@ EXPORT_SYMBOL(start_creating); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing - * an object from a directory. Permission checking (MAY_EXEC) is performed + * Locks are taken and a lookup is performed prior to removing an object + * from a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * * If the name doesn't exist, an error is returned. @@ -3423,7 +3423,7 @@ EXPORT_SYMBOL(start_creating); * * Returns: a positive dentry, or an error. */ -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { int err = lookup_one_common(idmap, name, parent); @@ -3440,7 +3440,7 @@ EXPORT_SYMBOL(start_removing); * @parent: directory in which to prepare to create the name * @name: the name to be created * - * Locks are taken and a lookup in performed prior to creating + * Locks are taken and a lookup is performed prior to creating * an object in a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * @@ -3451,7 +3451,7 @@ EXPORT_SYMBOL(start_removing); * * Returns: a negative or positive dentry, or an error. */ -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { @@ -3469,7 +3469,7 @@ EXPORT_SYMBOL(start_creating_killable); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing + * Locks are taken and a lookup is performed prior to removing * an object from a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * @@ -3482,7 +3482,7 @@ EXPORT_SYMBOL(start_creating_killable); * * Returns: a positive dentry, or an error. */ -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { @@ -3499,7 +3499,7 @@ EXPORT_SYMBOL(start_removing_killable); * @parent: directory in which to prepare to create the name * @name: the name to be created * - * Locks are taken and a lookup in performed prior to creating + * Locks are taken and a lookup is performed prior to creating * an object in a directory. * * If the name already exists, a positive dentry is returned. @@ -3522,7 +3522,7 @@ EXPORT_SYMBOL(start_creating_noperm); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing + * Locks are taken and a lookup is performed prior to removing * an object from a directory. * * If the name doesn't exist, an error is returned. @@ -3543,11 +3543,11 @@ struct dentry *start_removing_noperm(struct dentry *parent, EXPORT_SYMBOL(start_removing_noperm); /** - * start_creating_dentry - prepare to create a given dentry - * @parent: directory from which dentry should be removed - * @child: the dentry to be removed + * start_creating_dentry - prepare to access or create a given dentry + * @parent: directory of dentry + * @child: the dentry to be prepared * - * A lock is taken to protect the dentry again other dirops and + * A lock is taken to protect the dentry against other dirops and * the validity of the dentry is checked: correct parent and still hashed. * * If the dentry is valid and negative a reference is taken and @@ -3580,7 +3580,7 @@ EXPORT_SYMBOL(start_creating_dentry); * @parent: directory from which dentry should be removed * @child: the dentry to be removed * - * A lock is taken to protect the dentry again other dirops and + * A lock is taken to protect the dentry against other dirops and * the validity of the dentry is checked: correct parent and still hashed. * * If the dentry is valid and positive, a reference is taken and @@ -3642,7 +3642,7 @@ int user_path_at(int dfd, const char __user *name, unsigned flags, } EXPORT_SYMBOL(user_path_at); -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { kuid_t fsuid = current_fsuid(); @@ -3675,7 +3675,7 @@ EXPORT_SYMBOL(__check_sticky); * 11. We don't allow removal of NFS sillyrenamed files; it's handled by * nfs_async_unlink(). */ -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir) { struct inode *inode = d_backing_inode(victim); @@ -3728,7 +3728,7 @@ EXPORT_SYMBOL(may_delete_dentry); * 4. We should have write and exec permissions on dir * 5. We can't do it if dir is immutable (done in permission()) */ -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child) { audit_inode_child(dir, child, AUDIT_TYPE_CHILD_CREATE); @@ -4142,7 +4142,7 @@ EXPORT_SYMBOL(end_renaming); * * Returns: mode to be passed to the filesystem */ -static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap, +static inline umode_t vfs_prepare_mode(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode, umode_t mask_perms, umode_t type) { @@ -4174,7 +4174,7 @@ static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode, +int vfs_create(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode, struct delegated_inode *di) { struct inode *dir = d_inode(dentry->d_parent); @@ -4228,7 +4228,7 @@ bool may_open_dev(const struct path *path) !(path->mnt->mnt_sb->s_iflags & SB_I_NODEV); } -static int may_open(struct mnt_idmap *idmap, const struct path *path, +static int may_open(const struct mnt_idmap *idmap, const struct path *path, int acc_mode, int flag) { struct dentry *dentry = path->dentry; @@ -4287,7 +4287,7 @@ static int may_open(struct mnt_idmap *idmap, const struct path *path, return 0; } -static int handle_truncate(struct mnt_idmap *idmap, struct file *filp) +static int handle_truncate(const struct mnt_idmap *idmap, struct file *filp) { const struct path *path = &filp->f_path; struct inode *inode = path->dentry->d_inode; @@ -4312,7 +4312,7 @@ static inline int open_to_namei_flags(int flag) return flag; } -static int may_o_create(struct mnt_idmap *idmap, +static int may_o_create(const struct mnt_idmap *idmap, const struct path *dir, struct dentry *dentry, umode_t mode) { @@ -4432,7 +4432,7 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, const struct open_flags *op) { struct delegated_inode delegated_inode = { }; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *dir = nd->path.dentry; struct inode *dir_inode = dir->d_inode; int open_flag; @@ -4440,12 +4440,14 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, int error, create_error; umode_t mode; bool got_write; + unsigned int shared_flag; retry: open_flag = op->open_flag; got_write = false; mode = op->mode; create_error = 0; + shared_flag = (open_flag & O_CREAT) ? 0 : LOOKUP_SHARED; if (open_flag & (O_CREAT | O_TRUNC | O_WRONLY | O_RDWR)) { got_write = !mnt_want_write(nd->path.mnt); @@ -4454,10 +4456,10 @@ retry: * a different error; we'll be dropping this one anyway. */ } - if (open_flag & O_CREAT) - inode_lock(dir_inode); - else + if (shared_flag) inode_lock_shared(dir_inode); + else + inode_lock(dir_inode); if (unlikely(IS_DEADDIR(dir_inode))) { dentry = ERR_PTR(-ENOENT); @@ -4526,7 +4528,7 @@ retry: if (d_in_lookup(dentry)) { struct dentry *res = dir_inode->i_op->lookup(dir_inode, dentry, - nd->flags); + nd->flags | shared_flag); d_lookup_done(dentry); if (unlikely(res)) { if (IS_ERR(res)) { @@ -4574,10 +4576,10 @@ out: if (file->f_mode & FMODE_OPENED) fsnotify_open(file); } - if ((open_flag & O_CREAT) || create_error) - inode_unlock(dir_inode); - else + if (shared_flag) inode_unlock_shared(dir_inode); + else + inode_unlock(dir_inode); if (got_write) mnt_drop_write(nd->path.mnt); @@ -4789,7 +4791,8 @@ finish_lookup: static int do_open(struct nameidata *nd, struct file *file, const struct open_flags *op) { - struct mnt_idmap *idmap; + struct vfsmount *mnt; + const struct mnt_idmap *idmap; int open_flag = op->open_flag; bool do_truncate; int acc_mode; @@ -4830,11 +4833,17 @@ static int do_open(struct nameidata *nd, error = mnt_want_write(nd->path.mnt); if (error) return error; + /* + * A dedicated reference is needed because after the call to + * vfs_open_consume() we no longer own the reference in nd->path.mnt + * while we need to undo write acess below. + */ + mnt = mntget(nd->path.mnt); do_truncate = true; } error = may_open(idmap, &nd->path, acc_mode, open_flag); if (!error && !(file->f_mode & FMODE_OPENED)) - error = vfs_open(&nd->path, file); + error = vfs_open_consume(&nd->path, file); if (!error) error = security_file_post_open(file, op->acc_mode); if (!error && do_truncate) @@ -4843,8 +4852,10 @@ static int do_open(struct nameidata *nd, WARN_ON(1); error = -EINVAL; } - if (do_truncate) - mnt_drop_write(nd->path.mnt); + if (do_truncate) { + mnt_drop_write(mnt); + mntput(mnt); + } return error; } @@ -4863,7 +4874,7 @@ static int do_open(struct nameidata *nd, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_tmpfile(struct mnt_idmap *idmap, +int vfs_tmpfile(const struct mnt_idmap *idmap, const struct path *parentpath, struct file *file, umode_t mode) { @@ -4921,7 +4932,7 @@ int vfs_tmpfile(struct mnt_idmap *idmap, * hence this is only for kernel internal use, and must not be installed into * file tables or such. */ -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred) @@ -5169,7 +5180,7 @@ struct file *dentry_create(struct path *path, int flags, umode_t mode, struct dentry *orig_dentry = dentry; struct dentry *dir = dentry->d_parent; struct inode *dir_inode = d_inode(dir); - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; int error, create_error; file = alloc_empty_file(flags, cred); @@ -5238,7 +5249,7 @@ EXPORT_SYMBOL(dentry_create); * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +int vfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev, struct delegated_inode *delegated_inode) { @@ -5296,7 +5307,7 @@ int filename_mknodat(int dfd, struct filename *name, umode_t mode, unsigned int dev) { struct delegated_inode di = { }; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *dentry; struct path path; int error; @@ -5380,7 +5391,7 @@ SYSCALL_DEFINE3(mknod, const char __user *, filename, umode_t, mode, unsigned, d * * In case of an error the dentry is dput() and an ERR_PTR() is returned. */ -struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +struct dentry *vfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, struct delegated_inode *delegated_inode) { @@ -5487,7 +5498,7 @@ SYSCALL_DEFINE2(mkdir, const char __user *, pathname, umode_t, mode) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_rmdir(struct mnt_idmap *idmap, struct inode *dir, +int vfs_rmdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct delegated_inode *delegated_inode) { int error = may_delete_dentry(idmap, dir, dentry, true); @@ -5622,7 +5633,7 @@ SYSCALL_DEFINE1(rmdir, const char __user *, pathname) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_unlink(struct mnt_idmap *idmap, struct inode *dir, +int vfs_unlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct delegated_inode *delegated_inode) { struct inode *target = dentry->d_inode; @@ -5772,7 +5783,7 @@ SYSCALL_DEFINE1(unlink, const char __user *, pathname) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int vfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *oldname, struct delegated_inode *delegated_inode) { @@ -5874,7 +5885,7 @@ SYSCALL_DEFINE2(symlink, const char __user *, oldname, const char __user *, newn * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_link(struct dentry *old_dentry, struct mnt_idmap *idmap, +int vfs_link(struct dentry *old_dentry, const struct mnt_idmap *idmap, struct inode *dir, struct dentry *new_dentry, struct delegated_inode *delegated_inode) { @@ -5951,7 +5962,7 @@ EXPORT_SYMBOL(vfs_link); int filename_linkat(int olddfd, struct filename *old, int newdfd, struct filename *new, int flags) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *new_dentry; struct path old_path, new_path; struct delegated_inode delegated_inode = { }; diff --git a/fs/namespace.c b/fs/namespace.c index 580877e46b1a..973efee4b968 100644 --- a/fs/namespace.c +++ b/fs/namespace.c @@ -109,7 +109,7 @@ struct mount_kattr { unsigned int lookup_flags; enum mount_kattr_flags_t kflags; struct user_namespace *mnt_userns; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; }; /* /sys/fs */ @@ -249,16 +249,24 @@ void mnt_release_group_id(struct mount *mnt) mnt->mnt_group_id = 0; } -/* - * vfsmount lock must be held for read - */ -static inline void mnt_add_count(struct mount *mnt, int n) +static inline void mnt_inc_count(struct mount *mnt) { #ifdef CONFIG_SMP - this_cpu_add(mnt->mnt_pcp->mnt_count, n); + this_cpu_inc(mnt->mnt_pcp->mnt_gets); #else preempt_disable(); - mnt->mnt_count += n; + mnt->mnt_count++; + preempt_enable(); +#endif +} + +static inline void mnt_dec_count(struct mount *mnt) +{ +#ifdef CONFIG_SMP + this_cpu_inc(mnt->mnt_pcp->mnt_puts); +#else + preempt_disable(); + mnt->mnt_count--; preempt_enable(); #endif } @@ -269,14 +277,17 @@ static inline void mnt_add_count(struct mount *mnt, int n) int mnt_get_count(struct mount *mnt) { #ifdef CONFIG_SMP - int count = 0; + unsigned int gets = 0, puts = 0; int cpu; - for_each_possible_cpu(cpu) { - count += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_count; - } + /* puts first, so a put counted here has its get counted below */ + for_each_possible_cpu(cpu) + puts += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_puts; + smp_mb(); /* pairs with the smp_wmb() in mntput_no_expire() */ + for_each_possible_cpu(cpu) + gets += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_gets; - return count; + return gets - puts; #else return mnt->mnt_count; #endif @@ -305,7 +316,7 @@ static struct mount *alloc_vfsmnt(const char *name) if (!mnt->mnt_pcp) goto out_free_devname; - this_cpu_add(mnt->mnt_pcp->mnt_count, 1); + this_cpu_inc(mnt->mnt_pcp->mnt_gets); #else mnt->mnt_count = 1; mnt->mnt_writers = 0; @@ -746,13 +757,13 @@ int __legitimize_mnt(struct vfsmount *bastard, unsigned seq) if (bastard == NULL) return 0; mnt = real_mount(bastard); - mnt_add_count(mnt, 1); - smp_mb(); // see mntput_no_expire() and do_umount() + mnt_inc_count(mnt); + smp_mb(); /* see mntput_no_expire_slowpath() and do_umount() */ if (likely(!read_seqretry(&mount_lock, seq))) return 0; lock_mount_hash(); if (unlikely(bastard->mnt_flags & (MNT_SYNC_UMOUNT | MNT_DOOMED))) { - mnt_add_count(mnt, -1); + mnt_dec_count(mnt); unlock_mount_hash(); return 1; } @@ -1254,6 +1265,7 @@ static struct mount *clone_mnt(struct mount *old, struct dentry *root, mnt->mnt.mnt_flags = READ_ONCE(old->mnt.mnt_flags) & ~MNT_INTERNAL_FLAGS; + mnt->mnt_t_flags = old->mnt_t_flags & T_UNBINDABLE; if (flag & (CL_SLAVE | CL_PRIVATE)) mnt->mnt_group_id = 0; /* not a peer of original */ @@ -1347,7 +1359,7 @@ static void noinline mntput_no_expire_slowpath(struct mount *mnt) * mount_lock, we'll see their refcount increment here. */ smp_mb(); - mnt_add_count(mnt, -1); + mnt_dec_count(mnt); count = mnt_get_count(mnt); if (count != 0) { WARN_ON(count < 0); @@ -1404,7 +1416,8 @@ static void mntput_no_expire(struct mount *mnt) * non-NULL under rcu_read_lock(), the reference * we are dropping is not the final one. */ - mnt_add_count(mnt, -1); + smp_wmb(); /* pairs with the smp_mb() in mnt_get_count() */ + mnt_dec_count(mnt); rcu_read_unlock(); return; } @@ -1426,7 +1439,7 @@ EXPORT_SYMBOL(mntput); struct vfsmount *mntget(struct vfsmount *mnt) { if (mnt) - mnt_add_count(real_mount(mnt), 1); + mnt_inc_count(real_mount(mnt)); return mnt; } EXPORT_SYMBOL(mntget); @@ -3467,7 +3480,7 @@ static int do_set_group(const struct path *from_path, const struct path *to_path return -EINVAL; /* Setting sharing groups is only allowed on private mounts */ - if (IS_MNT_SHARED(to) || IS_MNT_SLAVE(to)) + if (IS_MNT_SHARED(to) || IS_MNT_SLAVE(to) || IS_MNT_UNBINDABLE(to)) return -EINVAL; /* From should not be private */ @@ -4115,7 +4128,7 @@ int path_mount(const char *dev_name, const struct path *path, if (flags & SB_MANDLOCK) warn_mandlock(); - /* Default to relatime unless overriden */ + /* Default to relatime unless overridden */ if (!(flags & MS_NOATIME)) mnt_flags |= MNT_RELATIME; @@ -4247,8 +4260,6 @@ struct mnt_namespace *copy_mnt_ns(u64 flags, struct mnt_namespace *ns, struct mount *new; int copy_flags; - BUG_ON(!ns); - if (likely(!(flags & CLONE_NEWNS))) { get_mnt_ns(ns); return ns; @@ -4544,16 +4555,16 @@ SYSCALL_DEFINE3(fsmount, int, fs_fd, unsigned int, flags, FD_PREPARE(fdf, (flags & FSMOUNT_CLOEXEC) ? O_CLOEXEC : 0, dentry_open(&new_path, O_PATH, fc->cred)); - if (fdf.err) { + if (fdf->fd < 0) { dissolve_on_fput(new_path.mnt); - return fdf.err; + return fdf->fd; } /* * Attach to an apparent O_PATH fd with a note that we * need to unmount it, not just simply put it. */ - fd_prepare_file(fdf)->f_mode |= FMODE_NEED_UNMOUNT; + fdf->file->f_mode |= FMODE_NEED_UNMOUNT; return fd_publish(fdf); } @@ -4898,7 +4909,7 @@ static int mount_setattr_prepare(struct mount_kattr *kattr, struct mount *mnt) static void do_idmap_mount(const struct mount_kattr *kattr, struct mount *mnt) { - struct mnt_idmap *old_idmap; + const struct mnt_idmap *old_idmap; if (!kattr->mnt_idmap) return; @@ -4941,7 +4952,7 @@ static int do_mount_setattr(const struct path *path, struct mount_kattr *kattr) return -EINVAL; if (kattr->mnt_userns) { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; mnt_idmap = alloc_mnt_idmap(kattr->mnt_userns); if (IS_ERR(mnt_idmap)) @@ -5198,12 +5209,12 @@ SYSCALL_DEFINE5(open_tree_attr, int, dfd, const char __user *, filename, return -EINVAL; FD_PREPARE(fdf, flags, vfs_open_tree(dfd, filename, flags)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; if (uattr) { struct mount_kattr kattr = {}; - struct file *file = fd_prepare_file(fdf); + struct file *file = fdf->file; int ret; if (flags & OPEN_TREE_CLONE) @@ -5246,7 +5257,7 @@ struct kstatmount { struct statmount __user *buf; size_t bufsize; struct vfsmount *mnt; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; u64 mask; struct path root; struct seq_file seq; diff --git a/fs/netfs/Kconfig b/fs/netfs/Kconfig index 7701c037c328..d0e7b0971fa3 100644 --- a/fs/netfs/Kconfig +++ b/fs/netfs/Kconfig @@ -22,6 +22,9 @@ config NETFS_STATS between CPUs. On the other hand, the stats are very useful for debugging purposes. Saying 'Y' here is recommended. +config NETFS_PGPRIV2 + bool + config NETFS_DEBUG bool "Enable dynamic debugging netfslib and FS-Cache" depends on NETFS_SUPPORT diff --git a/fs/netfs/Makefile b/fs/netfs/Makefile index b43188d64bd8..54834cde7e56 100644 --- a/fs/netfs/Makefile +++ b/fs/netfs/Makefile @@ -11,7 +11,6 @@ netfs-y := \ misc.o \ objects.o \ read_collect.o \ - read_pgpriv2.o \ read_retry.o \ read_single.o \ rolling_buffer.o \ @@ -19,6 +18,7 @@ netfs-y := \ write_issue.o \ write_retry.o +netfs-$(CONFIG_NETFS_PGPRIV2) += read_pgpriv2.o netfs-$(CONFIG_NETFS_STATS) += stats.o netfs-$(CONFIG_FSCACHE) += \ diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 105194de6e13..e30bde80276a 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -10,9 +10,9 @@ #include "internal.h" static void netfs_cache_expand_readahead(struct netfs_io_request *rreq, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size) + uoff_t *_start, + uoff_t *_len, + uoff_t i_size) { struct netfs_cache_resources *cres = &rreq->cache_resources; @@ -137,21 +137,6 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq) return subreq->len; } -static enum netfs_io_source netfs_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq, - loff_t i_size) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - enum netfs_io_source source; - - if (!cres->ops) - return NETFS_DOWNLOAD_FROM_SERVER; - source = cres->ops->prepare_read(subreq, i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - return source; - -} - /* * Issue a read against the cache. * - Eats the caller's ref on subreq. @@ -166,6 +151,19 @@ static void netfs_read_cache_to_pagecache(struct netfs_io_request *rreq, netfs_cache_read_terminated, subreq); } +int netfs_read_query_cache(struct netfs_io_request *rreq, struct fscache_occupancy *occ) +{ + struct netfs_cache_resources *cres = &rreq->cache_resources; + + occ->granularity = PAGE_SIZE; + if (occ->query_from >= occ->query_to) + return 0; + if (!cres->ops) + return 0; + occ->query_from = round_up(occ->query_from, occ->granularity); + return cres->ops->query_occupancy(cres, occ); +} + void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -242,7 +240,7 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, if (overlap > 0 && copy) { folio = folioq_folio(*fq, *slot); - if (unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags))) { + if (netfs_using_pgpriv2(rreq)) { if (!folio_test_private_2(folio)) folio_start_private_2(folio); } else { @@ -268,18 +266,113 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, */ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) { + struct fscache_occupancy _occ = { + .query_from = rreq->start, + .query_to = rreq->start + rreq->len, + .cached_from[0] = 0, + .cached_to[0] = 0, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; + struct fscache_occupancy *occ = &_occ; struct folio_queue *fq = rreq->buffer.tail; - unsigned long long start = rreq->start; unsigned int offset = 0; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret = 0, slot = 0; do { + int (*prepare_read)(struct netfs_io_subrequest *subreq) = NULL; struct netfs_io_subrequest *subreq; - enum netfs_io_source source = NETFS_SOURCE_UNKNOWN; + enum netfs_io_source source; ssize_t slice; + uoff_t hole_to, cache_to; + size_t len = size; + bool copy = false; + + /* If we don't have any, find out the next couple of data + * extents from the cache, containing of following the + * specified start offset. Holes have to be fetched from the + * server; data regions from the cache. + */ + hole_to = occ->cached_from[0]; + cache_to = occ->cached_to[0]; + if (start >= cache_to) { + /* Extent exhausted; shuffle down. */ + int i; + + for (i = 0; i < ARRAY_SIZE(occ->cached_from) - 1; i++) { + occ->cached_from[i] = occ->cached_from[i + 1]; + occ->cached_to[i] = occ->cached_to[i + 1]; + occ->cached_type[i] = occ->cached_type[i + 1]; + } + occ->cached_from[i] = ULLONG_MAX; + occ->cached_to[i] = ULLONG_MAX; + + if (occ->cached_from[0] != ULLONG_MAX) + continue; + + /* Get new extents */ + ret = netfs_read_query_cache(rreq, occ); + if (ret < 0) + break; + continue; + } - subreq = netfs_alloc_subrequest(rreq); + uoff_t zero_point = netfs_read_zero_point(rreq->inode); + uoff_t zlimit = umin(zero_point, rreq->i_size); + + _debug("rsub %llx %llx-%llx", start, hole_to, cache_to); + + if (start >= hole_to && start < cache_to) { + /* Overlap with a cached region, where the cache may + * record a block of zeroes. + */ + _debug("cached s=%llx c=%llx l=%zx", start, cache_to, size); + len = umin(cache_to - start, size); + len = round_up(len, occ->granularity); + if (occ->cached_type[0] == FSCACHE_EXTENT_ZERO) { + source = NETFS_FILL_WITH_ZEROES; + netfs_stat(&netfs_n_rh_zero); + } else { + source = NETFS_READ_FROM_CACHE; + prepare_read = rreq->cache_resources.ops->prepare_read; + } + } else if (start >= zlimit && size > 0) { + /* If this range lies beyond the zero-point, that part + * can just be cleared locally. + */ + _debug("zero %llx-%llx", start, start + size); + len = size; + source = NETFS_FILL_WITH_ZEROES; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_zero); + } else { + /* Read a cache hole from the server. If any part of + * this range lies beyond the zero-point or the EOF, + * that part can just be cleared locally. + */ + uoff_t limit = min3(zlimit, start + size, hole_to); + + _debug("limit %llx %llx", rreq->i_size, zero_point); + _debug("download %llx-%llx", start, start + size); + len = umin(limit - start, ULONG_MAX); + source = NETFS_DOWNLOAD_FROM_SERVER; + prepare_read = rreq->netfs_ops->prepare_read; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_download); + } + + if (len == 0) { + pr_err("ZERO-LEN READ: R=%08x l=%zx/%zx s=%llx z=%llx i=%llx", + rreq->debug_id, len, size, + start, zero_point, rreq->i_size); + break; + } + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) { ret = -ENOMEM; break; @@ -287,66 +380,23 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) subreq->start = start; subreq->len = size; + if (copy) + __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_queue_read(rreq, subreq); - source = netfs_cache_prepare_read(rreq, subreq, rreq->i_size); - subreq->source = source; - if (source == NETFS_DOWNLOAD_FROM_SERVER) { - unsigned long long zero_point = netfs_read_zero_point(rreq->inode); - unsigned long long zp = umin(zero_point, rreq->i_size); - size_t len = subreq->len; - - if (unlikely(rreq->origin == NETFS_READ_SINGLE)) - zp = rreq->i_size; - if (subreq->start >= zp) { - subreq->source = source = NETFS_FILL_WITH_ZEROES; - goto fill_with_zeroes; - } + rreq->io_streams[0].sreq_max_len = MAX_RW_COUNT; + rreq->io_streams[0].sreq_max_segs = INT_MAX; - if (len > zp - subreq->start) - len = zp - subreq->start; - if (len == 0) { - pr_err("ZERO-LEN READ: R=%08x[%x] l=%zx/%zx s=%llx z=%llx i=%llx", - rreq->debug_id, subreq->debug_index, - subreq->len, size, - subreq->start, zero_point, rreq->i_size); + if (prepare_read) { + ret = prepare_read(subreq); + if (ret < 0) { netfs_cancel_read(subreq, ret); break; } - subreq->len = len; - - netfs_stat(&netfs_n_rh_download); - if (rreq->netfs_ops->prepare_read) { - ret = rreq->netfs_ops->prepare_read(subreq); - if (ret < 0) { - netfs_cancel_read(subreq, ret); - break; - } - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - } - goto issue; - } - - fill_with_zeroes: - if (source == NETFS_FILL_WITH_ZEROES) { - subreq->source = NETFS_FILL_WITH_ZEROES; - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - netfs_stat(&netfs_n_rh_zero); - goto issue; - } - - if (source == NETFS_READ_FROM_CACHE) { - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - goto issue; + trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); } - pr_err("Unexpected read source %u\n", source); - WARN_ON_ONCE(1); - netfs_cancel_read(subreq, ret); - break; - - issue: slice = netfs_prepare_read_iterator(subreq); if (slice < 0) { ret = slice; @@ -355,18 +405,16 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } start += slice; size -= slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); if (fq) { /* See if the cache indicated this should be cached. */ - bool copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); - + copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_mark_copy_to_cache(rreq, &fq, &slot, &offset, slice, copy); } + trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_issue_read(rreq, subreq); netfs_maybe_bulk_drop_ra_refs(rreq); @@ -378,8 +426,7 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } @@ -646,11 +693,11 @@ EXPORT_SYMBOL(netfs_read_folio); * If any of these criteria are met, then zero out the unwritten parts * of the folio and return true. Otherwise, return false. */ -static bool netfs_skip_folio_read(struct folio *folio, loff_t pos, size_t len, +static bool netfs_skip_folio_read(struct folio *folio, uoff_t pos, size_t len, bool always_fill) { struct inode *inode = folio_inode(folio); - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); size_t offset = offset_in_folio(folio, pos); size_t plen = folio_size(folio); @@ -715,7 +762,7 @@ zero_out: */ int netfs_write_begin(struct netfs_inode *ctx, struct file *file, struct address_space *mapping, - loff_t pos, unsigned int len, struct folio **_folio, + uoff_t pos, unsigned int len, struct folio **_folio, void **_fsdata) { struct netfs_io_request *rreq; @@ -811,7 +858,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, struct netfs_io_request *rreq; struct address_space *mapping = folio->mapping; struct netfs_inode *ctx = netfs_inode(mapping->host); - unsigned long long start = folio_pos(folio); + uoff_t start = folio_pos(folio); size_t flen = folio_size(folio); int ret; diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index 2cdb68e6b16f..49b47252f675 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -17,7 +17,7 @@ * as possible to hold as much of the remaining length as possible in one go. */ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, - loff_t pos, size_t part) + uoff_t pos, size_t part) { pgoff_t index = pos / PAGE_SIZE; fgf_t fgp_flags = FGP_WRITEBEGIN; @@ -35,9 +35,9 @@ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, * the values actually are. */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied) + uoff_t pos, size_t copied) { - loff_t i_size, end = pos + copied; + uoff_t i_size, end = pos + copied; blkcnt_t add; size_t gap; @@ -54,9 +54,6 @@ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, i_size = i_size_read(inode); if (end > i_size) { i_size_write(inode, end); -#if IS_ENABLED(CONFIG_FSCACHE) - fscache_update_cookie(ctx->cache, NULL, &end); -#endif gap = SECTOR_SIZE - (i_size & (SECTOR_SIZE - 1)); if (copied > gap) { @@ -91,50 +88,20 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, struct inode *inode = file_inode(file); struct address_space *mapping = inode->i_mapping; struct netfs_inode *ctx = netfs_inode(inode); - struct writeback_control wbc = { - .sync_mode = WB_SYNC_NONE, - .for_sync = true, - .nr_to_write = LONG_MAX, - .range_start = iocb->ki_pos, - .range_end = iocb->ki_pos + iter->count, - }; - struct netfs_io_request *wreq = NULL; - struct folio *folio = NULL, *writethrough = NULL; + struct folio *folio = NULL; unsigned int bdp_flags = (iocb->ki_flags & IOCB_NOWAIT) ? BDP_ASYNC : 0; - ssize_t written = 0, ret, ret2; - loff_t pos = iocb->ki_pos; + ssize_t written = 0, ret; + uoff_t pos = iocb->ki_pos; size_t max_chunk = mapping_max_folio_size(mapping); bool maybe_trouble = false; - if (unlikely(iocb->ki_flags & (IOCB_DSYNC | IOCB_SYNC)) - ) { - wbc_attach_fdatawrite_inode(&wbc, mapping->host); - - ret = filemap_write_and_wait_range(mapping, pos, pos + iter->count); - if (ret < 0) { - wbc_detach_inode(&wbc); - goto out; - } - - wreq = netfs_begin_writethrough(iocb, iter->count); - if (IS_ERR(wreq)) { - wbc_detach_inode(&wbc); - ret = PTR_ERR(wreq); - wreq = NULL; - goto out; - } - if (!is_sync_kiocb(iocb)) - wreq->iocb = iocb; - netfs_stat(&netfs_n_wh_writethrough); - } else { - netfs_stat(&netfs_n_wh_buffered_write); - } + netfs_stat(&netfs_n_wh_buffered_write); do { enum netfs_folio_trace trace; struct netfs_folio *finfo; struct netfs_group *group; - unsigned long long fpos; + uoff_t fpos; size_t flen; size_t offset; /* Offset into pagecache folio */ size_t part; /* Bytes to write to folio */ @@ -390,15 +357,8 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, pos += copied; written += copied; - if (likely(!wreq)) { - folio_mark_dirty(folio); - folio_unlock(folio); - } else { - netfs_advance_writethrough(wreq, &wbc, folio, copied, - offset + copied == flen, - &writethrough); - /* Folio unlocked */ - } + folio_mark_dirty(folio); + folio_unlock(folio); retry: folio_put(folio); folio = NULL; @@ -420,15 +380,6 @@ out: ctx->ops->post_modify(inode); } - if (unlikely(wreq)) { - ret2 = netfs_end_writethrough(wreq, &wbc, writethrough); - wbc_detach_inode(&wbc); - if (ret2 == -EIOCBQUEUED) - return ret2; - if (ret == 0 && ret2 < 0) - ret = ret2; - } - iocb->ki_pos += written; _leave(" = %zd [%zd]", written, ret); return written ? written : ret; diff --git a/fs/netfs/direct_read.c b/fs/netfs/direct_read.c index 6a8fb0d55e04..8c15f3079723 100644 --- a/fs/netfs/direct_read.c +++ b/fs/netfs/direct_read.c @@ -47,15 +47,15 @@ static void netfs_prepare_dio_read_iterator(struct netfs_io_subrequest *subreq) */ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) { - unsigned long long start = rreq->start; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret; do { struct netfs_io_subrequest *subreq; ssize_t slice; - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { /* Stash the error in the request if there's not * already an error set. @@ -64,7 +64,6 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) break; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = size; @@ -84,10 +83,8 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) size -= slice; start += slice; rreq->submitted += slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); @@ -99,8 +96,7 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } } diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c index 2361277416c7..32200c10d2a4 100644 --- a/fs/netfs/direct_write.c +++ b/fs/netfs/direct_write.c @@ -225,9 +225,9 @@ ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter * struct netfs_group *netfs_group) { struct netfs_io_request *wreq; - unsigned long long start = iocb->ki_pos; - unsigned long long end = start + iov_iter_count(iter); ssize_t ret, n; + uoff_t start = iocb->ki_pos; + uoff_t end = start + iov_iter_count(iter); size_t len = iov_iter_count(iter); bool async = !is_sync_kiocb(iocb); @@ -336,8 +336,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from) struct inode *inode = mapping->host; struct netfs_inode *ictx = netfs_inode(inode); ssize_t ret; - loff_t pos = iocb->ki_pos; - unsigned long long end = pos + iov_iter_count(from) - 1; + uoff_t pos = iocb->ki_pos; + uoff_t end = pos + iov_iter_count(from) - 1; _enter("%llx,%zx,%llx", pos, iov_iter_count(from), i_size_read(inode)); diff --git a/fs/netfs/fscache_cookie.c b/fs/netfs/fscache_cookie.c index 3d56fc73435f..5a226f9cbdea 100644 --- a/fs/netfs/fscache_cookie.c +++ b/fs/netfs/fscache_cookie.c @@ -327,7 +327,7 @@ static struct fscache_cookie *fscache_alloc_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -452,7 +452,7 @@ struct fscache_cookie *__fscache_acquire_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -663,7 +663,7 @@ static void fscache_unuse_cookie_locked(struct fscache_cookie *cookie) * Stop using the cookie for I/O. */ void __fscache_unuse_cookie(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { unsigned int debug_id = cookie->debug_id; unsigned int r = refcount_read(&cookie->ref); @@ -1049,7 +1049,7 @@ static void fscache_perform_invalidation(struct fscache_cookie *cookie) * Invalidate an object. */ void __fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t new_size, + const void *aux_data, uoff_t new_size, unsigned int flags) { bool is_caching; diff --git a/fs/netfs/fscache_internal.h b/fs/netfs/fscache_internal.h deleted file mode 100644 index a09b948fcef2..000000000000 --- a/fs/netfs/fscache_internal.h +++ /dev/null @@ -1,14 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-or-later */ -/* Internal definitions for FS-Cache - * - * Copyright (C) 2021 Red Hat, Inc. All Rights Reserved. - * Written by David Howells (dhowells@redhat.com) - */ - -#include "internal.h" - -#ifdef pr_fmt -#undef pr_fmt -#endif - -#define pr_fmt(fmt) "FS-Cache: " fmt diff --git a/fs/netfs/fscache_io.c b/fs/netfs/fscache_io.c index 37f05b4d3469..056a2bae5d99 100644 --- a/fs/netfs/fscache_io.c +++ b/fs/netfs/fscache_io.c @@ -79,7 +79,7 @@ static int fscache_begin_operation(struct netfs_cache_resources *cres, cres->ops = NULL; cres->cache_priv = cookie; cres->cache_priv2 = NULL; - cres->debug_id = cookie->debug_id; + cres->cookie_id = cookie->debug_id; cres->inval_counter = cookie->inval_counter; if (!fscache_begin_cookie_access(cookie, why)) { @@ -162,7 +162,7 @@ EXPORT_SYMBOL(__fscache_begin_write_operation); struct fscache_write_request { struct netfs_cache_resources cache_resources; struct address_space *mapping; - loff_t start; + uoff_t start; size_t len; bool set_bits; bool using_pgpriv2; @@ -171,7 +171,7 @@ struct fscache_write_request { }; void __fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len) + uoff_t start, size_t len) { pgoff_t first = start / PAGE_SIZE; pgoff_t last = (start + len - 1) / PAGE_SIZE; @@ -208,7 +208,7 @@ static void fscache_wreq_done(void *priv, ssize_t transferred_or_error) void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond) @@ -267,7 +267,7 @@ EXPORT_SYMBOL(__fscache_write_to_cache); /* * Change the size of a backing object. */ -void __fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void __fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { struct netfs_cache_resources cres; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index c79c8e69d60c..b8591abc90a9 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -23,6 +23,8 @@ /* * buffered_read.c */ +int netfs_read_query_cache(struct netfs_io_request *rreq, + struct fscache_occupancy *occ); void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq); void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); @@ -33,7 +35,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, * buffered_write.c */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied); + uoff_t pos, size_t copied); /* * main.c @@ -86,13 +88,14 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin); void netfs_get_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_clear_subrequests(struct netfs_io_request *rreq); void netfs_put_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_put_failed_request(struct netfs_io_request *rreq); -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq); +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source); static inline void netfs_see_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what) @@ -120,9 +123,37 @@ void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); /* * read_pgpriv2.c */ +#ifdef CONFIG_NETFS_PGPRIV2 +int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs); void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio); void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq); bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq); +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)); +} +#else +static inline int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs) +{ + return -EIO; +} +static inline void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) +{ +} +static inline void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) +{ +} +static inline bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq) +{ + return true; +} +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return false; +} +#endif /* * read_retry.c @@ -157,7 +188,6 @@ extern atomic_t netfs_n_rh_write_zskip; extern atomic_t netfs_n_rh_retry_read_req; extern atomic_t netfs_n_rh_retry_read_subreq; extern atomic_t netfs_n_wh_buffered_write; -extern atomic_t netfs_n_wh_writethrough; extern atomic_t netfs_n_wh_dio_write; extern atomic_t netfs_n_wh_writepages; extern atomic_t netfs_n_wh_copy_to_cache; @@ -203,11 +233,11 @@ void netfs_write_collection_worker(struct work_struct *work); */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin); void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start); + uoff_t start); void netfs_reissue_write(struct netfs_io_stream *stream, struct netfs_io_subrequest *subreq, struct iov_iter *source); @@ -215,13 +245,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream); size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof); -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len); -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache); -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache); + uoff_t start, size_t len, bool to_eof); /* * write_retry.c @@ -321,6 +345,26 @@ static inline bool netfs_check_subreq_in_progress(const struct netfs_io_subreque } /* + * Indicate that we've generated and queued all the subrequests we're going to. + */ +static inline void netfs_all_subreqs_queued(struct netfs_io_request *rreq) +{ + smp_wmb(); /* Write lists before ALL_QUEUED. */ + set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + smp_mb__after_atomic(); + trace_netfs_rreq(rreq, netfs_rreq_trace_all_queued); +} + +/* + * Query if all subrequests are queued. + */ +static inline bool netfs_are_all_subreqs_queued(const struct netfs_io_request *rreq) +{ + /* Read lists after ALL_QUEUED. */ + return test_bit_acquire(NETFS_RREQ_ALL_QUEUED, &rreq->flags); +} + +/* * fscache-cache.c */ #ifdef CONFIG_PROC_FS diff --git a/fs/netfs/iterator.c b/fs/netfs/iterator.c index b375567e0520..eb1efb17f53a 100644 --- a/fs/netfs/iterator.c +++ b/fs/netfs/iterator.c @@ -209,7 +209,7 @@ static size_t netfs_limit_xarray(const struct iov_iter *iter, size_t start_offse { struct folio *folio; unsigned int nsegs = 0; - loff_t pos = iter->xarray_start + iter->iov_offset; + uoff_t pos = iter->xarray_start + iter->iov_offset; pgoff_t index = pos / PAGE_SIZE; size_t span = 0, n = iter->count; diff --git a/fs/netfs/main.c b/fs/netfs/main.c index 927badf3989d..609e22e8f76a 100644 --- a/fs/netfs/main.c +++ b/fs/netfs/main.c @@ -44,7 +44,6 @@ static const char *netfs_origins[nr__netfs_io_origin] = { [NETFS_DIO_READ] = "DR", [NETFS_WRITEBACK] = "WB", [NETFS_WRITEBACK_SINGLE] = "W1", - [NETFS_WRITETHROUGH] = "WT", [NETFS_UNBUFFERED_WRITE] = "UW", [NETFS_DIO_WRITE] = "DW", [NETFS_PGPRIV2_COPY_TO_CACHE] = "2C", diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c index f5c1c463f4ff..a3cd76d584b8 100644 --- a/fs/netfs/misc.c +++ b/fs/netfs/misc.c @@ -193,7 +193,7 @@ void netfs_clear_inode_writeback(struct inode *inode, const void *aux) struct fscache_cookie *cookie = netfs_i_cookie(netfs_inode(inode)); if (inode_state_read_once(inode) & I_PINNING_NETFS_WB) { - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); fscache_unuse_cookie(cookie, aux, &i_size); } } @@ -218,8 +218,8 @@ void netfs_invalidate_folio(struct folio *folio, size_t offset, size_t length) _enter("{%lx},%zx,%zx", folio->index, offset, length); if (offset == 0 && length == flen) { - unsigned long long i_size, remote_i_size, zero_point; - unsigned long long fpos = folio_pos(folio), end; + uoff_t i_size, remote_i_size, zero_point; + uoff_t fpos = folio_pos(folio), end; netfs_read_sizes(inode, &i_size, &remote_i_size, &zero_point); end = umin(fpos + flen, i_size); @@ -305,7 +305,7 @@ bool netfs_release_folio(struct folio *folio, gfp_t gfp) { struct inode *inode = folio_inode(folio); struct netfs_inode *ctx = netfs_inode(inode); - unsigned long long i_size, remote_i_size, zero_point, end; + uoff_t i_size, remote_i_size, zero_point, end; if (folio_test_dirty(folio)) return false; @@ -424,7 +424,7 @@ static int netfs_collect_in_app(struct netfs_io_request *rreq, need_collect = true; break; } - if (subreq || !test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (subreq || !netfs_are_all_subreqs_queued(rreq)) done = false; } diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index ad549daa9c79..4b8d20559b0e 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -16,7 +16,7 @@ static void netfs_free_request(struct work_struct *work); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin) { static atomic_t debug_ids; @@ -44,6 +44,7 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, rreq->gfp = gfp; rreq->start = start; rreq->collected_to = start; + rreq->cache_coll_to = start; rreq->cleaned_to = start; rreq->len = len; rreq->progress_at = 0; @@ -207,7 +208,8 @@ void netfs_put_failed_request(struct netfs_io_request *rreq) /* * Allocate and partially initialise an I/O request structure. */ -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq) +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source) { struct netfs_io_subrequest *subreq; mempool_t *mempool = rreq->netfs_ops->subrequest_pool ?: &netfs_subrequest_pool; @@ -224,6 +226,7 @@ struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq INIT_WORK(&subreq->work, NULL); INIT_LIST_HEAD(&subreq->rreq_link); refcount_set(&subreq->ref, 2); + subreq->source = source; subreq->rreq = rreq; subreq->debug_index = atomic_inc_return(&rreq->subreq_counter); netfs_get_request(rreq, netfs_rreq_trace_get_subreq); diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index a94197ef0181..2625efd48a9b 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -38,7 +38,7 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq) */ void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) { - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (folio_get_private(folio) == NETFS_FOLIO_COPY_TO_CACHE) { folio_detach_private(folio); trace_netfs_folio(folio, netfs_folio_trace_cancel_copy); @@ -81,7 +81,7 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq, if (unlikely(test_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags))) netfs_cancel_copy_to_cache(rreq, folio); - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) { trace_netfs_folio(folio, netfs_folio_trace_sched_copy); folio_mark_dirty(folio); @@ -153,8 +153,8 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, unsigned int *notes) { struct folio_queue *folioq = rreq->buffer.tail; - unsigned long long collected_to = rreq->collected_to; unsigned int slot = rreq->buffer.first_tail_slot; + uoff_t collected_to = rreq->collected_to; if (rreq->cleaned_to >= rreq->collected_to) return; @@ -179,7 +179,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize; folio = folioq_folio(folioq, slot); @@ -192,7 +192,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, fpos = folio_pos(folio); fend = fpos + fsize; - trace_netfs_collect_folio(rreq, folio, fend, collected_to); + trace_netfs_collect_folio(rreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) @@ -467,10 +467,8 @@ bool netfs_read_collection(struct netfs_io_request *rreq) /* We're done when the app thread has finished posting subreqs and the * queue is empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (!netfs_are_all_subreqs_queued(rreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before subreq lists. */ - if (!list_empty(&stream->subrequests)) return false; diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c index a4b7bb88cbdb..5280b606fda4 100644 --- a/fs/netfs/read_pgpriv2.c +++ b/fs/netfs/read_pgpriv2.c @@ -20,7 +20,7 @@ static void netfs_pgpriv2_copy_folio(struct netfs_io_request *creq, struct folio { struct netfs_io_stream *cache = &creq->io_streams[1]; size_t fsize = folio_size(folio), flen = fsize; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false; _enter(""); @@ -158,8 +158,7 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) return; netfs_issue_write(creq, &creq->io_streams[1]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &creq->flags); + netfs_all_subreqs_queued(creq); trace_netfs_rreq(rreq, netfs_rreq_trace_end_copy_to_cache); if (list_empty_careful(&creq->io_streams[1].subrequests)) netfs_wake_collector(creq); @@ -175,8 +174,8 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) { struct folio_queue *folioq = creq->buffer.tail; - unsigned long long collected_to = creq->collected_to; unsigned int slot = creq->buffer.first_tail_slot; + uoff_t collected_to = creq->collected_to; bool made_progress = false; if (slot >= folioq_nr_slots(folioq)) { @@ -186,7 +185,7 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -199,9 +198,9 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) fsize = folio_size(folio); flen = fsize; - fend = min_t(unsigned long long, fpos + flen, creq->i_size); + fend = min_t(uoff_t, fpos + flen, creq->i_size); - trace_netfs_collect_folio(creq, folio, fend, collected_to); + trace_netfs_collect_folio(creq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c index 4f6a36c6e214..5bd8dee5a834 100644 --- a/fs/netfs/read_retry.c +++ b/fs/netfs/read_retry.c @@ -75,7 +75,7 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) do { struct netfs_io_subrequest *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false, subreq_superfluous = false; @@ -195,12 +195,11 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) * and insert them after. */ do { - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { subreq = to; goto abandon_after; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = len; subreq->stream_nr = stream->stream_nr; @@ -272,6 +271,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) struct netfs_io_stream *stream = &rreq->io_streams[0]; netfs_stat(&netfs_n_rh_retry_read_req); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -282,6 +282,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) trace_netfs_rreq(rreq, netfs_rreq_trace_resubmit); netfs_retry_read_subrequests(rreq); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_end); } /* diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c index de67ac41548d..b248e34bd0c8 100644 --- a/fs/netfs/read_single.c +++ b/fs/netfs/read_single.c @@ -58,20 +58,6 @@ static int netfs_single_begin_cache_read(struct netfs_io_request *rreq, struct n return fscache_begin_read_operation(&rreq->cache_resources, netfs_i_cookie(ctx)); } -static void netfs_single_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - - if (!cres->ops) { - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; - return; - } - subreq->source = cres->ops->prepare_read(subreq, rreq->i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - -} - static void netfs_single_read_cache(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -89,21 +75,36 @@ static void netfs_single_read_cache(struct netfs_io_request *rreq, */ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) { + struct fscache_occupancy occ = { + .query_from = 0, + .query_to = rreq->len, + .cached_from[0] = ULLONG_MAX, + .cached_to[0] = ULLONG_MAX, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; struct netfs_io_subrequest *subreq; + enum netfs_io_source source = NETFS_DOWNLOAD_FROM_SERVER; int ret = 0; - subreq = netfs_alloc_subrequest(rreq); + /* Try to use the cache if the cache content matches the size of the + * remote file. + */ + netfs_read_query_cache(rreq, &occ); + if (occ.cached_from[0] == 0 && + occ.cached_to[0] >= rreq->len) + source = NETFS_READ_FROM_CACHE; + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) return -ENOMEM; - subreq->source = NETFS_SOURCE_UNKNOWN; subreq->start = 0; subreq->len = rreq->len; subreq->io_iter = rreq->buffer.iter; netfs_queue_read(rreq, subreq); - netfs_single_cache_prepare_read(rreq, subreq); switch (subreq->source) { case NETFS_DOWNLOAD_FROM_SERVER: netfs_stat(&netfs_n_rh_download); @@ -113,14 +114,18 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) goto cancel; } - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); rreq->submitted += subreq->len; break; case NETFS_READ_FROM_CACHE: - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + if (rreq->cache_resources.ops->prepare_read) { + ret = rreq->cache_resources.ops->prepare_read(subreq); + if (ret < 0) + goto cancel; + } + + netfs_all_subreqs_queued(rreq); trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_single_read_cache(rreq, subreq); rreq->submitted += subreq->len; @@ -136,8 +141,7 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) return ret; cancel: netfs_cancel_read(subreq, ret); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); return ret; } diff --git a/fs/netfs/stats.c b/fs/netfs/stats.c index ab6b916addc4..9a607c4e62dd 100644 --- a/fs/netfs/stats.c +++ b/fs/netfs/stats.c @@ -32,7 +32,6 @@ atomic_t netfs_n_rh_write_zskip; atomic_t netfs_n_rh_retry_read_req; atomic_t netfs_n_rh_retry_read_subreq; atomic_t netfs_n_wh_buffered_write; -atomic_t netfs_n_wh_writethrough; atomic_t netfs_n_wh_dio_write; atomic_t netfs_n_wh_writepages; atomic_t netfs_n_wh_copy_to_cache; @@ -58,9 +57,8 @@ int netfs_stats_show(struct seq_file *m, void *v) atomic_read(&netfs_n_rh_read_single), atomic_read(&netfs_n_rh_write_begin), atomic_read(&netfs_n_rh_write_zskip)); - seq_printf(m, "Writes : BW=%u WT=%u DW=%u WP=%u 2C=%u\n", + seq_printf(m, "Writes : BW=%u DW=%u WP=%u 2C=%u\n", atomic_read(&netfs_n_wh_buffered_write), - atomic_read(&netfs_n_wh_writethrough), atomic_read(&netfs_n_wh_dio_write), atomic_read(&netfs_n_wh_writepages), atomic_read(&netfs_n_wh_copy_to_cache)); diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 210eb8f3958d..6e8ea534230d 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -56,7 +56,7 @@ static void netfs_dump_request(const struct netfs_io_request *rreq) */ int netfs_folio_written_back(struct folio *folio) { - enum netfs_folio_trace why = netfs_folio_trace_clear; + enum netfs_folio_trace why = netfs_folio_trace_endwb; struct inode *inode = folio_inode(folio); struct netfs_inode *ictx = netfs_inode(inode); struct netfs_folio *finfo; @@ -67,7 +67,7 @@ int netfs_folio_written_back(struct folio *folio) /* Streaming writes cannot be redirtied whilst under writeback, * so discard the streaming record. */ - unsigned long long fend; + uoff_t fend; fend = folio_pos(folio) + finfo->dirty_offset + finfo->dirty_len; spin_lock(&ictx->inode.i_lock); @@ -79,13 +79,13 @@ int netfs_folio_written_back(struct folio *folio) group = finfo->netfs_group; gcount++; kfree(finfo); - why = netfs_folio_trace_clear_s; + why = netfs_folio_trace_endwb_s; goto end_wb; } if ((group = netfs_folio_group(folio))) { if (group == NETFS_FOLIO_COPY_TO_CACHE) { - why = netfs_folio_trace_clear_cc; + why = netfs_folio_trace_endwb_cc; folio_detach_private(folio); goto end_wb; } @@ -98,7 +98,7 @@ int netfs_folio_written_back(struct folio *folio) if (!folio_test_dirty(folio)) { folio_detach_private(folio); gcount++; - why = netfs_folio_trace_clear_g; + why = netfs_folio_trace_endwb_g; } } @@ -115,8 +115,8 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, unsigned int *notes) { struct folio_queue *folioq = wreq->buffer.tail; - unsigned long long collected_to = wreq->collected_to; unsigned int slot = wreq->buffer.first_tail_slot; + uoff_t collected_to = wreq->collected_to; if (WARN_ON_ONCE(!folioq)) { pr_err("[!] Writeback unlock found empty rolling buffer!\n"); @@ -140,7 +140,7 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, for (;;) { struct folio *folio; struct netfs_folio *finfo; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -154,9 +154,9 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, finfo = netfs_folio_info(folio); flen = finfo ? finfo->dirty_offset + finfo->dirty_len : fsize; - fend = min_t(unsigned long long, fpos + flen, wreq->i_size); + fend = min_t(uoff_t, fpos + flen, wreq->i_size); - trace_netfs_collect_folio(wreq, folio, fend, collected_to); + trace_netfs_collect_folio(wreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) @@ -189,6 +189,26 @@ done: } /* + * Collect cache results. + */ +static void netfs_cache_collect(struct netfs_io_request *wreq, + struct netfs_io_stream *stream, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + + if (stream->source != NETFS_WRITE_TO_CACHE || + wreq->cache_coll_to >= stream->collected_to) + return; + + if (cres->ops && cres->ops->collect_write) + cres->ops->collect_write(wreq, wreq->cache_coll_to, + stream->collected_to - wreq->cache_coll_to, + block_type); + wreq->cache_coll_to = stream->collected_to; +} + +/* * Collect and assess the results of various write subrequests. We may need to * retry some of the results - or even do an RMW cycle for content crypto. * @@ -201,8 +221,8 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) { struct netfs_io_subrequest *front, *remove; struct netfs_io_stream *stream; - unsigned long long collected_to, issued_to; unsigned int notes; + uoff_t collected_to, issued_to; int s; _enter("%llx-%llx", wreq->start, wreq->start + wreq->len); @@ -214,7 +234,6 @@ reassess_streams: smp_rmb(); collected_to = ULLONG_MAX; if (wreq->origin == NETFS_WRITEBACK || - wreq->origin == NETFS_WRITETHROUGH || wreq->origin == NETFS_PGPRIV2_COPY_TO_CACHE) notes = NEED_UNLOCK; else @@ -236,13 +255,19 @@ reassess_streams: /* Read first subreq pointer before IN_PROGRESS flag. */ while (front) { + enum netfs_cache_collect cache_collect; + trace_netfs_collect_sreq(wreq, front); //_debug("sreq [%x] %llx %zx/%zx", // front->debug_index, front->start, front->transferred, front->len); if (stream->collected_to < front->start) { trace_netfs_collect_gap(wreq, stream, issued_to, 'F'); + if (stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) + netfs_cache_collect(wreq, stream, stream->cache_collect); stream->collected_to = front->start; + netfs_cache_collect(wreq, stream, NETFS_CACHE_COLLECT_WRITE_GAP); + stream->cache_collect = NETFS_CACHE_COLLECT_WRITE_GAP; } /* Stall if the front is still undergoing I/O. */ @@ -250,7 +275,6 @@ reassess_streams: notes |= HIT_PENDING; break; } - smp_rmb(); /* Read counters after I-P flag. */ if (stream->failed) { stream->collected_to = front->start + front->len; @@ -263,15 +287,44 @@ reassess_streams: stream->transferred_valid = true; notes |= MADE_PROGRESS; } - if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { - stream->failed = true; - stream->error = front->error; - if (stream->source == NETFS_UPLOAD_TO_SERVER) - mapping_set_error(wreq->mapping, front->error); - notes |= NEED_REASSESS | SAW_FAILURE; + + /* Handle failed or cancelled subreqs. Failure of + * cache writes are handled differently to upload + * failures. Cache writes aren't fatal, provided we're + * not doing disconnected operation, and so we can kind + * of treat them as if they had succeeded - except that + * we need to log any holes they cause. + */ + switch (stream->source) { + case NETFS_UPLOAD_TO_SERVER: + if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { + if (!stream->failed) { + stream->failed = true; + stream->error = front->error; + mapping_set_error(wreq->mapping, front->error); + break; + } + notes |= NEED_REASSESS | SAW_FAILURE; + } + break; + + case NETFS_WRITE_TO_CACHE: + cache_collect = test_bit(NETFS_SREQ_CANCELLED, &front->flags) ? + NETFS_CACHE_COLLECT_WRITE_CANCEL : + NETFS_CACHE_COLLECT_WRITE_DATA; + if (cache_collect != stream->cache_collect && + stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) { + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_fail_collect); + netfs_cache_collect(wreq, stream, stream->cache_collect); + } + stream->cache_collect = cache_collect; + break; + + default: + WARN_ON(1); break; } - if (front->transferred < front->len) { + if (test_bit(NETFS_SREQ_NEED_RETRY, &front->flags)) { stream->need_retry = true; notes |= NEED_RETRY | MADE_PROGRESS; break; @@ -360,6 +413,7 @@ need_retry: */ bool netfs_write_collection(struct netfs_io_request *wreq) { + struct netfs_io_stream *cstream = &wreq->io_streams[1]; struct netfs_inode *ictx = netfs_inode(wreq->inode); size_t transferred; bool transferred_valid = false; @@ -372,9 +426,8 @@ bool netfs_write_collection(struct netfs_io_request *wreq) /* We're done when the app thread has finished posting subreqs and all * the queues in all the streams are empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags)) + if (!netfs_are_all_subreqs_queued(wreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before lists. */ transferred = LONG_MAX; for (s = 0; s < NR_IO_STREAMS; s++) { @@ -395,13 +448,19 @@ bool netfs_write_collection(struct netfs_io_request *wreq) wreq->transferred = transferred; trace_netfs_rreq(wreq, netfs_rreq_trace_write_done); - if (wreq->io_streams[1].active && - wreq->io_streams[1].failed && - ictx->ops->invalidate_cache) { - /* Cache write failure doesn't prevent writeback completion - * unless we're in disconnected mode. - */ - ictx->ops->invalidate_cache(wreq); + if (cstream->active) { + if (test_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) { + if (ictx->ops->invalidate_cache) { + /* Cache write failure doesn't prevent + * writeback completion unless we're in + * disconnected mode. + */ + trace_netfs_rreq(wreq, netfs_rreq_trace_inval_cache); + ictx->ops->invalidate_cache(wreq); + } + } else if (!cstream->failed) { + netfs_cache_collect(wreq, cstream, cstream->cache_collect); + } } _debug("finished"); @@ -411,7 +470,6 @@ bool netfs_write_collection(struct netfs_io_request *wreq) switch (wreq->origin) { case NETFS_WRITEBACK: case NETFS_WRITEBACK_SINGLE: - case NETFS_WRITETHROUGH: netfs_wb_end(ictx); break; default: @@ -486,24 +544,51 @@ void netfs_write_subrequest_terminated(void *_op, ssize_t transferred_or_error) if (IS_ERR_VALUE(transferred_or_error)) { subreq->error = transferred_or_error; - /* if need retry is set, error should not matter */ - if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { - set_bit(NETFS_SREQ_FAILED, &subreq->flags); - trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); - } switch (subreq->source) { case NETFS_WRITE_TO_CACHE: + /* We don't mark a cache-write subreq as failed. + * Instead we tell the issuer to produce dummy subreqs + * instead and make a note if we need to invalidate the + * cache at the end. We also don't pause the loop that + * grabs pages and launches upload subreqs. + * + * Note that we need to distinguish between -ENOBUFS + * (no space available in the cache) and other errors. + * In the former case, we can keep the data we have, + * though we might have to change the way the on-disk + * data is tracked. + */ netfs_stat(&netfs_n_wh_write_failed); + if (test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) + break; + + trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + set_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags); + if (transferred_or_error == -ENOBUFS) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_no_space); + else if (!test_and_set_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_failed); + subreq->transferred = subreq->len; break; + case NETFS_UPLOAD_TO_SERVER: + /* If need_retry is set, error should not matter */ + if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { + set_bit(NETFS_SREQ_FAILED, &subreq->flags); + trace_netfs_failure(wreq, subreq, transferred_or_error, + netfs_fail_upload); + } + + set_bit(NETFS_RREQ_PAUSE, &wreq->flags); + trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); netfs_stat(&netfs_n_wh_upload_failed); break; + default: break; } - trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); - set_bit(NETFS_RREQ_PAUSE, &wreq->flags); } else { if (WARN(transferred_or_error > subreq->len - subreq->transferred, "Subreq excess write: R=%x[%x] %zd > %zu - %zu", diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 851f6f93ad45..3989b4ec0c4b 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -89,14 +89,13 @@ static void netfs_kill_dirty_pages(struct address_space *mapping, */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin) { struct netfs_io_request *wreq; struct netfs_inode *ictx; bool is_cacheable = (origin == NETFS_WRITEBACK || origin == NETFS_WRITEBACK_SINGLE || - origin == NETFS_WRITETHROUGH || origin == NETFS_PGPRIV2_COPY_TO_CACHE); wreq = netfs_alloc_request(mapping, file, start, 0, origin); @@ -112,6 +111,8 @@ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, goto nomem; wreq->cleaned_to = wreq->start; + if (wreq->cache_resources.dio_size > 1) + wreq->cache_coll_to = round_down(wreq->start, wreq->cache_resources.dio_size); wreq->io_streams[0].stream_nr = 0; wreq->io_streams[0].source = NETFS_UPLOAD_TO_SERVER; @@ -156,7 +157,7 @@ EXPORT_SYMBOL(netfs_prepare_write_failed); */ void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start) + uoff_t start) { struct netfs_io_subrequest *subreq; struct iov_iter *wreq_iter = &wreq->buffer.iter; @@ -169,10 +170,9 @@ void netfs_prepare_write(struct netfs_io_request *wreq, wreq_iter->folioq_slot >= folioq_nr_slots(wreq_iter->folioq)) rolling_buffer_make_space(&wreq->buffer, wreq->gfp); - subreq = netfs_alloc_subrequest(wreq); + subreq = netfs_alloc_subrequest(wreq, stream->source); if (!subreq) return; - subreq->source = stream->source; subreq->start = start; subreq->stream_nr = stream->stream_nr; subreq->io_iter = *wreq_iter; @@ -233,6 +233,21 @@ static void netfs_do_issue_write(struct netfs_io_stream *stream, _enter("R=%x[%x],%zx", wreq->debug_id, subreq->debug_index, subreq->len); + if (stream->source == NETFS_WRITE_TO_CACHE && + unlikely(test_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags))) { + size_t dio_size = wreq->cache_resources.dio_size; + size_t len, disp; + + disp = subreq->start & (dio_size - 1); + len = round_up(subreq->len + disp, dio_size); + + subreq->start -= disp; + subreq->len = len; + + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + return netfs_write_subrequest_terminated(subreq, subreq->len); + } + if (test_bit(NETFS_SREQ_FAILED, &subreq->flags)) return netfs_write_subrequest_terminated(subreq, subreq->error); @@ -266,6 +281,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, if (!subreq) return; + stream->construct = NULL; subreq->io_iter.count = subreq->len; netfs_do_issue_write(stream, subreq); @@ -279,7 +295,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, */ size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof) + uoff_t start, size_t len, bool to_eof) { struct netfs_io_subrequest *subreq = stream->construct; size_t part; @@ -330,7 +346,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, struct netfs_folio *finfo; size_t iter_off = 0; size_t fsize = folio_size(folio), flen = fsize, foff = 0; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false, streamw = false; bool debug = false; @@ -367,11 +383,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, streamw = true; } - if (wreq->origin == NETFS_WRITETHROUGH) { - to_eof = false; - if (flen > i_size - fpos) - flen = i_size - fpos; - } else if (flen > i_size - fpos) { + if (flen > i_size - fpos) { flen = i_size - fpos; if (!streamw) folio_zero_segment(folio, flen, fsize); @@ -525,8 +537,7 @@ static void netfs_end_issue_write(struct netfs_io_request *wreq) { bool needs_poke = true; - smp_wmb(); /* Write subreq lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); for (int s = 0; s < NR_IO_STREAMS; s++) { struct netfs_io_stream *stream = &wreq->io_streams[s]; @@ -614,103 +625,6 @@ out: EXPORT_SYMBOL(netfs_writepages); /* - * Begin a write operation for writing through the pagecache. - */ -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len) -{ - struct netfs_io_request *wreq = NULL; - struct netfs_inode *ictx = netfs_inode(file_inode(iocb->ki_filp)); - - netfs_wb_begin(ictx, false); - - wreq = netfs_create_write_req(iocb->ki_filp->f_mapping, iocb->ki_filp, - iocb->ki_pos, NETFS_WRITETHROUGH); - if (IS_ERR(wreq)) { - netfs_wb_end(ictx); - return wreq; - } - - wreq->io_streams[0].avail = true; - __set_bit(NETFS_RREQ_OFFLOAD_COLLECTION, &wreq->flags); - trace_netfs_write(wreq, netfs_write_trace_writethrough); - return wreq; -} - -/* - * Advance the state of the write operation used when writing through the - * pagecache. Data has been copied into the pagecache that we need to append - * to the request. If we've added more than wsize then we need to create a new - * subrequest. - */ -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache) -{ - int ret; - - _enter("R=%x ic=%zu ws=%u cp=%zu tp=%u", - wreq->debug_id, wreq->buffer.iter.count, wreq->wsize, copied, to_page_end); - - /* The folio is locked. */ - - if (*writethrough_cache != folio) { - if (*writethrough_cache) { - /* Did the folio get moved? */ - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - } - /* We can make multiple writes to the folio... */ - if (wreq->len == 0) - trace_netfs_folio(folio, netfs_folio_trace_wthru); - else - trace_netfs_folio(folio, netfs_folio_trace_wthru_plus); - *writethrough_cache = folio; - folio_get(folio); - } - - wreq->len += copied; - - if (!to_page_end) { - folio_mark_dirty(folio); - folio_unlock(folio); - return 0; - } - - ret = netfs_write_folio(wreq, wbc, folio); - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - wreq->submitted = wreq->len; - return ret; -} - -/* - * End a write operation used when writing through the pagecache. - */ -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache) -{ - ssize_t ret; - - _enter("R=%x", wreq->debug_id); - - if (writethrough_cache) { - folio_lock(writethrough_cache); - netfs_write_folio(wreq, wbc, writethrough_cache); - folio_put(writethrough_cache); - wreq->submitted = wreq->len; - } - - netfs_end_issue_write(wreq); - - if (wreq->iocb) - ret = -EIOCBQUEUED; - else - ret = netfs_wait_for_write(wreq); - netfs_put_request(wreq, netfs_rreq_trace_put_return); - return ret; -} - -/* * Write some of a pending folio data back to the server and/or the cache. */ static int netfs_write_folio_single(struct netfs_io_request *wreq, @@ -721,7 +635,7 @@ static int netfs_write_folio_single(struct netfs_io_request *wreq, struct netfs_io_stream *stream; size_t iter_off = 0; size_t fsize = folio_size(folio), flen; - loff_t fpos = folio_pos(folio); + uoff_t fpos = folio_pos(folio); ssize_t ret; bool to_eof = false; bool no_debug = false; @@ -891,8 +805,7 @@ int netfs_writeback_single(struct address_space *mapping, stop: for (int s = 0; s < NR_IO_STREAMS; s++) netfs_issue_write(wreq, &wreq->io_streams[s]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); netfs_wake_collector(wreq); diff --git a/fs/netfs/write_retry.c b/fs/netfs/write_retry.c index 058bc7a166a5..2f20577563e1 100644 --- a/fs/netfs/write_retry.c +++ b/fs/netfs/write_retry.c @@ -55,7 +55,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, do { struct netfs_io_subrequest *subreq = NULL, *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false; @@ -149,8 +149,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, * and insert them after. */ do { - subreq = netfs_alloc_subrequest(wreq); - subreq->source = to->source; + subreq = netfs_alloc_subrequest(wreq, stream->source); subreq->start = start; subreq->stream_nr = to->stream_nr; subreq->retry_count = 1; @@ -211,6 +210,7 @@ void netfs_retry_writes(struct netfs_io_request *wreq) int s; netfs_stat(&netfs_n_wh_retry_write_req); + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -235,4 +235,6 @@ void netfs_retry_writes(struct netfs_io_request *wreq) netfs_retry_write_stream(wreq, stream); } } + + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_end); } diff --git a/fs/nfs/Kconfig b/fs/nfs/Kconfig index 64c249f800a9..fd430718d02b 100644 --- a/fs/nfs/Kconfig +++ b/fs/nfs/Kconfig @@ -189,6 +189,7 @@ config NFS_FSCACHE bool "Provide NFS client caching support" depends on NFS_FS select NETFS_SUPPORT + select NETFS_PGPRIV2 select FSCACHE help Say Y here if you want NFS data to be cached locally on disc through diff --git a/fs/nfs/dir.c b/fs/nfs/dir.c index 49394123bd09..354f986e60c4 100644 --- a/fs/nfs/dir.c +++ b/fs/nfs/dir.c @@ -2437,7 +2437,7 @@ out_err: return error; } -int nfs_create(struct mnt_idmap *idmap, struct inode *dir, +int nfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return nfs_do_create(dir, dentry, mode, O_EXCL); @@ -2448,7 +2448,7 @@ EXPORT_SYMBOL_GPL(nfs_create); * See comments for nfs_proc_create regarding failed operations. */ int -nfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +nfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct iattr attr; @@ -2475,7 +2475,7 @@ EXPORT_SYMBOL_GPL(nfs_mknod); /* * See comments for nfs_proc_create regarding failed operations. */ -struct dentry *nfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +struct dentry *nfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct iattr attr; @@ -2641,7 +2641,7 @@ EXPORT_SYMBOL_GPL(nfs_unlink); * now have a new file handle and can instantiate an in-core NFS inode * and move the raw page into its mapping. */ -int nfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int nfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct folio *folio; @@ -2771,7 +2771,7 @@ static bool nfs_rename_is_unsafe_cross_dir(struct dentry *old_dentry, * If these conditions are met, we can drop the dentries before doing * the rename. */ -int nfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +int nfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -3393,7 +3393,7 @@ static int nfs_execute_ok(struct inode *inode, int mask) return ret; } -int nfs_permission(struct mnt_idmap *idmap, +int nfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { diff --git a/fs/nfs/inode.c b/fs/nfs/inode.c index 3022454f7698..3c9b2ec4e244 100644 --- a/fs/nfs/inode.c +++ b/fs/nfs/inode.c @@ -690,7 +690,7 @@ EXPORT_SYMBOL_GPL(nfs_update_delegated_mtime); #define NFS_VALID_ATTRS (ATTR_MODE|ATTR_UID|ATTR_GID|ATTR_SIZE|ATTR_ATIME|ATTR_ATIME_SET|ATTR_MTIME|ATTR_MTIME_SET|ATTR_FILE|ATTR_OPEN) int -nfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +nfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -955,7 +955,7 @@ static u32 nfs_get_valid_attrmask(struct inode *inode) return reply_mask; } -int nfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int nfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h index 48f7c0e25da1..d6ea41a3f9b4 100644 --- a/fs/nfs/internal.h +++ b/fs/nfs/internal.h @@ -397,18 +397,18 @@ extern unsigned long nfs_access_cache_scan(struct shrinker *shrink, struct shrink_control *sc); struct dentry *nfs_lookup(struct inode *, struct dentry *, unsigned int); void nfs_d_prune_case_insensitive_aliases(struct inode *inode); -int nfs_create(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_create(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); -struct dentry *nfs_mkdir(struct mnt_idmap *, struct inode *, struct dentry *, +struct dentry *nfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int nfs_rmdir(struct inode *, struct dentry *); int nfs_unlink(struct inode *, struct dentry *); -int nfs_symlink(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *); int nfs_link(struct dentry *, struct inode *, struct dentry *); -int nfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, umode_t, +int nfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t); -int nfs_rename(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); #ifdef CONFIG_NFS_V4_2 diff --git a/fs/nfs/namespace.c b/fs/nfs/namespace.c index 6d0073c24771..c50d59c52c50 100644 --- a/fs/nfs/namespace.c +++ b/fs/nfs/namespace.c @@ -222,7 +222,7 @@ out_fc: } static int -nfs_namespace_getattr(struct mnt_idmap *idmap, +nfs_namespace_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -235,7 +235,7 @@ nfs_namespace_getattr(struct mnt_idmap *idmap, } static int -nfs_namespace_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +nfs_namespace_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { if (NFS_FH(d_inode(dentry))->size != 0) diff --git a/fs/nfs/nfs3_fs.h b/fs/nfs/nfs3_fs.h index b333ea119ef5..ffcabadb3546 100644 --- a/fs/nfs/nfs3_fs.h +++ b/fs/nfs/nfs3_fs.h @@ -12,7 +12,7 @@ */ #ifdef CONFIG_NFS_V3_ACL extern struct posix_acl *nfs3_get_acl(struct inode *inode, int type, bool rcu); -extern int nfs3_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int nfs3_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int nfs3_proc_setacls(struct inode *inode, struct posix_acl *acl, struct posix_acl *dfacl); diff --git a/fs/nfs/nfs3acl.c b/fs/nfs/nfs3acl.c index a126eb31f62f..2549a1985b9a 100644 --- a/fs/nfs/nfs3acl.c +++ b/fs/nfs/nfs3acl.c @@ -254,7 +254,7 @@ int nfs3_proc_setacls(struct inode *inode, struct posix_acl *acl, } -int nfs3_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int nfs3_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct posix_acl *orig = acl, *dfacl = NULL, *alloc; diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index 518348e87dd8..beb659744760 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -7866,7 +7866,7 @@ int nfs4_lock_delegation_recall(struct file_lock *fl, struct nfs4_state *state, #define XATTR_NAME_NFSV4_ACL "system.nfs4_acl" static int nfs4_xattr_set_nfs4_acl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7889,7 +7889,7 @@ static bool nfs4_xattr_list_nfs4_acl(struct dentry *dentry) #define XATTR_NAME_NFSV4_DACL "system.nfs4_dacl" static int nfs4_xattr_set_nfs4_dacl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7912,7 +7912,7 @@ static bool nfs4_xattr_list_nfs4_dacl(struct dentry *dentry) #define XATTR_NAME_NFSV4_SACL "system.nfs4_sacl" static int nfs4_xattr_set_nfs4_sacl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7935,7 +7935,7 @@ static bool nfs4_xattr_list_nfs4_sacl(struct dentry *dentry) #ifdef CONFIG_NFS_V4_SECURITY_LABEL static int nfs4_xattr_set_nfs4_label(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7965,7 +7965,7 @@ static const struct xattr_handler nfs4_xattr_nfs4_label_handler = { #ifdef CONFIG_NFS_V4_2 static int nfs4_xattr_set_nfs4_user(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) diff --git a/fs/nfs/unlink.c b/fs/nfs/unlink.c index b57cfaa4d516..c8d712204e64 100644 --- a/fs/nfs/unlink.c +++ b/fs/nfs/unlink.c @@ -67,6 +67,7 @@ static void nfs_async_unlink_release(void *calldata) struct super_block *sb = dentry->d_sb; up_read_non_owner(&NFS_I(d_inode(dentry->d_parent))->rmdir_sem); + d_lookup_acquire(dentry); d_lookup_done(dentry); nfs_free_unlinkdata(data); dput(dentry); @@ -159,6 +160,8 @@ static int nfs_call_unlink(struct dentry *dentry, struct inode *inode, struct nf return ret; } data->dentry = alias; + d_lookup_release(alias); + nfs_do_call_unlink(inode, data); return 1; } diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index f9131827d391..4789f2ec2078 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -1046,7 +1046,6 @@ static __be32 nfsd_finish_read(struct svc_rqst *rqstp, struct svc_fh *fhp, nfsd_stats_io_read_add(nn, fhp->fh_export, host_err); *eof = nfsd_eof_on_read(file, offset, host_err, *count); *count = host_err; - fsnotify_access(file); trace_nfsd_read_io_done(rqstp, fhp, offset, *count); return 0; } else { @@ -1071,19 +1070,11 @@ __be32 nfsd_splice_read(struct svc_rqst *rqstp, struct svc_fh *fhp, struct file *file, loff_t offset, unsigned long *count, u32 *eof) { - struct splice_desc sd = { - .len = 0, - .total_len = *count, - .pos = offset, - .u.data = rqstp, - }; ssize_t host_err; trace_nfsd_read_splice(rqstp, fhp, offset, *count); - host_err = rw_verify_area(READ, file, &offset, *count); - if (!host_err) - host_err = splice_direct_to_actor(file, &sd, - nfsd_direct_splice_actor); + host_err = vfs_splice_to_actor(file, offset, *count, + nfsd_direct_splice_actor, rqstp); return nfsd_finish_read(rqstp, fhp, file, offset, count, eof, host_err); } diff --git a/fs/nilfs2/inode.c b/fs/nilfs2/inode.c index 34e6096069ad..8953719c4950 100644 --- a/fs/nilfs2/inode.c +++ b/fs/nilfs2/inode.c @@ -904,7 +904,7 @@ void nilfs_evict_inode(struct inode *inode) */ } -int nilfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int nilfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct nilfs_transaction_info ti; @@ -943,7 +943,7 @@ out_err: return err; } -int nilfs_permission(struct mnt_idmap *idmap, struct inode *inode, +int nilfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct nilfs_root *root = NILFS_I(inode)->i_root; diff --git a/fs/nilfs2/ioctl.c b/fs/nilfs2/ioctl.c index 01a04080ef70..f56a79767c69 100644 --- a/fs/nilfs2/ioctl.c +++ b/fs/nilfs2/ioctl.c @@ -135,7 +135,7 @@ int nilfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) * * Return: 0 on success, or a negative error code on failure. */ -int nilfs_fileattr_set(struct mnt_idmap *idmap, +int nilfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/nilfs2/namei.c b/fs/nilfs2/namei.c index e037e0c6e31a..d0ae24f37854 100644 --- a/fs/nilfs2/namei.c +++ b/fs/nilfs2/namei.c @@ -85,7 +85,7 @@ nilfs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags) * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int nilfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int nilfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -113,7 +113,7 @@ static int nilfs_create(struct mnt_idmap *idmap, struct inode *dir, } static int -nilfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +nilfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -138,7 +138,7 @@ nilfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return err; } -static int nilfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int nilfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct nilfs_transaction_info ti; @@ -218,7 +218,7 @@ static int nilfs_link(struct dentry *old_dentry, struct inode *dir, return err; } -static struct dentry *nilfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *nilfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -358,7 +358,7 @@ static int nilfs_rmdir(struct inode *dir, struct dentry *dentry) return err; } -static int nilfs_rename(struct mnt_idmap *idmap, +static int nilfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/nilfs2/nilfs.h b/fs/nilfs2/nilfs.h index 4fc42d3787a4..28d35b2fa7f7 100644 --- a/fs/nilfs2/nilfs.h +++ b/fs/nilfs2/nilfs.h @@ -270,7 +270,7 @@ extern int nilfs_sync_file(struct file *, loff_t, loff_t, int); /* ioctl.c */ int nilfs_fileattr_get(struct dentry *dentry, struct file_kattr *m); -int nilfs_fileattr_set(struct mnt_idmap *idmap, +int nilfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long nilfs_ioctl(struct file *, unsigned int, unsigned long); long nilfs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); @@ -299,10 +299,10 @@ struct inode *nilfs_iget_for_shadow(struct inode *inode); extern void nilfs_update_inode(struct inode *, struct buffer_head *, int); extern void nilfs_truncate(struct inode *); extern void nilfs_evict_inode(struct inode *); -extern int nilfs_setattr(struct mnt_idmap *, struct dentry *, +extern int nilfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nilfs_write_failed(struct address_space *mapping, loff_t to); -int nilfs_permission(struct mnt_idmap *idmap, struct inode *inode, +int nilfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int nilfs_load_inode_block(struct inode *inode, struct buffer_head **pbh); extern int nilfs_inode_dirty(struct inode *); diff --git a/fs/nls/nls_iso8859-14.c b/fs/nls/nls_iso8859-14.c index c789eccb8a69..60b9400f915a 100644 --- a/fs/nls/nls_iso8859-14.c +++ b/fs/nls/nls_iso8859-14.c @@ -138,24 +138,23 @@ static const unsigned char page00[256] = { }; static const unsigned char page01[256] = { - 0x00, 0x00, 0xa1, 0xa2, 0x00, 0x00, 0x00, 0x00, /* 0x00-0x07 */ - 0x00, 0x00, 0xa6, 0xab, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x00-0x07 */ + 0x00, 0x00, 0xa4, 0xa5, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x10-0x17 */ - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb0, 0xb1, /* 0x18-0x1f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x18-0x1f */ 0xb2, 0xb3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x28-0x2f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x30-0x37 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x38-0x3f */ - 0xb4, 0xb5, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x40-0x47 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x40-0x47 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x48-0x4f */ - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb7, 0xb9, /* 0x50-0x57 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x50-0x57 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x58-0x5f */ - 0xbb, 0xbf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ - 0x00, 0x00, 0xd7, 0xf7, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ 0x00, 0x00, 0x00, 0x00, 0xd0, 0xf0, 0xde, 0xfe, /* 0x70-0x77 */ 0xaf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ - - 0xa8, 0xb8, 0xaa, 0xba, 0xbd, 0xbe, 0x00, 0x00, /* 0x80-0x87 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x80-0x87 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x88-0x8f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x90-0x97 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x98-0x9f */ @@ -169,7 +168,7 @@ static const unsigned char page01[256] = { 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xd8-0xdf */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xe0-0xe7 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xe8-0xef */ - 0x00, 0x00, 0xac, 0xbc, 0x00, 0x00, 0x00, 0x00, /* 0xf0-0xf7 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xf0-0xf7 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xf8-0xff */ }; @@ -178,7 +177,7 @@ static const unsigned char page1e[256] = { 0x00, 0x00, 0xa6, 0xab, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x10-0x17 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb0, 0xb1, /* 0x18-0x1f */ - 0xb2, 0xb3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x28-0x2f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x30-0x37 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x38-0x3f */ @@ -188,9 +187,8 @@ static const unsigned char page1e[256] = { 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x58-0x5f */ 0xbb, 0xbf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ 0x00, 0x00, 0xd7, 0xf7, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ - 0x00, 0x00, 0x00, 0x00, 0xd0, 0xf0, 0xde, 0xfe, /* 0x70-0x77 */ - 0xaf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ - + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x70-0x77 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ 0xa8, 0xb8, 0xaa, 0xba, 0xbd, 0xbe, 0x00, 0x00, /* 0x80-0x87 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x88-0x8f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x90-0x97 */ diff --git a/fs/nsfs.c b/fs/nsfs.c index c3b6ae76594a..56ea0bb9ef3a 100644 --- a/fs/nsfs.c +++ b/fs/nsfs.c @@ -348,8 +348,8 @@ static long ns_ioctl(struct file *filp, unsigned int ioctl, return ret; FD_PREPARE(fdf, O_CLOEXEC, dentry_open(&path, O_RDONLY, current_cred())); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; /* * If @uinfo is passed return all information about the * mount namespace as well. diff --git a/fs/ntfs/ea.c b/fs/ntfs/ea.c index b4fcfbe2da4c..ddc201e3e2aa 100644 --- a/fs/ntfs/ea.c +++ b/fs/ntfs/ea.c @@ -875,7 +875,7 @@ static int ntfs_validate_fattr(struct ntfs_inode *ni, __le32 fattr) } static int ntfs_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) { @@ -977,7 +977,7 @@ const struct xattr_handler * const ntfs_xattr_handlers[] = { // clang-format on #ifdef CONFIG_NTFS_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -1021,7 +1021,7 @@ struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, return acl; } -static noinline int ntfs_set_acl_ex(struct mnt_idmap *idmap, +static noinline int ntfs_set_acl_ex(const struct mnt_idmap *idmap, struct inode *inode, struct posix_acl *acl, int type, bool init_acl) { @@ -1103,13 +1103,13 @@ out: return err; } -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { return ntfs_set_acl_ex(idmap, d_inode(dentry), acl, type, false); } -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir) { struct posix_acl *default_acl, *acl; diff --git a/fs/ntfs/ea.h b/fs/ntfs/ea.h index acb39c2a6fbc..690fafe181fb 100644 --- a/fs/ntfs/ea.h +++ b/fs/ntfs/ea.h @@ -17,11 +17,11 @@ int ntfs_ea_set_wsl_inode(struct inode *inode, dev_t rdev, __le16 *ea_size, ssize_t ntfs_listxattr(struct dentry *dentry, char *buffer, size_t size); #ifdef CONFIG_NTFS_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir); #else #define ntfs_get_acl NULL diff --git a/fs/ntfs/file.c b/fs/ntfs/file.c index 3ec82715a588..bb8641103d26 100644 --- a/fs/ntfs/file.c +++ b/fs/ntfs/file.c @@ -318,7 +318,7 @@ static int ntfs_setattr_size(struct inode *vi, struct iattr *attr) * NOTE: Changes in inode size are not supported yet for compressed or * encrypted files. */ -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *vi = d_inode(dentry); @@ -387,7 +387,7 @@ out: return err; } -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags) { diff --git a/fs/ntfs/inode.h b/fs/ntfs/inode.h index ff61bd402df0..45a396c97846 100644 --- a/fs/ntfs/inode.h +++ b/fs/ntfs/inode.h @@ -330,9 +330,9 @@ int ntfs_read_inode_mount(struct inode *vi); int ntfs_show_options(struct seq_file *sf, struct dentry *root); int ntfs_truncate_vfs(struct inode *vi, loff_t new_size, loff_t i_size); -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags); diff --git a/fs/ntfs/namei.c b/fs/ntfs/namei.c index 75e201096525..028efad7d9c0 100644 --- a/fs/ntfs/namei.c +++ b/fs/ntfs/namei.c @@ -391,7 +391,7 @@ static int ntfs_sd_add_everyone(struct ntfs_inode *ni) return ret; } -static struct ntfs_inode *__ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static struct ntfs_inode *__ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, __le16 *name, u8 name_len, mode_t mode, dev_t dev, const char *target, int target_len) { @@ -731,7 +731,7 @@ err_out: return ERR_PTR(err); } -static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct ntfs_volume *vol = NTFS_SB(dir->i_sb); @@ -1046,7 +1046,7 @@ out: return err; } -static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ntfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1243,7 +1243,7 @@ err_out: return err; } -static int ntfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ntfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1393,7 +1393,7 @@ err_out: return err; } -static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct super_block *sb = dir->i_sb; @@ -1440,7 +1440,7 @@ out: return err; } -static int ntfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct super_block *sb = dir->i_sb; diff --git a/fs/ntfs3/file.c b/fs/ntfs3/file.c index 95dfe6878860..4cbdd9e4222b 100644 --- a/fs/ntfs3/file.c +++ b/fs/ntfs3/file.c @@ -129,7 +129,7 @@ int ntfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) /* * ntfs_fileattr_set - inode_operations::fileattr_set */ -int ntfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -264,7 +264,7 @@ long ntfs_compat_ioctl(struct file *filp, u32 cmd, unsigned long arg) /* * ntfs_getattr - inode_operations::getattr */ -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, u32 flags) { struct inode *inode = d_inode(path->dentry); @@ -706,7 +706,7 @@ out: /* * ntfs_setattr - inode_operations::setattr */ -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ntfs3/inode.c b/fs/ntfs3/inode.c index 5276c7a00db9..366af3038248 100644 --- a/fs/ntfs3/inode.c +++ b/fs/ntfs3/inode.c @@ -1385,7 +1385,7 @@ out: * * NOTE: if fnd != NULL (ntfs_atomic_open) then @dir is locked */ -int ntfs_create_inode(struct mnt_idmap *idmap, struct inode *dir, +int ntfs_create_inode(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const struct cpu_str *uni, umode_t mode, dev_t dev, const char *symname, u32 size, struct ntfs_fnd *fnd) diff --git a/fs/ntfs3/namei.c b/fs/ntfs3/namei.c index ec59bbabd3c5..ae88cb66fb7e 100644 --- a/fs/ntfs3/namei.c +++ b/fs/ntfs3/namei.c @@ -111,7 +111,7 @@ static struct dentry *ntfs_lookup(struct inode *dir, struct dentry *dentry, /* * ntfs_create - inode_operations::create */ -static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ntfs_create_inode(idmap, dir, dentry, NULL, S_IFREG | mode, 0, @@ -121,7 +121,7 @@ static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, /* * ntfs_mknod - inode_operations::mknod */ -static int ntfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { return ntfs_create_inode(idmap, dir, dentry, NULL, mode, rdev, NULL, 0, @@ -209,7 +209,7 @@ static int ntfs_unlink(struct inode *dir, struct dentry *dentry) /* * ntfs_symlink - inode_operations::symlink */ -static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { u32 size = strlen(symname); @@ -228,7 +228,7 @@ static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, /* * ntfs_mkdir - inode_operations::mkdir */ -static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ntfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(ntfs_create_inode(idmap, dir, dentry, NULL, @@ -262,7 +262,7 @@ static int ntfs_rmdir(struct inode *dir, struct dentry *dentry) /* * ntfs_rename - inode_operations::rename */ -static int ntfs_rename(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_rename(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct inode *new_dir, struct dentry *new_dentry, u32 flags) { diff --git a/fs/ntfs3/ntfs_fs.h b/fs/ntfs3/ntfs_fs.h index 5811d89d67b3..ea24126c70db 100644 --- a/fs/ntfs3/ntfs_fs.h +++ b/fs/ntfs3/ntfs_fs.h @@ -551,11 +551,11 @@ extern const struct file_operations ntfs_dir_operations; /* Globals from file.c */ int ntfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ntfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, u32 flags); -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int ntfs_file_open(struct inode *inode, struct file *file); int ntfs_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, @@ -803,7 +803,7 @@ int ntfs_set_size(struct inode *inode, u64 new_size); int ntfs3_write_inode(struct inode *inode, struct writeback_control *wbc); int ntfs_sync_inode(struct inode *inode); int inode_read_data(struct inode *inode, void *data, size_t bytes); -int ntfs_create_inode(struct mnt_idmap *idmap, struct inode *dir, +int ntfs_create_inode(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const struct cpu_str *uni, umode_t mode, dev_t dev, const char *symname, u32 size, struct ntfs_fnd *fnd); @@ -963,18 +963,18 @@ unsigned long ntfs_names_hash(const u16 *name, size_t len, const u16 *upcase, /* globals from xattr.c */ #ifdef CONFIG_NTFS3_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir); #else #define ntfs_get_acl NULL #define ntfs_set_acl NULL #endif -int ntfs_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry); +int ntfs_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry); ssize_t ntfs_listxattr(struct dentry *dentry, char *buffer, size_t size); extern const struct xattr_handler *const ntfs_xattr_handlers[]; diff --git a/fs/ntfs3/xattr.c b/fs/ntfs3/xattr.c index 7f77df9c46f8..e9824bd80322 100644 --- a/fs/ntfs3/xattr.c +++ b/fs/ntfs3/xattr.c @@ -545,7 +545,7 @@ out: /* * ntfs_get_acl - inode_operations::get_acl */ -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -597,7 +597,7 @@ struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, return acl; } -static noinline int ntfs_set_acl_ex(struct mnt_idmap *idmap, +static noinline int ntfs_set_acl_ex(const struct mnt_idmap *idmap, struct inode *inode, struct posix_acl *acl, int type, bool init_acl) { @@ -679,7 +679,7 @@ out: /* * ntfs_set_acl - inode_operations::set_acl */ -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { return ntfs_set_acl_ex(idmap, d_inode(dentry), acl, type, false); @@ -690,7 +690,7 @@ int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, * * Called from ntfs_create_inode(). */ -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir) { struct posix_acl *default_acl, *acl; @@ -724,7 +724,7 @@ int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, /* * ntfs_acl_chmod - Helper for ntfs_setattr(). */ -int ntfs_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry) +int ntfs_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry) { struct inode *inode = d_inode(dentry); struct super_block *sb = inode->i_sb; @@ -865,7 +865,7 @@ static bool ntfs_is_reserved_lxattr(const char *name) * ntfs_setxattr - inode_operations::setxattr */ static noinline int ntfs_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *de, + const struct mnt_idmap *idmap, struct dentry *de, struct inode *inode, const char *name, const void *value, size_t size, int flags) { diff --git a/fs/ocfs2/acl.c b/fs/ocfs2/acl.c index 090ec60fb576..801a2f56ad08 100644 --- a/fs/ocfs2/acl.c +++ b/fs/ocfs2/acl.c @@ -260,7 +260,7 @@ static int ocfs2_set_acl(handle_t *handle, return ret; } -int ocfs2_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct buffer_head *bh = NULL; diff --git a/fs/ocfs2/acl.h b/fs/ocfs2/acl.h index a91f9ce278d6..1ed05899cce1 100644 --- a/fs/ocfs2/acl.h +++ b/fs/ocfs2/acl.h @@ -17,7 +17,7 @@ struct ocfs2_acl_entry { }; struct posix_acl *ocfs2_iop_get_acl(struct inode *inode, int type, bool rcu); -int ocfs2_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ocfs2_acl_chmod(struct inode *, struct buffer_head *); struct ocfs2_acl_state { diff --git a/fs/ocfs2/buffer_head_io.c b/fs/ocfs2/buffer_head_io.c index 7bfe377af2df..733ceda79ca1 100644 --- a/fs/ocfs2/buffer_head_io.c +++ b/fs/ocfs2/buffer_head_io.c @@ -66,12 +66,14 @@ int ocfs2_write_block(struct ocfs2_super *osb, struct buffer_head *bh, wait_on_buffer(bh); - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { ocfs2_set_buffer_uptodate(ci, bh); } else { - /* We don't need to remove the clustered uptodate - * information for this bh as it's not marked locally - * uptodate. */ + /* + * The buffer still holds what we tried to write, but it did + * not reach the disk, so don't advertise it to the cluster + * as up to date. + */ ret = -EIO; mlog_errno(ret); } @@ -446,7 +448,7 @@ int ocfs2_write_super_or_backup(struct ocfs2_super *osb, wait_on_buffer(bh); - if (!buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ret = -EIO; mlog_errno(ret); } diff --git a/fs/ocfs2/dlmfs/dlmfs.c b/fs/ocfs2/dlmfs/dlmfs.c index 53df5dd10ad0..d3bfcada3e3b 100644 --- a/fs/ocfs2/dlmfs/dlmfs.c +++ b/fs/ocfs2/dlmfs/dlmfs.c @@ -188,7 +188,7 @@ static int dlmfs_file_release(struct inode *inode, * We do ->setattr() just to override size changes. Our size is the size * of the LVB and nothing else. */ -static int dlmfs_file_setattr(struct mnt_idmap *idmap, +static int dlmfs_file_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int error; @@ -402,7 +402,7 @@ static struct inode *dlmfs_get_inode(struct inode *parent, * File creation. Allocate an inode, and we're done.. */ /* SMP-safe */ -static struct dentry *dlmfs_mkdir(struct mnt_idmap * idmap, +static struct dentry *dlmfs_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) @@ -450,7 +450,7 @@ bail: return ERR_PTR(status); } -static int dlmfs_create(struct mnt_idmap *idmap, +static int dlmfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) diff --git a/fs/ocfs2/file.c b/fs/ocfs2/file.c index d6e977ba6565..62f45a1b5ca1 100644 --- a/fs/ocfs2/file.c +++ b/fs/ocfs2/file.c @@ -1117,7 +1117,7 @@ out: return ret; } -int ocfs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int status = 0, size_change; @@ -1317,7 +1317,7 @@ bail: return status; } -int ocfs2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ocfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); @@ -1349,7 +1349,7 @@ bail: return err; } -int ocfs2_permission(struct mnt_idmap *idmap, struct inode *inode, +int ocfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret, had_lock; diff --git a/fs/ocfs2/file.h b/fs/ocfs2/file.h index 41e65e45a9f3..97492ee5789e 100644 --- a/fs/ocfs2/file.h +++ b/fs/ocfs2/file.h @@ -50,11 +50,11 @@ int ocfs2_extend_no_holes(struct inode *inode, struct buffer_head *di_bh, u64 new_i_size, u64 zero_to); int ocfs2_zero_extend(struct inode *inode, struct buffer_head *di_bh, loff_t zero_to); -int ocfs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ocfs2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ocfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int ocfs2_permission(struct mnt_idmap *idmap, +int ocfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); diff --git a/fs/ocfs2/ioctl.c b/fs/ocfs2/ioctl.c index cbe59d231666..36c7c9ac8b5d 100644 --- a/fs/ocfs2/ioctl.c +++ b/fs/ocfs2/ioctl.c @@ -82,7 +82,7 @@ int ocfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return status; } -int ocfs2_fileattr_set(struct mnt_idmap *idmap, +int ocfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ocfs2/ioctl.h b/fs/ocfs2/ioctl.h index 4a1c2313b429..b1cb529fc5f9 100644 --- a/fs/ocfs2/ioctl.h +++ b/fs/ocfs2/ioctl.h @@ -12,7 +12,7 @@ #define OCFS2_IOCTL_PROTO_H int ocfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ocfs2_fileattr_set(struct mnt_idmap *idmap, +int ocfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long ocfs2_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); long ocfs2_compat_ioctl(struct file *file, unsigned cmd, unsigned long arg); diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb..ea6802d894c2 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -676,19 +676,20 @@ static int __ocfs2_journal_access(handle_t *handle, mlog(ML_ERROR, "giving me a buffer that's not uptodate!\n"); mlog(ML_ERROR, "b_blocknr=%llu, b_state=0x%lx\n", (unsigned long long)bh->b_blocknr, bh->b_state); - + } + /* + * A previous transaction with a couple of buffer heads fail + * to checkpoint, so all the bhs are marked as BH_Write_EIO. + * For current transaction, the bh is just among those error + * bhs which previous transaction handle. We can't just clear + * its BH_Write_EIO and reuse directly, since other bhs are + * not written to disk yet and that will cause metadata + * inconsistency. So we should set fs read-only to avoid + * further damage. + */ + if (buffer_write_io_error(bh)) { lock_buffer(bh); - /* - * A previous transaction with a couple of buffer heads fail - * to checkpoint, so all the bhs are marked as BH_Write_EIO. - * For current transaction, the bh is just among those error - * bhs which previous transaction handle. We can't just clear - * its BH_Write_EIO and reuse directly, since other bhs are - * not written to disk yet and that will cause metadata - * inconsistency. So we should set fs read-only to avoid - * further damage. - */ - if (buffer_write_io_error(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { unlock_buffer(bh); return ocfs2_error(osb->sb, "A previous attempt to " "write this buffer head failed\n"); diff --git a/fs/ocfs2/namei.c b/fs/ocfs2/namei.c index 58c6061ed983..fce9a31a3671 100644 --- a/fs/ocfs2/namei.c +++ b/fs/ocfs2/namei.c @@ -227,7 +227,7 @@ static void ocfs2_cleanup_add_entry_failure(struct ocfs2_super *osb, iput(inode); } -static int ocfs2_mknod(struct mnt_idmap *idmap, +static int ocfs2_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -650,7 +650,7 @@ static int ocfs2_mknod_locked(struct ocfs2_super *osb, suballoc_loc, suballoc_bit); } -static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap, +static struct dentry *ocfs2_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -666,7 +666,7 @@ static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap, return ERR_PTR(ret); } -static int ocfs2_create(struct mnt_idmap *idmap, +static int ocfs2_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -1206,7 +1206,7 @@ static void ocfs2_double_unlock(struct inode *inode1, struct inode *inode2) ocfs2_inode_unlock(inode2, 1); } -static int ocfs2_rename(struct mnt_idmap *idmap, +static int ocfs2_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, @@ -1810,7 +1810,7 @@ bail: return status; } -static int ocfs2_symlink(struct mnt_idmap *idmap, +static int ocfs2_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index bfafe059bedf..5274f4d571b6 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -7500,7 +7500,7 @@ static int ocfs2_xattr_security_get(const struct xattr_handler *handler, } static int ocfs2_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -7595,7 +7595,7 @@ static int ocfs2_xattr_trusted_get(const struct xattr_handler *handler, } static int ocfs2_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -7626,7 +7626,7 @@ static int ocfs2_xattr_user_get(const struct xattr_handler *handler, } static int ocfs2_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/omfs/dir.c b/fs/omfs/dir.c index 692297cf84e7..18f4b4543cc8 100644 --- a/fs/omfs/dir.c +++ b/fs/omfs/dir.c @@ -279,13 +279,13 @@ out_free_inode: return err; } -static struct dentry *omfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *omfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(omfs_add_node(dir, dentry, mode)); } -static int omfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int omfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return omfs_add_node(dir, dentry, mode | S_IFREG); @@ -370,7 +370,7 @@ static bool omfs_fill_chain(struct inode *dir, struct dir_context *ctx, return true; } -static int omfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int omfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/omfs/file.c b/fs/omfs/file.c index 28f3b113340e..79a413f1dc0d 100644 --- a/fs/omfs/file.c +++ b/fs/omfs/file.c @@ -338,7 +338,7 @@ const struct file_operations omfs_file_operations = { .splice_read = filemap_splice_read, }; -static int omfs_setattr(struct mnt_idmap *idmap, +static int omfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/omfs/inode.c b/fs/omfs/inode.c index 1d915ef72119..bc37029a4afb 100644 --- a/fs/omfs/inode.c +++ b/fs/omfs/inode.c @@ -145,7 +145,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh); if (wait) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) sync_failed = 1; } @@ -159,7 +159,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh2); if (wait) { sync_dirty_buffer(bh2); - if (buffer_req(bh2) && !buffer_uptodate(bh2)) + if (buffer_write_io_error(bh2)) sync_failed = 1; } brelse(bh2); diff --git a/fs/open.c b/fs/open.c index 6b1c14e684a9..e43f02ff64ac 100644 --- a/fs/open.c +++ b/fs/open.c @@ -36,7 +36,7 @@ #include "internal.h" -int do_truncate(struct mnt_idmap *idmap, struct dentry *dentry, +int do_truncate(const struct mnt_idmap *idmap, struct dentry *dentry, loff_t length, unsigned int time_attrs, struct file *filp) { int ret; @@ -72,7 +72,7 @@ int do_truncate(struct mnt_idmap *idmap, struct dentry *dentry, int vfs_truncate(const struct path *path, loff_t length) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode; int error; @@ -787,7 +787,7 @@ static inline bool setattr_vfsgid(struct iattr *attr, kgid_t kgid) int chown_common(const struct path *path, uid_t user, gid_t group) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct user_namespace *fs_userns; struct inode *inode = path->dentry->d_inode; struct delegated_inode delegated_inode = { }; @@ -931,6 +931,11 @@ cleanup_inode: return error; } +/* + * Populate struct file + * + * NOTE: it assumes f_path is populated and consumes the caller's reference. + */ static int do_dentry_open(struct file *f, int (*open)(struct inode *, struct file *)) { @@ -938,7 +943,6 @@ static int do_dentry_open(struct file *f, struct inode *inode = f->f_path.dentry->d_inode; int error; - path_get(&f->f_path); f->f_inode = inode; f->f_mapping = inode->i_mapping; f->f_wb_err = filemap_sample_wb_err(f->f_mapping); @@ -1055,6 +1059,7 @@ int finish_open(struct file *file, struct dentry *dentry, BUG_ON(file->f_mode & FMODE_OPENED); /* once it's opened, it's opened */ file->__f_path.dentry = dentry; + path_get(&file->f_path); return do_dentry_open(file, open); } EXPORT_SYMBOL(finish_open); @@ -1098,6 +1103,7 @@ int vfs_open(const struct path *path, struct file *file) int ret; file->__f_path = *path; + path_get(&file->f_path); ret = do_dentry_open(file, NULL); if (!ret) { /* @@ -1110,6 +1116,25 @@ int vfs_open(const struct path *path, struct file *file) return ret; } +/** + * vfs_open_consume - open the file at the given path and consume the reference + * @path: path to open + * @file: newly allocated file with f_flag initialized + */ +int vfs_open_consume(struct path *path, struct file *file) +{ + int ret; + + file->__f_path = *path; + path->mnt = NULL; + path->dentry = NULL; + ret = do_dentry_open(file, NULL); + if (!ret) { + fsnotify_open(file); + } + return ret; +} + struct file *dentry_open(const struct path *path, int flags, const struct cred *cred) { @@ -1537,6 +1562,19 @@ int filp_close(struct file *filp, fl_owner_t id) } EXPORT_SYMBOL(filp_close); +/* Like filp_close() but the last reference is put right here. */ +int filp_close_sync(struct file *filp, fl_owner_t id) +{ + int retval; + + /* Kernel threads must never put their final reference here. */ + VFS_WARN_ON_ONCE(current->flags & PF_KTHREAD); + retval = filp_flush(filp, id); + fput_close_sync(filp); + + return retval; +} + /* * Careful here! We test whether the file pointer is NULL before * releasing the fd. This ensures that one clone task can't release @@ -1551,13 +1589,11 @@ SYSCALL_DEFINE1(close, unsigned int, fd) if (!file) return -EBADF; - retval = filp_flush(file, current->files); - /* * We're returning to user space. Don't bother * with any delayed fput() cases. */ - fput_close_sync(file); + retval = filp_close_sync(file, current->files); if (likely(retval == 0)) return 0; diff --git a/fs/orangefs/acl.c b/fs/orangefs/acl.c index a01ef0c1b1bf..f31196e4bfaa 100644 --- a/fs/orangefs/acl.c +++ b/fs/orangefs/acl.c @@ -112,7 +112,7 @@ int __orangefs_set_acl(struct inode *inode, struct posix_acl *acl, int type) return error; } -int orangefs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int orangefs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; diff --git a/fs/orangefs/inode.c b/fs/orangefs/inode.c index c088a02e8215..b1fed1c81a4d 100644 --- a/fs/orangefs/inode.c +++ b/fs/orangefs/inode.c @@ -847,7 +847,7 @@ int __orangefs_setattr_mode(struct dentry *dentry, struct iattr *iattr) /* * Change attributes of an object referenced by dentry. */ -int orangefs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int orangefs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int ret; @@ -867,7 +867,7 @@ out: /* * Obtain attributes of an object given a dentry */ -int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, +int orangefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { int ret; @@ -891,7 +891,7 @@ int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, return ret; } -int orangefs_permission(struct mnt_idmap *idmap, +int orangefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret; @@ -953,7 +953,7 @@ static int orangefs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -static int orangefs_fileattr_set(struct mnt_idmap *idmap, +static int orangefs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { u64 val = 0; diff --git a/fs/orangefs/namei.c b/fs/orangefs/namei.c index 8ebc34e112d5..32b7769ea49c 100644 --- a/fs/orangefs/namei.c +++ b/fs/orangefs/namei.c @@ -15,7 +15,7 @@ /* * Get a newly allocated inode to go with a negative dentry. */ -static int orangefs_create(struct mnt_idmap *idmap, +static int orangefs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -211,7 +211,7 @@ static int orangefs_unlink(struct inode *dir, struct dentry *dentry) return ret; } -static int orangefs_symlink(struct mnt_idmap *idmap, +static int orangefs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) @@ -296,7 +296,7 @@ out: return ret; } -static struct dentry *orangefs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *orangefs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct orangefs_inode_s *parent = ORANGEFS_I(dir); @@ -364,7 +364,7 @@ out: return ret ? ERR_PTR(ret) : NULL; } -static int orangefs_rename(struct mnt_idmap *idmap, +static int orangefs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, diff --git a/fs/orangefs/orangefs-kernel.h b/fs/orangefs/orangefs-kernel.h index 1451fc2c1917..348fe340c5d5 100644 --- a/fs/orangefs/orangefs-kernel.h +++ b/fs/orangefs/orangefs-kernel.h @@ -98,7 +98,7 @@ enum orangefs_vfs_op_states { extern const struct xattr_handler * const orangefs_xattr_handlers[]; extern struct posix_acl *orangefs_get_acl(struct inode *inode, int type, bool rcu); -extern int orangefs_set_acl(struct mnt_idmap *idmap, +extern int orangefs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int __orangefs_set_acl(struct inode *inode, struct posix_acl *acl, int type); @@ -352,12 +352,12 @@ struct inode *orangefs_new_inode(struct super_block *sb, int __orangefs_setattr(struct inode *, struct iattr *); int __orangefs_setattr_mode(struct dentry *dentry, struct iattr *iattr); -int orangefs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int orangefs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, +int orangefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int orangefs_permission(struct mnt_idmap *idmap, +int orangefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int orangefs_update_time(struct inode *inode, enum fs_update_time type, diff --git a/fs/orangefs/xattr.c b/fs/orangefs/xattr.c index 885fd3bd5a3d..a49e64566e1e 100644 --- a/fs/orangefs/xattr.c +++ b/fs/orangefs/xattr.c @@ -527,7 +527,7 @@ out_unlock: } static int orangefs_xattr_set_default(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, diff --git a/fs/overlayfs/dir.c b/fs/overlayfs/dir.c index 7beb0af26498..1194ccf981c3 100644 --- a/fs/overlayfs/dir.c +++ b/fs/overlayfs/dir.c @@ -688,7 +688,7 @@ static int ovl_create_or_link(struct dentry *dentry, struct inode *inode, return err; } -static int ovl_create_object(struct mnt_idmap *idmap, struct dentry *dentry, +static int ovl_create_object(const struct mnt_idmap *idmap, struct dentry *dentry, int mode, dev_t rdev, const char *link) { int err; @@ -730,19 +730,19 @@ out: return err; } -static int ovl_create(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ovl_create_object(idmap, dentry, (mode & 07777) | S_IFREG, 0, NULL); } -static struct dentry *ovl_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ovl_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(ovl_create_object(idmap, dentry, (mode & 07777) | S_IFDIR, 0, NULL)); } -static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { /* Don't allow creation of "whiteout" on overlay */ @@ -752,7 +752,7 @@ static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir, return ovl_create_object(idmap, dentry, mode, rdev, NULL); } -static int ovl_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *link) { return ovl_create_object(idmap, dentry, S_IFLNK, 0, link); @@ -1344,7 +1344,7 @@ static void ovl_rename_end(struct ovl_renamedata *ovlrd) ovl_drop_write(ovlrd->old_dentry); } -static int ovl_rename(struct mnt_idmap *idmap, struct inode *olddir, +static int ovl_rename(const struct mnt_idmap *idmap, struct inode *olddir, struct dentry *old, struct inode *newdir, struct dentry *new, unsigned int flags) { @@ -1420,7 +1420,7 @@ static int ovl_dummy_open(struct inode *inode, struct file *file) return 0; } -static int ovl_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { int err; diff --git a/fs/overlayfs/file.c b/fs/overlayfs/file.c index f3d97eb146e8..7433220d4ad6 100644 --- a/fs/overlayfs/file.c +++ b/fs/overlayfs/file.c @@ -30,7 +30,7 @@ static struct file *ovl_open_realfile(const struct file *file, { struct inode *realinode = d_inode(realpath->dentry); struct inode *inode = file_inode(file); - struct mnt_idmap *real_idmap; + const struct mnt_idmap *real_idmap; struct file *realfile; int flags = file->f_flags | OVL_OPEN_FLAGS; int acc_mode = ACC_MODE(flags); diff --git a/fs/overlayfs/inode.c b/fs/overlayfs/inode.c index 401cb8c75520..70183d516e5e 100644 --- a/fs/overlayfs/inode.c +++ b/fs/overlayfs/inode.c @@ -18,7 +18,7 @@ #include "overlayfs.h" -int ovl_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int err; @@ -168,7 +168,7 @@ static inline int ovl_real_getattr_nosec(struct super_block *sb, return vfs_getattr_nosec(path, stat, request_mask, flags); } -int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, +int ovl_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct dentry *dentry = path->dentry; @@ -303,7 +303,7 @@ int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, return err; } -int ovl_permission(struct mnt_idmap *idmap, +int ovl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct inode *upperinode = ovl_inode_upper(inode); @@ -355,7 +355,7 @@ static const char *ovl_get_link(struct dentry *dentry, * alter the POSIX ACLs for the underlying filesystem. */ static void ovl_idmap_posix_acl(const struct inode *realinode, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct posix_acl *acl) { struct user_namespace *fs_userns = i_user_ns(realinode); @@ -406,7 +406,7 @@ struct posix_acl *ovl_get_acl_path(const struct path *path, const char *acl_name, bool noperm) { struct posix_acl *real_acl, *clone; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *realinode = d_inode(path->dentry); idmap = mnt_idmap(path->mnt); @@ -447,7 +447,7 @@ struct posix_acl *ovl_get_acl_path(const struct path *path, * * This is obviously only relevant when idmapped layers are used. */ -struct posix_acl *do_ovl_get_acl(struct mnt_idmap *idmap, +struct posix_acl *do_ovl_get_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, bool rcu, bool noperm) { @@ -536,7 +536,7 @@ out: return err; } -int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int err; @@ -650,7 +650,7 @@ int ovl_real_fileattr_set(const struct path *realpath, struct file_kattr *fa) return vfs_fileattr_set(mnt_idmap(realpath->mnt), realpath->dentry, fa); } -int ovl_fileattr_set(struct mnt_idmap *idmap, +int ovl_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/overlayfs/overlayfs.h b/fs/overlayfs/overlayfs.h index 7f3558372c59..53fbbe15c31d 100644 --- a/fs/overlayfs/overlayfs.h +++ b/fs/overlayfs/overlayfs.h @@ -804,11 +804,11 @@ int ovl_set_nlink_lower(struct dentry *dentry); unsigned int ovl_get_nlink(struct ovl_fs *ofs, struct dentry *lowerdentry, struct dentry *upperdentry, unsigned int fallback); -int ovl_permission(struct mnt_idmap *idmap, struct inode *inode, +int ovl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); #ifdef CONFIG_FS_POSIX_ACL -struct posix_acl *do_ovl_get_acl(struct mnt_idmap *idmap, +struct posix_acl *do_ovl_get_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, bool rcu, bool noperm); static inline struct posix_acl *ovl_get_inode_acl(struct inode *inode, int type, @@ -816,12 +816,12 @@ static inline struct posix_acl *ovl_get_inode_acl(struct inode *inode, int type, { return do_ovl_get_acl(&nop_mnt_idmap, inode, type, rcu, true); } -static inline struct posix_acl *ovl_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *ovl_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { return do_ovl_get_acl(idmap, d_inode(dentry), type, false, false); } -int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); struct posix_acl *ovl_get_acl_path(const struct path *path, const char *acl_name, bool noperm); @@ -916,7 +916,7 @@ extern const struct file_operations ovl_file_operations; int ovl_real_fileattr_get(const struct path *realpath, struct file_kattr *fa); int ovl_real_fileattr_set(const struct path *realpath, struct file_kattr *fa); int ovl_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ovl_fileattr_set(struct mnt_idmap *idmap, +int ovl_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); struct ovl_file; struct ovl_file *ovl_file_alloc(struct file *realfile); @@ -950,8 +950,8 @@ static inline bool ovl_force_readonly(struct ovl_fs *ofs) /* xattr.c */ const struct xattr_handler * const *ovl_xattr_handlers(struct ovl_fs *ofs); -int ovl_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, +int ovl_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); ssize_t ovl_listxattr(struct dentry *dentry, char *list, size_t size); diff --git a/fs/overlayfs/ovl_entry.h b/fs/overlayfs/ovl_entry.h index 80cad4ea96a3..ac07e8769f9b 100644 --- a/fs/overlayfs/ovl_entry.h +++ b/fs/overlayfs/ovl_entry.h @@ -105,7 +105,7 @@ static inline struct vfsmount *ovl_upper_mnt(struct ovl_fs *ofs) return ofs->layers[0].mnt; } -static inline struct mnt_idmap *ovl_upper_mnt_idmap(struct ovl_fs *ofs) +static inline const struct mnt_idmap *ovl_upper_mnt_idmap(struct ovl_fs *ofs) { return mnt_idmap(ovl_upper_mnt(ofs)); } diff --git a/fs/overlayfs/util.c b/fs/overlayfs/util.c index b41f4788e4f0..521717209b2e 100644 --- a/fs/overlayfs/util.c +++ b/fs/overlayfs/util.c @@ -657,7 +657,7 @@ bool ovl_path_is_whiteout(struct ovl_fs *ofs, const struct path *path) struct file *ovl_path_open(const struct path *path, int flags) { struct inode *inode = d_inode(path->dentry); - struct mnt_idmap *real_idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *real_idmap = mnt_idmap(path->mnt); int err, acc_mode; if (flags & ~(O_ACCMODE | O_LARGEFILE)) @@ -1496,7 +1496,7 @@ void ovl_copyattr(struct inode *inode) { struct path realpath; struct inode *realinode; - struct mnt_idmap *real_idmap; + const struct mnt_idmap *real_idmap; vfsuid_t vfsuid; vfsgid_t vfsgid; diff --git a/fs/overlayfs/xattrs.c b/fs/overlayfs/xattrs.c index 5ae44b9c8790..acc54f6138ef 100644 --- a/fs/overlayfs/xattrs.c +++ b/fs/overlayfs/xattrs.c @@ -190,7 +190,7 @@ static int ovl_own_xattr_get(const struct xattr_handler *handler, } static int ovl_own_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -217,7 +217,7 @@ static int ovl_other_xattr_get(const struct xattr_handler *handler, } static int ovl_other_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/pidfs.c b/fs/pidfs.c index a6a643f15d08..c37c17bcdbdc 100644 --- a/fs/pidfs.c +++ b/fs/pidfs.c @@ -823,13 +823,13 @@ static struct vfsmount *pidfs_mnt __ro_after_init; * implemented. Let's reject it completely until we have a clean * permission concept for pidfds. */ -static int pidfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int pidfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return anon_inode_setattr(idmap, dentry, attr); } -static int pidfs_getattr(struct mnt_idmap *idmap, const struct path *path, +static int pidfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -1102,7 +1102,7 @@ static int pidfs_xattr_get(const struct xattr_handler *handler, } static int pidfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) { diff --git a/fs/pipe.c b/fs/pipe.c index 292425a834dd..5791db016e24 100644 --- a/fs/pipe.c +++ b/fs/pipe.c @@ -1433,7 +1433,7 @@ int pipe_resize_ring(struct pipe_inode_info *pipe, unsigned int nr_slots) spin_unlock_irq(&pipe->rd_wait.lock); /* This might have made more room for writers */ - wake_up_interruptible(&pipe->wr_wait); + wake_up_interruptible_poll(&pipe->wr_wait, EPOLLOUT | EPOLLWRNORM); return 0; } diff --git a/fs/pnode.c b/fs/pnode.c index 5d91c3e58d2a..2cd667958efe 100644 --- a/fs/pnode.c +++ b/fs/pnode.c @@ -410,19 +410,99 @@ bool propagation_would_overmount(const struct mount *from, return false; } +/* Does @m receive propagation from @parent? */ +static bool receives_from(struct mount *m, struct mount *parent) +{ + if (m == parent) + return false; + for (; m; m = m->mnt_master) + if (m == parent || peers(m, parent)) + return true; + return false; +} + +/* + * Does @m receive propagation from the victim's parent as well? If so, then + * the mount at the victim's mountpoint inside of @m is a umount candidate as + * well. So it's the next candidate in the chain. Otherwise the chain ends at + * @m. + */ +static struct mount *next_candidate(struct mount *m, struct mount *victim) +{ + if (!receives_from(m, victim->mnt_parent)) + return NULL; + return __lookup_mnt(&m->mnt, victim->mnt_mountpoint); +} + +/* + * Would propagate_umount() pull out a mount of the chain of candidates that + * starts at @c, and does that mount have references beyond its own? + * + * This mirrors how trim_one(), trim_ancestors() and handle_locked() handle a + * synchronous umount: + * + * - single victim + * - without children + * - with MNT_LOCKED already cleared on every candidate by propagate_mount_unlock() + * + * A copy of the victim gets unmounted when each of its children is + * the next candidate in the chain or its overmount, unless the next + * unmount candidate is not its overmount and some unmount candidate further + * down has a child outside the chain. Keep this in sync with + * Documentation/filesystems/propagate_umount.txt. + */ +static bool chain_busy(struct mount *c, struct mount *victim) +{ + struct mount *m, *n, *next, *deepest = NULL; + bool above; + + /* the deepest candidate with a child outside the chain */ + for (m = c; m; m = next) { + next = next_candidate(m, victim); + list_for_each_entry(n, &m->mnt_mounts, mnt_child) { + if (n != next && n != victim) { + deepest = m; + break; + } + } + } + + above = deepest != NULL; /* @deepest is at or below @m */ + for (m = c; m; m = next) { + bool goes = true; + + next = next_candidate(m, victim); + list_for_each_entry(n, &m->mnt_mounts, mnt_child) { + if (n != next && n != m->overmount && n != victim) { + goes = false; + break; + } + } + if (goes && next && next != m->overmount && above && m != deepest) + goes = false; + if (m == deepest) + above = false; + if (goes && do_refcount_check(m, 1)) + return true; + } + return false; +} + /* * check if the mount 'mnt' can be unmounted successfully. * @mnt: the mount to be checked for unmount * NOTE: unmounting 'mnt' would naturally propagate to all * other mounts its parent propagates to. - * Check if any of these mounts that **do not have submounts** - * have more references than 'refcnt'. If so return busy. + * Check if any of the mounts that propagate_umount() would pull out + * along with it have more references than their own. If so return busy. * * vfsmount lock must be held for write */ int propagate_mount_busy(struct mount *mnt, int refcnt) { struct mount *parent = mnt->mnt_parent; + struct dentry *mp = mnt->mnt_mountpoint; + struct mount *m; /* * quickly check if the current mount can be unmounted. @@ -435,24 +515,16 @@ int propagate_mount_busy(struct mount *mnt, int refcnt) if (mnt == parent) return 0; - for (struct mount *m = propagation_next(parent, parent); m; - m = propagation_next(m, parent)) { - struct list_head *head; - struct mount *child = __lookup_mnt(&m->mnt, mnt->mnt_mountpoint); + /* the candidates are the mounts at @mp below the receivers */ + for (m = propagation_next(parent, parent); m; + m = propagation_next(m, parent)) { + struct mount *c = __lookup_mnt(&m->mnt, mp); - if (!child) + /* each chain once, from its top: skip receivers that are candidates */ + if (!c || (mnt_has_parent(m) && m->mnt_mountpoint == mp && + receives_from(m->mnt_parent, parent))) continue; - - head = &child->mnt_mounts; - if (!list_empty(head)) { - /* - * a mount that covers child completely wouldn't prevent - * it being pulled out; any other would. - */ - if (!list_is_singular(head) || !child->overmount) - continue; - } - if (do_refcount_check(child, 1)) + if (chain_busy(c, mnt)) return 1; } return 0; diff --git a/fs/posix_acl.c b/fs/posix_acl.c index 18b302f94174..fe77934ea8f2 100644 --- a/fs/posix_acl.c +++ b/fs/posix_acl.c @@ -118,7 +118,7 @@ void forget_all_cached_acls(struct inode *inode) } EXPORT_SYMBOL(forget_all_cached_acls); -static struct posix_acl *__get_acl(struct mnt_idmap *idmap, +static struct posix_acl *__get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, int type) { @@ -378,7 +378,7 @@ EXPORT_SYMBOL(posix_acl_from_mode); * by the acl. Returns -E... otherwise. */ int -posix_acl_permission(struct mnt_idmap *idmap, struct inode *inode, +posix_acl_permission(const struct mnt_idmap *idmap, struct inode *inode, const struct posix_acl *acl, int want) { const struct posix_acl_entry *pa, *pe, *mask_obj; @@ -608,7 +608,7 @@ EXPORT_SYMBOL(__posix_acl_chmod); * performed on the raw inode simply pass @nop_mnt_idmap. */ int - posix_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry, + posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { struct inode *inode = d_inode(dentry); @@ -709,7 +709,7 @@ EXPORT_SYMBOL_GPL(posix_acl_create); * * Called from set_acl inode operations. */ -int posix_acl_update_mode(struct mnt_idmap *idmap, +int posix_acl_update_mode(const struct mnt_idmap *idmap, struct inode *inode, umode_t *mode_p, struct posix_acl **acl) { @@ -889,7 +889,7 @@ EXPORT_SYMBOL (posix_acl_to_xattr); * Return: On success, the size of the stored uapi posix acls, on error a * negative errno. */ -static ssize_t vfs_posix_acl_to_xattr(struct mnt_idmap *idmap, +static ssize_t vfs_posix_acl_to_xattr(const struct mnt_idmap *idmap, struct inode *inode, const struct posix_acl *acl, void *buffer, size_t size) @@ -937,7 +937,7 @@ static ssize_t vfs_posix_acl_to_xattr(struct mnt_idmap *idmap, } int -set_posix_acl(struct mnt_idmap *idmap, struct dentry *dentry, +set_posix_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type, struct posix_acl *acl) { struct inode *inode = d_inode(dentry); @@ -1018,7 +1018,7 @@ const struct xattr_handler nop_posix_acl_default = { }; EXPORT_SYMBOL_GPL(nop_posix_acl_default); -int simple_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int simple_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; @@ -1057,7 +1057,7 @@ int simple_acl_create(struct inode *dir, struct inode *inode) return 0; } -static int vfs_set_acl_idmapped_mnt(struct mnt_idmap *idmap, +static int vfs_set_acl_idmapped_mnt(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, struct posix_acl *acl) { @@ -1091,7 +1091,7 @@ static int vfs_set_acl_idmapped_mnt(struct mnt_idmap *idmap, * * Return: On success 0, on error negative errno. */ -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { int acl_type; @@ -1168,7 +1168,7 @@ EXPORT_SYMBOL_GPL(vfs_set_acl); * * Return: On success POSIX ACLs in VFS format, on error negative errno. */ -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { struct inode *inode = d_inode(dentry); @@ -1212,7 +1212,7 @@ EXPORT_SYMBOL_GPL(vfs_get_acl); * * Return: On success 0, on error negative errno. */ -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { int acl_type; @@ -1265,7 +1265,7 @@ out_inode_unlock: } EXPORT_SYMBOL_GPL(vfs_remove_acl); -int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size) { int error; @@ -1286,7 +1286,7 @@ int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, return error; } -ssize_t do_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size) { ssize_t error; diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62..c9ee0946ecaf 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -702,7 +702,7 @@ static int proc_pid_syscall(struct seq_file *m, struct pid_namespace *ns, /* Here the fs part begins */ /************************************************************************/ -int proc_nochmod_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int proc_nochmod_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int error; @@ -743,7 +743,7 @@ static bool has_pid_permissions(struct proc_fs_info *fs_info, } -static int proc_pid_permission(struct mnt_idmap *idmap, +static int proc_pid_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct proc_fs_info *fs_info = proc_sb_info(inode->i_sb); @@ -848,15 +848,35 @@ static int __mem_open(struct inode *inode, struct file *file, unsigned int mode) return 0; } +/* private_data for proc_mem_operations */ +struct mem_private { + struct mm_struct *mm; + /* + * Was the ptrace access check on open bypassed because the opener used + * the same MM (introspection)? + */ + bool opened_by_owner; +}; + static int mem_open(struct inode *inode, struct file *file) { + struct mem_private *priv __free(kfree) = kmalloc_obj(struct mem_private); + + if (!priv) + return -ENOMEM; if (WARN_ON_ONCE(!(file->f_op->fop_flags & FOP_UNSIGNED_OFFSET))) return -EINVAL; - return __mem_open(inode, file, PTRACE_MODE_ATTACH); + priv->mm = proc_mem_open(inode, PTRACE_MODE_ATTACH); + if (IS_ERR_OR_NULL(priv->mm)) + return priv->mm ? PTR_ERR(priv->mm) : -ESRCH; + priv->opened_by_owner = priv->mm == current->mm; + file->private_data = no_free_ptr(priv); + return 0; } static bool proc_mem_foll_force(struct file *file, struct mm_struct *mm) { + struct mem_private *priv = file->private_data; struct task_struct *task; bool ptrace_active = false; @@ -871,16 +891,20 @@ static bool proc_mem_foll_force(struct file *file, struct mm_struct *mm) READ_ONCE(task->parent) == current; put_task_struct(task); } - return ptrace_active; + if (!ptrace_active) + return false; + break; default: - return true; + break; } + return security_mem_foll_force(file->f_cred, priv->opened_by_owner) == 0; } static ssize_t mem_rw(struct file *file, char __user *buf, size_t count, loff_t *ppos, int write) { - struct mm_struct *mm = file->private_data; + struct mem_private *priv = file->private_data; + struct mm_struct *mm = priv->mm; unsigned long addr = *ppos; ssize_t copied; char *page; @@ -970,12 +994,21 @@ static int mem_release(struct inode *inode, struct file *file) return 0; } +static int mem_release_with_private(struct inode *inode, struct file *file) +{ + struct mem_private *priv = file->private_data; + + mmdrop(priv->mm); + kfree(priv); + return 0; +} + static const struct file_operations proc_mem_operations = { .llseek = mem_lseek, .read = mem_read, .write = mem_write, .open = mem_open, - .release = mem_release, + .release = mem_release_with_private, .fop_flags = FOP_UNSIGNED_OFFSET, }; @@ -1994,7 +2027,7 @@ static struct inode *proc_pid_make_base_inode(struct super_block *sb, return inode; } -int pid_getattr(struct mnt_idmap *idmap, const struct path *path, +int pid_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -3607,7 +3640,7 @@ int proc_pid_readdir(struct file *file, struct dir_context *ctx) * This function makes sure that the node is always accessible for members of * same thread group. */ -static int proc_tid_comm_permission(struct mnt_idmap *idmap, +static int proc_tid_comm_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { bool is_same_tgroup; @@ -3936,7 +3969,7 @@ static int proc_task_readdir(struct file *file, struct dir_context *ctx) return 0; } -static int proc_task_getattr(struct mnt_idmap *idmap, +static int proc_task_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/fd.c b/fs/proc/fd.c index 0f9a1556f2a3..7214a7495380 100644 --- a/fs/proc/fd.c +++ b/fs/proc/fd.c @@ -82,7 +82,7 @@ static int seq_fdinfo_open(struct inode *inode, struct file *file) * that the current task has PTRACE_MODE_READ in addition to the normal * POSIX-like checks. */ -static int proc_fdinfo_permission(struct mnt_idmap *idmap, struct inode *inode, +static int proc_fdinfo_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { bool allowed = false; @@ -323,7 +323,7 @@ static struct dentry *proc_lookupfd(struct inode *dir, struct dentry *dentry, * /proc/pid/fd needs a special permission handler so that a process can still * access /proc/self/fd after it has executed a setuid(). */ -int proc_fd_permission(struct mnt_idmap *idmap, +int proc_fd_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct task_struct *p; @@ -342,7 +342,7 @@ int proc_fd_permission(struct mnt_idmap *idmap, return rv; } -static int proc_fd_getattr(struct mnt_idmap *idmap, +static int proc_fd_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/fd.h b/fs/proc/fd.h index 7e7265f7e06f..77f2e4f38592 100644 --- a/fs/proc/fd.h +++ b/fs/proc/fd.h @@ -10,7 +10,7 @@ extern const struct inode_operations proc_fd_inode_operations; extern const struct file_operations proc_fdinfo_operations; extern const struct inode_operations proc_fdinfo_inode_operations; -extern int proc_fd_permission(struct mnt_idmap *idmap, +extern int proc_fd_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); static inline unsigned int proc_fd(struct inode *inode) diff --git a/fs/proc/generic.c b/fs/proc/generic.c index 26086a283672..2b1971da4a85 100644 --- a/fs/proc/generic.c +++ b/fs/proc/generic.c @@ -117,7 +117,7 @@ static bool pde_subdir_insert(struct proc_dir_entry *dir, return true; } -static int proc_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int proc_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -135,7 +135,7 @@ static int proc_setattr(struct mnt_idmap *idmap, struct dentry *dentry, return 0; } -static int proc_getattr(struct mnt_idmap *idmap, +static int proc_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 04bd6c9e65a7..b9aaac41c283 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -258,9 +258,9 @@ extern int proc_pid_statm(struct seq_file *, struct pid_namespace *, * base.c */ extern const struct dentry_operations pid_dentry_operations; -extern int pid_getattr(struct mnt_idmap *, const struct path *, +extern int pid_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); -int proc_nochmod_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int proc_nochmod_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void proc_pid_evict_inode(struct proc_inode *); extern struct inode *proc_pid_make_inode(struct super_block *, struct task_struct *, umode_t); diff --git a/fs/proc/proc_net.c b/fs/proc/proc_net.c index 00cc385bce21..b1f5eafb069a 100644 --- a/fs/proc/proc_net.c +++ b/fs/proc/proc_net.c @@ -308,7 +308,7 @@ static struct dentry *proc_tgid_net_lookup(struct inode *dir, return de; } -static int proc_tgid_net_getattr(struct mnt_idmap *idmap, +static int proc_tgid_net_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/proc_sysctl.c b/fs/proc/proc_sysctl.c index 04a382178c65..d1cfd2941359 100644 --- a/fs/proc/proc_sysctl.c +++ b/fs/proc/proc_sysctl.c @@ -788,7 +788,7 @@ out: return 0; } -static int proc_sys_permission(struct mnt_idmap *idmap, +static int proc_sys_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { /* @@ -817,7 +817,7 @@ static int proc_sys_permission(struct mnt_idmap *idmap, return error; } -static int proc_sys_setattr(struct mnt_idmap *idmap, +static int proc_sys_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -834,7 +834,7 @@ static int proc_sys_setattr(struct mnt_idmap *idmap, return 0; } -static int proc_sys_getattr(struct mnt_idmap *idmap, +static int proc_sys_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/root.c b/fs/proc/root.c index 99adddfeb4a4..7fbbe92bf73a 100644 --- a/fs/proc/root.c +++ b/fs/proc/root.c @@ -402,7 +402,7 @@ void __init proc_root_init(void) register_filesystem(&proc_fs_type); } -static int proc_root_getattr(struct mnt_idmap *idmap, +static int proc_root_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/vmcore.c b/fs/proc/vmcore.c index 44d15436439f..406898247d7a 100644 --- a/fs/proc/vmcore.c +++ b/fs/proc/vmcore.c @@ -1709,6 +1709,24 @@ static void vmcore_free_device_dumps(void) #endif /* CONFIG_PROC_VMCORE_DEVICE_DUMP */ } +#define VMCOREINFO_OSRELEASE_KEY "OSRELEASE=" + +static void __init vmcore_report_crashed_release(void) +{ + const char *ver, *eol; + + ver = strnstr(elfnotes_buf, VMCOREINFO_OSRELEASE_KEY, elfnotes_sz); + if (!ver) + return; + + ver += sizeof(VMCOREINFO_OSRELEASE_KEY) - 1; + eol = memchr(ver, '\n', elfnotes_buf + elfnotes_sz - ver); + if (!eol) + return; + + pr_notice("dump is from kernel %.*s\n", (int)(eol - ver), ver); +} + /* Init function for vmcore module. */ static int __init vmcore_init(void) { @@ -1733,6 +1751,8 @@ static int __init vmcore_init(void) elfcorehdr_free(elfcorehdr_addr); elfcorehdr_addr = ELFCORE_ADDR_ERR; + vmcore_report_crashed_release(); + proc_vmcore = proc_create("vmcore", S_IRUSR, NULL, &vmcore_proc_ops); if (proc_vmcore) proc_vmcore->size = vmcore_size; diff --git a/fs/quota/dquot.c b/fs/quota/dquot.c index cde707fad46f..3ad19d7e472f 100644 --- a/fs/quota/dquot.c +++ b/fs/quota/dquot.c @@ -2095,7 +2095,7 @@ EXPORT_SYMBOL(__dquot_transfer); /* Wrapper for transferring ownership of an inode for uid/gid only * Called from FSXXX_setattr() */ -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { struct dquot *transfer_to[MAXQUOTAS] = {}; diff --git a/fs/ramfs/file-nommu.c b/fs/ramfs/file-nommu.c index fb471bf88ab7..7ed6fba134c6 100644 --- a/fs/ramfs/file-nommu.c +++ b/fs/ramfs/file-nommu.c @@ -22,7 +22,7 @@ #include <linux/uaccess.h> #include "internal.h" -static int ramfs_nommu_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +static int ramfs_nommu_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); static unsigned long ramfs_nommu_get_unmapped_area(struct file *file, unsigned long addr, unsigned long len, @@ -161,7 +161,7 @@ static int ramfs_nommu_resize(struct inode *inode, loff_t newsize, loff_t size) * handle a change of attributes * - we're specifically interested in a change of size */ -static int ramfs_nommu_setattr(struct mnt_idmap *idmap, +static int ramfs_nommu_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { struct inode *inode = d_inode(dentry); diff --git a/fs/ramfs/inode.c b/fs/ramfs/inode.c index 0a88ede48e0a..ef91b933d4e6 100644 --- a/fs/ramfs/inode.c +++ b/fs/ramfs/inode.c @@ -95,7 +95,7 @@ struct inode *ramfs_get_inode(struct super_block *sb, */ /* SMP-safe */ static int -ramfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +ramfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode * inode = ramfs_get_inode(dir->i_sb, dir, mode, dev); @@ -118,7 +118,7 @@ out: return error; } -static struct dentry *ramfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ramfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int retval = ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); @@ -127,13 +127,13 @@ static struct dentry *ramfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, return ERR_PTR(retval); } -static int ramfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ramfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFREG, 0); } -static int ramfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ramfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -163,7 +163,7 @@ out: return error; } -static int ramfs_tmpfile(struct mnt_idmap *idmap, +static int ramfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode; diff --git a/fs/read_write.c b/fs/read_write.c index e8c14e2760b2..4da846a2bf17 100644 --- a/fs/read_write.c +++ b/fs/read_write.c @@ -274,7 +274,7 @@ loff_t fixed_size_llseek(struct file *file, loff_t offset, int whence, loff_t si EXPORT_SYMBOL(fixed_size_llseek); /** - * no_seek_end_llseek - llseek implementation for fixed-sized devices + * no_seek_end_llseek - llseek implementation for files without SEEK_END * @file: file structure to seek on * @offset: file offset to seek to * @whence: type of seek @@ -293,7 +293,7 @@ loff_t no_seek_end_llseek(struct file *file, loff_t offset, int whence) EXPORT_SYMBOL(no_seek_end_llseek); /** - * no_seek_end_llseek_size - llseek implementation for fixed-sized devices + * no_seek_end_llseek_size - llseek implementation for files without SEEK_END * @file: file structure to seek on * @offset: file offset to seek to * @whence: type of seek @@ -1761,8 +1761,8 @@ EXPORT_SYMBOL(generic_write_checks_count); * Performs necessary checks before doing a write * * Can adjust writing position or amount of bytes to write. - * Returns appropriate error code that caller should return or - * zero in case that write should be allowed. + * Returns the number of bytes that may be written on success (which + * may be less than requested if truncated), or a negative error code. */ ssize_t generic_write_checks(struct kiocb *iocb, struct iov_iter *from) { diff --git a/fs/remap_range.c b/fs/remap_range.c index 26afbbbfb10c..6eb7d845de5d 100644 --- a/fs/remap_range.c +++ b/fs/remap_range.c @@ -415,7 +415,7 @@ EXPORT_SYMBOL(vfs_clone_file_range); /* Check whether we are allowed to dedupe the destination file */ static bool may_dedupe_file(struct file *file) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct inode *inode = file_inode(file); if (capable(CAP_SYS_ADMIN)) diff --git a/fs/smb/client/cifsacl.c b/fs/smb/client/cifsacl.c index c5e47a835f99..d1a92bb4d9a5 100644 --- a/fs/smb/client/cifsacl.c +++ b/fs/smb/client/cifsacl.c @@ -1904,7 +1904,7 @@ id_mode_to_cifs_acl_exit: return rc; } -struct posix_acl *cifs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *cifs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { #if defined(CONFIG_CIFS_ALLOW_INSECURE_LEGACY) && defined(CONFIG_CIFS_POSIX) @@ -1968,7 +1968,7 @@ out: #endif } -int cifs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int cifs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { #if defined(CONFIG_CIFS_ALLOW_INSECURE_LEGACY) && defined(CONFIG_CIFS_POSIX) diff --git a/fs/smb/client/cifsfs.c b/fs/smb/client/cifsfs.c index 7ecd70efdfea..b1ecbcfb154e 100644 --- a/fs/smb/client/cifsfs.c +++ b/fs/smb/client/cifsfs.c @@ -402,7 +402,7 @@ out_unlock: return rc; } -static int cifs_permission(struct mnt_idmap *idmap, +static int cifs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { unsigned int sbflags = cifs_sb_flags(CIFS_SB(inode)); diff --git a/fs/smb/client/cifsfs.h b/fs/smb/client/cifsfs.h index 0c85daa8386e..51692e14c4dd 100644 --- a/fs/smb/client/cifsfs.h +++ b/fs/smb/client/cifsfs.h @@ -53,23 +53,23 @@ void cifs_sb_deactive(struct super_block *sb); /* Functions related to inodes */ extern const struct inode_operations cifs_dir_inode_ops; struct inode *cifs_root_iget(struct super_block *sb); -int cifs_create(struct mnt_idmap *idmap, struct inode *dir, +int cifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *direntry, umode_t mode); int cifs_atomic_open(struct inode *dir, struct dentry *direntry, struct file *file, unsigned int oflags, umode_t mode); -int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int cifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode); struct dentry *cifs_lookup(struct inode *parent_dir_inode, struct dentry *direntry, unsigned int flags); int cifs_unlink(struct inode *dir, struct dentry *dentry); int cifs_hardlink(struct dentry *old_file, struct inode *inode, struct dentry *direntry); -int cifs_mknod(struct mnt_idmap *idmap, struct inode *inode, +int cifs_mknod(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode, dev_t device_number); -struct dentry *cifs_mkdir(struct mnt_idmap *idmap, struct inode *inode, +struct dentry *cifs_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode); int cifs_rmdir(struct inode *inode, struct dentry *direntry); -int cifs_rename2(struct mnt_idmap *idmap, struct inode *source_dir, +int cifs_rename2(const struct mnt_idmap *idmap, struct inode *source_dir, struct dentry *source_dentry, struct inode *target_dir, struct dentry *target_dentry, unsigned int flags); int cifs_revalidate_file_attr(struct file *filp); @@ -78,9 +78,9 @@ int cifs_revalidate_file(struct file *filp); int cifs_revalidate_dentry(struct dentry *dentry); int cifs_revalidate_mapping(struct inode *inode); int cifs_zap_mapping(struct inode *inode); -int cifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int cifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int cifs_setattr(struct mnt_idmap *idmap, struct dentry *direntry, +int cifs_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs); int cifs_fiemap(struct inode *inode, struct fiemap_extent_info *fei, u64 start, u64 len); @@ -129,7 +129,7 @@ struct vfsmount *cifs_d_automount(struct path *path); /* Functions related to symlinks */ const char *cifs_get_link(struct dentry *dentry, struct inode *inode, struct delayed_call *done); -int cifs_symlink(struct mnt_idmap *idmap, struct inode *inode, +int cifs_symlink(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, const char *symname); #ifdef CONFIG_CIFS_XATTR diff --git a/fs/smb/client/cifsproto.h b/fs/smb/client/cifsproto.h index 00168839c123..e6beff8aafe0 100644 --- a/fs/smb/client/cifsproto.h +++ b/fs/smb/client/cifsproto.h @@ -212,9 +212,9 @@ struct smb_ntsd *get_cifs_acl(struct cifs_sb_info *cifs_sb, struct smb_ntsd *get_cifs_acl_by_fid(struct cifs_sb_info *cifs_sb, const struct cifs_fid *cifsfid, u32 *pacllen, u32 info); -struct posix_acl *cifs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *cifs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int cifs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int cifs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int set_cifs_acl(struct smb_ntsd *pnntsd, __u32 acllen, struct inode *inode, const char *path, int aclflag); diff --git a/fs/smb/client/dir.c b/fs/smb/client/dir.c index 1a56fa4d0e89..a2cf3d35ca5d 100644 --- a/fs/smb/client/dir.c +++ b/fs/smb/client/dir.c @@ -664,7 +664,7 @@ out_free_xid: * The initial dentry state is hashed-negative. On success, dentry will become * hashed-positive by calling d_instantiate(). */ -int cifs_create(struct mnt_idmap *idmap, struct inode *dir, +int cifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *direntry, umode_t mode) { struct cifs_sb_info *cifs_sb = CIFS_SB(dir); @@ -721,7 +721,7 @@ out_free_xid: return rc; } -int cifs_mknod(struct mnt_idmap *idmap, struct inode *inode, +int cifs_mknod(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode, dev_t device_number) { int rc = -EPERM; @@ -1084,7 +1084,7 @@ static int set_tmpfile_attr(const unsigned int xid, unsigned int oflags, * The initial dentry state is unhashed-negative. On success, dentry will * become unhashed-positive by calling d_instantiate(). */ -int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int cifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct dentry *dentry = file->f_path.dentry; diff --git a/fs/smb/client/inode.c b/fs/smb/client/inode.c index 1fe0ef0a95db..4d7a87c7b73a 100644 --- a/fs/smb/client/inode.c +++ b/fs/smb/client/inode.c @@ -2281,7 +2281,7 @@ posix_mkdir_get_info: } #endif /* CONFIG_CIFS_ALLOW_INSECURE_LEGACY */ -struct dentry *cifs_mkdir(struct mnt_idmap *idmap, struct inode *inode, +struct dentry *cifs_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode) { int rc = 0; @@ -2531,7 +2531,7 @@ do_rename_exit: } int -cifs_rename2(struct mnt_idmap *idmap, struct inode *source_dir, +cifs_rename2(const struct mnt_idmap *idmap, struct inode *source_dir, struct dentry *source_dentry, struct inode *target_dir, struct dentry *target_dentry, unsigned int flags) { @@ -2937,7 +2937,7 @@ int cifs_revalidate_dentry(struct dentry *dentry) return cifs_revalidate_mapping(inode); } -int cifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int cifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct cifs_sb_info *cifs_sb = CIFS_SB(path->dentry); @@ -3554,7 +3554,7 @@ cifs_setattr_exit: } int -cifs_setattr(struct mnt_idmap *idmap, struct dentry *direntry, +cifs_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs) { struct cifs_sb_info *cifs_sb = CIFS_SB(direntry->d_sb); diff --git a/fs/smb/client/link.c b/fs/smb/client/link.c index 8d5d6aca742a..76df31abeaea 100644 --- a/fs/smb/client/link.c +++ b/fs/smb/client/link.c @@ -533,7 +533,7 @@ cifs_hl_exit: } int -cifs_symlink(struct mnt_idmap *idmap, struct inode *inode, +cifs_symlink(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, const char *symname) { struct cifs_sb_info *cifs_sb = CIFS_SB(inode); diff --git a/fs/smb/client/transport.c b/fs/smb/client/transport.c index 6e21b5f8754a..93ff6a4dbb35 100644 --- a/fs/smb/client/transport.c +++ b/fs/smb/client/transport.c @@ -22,7 +22,6 @@ #include <linux/mempool.h> #include <linux/sched/signal.h> #include <linux/task_io_accounting_ops.h> -#include <linux/task_work.h> #include "cifsglob.h" #include "cifsproto.h" #include "cifs_debug.h" @@ -171,15 +170,11 @@ smb_send_kvec(struct TCP_Server_Info *server, struct msghdr *smb_msg, * after the retries we will kill the socket and * reconnect which may clear the network problem. * - * Even if regular signals are masked, EINTR might be - * propagated from sk_stream_wait_memory() to here when - * TIF_NOTIFY_SIGNAL is used for task work. For example, - * certain io_uring completions will use that. Treat - * having EINTR with pending task work the same as EAGAIN - * to avoid unnecessary reconnects. + * Task work must not abort the send, see signal_pending(). */ - rc = sock_sendmsg(ssocket, smb_msg); - if (rc == -EAGAIN || unlikely(rc == -EINTR && task_work_pending(current))) { + scoped_guard(no_notify_signal) + rc = sock_sendmsg(ssocket, smb_msg); + if (rc == -EAGAIN) { retries++; if (retries >= 14 || (!server->noblocksnd && (retries > 2))) { diff --git a/fs/smb/client/xattr.c b/fs/smb/client/xattr.c index 5091f6c0d7fe..f6c9343016f7 100644 --- a/fs/smb/client/xattr.c +++ b/fs/smb/client/xattr.c @@ -91,7 +91,7 @@ static int cifs_creation_time_set(unsigned int xid, struct cifs_tcon *pTcon, } static int cifs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/smb/server/ndr.c b/fs/smb/server/ndr.c index 58d71560f626..7e546c22e284 100644 --- a/fs/smb/server/ndr.c +++ b/fs/smb/server/ndr.c @@ -338,7 +338,7 @@ static int ndr_encode_posix_acl_entry(struct ndr *n, struct xattr_smb_acl *acl) } int ndr_encode_posix_acl(struct ndr *n, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode, struct xattr_smb_acl *acl, struct xattr_smb_acl *def_acl) diff --git a/fs/smb/server/ndr.h b/fs/smb/server/ndr.h index f3c108c8cf4d..646568c42e4d 100644 --- a/fs/smb/server/ndr.h +++ b/fs/smb/server/ndr.h @@ -14,7 +14,7 @@ struct ndr { int ndr_encode_dos_attr(struct ndr *n, struct xattr_dos_attrib *da); int ndr_decode_dos_attr(struct ndr *n, struct xattr_dos_attrib *da); -int ndr_encode_posix_acl(struct ndr *n, struct mnt_idmap *idmap, +int ndr_encode_posix_acl(struct ndr *n, const struct mnt_idmap *idmap, struct inode *inode, struct xattr_smb_acl *acl, struct xattr_smb_acl *def_acl); int ndr_encode_v4_ntacl(struct ndr *n, struct xattr_ntacl *acl); diff --git a/fs/smb/server/oplock.c b/fs/smb/server/oplock.c index 1b8c3482d1e4..d0f18ebf471b 100644 --- a/fs/smb/server/oplock.c +++ b/fs/smb/server/oplock.c @@ -2270,7 +2270,7 @@ void create_posix_rsp_buf(char *cc, struct ksmbd_file *fp) { struct create_posix_rsp *buf; struct inode *inode = file_inode(fp->filp); - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c index 2f19ec9afa50..45dce9c30b6b 100644 --- a/fs/smb/server/smb2pdu.c +++ b/fs/smb/server/smb2pdu.c @@ -3297,7 +3297,7 @@ static bool smb2_is_private_ea(const char *name, size_t name_len) static int smb2_set_ea(struct smb2_ea_info *eabuf, unsigned int buf_len, const struct path *path, bool get_write) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); char *attr_name = NULL, *value; int rc = 0; unsigned int next = 0; @@ -3398,7 +3398,7 @@ static noinline int smb2_set_stream_name_xattr(const struct path *path, struct ksmbd_file *fp, char *stream_name, int s_type) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); size_t xattr_stream_size; char *xattr_stream_name; int rc; @@ -3475,7 +3475,7 @@ static loff_t ksmbd_stream_eof(struct ksmbd_file *fp) static int smb2_remove_smb_xattrs(const struct path *path) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); char *name, *xattr_list = NULL; ssize_t xattr_list_len; int err = 0; @@ -3668,7 +3668,7 @@ static int smb2_create_sd_buffer(struct ksmbd_work *work, } static int ksmbd_acls_fattr(struct smb_fattr *fattr, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode) { struct posix_acl *acl; @@ -4171,7 +4171,7 @@ int smb2_open(struct ksmbd_work *work) struct ksmbd_share_config *share = tcon->share_conf; struct ksmbd_file *fp = NULL; struct file *filp = NULL; - struct mnt_idmap *idmap = NULL; + const struct mnt_idmap *idmap = NULL; struct kstat stat; struct create_context *context; struct lease_ctx_info *lc = NULL; @@ -5839,7 +5839,7 @@ struct smb2_query_dir_private { static int process_query_dir_entries(struct smb2_query_dir_private *priv) { - struct mnt_idmap *idmap = file_mnt_idmap(priv->dir_fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(priv->dir_fp->filp); struct kstat kstat; struct ksmbd_kstat ksmbd_kstat; int rc; @@ -6432,7 +6432,7 @@ static int smb2_get_ea(struct ksmbd_work *work, struct ksmbd_file *fp, ssize_t buf_free_len, alignment_bytes, next_offset, rsp_data_cnt = 0; struct smb2_ea_info_req *ea_req = NULL; const struct path *path; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); if (!(fp->daccess & FILE_READ_EA_LE)) { pr_err("Not permitted to read ext attr : 0x%x\n", @@ -7141,7 +7141,7 @@ static int find_file_posix_info(struct smb2_query_info_rsp *rsp, { struct smb311_posix_qinfo *file_info; struct inode *inode = file_inode(fp->filp); - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); struct kstat stat; @@ -7633,7 +7633,7 @@ static int smb2_get_info_sec(struct ksmbd_work *work, struct smb2_query_info_rsp *rsp) { struct ksmbd_file *fp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct smb_ntsd *pntsd = NULL, *ppntsd = NULL; struct smb_fattr fattr = {{0}}; struct inode *inode; @@ -8175,7 +8175,7 @@ static int set_file_basic_info(struct ksmbd_file *fp, struct iattr attrs; struct file *filp; struct inode *inode; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; __le32 attrs_mask = FILE_ATTRIBUTE_DIRECTORY_LE | FILE_ATTRIBUTE_COMPRESSED_LE; int rc = 0; @@ -10793,7 +10793,7 @@ static inline int fsctl_set_sparse(struct ksmbd_work *work, u64 id, struct file_sparse *sparse) { struct ksmbd_file *fp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; int ret = 0; __le32 old_fattr; diff --git a/fs/smb/server/smb_common.c b/fs/smb/server/smb_common.c index 086a1b85e5f4..7dbcfa658edd 100644 --- a/fs/smb/server/smb_common.c +++ b/fs/smb/server/smb_common.c @@ -467,7 +467,7 @@ int ksmbd_populate_dot_dotdot_entries(struct ksmbd_work *work, int info_level, { int i, rc = 0; struct ksmbd_conn *conn = work->conn; - struct mnt_idmap *idmap = file_mnt_idmap(dir->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(dir->filp); for (i = 0; i < 2; i++) { struct kstat kstat; diff --git a/fs/smb/server/smbacl.c b/fs/smb/server/smbacl.c index e75247915c87..96428df33b43 100644 --- a/fs/smb/server/smbacl.c +++ b/fs/smb/server/smbacl.c @@ -258,7 +258,7 @@ void id_to_sid(unsigned int cid, uint sidtype, struct smb_sid *ssid) ssid->num_subauth++; } -static int sid_to_id(struct mnt_idmap *idmap, +static int sid_to_id(const struct mnt_idmap *idmap, struct smb_sid *psid, uint sidtype, struct smb_fattr *fattr) { @@ -384,7 +384,7 @@ void free_acl_state(struct posix_acl_state *state) kfree(state->groups); } -static int parse_dacl(struct mnt_idmap *idmap, +static int parse_dacl(const struct mnt_idmap *idmap, struct smb_acl *pdacl, char *end_of_acl, struct smb_sid *pownersid, struct smb_sid *pgrpsid, struct smb_fattr *fattr) @@ -620,7 +620,7 @@ out: return ret; } -static void set_posix_acl_entries_dacl(struct mnt_idmap *idmap, +static void set_posix_acl_entries_dacl(const struct mnt_idmap *idmap, struct smb_ace *pndace, struct smb_fattr *fattr, u16 *num_aces, u16 *size, u16 existing_nt_aces, @@ -751,7 +751,7 @@ posix_default_acl: } } -static void set_ntacl_dacl(struct mnt_idmap *idmap, +static void set_ntacl_dacl(const struct mnt_idmap *idmap, struct smb_acl *pndacl, struct smb_acl *nt_dacl, unsigned int aces_size, @@ -810,7 +810,7 @@ next_ace: pndacl->size = cpu_to_le16(le16_to_cpu(pndacl->size) + size); } -static void set_mode_dacl(struct mnt_idmap *idmap, +static void set_mode_dacl(const struct mnt_idmap *idmap, struct smb_acl *pndacl, struct smb_fattr *fattr) { struct smb_ace *pace, *pndace; @@ -896,7 +896,7 @@ static int parse_sid(struct smb_sid *psid, char *end_of_acl) } /* Convert CIFS ACL to POSIX form */ -int parse_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int parse_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, int acl_len, struct smb_fattr *fattr) { int rc = 0; @@ -1031,7 +1031,7 @@ size_t smb_acl_sec_desc_scratch_len(struct smb_fattr *fattr, } /* Convert permission bits from mode to equivalent CIFS ACL */ -int build_sec_desc(struct mnt_idmap *idmap, +int build_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info, __u32 *secdesclen, struct smb_fattr *fattr) @@ -1200,7 +1200,7 @@ int smb_inherit_dacl(struct ksmbd_conn *conn, struct smb_ntsd *parent_pntsd = NULL; struct smb_sid owner_sid, group_sid; struct dentry *parent = path->dentry->d_parent; - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); int inherited_flags = 0, flags = 0, i, nt_size = 0, pdacl_size; int rc = 0, pntsd_type, ppntsd_size, acl_len, aces_size; unsigned int dacloffset; @@ -1455,7 +1455,7 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path, __le32 *pdaccess, __le32 raw_daccess, int uid, bool strict) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); struct smb_ntsd *pntsd = NULL; struct smb_acl *pdacl; struct posix_acl *posix_acls; @@ -1678,7 +1678,7 @@ int set_info_sec(struct ksmbd_conn *conn, struct ksmbd_tree_connect *tcon, int rc; struct smb_fattr fattr = {{0}}; struct inode *inode = d_inode(path->dentry); - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); struct iattr newattrs; fattr.cf_uid = INVALID_UID; diff --git a/fs/smb/server/smbacl.h b/fs/smb/server/smbacl.h index 01810c16cc04..28d215807faa 100644 --- a/fs/smb/server/smbacl.h +++ b/fs/smb/server/smbacl.h @@ -81,9 +81,9 @@ struct posix_acl_state { struct posix_ace_state_array *groups; }; -int parse_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int parse_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, int acl_len, struct smb_fattr *fattr); -int build_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int build_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info, __u32 *secdesclen, struct smb_fattr *fattr); int init_acl_state(struct posix_acl_state *state, u16 cnt); @@ -105,7 +105,7 @@ void ksmbd_init_domain(u32 *sub_auth); size_t smb_acl_sec_desc_scratch_len(struct smb_fattr *fattr, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info); -static inline uid_t posix_acl_uid_translate(struct mnt_idmap *idmap, +static inline uid_t posix_acl_uid_translate(const struct mnt_idmap *idmap, struct posix_acl_entry *pace) { vfsuid_t vfsuid; @@ -117,7 +117,7 @@ static inline uid_t posix_acl_uid_translate(struct mnt_idmap *idmap, return from_kuid(&init_user_ns, vfsuid_into_kuid(vfsuid)); } -static inline gid_t posix_acl_gid_translate(struct mnt_idmap *idmap, +static inline gid_t posix_acl_gid_translate(const struct mnt_idmap *idmap, struct posix_acl_entry *pace) { vfsgid_t vfsgid; diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c index db0f2de2bab3..2e2c554bc2c1 100644 --- a/fs/smb/server/vfs.c +++ b/fs/smb/server/vfs.c @@ -116,7 +116,7 @@ static int ksmbd_vfs_path_lookup(struct ksmbd_share_config *share_conf, return 0; } -void ksmbd_vfs_query_maximal_access(struct mnt_idmap *idmap, +void ksmbd_vfs_query_maximal_access(const struct mnt_idmap *idmap, struct dentry *dentry, __le32 *daccess) { *daccess = cpu_to_le32(FILE_READ_ATTRIBUTES | READ_CONTROL); @@ -184,7 +184,7 @@ int ksmbd_vfs_create(struct ksmbd_work *work, const char *name, umode_t mode) */ int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct path path; struct dentry *dentry, *d; int err = 0; @@ -217,7 +217,7 @@ int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode) return err; } -ssize_t ksmbd_vfs_getcasexattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getcasexattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len, char **attr_value) { @@ -387,7 +387,7 @@ static int ksmbd_vfs_stream_write(struct ksmbd_file *fp, char *buf, loff_t *pos, { const struct cred *saved_cred; char *stream_buf = NULL, *wbuf; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); size_t size; ssize_t v_len; int err = 0; @@ -578,7 +578,7 @@ int ksmbd_vfs_fsync(struct ksmbd_work *work, u64 fid, u64 p_id) */ int ksmbd_vfs_remove_file(struct ksmbd_work *work, const struct path *path) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *parent = path->dentry->d_parent; int err; @@ -842,7 +842,7 @@ ssize_t ksmbd_vfs_listxattr(struct dentry *dentry, char **list) return size; } -ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_xattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name) { return vfs_getxattr(idmap, dentry, xattr_name, NULL, 0); @@ -857,7 +857,7 @@ ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, * * Return: read xattr value length on success, otherwise error */ -ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name, char **xattr_buf) { @@ -894,7 +894,7 @@ ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, * * Return: 0 on success, otherwise error */ -int ksmbd_vfs_setxattr(struct mnt_idmap *idmap, +int ksmbd_vfs_setxattr(const struct mnt_idmap *idmap, const struct path *path, const char *attr_name, void *attr_value, size_t attr_size, int flags, bool get_write) @@ -1178,7 +1178,7 @@ int ksmbd_vfs_query_allocated_ranges(struct ksmbd_file *fp, loff_t start, return ret; } -int ksmbd_vfs_remove_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_xattr(const struct mnt_idmap *idmap, const struct path *path, char *attr_name, bool get_write) { @@ -1203,7 +1203,7 @@ int ksmbd_vfs_unlink(struct file *filp) const struct cred *saved_cred; int err = 0; struct dentry *dir, *dentry = filp->f_path.dentry; - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); saved_cred = override_creds(filp->f_cred); err = mnt_want_write(filp->f_path.mnt); @@ -1472,7 +1472,7 @@ struct dentry *ksmbd_vfs_kern_path_create(struct ksmbd_work *work, return dent; } -int ksmbd_vfs_remove_acl_xattrs(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_acl_xattrs(const struct mnt_idmap *idmap, const struct path *path) { char *name, *xattr_list = NULL; @@ -1512,7 +1512,7 @@ out: return err; } -int ksmbd_vfs_remove_sd_xattrs(struct mnt_idmap *idmap, const struct path *path) +int ksmbd_vfs_remove_sd_xattrs(const struct mnt_idmap *idmap, const struct path *path) { char *name, *xattr_list = NULL; ssize_t xattr_list_len; @@ -1541,7 +1541,7 @@ out: return err; } -static struct xattr_smb_acl *ksmbd_vfs_make_xattr_posix_acl(struct mnt_idmap *idmap, +static struct xattr_smb_acl *ksmbd_vfs_make_xattr_posix_acl(const struct mnt_idmap *idmap, struct inode *inode, int acl_type) { @@ -1607,7 +1607,7 @@ out: } int ksmbd_vfs_set_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct smb_ntsd *pntsd, int len, bool get_write) @@ -1675,7 +1675,7 @@ out: EXPORT_SYMBOL_IF_KUNIT(ksmbd_vfs_set_sd_xattr); int ksmbd_vfs_get_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct smb_ntsd **pntsd) { @@ -1744,7 +1744,7 @@ out_free: return rc; } -int ksmbd_vfs_set_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_set_dos_attrib_xattr(const struct mnt_idmap *idmap, const struct path *path, struct xattr_dos_attrib *da, bool get_write) @@ -1766,7 +1766,7 @@ out: return err; } -int ksmbd_vfs_get_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_get_dos_attrib_xattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct xattr_dos_attrib *da) { @@ -1822,7 +1822,7 @@ void *ksmbd_vfs_init_kstat(char **p, struct ksmbd_kstat *ksmbd_kstat) } int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct ksmbd_kstat *ksmbd_kstat) { @@ -1897,7 +1897,7 @@ int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, return 0; } -ssize_t ksmbd_vfs_casexattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_casexattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len) { @@ -2212,7 +2212,7 @@ void ksmbd_vfs_posix_lock_unblock(struct file_lock *flock) locks_delete_block(flock); } -int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_set_init_posix_acl(const struct mnt_idmap *idmap, const struct path *path) { struct posix_acl_state acl_state; @@ -2265,7 +2265,7 @@ int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, return rc; } -int ksmbd_vfs_inherit_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_inherit_posix_acl(const struct mnt_idmap *idmap, const struct path *path, struct inode *parent_inode) { struct posix_acl *acls; @@ -2328,7 +2328,7 @@ static int __ksmbd_vfs_set_compression(struct ksmbd_work *work, const struct cred *saved_cred = NULL; struct file_kattr fa; struct dentry *dentry = fp->filp->f_path.dentry; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); u32 flags; __le32 old_fattr; int rc; diff --git a/fs/smb/server/vfs.h b/fs/smb/server/vfs.h index 566c670c90be..216c76291fc1 100644 --- a/fs/smb/server/vfs.h +++ b/fs/smb/server/vfs.h @@ -74,7 +74,7 @@ struct ksmbd_kstat { }; int ksmbd_vfs_lock_parent(struct dentry *parent, struct dentry *child); -void ksmbd_vfs_query_maximal_access(struct mnt_idmap *idmap, +void ksmbd_vfs_query_maximal_access(const struct mnt_idmap *idmap, struct dentry *dentry, __le32 *daccess); int ksmbd_vfs_create(struct ksmbd_work *work, const char *name, umode_t mode); int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode); @@ -104,25 +104,25 @@ int ksmbd_vfs_copy_file_ranges(struct ksmbd_work *work, unsigned int *chunk_size_written, loff_t *total_size_written); ssize_t ksmbd_vfs_listxattr(struct dentry *dentry, char **list); -ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name, char **xattr_buf); -ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_xattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name); -ssize_t ksmbd_vfs_getcasexattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getcasexattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len, char **attr_value); -ssize_t ksmbd_vfs_casexattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_casexattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len); -int ksmbd_vfs_setxattr(struct mnt_idmap *idmap, +int ksmbd_vfs_setxattr(const struct mnt_idmap *idmap, const struct path *path, const char *attr_name, void *attr_value, size_t attr_size, int flags, bool get_write); int ksmbd_vfs_xattr_stream_name(char *stream_name, char **xattr_stream_name, size_t *xattr_stream_name_size, int s_type); -int ksmbd_vfs_remove_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_xattr(const struct mnt_idmap *idmap, const struct path *path, char *attr_name, bool get_write); int ksmbd_vfs_kern_path(struct ksmbd_work *work, char *name, @@ -152,33 +152,33 @@ int ksmbd_vfs_query_allocated_ranges(struct ksmbd_file *fp, loff_t start, int ksmbd_vfs_unlink(struct file *filp); void *ksmbd_vfs_init_kstat(char **p, struct ksmbd_kstat *ksmbd_kstat); int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct ksmbd_kstat *ksmbd_kstat); void ksmbd_vfs_posix_lock_wait(struct file_lock *flock); void ksmbd_vfs_posix_lock_unblock(struct file_lock *flock); -int ksmbd_vfs_remove_acl_xattrs(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_acl_xattrs(const struct mnt_idmap *idmap, const struct path *path); -int ksmbd_vfs_remove_sd_xattrs(struct mnt_idmap *idmap, const struct path *path); +int ksmbd_vfs_remove_sd_xattrs(const struct mnt_idmap *idmap, const struct path *path); int ksmbd_vfs_set_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct smb_ntsd *pntsd, int len, bool get_write); int ksmbd_vfs_get_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct smb_ntsd **pntsd); -int ksmbd_vfs_set_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_set_dos_attrib_xattr(const struct mnt_idmap *idmap, const struct path *path, struct xattr_dos_attrib *da, bool get_write); -int ksmbd_vfs_get_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_get_dos_attrib_xattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct xattr_dos_attrib *da); -int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_set_init_posix_acl(const struct mnt_idmap *idmap, const struct path *path); -int ksmbd_vfs_inherit_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_inherit_posix_acl(const struct mnt_idmap *idmap, const struct path *path, struct inode *parent_inode); void ksmbd_vfs_update_compressed_fattr(struct dentry *dentry, __le32 *fattr); diff --git a/fs/splice.c b/fs/splice.c index 9d8f63e2fd1a..bc243ab8dc43 100644 --- a/fs/splice.c +++ b/fs/splice.c @@ -177,9 +177,9 @@ static const struct pipe_buf_operations user_page_pipe_buf_ops = { static void wakeup_pipe_readers(struct pipe_inode_info *pipe) { - smp_mb(); - if (waitqueue_active(&pipe->rd_wait)) - wake_up_interruptible(&pipe->rd_wait); + if (wq_has_sleeper(&pipe->rd_wait)) + wake_up_interruptible_poll(&pipe->rd_wait, + EPOLLIN | EPOLLRDNORM); kill_fasync(&pipe->fasync_readers, SIGIO, POLL_IN); } @@ -413,9 +413,9 @@ EXPORT_SYMBOL(nosteal_pipe_buf_ops); static void wakeup_pipe_writers(struct pipe_inode_info *pipe) { - smp_mb(); - if (waitqueue_active(&pipe->wr_wait)) - wake_up_interruptible(&pipe->wr_wait); + if (wq_has_sleeper(&pipe->wr_wait)) + wake_up_interruptible_poll(&pipe->wr_wait, + EPOLLOUT | EPOLLWRNORM); kill_fasync(&pipe->fasync_writers, SIGIO, POLL_OUT); } @@ -1009,21 +1009,14 @@ ssize_t vfs_splice_read(struct file *in, loff_t *ppos, } EXPORT_SYMBOL_GPL(vfs_splice_read); -/** - * splice_direct_to_actor - splices data directly between two non-pipes - * @in: file to splice from - * @sd: actor information on where to splice to - * @actor: handles the data splicing - * - * Description: - * This is a special case helper to splice directly between two - * points, without requiring an explicit pipe. Internally an allocated - * pipe is cached in the process, and reused during the lifetime of - * that process. - * +/* + * This is a special case helper to splice directly between two + * points, without requiring an explicit pipe. Internally an allocated + * pipe is cached in the process, and reused during the lifetime of + * that process. */ -ssize_t splice_direct_to_actor(struct file *in, struct splice_desc *sd, - splice_direct_actor *actor) +static ssize_t splice_direct_to_actor(struct file *in, struct splice_desc *sd, + splice_direct_actor *actor) { struct pipe_inode_info *pipe; ssize_t ret, bytes; @@ -1147,7 +1140,42 @@ out_release: goto done; } -EXPORT_SYMBOL(splice_direct_to_actor); + +/** + * vfs_splice_to_actor - call an actor on data read from a file + * @in: file to read from + * @pos: file offset + * @count: maximum number of bytes to read + * @actor: callback to process a pipe's worth of data + * @private: private data passed to @actor + * + * Read up to @count worth of data from @in at @pos, and call @actor + * when the hidden pipe used to buffer the data is full. Ensures the + * read is allowed using rw_verify_area() and emits fsnotify access + * events. @in must be seekable (FMODE_LSEEK). + * + * Return: The number of bytes spliced, or a negative errno. + */ +ssize_t vfs_splice_to_actor(struct file *in, loff_t pos, size_t count, + splice_direct_actor *actor, void *private) +{ + struct splice_desc sd = { + .total_len = count, + .pos = pos, + .u.data = private, + }; + ssize_t ret; + + ret = rw_verify_area(READ, in, &sd.pos, sd.total_len); + if (ret < 0) + return ret; + + ret = splice_direct_to_actor(in, &sd, actor); + if (ret >= 0) + fsnotify_access(in); + return ret; +} +EXPORT_SYMBOL(vfs_splice_to_actor); static int direct_splice_actor(struct pipe_inode_info *pipe, struct splice_desc *sd) diff --git a/fs/stat.c b/fs/stat.c index c461c3054234..a9b7383d538d 100644 --- a/fs/stat.c +++ b/fs/stat.c @@ -79,7 +79,7 @@ EXPORT_SYMBOL(fill_mg_cmtime); * uid and gid filds. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -void generic_fillattr(struct mnt_idmap *idmap, u32 request_mask, +void generic_fillattr(const struct mnt_idmap *idmap, u32 request_mask, struct inode *inode, struct kstat *stat) { vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); @@ -181,7 +181,7 @@ EXPORT_SYMBOL_GPL(generic_fill_statx_atomic_writes); int vfs_getattr_nosec(const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode = d_backing_inode(path->dentry); memset(stat, 0, sizeof(*stat)); diff --git a/fs/super.c b/fs/super.c index 1d5ccf540a9b..b1d4add11b77 100644 --- a/fs/super.c +++ b/fs/super.c @@ -1374,7 +1374,17 @@ static int test_single_super(struct super_block *s, struct fs_context *fc) return 1; } -static int vfs_get_super(struct fs_context *fc, +/** + * get_tree_super - Get a superblock, optionally sharing an existing one + * @fc: The filesystem context holding the parameters + * @test: Comparison function to find a matching existing superblock, or NULL + * @fill_super: Helper to initialise a new superblock + * + * If @test is non-NULL and matches an existing superblock, that superblock is + * reused; otherwise a new anonymous superblock is created and initialised with + * @fill_super. Passing NULL for @test always creates a new superblock. + */ +int get_tree_super(struct fs_context *fc, int (*test)(struct super_block *, struct fs_context *), int (*fill_super)(struct super_block *sb, struct fs_context *fc)) @@ -1401,12 +1411,13 @@ error: deactivate_locked_super(sb); return err; } +EXPORT_SYMBOL(get_tree_super); int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)) { - return vfs_get_super(fc, NULL, fill_super); + return get_tree_super(fc, NULL, fill_super); } EXPORT_SYMBOL(get_tree_nodev); @@ -1414,7 +1425,7 @@ int get_tree_single(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)) { - return vfs_get_super(fc, test_single_super, fill_super); + return get_tree_super(fc, test_single_super, fill_super); } EXPORT_SYMBOL(get_tree_single); @@ -1424,7 +1435,7 @@ int get_tree_keyed(struct fs_context *fc, void *key) { fc->s_fs_info = key; - return vfs_get_super(fc, test_keyed_super, fill_super); + return get_tree_super(fc, test_keyed_super, fill_super); } EXPORT_SYMBOL(get_tree_keyed); diff --git a/fs/tests/.kunitconfig b/fs/tests/.kunitconfig new file mode 100644 index 000000000000..de67125a9421 --- /dev/null +++ b/fs/tests/.kunitconfig @@ -0,0 +1,2 @@ +CONFIG_KUNIT=y +CONFIG_FDTABLE_KUNIT_TEST=y diff --git a/fs/tests/fdtable_kunit.c b/fs/tests/fdtable_kunit.c new file mode 100644 index 000000000000..c5b028264557 --- /dev/null +++ b/fs/tests/fdtable_kunit.c @@ -0,0 +1,72 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <kunit/test.h> +#include <linux/fdtable.h> +#include <linux/file.h> + +static void test_alloc_fdtable(struct kunit *test) +{ + struct fdtable *fdt; + unsigned int slots = 64; + + fdt = alloc_fdtable(slots); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt); + + /* Check that max_fds is set correctly and is >= slots */ + KUNIT_EXPECT_GE(test, fdt->max_fds, slots); + + /* Check that fd is allocated */ + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt->fd); + + /* + * Check dynamic object size of fdt->fd if compiler supports + * __counted_by_ptr. + */ +#ifdef CONFIG_CC_HAS_COUNTED_BY_PTR + KUNIT_EXPECT_EQ(test, __struct_size(fdt->fd), + fdt->max_fds * sizeof(struct file *)); +#endif + + __free_fdtable(fdt); +} + +static void test_dup_fd(struct kunit *test) +{ + struct files_struct *newf; + struct fdtable *fdt; + + newf = dup_fd(&init_files, NULL); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, newf); + + fdt = rcu_dereference_raw(newf->fdt); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt); + + /* Check that max_fds is set correctly and is >= NR_OPEN_DEFAULT */ + KUNIT_EXPECT_GE(test, fdt->max_fds, NR_OPEN_DEFAULT); + + /* Check that fd is allocated */ + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt->fd); + + /* + * Check dynamic object size of fdt->fd if compiler supports + * __counted_by_ptr. + */ +#ifdef CONFIG_CC_HAS_COUNTED_BY_PTR + KUNIT_EXPECT_EQ(test, __struct_size(fdt->fd), + fdt->max_fds * sizeof(struct file *)); +#endif + + put_files_struct(newf); +} + +static struct kunit_case fdtable_test_cases[] = { + KUNIT_CASE(test_alloc_fdtable), + KUNIT_CASE(test_dup_fd), + {} +}; + +static struct kunit_suite fdtable_test_suite = { + .name = "fdtable", + .test_cases = fdtable_test_cases, +}; + +kunit_test_suite(fdtable_test_suite); diff --git a/fs/tracefs/event_inode.c b/fs/tracefs/event_inode.c index 6e3513b13cfa..6f6daac88621 100644 --- a/fs/tracefs/event_inode.c +++ b/fs/tracefs/event_inode.c @@ -182,7 +182,7 @@ static void update_attr(struct eventfs_attr *attr, struct iattr *iattr) } } -static int eventfs_set_attr(struct mnt_idmap *idmap, struct dentry *dentry, +static int eventfs_set_attr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { const struct eventfs_entry *entry; diff --git a/fs/tracefs/inode.c b/fs/tracefs/inode.c index f3d6188a3b7b..5020365ca704 100644 --- a/fs/tracefs/inode.c +++ b/fs/tracefs/inode.c @@ -94,7 +94,7 @@ static struct tracefs_dir_ops { int (*rmdir)(const char *name); } tracefs_ops __ro_after_init; -static struct dentry *tracefs_syscall_mkdir(struct mnt_idmap *idmap, +static struct dentry *tracefs_syscall_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *dentry, umode_t mode) { @@ -189,14 +189,14 @@ static void set_tracefs_inode_owner(struct inode *inode) inode->i_gid = gid; } -static int tracefs_permission(struct mnt_idmap *idmap, +static int tracefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { set_tracefs_inode_owner(inode); return generic_permission(idmap, inode, mask); } -static int tracefs_getattr(struct mnt_idmap *idmap, +static int tracefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -207,7 +207,7 @@ static int tracefs_getattr(struct mnt_idmap *idmap, return 0; } -static int tracefs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int tracefs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; diff --git a/fs/ubifs/dir.c b/fs/ubifs/dir.c index 23ec924162d6..c67954f6bee0 100644 --- a/fs/ubifs/dir.c +++ b/fs/ubifs/dir.c @@ -302,7 +302,7 @@ static int ubifs_prepare_create(struct inode *dir, struct dentry *dentry, return fscrypt_setup_filename(dir, &dentry->d_name, 0, nm); } -static int ubifs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -440,7 +440,7 @@ static void unlock_2_inodes(struct inode *inode1, struct inode *inode2) mutex_unlock(&ubifs_inode(inode1)->ui_mutex); } -static int ubifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct dentry *dentry = file->f_path.dentry; @@ -1002,7 +1002,7 @@ out_fname: return err; } -static struct dentry *ubifs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ubifs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -1077,7 +1077,7 @@ out_budg: return ERR_PTR(err); } -static int ubifs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -1170,7 +1170,7 @@ out_budg: return err; } -static int ubifs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -1642,7 +1642,7 @@ out: return err; } -static int ubifs_rename(struct mnt_idmap *idmap, +static int ubifs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1667,7 +1667,7 @@ static int ubifs_rename(struct mnt_idmap *idmap, return do_rename(old_dir, old_dentry, new_dir, new_dentry, flags); } -int ubifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ubifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { loff_t size; diff --git a/fs/ubifs/file.c b/fs/ubifs/file.c index e73c28b12f97..244b835fc82a 100644 --- a/fs/ubifs/file.c +++ b/fs/ubifs/file.c @@ -1251,7 +1251,7 @@ static int do_setattr(struct ubifs_info *c, struct inode *inode, return err; } -int ubifs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ubifs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int err; @@ -1611,7 +1611,7 @@ static const char *ubifs_get_link(struct dentry *dentry, return fscrypt_get_symlink(inode, ui->data, ui->data_len, done); } -static int ubifs_symlink_getattr(struct mnt_idmap *idmap, +static int ubifs_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/ubifs/ioctl.c b/fs/ubifs/ioctl.c index 79536b2e3d7a..5c34f895bd4e 100644 --- a/fs/ubifs/ioctl.c +++ b/fs/ubifs/ioctl.c @@ -144,7 +144,7 @@ int ubifs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ubifs_fileattr_set(struct mnt_idmap *idmap, +int ubifs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ubifs/ubifs.h b/fs/ubifs/ubifs.h index 00db0d19a85e..b85b45a0564b 100644 --- a/fs/ubifs/ubifs.h +++ b/fs/ubifs/ubifs.h @@ -2020,7 +2020,7 @@ int ubifs_calc_dark(const struct ubifs_info *c, int spc); /* file.c */ int ubifs_fsync(struct file *file, loff_t start, loff_t end, int datasync); -int ubifs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ubifs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int ubifs_update_time(struct inode *inode, enum fs_update_time type, unsigned int flags); @@ -2028,7 +2028,7 @@ int ubifs_update_time(struct inode *inode, enum fs_update_time type, /* dir.c */ struct inode *ubifs_new_inode(struct ubifs_info *c, struct inode *dir, umode_t mode, bool is_xattr); -int ubifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ubifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); int ubifs_check_dir_empty(struct inode *dir); @@ -2083,7 +2083,7 @@ void ubifs_destroy_size_tree(struct ubifs_info *c); /* ioctl.c */ int ubifs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ubifs_fileattr_set(struct mnt_idmap *idmap, +int ubifs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long ubifs_ioctl(struct file *file, unsigned int cmd, unsigned long arg); void ubifs_set_inode_flags(struct inode *inode); diff --git a/fs/ubifs/xattr.c b/fs/ubifs/xattr.c index b5a9ab9d8a10..3b0e8a270f59 100644 --- a/fs/ubifs/xattr.c +++ b/fs/ubifs/xattr.c @@ -660,7 +660,7 @@ static int xattr_get(const struct xattr_handler *handler, } static int xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/udf/file.c b/fs/udf/file.c index 57d11606a2a7..02e9314818dc 100644 --- a/fs/udf/file.c +++ b/fs/udf/file.c @@ -212,7 +212,7 @@ const struct file_operations udf_file_operations = { .setlease = generic_setlease, }; -static int udf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int udf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/udf/namei.c b/fs/udf/namei.c index b90841ac0a40..42d000fe9d6b 100644 --- a/fs/udf/namei.c +++ b/fs/udf/namei.c @@ -370,7 +370,7 @@ static int udf_add_nondir(struct dentry *dentry, struct inode *inode) return 0; } -static int udf_create(struct mnt_idmap *idmap, struct inode *dir, +static int udf_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode = udf_new_inode(dir, mode); @@ -386,7 +386,7 @@ static int udf_create(struct mnt_idmap *idmap, struct inode *dir, return udf_add_nondir(dentry, inode); } -static int udf_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int udf_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = udf_new_inode(dir, mode); @@ -403,7 +403,7 @@ static int udf_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int udf_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int udf_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -419,7 +419,7 @@ static int udf_mknod(struct mnt_idmap *idmap, struct inode *dir, return udf_add_nondir(dentry, inode); } -static struct dentry *udf_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *udf_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -567,7 +567,7 @@ out: return ret; } -static int udf_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int udf_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -762,7 +762,7 @@ static int udf_link(struct dentry *old_dentry, struct inode *dir, /* Anybody can rename anything with this: the permission checks are left to the * higher-level routines. */ -static int udf_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int udf_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/udf/symlink.c b/fs/udf/symlink.c index a05d1888a2ba..df41bf5a05b1 100644 --- a/fs/udf/symlink.c +++ b/fs/udf/symlink.c @@ -133,7 +133,7 @@ out: return err; } -static int udf_symlink_getattr(struct mnt_idmap *idmap, +static int udf_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/ufs/dir.c b/fs/ufs/dir.c index ce43cf20b07c..e96174b738b4 100644 --- a/fs/ufs/dir.c +++ b/fs/ufs/dir.c @@ -213,7 +213,7 @@ fail: static unsigned ufs_last_byte(struct inode *inode, unsigned long page_nr) { - unsigned last_byte = inode->i_size; + u64 last_byte = inode->i_size; last_byte -= page_nr << PAGE_SHIFT; if (last_byte > PAGE_SIZE) diff --git a/fs/ufs/inode.c b/fs/ufs/inode.c index 440d014cc5ed..c9ff8673fa66 100644 --- a/fs/ufs/inode.c +++ b/fs/ufs/inode.c @@ -1195,7 +1195,7 @@ out: return err; } -int ufs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ufs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ufs/namei.c b/fs/ufs/namei.c index 6703f3bcf76f..d45347a25741 100644 --- a/fs/ufs/namei.c +++ b/fs/ufs/namei.c @@ -69,7 +69,7 @@ static struct dentry *ufs_lookup(struct inode * dir, struct dentry *dentry, unsi * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ufs_create (struct mnt_idmap * idmap, +static int ufs_create (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { struct inode *inode; @@ -85,7 +85,7 @@ static int ufs_create (struct mnt_idmap * idmap, return ufs_add_nondir(dentry, inode); } -static int ufs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ufs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -105,7 +105,7 @@ static int ufs_mknod(struct mnt_idmap *idmap, struct inode *dir, return err; } -static int ufs_symlink (struct mnt_idmap * idmap, struct inode * dir, +static int ufs_symlink (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, const char * symname) { struct super_block * sb = dir->i_sb; @@ -165,7 +165,7 @@ static int ufs_link (struct dentry * old_dentry, struct inode * dir, return error; } -static struct dentry *ufs_mkdir(struct mnt_idmap * idmap, struct inode * dir, +static struct dentry *ufs_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { struct inode * inode; @@ -240,7 +240,7 @@ static int ufs_rmdir (struct inode * dir, struct dentry *dentry) return err; } -static int ufs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ufs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/ufs/ufs.h b/fs/ufs/ufs.h index 788e025056b2..541566f5b5fb 100644 --- a/fs/ufs/ufs.h +++ b/fs/ufs/ufs.h @@ -120,7 +120,7 @@ extern struct inode *ufs_iget(struct super_block *, unsigned long); extern int ufs_write_inode (struct inode *, struct writeback_control *); extern int ufs_sync_inode (struct inode *); extern void ufs_evict_inode (struct inode *); -extern int ufs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int ufs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); /* namei.c */ diff --git a/fs/vboxsf/dir.c b/fs/vboxsf/dir.c index 0b9eab157432..b9c46c769c4e 100644 --- a/fs/vboxsf/dir.c +++ b/fs/vboxsf/dir.c @@ -296,14 +296,14 @@ out: return err; } -static int vboxsf_dir_mkfile(struct mnt_idmap *idmap, +static int vboxsf_dir_mkfile(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, umode_t mode) { return vboxsf_dir_create(parent, dentry, mode, false, true, NULL); } -static struct dentry *vboxsf_dir_mkdir(struct mnt_idmap *idmap, +static struct dentry *vboxsf_dir_mkdir(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, umode_t mode) { @@ -382,7 +382,7 @@ static int vboxsf_dir_unlink(struct inode *parent, struct dentry *dentry) return 0; } -static int vboxsf_dir_rename(struct mnt_idmap *idmap, +static int vboxsf_dir_rename(const struct mnt_idmap *idmap, struct inode *old_parent, struct dentry *old_dentry, struct inode *new_parent, @@ -425,7 +425,7 @@ err_put_old_path: return err; } -static int vboxsf_dir_symlink(struct mnt_idmap *idmap, +static int vboxsf_dir_symlink(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, const char *symname) { diff --git a/fs/vboxsf/utils.c b/fs/vboxsf/utils.c index 298bfc93255c..8775fbee1ce6 100644 --- a/fs/vboxsf/utils.c +++ b/fs/vboxsf/utils.c @@ -233,7 +233,7 @@ int vboxsf_inode_revalidate(struct dentry *dentry) return 0; } -int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, +int vboxsf_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *kstat, u32 request_mask, unsigned int flags) { int err; @@ -258,7 +258,7 @@ int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int vboxsf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int vboxsf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct vboxsf_inode *sf_i = VBOXSF_I(d_inode(dentry)); diff --git a/fs/vboxsf/vfsmod.h b/fs/vboxsf/vfsmod.h index b61afd0ce842..59a4e44c4005 100644 --- a/fs/vboxsf/vfsmod.h +++ b/fs/vboxsf/vfsmod.h @@ -98,10 +98,10 @@ int vboxsf_stat(struct vboxsf_sbi *sbi, struct shfl_string *path, struct shfl_fsobjinfo *info); int vboxsf_stat_dentry(struct dentry *dentry, struct shfl_fsobjinfo *info); int vboxsf_inode_revalidate(struct dentry *dentry); -int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, +int vboxsf_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *kstat, u32 request_mask, unsigned int query_flags); -int vboxsf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int vboxsf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); struct shfl_string *vboxsf_path_from_dentry(struct vboxsf_sbi *sbi, struct dentry *dentry); diff --git a/fs/xattr.c b/fs/xattr.c index d58979115200..d9f035610f0b 100644 --- a/fs/xattr.c +++ b/fs/xattr.c @@ -100,7 +100,7 @@ xattr_resolve_name(struct inode *inode, const char **name) * * Return: On success zero is returned. On error a negative errno is returned. */ -int may_write_xattr(struct mnt_idmap *idmap, struct inode *inode) +int may_write_xattr(const struct mnt_idmap *idmap, struct inode *inode) { if (IS_IMMUTABLE(inode)) return -EPERM; @@ -123,7 +123,7 @@ static inline int xattr_permission_error(int mask) * because different namespaces have very different rules. */ static int -xattr_permission(struct mnt_idmap *idmap, struct inode *inode, +xattr_permission(const struct mnt_idmap *idmap, struct inode *inode, const char *name, int mask) { if (mask & MAY_WRITE) { @@ -204,7 +204,7 @@ xattr_supports_user_prefix(struct inode *inode) EXPORT_SYMBOL(xattr_supports_user_prefix); int -__vfs_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) { @@ -242,7 +242,7 @@ EXPORT_SYMBOL(__vfs_setxattr); * is executed. It also assumes that the caller will make the appropriate * permission checks. */ -int __vfs_setxattr_noperm(struct mnt_idmap *idmap, +int __vfs_setxattr_noperm(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -295,7 +295,7 @@ int __vfs_setxattr_noperm(struct mnt_idmap *idmap, * a delegation was broken on, NULL if none. */ int -__vfs_setxattr_locked(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_setxattr_locked(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags, struct delegated_inode *delegated_inode) { @@ -324,7 +324,7 @@ out: EXPORT_SYMBOL_GPL(__vfs_setxattr_locked); int -vfs_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { struct inode *inode = dentry->d_inode; @@ -358,7 +358,7 @@ retry_deleg: EXPORT_SYMBOL_GPL(vfs_setxattr); static ssize_t -xattr_getsecurity(struct mnt_idmap *idmap, struct inode *inode, +xattr_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void *value, size_t size) { void *buffer = NULL; @@ -395,7 +395,7 @@ out_noalloc: * Returns the result of alloc, if failed, or the getxattr operation. */ int -vfs_getxattr_alloc(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t xattr_size, gfp_t flags) { @@ -448,7 +448,7 @@ __vfs_getxattr(struct dentry *dentry, struct inode *inode, const char *name, EXPORT_SYMBOL(__vfs_getxattr); ssize_t -vfs_getxattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, void *value, size_t size) { struct inode *inode = dentry->d_inode; @@ -527,7 +527,7 @@ vfs_listxattr(struct dentry *dentry, char *list, size_t size) EXPORT_SYMBOL_GPL(vfs_listxattr); int -__vfs_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct inode *inode = d_inode(dentry); @@ -557,7 +557,7 @@ EXPORT_SYMBOL(__vfs_removexattr); * a delegation was broken on, NULL if none. */ int -__vfs_removexattr_locked(struct mnt_idmap *idmap, +__vfs_removexattr_locked(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct delegated_inode *delegated_inode) { @@ -589,7 +589,7 @@ out: EXPORT_SYMBOL_GPL(__vfs_removexattr_locked); int -vfs_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct inode *inode = dentry->d_inode; @@ -652,7 +652,7 @@ int setxattr_copy(const char __user *name, struct kernel_xattr_ctx *ctx) return error; } -static int do_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int do_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct kernel_xattr_ctx *ctx) { if (is_posix_acl_xattr(ctx->kname->name)) @@ -787,7 +787,7 @@ SYSCALL_DEFINE5(fsetxattr, int, fd, const char __user *, name, * Extended attribute GET operations */ static ssize_t -do_getxattr(struct mnt_idmap *idmap, struct dentry *d, +do_getxattr(const struct mnt_idmap *idmap, struct dentry *d, struct kernel_xattr_ctx *ctx) { ssize_t error; @@ -1029,7 +1029,7 @@ SYSCALL_DEFINE3(flistxattr, int, fd, char __user *, list, size_t, size) * Extended attribute REMOVE operations */ static long -removexattr(struct mnt_idmap *idmap, struct dentry *d, const char *name) +removexattr(const struct mnt_idmap *idmap, struct dentry *d, const char *name) { if (is_posix_acl_xattr(name)) return vfs_remove_acl(idmap, d, name); diff --git a/fs/xfs/libxfs/xfs_errortag.h b/fs/xfs/libxfs/xfs_errortag.h index f0c83f1f0b3b..d14aa289699f 100644 --- a/fs/xfs/libxfs/xfs_errortag.h +++ b/fs/xfs/libxfs/xfs_errortag.h @@ -75,7 +75,8 @@ #define XFS_ERRTAG_METAFILE_RESV_CRITICAL 45 #define XFS_ERRTAG_FORCE_ZERO_RANGE 46 #define XFS_ERRTAG_ZONE_RESET 47 -#define XFS_ERRTAG_MAX 48 +#define XFS_ERRTAG_BOUNCE_REREAD 48 +#define XFS_ERRTAG_MAX 49 /* * Random factors for above tags, 1 means always, 2 means 1/2 time, etc. @@ -137,7 +138,8 @@ XFS_ERRTAG(WRITE_DELAY_MS, write_delay_ms, 3000) \ XFS_ERRTAG(EXCHMAPS_FINISH_ONE, exchmaps_finish_one, 1) \ XFS_ERRTAG(METAFILE_RESV_CRITICAL, metafile_resv_crit, 4) \ XFS_ERRTAG(FORCE_ZERO_RANGE, force_zero_range, 4) \ -XFS_ERRTAG(ZONE_RESET, zone_reset, 1) +XFS_ERRTAG(ZONE_RESET, zone_reset, 1) \ +XFS_ERRTAG(BOUNCE_REREAD, bounce_reread, XFS_RANDOM_DEFAULT) #endif /* XFS_ERRTAG */ #endif /* __XFS_ERRORTAG_H_ */ diff --git a/fs/xfs/libxfs/xfs_inode_util.h b/fs/xfs/libxfs/xfs_inode_util.h index 060242998a23..e9eac35159c3 100644 --- a/fs/xfs/libxfs/xfs_inode_util.h +++ b/fs/xfs/libxfs/xfs_inode_util.h @@ -27,7 +27,7 @@ prid_t xfs_get_initial_prid(struct xfs_inode *dp); * idmap to NULL. To create a tree root, set pip to NULL. */ struct xfs_icreate_args { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct xfs_inode *pip; /* parent inode or null */ dev_t rdev; umode_t mode; diff --git a/fs/xfs/xfs_acl.c b/fs/xfs/xfs_acl.c index fdfca6fc75b6..20d87b52c4fc 100644 --- a/fs/xfs/xfs_acl.c +++ b/fs/xfs/xfs_acl.c @@ -243,7 +243,7 @@ xfs_acl_set_mode( } int -xfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +xfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { umode_t mode; diff --git a/fs/xfs/xfs_acl.h b/fs/xfs/xfs_acl.h index bf7f960997d3..183526bec32c 100644 --- a/fs/xfs/xfs_acl.h +++ b/fs/xfs/xfs_acl.h @@ -11,7 +11,7 @@ struct posix_acl; #ifdef CONFIG_XFS_POSIX_ACL extern struct posix_acl *xfs_get_acl(struct inode *inode, int type, bool rcu); -extern int xfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int xfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int __xfs_set_acl(struct inode *inode, struct posix_acl *acl, int type); void xfs_forget_acl(struct inode *inode, const char *name); diff --git a/fs/xfs/xfs_aops.c b/fs/xfs/xfs_aops.c index 8b6119776fb3..c30e688cfc9f 100644 --- a/fs/xfs/xfs_aops.c +++ b/fs/xfs/xfs_aops.c @@ -23,7 +23,6 @@ #include "xfs_ioend.h" #include "xfs_zone_alloc.h" #include "xfs_rtgroup.h" -#include <linux/bio-integrity.h> struct xfs_writepage_ctx { struct iomap_writepage_ctx ctx; @@ -498,8 +497,7 @@ xfs_zoned_writeback_submit( bio_endio(&ioend->io_bio); return error; } - if (wpc->iomap.flags & IOMAP_F_INTEGRITY) - fs_bio_integrity_generate(&ioend->io_bio); + xfs_zone_alloc_and_submit(ioend, &XFS_ZWPC(wpc)->open_zone); return 0; } @@ -585,11 +583,10 @@ xfs_bio_submit_read( const struct iomap_iter *iter, struct iomap_read_folio_ctx *ctx) { - struct bio *bio = ctx->read_ctx; - - /* defer read completions to the ioend workqueue */ - iomap_init_ioend(iter->inode, bio, ctx->read_ctx_file_offset, 0); - iomap_bio_submit_read_endio(iter, ctx, xfs_end_bio); + xfs_ioend_submit_read(iter->inode, ctx->read_ctx, + ctx->read_ctx_file_offset, + iomap_ioend_flags(&iter->iomap)); + ctx->read_ctx = NULL; } static const struct iomap_read_ops xfs_iomap_read_ops = { diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c index 8256c1d13ce2..6c93b4f5629c 100644 --- a/fs/xfs/xfs_buf.c +++ b/fs/xfs/xfs_buf.c @@ -5,6 +5,7 @@ */ #include "xfs_platform.h" #include <linux/backing-dev.h> +#include <linux/blk-integrity.h> #include <linux/dax.h> #include "xfs_shared.h" @@ -1694,6 +1695,7 @@ xfs_configure_buftarg( struct xfs_mount *mp = btp->bt_mount; if (btp->bt_bdev) { + struct blk_integrity *bi = bdev_get_integrity(btp->bt_bdev); int error; error = bdev_validate_blocksize(btp->bt_bdev, sectorsize); @@ -1706,6 +1708,15 @@ xfs_configure_buftarg( if (bdev_can_atomic_write(btp->bt_bdev)) xfs_configure_buftarg_atomic_writes(btp); + + if (!bi) + ; + else if (btp->bt_bdev == btp->bt_mount->m_super->s_bdev) + xfs_info(mp, "using %s integrity profile", + blk_integrity_profile_name(bi)); + else + xfs_info(mp, "using %s integrity profile for %pg", + blk_integrity_profile_name(bi), btp->bt_bdev); } btp->bt_meta_sectorsize = sectorsize; diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c index dd6d2e08faff..d164de6ff98b 100644 --- a/fs/xfs/xfs_file.c +++ b/fs/xfs/xfs_file.c @@ -37,6 +37,7 @@ #include <linux/fadvise.h> #include <linux/mount.h> #include <linux/filelock.h> +#include <linux/bio-integrity.h> static const struct vm_operations_struct xfs_file_vm_ops; @@ -222,9 +223,8 @@ xfs_dio_read_bounce_submit_io( struct bio *bio, loff_t file_offset) { - iomap_init_ioend(iter->inode, bio, file_offset, IOMAP_IOEND_DIRECT); - bio->bi_end_io = xfs_end_bio; - submit_bio(bio); + xfs_ioend_submit_read(iter->inode, bio, file_offset, + iomap_ioend_flags(&iter->iomap) | IOMAP_IOEND_DIRECT); } static const struct iomap_dio_ops xfs_dio_read_bounce_ops = { @@ -252,8 +252,7 @@ xfs_file_dio_read( return ret; if (mapping_stable_writes(iocb->ki_filp->f_mapping)) { ret = iomap_dio_rw(iocb, to, &xfs_read_iomap_ops, - &xfs_dio_read_bounce_ops, IOMAP_DIO_BOUNCE, - NULL, 0); + &xfs_dio_read_bounce_ops, 0, NULL, 0); } else { ret = iomap_dio_read_simple(iocb, to, xfs_read_iomap_begin); if (ret == -ENOTBLK) @@ -713,7 +712,7 @@ xfs_dio_zoned_submit_io( bio->bi_end_io = xfs_end_bio; ioend = iomap_init_ioend(iter->inode, bio, file_offset, - IOMAP_IOEND_DIRECT); + iomap_ioend_flags(&iter->iomap) | IOMAP_IOEND_DIRECT); xfs_zone_alloc_and_submit(ioend, &ac->open_zone); } diff --git a/fs/xfs/xfs_handle.c b/fs/xfs/xfs_handle.c index fd9d4d8258ff..4924e676ae91 100644 --- a/fs/xfs/xfs_handle.c +++ b/fs/xfs/xfs_handle.c @@ -272,11 +272,11 @@ xfs_open_by_handle( path.mnt = mntget(parfilp->f_path.mnt); FD_PREPARE(fdf, 0, dentry_open(&path, hreq->oflags, cred)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; if (S_ISREG(inode->i_mode)) { - struct file *filp = fd_prepare_file(fdf); + struct file *filp = fdf->file; filp->f_flags |= O_NOATIME; filp->f_mode |= FMODE_NOCMTIME; diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c index 15b62574b8d4..05a14da28031 100644 --- a/fs/xfs/xfs_inode.c +++ b/fs/xfs/xfs_inode.c @@ -2084,7 +2084,7 @@ xfs_sort_inodes( */ static int xfs_rename_alloc_whiteout( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_name *src_name, struct xfs_inode *dp, struct xfs_inode **wip) @@ -2130,7 +2130,7 @@ xfs_rename_alloc_whiteout( */ int xfs_rename( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_inode *src_dp, struct xfs_name *src_name, struct xfs_inode *src_ip, diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h index 1602027cd0aa..ca96ba096359 100644 --- a/fs/xfs/xfs_inode.h +++ b/fs/xfs/xfs_inode.h @@ -568,7 +568,7 @@ int xfs_remove(struct xfs_inode *dp, struct xfs_name *name, struct xfs_inode *ip); int xfs_link(struct xfs_inode *tdp, struct xfs_inode *sip, struct xfs_name *target_name); -int xfs_rename(struct mnt_idmap *idmap, +int xfs_rename(const struct mnt_idmap *idmap, struct xfs_inode *src_dp, struct xfs_name *src_name, struct xfs_inode *src_ip, struct xfs_inode *target_dp, struct xfs_name *target_name, diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index c0fc9b34f393..f81b6e52ac40 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -748,7 +748,7 @@ xfs_ioctl_setattr_check_projid( int xfs_fileattr_set( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { diff --git a/fs/xfs/xfs_ioctl.h b/fs/xfs/xfs_ioctl.h index e57d8f5148bf..6e55cc847654 100644 --- a/fs/xfs/xfs_ioctl.h +++ b/fs/xfs/xfs_ioctl.h @@ -19,7 +19,7 @@ xfs_fileattr_get( extern int xfs_fileattr_set( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c index 40695d18dac0..e70be5b86f0b 100644 --- a/fs/xfs/xfs_ioend.c +++ b/fs/xfs/xfs_ioend.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * Copyright (c) 2016-2025 Christoph Hellwig. + * Copyright (c) 2016-2026 Christoph Hellwig. * All Rights Reserved. */ #include "xfs_platform.h" @@ -16,6 +16,135 @@ #include "xfs_reflink.h" #include "xfs_zone_alloc.h" #include "xfs_ioend.h" +#include "xfs_error.h" +#include "xfs_errortag.h" +#include <linux/bio-integrity.h> + +static void +xfs_dio_bounce_end_io( + struct bio *bio) +{ + struct iomap_ioend *ioend = iomap_ioend_from_bio(bio); + int error = blk_status_to_errno(bio->bi_status); + struct bio *orig_bio = bio->bi_private; + + if ((ioend->io_flags & IOMAP_IOEND_INTEGRITY) && !bio->bi_status) + error = iomap_ioend_integrity_verify(ioend); + iomap_bounce_read_end_io(ioend, orig_bio, error); +} + +static void +xfs_bounce_submit_ioend( + struct iomap_ioend *ioend) +{ + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_alloc(&ioend->io_bio); + ioend->io_bio.bi_end_io = xfs_dio_bounce_end_io; + bio_set_flag(&ioend->io_bio, BIO_COMPLETE_IN_TASK); + submit_bio(&ioend->io_bio); +} + +static void +xfs_end_bio_bounced( + struct bio *bio) +{ + /* + * Just complete the original ioends as all verification is done by the + * end_io handlers for the clone bio(s). + */ + iomap_finish_ioends(iomap_ioend_from_bio(bio), + blk_status_to_errno(bio->bi_status)); +} + +static void +xfs_read_bounce_and_resubmit( + struct iomap_ioend *ioend) +{ + struct bio *bio = &ioend->io_bio; + struct xfs_inode *ip = XFS_I(ioend->io_inode); + unsigned int nofs_flag = memalloc_nofs_save(); + + trace_xfs_bounce_reread(ip, ioend->io_offset, ioend->io_size); + + /* + * Free the bio integrity data for the original bio, as we'll allocate + * a new one for each sub-I/O, which could deadlock if we keep the + * integrity data for the original bio around. + */ + if (bio_integrity(bio)) + fs_bio_integrity_free(bio); + + /* + * Resubmit the bio through the iomap bounce machinery. The original + * bio itself is not resubmitted to the block layer, but just used to + * track I/O completion of the cloned bios. + */ + bio_prepare_reissue(bio, xfs_inode_buftarg(ip)->bt_bdev); + bio->bi_iter = (struct bvec_iter) { + .bi_sector = ioend->io_sector, + .bi_size = ioend->io_size, + .bi_offset = ioend->io_bvec_offset, + }; + bio->bi_end_io = xfs_end_bio_bounced; + iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev), + xfs_bounce_submit_ioend); + memalloc_nofs_restore(nofs_flag); +} + +static void +xfs_end_io_read( + struct bio *bio) +{ + struct iomap_ioend *ioend = iomap_ioend_from_bio(bio); + struct xfs_inode *ip = XFS_I(ioend->io_inode); + struct xfs_mount *mp = ip->i_mount; + int error = blk_status_to_errno(bio->bi_status); + + if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY)) { + error = iomap_ioend_integrity_verify(ioend); + if ((ioend->io_flags & IOMAP_IOEND_DIRECT) && + READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_LAZY) { + /* + * We only really need to retry for guard tag errors, + * but right now we can't distinguish them from other + * (i.e, reftag) errors. + */ + if (error || + XFS_TEST_ERROR(mp, XFS_ERRTAG_BOUNCE_REREAD)) { + xfs_read_bounce_and_resubmit(ioend); + return; + } + } + } + + iomap_finish_ioends(ioend, error); +} + +void +xfs_ioend_submit_read( + struct inode *inode, + struct bio *bio, + loff_t file_offset, + u16 ioend_flags) +{ + struct xfs_inode *ip = XFS_I(inode); + struct xfs_mount *mp = ip->i_mount; + struct iomap_ioend *ioend; + + ioend = iomap_init_ioend(inode, bio, file_offset, ioend_flags); + if ((ioend_flags & IOMAP_IOEND_DIRECT) && + READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_ALWAYS) { + iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev), + xfs_bounce_submit_ioend); + return; + } + + if (ioend_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_alloc(bio); + bio->bi_end_io = xfs_end_io_read; + bio_set_flag(bio, BIO_COMPLETE_IN_TASK); + submit_bio(bio); +} static void xfs_ioend_put_open_zones( @@ -148,11 +277,7 @@ xfs_end_io( io_list))) { list_del_init(&ioend->io_list); iomap_ioend_try_merge(ioend, &tmp); - if (bio_op(&ioend->io_bio) == REQ_OP_READ) - iomap_finish_ioends(ioend, - blk_status_to_errno(ioend->io_bio.bi_status)); - else - xfs_end_ioend_write(ioend); + xfs_end_ioend_write(ioend); cond_resched(); } } diff --git a/fs/xfs/xfs_ioend.h b/fs/xfs/xfs_ioend.h index 525865767fca..7c2a1ea3e6ed 100644 --- a/fs/xfs/xfs_ioend.h +++ b/fs/xfs/xfs_ioend.h @@ -12,5 +12,7 @@ static inline bool xfs_ioend_is_append(struct iomap_ioend *ioend) } void xfs_end_bio(struct bio *bio); +void xfs_ioend_submit_read(struct inode *inode, struct bio *bio, + loff_t file_offset, u16 ioend_flags); #endif /* __XFS_IOEND_H */ diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c index d1306e723899..f67a541ef35a 100644 --- a/fs/xfs/xfs_iops.c +++ b/fs/xfs/xfs_iops.c @@ -169,7 +169,7 @@ xfs_create_need_xattr( STATIC int xfs_generic_create( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -279,7 +279,7 @@ xfs_generic_create( STATIC int xfs_vn_mknod( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -290,7 +290,7 @@ xfs_vn_mknod( STATIC int xfs_vn_create( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -300,7 +300,7 @@ xfs_vn_create( STATIC struct dentry * xfs_vn_mkdir( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -425,7 +425,7 @@ xfs_vn_unlink( STATIC int xfs_vn_symlink( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) @@ -466,7 +466,7 @@ xfs_vn_symlink( STATIC int xfs_vn_rename( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *odir, struct dentry *odentry, struct inode *ndir, @@ -679,7 +679,7 @@ xfs_report_atomic_write( STATIC int xfs_vn_getattr( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, @@ -754,7 +754,7 @@ xfs_vn_getattr( static int xfs_vn_change_ok( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -777,7 +777,7 @@ xfs_vn_change_ok( */ static int xfs_setattr_nonsize( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct xfs_inode *ip, struct iattr *iattr) @@ -903,7 +903,7 @@ out_dqrele: */ int xfs_vn_setattr_size( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -1130,7 +1130,7 @@ out_trans_cancel: STATIC int xfs_vn_setattr( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -1250,7 +1250,7 @@ xfs_vn_fiemap( STATIC int xfs_vn_tmpfile( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) diff --git a/fs/xfs/xfs_iops.h b/fs/xfs/xfs_iops.h index 0896f6b8b3b8..328305bba19d 100644 --- a/fs/xfs/xfs_iops.h +++ b/fs/xfs/xfs_iops.h @@ -10,7 +10,7 @@ struct xfs_inode; extern ssize_t xfs_vn_listxattr(struct dentry *, char *data, size_t size); -int xfs_vn_setattr_size(struct mnt_idmap *idmap, +int xfs_vn_setattr_size(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *vap); int xfs_inode_init_security(struct inode *inode, struct inode *dir, diff --git a/fs/xfs/xfs_itable.c b/fs/xfs/xfs_itable.c index 159295c63e8f..a4cf1effa5e6 100644 --- a/fs/xfs/xfs_itable.c +++ b/fs/xfs/xfs_itable.c @@ -63,7 +63,7 @@ want_metadir_file( STATIC int xfs_bulkstat_one_int( struct xfs_mount *mp, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_trans *tp, xfs_ino_t ino, struct xfs_bstat_chunk *bc) diff --git a/fs/xfs/xfs_itable.h b/fs/xfs/xfs_itable.h index 2d0612f14d6e..c0567bfc30fb 100644 --- a/fs/xfs/xfs_itable.h +++ b/fs/xfs/xfs_itable.h @@ -8,7 +8,7 @@ /* In-memory representation of a userspace request for batch inode data. */ struct xfs_ibulk { struct xfs_mount *mp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; void __user *ubuffer; /* user output buffer */ xfs_ino_t startino; /* start with this inode */ unsigned int icount; /* number of elements in ubuffer */ diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h index 216a38a354e7..894ff2f4ecbd 100644 --- a/fs/xfs/xfs_mount.h +++ b/fs/xfs/xfs_mount.h @@ -142,6 +142,12 @@ struct xfs_freecounter { uint64_t res_saved; }; +enum xfs_read_bounce { + XFS_READ_BOUNCE_NEVER, + XFS_READ_BOUNCE_ALWAYS, + XFS_READ_BOUNCE_LAZY, +}; + /* * The struct xfsmount layout is optimised to separate read-mostly variables * from variables that are frequently modified. We put the read-mostly variables @@ -177,6 +183,7 @@ typedef struct xfs_mount { struct workqueue_struct *m_sync_workqueue; struct workqueue_struct *m_blockgc_wq; struct workqueue_struct *m_inodegc_wq; + enum xfs_read_bounce m_read_bounce; int m_bsize; /* fs logical block size */ uint8_t m_blkbit_log; /* blocklog + NBBY */ @@ -291,6 +298,7 @@ typedef struct xfs_mount { struct xfs_zone_info *m_zone_info; /* zone allocator information */ struct dentry *m_debugfs; /* debugfs parent */ struct xfs_kobj m_kobj; + struct xfs_kobj m_csum_kobj; struct xfs_kobj m_error_kobj; struct xfs_kobj m_error_meta_kobj; struct xfs_error_cfg m_error_cfg[XFS_ERR_CLASS_MAX][XFS_ERR_ERRNO_MAX]; diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c index 2edc2a497883..5a06132aa384 100644 --- a/fs/xfs/xfs_super.c +++ b/fs/xfs/xfs_super.c @@ -2317,6 +2317,7 @@ xfs_init_fs_context( mp->m_logbufs = -1; mp->m_logbsize = -1; mp->m_allocsize_log = 16; /* 64k */ + mp->m_read_bounce = XFS_READ_BOUNCE_LAZY; xfs_hooks_init(&mp->m_dir_update_hooks); diff --git a/fs/xfs/xfs_symlink.c b/fs/xfs/xfs_symlink.c index cc13819df6f2..709cd22248f8 100644 --- a/fs/xfs/xfs_symlink.c +++ b/fs/xfs/xfs_symlink.c @@ -82,7 +82,7 @@ xfs_readlink( int xfs_symlink( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_inode *dp, struct xfs_name *link_name, const char *target_path, diff --git a/fs/xfs/xfs_symlink.h b/fs/xfs/xfs_symlink.h index 0d29a50e66fd..3c5a969f9fc5 100644 --- a/fs/xfs/xfs_symlink.h +++ b/fs/xfs/xfs_symlink.h @@ -7,7 +7,7 @@ /* Kernel only symlink definitions */ -int xfs_symlink(struct mnt_idmap *idmap, struct xfs_inode *dp, +int xfs_symlink(const struct mnt_idmap *idmap, struct xfs_inode *dp, struct xfs_name *link_name, const char *target_path, umode_t mode, struct xfs_inode **ipp); int xfs_readlink(struct xfs_inode *ip, char *link); diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c index b62712187324..e77917ac179d 100644 --- a/fs/xfs/xfs_sysfs.c +++ b/fs/xfs/xfs_sysfs.c @@ -392,6 +392,71 @@ const struct kobj_type xfs_stats_ktype = { .default_groups = xfs_stats_groups, }; +static inline struct xfs_mount *csum_to_mp(struct kobject *kobj) +{ + return container_of(to_kobj(kobj), struct xfs_mount, m_csum_kobj); +} + +static bool +xfs_has_read_bounce( + struct xfs_mount *mp) +{ + if (bdev_has_integrity_csum(mp->m_ddev_targp->bt_bdev)) + return true; + if (mp->m_rtdev_targp && + bdev_has_integrity_csum(mp->m_rtdev_targp->bt_bdev)) + return true; + return false; +} + +static const char * const bounce_modes[] = { + [XFS_READ_BOUNCE_NEVER] = "never", + [XFS_READ_BOUNCE_ALWAYS] = "always", + [XFS_READ_BOUNCE_LAZY] = "lazy", +}; + +static ssize_t +read_bounce_show( + struct kobject *kobj, + char *buf) +{ + struct xfs_mount *mp = csum_to_mp(kobj); + + return sysfs_emit(buf, "%s\n", + bounce_modes[READ_ONCE(mp->m_read_bounce)]); +} + +static ssize_t +read_bounce_store( + struct kobject *kobj, + const char *buf, + size_t count) +{ + struct xfs_mount *mp = csum_to_mp(kobj); + int ret; + + if (!xfs_has_read_bounce(mp)) + return -EINVAL; + ret = sysfs_match_string(bounce_modes, buf); + if (ret < 0) + return ret; + WRITE_ONCE(mp->m_read_bounce, ret); + return count; +} +XFS_SYSFS_ATTR_RW(read_bounce); + +static struct attribute *xfs_csum_attrs[] = { + ATTR_LIST(read_bounce), + NULL, +}; +ATTRIBUTE_GROUPS(xfs_csum); + +static const struct kobj_type xfs_csum_ktype = { + .release = xfs_sysfs_release, + .sysfs_ops = &xfs_sysfs_ops, + .default_groups = xfs_csum_groups, +}; + /* xlog */ static inline struct xlog * @@ -817,11 +882,17 @@ xfs_mount_sysfs_init( if (error) goto out_remove_fsdir; + /* .../xfs/<dev>/csum/ */ + error = xfs_sysfs_init(&mp->m_csum_kobj, &xfs_csum_ktype, &mp->m_kobj, + "csum"); + if (error) + goto out_remove_stats_dir; + /* .../xfs/<dev>/error/ */ error = xfs_sysfs_init(&mp->m_error_kobj, &xfs_error_ktype, &mp->m_kobj, "error"); if (error) - goto out_remove_stats_dir; + goto out_remove_csum_dir; /* .../xfs/<dev>/error/fail_at_unmount */ error = sysfs_create_file(&mp->m_error_kobj.kobject, @@ -835,12 +906,14 @@ xfs_mount_sysfs_init( "metadata", &mp->m_error_meta_kobj, xfs_error_meta_init); if (error) - goto out_remove_error_dir; + goto out_remove_csum_dir; return 0; out_remove_error_dir: xfs_sysfs_del(&mp->m_error_kobj); +out_remove_csum_dir: + xfs_sysfs_del(&mp->m_csum_kobj); out_remove_stats_dir: xfs_sysfs_del(&mp->m_stats.xs_kobj); out_remove_fsdir: @@ -864,6 +937,7 @@ xfs_mount_sysfs_del( } xfs_sysfs_del(&mp->m_error_meta_kobj); xfs_sysfs_del(&mp->m_error_kobj); + xfs_sysfs_del(&mp->m_csum_kobj); xfs_sysfs_del(&mp->m_stats.xs_kobj); xfs_sysfs_del(&mp->m_kobj); } diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h index 0fc8927339b5..28d49f158b22 100644 --- a/fs/xfs/xfs_trace.h +++ b/fs/xfs/xfs_trace.h @@ -1896,6 +1896,7 @@ DEFINE_SIMPLE_IO_EVENT(xfs_zero_eof); DEFINE_SIMPLE_IO_EVENT(xfs_end_io_direct_write); DEFINE_SIMPLE_IO_EVENT(xfs_file_splice_read); DEFINE_SIMPLE_IO_EVENT(xfs_zoned_map_blocks); +DEFINE_SIMPLE_IO_EVENT(xfs_bounce_reread); DECLARE_EVENT_CLASS(xfs_itrunc_class, TP_PROTO(struct xfs_inode *ip, xfs_fsize_t new_size), diff --git a/fs/xfs/xfs_xattr.c b/fs/xfs/xfs_xattr.c index 1efe6c8139b2..b059a9714d11 100644 --- a/fs/xfs/xfs_xattr.c +++ b/fs/xfs/xfs_xattr.c @@ -169,7 +169,7 @@ xfs_xattr_flags_to_op( static int xfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) { diff --git a/fs/xfs/xfs_zone_alloc.c b/fs/xfs/xfs_zone_alloc.c index b75cf3bfe33c..5f0af0c2c5e5 100644 --- a/fs/xfs/xfs_zone_alloc.c +++ b/fs/xfs/xfs_zone_alloc.c @@ -26,6 +26,7 @@ #include "xfs_zones.h" #include "xfs_trace.h" #include "xfs_mru_cache.h" +#include <linux/bio-integrity.h> static void xfs_open_zone_free_rcu( @@ -911,6 +912,9 @@ xfs_zone_alloc_and_submit( if (xfs_is_shutdown(mp)) goto out_error; + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_generate(&ioend->io_bio); + /* * If we don't have a locally cached zone in this write context, see if * the inode is still associated with a zone and use that if so. diff --git a/fs/zonefs/super.c b/fs/zonefs/super.c index ff43d6d1ea30..b97f1f2b8dda 100644 --- a/fs/zonefs/super.c +++ b/fs/zonefs/super.c @@ -533,7 +533,7 @@ static int zonefs_show_options(struct seq_file *seq, struct dentry *root) return 0; } -static int zonefs_inode_setattr(struct mnt_idmap *idmap, +static int zonefs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h index f686a37f7a0a..2e87faf9a8c2 100644 --- a/include/linux/binfmts.h +++ b/include/linux/binfmts.h @@ -128,7 +128,8 @@ struct linux_binfmt { struct module *module; int (*load_binary)(struct linux_binprm *); #ifdef CONFIG_COREDUMP - int (*core_dump)(struct coredump_params *cprm); + /* Returns true if the whole coredump was written. */ + bool (*core_dump)(struct coredump_params *cprm); unsigned long min_coredump; /* minimal dump size */ #endif } __randomize_layout; diff --git a/include/linux/bio-integrity.h b/include/linux/bio-integrity.h index 0ea2a8bf7efb..a954c97be0b3 100644 --- a/include/linux/bio-integrity.h +++ b/include/linux/bio-integrity.h @@ -151,7 +151,6 @@ void bio_integrity_setup_default(struct bio *bio); unsigned int fs_bio_integrity_alloc(struct bio *bio); void fs_bio_integrity_free(struct bio *bio); void fs_bio_integrity_generate(struct bio *bio); -int fs_bio_integrity_verify(struct bio *bio, sector_t sector, - unsigned int size); +int fs_bio_integrity_verify(struct bio *bio, struct bvec_iter *data_iter); #endif /* _LINUX_BIO_INTEGRITY_H */ diff --git a/include/linux/bio.h b/include/linux/bio.h index bb3235497e67..17944e44b584 100644 --- a/include/linux/bio.h +++ b/include/linux/bio.h @@ -479,6 +479,7 @@ static inline void bio_init_inline(struct bio *bio, struct block_device *bdev, extern void bio_uninit(struct bio *); void bio_reset(struct bio *bio, struct block_device *bdev, blk_opf_t opf); void bio_reuse(struct bio *bio, blk_opf_t opf); +void bio_prepare_reissue(struct bio *bio, struct block_device *bdev); void bio_chain(struct bio *, struct bio *); void bio_await(struct bio *bio, void *priv, void (*submit)(struct bio *bio, void *priv)); @@ -516,16 +517,18 @@ int bdev_rw_virt(struct block_device *bdev, sector_t sector, void *data, size_t len, enum req_op op); int bio_iov_iter_get_pages(struct bio *bio, struct iov_iter *iter, - unsigned mem_align_mask, unsigned len_align_mask); + unsigned maxlen, unsigned mem_align_mask, + unsigned len_align_mask); bool bio_iov_iter_set(struct bio *bio, const struct iov_iter *iter); void __bio_release_pages(struct bio *bio, bool mark_dirty); extern void bio_set_pages_dirty(struct bio *bio); extern void bio_check_pages_dirty(struct bio *bio); -int bio_iov_iter_bounce(struct bio *bio, struct iov_iter *iter, size_t maxlen, - size_t minsize); -void bio_iov_iter_unbounce(struct bio *bio, bool is_error, bool mark_dirty); +int bio_alloc_bounce_folios(struct bio *bio, size_t total_len, size_t minsize); +void bio_free_folios(struct bio *bio); +int bio_iov_iter_bounce_write(struct bio *bio, struct iov_iter *iter, + size_t maxlen, size_t minsize); extern void bio_copy_data(struct bio *dst, struct bio *src); extern void bio_free_pages(struct bio *bio); diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h index 4f7905c3412b..098a65f3e48b 100644 --- a/include/linux/blkdev.h +++ b/include/linux/blkdev.h @@ -1816,9 +1816,11 @@ static inline int bio_split_rw_at(struct bio *bio, */ static inline unsigned int max_integrity_io_size(struct queue_limits *lim) { - return min_t(unsigned int, lim->max_segment_size, - (BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size) << - lim->integrity.interval_exp); + u64 max_intervals; + + max_intervals = BLK_INTEGRITY_MAX_SIZE / lim->integrity.metadata_size; + return min_t(u64, lim->max_segment_size, + max_intervals << lim->integrity.interval_exp); } #define DEFINE_IO_COMP_BATCH(name) struct io_comp_batch name = { } diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c054..e7a701ce029d 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -59,10 +59,7 @@ struct address_space; struct buffer_head { unsigned long b_state; /* buffer state bitmap (see above) */ struct buffer_head *b_this_page;/* circular list of page's buffers */ - union { - struct page *b_page; /* the page this bh is mapped to */ - struct folio *b_folio; /* the folio this bh is mapped to */ - }; + struct folio *b_folio; /* the folio this bh is mapped to */ sector_t b_blocknr; /* start block number */ size_t b_size; /* size of mapping */ @@ -172,7 +169,36 @@ static __always_inline int buffer_uptodate(const struct buffer_head *bh) static inline unsigned long bh_offset(const struct buffer_head *bh) { - return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); + return (unsigned long)(bh)->b_data & (folio_size(bh->b_folio) - 1); +} + +/** + * kmap_local_bh - Map the data of a buffer. + * @bh: The buffer. + * + * Buffers usually live in the page cache, but a few are built over memory + * which is not. Those carry no folio and b_data is already a kernel address + * which is always mapped, so there is nothing to do for them. Pair with + * kunmap_local_bh(). + * + * Return: A pointer to the buffer's data. + */ +static inline void *kmap_local_bh(const struct buffer_head *bh) +{ + if (!bh->b_folio) + return bh->b_data; + return kmap_local_folio(bh->b_folio, bh_offset(bh)); +} + +/** + * kunmap_local_bh - Unmap the data of a buffer. + * @bh: The buffer. + * @addr: The address returned by kmap_local_bh(). + */ +static inline void kunmap_local_bh(const struct buffer_head *bh, void *addr) +{ + if (bh->b_folio) + kunmap_local(addr); } /* If we *know* page->private refers to buffer_heads */ @@ -338,20 +364,58 @@ static inline void bforget(struct buffer_head *bh) __bforget(bh); } -static inline struct buffer_head * -sb_bread(struct super_block *sb, sector_t block) +/** + * sb_bread - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers + * to it. The memory is allocated from the movable area so that it can + * be migrated. The returned buffer head has its refcount increased. + * The caller should call brelse() when it has finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, __GFP_MOVABLE); } -static inline struct buffer_head * -sb_bread_unmovable(struct super_block *sb, sector_t block) +/** + * sb_bread_unmovable - Read a block. + * @sb: The superblock to read from. + * @block: Block number in units of block size. + * + * Read a specified block, and return the buffer head that refers to it. + * The memory is allocated from the unmovable area so that pointers into + * it remain valid after compaction runs. The returned buffer head has + * its refcount increased. The caller should call brelse() when it has + * finished with the buffer. + * + * Context: May sleep waiting for I/O. + * Return: NULL if the block was unreadable. + */ +static inline +struct buffer_head *sb_bread_unmovable(struct super_block *sb, sector_t block) { return __bread_gfp(sb->s_bdev, block, sb->s_blocksize, 0); } -static inline void -sb_breadahead(struct super_block *sb, sector_t block) +/** + * sb_breadahead - Start readahead. + * @sb: Superblock identifying the block device. + * @block: The block to read. + * + * Read this block. The I/O will be flagged as being readahead rather + * than immediate read, but (unlike the page cache), surrounding blocks + * will not be read. + * + * Context: May sleep in order to allocate memory. + */ +static inline +void sb_breadahead(struct super_block *sb, sector_t block) { __breadahead(sb->s_bdev, block, sb->s_blocksize); } diff --git a/include/linux/capability.h b/include/linux/capability.h index f8532d92fcad..622137f66f09 100644 --- a/include/linux/capability.h +++ b/include/linux/capability.h @@ -186,9 +186,9 @@ static inline bool ns_capable_setid(struct user_namespace *ns, int cap) } #endif /* CONFIG_MULTIUSER */ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct inode *inode); -bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap, +bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap, const struct inode *inode, int cap); extern bool file_ns_capable(const struct file *file, struct user_namespace *ns, int cap); extern bool ptracer_capable(struct task_struct *tsk, struct user_namespace *ns); @@ -215,11 +215,11 @@ static inline bool checkpoint_restore_ns_capable_noaudit(struct user_namespace * } /* audit system wants to get cap info from files as well */ -int get_vfs_caps_from_disk(struct mnt_idmap *idmap, +int get_vfs_caps_from_disk(const struct mnt_idmap *idmap, const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps); -int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry, +int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry, const void **ivalue, size_t size); #endif /* !_LINUX_CAPABILITY_H */ diff --git a/include/linux/cleanup.h b/include/linux/cleanup.h index b1b5698cbf1b..1fb8058b897d 100644 --- a/include/linux/cleanup.h +++ b/include/linux/cleanup.h @@ -261,10 +261,6 @@ const volatile void * __must_check_fn(const volatile void *val) * CLASS(name, var)(args...): * declare the variable @var as an instance of the named class * - * CLASS_INIT(name, var, init_expr): - * declare the variable @var as an instance of the named class with - * custom initialization expression. - * * Ex. * * DEFINE_CLASS(fdget, struct fd, fdput(_T), fdget(fd), int fd) @@ -302,9 +298,6 @@ static __always_inline class_##_name##_t class_##_name##ext##_constructor(_init_ class_##_name##_t var __cleanup(class_##_name##_destructor) = \ class_##_name##_constructor -#define CLASS_INIT(_name, _var, _init_expr) \ - class_##_name##_t _var __cleanup(class_##_name##_destructor) = (_init_expr) - #define __scoped_class(_name, var, _label, args...) \ for (CLASS(_name, var)(args); ; ({ goto _label; })) \ if (0) { \ diff --git a/include/linux/coredump.h b/include/linux/coredump.h index 7b38ee2e7913..74af57b9406b 100644 --- a/include/linux/coredump.h +++ b/include/linux/coredump.h @@ -6,9 +6,20 @@ #include <linux/mm.h> #include <linux/fs.h> #include <linux/sched/coredump.h> +#include <uapi/linux/coredump.h> #include <asm/siginfo.h> #ifdef CONFIG_COREDUMP +/** + * enum coredump_state - what happened while the coredump was written + * @COREDUMP_STATE_STARTED: the dumper committed to writing a coredump + * @COREDUMP_STATE_TRUNCATED: the dumper stopped before it had written all of it + */ +enum coredump_state { + COREDUMP_STATE_STARTED = (1U << 0), + COREDUMP_STATE_TRUNCATED = (1U << 1), +}; + struct core_vma_metadata { unsigned long start, end; vm_flags_t flags; @@ -21,12 +32,20 @@ struct coredump_params { const kernel_siginfo_t *siginfo; struct file *file; unsigned long limit; - /* MMF_DUMP_FILTER_* bits, snapshot of mm->flags at dump start. */ - unsigned long mm_flags; + /* COREDUMP_MEMORY_* types to dump, the task's or the server's. */ + u64 memory_types; /* Snapshot of dumpable at dump start. */ enum task_dumpable dumpable; int cpu; + /* COREDUMP_* options negotiated with the coredump server. */ + u64 mask; + /* COREDUMP_STATE_* raised while the coredump is written. */ + enum coredump_state state; + /* Record header scratch, NULL unless the coredump is a record stream. */ + struct coredump_record_header *record_hdr; + /* Bytes handed to the file, record headers included. */ loff_t written; + /* Offset in the coredump, record headers excluded. */ loff_t pos; loff_t to_skip; int vma_count; @@ -41,13 +60,13 @@ extern unsigned int core_file_note_size_limit; * These are the only things you should do on a core-file: use only these * functions to write out all the necessary info. */ -extern void dump_skip_to(struct coredump_params *cprm, unsigned long to); -extern void dump_skip(struct coredump_params *cprm, size_t nr); -extern int dump_emit(struct coredump_params *cprm, const void *addr, int nr); -extern int dump_align(struct coredump_params *cprm, int align); -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len); -extern void vfs_coredump(const kernel_siginfo_t *siginfo); +void dump_skip_to(struct coredump_params *cprm, unsigned long to); +void dump_skip(struct coredump_params *cprm, size_t nr); +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr); +bool dump_align(struct coredump_params *cprm, int align); +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len); +void vfs_coredump(const kernel_siginfo_t *siginfo); /* * Logging for the coredump code, ratelimited. diff --git a/include/linux/dax.h b/include/linux/dax.h index fe6c3ded1b50..f2d47975d905 100644 --- a/include/linux/dax.h +++ b/include/linux/dax.h @@ -155,8 +155,6 @@ int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc); int dax_folio_reset_order(struct folio *folio); -struct page *dax_layout_busy_page(struct address_space *mapping); -struct page *dax_layout_busy_page_range(struct address_space *mapping, loff_t start, loff_t end); dax_entry_t dax_lock_folio(struct folio *folio); void dax_unlock_folio(struct folio *folio, dax_entry_t cookie); dax_entry_t dax_lock_mapping_entry(struct address_space *mapping, @@ -173,16 +171,6 @@ static inline int fs_dax_get(struct dax_device *dax_dev, void *holder, { return -EOPNOTSUPP; } -static inline struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return NULL; -} - -static inline struct page *dax_layout_busy_page_range(struct address_space *mapping, pgoff_t start, pgoff_t nr_pages) -{ - return NULL; -} - static inline int dax_writeback_mapping_range(struct address_space *mapping, struct dax_device *dax_dev, struct writeback_control *wbc) { diff --git a/include/linux/dcache.h b/include/linux/dcache.h index 4b1ff99608e0..adf239f8205f 100644 --- a/include/linux/dcache.h +++ b/include/linux/dcache.h @@ -116,6 +116,8 @@ struct dentry { * possible! */ + /* lockdep tracking of DCACHE_PAR_LOOKUP locks */ + struct lockdep_map lookup_map; struct list_head d_lru; /* LRU list */ struct hlist_node d_sib; /* child of parent list */ struct hlist_head d_children; /* our children */ @@ -236,7 +238,9 @@ enum dentry_flags { DCACHE_PAR_LOOKUP = BIT(24), /* being looked up (with parent locked shared) */ DCACHE_DENTRY_CURSOR = BIT(25), DCACHE_NORCU = BIT(26), /* No RCU delay for freeing */ - DCACHE_PERSISTENT = BIT(27) + DCACHE_PERSISTENT = BIT(27), +/* 28, 29, 30 free */ + DCACHE_PRIVATE = BIT(31) /* fs-specific flag */ }; #define DCACHE_MANAGED_DENTRY \ @@ -257,7 +261,9 @@ extern void d_delete(struct dentry *); extern struct dentry * d_alloc(struct dentry *, const struct qstr *); extern struct dentry * d_alloc_anon(struct super_block *); extern struct dentry * d_alloc_parallel(struct dentry *, const struct qstr *); +extern struct dentry * d_alloc_trylock(struct dentry *, struct qstr *); extern struct dentry * d_splice_alias(struct inode *, struct dentry *); +struct dentry *d_duplicate(struct dentry *dentry); /* weird procfs mess; *NOT* exported */ extern struct dentry * d_splice_alias_ops(struct inode *, struct dentry *, const struct dentry_operations *); @@ -553,6 +559,36 @@ static inline int simple_positive(const struct dentry *dentry) unsigned long vfs_pressure_ratio(unsigned long val); /** + * d_lookup_release - release ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry is to be passed to another thread which + * will drop the in-lookup lock, then d_lookup_release() must be called + * to tell lockdep that this thread no lock holds the lock. The + * thread that receives the lock must call d_lookup_acquire() to + * acquire the lock. + */ +static inline void d_lookup_release(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_release(&dentry->lookup_map); +} + +/** + * d_lookup_acquire - acquire ownership of DCACHE_PAR_LOOKUP lock + * @dentry: dentry that is locked + * + * If an in-lookup dentry was passed to this thread, the + * d_lookup_acquire() must be called to tell lockdep that this + * thread now owns the DCACHE_PAR_LOOKUP lock. + */ +static inline void d_lookup_acquire(struct dentry *dentry) +{ + if (d_in_lookup(dentry)) + lock_map_acquire_try(&dentry->lookup_map); +} + +/** * d_inode - Get the actual inode of this dentry * @dentry: The dentry to query * diff --git a/include/linux/fdtable.h b/include/linux/fdtable.h index c45306a9f007..a46781058729 100644 --- a/include/linux/fdtable.h +++ b/include/linux/fdtable.h @@ -25,7 +25,7 @@ struct fdtable { unsigned int max_fds; - struct file __rcu **fd; /* current fd array */ + struct file __rcu **fd __counted_by_ptr(max_fds); /* current fd array */ unsigned long *close_on_exec; unsigned long *open_fds; unsigned long *full_fds_bits; @@ -101,11 +101,22 @@ struct task_struct; void put_files_struct(struct files_struct *fs); int unshare_files(void); +void switch_files_struct(struct task_struct *tsk, struct files_struct *files); +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp); +enum fd_range_flags { + /* Leave behind all descriptors outside of the specified range. */ + FD_RANGE_EXCEPT = (1U << 0), + + /* Only select descriptors that have close-on-exec set. */ + FD_RANGE_CLOEXEC_ONLY = (1U << 1), +}; + struct fd_range { unsigned int from, to; + enum fd_range_flags flags; }; struct files_struct *dup_fd(struct files_struct *, struct fd_range *) __latent_entropy; -void do_close_on_exec(struct files_struct *); +void close_cloexec_files(struct files_struct *); int iterate_fd(struct files_struct *, unsigned, int (*)(const void *, struct file *, unsigned), const void *); diff --git a/include/linux/file.h b/include/linux/file.h index 27484b444d31..41c3c0be1064 100644 --- a/include/linux/file.h +++ b/include/linux/file.h @@ -12,6 +12,7 @@ #include <linux/errno.h> #include <linux/cleanup.h> #include <linux/err.h> +#include <linux/vfsdebug.h> struct file; @@ -129,117 +130,84 @@ extern unsigned int sysctl_nr_open_min, sysctl_nr_open_max; /* * fd_prepare: Combined fd + file allocation cleanup class. - * @err: Error code to indicate if allocation succeeded. - * @__fd: Allocated fd (may not be accessed directly) - * @__file: Allocated struct file pointer (may not be accessed directly) + * @fd: Allocated fd + * @file: Allocated struct file pointer * * Allocates an fd and a file together. On error paths, automatically cleans * up whichever resource was successfully allocated. Allows flexible file * allocation with different functions per usage. * - * Do not use directly. + * Do not declare directly, use FD_PREPARE(). */ struct fd_prepare { - s32 err; - s32 __fd; /* do not access directly */ - struct file *__file; /* do not access directly */ + int fd; + struct file *file; }; -/* Typedef for fd_prepare cleanup guards. */ -typedef struct fd_prepare class_fd_prepare_t; - -/* - * Accessors for fd_prepare class members. - * _Generic() is used for zero-cost type safety. - */ -#define fd_prepare_fd(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__fd)) - -#define fd_prepare_file(_fdf) \ - (_Generic((_fdf), struct fd_prepare: (_fdf).__file)) - /* Do not use directly. */ -static inline void class_fd_prepare_destructor(const struct fd_prepare *fdf) +static __always_inline void __fd_prepare_cleanup(const struct fd_prepare *fdf) { - if (unlikely(fdf->__fd >= 0)) - put_unused_fd(fdf->__fd); - if (unlikely(!IS_ERR_OR_NULL(fdf->__file))) - fput(fdf->__file); + if (unlikely(fdf->fd >= 0)) { + put_unused_fd(fdf->fd); + fput(fdf->file); + } } /* Do not use directly. */ -static inline int class_fd_prepare_lock_err(const struct fd_prepare *fdf) +static __always_inline struct fd_prepare __fd_prepare(int fd, struct file *file) { - if (unlikely(fdf->err)) - return fdf->err; - if (unlikely(fdf->__fd < 0)) - return fdf->__fd; - if (unlikely(IS_ERR(fdf->__file))) - return PTR_ERR(fdf->__file); - if (unlikely(!fdf->__file)) - return -ENOMEM; - return 0; + if (fd >= 0 && IS_ERR_OR_NULL(file)) { + int err = file ? PTR_ERR(file) : -ENOMEM; + + put_unused_fd(fd); + fd = err; + file = NULL; + } + return (struct fd_prepare){ .fd = fd, .file = file }; } /* - * __FD_PREPARE_INIT - Helper to initialize fd_prepare class. - * @_fd_flags: flags for get_unused_fd_flags() - * @_file_owned: expression that returns struct file * - * - * Returns a struct fd_prepare with fd, file, and err set. - * If fd allocation fails, fd will be negative and err will be set. If - * fd succeeds but file_init_expr fails, file will be ERR_PTR and err - * will be set. The err field is the single source of truth for error - * checking. - */ -#define __FD_PREPARE_INIT(_fd_flags, _file_owned) \ - ({ \ - struct fd_prepare fdf = { \ - .__fd = get_unused_fd_flags((_fd_flags)), \ - }; \ - if (likely(fdf.__fd >= 0)) \ - fdf.__file = (_file_owned); \ - fdf.err = ACQUIRE_ERR(fd_prepare, &fdf); \ - fdf; \ - }) - -/* - * FD_PREPARE - Macro to declare and initialize an fd_prepare variable. + * FD_PREPARE - Declare and initialize an fd_prepare instance. * - * Declares and initializes an fd_prepare variable with automatic - * cleanup. No separate scope required - cleanup happens when variable - * goes out of scope. + * This allocates a new fd and only evaluates @_file_owned if the + * allocation succeeded. Cleanup happens when the variable goes out of + * scope and the guard releases whichever of the descriptor and the file + * was allocated. If fd_publish() was called the fd and file are + * published and cleanup becomes a nop. * - * @_fdf: name of struct fd_prepare variable to define + * @_fdf: name of the const struct fd_prepare pointer to define * @_fd_flags: flags for get_unused_fd_flags() * @_file_owned: struct file to take ownership of (can be expression) */ +#define __FD_PREPARE(_guard, _fdf, _fd_flags, _file_owned) \ + struct fd_prepare _guard __cleanup(__fd_prepare_cleanup) = ({ \ + int __fd = get_unused_fd_flags(_fd_flags); \ + __fd_prepare(__fd, __fd < 0 ? NULL : (_file_owned)); \ + }); \ + const struct fd_prepare *const _fdf = &_guard + #define FD_PREPARE(_fdf, _fd_flags, _file_owned) \ - CLASS_INIT(fd_prepare, _fdf, __FD_PREPARE_INIT(_fd_flags, _file_owned)) + __FD_PREPARE(__UNIQUE_ID(fd_prepare), _fdf, _fd_flags, _file_owned) /* * fd_publish - Publish prepared fd and file to the fd table. - * @_fdf: struct fd_prepare variable + * @fdf: struct fd_prepare pointer defined by FD_PREPARE() */ -#define fd_publish(_fdf) \ - ({ \ - struct fd_prepare *fdp = &(_fdf); \ - VFS_WARN_ON_ONCE(fdp->err); \ - VFS_WARN_ON_ONCE(fdp->__fd < 0); \ - VFS_WARN_ON_ONCE(IS_ERR_OR_NULL(fdp->__file)); \ - fd_install(fdp->__fd, fdp->__file); \ - retain_and_null_ptr(fdp->__file); \ - take_fd(fdp->__fd); \ - }) +static __always_inline int fd_publish(const struct fd_prepare *fdf) +{ + /* Callers only get a const view, the guard itself is writable. */ + struct fd_prepare *guard = (struct fd_prepare *)fdf; + + VFS_WARN_ON_ONCE(guard->fd < 0); + fd_install(guard->fd, guard->file); + return take_fd(guard->fd); +} /* Do not use directly. */ -#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ - ({ \ - FD_PREPARE(_fdf, _fd_flags, _file_owned); \ - s32 ret = _fdf.err; \ - if (likely(!ret)) \ - ret = fd_publish(_fdf); \ - ret; \ +#define __FD_ADD(_fdf, _fd_flags, _file_owned) \ + ({ \ + FD_PREPARE(_fdf, _fd_flags, _file_owned); \ + _fdf->fd < 0 ? _fdf->fd : fd_publish(_fdf); \ }) /* diff --git a/include/linux/fileattr.h b/include/linux/fileattr.h index 58044b598016..09e32b84e02a 100644 --- a/include/linux/fileattr.h +++ b/include/linux/fileattr.h @@ -74,7 +74,7 @@ static inline bool fileattr_has_fsx(const struct file_kattr *fa) } int vfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ioctl_getflags(struct file *file, unsigned int __user *argp); int ioctl_setflags(struct file *file, unsigned int __user *argp); diff --git a/include/linux/fs.h b/include/linux/fs.h index f9d1e05e8ae6..3db90996756f 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1436,10 +1436,10 @@ static inline void i_gid_write(struct inode *inode, gid_t gid) * @idmap: idmap of the mount the inode was found from * @inode: inode to map * - * Return: whe inode's i_uid mapped down according to @idmap. + * Return: the inode's i_uid mapped down according to @idmap. * If the inode's i_uid has no mapping INVALID_VFSUID is returned. */ -static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, +static inline vfsuid_t i_uid_into_vfsuid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsuid(idmap, i_user_ns(inode), inode->i_uid); @@ -1456,7 +1456,7 @@ static inline vfsuid_t i_uid_into_vfsuid(struct mnt_idmap *idmap, * * Return: true if @inode's i_uid field needs to be updated, false if not. */ -static inline bool i_uid_needs_update(struct mnt_idmap *idmap, +static inline bool i_uid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1474,7 +1474,7 @@ static inline bool i_uid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_uid field translating the vfsuid of any idmapped * mount into the filesystem kuid. */ -static inline void i_uid_update(struct mnt_idmap *idmap, +static inline void i_uid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1491,7 +1491,7 @@ static inline void i_uid_update(struct mnt_idmap *idmap, * Return: the inode's i_gid mapped down according to @idmap. * If the inode's i_gid has no mapping INVALID_VFSGID is returned. */ -static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, +static inline vfsgid_t i_gid_into_vfsgid(const struct mnt_idmap *idmap, const struct inode *inode) { return make_vfsgid(idmap, i_user_ns(inode), inode->i_gid); @@ -1508,7 +1508,7 @@ static inline vfsgid_t i_gid_into_vfsgid(struct mnt_idmap *idmap, * * Return: true if @inode's i_gid field needs to be updated, false if not. */ -static inline bool i_gid_needs_update(struct mnt_idmap *idmap, +static inline bool i_gid_needs_update(const struct mnt_idmap *idmap, const struct iattr *attr, const struct inode *inode) { @@ -1526,7 +1526,7 @@ static inline bool i_gid_needs_update(struct mnt_idmap *idmap, * Safely update @inode's i_gid field translating the vfsgid of any idmapped * mount into the filesystem kgid. */ -static inline void i_gid_update(struct mnt_idmap *idmap, +static inline void i_gid_update(const struct mnt_idmap *idmap, const struct iattr *attr, struct inode *inode) { @@ -1544,7 +1544,7 @@ static inline void i_gid_update(struct mnt_idmap *idmap, * an idmapped mount map the caller's fsuid according to @idmap. */ static inline void inode_fsuid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_uid = mapped_fsuid(idmap, i_user_ns(inode)); } @@ -1558,7 +1558,7 @@ static inline void inode_fsuid_set(struct inode *inode, * an idmapped mount map the caller's fsgid according to @idmap. */ static inline void inode_fsgid_set(struct inode *inode, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { inode->i_gid = mapped_fsgid(idmap, i_user_ns(inode)); } @@ -1575,7 +1575,7 @@ static inline void inode_fsgid_set(struct inode *inode, * Return: true if fsuid and fsgid is mapped, false if not. */ static inline bool fsuidgid_has_mapping(struct super_block *sb, - struct mnt_idmap *idmap) + const struct mnt_idmap *idmap) { struct user_namespace *fs_userns = sb->s_user_ns; kuid_t kuid; @@ -1755,25 +1755,25 @@ static inline bool file_write_not_started(const struct file *file) return sb_write_not_started(file_inode(file)->i_sb); } -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode); /* * VFS helper functions.. */ -int vfs_create(struct mnt_idmap *, struct dentry *, umode_t, +int vfs_create(const struct mnt_idmap *, struct dentry *, umode_t, struct delegated_inode *); -struct dentry *vfs_mkdir(struct mnt_idmap *, struct inode *, +struct dentry *vfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, struct delegated_inode *); -int vfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t, struct delegated_inode *); -int vfs_symlink(struct mnt_idmap *, struct inode *, +int vfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *, struct delegated_inode *); -int vfs_link(struct dentry *, struct mnt_idmap *, struct inode *, +int vfs_link(struct dentry *, const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_rmdir(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_rmdir(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); -int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, +int vfs_unlink(const struct mnt_idmap *, struct inode *, struct dentry *, struct delegated_inode *); /** @@ -1787,7 +1787,7 @@ int vfs_unlink(struct mnt_idmap *, struct inode *, struct dentry *, * @flags: rename flags */ struct renamedata { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; struct dentry *old_parent; struct dentry *old_dentry; struct dentry *new_parent; @@ -1798,14 +1798,14 @@ struct renamedata { int vfs_rename(struct renamedata *); -static inline int vfs_whiteout(struct mnt_idmap *idmap, +static inline int vfs_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry) { return vfs_mknod(idmap, dir, dentry, S_IFCHR | WHITEOUT_MODE, WHITEOUT_DEV, NULL); } -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred); @@ -1830,12 +1830,12 @@ extern long compat_ptr_ioctl(struct file *file, unsigned int cmd, /* * VFS file helper functions. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode); extern bool may_open_dev(const struct path *path); -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode); -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -1994,26 +1994,26 @@ enum fs_update_time { struct inode_operations { struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); const char * (*get_link) (struct dentry *, struct inode *, struct delayed_call *); - int (*permission) (struct mnt_idmap *, struct inode *, int); + int (*permission) (const struct mnt_idmap *, struct inode *, int); struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); int (*readlink) (struct dentry *, char __user *,int); - int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*create) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); int (*link) (struct dentry *,struct inode *,struct dentry *); int (*unlink) (struct inode *,struct dentry *); - int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*symlink) (const struct mnt_idmap *, struct inode *,struct dentry *, const char *); - struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *, + struct dentry *(*mkdir) (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int (*rmdir) (struct inode *,struct dentry *); - int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *, + int (*mknod) (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); - int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, + int (*rename) (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); - int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); - int (*getattr) (struct mnt_idmap *, const struct path *, + int (*setattr) (const struct mnt_idmap *, struct dentry *, struct iattr *); + int (*getattr) (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); ssize_t (*listxattr) (struct dentry *, char *, size_t); int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, @@ -2024,13 +2024,13 @@ struct inode_operations { int (*atomic_open)(struct inode *, struct dentry *, struct file *, unsigned open_flag, umode_t create_mode); - int (*tmpfile) (struct mnt_idmap *, struct inode *, + int (*tmpfile) (const struct mnt_idmap *, struct inode *, struct file *, umode_t); - struct posix_acl *(*get_acl)(struct mnt_idmap *, struct dentry *, + struct posix_acl *(*get_acl)(const struct mnt_idmap *, struct dentry *, int); - int (*set_acl)(struct mnt_idmap *, struct dentry *, + int (*set_acl)(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); - int (*fileattr_set)(struct mnt_idmap *idmap, + int (*fileattr_set)(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); struct offset_ctx *(*get_offset_ctx)(struct inode *inode); @@ -2173,7 +2173,7 @@ extern loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, (inode)->i_rdev == WHITEOUT_DEV) #define IS_ANON_FILE(inode) ((inode)->i_flags & S_ANON_INODE) -static inline bool HAS_UNMAPPED_ID(struct mnt_idmap *idmap, +static inline bool HAS_UNMAPPED_ID(const struct mnt_idmap *idmap, struct inode *inode) { return !vfsuid_valid(i_uid_into_vfsuid(idmap, inode)) || @@ -2459,7 +2459,7 @@ struct filename { static_assert(offsetof(struct filename, iname) % sizeof(long) == 0); static_assert(sizeof(struct filename) % 64 == 0); -static inline struct mnt_idmap *file_mnt_idmap(const struct file *file) +static inline const struct mnt_idmap *file_mnt_idmap(const struct file *file) { return mnt_idmap(file->f_path.mnt); } @@ -2483,7 +2483,7 @@ static inline bool is_idmapped_mnt(const struct vfsmount *mnt) } int vfs_truncate(const struct path *, loff_t); -int do_truncate(struct mnt_idmap *, struct dentry *, loff_t start, +int do_truncate(const struct mnt_idmap *, struct dentry *, loff_t start, unsigned int time_attrs, struct file *filp); extern int vfs_fallocate(struct file *file, int mode, loff_t offset, loff_t len); @@ -2707,10 +2707,10 @@ static inline int bmap(struct inode *inode, sector_t *block) } #endif -int notify_change(struct mnt_idmap *, struct dentry *, +int notify_change(const struct mnt_idmap *, struct dentry *, struct iattr *, struct delegated_inode *); -int inode_permission(struct mnt_idmap *, struct inode *, int); -int generic_permission(struct mnt_idmap *, struct inode *, int); +int inode_permission(const struct mnt_idmap *, struct inode *, int); +int generic_permission(const struct mnt_idmap *, struct inode *, int); static inline int file_permission(struct file *file, int mask) { return inode_permission(file_mnt_idmap(file), @@ -2721,12 +2721,12 @@ static inline int path_permission(const struct path *path, int mask) return inode_permission(mnt_idmap(path->mnt), d_inode(path->dentry), mask); } -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode); -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir); -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child); static inline bool execute_ok(struct inode *inode) @@ -3045,9 +3045,9 @@ static inline struct inode *new_inode_pseudo(struct super_block *sb) } extern struct inode *new_inode(struct super_block *sb); extern void free_inode_nonrcu(struct inode *inode); -extern int setattr_should_drop_suidgid(struct mnt_idmap *, struct inode *); +extern int setattr_should_drop_suidgid(const struct mnt_idmap *, struct inode *); extern int file_remove_privs(struct file *); -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode); /* @@ -3204,7 +3204,7 @@ extern int page_symlink(struct inode *inode, const char *symname, int len); extern const struct inode_operations page_symlink_inode_operations; extern void kfree_link(void *); void fill_mg_cmtime(struct kstat *stat, u32 request_mask, struct inode *inode); -void generic_fillattr(struct mnt_idmap *, u32, struct inode *, struct kstat *); +void generic_fillattr(const struct mnt_idmap *, u32, struct inode *, struct kstat *); void generic_fill_statx_attr(struct inode *inode, struct kstat *stat); void generic_fill_statx_atomic_writes(struct kstat *stat, unsigned int unit_min, @@ -3261,9 +3261,9 @@ extern int dcache_dir_open(struct inode *, struct file *); extern int dcache_dir_close(struct inode *, struct file *); extern loff_t dcache_dir_lseek(struct file *, loff_t, int); extern int dcache_readdir(struct file *, struct dir_context *); -extern int simple_setattr(struct mnt_idmap *, struct dentry *, +extern int simple_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -extern int simple_getattr(struct mnt_idmap *, const struct path *, +extern int simple_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern int simple_statfs(struct dentry *, struct kstatfs *); extern int simple_open(struct inode *inode, struct file *file); @@ -3276,7 +3276,7 @@ void simple_rename_timestamp(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); extern int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry); -extern int simple_rename(struct mnt_idmap *, struct inode *, +extern int simple_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); extern void simple_recursive_removal(struct dentry *, @@ -3397,11 +3397,11 @@ static inline bool generic_ci_validate_strict_name(struct inode *dir, } #endif -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid); -int setattr_prepare(struct mnt_idmap *, struct dentry *, struct iattr *); +int setattr_prepare(const struct mnt_idmap *, struct dentry *, struct iattr *); extern int inode_newsize_ok(const struct inode *, loff_t offset); -void setattr_copy(struct mnt_idmap *, struct inode *inode, +void setattr_copy(const struct mnt_idmap *, struct inode *inode, const struct iattr *attr); extern int file_update_time(struct file *file); @@ -3578,7 +3578,7 @@ static inline bool is_sxid(umode_t mode) return mode & (S_ISUID | S_ISGID); } -static inline int check_sticky(struct mnt_idmap *idmap, +static inline int check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { if (!(dir->i_mode & S_ISVTX)) @@ -3653,23 +3653,6 @@ extern int vfs_fadvise(struct file *file, loff_t offset, loff_t len, extern int generic_fadvise(struct file *file, loff_t offset, loff_t len, int advice); -static inline bool vfs_empty_path(int dfd, const char __user *path) -{ - char c; - - if (dfd < 0) - return false; - - /* We now allow NULL to be used for empty path. */ - if (!path) - return true; - - if (unlikely(get_user(c, path))) - return false; - - return !c; -} - int generic_atomic_write_valid(struct kiocb *iocb, struct iov_iter *iter); static inline bool extensible_ioctl_valid(unsigned int cmd_a, diff --git a/include/linux/fs_context.h b/include/linux/fs_context.h index 0d6c8a6d7be2..c920aba5177c 100644 --- a/include/linux/fs_context.h +++ b/include/linux/fs_context.h @@ -150,6 +150,10 @@ extern int vfs_parse_fs_param_source(struct fs_context *fc, struct fs_parameter *param); extern void fc_drop_locked(struct fs_context *fc); +extern int get_tree_super(struct fs_context *fc, + int (*test)(struct super_block *, struct fs_context *), + int (*fill_super)(struct super_block *sb, + struct fs_context *fc)); extern int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)); diff --git a/include/linux/fscache-cache.h b/include/linux/fscache-cache.h index 4c91a019972b..ee524c863fa9 100644 --- a/include/linux/fscache-cache.h +++ b/include/linux/fscache-cache.h @@ -67,7 +67,7 @@ struct fscache_cache_ops { /* Change the size of a data object */ void (*resize_cookie)(struct netfs_cache_resources *cres, - loff_t new_size); + uoff_t new_size); /* Invalidate an object */ bool (*invalidate_cookie)(struct fscache_cookie *cookie); diff --git a/include/linux/fscache.h b/include/linux/fscache.h index 58fdb9605425..f2d958bd1f48 100644 --- a/include/linux/fscache.h +++ b/include/linux/fscache.h @@ -112,7 +112,7 @@ struct fscache_cookie { struct list_head proc_link; /* Link in proc list */ struct list_head commit_link; /* Link in commit queue */ struct work_struct work; /* Commit/relinq/withdraw work */ - loff_t object_size; /* Size of the netfs object */ + uoff_t object_size; /* Size of the netfs object */ unsigned long unused_at; /* Time at which unused (jiffies) */ unsigned long flags; #define FSCACHE_COOKIE_RELINQUISHED 0 /* T if cookie has been relinquished */ @@ -147,6 +147,23 @@ struct fscache_cookie { }; }; +enum fscache_extent_type { + FSCACHE_EXTENT_DATA, + FSCACHE_EXTENT_ZERO, +} __mode(byte); + +/* + * Cache occupancy information. + */ +struct fscache_occupancy { + unsigned long long query_from; /* Point to query from */ + unsigned long long query_to; /* Point to query to */ + unsigned long long cached_from[2]; /* Point at which cache extents start */ + unsigned long long cached_to[2]; /* Point at which cache extents end */ + unsigned int granularity; /* Granularity desired */ + enum fscache_extent_type cached_type[2]; /* Type of cache extent */ +}; + /* * slow-path functions for when there is actually caching available, and the * netfs does actually have a valid token @@ -163,22 +180,22 @@ extern struct fscache_cookie *__fscache_acquire_cookie( u8, const void *, size_t, const void *, size_t, - loff_t); + uoff_t); extern void __fscache_use_cookie(struct fscache_cookie *, bool); -extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const loff_t *); +extern void __fscache_unuse_cookie(struct fscache_cookie *, const void *, const uoff_t *); extern void __fscache_relinquish_cookie(struct fscache_cookie *, bool); -extern void __fscache_resize_cookie(struct fscache_cookie *, loff_t); -extern void __fscache_invalidate(struct fscache_cookie *, const void *, loff_t, unsigned int); +extern void __fscache_resize_cookie(struct fscache_cookie *, uoff_t); +extern void __fscache_invalidate(struct fscache_cookie *, const void *, uoff_t, unsigned int); extern int __fscache_begin_read_operation(struct netfs_cache_resources *, struct fscache_cookie *); extern int __fscache_begin_write_operation(struct netfs_cache_resources *, struct fscache_cookie *); void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond); -extern void __fscache_clear_page_bits(struct address_space *, loff_t, size_t); +extern void __fscache_clear_page_bits(struct address_space *, uoff_t, size_t); /** * fscache_acquire_volume - Register a volume as desiring caching services @@ -249,7 +266,7 @@ struct fscache_cookie *fscache_acquire_cookie(struct fscache_volume *volume, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { if (!fscache_volume_valid(volume)) return NULL; @@ -286,7 +303,7 @@ static inline void fscache_use_cookie(struct fscache_cookie *cookie, */ static inline void fscache_unuse_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_valid(cookie)) __fscache_unuse_cookie(cookie, aux_data, object_size); @@ -327,7 +344,7 @@ static inline void *fscache_get_aux(struct fscache_cookie *cookie) */ static inline void fscache_update_aux(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { void *p = fscache_get_aux(cookie); @@ -343,7 +360,7 @@ extern atomic_t fscache_n_updates; static inline void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { #ifdef CONFIG_FSCACHE_STATS atomic_inc(&fscache_n_updates); @@ -369,7 +386,7 @@ void __fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data */ static inline void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, - const loff_t *object_size) + const uoff_t *object_size) { if (fscache_cookie_enabled(cookie)) __fscache_update_cookie(cookie, aux_data, object_size); @@ -386,7 +403,7 @@ void fscache_update_cookie(struct fscache_cookie *cookie, const void *aux_data, * description. */ static inline -void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { if (fscache_cookie_enabled(cookie)) __fscache_resize_cookie(cookie, new_size); @@ -413,7 +430,7 @@ void fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) */ static inline void fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t size, unsigned int flags) + const void *aux_data, uoff_t size, unsigned int flags) { if (fscache_cookie_enabled(cookie)) __fscache_invalidate(cookie, aux_data, size, flags); @@ -502,7 +519,7 @@ static inline void fscache_end_operation(struct netfs_cache_resources *cres) */ static inline int fscache_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -561,7 +578,7 @@ int fscache_begin_write_operation(struct netfs_cache_resources *cres, */ static inline int fscache_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -581,7 +598,7 @@ int fscache_write(struct netfs_cache_resources *cres, * waiting. */ static inline void fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len, + uoff_t start, size_t len, bool caching) { if (caching) @@ -615,7 +632,7 @@ static inline void fscache_clear_page_bits(struct address_space *mapping, */ static inline void fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool caching) diff --git a/include/linux/iomap.h b/include/linux/iomap.h index bc7ae6327dbf..59718f73c15a 100644 --- a/include/linux/iomap.h +++ b/include/linux/iomap.h @@ -483,13 +483,35 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, #define IOMAP_IOEND_BOUNDARY (1U << 2) /* is direct I/O */ #define IOMAP_IOEND_DIRECT (1U << 3) +/* generate integrity (PI) information */ +#ifdef CONFIG_BLK_DEV_INTEGRITY +#define IOMAP_IOEND_INTEGRITY (1U << 4) +#else +#define IOMAP_IOEND_INTEGRITY 0 +#endif /* CONFIG_BLK_DEV_INTEGRITY */ /* * Flags that if set on either ioend prevent the merge of two ioends. * (IOMAP_IOEND_BOUNDARY also prevents merges, but only one-way) */ #define IOMAP_IOEND_NOMERGE_FLAGS \ - (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT) + (IOMAP_IOEND_SHARED | IOMAP_IOEND_UNWRITTEN | IOMAP_IOEND_DIRECT | \ + IOMAP_IOEND_INTEGRITY) + +/* ioend flags directly implied by iomap flags */ +static inline u16 iomap_ioend_flags(const struct iomap *iomap) +{ + unsigned int flags = 0; + + if (iomap->type == IOMAP_UNWRITTEN) + flags |= IOMAP_IOEND_UNWRITTEN; + if (iomap->flags & IOMAP_F_SHARED) + flags |= IOMAP_IOEND_SHARED; + if (iomap->flags & IOMAP_F_INTEGRITY) + flags |= IOMAP_IOEND_INTEGRITY; + + return flags; +} /* * Structure for writeback I/O completions. @@ -500,6 +522,7 @@ sector_t iomap_bmap(struct address_space *mapping, sector_t bno, struct iomap_ioend { struct list_head io_list; /* next ioend in chain */ u16 io_flags; /* IOMAP_IOEND_* */ + u32 io_bvec_offset; /* offset into first bvec */ struct inode *io_inode; /* file being written to */ size_t io_size; /* size of the extent */ atomic_t io_remaining; /* completetion defer count */ @@ -517,6 +540,13 @@ static inline struct iomap_ioend *iomap_ioend_from_bio(struct bio *bio) return container_of(bio, struct iomap_ioend, io_bio); } +#define BVEC_ITER_IOEND(_ioend) \ +{ \ + .bi_sector = (_ioend)->io_sector, \ + .bi_size = (_ioend)->io_size, \ + .bi_offset = (_ioend)->io_bvec_offset, \ +} + struct iomap_writeback_ops { /* * Performs writeback on the passed in range @@ -565,6 +595,7 @@ void iomap_finish_ioends(struct iomap_ioend *ioend, int error); void iomap_ioend_try_merge(struct iomap_ioend *ioend, struct list_head *more_ioends); void iomap_sort_ioends(struct list_head *ioend_list); +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend); ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, loff_t pos, loff_t end_pos, unsigned int dirty_len); int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error); @@ -577,6 +608,11 @@ void iomap_finish_folio_write(struct inode *inode, struct folio *folio, int iomap_writeback_folio(struct iomap_writepage_ctx *wpc, struct folio *folio); int iomap_writepages(struct iomap_writepage_ctx *wpc); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)); +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error); + struct iomap_read_folio_ctx { const struct iomap_read_ops *ops; struct folio *cur_folio; diff --git a/include/linux/lsm_hook_defs.h b/include/linux/lsm_hook_defs.h index 65c9609ec207..c9561564585e 100644 --- a/include/linux/lsm_hook_defs.h +++ b/include/linux/lsm_hook_defs.h @@ -36,6 +36,7 @@ LSM_HOOK(int, 0, binder_transfer_file, const struct cred *from, LSM_HOOK(int, 0, ptrace_access_check, struct task_struct *child, unsigned int mode) LSM_HOOK(int, 0, ptrace_traceme, struct task_struct *parent) +LSM_HOOK(int, 0, mem_foll_force, const struct cred *subject, bool opened_by_owner) LSM_HOOK(int, 0, capget, const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, kernel_cap_t *permitted) LSM_HOOK(int, 0, capset, struct cred *new, const struct cred *old, @@ -94,7 +95,7 @@ LSM_HOOK(int, 0, path_mkdir, const struct path *dir, struct dentry *dentry, LSM_HOOK(int, 0, path_rmdir, const struct path *dir, struct dentry *dentry) LSM_HOOK(int, 0, path_mknod, const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev) -LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, path_post_mknod, const struct mnt_idmap *idmap, struct dentry *dentry) LSM_HOOK(int, 0, path_truncate, const struct path *path) LSM_HOOK(int, 0, path_symlink, const struct path *dir, struct dentry *dentry, @@ -122,7 +123,7 @@ LSM_HOOK(int, 0, inode_init_security_anon, struct inode *inode, const struct qstr *name, const struct inode *context_inode) LSM_HOOK(int, 0, inode_create, struct inode *dir, struct dentry *dentry, umode_t mode) -LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_create_tmpfile, const struct mnt_idmap *idmap, struct inode *inode) LSM_HOOK(int, 0, inode_link, struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry) @@ -140,39 +141,39 @@ LSM_HOOK(int, 0, inode_readlink, struct dentry *dentry) LSM_HOOK(int, 0, inode_follow_link, struct dentry *dentry, struct inode *inode, bool rcu) LSM_HOOK(int, 0, inode_permission, struct inode *inode, int mask) -LSM_HOOK(int, 0, inode_setattr, struct mnt_idmap *idmap, struct dentry *dentry, +LSM_HOOK(int, 0, inode_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) -LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_setattr, const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) LSM_HOOK(int, 0, inode_getattr, const struct path *path) LSM_HOOK(int, 0, inode_xattr_skipcap, const char *name) -LSM_HOOK(int, 0, inode_setxattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_setxattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(void, LSM_RET_VOID, inode_post_setxattr, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) LSM_HOOK(int, 0, inode_getxattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_listxattr, struct dentry *dentry) -LSM_HOOK(int, 0, inode_removexattr, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_removexattr, const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) LSM_HOOK(void, LSM_RET_VOID, inode_post_removexattr, struct dentry *dentry, const char *name) LSM_HOOK(int, 0, inode_file_setattr, struct dentry *dentry, struct file_kattr *fa) LSM_HOOK(int, 0, inode_file_getattr, struct dentry *dentry, struct file_kattr *fa) -LSM_HOOK(int, 0, inode_set_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_set_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) LSM_HOOK(void, LSM_RET_VOID, inode_post_set_acl, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) -LSM_HOOK(int, 0, inode_get_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_get_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(int, 0, inode_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) -LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, struct mnt_idmap *idmap, +LSM_HOOK(void, LSM_RET_VOID, inode_post_remove_acl, const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) LSM_HOOK(int, 0, inode_need_killpriv, struct dentry *dentry) -LSM_HOOK(int, 0, inode_killpriv, struct mnt_idmap *idmap, +LSM_HOOK(int, 0, inode_killpriv, const struct mnt_idmap *idmap, struct dentry *dentry) -LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, struct mnt_idmap *idmap, +LSM_HOOK(int, -EOPNOTSUPP, inode_getsecurity, const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) LSM_HOOK(int, -EOPNOTSUPP, inode_setsecurity, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/include/linux/mnt_idmapping.h b/include/linux/mnt_idmapping.h index e71a6070a8f8..78eeef4c2996 100644 --- a/include/linux/mnt_idmapping.h +++ b/include/linux/mnt_idmapping.h @@ -8,8 +8,8 @@ struct mnt_idmap; struct user_namespace; -extern struct mnt_idmap nop_mnt_idmap; -extern struct mnt_idmap invalid_mnt_idmap; +extern const struct mnt_idmap nop_mnt_idmap; +extern const struct mnt_idmap invalid_mnt_idmap; extern struct user_namespace init_user_ns; typedef struct { @@ -121,19 +121,19 @@ static inline bool vfsgid_eq_kgid(vfsgid_t vfsgid, kgid_t kgid) int vfsgid_in_group_p(vfsgid_t vfsgid); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid); -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid); -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid); -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid); /** @@ -148,7 +148,7 @@ kgid_t from_vfsgid(struct mnt_idmap *idmap, * * Return: true if @vfsuid has a mapping in the filesystem, false if not. */ -static inline bool vfsuid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsuid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { @@ -186,7 +186,7 @@ static inline kuid_t vfsuid_into_kuid(vfsuid_t vfsuid) * * Return: true if @vfsgid has a mapping in the filesystem, false if not. */ -static inline bool vfsgid_has_fsmapping(struct mnt_idmap *idmap, +static inline bool vfsgid_has_fsmapping(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { @@ -225,7 +225,7 @@ static inline kgid_t vfsgid_into_kgid(vfsgid_t vfsgid) * * Return: the caller's current fsuid mapped up according to @idmap. */ -static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, +static inline kuid_t mapped_fsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsuid(idmap, fs_userns, VFSUIDT_INIT(current_fsuid())); @@ -244,7 +244,7 @@ static inline kuid_t mapped_fsuid(struct mnt_idmap *idmap, * * Return: the caller's current fsgid mapped up according to @idmap. */ -static inline kgid_t mapped_fsgid(struct mnt_idmap *idmap, +static inline kgid_t mapped_fsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns) { return from_vfsgid(idmap, fs_userns, VFSGIDT_INIT(current_fsgid())); diff --git a/include/linux/mount.h b/include/linux/mount.h index acfe7ef86a1b..e90ccafef281 100644 --- a/include/linux/mount.h +++ b/include/linux/mount.h @@ -59,10 +59,10 @@ struct vfsmount { struct dentry *mnt_root; /* root of the mounted tree */ struct super_block *mnt_sb; /* pointer to superblock */ int mnt_flags; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; } __randomize_layout; -static inline struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) +static inline const struct mnt_idmap *mnt_idmap(const struct vfsmount *mnt) { /* Pairs with smp_store_release() in do_idmap_mount(). */ return READ_ONCE(mnt->mnt_idmap); diff --git a/include/linux/namei.h b/include/linux/namei.h index 86d657b24fc6..c4436e5c2ba6 100644 --- a/include/linux/namei.h +++ b/include/linux/namei.h @@ -32,8 +32,9 @@ enum { MAX_NESTED_LINKS = 8 }; #define LOOKUP_CREATE BIT(17) /* ... in object creation */ #define LOOKUP_EXCL BIT(18) /* ... in target must not exist */ #define LOOKUP_RENAME_TARGET BIT(19) /* ... in destination of rename() */ +#define LOOKUP_SHARED BIT(20) /* Parent lock is held shared */ -/* 4 spare bits for intent */ +/* 3 spare bits for intent */ /* Scoping flags for lookup. */ #define LOOKUP_NO_SYMLINKS BIT(24) /* No symlink crossing. */ @@ -70,24 +71,24 @@ extern struct dentry *try_lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_unlocked(struct qstr *, struct dentry *); extern struct dentry *lookup_noperm_positive_unlocked(struct qstr *, struct dentry *); -struct dentry *lookup_one(struct mnt_idmap *, struct qstr *, struct dentry *); -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one(const struct mnt_idmap *, struct qstr *, struct dentry *); +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base); -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name); struct dentry *start_creating_noperm(struct dentry *parent, struct qstr *name); diff --git a/include/linux/netfs.h b/include/linux/netfs.h index b4dd32863dd4..67e010b6994b 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -22,6 +22,7 @@ enum netfs_sreq_ref_trace; typedef struct mempool mempool_t; +struct fscache_occupancy; struct folio_queue; /** @@ -62,8 +63,8 @@ struct netfs_inode { struct fscache_cookie *cache; #endif struct list_head wb_queue; /* Queue of processes wanting to do writeback */ - loff_t _remote_i_size; /* Size of the remote file */ - loff_t _zero_point; /* Size after which we assume there's no data + uoff_t _remote_i_size; /* Size of the remote file */ + uoff_t _zero_point; /* Size after which we assume there's no data * on the server */ spinlock_t lock; /* Lock covering wb_queue */ atomic_t io_count; /* Number of outstanding reqs */ @@ -125,6 +126,12 @@ static inline struct netfs_group *netfs_folio_group(struct folio *folio) return priv; } +enum netfs_cache_collect { + NETFS_CACHE_COLLECT_WRITE_GAP, /* Gap in collection, no state either way */ + NETFS_CACHE_COLLECT_WRITE_DATA, /* Currently collecting good writes */ + NETFS_CACHE_COLLECT_WRITE_CANCEL, /* Currently collecting cancelled writes */ +}; + /* * Stream of I/O subrequests going to a particular destination, such as the * server or the local cache. This is mainly intended for writing where we may @@ -142,7 +149,7 @@ struct netfs_io_stream { void (*issue_write)(struct netfs_io_subrequest *subreq); /* Collection tracking */ struct list_head subrequests; /* Contributory I/O operations */ - unsigned long long collected_to; /* Position we've collected results to */ + uoff_t collected_to; /* Position we've collected results to */ size_t transferred; /* The amount transferred from this stream */ unsigned short error; /* Aggregate error for the stream */ enum netfs_io_source source; /* Where to read from/write to */ @@ -152,6 +159,7 @@ struct netfs_io_stream { bool need_retry; /* T if this stream needs retrying */ bool failed; /* T if this stream failed */ bool transferred_valid; /* T is ->transferred is valid */ + enum netfs_cache_collect cache_collect; /* Current writeback cache collect state */ }; /* @@ -161,8 +169,11 @@ struct netfs_cache_resources { const struct netfs_cache_ops *ops; void *cache_priv; void *cache_priv2; - unsigned int debug_id; /* Cookie debug ID */ + uoff_t cache_i_size; /* Initial size of cache file */ + unsigned int cookie_id; /* Cache cookie debug ID */ + unsigned int object_id; /* Cache object debug ID */ unsigned int inval_counter; /* object->inval_counter at begin_op */ + unsigned int dio_size; /* DIO block size */ }; /* @@ -177,7 +188,7 @@ struct netfs_io_subrequest { struct work_struct work; struct list_head rreq_link; /* Link in rreq->subrequests */ struct iov_iter io_iter; /* Iterator for this subrequest */ - unsigned long long start; /* Where to start the I/O */ + uoff_t start; /* Where to start the I/O */ size_t len; /* Size of the I/O */ size_t transferred; /* Amount of data transferred */ refcount_t ref; @@ -196,6 +207,7 @@ struct netfs_io_subrequest { #define NETFS_SREQ_IN_PROGRESS 8 /* Unlocked when the subrequest completes */ #define NETFS_SREQ_NEED_RETRY 9 /* Set if the filesystem requests a retry */ #define NETFS_SREQ_FAILED 10 /* Set if the subreq failed unretryably */ +#define NETFS_SREQ_CANCELLED 11 /* Set if the subreq was cancelled by netfslib */ }; enum netfs_io_origin { @@ -208,7 +220,6 @@ enum netfs_io_origin { NETFS_DIO_READ, /* This is a direct I/O read */ NETFS_WRITEBACK, /* This write was triggered by writepages */ NETFS_WRITEBACK_SINGLE, /* This monolithic write was triggered by writepages */ - NETFS_WRITETHROUGH, /* This write was made by netfs_perform_write() */ NETFS_UNBUFFERED_WRITE, /* This is an unbuffered write */ NETFS_DIO_WRITE, /* This is a direct I/O write */ NETFS_PGPRIV2_COPY_TO_CACHE, /* [DEPRECATED] This is writing read data to the cache */ @@ -243,17 +254,18 @@ struct netfs_io_request { void *netfs_priv; /* Private data for the netfs */ void *netfs_priv2; /* Private data for the netfs */ struct bio_vec *direct_bv; /* DIO buffer list (when handling iovec-iter) */ - unsigned long long submitted; /* Amount submitted for I/O so far */ - unsigned long long len; /* Length of the request */ + uoff_t submitted; /* Amount submitted for I/O so far */ + uoff_t len; /* Length of the request */ size_t transferred; /* Amount to be indicated as transferred */ size_t progress_at; /* Report read progress when hit this much read */ long error; /* 0 or error that occurred */ - unsigned long long i_size; /* Size of the file */ - unsigned long long start; /* Start position */ + uoff_t i_size; /* Size of the file */ + uoff_t start; /* Start position */ atomic64_t issued_to; /* Write issuer folio cursor */ - unsigned long long collected_to; /* Point we've collected to */ - unsigned long long cleaned_to; /* Position we've cleaned folios to */ - unsigned long long abandon_to; /* Position to abandon folios to */ + uoff_t collected_to; /* Point we've collected to */ + uoff_t cache_coll_to; /* Point the cache has collected to */ + uoff_t cleaned_to; /* Position we've cleaned folios to */ + uoff_t abandon_to; /* Position to abandon folios to */ const struct folio *no_unlock_folio; /* Don't unlock this folio after read */ gfp_t gfp; /* GFP flags to use */ unsigned int direct_bv_count; /* Number of elements in direct_bv[] */ @@ -273,14 +285,18 @@ struct netfs_io_request { #define NETFS_RREQ_FAILED 3 /* The request failed */ #define NETFS_RREQ_RETRYING 4 /* Set if we're in the retry path */ #define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */ -#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */ -#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_CACHE_STOP 8 /* Set to stop caching (ENOBUFS or error) */ +#define NETFS_RREQ_CACHE_ERROR 9 /* Set if we got an error from the cache */ #define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */ -#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */ -#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */ +#define NETFS_RREQ_OFFLOAD_COLLECTION 12 /* Offload collection to workqueue */ +#define NETFS_RREQ_NO_UNLOCK_FOLIO 13 /* Don't unlock no_unlock_folio on completion */ +#define NETFS_RREQ_UPLOAD_TO_SERVER 14 /* Need to write to the server */ +#define NETFS_RREQ_USE_IO_ITER 15 /* Use ->io_iter rather than ->i_pages */ #define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */ +#ifdef CONFIG_NETFS_PGPRIV2 #define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark * write to cache on read */ +#endif const struct netfs_request_ops *netfs_ops; }; @@ -299,12 +315,12 @@ struct netfs_request_ops { int (*prepare_read)(struct netfs_io_subrequest *subreq); void (*issue_read)(struct netfs_io_subrequest *subreq); bool (*is_still_valid)(struct netfs_io_request *rreq); - int (*check_write_begin)(struct file *file, loff_t pos, unsigned len, + int (*check_write_begin)(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata); void (*done)(struct netfs_io_request *rreq); /* Modification handling */ - void (*update_i_size)(struct inode *inode, loff_t i_size); + void (*update_i_size)(struct inode *inode, uoff_t i_size); void (*post_modify)(struct inode *inode); /* Write request handling */ @@ -332,7 +348,7 @@ struct netfs_cache_ops { /* Read data from the cache */ int (*read)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -340,7 +356,7 @@ struct netfs_cache_ops { /* Write data to the cache */ int (*write)(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -350,15 +366,14 @@ struct netfs_cache_ops { /* Expand readahead request */ void (*expand_readahead)(struct netfs_cache_resources *cres, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size); + uoff_t *_start, + uoff_t *_len, + uoff_t i_size); /* Prepare a read operation, shortening it to a cached/uncached * boundary as appropriate. */ - enum netfs_io_source (*prepare_read)(struct netfs_io_subrequest *subreq, - unsigned long long i_size); + int (*prepare_read)(struct netfs_io_subrequest *subreq); /* Prepare a write subrequest, working out if we're allowed to do it * and finding out the maximum amount of data to gather before @@ -371,15 +386,24 @@ struct netfs_cache_ops { * actually do. */ int (*prepare_write)(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet); + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet); /* Query the occupancy of the cache in a region, returning where the * next chunk of data starts and how long it is. */ int (*query_occupancy)(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len); + struct fscache_occupancy *occ); + + /* Collect the result of buffered writeback to the cache. This + * includes copying a read to the cache. block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a cancellation gap + */ + void (*collect_write)(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type); }; /* High-level read API. */ @@ -410,7 +434,7 @@ struct readahead_control; void netfs_readahead(struct readahead_control *); int netfs_read_folio(struct file *, struct folio *); int netfs_write_begin(struct netfs_inode *, struct file *, - struct address_space *, loff_t pos, unsigned int len, + struct address_space *, uoff_t pos, unsigned int len, struct folio **, void **fsdata); int netfs_writepages(struct address_space *mapping, struct writeback_control *wbc); @@ -488,10 +512,10 @@ static inline struct netfs_inode *netfs_inode(struct inode *inode) * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_remote_i_size(const struct inode *inode) +static inline uoff_t netfs_read_remote_i_size(const struct inode *inode) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long remote_i_size; + uoff_t remote_i_size; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -526,7 +550,7 @@ static inline unsigned long long netfs_read_remote_i_size(const struct inode *in * spinning forever. */ static inline void netfs_write_remote_i_size(struct inode *inode, - unsigned long long remote_i_size) + uoff_t remote_i_size) { struct netfs_inode *ictx = netfs_inode(inode); @@ -563,10 +587,10 @@ static inline void netfs_write_remote_i_size(struct inode *inode, * cmpxchg8b without the need of the lock prefix). For SMP compiles and 64bit * archs it makes no difference if preempt is enabled or not. */ -static inline unsigned long long netfs_read_zero_point(const struct inode *inode) +static inline uoff_t netfs_read_zero_point(const struct inode *inode) { struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); - unsigned long long zero_point; + uoff_t zero_point; #if BITS_PER_LONG==32 && defined(CONFIG_SMP) unsigned int seq; @@ -601,7 +625,7 @@ static inline unsigned long long netfs_read_zero_point(const struct inode *inode * forever. */ static inline void netfs_write_zero_point(struct inode *inode, - unsigned long long zero_point) + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -642,9 +666,9 @@ static inline void netfs_write_zero_point(struct inode *inode, * archs it makes no difference if preempt is enabled or not. */ static inline void netfs_read_sizes(const struct inode *inode, - unsigned long long *i_size, - unsigned long long *remote_i_size, - unsigned long long *zero_point) + uoff_t *i_size, + uoff_t *remote_i_size, + uoff_t *zero_point) { const struct netfs_inode *ictx = container_of(inode, struct netfs_inode, inode); #if BITS_PER_LONG==32 && defined(CONFIG_SMP) @@ -690,9 +714,9 @@ static inline void netfs_read_sizes(const struct inode *inode, * forever. */ static inline void netfs_write_sizes(struct inode *inode, - unsigned long long i_size, - unsigned long long remote_i_size, - unsigned long long zero_point) + uoff_t i_size, + uoff_t remote_i_size, + uoff_t zero_point) { struct netfs_inode *ictx = netfs_inode(inode); @@ -760,7 +784,7 @@ static inline void netfs_inode_init(struct netfs_inode *ctx, * Inform the netfs lib that a file got resized so that it can adjust its state. */ static inline void netfs_resize_file(struct netfs_inode *ictx, - unsigned long long new_i_size, + uoff_t new_i_size, bool changed_on_server) { #if BITS_PER_LONG==32 && defined(CONFIG_SMP) diff --git a/include/linux/nfs_fs.h b/include/linux/nfs_fs.h index b85a73ae7919..d2c716322c6f 100644 --- a/include/linux/nfs_fs.h +++ b/include/linux/nfs_fs.h @@ -437,11 +437,11 @@ extern int nfs_refresh_inode(struct inode *, struct nfs_fattr *); extern int nfs_post_op_update_inode(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc(struct inode *inode, struct nfs_fattr *fattr); extern int nfs_post_op_update_inode_force_wcc_locked(struct inode *inode, struct nfs_fattr *fattr); -extern int nfs_getattr(struct mnt_idmap *, const struct path *, +extern int nfs_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void nfs_access_add_cache(struct inode *, struct nfs_access_entry *, const struct cred *); extern void nfs_access_set_mask(struct nfs_access_entry *, u32); -extern int nfs_permission(struct mnt_idmap *, struct inode *, int); +extern int nfs_permission(const struct mnt_idmap *, struct inode *, int); extern int nfs_open(struct inode *, struct file *); extern int nfs_attribute_cache_expired(struct inode *inode); extern int nfs_revalidate_inode(struct inode *inode, unsigned long flags); @@ -450,7 +450,7 @@ extern int nfs_clear_invalid_mapping(struct address_space *mapping); extern bool nfs_mapping_need_revalidate_inode(struct inode *inode); extern int nfs_revalidate_mapping(struct inode *inode, struct address_space *mapping); extern int nfs_revalidate_mapping_rcu(struct inode *inode); -extern int nfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int nfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nfs_setattr_update_inode(struct inode *inode, struct iattr *attr, struct nfs_fattr *); extern void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr); extern struct nfs_open_context *get_nfs_open_context(struct nfs_open_context *ctx); diff --git a/include/linux/posix_acl.h b/include/linux/posix_acl.h index 62d497763e25..caf500bed993 100644 --- a/include/linux/posix_acl.h +++ b/include/linux/posix_acl.h @@ -74,20 +74,20 @@ extern int __posix_acl_create(struct posix_acl **, gfp_t, umode_t *); extern int __posix_acl_chmod(struct posix_acl **, gfp_t, umode_t); extern struct posix_acl *get_posix_acl(struct inode *, int); -int set_posix_acl(struct mnt_idmap *, struct dentry *, int, +int set_posix_acl(const struct mnt_idmap *, struct dentry *, int, struct posix_acl *); struct posix_acl *get_cached_acl_rcu(struct inode *inode, int type); struct posix_acl *posix_acl_clone(const struct posix_acl *acl, gfp_t flags); #ifdef CONFIG_FS_POSIX_ACL -int posix_acl_chmod(struct mnt_idmap *, struct dentry *, umode_t); +int posix_acl_chmod(const struct mnt_idmap *, struct dentry *, umode_t); extern int posix_acl_create(struct inode *, umode_t *, struct posix_acl **, struct posix_acl **); -int posix_acl_update_mode(struct mnt_idmap *, struct inode *, umode_t *, +int posix_acl_update_mode(const struct mnt_idmap *, struct inode *, umode_t *, struct posix_acl **); -int simple_set_acl(struct mnt_idmap *, struct dentry *, +int simple_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); extern int simple_acl_create(struct inode *, struct inode *); @@ -96,7 +96,7 @@ void set_cached_acl(struct inode *inode, int type, struct posix_acl *acl); void forget_cached_acl(struct inode *inode, int type); void forget_all_cached_acls(struct inode *inode); int posix_acl_valid(struct user_namespace *, const struct posix_acl *); -int posix_acl_permission(struct mnt_idmap *, struct inode *, +int posix_acl_permission(const struct mnt_idmap *, struct inode *, const struct posix_acl *, int); static inline void cache_no_acl(struct inode *inode) @@ -105,16 +105,16 @@ static inline void cache_no_acl(struct inode *inode) inode->i_default_acl = NULL; } -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); int posix_acl_listxattr(struct inode *inode, char **buffer, ssize_t *remaining_size); #else -static inline int posix_acl_chmod(struct mnt_idmap *idmap, +static inline int posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { return 0; @@ -141,21 +141,21 @@ static inline void forget_all_cached_acls(struct inode *inode) { } -static inline int vfs_set_acl(struct mnt_idmap *idmap, +static inline int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *acl) { return -EOPNOTSUPP; } -static inline struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return ERR_PTR(-EOPNOTSUPP); } -static inline int vfs_remove_acl(struct mnt_idmap *idmap, +static inline int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return -EOPNOTSUPP; diff --git a/include/linux/quotaops.h b/include/linux/quotaops.h index f9c0f9d7c9d9..0c64ca674e77 100644 --- a/include/linux/quotaops.h +++ b/include/linux/quotaops.h @@ -20,7 +20,7 @@ static inline struct quota_info *sb_dqopt(struct super_block *sb) } /* i_rwsem must being held */ -static inline bool is_quota_modification(struct mnt_idmap *idmap, +static inline bool is_quota_modification(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *ia) { return ((ia->ia_valid & ATTR_SIZE) || @@ -109,7 +109,7 @@ int dquot_set_dqblk(struct super_block *sb, struct kqid id, struct qc_dqblk *di); int __dquot_transfer(struct inode *inode, struct dquot **transfer_to); -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr); static inline struct mem_dqinfo *sb_dqinfo(struct super_block *sb, int type) @@ -229,7 +229,7 @@ static inline void dquot_free_inode(struct inode *inode) { } -static inline int dquot_transfer(struct mnt_idmap *idmap, +static inline int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { return 0; diff --git a/include/linux/sched.h b/include/linux/sched.h index d35ae49a991f..78bfc0e32e56 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1817,7 +1817,7 @@ extern struct pid __rcu *cad_pid; * I am cleaning dirty pages from some other bdi. */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_RANDOMIZE 0x00400000 /* Randomize virtual address space */ -#define PF__HOLE__00800000 0x00800000 +#define PF_NO_NOTIFY_SIGNAL 0x00800000 /* see no_notify_signal_save() */ #define PF__HOLE__01000000 0x01000000 #define PF__HOLE__02000000 0x02000000 #define PF_NO_SETAFFINITY 0x04000000 /* Userland is not allowed to meddle with cpus_mask */ diff --git a/include/linux/sched/signal.h b/include/linux/sched/signal.h index d45a5476b97d..d9419dc902f6 100644 --- a/include/linux/sched/signal.h +++ b/include/linux/sched/signal.h @@ -2,6 +2,7 @@ #ifndef _LINUX_SCHED_SIGNAL_H #define _LINUX_SCHED_SIGNAL_H +#include <linux/cleanup.h> #include <linux/rculist.h> #include <linux/signal.h> #include <linux/sched.h> @@ -79,9 +80,9 @@ struct core_thread { }; struct core_state { - atomic_t nr_threads; - struct core_thread dumper; - struct completion startup; + /* Threads the dumper still waits for. */ + atomic_t threads_remaining; + struct core_thread *tasks; }; /* @@ -384,14 +385,36 @@ static inline int task_sigpending(struct task_struct *p) return unlikely(test_tsk_thread_flag(p,TIF_SIGPENDING)); } +/* Prevent TIF_NOTIFY_SIGNAL from interrupting this task. */ +static inline unsigned int no_notify_signal_save(void) +{ + unsigned int flags = current->flags; + + current->flags |= PF_NO_NOTIFY_SIGNAL; + return flags; +} + +/* Restore the previous PF_NO_NOTIFY_SIGNAL state. */ +static inline void no_notify_signal_restore(unsigned int flags) +{ + current_restore_flags(flags, PF_NO_NOTIFY_SIGNAL); +} + +DEFINE_LOCK_GUARD_0(no_notify_signal, + _T->flags = no_notify_signal_save(), + no_notify_signal_restore(_T->flags), + unsigned int flags) + static inline int signal_pending(struct task_struct *p) { /* * TIF_NOTIFY_SIGNAL isn't really a signal, but it requires the same * behavior in terms of ensuring that we break out of wait loops - * so that notify signal callbacks can be processed. + * so that notify signal callbacks can be processed. Not for a task + * that asked not to be interrupted by it, see no_notify_signal_save(). */ - if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL))) + if (unlikely(test_tsk_thread_flag(p, TIF_NOTIFY_SIGNAL)) && + likely(!(READ_ONCE(p->flags) & PF_NO_NOTIFY_SIGNAL))) return 1; return task_sigpending(p); } diff --git a/include/linux/security.h b/include/linux/security.h index 153e9043058f..f7ff72ff956b 100644 --- a/include/linux/security.h +++ b/include/linux/security.h @@ -185,11 +185,11 @@ extern int cap_capset(struct cred *new, const struct cred *old, extern int cap_bprm_creds_from_file(struct linux_binprm *bprm, const struct file *file); int cap_inode_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int cap_inode_removexattr(struct mnt_idmap *idmap, +int cap_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); int cap_inode_need_killpriv(struct dentry *dentry); -int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int cap_inode_getsecurity(struct mnt_idmap *idmap, +int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int cap_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); extern int cap_mmap_addr(unsigned long addr); @@ -338,6 +338,7 @@ int security_binder_transfer_file(const struct cred *from, const struct cred *to, const struct file *file); int security_ptrace_access_check(struct task_struct *child, unsigned int mode); int security_ptrace_traceme(struct task_struct *parent); +int security_mem_foll_force(const struct cred *subject, bool opened_by_owner); int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -405,7 +406,7 @@ int security_inode_init_security_anon(struct inode *inode, const struct qstr *name, const struct inode *context_inode); int security_inode_create(struct inode *dir, struct dentry *dentry, umode_t mode); -void security_inode_post_create_tmpfile(struct mnt_idmap *idmap, +void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode); int security_inode_link(struct dentry *old_dentry, struct inode *dir, struct dentry *new_dentry); @@ -422,31 +423,31 @@ int security_inode_readlink(struct dentry *dentry); int security_inode_follow_link(struct dentry *dentry, struct inode *inode, bool rcu); int security_inode_permission(struct inode *inode, int mask); -int security_inode_setattr(struct mnt_idmap *idmap, +int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid); int security_inode_getattr(const struct path *path); -int security_inode_setxattr(struct mnt_idmap *idmap, +int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags); -int security_inode_set_acl(struct mnt_idmap *idmap, +int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name, struct posix_acl *kacl); -int security_inode_get_acl(struct mnt_idmap *idmap, +int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -int security_inode_remove_acl(struct mnt_idmap *idmap, +int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); -void security_inode_post_remove_acl(struct mnt_idmap *idmap, +void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name); void security_inode_post_setxattr(struct dentry *dentry, const char *name, const void *value, size_t size, int flags); int security_inode_getxattr(struct dentry *dentry, const char *name); int security_inode_listxattr(struct dentry *dentry); -int security_inode_removexattr(struct mnt_idmap *idmap, +int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name); void security_inode_post_removexattr(struct dentry *dentry, const char *name); int security_inode_file_setattr(struct dentry *dentry, @@ -454,8 +455,8 @@ int security_inode_file_setattr(struct dentry *dentry, int security_inode_file_getattr(struct dentry *dentry, struct file_kattr *fa); int security_inode_need_killpriv(struct dentry *dentry); -int security_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry); -int security_inode_getsecurity(struct mnt_idmap *idmap, +int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry); +int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc); int security_inode_setsecurity(struct inode *inode, const char *name, const void *value, size_t size, int flags); @@ -676,6 +677,12 @@ static inline int security_ptrace_traceme(struct task_struct *parent) return cap_ptrace_traceme(parent); } +static inline int security_mem_foll_force(const struct cred *subject, + bool opened_by_owner) +{ + return 0; +} + static inline int security_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, @@ -910,7 +917,7 @@ static inline int security_inode_create(struct inode *dir, } static inline void -security_inode_post_create_tmpfile(struct mnt_idmap *idmap, struct inode *inode) +security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { } static inline int security_inode_link(struct dentry *old_dentry, @@ -979,7 +986,7 @@ static inline int security_inode_permission(struct inode *inode, int mask) return 0; } -static inline int security_inode_setattr(struct mnt_idmap *idmap, +static inline int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { @@ -987,7 +994,7 @@ static inline int security_inode_setattr(struct mnt_idmap *idmap, } static inline void -security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { } @@ -996,14 +1003,14 @@ static inline int security_inode_getattr(const struct path *path) return 0; } -static inline int security_inode_setxattr(struct mnt_idmap *idmap, +static inline int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { return cap_inode_setxattr(dentry, name, value, size, flags); } -static inline int security_inode_set_acl(struct mnt_idmap *idmap, +static inline int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) @@ -1016,21 +1023,21 @@ static inline void security_inode_post_set_acl(struct dentry *dentry, struct posix_acl *kacl) { } -static inline int security_inode_get_acl(struct mnt_idmap *idmap, +static inline int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline int security_inode_remove_acl(struct mnt_idmap *idmap, +static inline int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return 0; } -static inline void security_inode_post_remove_acl(struct mnt_idmap *idmap, +static inline void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { } @@ -1050,7 +1057,7 @@ static inline int security_inode_listxattr(struct dentry *dentry) return 0; } -static inline int security_inode_removexattr(struct mnt_idmap *idmap, +static inline int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { @@ -1078,13 +1085,13 @@ static inline int security_inode_need_killpriv(struct dentry *dentry) return cap_inode_need_killpriv(dentry); } -static inline int security_inode_killpriv(struct mnt_idmap *idmap, +static inline int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { return cap_inode_killpriv(idmap, dentry); } -static inline int security_inode_getsecurity(struct mnt_idmap *idmap, +static inline int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) @@ -2085,7 +2092,7 @@ int security_path_mkdir(const struct path *dir, struct dentry *dentry, umode_t m int security_path_rmdir(const struct path *dir, struct dentry *dentry); int security_path_mknod(const struct path *dir, struct dentry *dentry, umode_t mode, unsigned int dev); -void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry); +void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry); int security_path_truncate(const struct path *path); int security_path_symlink(const struct path *dir, struct dentry *dentry, const char *old_name); @@ -2120,7 +2127,7 @@ static inline int security_path_mknod(const struct path *dir, struct dentry *den return 0; } -static inline void security_path_post_mknod(struct mnt_idmap *idmap, +static inline void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { } diff --git a/include/linux/splice.h b/include/linux/splice.h index 9dec4861d09f..0e6c955dc6ff 100644 --- a/include/linux/splice.h +++ b/include/linux/splice.h @@ -79,8 +79,8 @@ ssize_t add_to_pipe(struct pipe_inode_info *pipe, struct pipe_buffer *buf); ssize_t vfs_splice_read(struct file *in, loff_t *ppos, struct pipe_inode_info *pipe, size_t len, unsigned int flags); -ssize_t splice_direct_to_actor(struct file *file, struct splice_desc *sd, - splice_direct_actor *actor); +ssize_t vfs_splice_to_actor(struct file *file, loff_t pos, size_t count, + splice_direct_actor *actor, void *private); ssize_t do_splice(struct file *in, loff_t *off_in, struct file *out, loff_t *off_out, size_t len, unsigned int flags); ssize_t do_splice_direct(struct file *in, loff_t *ppos, struct file *out, diff --git a/include/linux/uidgid.h b/include/linux/uidgid.h index 2dc767e08f54..02403629b49f 100644 --- a/include/linux/uidgid.h +++ b/include/linux/uidgid.h @@ -130,9 +130,9 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return from_kgid(ns, gid) != (gid_t) -1; } -u32 map_id_down(struct uid_gid_map *map, u32 id); -u32 map_id_up(struct uid_gid_map *map, u32 id); -u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count); +u32 map_id_down(const struct uid_gid_map *map, u32 id); +u32 map_id_up(const struct uid_gid_map *map, u32 id); +u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count); #else @@ -182,17 +182,17 @@ static inline bool kgid_has_mapping(struct user_namespace *ns, kgid_t gid) return gid_valid(gid); } -static inline u32 map_id_down(struct uid_gid_map *map, u32 id) +static inline u32 map_id_down(const struct uid_gid_map *map, u32 id) { return id; } -static inline u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) +static inline u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count) { return id; } -static inline u32 map_id_up(struct uid_gid_map *map, u32 id) +static inline u32 map_id_up(const struct uid_gid_map *map, u32 id) { return id; } diff --git a/include/linux/user_namespace.h b/include/linux/user_namespace.h index e38d9e60569f..91232053775d 100644 --- a/include/linux/user_namespace.h +++ b/include/linux/user_namespace.h @@ -29,8 +29,8 @@ struct uid_gid_map { /* 64 bytes -- 1 cache line */ u32 nr_extents; }; struct { - struct uid_gid_extent *forward; - struct uid_gid_extent *reverse; + struct uid_gid_extent *forward __counted_by_ptr(nr_extents); + struct uid_gid_extent *reverse __counted_by_ptr(nr_extents); }; }; }; @@ -207,6 +207,13 @@ extern bool in_userns(const struct user_namespace *ancestor, const struct user_namespace *child); extern bool current_in_userns(const struct user_namespace *target_ns); struct ns_common *ns_get_owner(struct ns_common *ns); + +#if IS_ENABLED(CONFIG_KUNIT) +extern int uid_gid_map_insert_extent(struct uid_gid_map *map, + struct uid_gid_extent *extent); +extern int uid_gid_map_sort(struct uid_gid_map *map); +#endif /* CONFIG_KUNIT */ + #else static inline struct user_namespace *get_user_ns(struct user_namespace *ns) diff --git a/include/linux/wait_bit.h b/include/linux/wait_bit.h index 553d7b23e3ad..af077ed4caf6 100644 --- a/include/linux/wait_bit.h +++ b/include/linux/wait_bit.h @@ -433,6 +433,32 @@ do { \ }) /** + * wait_var_event_state - wait for a variable to be updated and notified + * @var: the address of variable being waited on + * @condition: the condition to wait for + * @state: the task state to sleep in, %TASK_UNINTERRUPTIBLE etc. + * + * Wait for a @condition to be true, only re-checking when a wake up is + * received for the given @var (an arbitrary kernel address which need + * not be directly related to the given condition, but usually is). + * + * Returns 0 if the condition became true, or %-ERESTARTSYS if a signal + * arrived which @state allows to interrupt. + * + * The condition should normally use smp_load_acquire() or a similarly + * ordered access to ensure that any changes to memory made before the + * condition became true will be visible after the wait completes. + */ +#define wait_var_event_state(var, condition, state) \ +({ \ + int __ret = 0; \ + might_sleep(); \ + if (!(condition)) \ + __ret = ___wait_var_event(var, condition, (state), 0, 0, schedule()); \ + __ret; \ +}) + +/** * wait_var_event_any_lock - wait for a variable to be updated under a lock * @var: the address of the variable being waited on * @condition: condition to wait for diff --git a/include/linux/xattr.h b/include/linux/xattr.h index 54ac3cbc133f..4cc4257de084 100644 --- a/include/linux/xattr.h +++ b/include/linux/xattr.h @@ -47,7 +47,7 @@ struct xattr_handler { struct inode *inode, const char *name, void *buffer, size_t size); int (*set)(const struct xattr_handler *, - struct mnt_idmap *idmap, struct dentry *dentry, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags); }; @@ -77,25 +77,25 @@ struct xattr { }; ssize_t __vfs_getxattr(struct dentry *, struct inode *, const char *, void *, size_t); -ssize_t vfs_getxattr(struct mnt_idmap *, struct dentry *, const char *, +ssize_t vfs_getxattr(const struct mnt_idmap *, struct dentry *, const char *, void *, size_t); ssize_t vfs_listxattr(struct dentry *d, char *list, size_t size); -int __vfs_setxattr(struct mnt_idmap *, struct dentry *, struct inode *, +int __vfs_setxattr(const struct mnt_idmap *, struct dentry *, struct inode *, const char *, const void *, size_t, int); -int __vfs_setxattr_noperm(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_noperm(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_setxattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_setxattr_locked(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int, struct delegated_inode *); -int vfs_setxattr(struct mnt_idmap *, struct dentry *, const char *, +int vfs_setxattr(const struct mnt_idmap *, struct dentry *, const char *, const void *, size_t, int); -int __vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); -int __vfs_removexattr_locked(struct mnt_idmap *, struct dentry *, +int __vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); +int __vfs_removexattr_locked(const struct mnt_idmap *, struct dentry *, const char *, struct delegated_inode *); -int vfs_removexattr(struct mnt_idmap *, struct dentry *, const char *); +int vfs_removexattr(const struct mnt_idmap *, struct dentry *, const char *); ssize_t generic_listxattr(struct dentry *dentry, char *buffer, size_t buffer_size); -int vfs_getxattr_alloc(struct mnt_idmap *idmap, +int vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t size, gfp_t flags); diff --git a/include/trace/events/cachefiles.h b/include/trace/events/cachefiles.h index e3101410e8b2..a19233e8ae78 100644 --- a/include/trace/events/cachefiles.h +++ b/include/trace/events/cachefiles.h @@ -52,6 +52,8 @@ enum cachefiles_coherency_trace { cachefiles_coherency_check_ok, cachefiles_coherency_check_type, cachefiles_coherency_check_xattr, + cachefiles_coherency_discontiguous, + cachefiles_coherency_remove, cachefiles_coherency_set_fail, cachefiles_coherency_set_ok, cachefiles_coherency_vol_check_cmp, @@ -63,9 +65,11 @@ enum cachefiles_coherency_trace { }; enum cachefiles_trunc_trace { + cachefiles_trunc_clear_padding, cachefiles_trunc_dio_adjust, cachefiles_trunc_expand_tmpfile, cachefiles_trunc_shrink, + cachefiles_trunc_zap, }; enum cachefiles_prepare_read_trace { @@ -80,11 +84,14 @@ enum cachefiles_prepare_read_trace { }; enum cachefiles_error_trace { + cachefiles_trace_alignment_error, + cachefiles_trace_create_nospace, cachefiles_trace_fallocate_error, cachefiles_trace_getxattr_error, cachefiles_trace_link_error, cachefiles_trace_lookup_error, cachefiles_trace_mkdir_error, + cachefiles_trace_mkdir_nospace, cachefiles_trace_notify_change_error, cachefiles_trace_open_error, cachefiles_trace_read_error, @@ -97,6 +104,8 @@ enum cachefiles_error_trace { cachefiles_trace_trunc_error, cachefiles_trace_unlink_error, cachefiles_trace_write_error, + cachefiles_trace_write_nospace, + cachefiles_trace_write_nospace_2, }; #endif @@ -136,6 +145,8 @@ enum cachefiles_error_trace { EM(cachefiles_coherency_check_ok, "OK ") \ EM(cachefiles_coherency_check_type, "BAD type") \ EM(cachefiles_coherency_check_xattr, "BAD xatt") \ + EM(cachefiles_coherency_discontiguous, "--- gap ") \ + EM(cachefiles_coherency_remove, "REMOVE ") \ EM(cachefiles_coherency_set_fail, "SET fail") \ EM(cachefiles_coherency_set_ok, "SET ok ") \ EM(cachefiles_coherency_vol_check_cmp, "VOL BAD cmp ") \ @@ -146,9 +157,11 @@ enum cachefiles_error_trace { E_(cachefiles_coherency_vol_set_ok, "VOL SET ok ") #define cachefiles_trunc_traces \ + EM(cachefiles_trunc_clear_padding, "CLRPAD") \ EM(cachefiles_trunc_dio_adjust, "DIOADJ") \ EM(cachefiles_trunc_expand_tmpfile, "EXPTMP") \ - E_(cachefiles_trunc_shrink, "SHRINK") + EM(cachefiles_trunc_shrink, "SHRINK") \ + E_(cachefiles_trunc_zap, "ZAP ") #define cachefiles_prepare_read_traces \ EM(cachefiles_trace_read_after_eof, "after-eof ") \ @@ -161,11 +174,14 @@ enum cachefiles_error_trace { E_(cachefiles_trace_read_seek_nxio, "seek-enxio") #define cachefiles_error_traces \ + EM(cachefiles_trace_alignment_error, "align") \ + EM(cachefiles_trace_create_nospace, "create-nospace") \ EM(cachefiles_trace_fallocate_error, "fallocate") \ EM(cachefiles_trace_getxattr_error, "getxattr") \ EM(cachefiles_trace_link_error, "link") \ EM(cachefiles_trace_lookup_error, "lookup") \ EM(cachefiles_trace_mkdir_error, "mkdir") \ + EM(cachefiles_trace_mkdir_nospace, "mkdir-nospace") \ EM(cachefiles_trace_notify_change_error, "notify_change") \ EM(cachefiles_trace_open_error, "open") \ EM(cachefiles_trace_read_error, "read") \ @@ -177,7 +193,9 @@ enum cachefiles_error_trace { EM(cachefiles_trace_tmpfile_error, "tmpfile") \ EM(cachefiles_trace_trunc_error, "trunc") \ EM(cachefiles_trace_unlink_error, "unlink") \ - E_(cachefiles_trace_write_error, "write") + EM(cachefiles_trace_write_error, "write") \ + EM(cachefiles_trace_write_nospace, "write-nospace") \ + E_(cachefiles_trace_write_nospace_2, "write-nospace-2") /* @@ -371,12 +389,12 @@ TRACE_EVENT(cachefiles_rename, TRACE_EVENT(cachefiles_coherency, TP_PROTO(struct cachefiles_object *obj, - ino_t ino, + ino_t ino, uoff_t obj_size, const void *disk_aux, enum cachefiles_content content, enum cachefiles_coherency_trace why), - TP_ARGS(obj, ino, disk_aux, content, why), + TP_ARGS(obj, ino, obj_size, disk_aux, content, why), /* Note that obj may be NULL */ TP_STRUCT__entry( @@ -384,6 +402,7 @@ TRACE_EVENT(cachefiles_coherency, __field(enum cachefiles_coherency_trace, why) __field(enum cachefiles_content, content) __field(u64, ino) + __field(u64, obj_size) __field(u64, aux) __field(u64, disk_aux) ), @@ -398,6 +417,7 @@ TRACE_EVENT(cachefiles_coherency, __entry->why = why; __entry->content = content; __entry->ino = ino; + __entry->obj_size = obj_size; __entry->aux = be64_to_cpup((__be64 *)obj->cookie->inline_aux); /* cachefiles_xattr::data is 2-byte aligned but not 8-byte aligned. */ @@ -412,10 +432,11 @@ TRACE_EVENT(cachefiles_coherency, } ), - TP_printk("o=%08x %s B=%llx c=%u aux=%llx dsk=%llx", + TP_printk("o=%08x %s B=%llx oz=%llx c=%u aux=%llx dsk=%llx", __entry->obj, __print_symbolic(__entry->why, cachefiles_coherency_traces), __entry->ino, + __entry->obj_size, __entry->content, __entry->aux, __entry->disk_aux) @@ -449,7 +470,7 @@ TRACE_EVENT(cachefiles_vol_coherency, TRACE_EVENT(cachefiles_prep_read, TP_PROTO(struct cachefiles_object *obj, - loff_t start, + uoff_t start, size_t len, unsigned short flags, enum netfs_io_source source, @@ -464,7 +485,7 @@ TRACE_EVENT(cachefiles_prep_read, __field(enum netfs_io_source, source) __field(enum cachefiles_prepare_read_trace, why) __field(size_t, len) - __field(loff_t, start) + __field(uoff_t, start) __field(unsigned int, netfs_inode) __field(unsigned int, cache_inode) ), @@ -492,16 +513,16 @@ TRACE_EVENT(cachefiles_prep_read, TRACE_EVENT(cachefiles_read, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -521,16 +542,16 @@ TRACE_EVENT(cachefiles_read, TRACE_EVENT(cachefiles_write, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t start, + uoff_t start, size_t len), TP_ARGS(obj, backer, start, len), TP_STRUCT__entry( - __field(unsigned int, obj) - __field(unsigned int, backer) - __field(size_t, len) - __field(loff_t, start) + __field(unsigned int, obj) + __field(unsigned int, backer) + __field(size_t, len) + __field(uoff_t, start) ), TP_fast_assign( @@ -549,7 +570,7 @@ TRACE_EVENT(cachefiles_write, TRACE_EVENT(cachefiles_trunc, TP_PROTO(struct cachefiles_object *obj, struct inode *backer, - loff_t from, loff_t to, enum cachefiles_trunc_trace why), + uoff_t from, uoff_t to, enum cachefiles_trunc_trace why), TP_ARGS(obj, backer, from, to, why), @@ -557,8 +578,8 @@ TRACE_EVENT(cachefiles_trunc, __field(unsigned int, obj) __field(unsigned int, backer) __field(enum cachefiles_trunc_trace, why) - __field(loff_t, from) - __field(loff_t, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( @@ -694,6 +715,26 @@ TRACE_EVENT(cachefiles_io_error, __entry->error) ); +TRACE_EVENT(cachefiles_no_space, + TP_PROTO(struct cachefiles_object *obj, enum cachefiles_error_trace trace), + + TP_ARGS(obj, trace), + + TP_STRUCT__entry( + __field(unsigned int, obj) + __field(enum cachefiles_error_trace, trace) + ), + + TP_fast_assign( + __entry->obj = obj ? obj->debug_id : 0; + __entry->trace = trace; + ), + + TP_printk("o=%08x %s", + __entry->obj, + __print_symbolic(__entry->trace, cachefiles_error_traces)) + ); + #endif /* _TRACE_CACHEFILES_H */ /* This part must be outside protection */ diff --git a/include/trace/events/fscache.h b/include/trace/events/fscache.h index f1a73aa83fbb..8735d428ebd9 100644 --- a/include/trace/events/fscache.h +++ b/include/trace/events/fscache.h @@ -460,13 +460,13 @@ TRACE_EVENT(fscache_relinquish, ); TRACE_EVENT(fscache_invalidate, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, new_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( @@ -479,14 +479,14 @@ TRACE_EVENT(fscache_invalidate, ); TRACE_EVENT(fscache_resize, - TP_PROTO(struct fscache_cookie *cookie, loff_t new_size), + TP_PROTO(struct fscache_cookie *cookie, uoff_t new_size), TP_ARGS(cookie, new_size), TP_STRUCT__entry( __field(unsigned int, cookie ) - __field(loff_t, old_size ) - __field(loff_t, new_size ) + __field(uoff_t, old_size ) + __field(uoff_t, new_size ) ), TP_fast_assign( diff --git a/include/trace/events/netfs.h b/include/trace/events/netfs.h index 3fec3e8f91c8..bf1e1f185b05 100644 --- a/include/trace/events/netfs.h +++ b/include/trace/events/netfs.h @@ -30,8 +30,7 @@ EM(netfs_write_trace_dio_write, "DIO-WRITE") \ EM(netfs_write_trace_unbuffered_write, "UNB-WRITE") \ EM(netfs_write_trace_writeback, "WRITEBACK") \ - EM(netfs_write_trace_writeback_single, "WB-SINGLE") \ - E_(netfs_write_trace_writethrough, "WRITETHRU") + E_(netfs_write_trace_writeback_single, "WB-SINGLE") #define netfs_rreq_origins \ EM(NETFS_READAHEAD, "RA") \ @@ -43,13 +42,17 @@ EM(NETFS_DIO_READ, "DR") \ EM(NETFS_WRITEBACK, "WB") \ EM(NETFS_WRITEBACK_SINGLE, "W1") \ - EM(NETFS_WRITETHROUGH, "WT") \ EM(NETFS_UNBUFFERED_WRITE, "UW") \ EM(NETFS_DIO_WRITE, "DW") \ E_(NETFS_PGPRIV2_COPY_TO_CACHE, "2C") #define netfs_rreq_traces \ + EM(netfs_rreq_trace_all_queued, "ALL-Q ") \ EM(netfs_rreq_trace_assess, "ASSESS ") \ + EM(netfs_rreq_trace_cache_cancelled, "CA-CNCL") \ + EM(netfs_rreq_trace_cache_failed, "CA-FAIL") \ + EM(netfs_rreq_trace_cache_fail_collect, "CA-F-CO") \ + EM(netfs_rreq_trace_cache_no_space, "CA-NOSP") \ EM(netfs_rreq_trace_collect, "COLLECT") \ EM(netfs_rreq_trace_complete, "COMPLET") \ EM(netfs_rreq_trace_copy, "COPY ") \ @@ -58,11 +61,14 @@ EM(netfs_rreq_trace_end_copy_to_cache, "END-C2C") \ EM(netfs_rreq_trace_free, "FREE ") \ EM(netfs_rreq_trace_intr, "INTR ") \ + EM(netfs_rreq_trace_inval_cache, "INVL-CA") \ EM(netfs_rreq_trace_ki_complete, "KI-CMPL") \ EM(netfs_rreq_trace_ra_put_ref, "RA-PUT ") \ EM(netfs_rreq_trace_recollect, "RECLLCT") \ EM(netfs_rreq_trace_redirty, "REDIRTY") \ EM(netfs_rreq_trace_resubmit, "RESUBMT") \ + EM(netfs_rreq_trace_retry_begin, "RETRY-BEGIN") \ + EM(netfs_rreq_trace_retry_end, "RETRY-END") \ EM(netfs_rreq_trace_set_abandon, "S-ABNDN") \ EM(netfs_rreq_trace_set_pause, "PAUSE ") \ EM(netfs_rreq_trace_unlock, "UNLOCK ") \ @@ -94,8 +100,10 @@ EM(netfs_sreq_trace_abandoned, "ABNDN") \ EM(netfs_sreq_trace_add_donations, "+DON ") \ EM(netfs_sreq_trace_added, "ADD ") \ + EM(netfs_sreq_trace_cache_nofile, "CA-!F") \ EM(netfs_sreq_trace_cache_nowrite, "CA-NW") \ EM(netfs_sreq_trace_cache_prepare, "CA-PR") \ + EM(netfs_sreq_trace_cache_waitfail, "CA-!W") \ EM(netfs_sreq_trace_cache_write, "CA-WR") \ EM(netfs_sreq_trace_cancel, "CANCL") \ EM(netfs_sreq_trace_clear, "CLEAR") \ @@ -134,12 +142,12 @@ #define netfs_failures \ EM(netfs_fail_check_write_begin, "check-write-begin") \ - EM(netfs_fail_copy_to_cache, "copy-to-cache") \ EM(netfs_fail_dio_read_short, "dio-read-short") \ EM(netfs_fail_dio_read_zero, "dio-read-zero") \ EM(netfs_fail_read, "read") \ EM(netfs_fail_short_read, "short-read") \ EM(netfs_fail_prepare_write, "prep-write") \ + EM(netfs_fail_upload, "upload") \ E_(netfs_fail_write, "write") #define netfs_rreq_ref_traces \ @@ -194,11 +202,11 @@ EM(netfs_folio_trace_alloc_buffer, "alloc-buf") \ EM(netfs_folio_trace_cancel_copy, "cancel-copy") \ EM(netfs_folio_trace_cancel_store, "cancel-store") \ - EM(netfs_folio_trace_clear, "clear") \ - EM(netfs_folio_trace_clear_cc, "clear-cc") \ - EM(netfs_folio_trace_clear_g, "clear-g") \ - EM(netfs_folio_trace_clear_s, "clear-s") \ EM(netfs_folio_trace_end_copy, "end-copy") \ + EM(netfs_folio_trace_endwb, "endwb") \ + EM(netfs_folio_trace_endwb_cc, "endwb-cc") \ + EM(netfs_folio_trace_endwb_g, "endwb-g") \ + EM(netfs_folio_trace_endwb_s, "endwb-s") \ EM(netfs_folio_trace_filled_gaps, "filled-gaps") \ EM(netfs_folio_trace_invalidate_all, "inval-all") \ EM(netfs_folio_trace_invalidate_front, "inval-front") \ @@ -223,9 +231,7 @@ EM(netfs_folio_trace_sched_copy, "sched-copy") \ EM(netfs_folio_trace_store, "store") \ EM(netfs_folio_trace_store_copy, "store-copy") \ - EM(netfs_folio_trace_store_plus, "store+") \ - EM(netfs_folio_trace_wthru, "wthru") \ - E_(netfs_folio_trace_wthru_plus, "wthru+") + E_(netfs_folio_trace_store_plus, "store+") #define netfs_collect_contig_traces \ EM(netfs_contig_trace_collect, "Collect") \ @@ -301,7 +307,7 @@ netfs_folioq_traces; TRACE_EVENT(netfs_read, TP_PROTO(struct netfs_io_request *rreq, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_read_trace what), TP_ARGS(rreq, start, len, what), @@ -309,8 +315,9 @@ TRACE_EVENT(netfs_read, TP_STRUCT__entry( __field(unsigned int, rreq) __field(unsigned int, cookie) - __field(loff_t, i_size) - __field(loff_t, start) + __field(unsigned int, object) + __field(uoff_t, i_size) + __field(uoff_t, start) __field(size_t, len) __field(enum netfs_read_trace, what) __field(u64, netfs_inode) @@ -318,7 +325,8 @@ TRACE_EVENT(netfs_read, TP_fast_assign( __entry->rreq = rreq->debug_id; - __entry->cookie = rreq->cache_resources.debug_id; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->i_size = rreq->i_size; __entry->start = start; __entry->len = len; @@ -326,10 +334,10 @@ TRACE_EVENT(netfs_read, __entry->netfs_inode = rreq->inode->i_ino; ), - TP_printk("R=%08x %s c=%08x ni=%llx s=%llx l=%zx sz=%llx", + TP_printk("R=%08x %s c=%08x o=%08x ni=%llx s=%llx l=%zx sz=%llx", __entry->rreq, __print_symbolic(__entry->what, netfs_read_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->netfs_inode, __entry->start, __entry->len, __entry->i_size) ); @@ -377,7 +385,7 @@ TRACE_EVENT(netfs_sreq, __field(u8, slot) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -418,7 +426,7 @@ TRACE_EVENT(netfs_failure, __field(enum netfs_failure, what) __field(size_t, len) __field(size_t, transferred) - __field(loff_t, start) + __field(uoff_t, start) ), TP_fast_assign( @@ -501,6 +509,7 @@ TRACE_EVENT(netfs_folio, TP_STRUCT__entry( __field(u64, ino) __field(pgoff_t, index) + __field(unsigned long, pfn) __field(unsigned int, nr) __field(enum netfs_folio_trace, why) ), @@ -511,9 +520,11 @@ TRACE_EVENT(netfs_folio, __entry->why = why; __entry->index = folio->index; __entry->nr = folio_nr_pages(folio); + __entry->pfn = folio_pfn(folio); ), - TP_printk("i=%05llx ix=%05lx-%05lx %s", + TP_printk("p=%lx i=%05llx ix=%05lx-%05lx %s", + __entry->pfn, __entry->ino, __entry->index, __entry->index + __entry->nr - 1, __print_symbolic(__entry->why, netfs_folio_traces)) ); @@ -524,10 +535,10 @@ TRACE_EVENT(netfs_write_iter, TP_ARGS(iocb, from), TP_STRUCT__entry( - __field(unsigned long long, start) - __field(size_t, len) - __field(unsigned int, flags) - __field(unsigned int, ino) + __field(uoff_t, start) + __field(size_t, len) + __field(unsigned int, flags) + __field(unsigned int, ino) ), TP_fast_assign( @@ -550,27 +561,27 @@ TRACE_EVENT(netfs_write, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, cookie) + __field(unsigned int, object) __field(unsigned int, ino) __field(enum netfs_write_trace, what) - __field(unsigned long long, start) - __field(unsigned long long, len) + __field(uoff_t, start) + __field(uoff_t, len) ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(wreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->wreq = wreq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = wreq->cache_resources.cookie_id; + __entry->object = wreq->cache_resources.object_id; __entry->ino = wreq->inode->i_ino; __entry->what = what; __entry->start = wreq->start; __entry->len = wreq->len; ), - TP_printk("R=%08x %s c=%08x i=%x by=%llx-%llx", + TP_printk("R=%08x %s c=%08x o=%08x i=%x by=%llx-%llx", __entry->wreq, __print_symbolic(__entry->what, netfs_write_traces), - __entry->cookie, + __entry->cookie, __entry->object, __entry->ino, __entry->start, __entry->start + __entry->len - 1) ); @@ -582,25 +593,26 @@ TRACE_EVENT(netfs_copy2cache, TP_ARGS(rreq, creq), TP_STRUCT__entry( - __field(unsigned int, rreq) - __field(unsigned int, creq) - __field(unsigned int, cookie) - __field(unsigned int, ino) + __field(unsigned int, rreq) + __field(unsigned int, creq) + __field(unsigned int, cookie) + __field(unsigned int, object) + __field(unsigned int, ino) ), TP_fast_assign( - struct netfs_inode *__ctx = netfs_inode(rreq->inode); - struct fscache_cookie *__cookie = netfs_i_cookie(__ctx); __entry->rreq = rreq->debug_id; __entry->creq = creq->debug_id; - __entry->cookie = __cookie ? __cookie->debug_id : 0; + __entry->cookie = rreq->cache_resources.cookie_id; + __entry->object = rreq->cache_resources.object_id; __entry->ino = rreq->inode->i_ino; ), - TP_printk("R=%08x CR=%08x c=%08x i=%x ", + TP_printk("R=%08x CR=%08x c=%08x o=%08x i=%x ", __entry->rreq, __entry->creq, __entry->cookie, + __entry->object, __entry->ino) ); @@ -610,10 +622,10 @@ TRACE_EVENT(netfs_collect, TP_ARGS(wreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, len) - __field(unsigned long long, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, len) + __field(uoff_t, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -636,12 +648,12 @@ TRACE_EVENT(netfs_collect_sreq, TP_ARGS(wreq, subreq), TP_STRUCT__entry( - __field(unsigned int, wreq) - __field(unsigned int, subreq) - __field(unsigned int, stream) - __field(unsigned int, len) - __field(unsigned int, transferred) - __field(unsigned long long, start) + __field(unsigned int, wreq) + __field(unsigned int, subreq) + __field(unsigned int, stream) + __field(unsigned int, len) + __field(unsigned int, transferred) + __field(uoff_t, start) ), TP_fast_assign( @@ -660,37 +672,30 @@ TRACE_EVENT(netfs_collect_sreq, TRACE_EVENT(netfs_collect_folio, TP_PROTO(const struct netfs_io_request *wreq, - const struct folio *folio, - unsigned long long fend, - unsigned long long collected_to), + const struct folio *folio), - TP_ARGS(wreq, folio, fend, collected_to), + TP_ARGS(wreq, folio), TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned long, index) - __field(unsigned long long, fend) - __field(unsigned long long, cleaned_to) - __field(unsigned long long, collected_to) + __field(unsigned int, nr) ), TP_fast_assign( __entry->wreq = wreq->debug_id; __entry->index = folio->index; - __entry->fend = fend; - __entry->cleaned_to = wreq->cleaned_to; - __entry->collected_to = collected_to; + __entry->nr = folio_nr_pages(folio); ), - TP_printk("R=%08x ix=%05lx r=%llx-%llx t=%llx/%llx", + TP_printk("R=%08x ix=%05lx-%05lx", __entry->wreq, __entry->index, - (unsigned long long)__entry->index * PAGE_SIZE, __entry->fend, - __entry->cleaned_to, __entry->collected_to) + __entry->index + __entry->nr - 1) ); TRACE_EVENT(netfs_collect_state, TP_PROTO(const struct netfs_io_request *wreq, - unsigned long long collected_to, + uoff_t collected_to, unsigned int notes), TP_ARGS(wreq, collected_to, notes), @@ -698,8 +703,8 @@ TRACE_EVENT(netfs_collect_state, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned int, notes) - __field(unsigned long long, collected_to) - __field(unsigned long long, cleaned_to) + __field(uoff_t, collected_to) + __field(uoff_t, cleaned_to) ), TP_fast_assign( @@ -718,7 +723,7 @@ TRACE_EVENT(netfs_collect_state, TRACE_EVENT(netfs_collect_gap, TP_PROTO(const struct netfs_io_request *wreq, const struct netfs_io_stream *stream, - unsigned long long jump_to, char type), + uoff_t jump_to, char type), TP_ARGS(wreq, stream, jump_to, type), @@ -726,8 +731,8 @@ TRACE_EVENT(netfs_collect_gap, __field(unsigned int, wreq) __field(unsigned char, stream) __field(unsigned char, type) - __field(unsigned long long, from) - __field(unsigned long long, to) + __field(uoff_t, from) + __field(uoff_t, to) ), TP_fast_assign( @@ -752,8 +757,8 @@ TRACE_EVENT(netfs_collect_stream, TP_STRUCT__entry( __field(unsigned int, wreq) __field(unsigned char, stream) - __field(unsigned long long, collected_to) - __field(unsigned long long, issued_to) + __field(uoff_t, collected_to) + __field(uoff_t, issued_to) ), TP_fast_assign( diff --git a/include/uapi/linux/close_range.h b/include/uapi/linux/close_range.h index 2d804281554c..7da9ed95258a 100644 --- a/include/uapi/linux/close_range.h +++ b/include/uapi/linux/close_range.h @@ -2,11 +2,34 @@ #ifndef _UAPI_LINUX_CLOSE_RANGE_H #define _UAPI_LINUX_CLOSE_RANGE_H -/* Unshare the file descriptor table before closing file descriptors. */ -#define CLOSE_RANGE_UNSHARE (1U << 1) +/* + * A macro of one of these names defined before this header is parsed, by + * a libc or by a program's own fallback, would replace the enumerator. + */ +#undef CLOSE_RANGE_UNSHARE +#undef CLOSE_RANGE_CLOEXEC +#undef CLOSE_RANGE_EXCEPT +#undef CLOSE_RANGE_CLOEXEC_ONLY -/* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ -#define CLOSE_RANGE_CLOEXEC (1U << 2) +enum close_range_flags { + /* Unshare the file descriptor table before closing file descriptors. */ + CLOSE_RANGE_UNSHARE = (1U << 1), + + /* Set the FD_CLOEXEC bit instead of closing the file descriptor. */ + CLOSE_RANGE_CLOEXEC = (1U << 2), + + /* Act on every file descriptor outside of the given range instead. */ + CLOSE_RANGE_EXCEPT = (1U << 3), + + /* Only close file descriptors that have the FD_CLOEXEC bit set. */ + CLOSE_RANGE_CLOEXEC_ONLY = (1U << 4), +}; + +/* Keep #ifdef working and let glibc skip its own definitions. */ +#define CLOSE_RANGE_UNSHARE CLOSE_RANGE_UNSHARE +#define CLOSE_RANGE_CLOEXEC CLOSE_RANGE_CLOEXEC +#define CLOSE_RANGE_EXCEPT CLOSE_RANGE_EXCEPT +#define CLOSE_RANGE_CLOEXEC_ONLY CLOSE_RANGE_CLOEXEC_ONLY #endif /* _UAPI_LINUX_CLOSE_RANGE_H */ diff --git a/include/uapi/linux/coredump.h b/include/uapi/linux/coredump.h index dc3789b78af0..6d0c53b534ea 100644 --- a/include/uapi/linux/coredump.h +++ b/include/uapi/linux/coredump.h @@ -11,12 +11,53 @@ * @COREDUMP_USERSPACE: userspace writes coredump * @COREDUMP_REJECT: don't generate coredump * @COREDUMP_WAIT: wait for coredump server + * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of + * as a plain byte stream, see struct coredump_record_header; + * requires COREDUMP_KERNEL + * @COREDUMP_SPARSE: describe the holes in the coredump as zero records + * instead of transferring them; requires COREDUMP_RECORDS + * @COREDUMP_MEMORY_TYPES: dump the memory types in + * coredump_ack->memory_types instead of the ones + * the task selected; requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), COREDUMP_USERSPACE = (1ULL << 1), COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), + COREDUMP_RECORDS = (1ULL << 4), + COREDUMP_SPARSE = (1ULL << 5), + COREDUMP_MEMORY_TYPES = (1ULL << 6), +}; + +/** + * coredump memory types + * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory + * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory + * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory + * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory + * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private + * mapping that starts an ELF file + * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory + * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory + * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory + * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory + * + * A bitmask of memory types a coredump may request to be included. New + * memory type bits must ensure that they do not steal memory from an + * existing one so a coredump server will continue to get the same + * coredumps even if a new bit is introduced. + */ +enum { + COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0), + COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1), + COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2), + COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3), + COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4), + COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5), + COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6), + COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7), + COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8), }; /** @@ -24,17 +65,19 @@ enum { * @size: size of struct coredump_req * @size_ack: known size of struct coredump_ack on this kernel * @mask: supported features + * @memory_types: the memory types the task selected + * @memory_types_mask: the memory types this kernel knows * * When a coredump happens the kernel will connect to the coredump * socket and send a coredump request to the coredump server. The @size * member is set to the size of struct coredump_req and provides a hint * to userspace how much data can be read. Userspace may use MSG_PEEK to * peek the size of struct coredump_req and then choose to consume it in - * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0 + * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0 * request. If the size the kernel sends is larger userspace simply * discards any remaining data. * - * The coredump_req->mask member is set to the currently know features. + * The coredump_req->mask member is set to the currently known features. * Userspace may only set coredump_ack->mask to the bits raised by the * kernel in coredump_req->mask. * @@ -42,15 +85,27 @@ enum { * struct coredump_ack the kernel knows. Userspace may only send up to * coredump_req->size_ack bytes to the kernel and must set * coredump_ack->size accordingly. + * + * @memory_types is set to the default memory types that are included in + * the coredump. This can be overridden by raising bits in + * coredump_ack->memory_types. + * + * @memory_types_mask contains a bitmask of all memory types the kernel + * knows about. A coredump server may only raise bits in + * coredump_ack->memory_types that are raised in + * coredump_req->memory_types_mask. */ struct coredump_req { __u32 size; __u32 size_ack; __u64 mask; + __u64 memory_types; + __u64 memory_types_mask; }; enum { COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */ }; /** @@ -58,6 +113,8 @@ enum { * @size: size of the struct * @spare: unused * @mask: features kernel is supposed to use + * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES + * in @mask * * The @size member must be set to the size of struct coredump_ack. It * may never exceed what the kernel returned in coredump_req->size_ack @@ -67,15 +124,30 @@ enum { * The @mask member must be set to the features the coredump server * wants the kernel to use. Only bits the kernel returned in * coredump_req->mask may be set. + * + * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the + * memory types set in the @memory_types mask. Zero is valid and dumps + * no memory apart from the mappings that are always dumped. + * + * Note that memory a task excluded via MADV_DONTDUMP is always left + * out. A coredump server wanting to add or drop memory types instead of + * outright replacing it should simply copy coredump_req->memory_types + * and then mask off or raise types as needed. + * + * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't + * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of + * at least COREDUMP_ACK_SIZE_VER1 bytes. */ struct coredump_ack { __u32 size; __u32 spare; __u64 mask; + __u64 memory_types; }; enum { COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */ }; /** @@ -83,11 +155,12 @@ enum { * * The kernel will place a single byte on the coredump socket. The * markers notify userspace whether the coredump ack succeeded or - * failed. + * failed. After any marker other than COREDUMP_MARK_REQACK the kernel + * closes the connection and no coredump is generated. * * @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small * @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big - * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid + * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid * @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options * @COREDUMP_MARK_REQACK: the coredump request and ack was successful * @__COREDUMP_MARK_MAX: the maximum coredump mark value @@ -101,4 +174,72 @@ enum coredump_mark { __COREDUMP_MARK_MAX = (1U << 31), }; +/** + * enum coredump_record_type - Type of a coredump record + * + * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data + * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed + * by any data and no further record is sent + * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not + * followed by any data + * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value + */ +enum coredump_record_type { + COREDUMP_RECORD_DATA = 0U, + COREDUMP_RECORD_END = 1U, + COREDUMP_RECORD_ZERO = 2U, + __COREDUMP_RECORD_TYPE_MAX = (1U << 31), +}; + +/** + * struct coredump_record_header - header of a coredump record + * @size: size of struct coredump_record_header + * @type: one of enum coredump_record_type + * @flags: modifiers for this record + * @offset: offset in the coredump this record starts at + * @len: number of coredump bytes this record accounts for + * + * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask + * the kernel doesn't send the coredump as a plain byte stream. It sends + * a sequence of records instead. A COREDUMP_RECORD_DATA record is + * followed by @len bytes of actual coredump data. A + * COREDUMP_RECORD_ZERO record is followed by nothing and stands for + * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never + * sees a zero record. Records arrive in order and leave no gaps. So + * @offset is the sum of the @len of all records before it. + * + * The last record is a COREDUMP_RECORD_END record. It is followed by + * nothing. Its @len is zero. Its @offset is the size of the coredump. + * The kernel only sends it once it has written the whole coredump. A + * server that hits end-of-file without having seen an end record must + * treat the coredump as incomplete. + * + * The @size member is set to the size of struct coredump_record_header + * the kernel knows and lets the header grow later. It comes first so it + * can be peeked. Userspace must consume @size bytes and discard + * anything beyond what it knows. It must refuse a @size smaller than + * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone. + * @offset and @len count coredump bytes. + * + * The @flags member carries modifiers that change how the record is to + * be interpreted. No flag is defined yet. Userspace must refuse a + * record carrying a flag or a type it doesn't know. Every new record + * type is raised in coredump_req->mask as a feature of its own. A + * server only ever sees the types it asked for. + * + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and + * COREDUMP_SPARSE with COREDUMP_RECORDS. + */ +struct coredump_record_header { + __u32 size; + __u32 type; + __u64 flags; + __u64 offset; + __u64 len; +}; + +enum { + COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */ +}; + #endif /* _UAPI_LINUX_COREDUMP_H */ diff --git a/include/uapi/linux/fs.h b/include/uapi/linux/fs.h index 34c6f219462a..a46c33692aa2 100644 --- a/include/uapi/linux/fs.h +++ b/include/uapi/linux/fs.h @@ -88,7 +88,7 @@ struct fstrim_range { * We include a length field because some filesystems (vfat) have an identifier * that we do want to expose as a UUID, but doesn't have the standard length. * - * We use a fixed size buffer beacuse this interface will, by fiat, never + * We use a fixed size buffer because this interface will, by fiat, never * support "UUIDs" longer than 16 bytes; we don't want to force all downstream * users to have to deal with that. */ diff --git a/init/Kconfig b/init/Kconfig index 8583d9f06c52..27c1ffc675bf 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1457,6 +1457,17 @@ config USER_NS If unsure, say N. +config USER_NS_MAP_KUNIT_TEST + tristate "KUint test for user namespace map insertion" if !KUNIT_ALL_TESTS + depends on USER_NS && KUNIT + default KUNIT_ALL_TESTS + help + This builds the KUnit test for user namespace uid/gid map insertion. + It validates map insertion, limits, dynamic allocation of the + extended extents array, and mapping sorting functions. + + If unsure, say N. + config PID_NS bool "PID Namespaces" default y diff --git a/init/initramfs.c b/init/initramfs.c index 3cee8b50ad82..ebddde9c8f0d 100644 --- a/init/initramfs.c +++ b/init/initramfs.c @@ -80,7 +80,7 @@ static __initdata struct hash { int ino, minor, major; umode_t mode; struct hash *next; - char name[N_ALIGN(PATH_MAX)]; + char name[]; } *head[32]; static __initdata bool hardlink_seen; @@ -92,7 +92,7 @@ static inline int hash(int major, int minor, int ino) } static char __init *find_link(int major, int minor, int ino, - umode_t mode, char *name) + umode_t mode, const char *name, size_t nlen) { struct hash **p, *q; for (p = head + hash(major, minor, ino); *p; p = &(*p)->next) { @@ -106,14 +106,15 @@ static char __init *find_link(int major, int minor, int ino, continue; return (*p)->name; } - q = kmalloc_obj(struct hash); + + q = kmalloc_flex(struct hash, name, nlen); if (!q) panic_show_mem("can't allocate link hash entry"); q->major = major; q->minor = minor; q->ino = ino; q->mode = mode; - strscpy(q->name, name); + strscpy(q->name, name, nlen); q->next = NULL; *p = q; hardlink_seen = true; @@ -355,7 +356,7 @@ static void __init clean_path(char *path, umode_t fmode) static int __init maybe_link(void) { if (nlink >= 2) { - char *old = find_link(major, minor, ino, mode, collected); + char *old = find_link(major, minor, ino, mode, collected, name_len); if (old) { clean_path(collected, 0); return (init_link(old, collected) < 0) ? -1 : 1; diff --git a/io_uring/io-wq.c b/io_uring/io-wq.c index 2ca223e47d41..2a980e86dd94 100644 --- a/io_uring/io-wq.c +++ b/io_uring/io-wq.c @@ -1324,6 +1324,8 @@ static bool io_task_work_match(struct callback_head *cb, void *data) void io_wq_exit_start(struct io_wq *wq) { set_bit(IO_WQ_BIT_EXIT, &wq->state); + /* Pairs with task_work_add() in io_queue_worker_create(). */ + smp_mb__after_atomic(); } static void io_wq_cancel_tw_create(struct io_wq *wq) diff --git a/io_uring/mock_file.c b/io_uring/mock_file.c index b318ed697998..9f0b4d850c12 100644 --- a/io_uring/mock_file.c +++ b/io_uring/mock_file.c @@ -257,17 +257,17 @@ static int io_create_mock_file(struct io_uring_cmd *cmd, unsigned int issue_flag FD_PREPARE(fdf, O_RDWR | O_CLOEXEC, anon_inode_create_getfile("[io_uring_mock]", fops, mf, O_RDWR | O_CLOEXEC, NULL)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; retain_and_null_ptr(mf); - file = fd_prepare_file(fdf); + file = fdf->file; file->f_mode |= FMODE_READ | FMODE_CAN_READ | FMODE_WRITE | FMODE_CAN_WRITE | FMODE_LSEEK; if (mc.flags & IORING_MOCK_CREATE_F_SUPPORT_NOWAIT) file->f_mode |= FMODE_NOWAIT; - mc.out_fd = fd_prepare_fd(fdf); + mc.out_fd = fdf->fd; if (copy_to_user(uarg, &mc, uarg_size)) return -EFAULT; diff --git a/ipc/mqueue.c b/ipc/mqueue.c index d1dd36a651b0..322c3dd98c44 100644 --- a/ipc/mqueue.c +++ b/ipc/mqueue.c @@ -528,6 +528,13 @@ static void mqueue_evict_inode(struct inode *inode) list_add_tail(&msg->m_list, &tmp_msg); kfree(info->node_cache); spin_unlock(&info->lock); + /* + * A shared file table can let the notification owner exit without + * running ->flush(). No users of the inode remain during eviction, so + * tear down any stale notification after dropping info->lock because + * netlink_sendskb() may release the final socket reference. + */ + remove_notification(info); list_for_each_entry_safe(msg, nmsg, &tmp_msg, m_list) { list_del(&msg->m_list); @@ -607,7 +614,7 @@ out_unlock: return error; } -static int mqueue_create(struct mnt_idmap *idmap, struct inode *dir, +static int mqueue_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return mqueue_create_attr(dentry, mode, NULL); diff --git a/kernel/Makefile b/kernel/Makefile index 1e1a31673577..2a64282749b8 100644 --- a/kernel/Makefile +++ b/kernel/Makefile @@ -141,6 +141,7 @@ obj-$(CONFIG_WATCH_QUEUE) += watch_queue.o obj-$(CONFIG_RESOURCE_KUNIT_TEST) += resource_kunit.o obj-$(CONFIG_SYSCTL_KUNIT_TEST) += sysctl-test.o +obj-$(CONFIG_USER_NS_MAP_KUNIT_TEST) += tests/user_ns_map_kunit.o CFLAGS_kstack_erase.o += $(DISABLE_KSTACK_ERASE) CFLAGS_kstack_erase.o += $(call cc-option,-mgeneral-regs-only) diff --git a/kernel/bpf/bpf_iter.c b/kernel/bpf/bpf_iter.c index b40eb404adab..d9191df5de22 100644 --- a/kernel/bpf/bpf_iter.c +++ b/kernel/bpf/bpf_iter.c @@ -643,11 +643,11 @@ int bpf_iter_new_fd(struct bpf_link *link) flags = O_RDONLY | O_CLOEXEC; FD_PREPARE(fdf, flags, anon_inode_getfile("bpf_iter", &bpf_iter_fops, NULL, flags)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; iter_link = container_of(link, struct bpf_iter_link, link); - err = prepare_seq_file(fd_prepare_file(fdf), iter_link); + err = prepare_seq_file(fdf->file, iter_link); if (err) return err; /* Automatic cleanup handles fput */ diff --git a/kernel/bpf/inode.c b/kernel/bpf/inode.c index 7837968c0842..c6f328e4752e 100644 --- a/kernel/bpf/inode.c +++ b/kernel/bpf/inode.c @@ -176,7 +176,7 @@ static void bpf_dentry_finalize(struct dentry *dentry, struct inode *inode, inode_set_mtime_to_ts(dir, inode_set_ctime_current(dir)); } -static struct dentry *bpf_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *bpf_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -424,7 +424,7 @@ bpf_lookup(struct inode *dir, struct dentry *dentry, unsigned flags) return simple_lookup(dir, dentry, flags); } -static int bpf_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int bpf_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *target) { struct inode *inode; @@ -874,7 +874,7 @@ enum { }; static int bpf_fs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) { diff --git a/kernel/bpf/token.c b/kernel/bpf/token.c index e85a179523f0..da915a4f972b 100644 --- a/kernel/bpf/token.c +++ b/kernel/bpf/token.c @@ -169,8 +169,8 @@ int bpf_token_create(union bpf_attr *attr) FD_PREPARE(fdf, O_CLOEXEC, alloc_file_pseudo(inode, path.mnt, BPF_TOKEN_INODE_NAME, O_RDWR, &bpf_token_fops)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; token = kzalloc_obj(*token, GFP_USER); if (!token) @@ -190,7 +190,7 @@ int bpf_token_create(union bpf_attr *attr) return err; get_user_ns(token->userns); - fd_prepare_file(fdf)->private_data = no_free_ptr(token); + fdf->file->private_data = no_free_ptr(token); return fd_publish(fdf); } diff --git a/kernel/capability.c b/kernel/capability.c index 90e6ab62f6db..a689dae590ff 100644 --- a/kernel/capability.c +++ b/kernel/capability.c @@ -470,7 +470,7 @@ EXPORT_SYMBOL(file_ns_capable); * Return true if the inode uid and gid are within the namespace. */ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct inode *inode) { return vfsuid_has_mapping(ns, i_uid_into_vfsuid(idmap, inode)) && @@ -487,7 +487,7 @@ bool privileged_wrt_inode_uidgid(struct user_namespace *ns, * its own user namespace and that the given inode's uid and gid are * mapped into the current user namespace. */ -bool capable_wrt_inode_uidgid(struct mnt_idmap *idmap, +bool capable_wrt_inode_uidgid(const struct mnt_idmap *idmap, const struct inode *inode, int cap) { struct user_namespace *ns = current_user_ns(); diff --git a/kernel/exit.c b/kernel/exit.c index 282328d2b4cf..29e853a36602 100644 --- a/kernel/exit.c +++ b/kernel/exit.c @@ -17,6 +17,7 @@ #include <linux/module.h> #include <linux/capability.h> #include <linux/completion.h> +#include <linux/wait_bit.h> #include <linux/personality.h> #include <linux/tty.h> #include <linux/iocontext.h> @@ -436,15 +437,14 @@ static void coredump_task_exit(struct task_struct *tsk, self.task = tsk; if (self.task->flags & PF_SIGNALED) - self.next = xchg(&core_state->dumper.next, &self); + self.next = xchg(&core_state->tasks, &self); else self.task = NULL; /* * Implies mb(), the result of xchg() must be visible - * to core_state->dumper. + * to the dumper. */ - if (atomic_dec_and_test(&core_state->nr_threads)) - complete(&core_state->startup); + atomic_dec_and_wake_up(&core_state->threads_remaining); for (;;) { set_current_state(TASK_IDLE|TASK_FREEZABLE); @@ -892,7 +892,7 @@ static void synchronize_group_exit(struct task_struct *tsk, long code) * Serialize with any possible pending coredump. * We must hold siglock around checking core_state * and setting PF_POSTCOREDUMP. The core-inducing thread - * will increment ->nr_threads for each thread in the + * will increment ->threads_remaining for each thread in the * group without PF_POSTCOREDUMP set. */ tsk->flags |= PF_POSTCOREDUMP; @@ -978,10 +978,11 @@ void __noreturn do_exit(long code) exit_sem(tsk); exit_shm(tsk); - exit_files(tsk); - exit_fs(tsk); + /* Hang the tty up before the last close of it can clear the session. */ if (group_dead) disassociate_ctty(1); + exit_files(tsk); + exit_fs(tsk); exit_nsproxy_namespaces(tsk); exit_task_work(tsk); exit_thread(tsk); diff --git a/kernel/fork.c b/kernel/fork.c index 10f2d05d816a..5a4cf4cf767a 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1677,6 +1677,7 @@ static int copy_files(u64 clone_flags, struct task_struct *tsk, if (clone_flags & CLONE_FILES) { atomic_inc(&oldf->count); + tsk->files = oldf; return 0; } @@ -2146,12 +2147,9 @@ __latent_entropy struct task_struct *copy_process( if (args->kthread) p->flags |= PF_KTHREAD; if (args->user_worker) { - /* - * Mark us a user worker, and block any signal that isn't - * fatal or STOP - */ + /* A user worker takes only the signals nobody can block. */ p->flags |= PF_USER_WORKER; - siginitsetinv(&p->blocked, sigmask(SIGKILL)|sigmask(SIGSTOP)); + siginitsetinv(&p->blocked, SIG_KERNEL_ONLY_MASK); } if (args->io_thread) p->flags |= PF_IO_WORKER; @@ -2200,6 +2198,8 @@ __latent_entropy struct task_struct *copy_process( INIT_LIST_HEAD(&p->sibling); rcu_copy_process(p); p->vfork_done = NULL; + /* Set by copy_files(), exit_files() on the error path skips NULL. */ + p->files = NULL; spin_lock_init(&p->alloc_lock); init_sigpending(&p->pending); @@ -2301,7 +2301,7 @@ __latent_entropy struct task_struct *copy_process( goto bad_fork_cleanup_semundo; retval = copy_fs(clone_flags, p, args->umh); if (retval) - goto bad_fork_cleanup_files; + goto bad_fork_cleanup_semundo; retval = copy_sighand(clone_flags, p); if (retval) goto bad_fork_cleanup_fs; @@ -2615,8 +2615,6 @@ bad_fork_cleanup_sighand: __cleanup_sighand(p->sighand); bad_fork_cleanup_fs: exit_fs(p); /* blocking */ -bad_fork_cleanup_files: - exit_files(p); /* blocking */ bad_fork_cleanup_semundo: exit_sem(p); bad_fork_cleanup_security: @@ -2627,6 +2625,8 @@ bad_fork_cleanup_perf: perf_event_free_task(p); bad_fork_sched_cancel_fork: sched_cancel_fork(p); + /* ->release() of a file may need scx_fork_rwsem for write. */ + exit_files(p); /* blocking */ bad_fork_cleanup_policy: lockdep_free_task(p); #ifdef CONFIG_NUMA @@ -2705,6 +2705,10 @@ struct task_struct *create_io_thread(int (*fn)(void *), void *arg, int node) .user_worker = 1, }; + /* A creator past its fatal signal or its coredump point gets no thread. */ + if (current->flags & (PF_SIGNALED | PF_POSTCOREDUMP)) + return ERR_PTR(-EINTR); + return copy_process(NULL, 0, node, &args); } @@ -3213,24 +3217,6 @@ static int unshare_fs(unsigned long unshare_flags, struct fs_struct **new_fsp) } /* - * Unshare file descriptor table if it is being shared - */ -static int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) -{ - struct files_struct *fd = current->files; - - if ((unshare_flags & CLONE_FILES) && - (fd && atomic_read(&fd->count) > 1)) { - fd = dup_fd(fd, NULL); - if (IS_ERR(fd)) - return PTR_ERR(fd); - *new_fdp = fd; - } - - return 0; -} - -/* * unshare allows a process to 'unshare' part of the process * context which was originally shared using clone. copy_* * functions used by kernel_clone() cannot be used here directly @@ -3325,10 +3311,8 @@ int ksys_unshare(unsigned long unshare_flags) if (new_fs) new_fs = switch_fs_struct(new_fs); - if (new_fd) { - guard(task_lock)(current); - swap(current->files, new_fd); - } + if (new_fd) + switch_files_struct(current, no_free_ptr(new_fd)); if (new_cred) { /* Install the new user namespace */ @@ -3361,30 +3345,6 @@ SYSCALL_DEFINE1(unshare, unsigned long, unshare_flags) return ksys_unshare(unshare_flags); } -/* - * Helper to unshare the files of the current task. - * We don't want to expose copy_files internals to - * the exec layer of the kernel. - */ - -int unshare_files(void) -{ - struct task_struct *task = current; - struct files_struct *old, *copy = NULL; - int error; - - error = unshare_fd(CLONE_FILES, ©); - if (error || !copy) - return error; - - old = task->files; - task_lock(task); - task->files = copy; - task_unlock(task); - put_files_struct(old); - return 0; -} - static int sysctl_max_threads(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { diff --git a/kernel/kthread.c b/kernel/kthread.c index a3f95c90456b..cc8cb5d3eab7 100644 --- a/kernel/kthread.c +++ b/kernel/kthread.c @@ -81,7 +81,7 @@ enum KTHREAD_BITS { static inline struct kthread *to_kthread(struct task_struct *k) { - WARN_ON(!(k->flags & PF_KTHREAD)); + WARN_ON(!(READ_ONCE(k->flags) & PF_KTHREAD)); return k->worker_private; } diff --git a/kernel/pid_namespace.c b/kernel/pid_namespace.c index d36afc58ee1d..8bc9edb40b78 100644 --- a/kernel/pid_namespace.c +++ b/kernel/pid_namespace.c @@ -238,9 +238,10 @@ void zap_pid_ns_processes(struct pid_namespace *pid_ns) * kernel_wait4() will also block until our children traced from the * parent namespace are detached and become EXIT_DEAD. */ + /* Task work must not busy-loop the reaper, see signal_pending(). */ + guard(no_notify_signal)(); do { clear_thread_flag(TIF_SIGPENDING); - clear_thread_flag(TIF_NOTIFY_SIGNAL); rc = kernel_wait4(-1, NULL, __WALL, NULL); } while (rc != -ECHILD); diff --git a/kernel/ptrace.c b/kernel/ptrace.c index d041645d9d17..4e9822a87aab 100644 --- a/kernel/ptrace.c +++ b/kernel/ptrace.c @@ -1227,6 +1227,12 @@ int ptrace_request(struct task_struct *child, long request, case PTRACE_SETSIGMASK: { sigset_t new_set; + /* A user worker only ever takes SIGKILL and SIGSTOP. */ + if (child->flags & PF_USER_WORKER) { + ret = -EPERM; + break; + } + if (addr != sizeof(sigset_t)) { ret = -EINVAL; break; diff --git a/kernel/signal.c b/kernel/signal.c index d31ebcb6ed4d..e433de93b430 100644 --- a/kernel/signal.c +++ b/kernel/signal.c @@ -3170,6 +3170,10 @@ static void retarget_shared_pending(struct task_struct *tsk, sigset_t *which) sigset_t retarget; struct task_struct *t; + /* Nobody dequeues them in a dying group, see get_signal(). */ + if (tsk->signal->flags & SIGNAL_GROUP_EXIT) + return; + sigandsets(&retarget, &tsk->signal->shared_pending.signal, which); if (sigisemptyset(&retarget)) return; @@ -3255,6 +3259,16 @@ long do_no_restart_syscall(struct restart_block *param) static void __set_task_blocked(struct task_struct *tsk, const sigset_t *newset) { + sigset_t floor, floored; + + /* A user worker never unblocks anything but SIGKILL and SIGSTOP. */ + if (unlikely(tsk->flags & PF_USER_WORKER)) { + siginitsetinv(&floor, SIG_KERNEL_ONLY_MASK); + sigorsets(&floored, newset, &floor); + WARN_ON_ONCE(!sigequalsets(&floored, newset)); + newset = &floored; + } + if (task_sigpending(tsk) && !thread_group_empty(tsk)) { sigset_t newblocked; /* A set of now blocked but previously unblocked signals. */ diff --git a/kernel/tests/.kunitconfig b/kernel/tests/.kunitconfig new file mode 100644 index 000000000000..b3d1206fd81a --- /dev/null +++ b/kernel/tests/.kunitconfig @@ -0,0 +1,4 @@ +CONFIG_KUNIT=y +CONFIG_NAMESPACES=y +CONFIG_USER_NS=y +CONFIG_USER_NS_MAP_KUNIT_TEST=y diff --git a/kernel/tests/user_ns_map_kunit.c b/kernel/tests/user_ns_map_kunit.c new file mode 100644 index 000000000000..033dccc6a535 --- /dev/null +++ b/kernel/tests/user_ns_map_kunit.c @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * KUnit test for user namespace map insertion and sorting. + */ + +#define pr_fmt(fmt) "user_namespace: " fmt + +#include <kunit/test.h> +#include <linux/user_namespace.h> + +#define NR_EXTENTS (UID_GID_MAP_MAX_BASE_EXTENTS + 5) + +static void user_ns_map_insert(struct kunit *test) +{ + struct uid_gid_map map; + struct uid_gid_extent extent; + int i, ret; + + memset(&map, 0, sizeof(map)); + + /* Insert up to UID_GID_MAP_MAX_BASE_EXTENTS elements */ + for (i = 0; i < UID_GID_MAP_MAX_BASE_EXTENTS; i++) { + extent.first = i * 10; + extent.lower_first = i * 100; + extent.count = 5; + + ret = uid_gid_map_insert_extent(&map, &extent); + KUNIT_ASSERT_EQ(test, ret, 0); + } + + KUNIT_EXPECT_EQ(test, map.nr_extents, UID_GID_MAP_MAX_BASE_EXTENTS); + + /* Verify the elements ended up in the 'extent' array */ + for (i = 0; i < UID_GID_MAP_MAX_BASE_EXTENTS; i++) { + KUNIT_EXPECT_EQ(test, map.extent[i].first, i * 10); + KUNIT_EXPECT_EQ(test, map.extent[i].lower_first, i * 100); + KUNIT_EXPECT_EQ(test, map.extent[i].count, 5); + } +} + +static void user_ns_map_insert_extended(struct kunit *test) +{ + struct uid_gid_map map; + struct uid_gid_extent extent; + int i, ret; + + memset(&map, 0, sizeof(map)); + + /* Insert more than UID_GID_MAP_MAX_BASE_EXTENTS elements */ + for (i = 0; i < NR_EXTENTS; i++) { + int value = 9 - i; + + extent.first = value * 10; + extent.lower_first = value * 100; + extent.count = 5; + + ret = uid_gid_map_insert_extent(&map, &extent); + KUNIT_ASSERT_EQ(test, ret, 0); + } + + KUNIT_EXPECT_EQ(test, map.nr_extents, NR_EXTENTS); + + /* Now sort the map to set up reverse mapping */ + ret = uid_gid_map_sort(&map); + KUNIT_ASSERT_EQ(test, ret, 0); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, map.reverse); + + /* Verify the elements are in 'forward' and that sorting is correct */ + for (i = 0; i < map.nr_extents; i++) { + KUNIT_EXPECT_EQ(test, map.forward[i].first, i * 10); + KUNIT_EXPECT_EQ(test, map.forward[i].lower_first, i * 100); + KUNIT_EXPECT_EQ(test, map.forward[i].count, 5); + + KUNIT_EXPECT_EQ(test, map.reverse[i].first, i * 10); + KUNIT_EXPECT_EQ(test, map.reverse[i].lower_first, i * 100); + KUNIT_EXPECT_EQ(test, map.reverse[i].count, 5); + } + + kfree(map.forward); + kfree(map.reverse); +} + +static struct kunit_case user_ns_map_test_cases[] = { + KUNIT_CASE(user_ns_map_insert), + KUNIT_CASE(user_ns_map_insert_extended), + {} +}; + +static struct kunit_suite user_ns_map_test_suite = { + .name = "user_ns_map", + .test_cases = user_ns_map_test_cases, +}; + +kunit_test_suite(user_ns_map_test_suite); + +MODULE_LICENSE("GPL"); +MODULE_DESCRIPTION("KUnit test for user namespace map insertion"); +MODULE_IMPORT_NS("EXPORTED_FOR_KUNIT_TESTING"); diff --git a/kernel/user_namespace.c b/kernel/user_namespace.c index 0bed462e9b2a..1b23d819d398 100644 --- a/kernel/user_namespace.c +++ b/kernel/user_namespace.c @@ -1,5 +1,6 @@ // SPDX-License-Identifier: GPL-2.0-only +#include <kunit/visibility.h> #include <linux/export.h> #include <linux/nsproxy.h> #include <linux/slab.h> @@ -162,9 +163,6 @@ int create_user_ns(struct cred *new) ns_tree_add(ns); return 0; fail_keyring: -#ifdef CONFIG_PERSISTENT_KEYRINGS - key_put(ns->persistent_keyring_register); -#endif ns_common_free(ns); fail_free: kmem_cache_free(user_ns_cachep, ns); @@ -278,8 +276,8 @@ static int cmp_map_id(const void *k, const void *e) * map_id_range_down_max - Find idmap via binary search in ordered idmap array. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS. */ -static struct uid_gid_extent * -map_id_range_down_max(unsigned extents, struct uid_gid_map *map, u32 id, u32 count) +static const struct uid_gid_extent * +map_id_range_down_max(unsigned extents, const struct uid_gid_map *map, u32 id, u32 count) { struct idmap_key key; @@ -296,8 +294,8 @@ map_id_range_down_max(unsigned extents, struct uid_gid_map *map, u32 id, u32 cou * Can only be called if number of mappings is equal or less than * UID_GID_MAP_MAX_BASE_EXTENTS. */ -static struct uid_gid_extent * -map_id_range_down_base(unsigned extents, struct uid_gid_map *map, u32 id, u32 count) +static const struct uid_gid_extent * +map_id_range_down_base(unsigned extents, const struct uid_gid_map *map, u32 id, u32 count) { unsigned idx; u32 first, last, id2; @@ -315,9 +313,9 @@ map_id_range_down_base(unsigned extents, struct uid_gid_map *map, u32 id, u32 co return NULL; } -static u32 map_id_range_down(struct uid_gid_map *map, u32 id, u32 count) +static u32 map_id_range_down(const struct uid_gid_map *map, u32 id, u32 count) { - struct uid_gid_extent *extent; + const struct uid_gid_extent *extent; unsigned extents = map->nr_extents; smp_rmb(); @@ -335,7 +333,7 @@ static u32 map_id_range_down(struct uid_gid_map *map, u32 id, u32 count) return id; } -u32 map_id_down(struct uid_gid_map *map, u32 id) +u32 map_id_down(const struct uid_gid_map *map, u32 id) { return map_id_range_down(map, id, 1); } @@ -345,8 +343,8 @@ u32 map_id_down(struct uid_gid_map *map, u32 id) * Can only be called if number of mappings is equal or less than * UID_GID_MAP_MAX_BASE_EXTENTS. */ -static struct uid_gid_extent * -map_id_range_up_base(unsigned extents, struct uid_gid_map *map, u32 id, u32 count) +static const struct uid_gid_extent * +map_id_range_up_base(unsigned extents, const struct uid_gid_map *map, u32 id, u32 count) { unsigned idx; u32 first, last, id2; @@ -368,8 +366,8 @@ map_id_range_up_base(unsigned extents, struct uid_gid_map *map, u32 id, u32 coun * map_id_up_max - Find idmap via binary search in ordered idmap array. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS. */ -static struct uid_gid_extent * -map_id_range_up_max(unsigned extents, struct uid_gid_map *map, u32 id, u32 count) +static const struct uid_gid_extent * +map_id_range_up_max(unsigned extents, const struct uid_gid_map *map, u32 id, u32 count) { struct idmap_key key; @@ -381,9 +379,9 @@ map_id_range_up_max(unsigned extents, struct uid_gid_map *map, u32 id, u32 count sizeof(struct uid_gid_extent), cmp_map_id); } -u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) +u32 map_id_range_up(const struct uid_gid_map *map, u32 id, u32 count) { - struct uid_gid_extent *extent; + const struct uid_gid_extent *extent; unsigned extents = map->nr_extents; smp_rmb(); @@ -401,7 +399,7 @@ u32 map_id_range_up(struct uid_gid_map *map, u32 id, u32 count) return id; } -u32 map_id_up(struct uid_gid_map *map, u32 id) +u32 map_id_up(const struct uid_gid_map *map, u32 id) { return map_id_range_up(map, id, 1); } @@ -782,11 +780,13 @@ static bool mappings_overlap(struct uid_gid_map *new_map, } /* - * insert_extent - Safely insert a new idmap extent into struct uid_gid_map. + * uid_gid_map_insert_extent - Safely insert a new idmap extent into + * struct uid_gid_map. * Takes care to allocate a 4K block of memory if the number of mappings exceeds * UID_GID_MAP_MAX_BASE_EXTENTS. */ -static int insert_extent(struct uid_gid_map *map, struct uid_gid_extent *extent) +VISIBLE_IF_KUNIT int uid_gid_map_insert_extent(struct uid_gid_map *map, + struct uid_gid_extent *extent) { struct uid_gid_extent *dest; @@ -809,15 +809,20 @@ static int insert_extent(struct uid_gid_map *map, struct uid_gid_extent *extent) map->reverse = NULL; } - if (map->nr_extents < UID_GID_MAP_MAX_BASE_EXTENTS) - dest = &map->extent[map->nr_extents]; + /* + * nr_extents must be updated before the extent and forward arrays are + * accessed, otherwise KSAN will assert an out-of-bounds error. + */ + map->nr_extents++; + if (map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS) + dest = &map->extent[map->nr_extents - 1]; else - dest = &map->forward[map->nr_extents]; + dest = &map->forward[map->nr_extents - 1]; *dest = *extent; - map->nr_extents++; return 0; } +EXPORT_SYMBOL_IF_KUNIT(uid_gid_map_insert_extent); /* cmp function to sort() forward mappings */ static int cmp_extents_forward(const void *a, const void *b) @@ -850,10 +855,10 @@ static int cmp_extents_reverse(const void *a, const void *b) } /* - * sort_idmaps - Sorts an array of idmap entries. + * uid_gid_map_sort - Sorts an array of idmap entries. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS. */ -static int sort_idmaps(struct uid_gid_map *map) +VISIBLE_IF_KUNIT int uid_gid_map_sort(struct uid_gid_map *map) { if (map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS) return 0; @@ -874,6 +879,7 @@ static int sort_idmaps(struct uid_gid_map *map) return 0; } +EXPORT_SYMBOL_IF_KUNIT(uid_gid_map_sort); /** * verify_root_map() - check the uid 0 mapping @@ -1042,7 +1048,7 @@ static ssize_t map_write(struct file *file, const char __user *buf, (next_line != NULL)) goto out; - ret = insert_extent(&new_map, &extent); + ret = uid_gid_map_insert_extent(&new_map, &extent); if (ret < 0) goto out; ret = -EINVAL; @@ -1086,7 +1092,7 @@ static ssize_t map_write(struct file *file, const char __user *buf, * If we want to use binary search for lookup, this clones the extent * array and sorts both copies. */ - ret = sort_idmaps(&new_map); + ret = uid_gid_map_sort(&new_map); if (ret < 0) goto out; diff --git a/kernel/utsname.c b/kernel/utsname.c index ebbfc578a9d3..1ebf87e24607 100644 --- a/kernel/utsname.c +++ b/kernel/utsname.c @@ -81,7 +81,6 @@ struct uts_namespace *copy_utsname(u64 flags, { struct uts_namespace *new_ns; - BUG_ON(!old_ns); get_uts_ns(old_ns); if (!(flags & CLONE_NEWUTS)) diff --git a/mm/secretmem.c b/mm/secretmem.c index 384f5cfc457f..af95b4dc943a 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -238,7 +238,7 @@ const struct address_space_operations secretmem_aops = { .migrate_folio = secretmem_migrate_folio, }; -static int secretmem_setattr(struct mnt_idmap *idmap, +static int secretmem_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/mm/shmem.c b/mm/shmem.c index 848316eaa7f4..0d891ade18b5 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1292,7 +1292,7 @@ void shmem_truncate_range(struct inode *inode, loff_t lstart, uoff_t lend) } EXPORT_SYMBOL_GPL(shmem_truncate_range); -static int shmem_getattr(struct mnt_idmap *idmap, +static int shmem_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -1326,7 +1326,7 @@ static int shmem_getattr(struct mnt_idmap *idmap, return 0; } -static int shmem_setattr(struct mnt_idmap *idmap, +static int shmem_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -3025,7 +3025,7 @@ static struct offset_ctx *shmem_get_offset_ctx(struct inode *inode) return &SHMEM_I(inode)->dir_offsets; } -static struct inode *__shmem_get_inode(struct mnt_idmap *idmap, +static struct inode *__shmem_get_inode(const struct mnt_idmap *idmap, struct super_block *sb, struct inode *dir, umode_t mode, dev_t dev, vma_flags_t flags) @@ -3105,7 +3105,7 @@ static struct inode *__shmem_get_inode(struct mnt_idmap *idmap, } #ifdef CONFIG_TMPFS_QUOTA -static struct inode *shmem_get_inode(struct mnt_idmap *idmap, +static struct inode *shmem_get_inode(const struct mnt_idmap *idmap, struct super_block *sb, struct inode *dir, umode_t mode, dev_t dev, vma_flags_t flags) { @@ -3133,7 +3133,7 @@ errout: return ERR_PTR(err); } #else -static struct inode *shmem_get_inode(struct mnt_idmap *idmap, +static struct inode *shmem_get_inode(const struct mnt_idmap *idmap, struct super_block *sb, struct inode *dir, umode_t mode, dev_t dev, vma_flags_t flags) { @@ -3821,7 +3821,7 @@ static int shmem_statfs(struct dentry *dentry, struct kstatfs *buf) * File creation. Allocate an inode, and we're done.. */ static int -shmem_mknod(struct mnt_idmap *idmap, struct inode *dir, +shmem_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode *inode; @@ -3860,7 +3860,7 @@ out_iput: } static int -shmem_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +shmem_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode; @@ -3888,7 +3888,7 @@ out_iput: return error; } -static struct dentry *shmem_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *shmem_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int error; @@ -3900,7 +3900,7 @@ static struct dentry *shmem_mkdir(struct mnt_idmap *idmap, struct inode *dir, return NULL; } -static int shmem_create(struct mnt_idmap *idmap, struct inode *dir, +static int shmem_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return shmem_mknod(idmap, dir, dentry, mode | S_IFREG, 0); @@ -3973,7 +3973,7 @@ static int shmem_rmdir(struct inode *dir, struct dentry *dentry) return shmem_unlink(dir, dentry); } -static int shmem_whiteout(struct mnt_idmap *idmap, +static int shmem_whiteout(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry) { struct dentry *whiteout; @@ -3994,7 +3994,7 @@ static int shmem_whiteout(struct mnt_idmap *idmap, * it exists so that the VFS layer correctly free's it when it * gets overwritten. */ -static int shmem_rename2(struct mnt_idmap *idmap, +static int shmem_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -4050,7 +4050,7 @@ static int shmem_rename2(struct mnt_idmap *idmap, return 0; } -static int shmem_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int shmem_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int error; @@ -4162,7 +4162,7 @@ static int shmem_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -static int shmem_fileattr_set(struct mnt_idmap *idmap, +static int shmem_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -4267,7 +4267,7 @@ static int shmem_xattr_handler_get(const struct xattr_handler *handler, } static int shmem_xattr_handler_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -5795,7 +5795,7 @@ static inline void shmem_unacct_size(unsigned long flags, loff_t size) { } -static inline struct inode *shmem_get_inode(struct mnt_idmap *idmap, +static inline struct inode *shmem_get_inode(const struct mnt_idmap *idmap, struct super_block *sb, struct inode *dir, umode_t mode, dev_t dev, vma_flags_t flags) { diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 74f04c323c50..6a033df14027 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -4811,12 +4811,12 @@ static int new_userfaultfd(int flags) anon_inode_create_getfile("[userfaultfd]", &userfaultfd_fops, ctx, O_RDONLY | (flags & UFFD_SHARED_FCNTL_FLAGS), NULL)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; /* prevent the mm struct to be freed */ mmgrab(ctx->mm); - fd_prepare_file(fdf)->f_mode |= FMODE_NOWAIT; + fdf->file->f_mode |= FMODE_NOWAIT; retain_and_null_ptr(ctx); return fd_publish(fdf); } diff --git a/net/core/scm.c b/net/core/scm.c index f0d44ecdb11f..15c330784a69 100644 --- a/net/core/scm.c +++ b/net/core/scm.c @@ -364,14 +364,14 @@ int scm_recv_one_fd(struct file *f, int __user *ufd, unsigned int flags, return notrunc ? put_user(error, ufd) : error; FD_PREPARE(fdf, flags, get_file(f)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; - error = put_user(fd_prepare_fd(fdf), ufd); + error = put_user(fdf->fd, ufd); if (error) return error; - __receive_sock(fd_prepare_file(fdf)); + __receive_sock(fdf->file); return fd_publish(fdf); } diff --git a/net/handshake/netlink.c b/net/handshake/netlink.c index 3fd4fef9bab1..73b8314d9010 100644 --- a/net/handshake/netlink.c +++ b/net/handshake/netlink.c @@ -107,17 +107,17 @@ int handshake_nl_accept_doit(struct sk_buff *skb, struct genl_info *info) req = handshake_req_next(hn, class); if (req) { FD_PREPARE(fdf, O_CLOEXEC, req->hr_file); - if (fdf.err) { + if (fdf->fd < 0) { fput(req->hr_file); /* drop ref from handshake_req_next() */ - err = fdf.err; + err = fdf->fd; goto out_complete; } - err = req->hr_proto->hp_accept(req, info, fd_prepare_fd(fdf)); + err = req->hr_proto->hp_accept(req, info, fdf->fd); if (err) goto out_complete; /* Automatic cleanup handles fput */ - trace_handshake_cmd_accept(net, req, req->hr_sk, fd_prepare_fd(fdf)); + trace_handshake_cmd_accept(net, req, req->hr_sk, fdf->fd); fd_publish(fdf); return 0; } diff --git a/net/kcm/kcmsock.c b/net/kcm/kcmsock.c index 71af69d442f2..962ee2c4acd2 100644 --- a/net/kcm/kcmsock.c +++ b/net/kcm/kcmsock.c @@ -1580,10 +1580,10 @@ static int kcm_ioctl(struct socket *sock, unsigned int cmd, unsigned long arg) struct kcm_clone info; FD_PREPARE(fdf, 0, kcm_clone(sock)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; - info.fd = fd_prepare_fd(fdf); + info.fd = fdf->fd; if (copy_to_user((void __user *)arg, &info, sizeof(info))) return -EFAULT; diff --git a/net/socket.c b/net/socket.c index c05d86e63abf..c0ab6a3ad8ed 100644 --- a/net/socket.c +++ b/net/socket.c @@ -422,7 +422,7 @@ static const struct xattr_handler sockfs_xattr_handler = { }; static int sockfs_security_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) @@ -447,7 +447,7 @@ static int sockfs_user_xattr_get(const struct xattr_handler *handler, } static int sockfs_user_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) @@ -670,7 +670,7 @@ static ssize_t sockfs_listxattr(struct dentry *dentry, char *buffer, return used; } -static int sockfs_setattr(struct mnt_idmap *idmap, +static int sockfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int err = simple_setattr(&nop_mnt_idmap, dentry, iattr); diff --git a/net/unix/af_unix.c b/net/unix/af_unix.c index 42cffeafc8c1..d3b459310523 100644 --- a/net/unix/af_unix.c +++ b/net/unix/af_unix.c @@ -1361,7 +1361,7 @@ static int unix_bind_bsd(struct sock *sk, struct sockaddr_un *sunaddr, struct unix_sock *u = unix_sk(sk); unsigned int new_hash, old_hash; struct net *net = sock_net(sk); - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct unix_address *addr; struct dentry *dentry; struct path parent; diff --git a/security/apparmor/apparmorfs.c b/security/apparmor/apparmorfs.c index 5b42140e12e8..d1386722537d 100644 --- a/security/apparmor/apparmorfs.c +++ b/security/apparmor/apparmorfs.c @@ -2073,7 +2073,7 @@ fail2: return error; } -static struct dentry *ns_mkdir_op(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ns_mkdir_op(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct aa_ns *ns, *parent; diff --git a/security/apparmor/lsm.c b/security/apparmor/lsm.c index d502ad0ac26f..c73681d820a0 100644 --- a/security/apparmor/lsm.c +++ b/security/apparmor/lsm.c @@ -397,7 +397,7 @@ static int apparmor_path_rename(const struct path *old_dir, struct dentry *old_d label = begin_current_label_crit_section(&needput); if (!unconfined(label)) { - struct mnt_idmap *idmap = mnt_idmap(old_dir->mnt); + const struct mnt_idmap *idmap = mnt_idmap(old_dir->mnt); vfsuid_t vfsuid; struct path old_path = { .mnt = old_dir->mnt, .dentry = old_dentry }; @@ -485,7 +485,7 @@ static int apparmor_file_open(struct file *file) label = aa_get_newest_cred_label_condref(file->f_cred, &needput); if (!unconfined(label)) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct inode *inode = file_inode(file); vfsuid_t vfsuid; struct path_cond cond = { diff --git a/security/commoncap.c b/security/commoncap.c index 3399535808fe..d47ab3022343 100644 --- a/security/commoncap.c +++ b/security/commoncap.c @@ -348,7 +348,7 @@ int cap_inode_need_killpriv(struct dentry *dentry) * * Return: 0 if successful, -ve on error. */ -int cap_inode_killpriv(struct mnt_idmap *idmap, struct dentry *dentry) +int cap_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { int error; @@ -417,7 +417,7 @@ static bool is_v3header(int size, const struct vfs_cap_data *cap) * by the integrity subsystem, which really wants the unconverted values - * so that's good. */ -int cap_inode_getsecurity(struct mnt_idmap *idmap, +int cap_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) { @@ -566,7 +566,7 @@ static bool validheader(size_t size, const struct vfs_cap_data *cap) * * Return: On success, return the new size; on error, return < 0. */ -int cap_convert_nscap(struct mnt_idmap *idmap, struct dentry *dentry, +int cap_convert_nscap(const struct mnt_idmap *idmap, struct dentry *dentry, const void **ivalue, size_t size) { struct vfs_ns_cap_data *nscap; @@ -672,7 +672,7 @@ static inline int bprm_caps_from_vfs_caps(struct cpu_vfs_cap_data *caps, * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -int get_vfs_caps_from_disk(struct mnt_idmap *idmap, +int get_vfs_caps_from_disk(const struct mnt_idmap *idmap, const struct dentry *dentry, struct cpu_vfs_cap_data *cpu_caps) { @@ -1063,7 +1063,7 @@ int cap_inode_setxattr(struct dentry *dentry, const char *name, * This is used to make sure security xattrs don't get removed by those who * aren't privileged to remove them. */ -int cap_inode_removexattr(struct mnt_idmap *idmap, +int cap_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct user_namespace *user_ns = dentry->d_sb->s_user_ns; diff --git a/security/integrity/evm/evm_main.c b/security/integrity/evm/evm_main.c index b59e3f121b8a..9bfea858df2d 100644 --- a/security/integrity/evm/evm_main.c +++ b/security/integrity/evm/evm_main.c @@ -481,7 +481,7 @@ static enum integrity_status evm_verify_current_integrity(struct dentry *dentry) * * Returns 1 if passed xattr value differs from current value, 0 otherwise. */ -static int evm_xattr_change(struct mnt_idmap *idmap, +static int evm_xattr_change(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name, const void *xattr_value, size_t xattr_value_len) { @@ -517,7 +517,7 @@ out: * For posix xattr acls only, permit security.evm, even if it currently * doesn't exist, to be updated unless the EVM signature is immutable. */ -static int evm_protect_xattr(struct mnt_idmap *idmap, +static int evm_protect_xattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name, const void *xattr_value, size_t xattr_value_len) { @@ -607,7 +607,7 @@ out: * userspace from writing HMAC value. Writing 'security.evm' requires * requires CAP_SYS_ADMIN privileges. */ -static int evm_inode_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int evm_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name, const void *xattr_value, size_t xattr_value_len, int flags) { @@ -639,7 +639,7 @@ static int evm_inode_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, * Removing 'security.evm' requires CAP_SYS_ADMIN privileges and that * the current value is valid. */ -static int evm_inode_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int evm_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name) { /* Policy permits modification of the protected xattrs even though @@ -652,7 +652,7 @@ static int evm_inode_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, } #ifdef CONFIG_FS_POSIX_ACL -static int evm_inode_set_acl_change(struct mnt_idmap *idmap, +static int evm_inode_set_acl_change(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *kacl) { @@ -671,7 +671,7 @@ static int evm_inode_set_acl_change(struct mnt_idmap *idmap, return 0; } #else -static inline int evm_inode_set_acl_change(struct mnt_idmap *idmap, +static inline int evm_inode_set_acl_change(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct posix_acl *kacl) @@ -693,7 +693,7 @@ static inline int evm_inode_set_acl_change(struct mnt_idmap *idmap, * * Return: zero on success, -EPERM on failure. */ -static int evm_inode_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +static int evm_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { enum integrity_status evm_status; @@ -745,7 +745,7 @@ static int evm_inode_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, * * Return: zero on success, -EPERM on failure. */ -static int evm_inode_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +static int evm_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return evm_inode_set_acl(idmap, dentry, acl_name, NULL); @@ -926,14 +926,14 @@ static void evm_inode_post_removexattr(struct dentry *dentry, * Update the 'security.evm' xattr with the EVM HMAC re-calculated after * removing posix acls. */ -static inline void evm_inode_post_remove_acl(struct mnt_idmap *idmap, +static inline void evm_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { evm_inode_post_removexattr(dentry, acl_name); } -static int evm_attr_change(struct mnt_idmap *idmap, +static int evm_attr_change(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_backing_inode(dentry); @@ -956,7 +956,7 @@ static int evm_attr_change(struct mnt_idmap *idmap, * Permit update of file attributes when files have a valid EVM signature, * except in the case of them having an immutable portable signature. */ -static int evm_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int evm_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; @@ -1008,7 +1008,7 @@ static int evm_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, * This function is called from notify_change(), which expects the caller * to lock the inode's i_mutex. */ -static void evm_inode_post_setattr(struct mnt_idmap *idmap, +static void evm_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { if (!evm_revalidate_status(NULL)) @@ -1140,7 +1140,7 @@ static void evm_file_release(struct file *file) iint->flags &= ~EVM_NEW_FILE; } -static void evm_post_path_mknod(struct mnt_idmap *idmap, struct dentry *dentry) +static void evm_post_path_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { struct inode *inode = d_backing_inode(dentry); struct evm_iint_cache *iint = evm_iint_inode(inode); diff --git a/security/integrity/ima/ima.h b/security/integrity/ima/ima.h index 10214f73ca1e..b502854f28ee 100644 --- a/security/integrity/ima/ima.h +++ b/security/integrity/ima/ima.h @@ -423,7 +423,7 @@ static inline void ima_process_queued_keys(void) {} #endif /* CONFIG_IMA_QUEUE_EARLY_BOOT_KEYS */ /* LIM API function definitions */ -int ima_get_action(struct mnt_idmap *idmap, struct inode *inode, +int ima_get_action(const struct mnt_idmap *idmap, struct inode *inode, const struct cred *cred, struct lsm_prop *prop, int mask, enum ima_hooks func, int *pcr, struct ima_template_desc **template_desc, @@ -437,7 +437,7 @@ void ima_store_measurement(struct ima_iint_cache *iint, struct file *file, struct evm_ima_xattr_data *xattr_value, int xattr_len, const struct modsig *modsig, int pcr, struct ima_template_desc *template_desc); -int process_buffer_measurement(struct mnt_idmap *idmap, +int process_buffer_measurement(const struct mnt_idmap *idmap, struct inode *inode, const void *buf, int size, const char *eventname, enum ima_hooks func, int pcr, const char *func_data, @@ -454,7 +454,7 @@ void ima_free_template_entry(struct ima_template_entry *entry); const char *ima_d_path(const struct path *path, char **pathbuf, char *filename); /* IMA policy related functions */ -int ima_match_policy(struct mnt_idmap *idmap, struct inode *inode, +int ima_match_policy(const struct mnt_idmap *idmap, struct inode *inode, const struct cred *cred, struct lsm_prop *prop, enum ima_hooks func, int mask, int flags, int *pcr, struct ima_template_desc **template_desc, @@ -489,7 +489,7 @@ int ima_appraise_measurement(enum ima_hooks func, struct ima_iint_cache *iint, struct evm_ima_xattr_data *xattr_value, int xattr_len, const struct modsig *modsig, bool bprm_is_check); -int ima_must_appraise(struct mnt_idmap *idmap, struct inode *inode, +int ima_must_appraise(const struct mnt_idmap *idmap, struct inode *inode, int mask, enum ima_hooks func); void ima_update_xattr(struct ima_iint_cache *iint, struct file *file); enum integrity_status ima_get_cache_status(struct ima_iint_cache *iint, @@ -519,7 +519,7 @@ static inline int ima_appraise_measurement(enum ima_hooks func, return INTEGRITY_UNKNOWN; } -static inline int ima_must_appraise(struct mnt_idmap *idmap, +static inline int ima_must_appraise(const struct mnt_idmap *idmap, struct inode *inode, int mask, enum ima_hooks func) { diff --git a/security/integrity/ima/ima_api.c b/security/integrity/ima/ima_api.c index 122d127e108d..3c5a23b4a2ff 100644 --- a/security/integrity/ima/ima_api.c +++ b/security/integrity/ima/ima_api.c @@ -188,7 +188,7 @@ err_out: * Returns IMA_MEASURE, IMA_APPRAISE mask. * */ -int ima_get_action(struct mnt_idmap *idmap, struct inode *inode, +int ima_get_action(const struct mnt_idmap *idmap, struct inode *inode, const struct cred *cred, struct lsm_prop *prop, int mask, enum ima_hooks func, int *pcr, struct ima_template_desc **template_desc, diff --git a/security/integrity/ima/ima_appraise.c b/security/integrity/ima/ima_appraise.c index b280488e15fc..63b0cb957c65 100644 --- a/security/integrity/ima/ima_appraise.c +++ b/security/integrity/ima/ima_appraise.c @@ -71,7 +71,7 @@ bool is_ima_appraise_enabled(void) * * Return 1 to appraise or hash */ -int ima_must_appraise(struct mnt_idmap *idmap, struct inode *inode, +int ima_must_appraise(const struct mnt_idmap *idmap, struct inode *inode, int mask, enum ima_hooks func) { struct lsm_prop prop; @@ -634,7 +634,7 @@ void ima_update_xattr(struct ima_iint_cache *iint, struct file *file) * This function is called from notify_change(), which expects the caller * to lock the inode's i_mutex. */ -static void ima_inode_post_setattr(struct mnt_idmap *idmap, +static void ima_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { struct inode *inode = d_backing_inode(dentry); @@ -759,7 +759,7 @@ static int validate_hash_algo(struct dentry *dentry, return -EACCES; } -static int ima_inode_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int ima_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name, const void *xattr_value, size_t xattr_value_len, int flags) { @@ -792,7 +792,7 @@ static int ima_inode_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, return result; } -static int ima_inode_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +static int ima_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { if (evm_revalidate_status(acl_name)) @@ -801,7 +801,7 @@ static int ima_inode_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, return 0; } -static int ima_inode_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int ima_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *xattr_name) { int result, digsig = -1; @@ -817,7 +817,7 @@ static int ima_inode_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, return result; } -static int ima_inode_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +static int ima_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return ima_inode_set_acl(idmap, dentry, acl_name, NULL); diff --git a/security/integrity/ima/ima_main.c b/security/integrity/ima/ima_main.c index ab1e53b3210d..7ac38a98b1f9 100644 --- a/security/integrity/ima/ima_main.c +++ b/security/integrity/ima/ima_main.c @@ -846,7 +846,7 @@ EXPORT_SYMBOL_GPL(ima_inode_hash); * Skip calling process_measurement(), but indicate which newly, created * tmpfiles are in policy. */ -static void ima_post_create_tmpfile(struct mnt_idmap *idmap, +static void ima_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { @@ -879,7 +879,7 @@ static void ima_post_create_tmpfile(struct mnt_idmap *idmap, * Mark files created via the mknodat syscall as new, so that the * file data can be written later. */ -static void ima_post_path_mknod(struct mnt_idmap *idmap, struct dentry *dentry) +static void ima_post_path_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { struct ima_iint_cache *iint; struct inode *inode = dentry->d_inode; @@ -1095,7 +1095,7 @@ static int ima_post_load_data(char *buf, loff_t size, * has been written to the passed location but not added to a measurement entry, * a negative value otherwise. */ -int process_buffer_measurement(struct mnt_idmap *idmap, +int process_buffer_measurement(const struct mnt_idmap *idmap, struct inode *inode, const void *buf, int size, const char *eventname, enum ima_hooks func, int pcr, const char *func_data, diff --git a/security/integrity/ima/ima_policy.c b/security/integrity/ima/ima_policy.c index 68d9a5e6c232..b34d9621929b 100644 --- a/security/integrity/ima/ima_policy.c +++ b/security/integrity/ima/ima_policy.c @@ -580,7 +580,7 @@ static bool ima_match_rule_data(struct ima_rule_entry *rule, * Returns true on rule match, false on failure. */ static bool ima_match_rules(struct ima_rule_entry *rule, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode, const struct cred *cred, struct lsm_prop *prop, enum ima_hooks func, int mask, const char *func_data) @@ -762,7 +762,7 @@ static int get_subaction(struct ima_rule_entry *rule, enum ima_hooks func) * list when walking it. Reads are many orders of magnitude more numerous * than writes so ima_match_policy() is classical RCU candidate. */ -int ima_match_policy(struct mnt_idmap *idmap, struct inode *inode, +int ima_match_policy(const struct mnt_idmap *idmap, struct inode *inode, const struct cred *cred, struct lsm_prop *prop, enum ima_hooks func, int mask, int flags, int *pcr, struct ima_template_desc **template_desc, diff --git a/security/security.c b/security/security.c index 2ee276ab15c5..f476b0736c9a 100644 --- a/security/security.c +++ b/security/security.c @@ -596,6 +596,31 @@ int security_ptrace_traceme(struct task_struct *parent) } /** + * security_mem_foll_force() - Check if FOLL_FORCE is allowed + * @subject: credentials using which /proc/$pid/mem was opened + * @opened_by_owner: whether checks on open() were bypassed because the opener + * has the same MM as the target + * + * Check if FOLL_FORCE is allowed for accessing process memory through + * /proc/$pid/mem. opened_by_owner signals whether the opener's MM was the same + * as the target MM, meaning the security_ptrace_access_check() hook was + * bypassed on open(). + * (Current current->mm does not matter for this; for example, if write() is + * called on an FD that was received from another process which obtained it with + * open("/proc/self/mem"), @opened_by_owner is still true.) + * + * Note that this hook is only designed to be useful in the opened_by_owner + * case, where the subject credentials effectively also describe the object. + * + * Return: Returns 0 if permission is granted. + */ +int security_mem_foll_force(const struct cred *subject, + bool opened_by_owner) +{ + return call_int_hook(mem_foll_force, subject, opened_by_owner); +} + +/** * security_capget() - Get the capability sets for a process * @target: target process * @effective: effective capability set @@ -1425,7 +1450,7 @@ EXPORT_SYMBOL(security_path_mknod); * * Update inode security field after a regular file has been created. */ -void security_path_post_mknod(struct mnt_idmap *idmap, struct dentry *dentry) +void security_path_post_mknod(const struct mnt_idmap *idmap, struct dentry *dentry) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) return; @@ -1638,7 +1663,7 @@ EXPORT_SYMBOL_GPL(security_inode_create); * * Update inode security data after a tmpfile has been created. */ -void security_inode_post_create_tmpfile(struct mnt_idmap *idmap, +void security_inode_post_create_tmpfile(const struct mnt_idmap *idmap, struct inode *inode) { if (unlikely(IS_PRIVATE(inode))) @@ -1855,7 +1880,7 @@ int security_inode_permission(struct inode *inode, int mask) * * Return: Returns 0 if permission is granted. */ -int security_inode_setattr(struct mnt_idmap *idmap, +int security_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) @@ -1872,7 +1897,7 @@ EXPORT_SYMBOL_GPL(security_inode_setattr); * * Update inode security field after successful setting file attributes. */ -void security_inode_post_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +void security_inode_post_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, int ia_valid) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) @@ -1921,7 +1946,7 @@ int security_inode_getattr(const struct path *path) * * Return: Returns 0 if permission is granted. */ -int security_inode_setxattr(struct mnt_idmap *idmap, +int security_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -1953,7 +1978,7 @@ int security_inode_setxattr(struct mnt_idmap *idmap, * * Return: Returns 0 if permission is granted. */ -int security_inode_set_acl(struct mnt_idmap *idmap, +int security_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { @@ -1990,7 +2015,7 @@ void security_inode_post_set_acl(struct dentry *dentry, const char *acl_name, * * Return: Returns 0 if permission is granted. */ -int security_inode_get_acl(struct mnt_idmap *idmap, +int security_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) @@ -2009,7 +2034,7 @@ int security_inode_get_acl(struct mnt_idmap *idmap, * * Return: Returns 0 if permission is granted. */ -int security_inode_remove_acl(struct mnt_idmap *idmap, +int security_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) @@ -2026,7 +2051,7 @@ int security_inode_remove_acl(struct mnt_idmap *idmap, * Update inode security data after successfully removing posix acls on * @dentry in @idmap. The posix acls are identified by @acl_name. */ -void security_inode_post_remove_acl(struct mnt_idmap *idmap, +void security_inode_post_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { if (unlikely(IS_PRIVATE(d_backing_inode(dentry)))) @@ -2108,7 +2133,7 @@ int security_inode_listxattr(struct dentry *dentry) * * Return: Returns 0 if permission is granted. */ -int security_inode_removexattr(struct mnt_idmap *idmap, +int security_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { int rc; @@ -2197,7 +2222,7 @@ int security_inode_need_killpriv(struct dentry *dentry) * Return: Return 0 on success. If error is returned, then the operation * causing setuid bit removal is failed. */ -int security_inode_killpriv(struct mnt_idmap *idmap, +int security_inode_killpriv(const struct mnt_idmap *idmap, struct dentry *dentry) { return call_int_hook(inode_killpriv, idmap, dentry); @@ -2219,7 +2244,7 @@ int security_inode_killpriv(struct mnt_idmap *idmap, * * Return: Returns size of buffer on success. */ -int security_inode_getsecurity(struct mnt_idmap *idmap, +int security_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) { diff --git a/security/selinux/hooks.c b/security/selinux/hooks.c index 3f4a6ddee322..73496690ab52 100644 --- a/security/selinux/hooks.c +++ b/security/selinux/hooks.c @@ -2165,6 +2165,36 @@ static int selinux_ptrace_traceme(struct task_struct *parent) SECCLASS_PROCESS, PROCESS__PTRACE, NULL); } +/** + * selinux_mem_foll_force() - Determine whether /proc/$pid/mem can use FOLL_FORCE + * @subject: credentials using which /proc/$pid/mem was opened + * @opened_by_owner: whether checks on open() were bypassed because the opener + * has the same MM as the target + * + * Decide whether it should be possible to read non-readable VMAs and write + * non-writable VMAs via /proc/self/mem. + * The @opened_by_owner case only applies to systems configured with + * PROC_MEM_FORCE_ALWAYS, and only happens on accesses that are not visible to + * selinux_ptrace_access_check() because of the introspection exceptions in + * may_access_mm() and __ptrace_may_access(). + * + * This allows a process to overwrite read-only code in its own address space. + * + * Creating an audit record on denial doesn't make sense here, since we can't + * tell whether FOLL_FORCE matters for the accessed VMAs. + */ +static int selinux_mem_foll_force(const struct cred *subject, bool opened_by_owner) +{ + struct av_decision avd; + u32 sid; + + if (!opened_by_owner) + return 0; + sid = cred_sid(subject); + + return avc_has_perm_noaudit(sid, sid, SECCLASS_PROCESS, PROCESS__PTRACE, 0, &avd); +} + static int selinux_capget(const struct task_struct *target, kernel_cap_t *effective, kernel_cap_t *inheritable, kernel_cap_t *permitted) { @@ -3303,7 +3333,7 @@ static int selinux_inode_permission(struct inode *inode, int requested) return rc; } -static int selinux_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int selinux_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { const struct cred *cred = current_cred(); @@ -3373,7 +3403,7 @@ static int selinux_inode_xattr_skipcap(const char *name) return !strcmp(name, XATTR_NAME_SELINUX); } -static int selinux_inode_setxattr(struct mnt_idmap *idmap, +static int selinux_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -3459,20 +3489,20 @@ static int selinux_inode_setxattr(struct mnt_idmap *idmap, &ad); } -static int selinux_inode_set_acl(struct mnt_idmap *idmap, +static int selinux_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { return dentry_has_perm(current_cred(), dentry, FILE__SETATTR); } -static int selinux_inode_get_acl(struct mnt_idmap *idmap, +static int selinux_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return dentry_has_perm(current_cred(), dentry, FILE__GETATTR); } -static int selinux_inode_remove_acl(struct mnt_idmap *idmap, +static int selinux_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { return dentry_has_perm(current_cred(), dentry, FILE__SETATTR); @@ -3532,7 +3562,7 @@ static int selinux_inode_listxattr(struct dentry *dentry) return dentry_has_perm(cred, dentry, FILE__GETATTR); } -static int selinux_inode_removexattr(struct mnt_idmap *idmap, +static int selinux_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { /* if not a selinux xattr, only check the ordinary setattr perm */ @@ -3612,7 +3642,7 @@ static int selinux_path_notify(const struct path *path, u64 mask, * * Permission check is handled by selinux_inode_getxattr hook. */ -static int selinux_inode_getsecurity(struct mnt_idmap *idmap, +static int selinux_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) { @@ -7664,6 +7694,7 @@ static struct security_hook_list selinux_hooks[] __ro_after_init = { LSM_HOOK_INIT(ptrace_access_check, selinux_ptrace_access_check), LSM_HOOK_INIT(ptrace_traceme, selinux_ptrace_traceme), + LSM_HOOK_INIT(mem_foll_force, selinux_mem_foll_force), LSM_HOOK_INIT(capget, selinux_capget), LSM_HOOK_INIT(capset, selinux_capset), LSM_HOOK_INIT(capable, selinux_capable), diff --git a/security/selinux/selinuxfs.c b/security/selinux/selinuxfs.c index c7d91476971c..a0be7f1b5993 100644 --- a/security/selinux/selinuxfs.c +++ b/security/selinux/selinuxfs.c @@ -1782,7 +1782,7 @@ static struct dentry *sel_make_dir(struct dentry *dir, const char *name, return sel_attach(dir, name, inode); } -static int reject_all(struct mnt_idmap *idmap, struct inode *inode, int mask) +static int reject_all(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return -EPERM; // no access for anyone, root or no root. } diff --git a/security/smack/smack_lsm.c b/security/smack/smack_lsm.c index 8e88ac65fd7f..bb78569b3d5b 100644 --- a/security/smack/smack_lsm.c +++ b/security/smack/smack_lsm.c @@ -1270,7 +1270,7 @@ static int smack_inode_permission(struct inode *inode, int mask) * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int smack_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct smk_audit_info ad; @@ -1348,7 +1348,7 @@ static int smack_inode_xattr_skipcap(const char *name) * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_setxattr(struct mnt_idmap *idmap, +static int smack_inode_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -1482,7 +1482,7 @@ static int smack_inode_getxattr(struct dentry *dentry, const char *name) * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_removexattr(struct mnt_idmap *idmap, +static int smack_inode_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct inode_smack *isp; @@ -1543,7 +1543,7 @@ static int smack_inode_removexattr(struct mnt_idmap *idmap, * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_set_acl(struct mnt_idmap *idmap, +static int smack_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { @@ -1566,7 +1566,7 @@ static int smack_inode_set_acl(struct mnt_idmap *idmap, * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_get_acl(struct mnt_idmap *idmap, +static int smack_inode_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { struct smk_audit_info ad; @@ -1588,7 +1588,7 @@ static int smack_inode_get_acl(struct mnt_idmap *idmap, * * Returns 0 if access is permitted, an error code otherwise */ -static int smack_inode_remove_acl(struct mnt_idmap *idmap, +static int smack_inode_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { struct smk_audit_info ad; @@ -1612,7 +1612,7 @@ static int smack_inode_remove_acl(struct mnt_idmap *idmap, * * Returns the size of the attribute or an error code */ -static int smack_inode_getsecurity(struct mnt_idmap *idmap, +static int smack_inode_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void **buffer, bool alloc) { diff --git a/tools/include/uapi/linux/coredump.h b/tools/include/uapi/linux/coredump.h index dc3789b78af0..6d0c53b534ea 100644 --- a/tools/include/uapi/linux/coredump.h +++ b/tools/include/uapi/linux/coredump.h @@ -11,12 +11,53 @@ * @COREDUMP_USERSPACE: userspace writes coredump * @COREDUMP_REJECT: don't generate coredump * @COREDUMP_WAIT: wait for coredump server + * @COREDUMP_RECORDS: send the coredump as a sequence of records instead of + * as a plain byte stream, see struct coredump_record_header; + * requires COREDUMP_KERNEL + * @COREDUMP_SPARSE: describe the holes in the coredump as zero records + * instead of transferring them; requires COREDUMP_RECORDS + * @COREDUMP_MEMORY_TYPES: dump the memory types in + * coredump_ack->memory_types instead of the ones + * the task selected; requires COREDUMP_KERNEL */ enum { COREDUMP_KERNEL = (1ULL << 0), COREDUMP_USERSPACE = (1ULL << 1), COREDUMP_REJECT = (1ULL << 2), COREDUMP_WAIT = (1ULL << 3), + COREDUMP_RECORDS = (1ULL << 4), + COREDUMP_SPARSE = (1ULL << 5), + COREDUMP_MEMORY_TYPES = (1ULL << 6), +}; + +/** + * coredump memory types + * @COREDUMP_MEMORY_ANON_PRIVATE: anonymous private memory + * @COREDUMP_MEMORY_ANON_SHARED: anonymous shared memory + * @COREDUMP_MEMORY_FILE_PRIVATE: file-backed private memory + * @COREDUMP_MEMORY_FILE_SHARED: file-backed shared memory + * @COREDUMP_MEMORY_ELF_HEADERS: the first page of a file-backed private + * mapping that starts an ELF file + * @COREDUMP_MEMORY_HUGETLB_PRIVATE: hugetlb private memory + * @COREDUMP_MEMORY_HUGETLB_SHARED: hugetlb shared memory + * @COREDUMP_MEMORY_DAX_PRIVATE: DAX private memory + * @COREDUMP_MEMORY_DAX_SHARED: DAX shared memory + * + * A bitmask of memory types a coredump may request to be included. New + * memory type bits must ensure that they do not steal memory from an + * existing one so a coredump server will continue to get the same + * coredumps even if a new bit is introduced. + */ +enum { + COREDUMP_MEMORY_ANON_PRIVATE = (1ULL << 0), + COREDUMP_MEMORY_ANON_SHARED = (1ULL << 1), + COREDUMP_MEMORY_FILE_PRIVATE = (1ULL << 2), + COREDUMP_MEMORY_FILE_SHARED = (1ULL << 3), + COREDUMP_MEMORY_ELF_HEADERS = (1ULL << 4), + COREDUMP_MEMORY_HUGETLB_PRIVATE = (1ULL << 5), + COREDUMP_MEMORY_HUGETLB_SHARED = (1ULL << 6), + COREDUMP_MEMORY_DAX_PRIVATE = (1ULL << 7), + COREDUMP_MEMORY_DAX_SHARED = (1ULL << 8), }; /** @@ -24,17 +65,19 @@ enum { * @size: size of struct coredump_req * @size_ack: known size of struct coredump_ack on this kernel * @mask: supported features + * @memory_types: the memory types the task selected + * @memory_types_mask: the memory types this kernel knows * * When a coredump happens the kernel will connect to the coredump * socket and send a coredump request to the coredump server. The @size * member is set to the size of struct coredump_req and provides a hint * to userspace how much data can be read. Userspace may use MSG_PEEK to * peek the size of struct coredump_req and then choose to consume it in - * one go. Userspace may also simply read a COREDUMP_ACK_SIZE_VER0 + * one go. Userspace may also simply read a COREDUMP_REQ_SIZE_VER0 * request. If the size the kernel sends is larger userspace simply * discards any remaining data. * - * The coredump_req->mask member is set to the currently know features. + * The coredump_req->mask member is set to the currently known features. * Userspace may only set coredump_ack->mask to the bits raised by the * kernel in coredump_req->mask. * @@ -42,15 +85,27 @@ enum { * struct coredump_ack the kernel knows. Userspace may only send up to * coredump_req->size_ack bytes to the kernel and must set * coredump_ack->size accordingly. + * + * @memory_types is set to the default memory types that are included in + * the coredump. This can be overridden by raising bits in + * coredump_ack->memory_types. + * + * @memory_types_mask contains a bitmask of all memory types the kernel + * knows about. A coredump server may only raise bits in + * coredump_ack->memory_types that are raised in + * coredump_req->memory_types_mask. */ struct coredump_req { __u32 size; __u32 size_ack; __u64 mask; + __u64 memory_types; + __u64 memory_types_mask; }; enum { COREDUMP_REQ_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_REQ_SIZE_VER1 = 32U, /* memory_types and memory_types_mask added */ }; /** @@ -58,6 +113,8 @@ enum { * @size: size of the struct * @spare: unused * @mask: features kernel is supposed to use + * @memory_types: memory types to dump, only with COREDUMP_MEMORY_TYPES + * in @mask * * The @size member must be set to the size of struct coredump_ack. It * may never exceed what the kernel returned in coredump_req->size_ack @@ -67,15 +124,30 @@ enum { * The @mask member must be set to the features the coredump server * wants the kernel to use. Only bits the kernel returned in * coredump_req->mask may be set. + * + * If COREDUMP_MEMORY_TYPES is raised in @mask the kernel dumps the + * memory types set in the @memory_types mask. Zero is valid and dumps + * no memory apart from the mappings that are always dumped. + * + * Note that memory a task excluded via MADV_DONTDUMP is always left + * out. A coredump server wanting to add or drop memory types instead of + * outright replacing it should simply copy coredump_req->memory_types + * and then mask off or raise types as needed. + * + * Note that @memory_types must be zero if COREDUMP_MEMORY_TYPES isn't + * raised. COREDUMP_MEMORY_TYPES requires COREDUMP_KERNEL and an ack of + * at least COREDUMP_ACK_SIZE_VER1 bytes. */ struct coredump_ack { __u32 size; __u32 spare; __u64 mask; + __u64 memory_types; }; enum { COREDUMP_ACK_SIZE_VER0 = 16U, /* size of first published struct */ + COREDUMP_ACK_SIZE_VER1 = 24U, /* memory_types added */ }; /** @@ -83,11 +155,12 @@ enum { * * The kernel will place a single byte on the coredump socket. The * markers notify userspace whether the coredump ack succeeded or - * failed. + * failed. After any marker other than COREDUMP_MARK_REQACK the kernel + * closes the connection and no coredump is generated. * * @COREDUMP_MARK_MINSIZE: the provided coredump_ack size was too small * @COREDUMP_MARK_MAXSIZE: the provided coredump_ack size was too big - * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask was invalid + * @COREDUMP_MARK_UNSUPPORTED: the provided coredump_ack mask or memory types were invalid * @COREDUMP_MARK_CONFLICTING: the provided coredump_ack mask has conflicting options * @COREDUMP_MARK_REQACK: the coredump request and ack was successful * @__COREDUMP_MARK_MAX: the maximum coredump mark value @@ -101,4 +174,72 @@ enum coredump_mark { __COREDUMP_MARK_MAX = (1U << 31), }; +/** + * enum coredump_record_type - Type of a coredump record + * + * @COREDUMP_RECORD_DATA: the header is followed by ->len bytes of data + * @COREDUMP_RECORD_END: the coredump ends here, the header is not followed + * by any data and no further record is sent + * @COREDUMP_RECORD_ZERO: the header stands for ->len zero bytes and is not + * followed by any data + * @__COREDUMP_RECORD_TYPE_MAX: the maximum coredump record type value + */ +enum coredump_record_type { + COREDUMP_RECORD_DATA = 0U, + COREDUMP_RECORD_END = 1U, + COREDUMP_RECORD_ZERO = 2U, + __COREDUMP_RECORD_TYPE_MAX = (1U << 31), +}; + +/** + * struct coredump_record_header - header of a coredump record + * @size: size of struct coredump_record_header + * @type: one of enum coredump_record_type + * @flags: modifiers for this record + * @offset: offset in the coredump this record starts at + * @len: number of coredump bytes this record accounts for + * + * If the coredump server raises COREDUMP_RECORDS in coredump_ack->mask + * the kernel doesn't send the coredump as a plain byte stream. It sends + * a sequence of records instead. A COREDUMP_RECORD_DATA record is + * followed by @len bytes of actual coredump data. A + * COREDUMP_RECORD_ZERO record is followed by nothing and stands for + * @len zero bytes. A server that didn't raise COREDUMP_SPARSE never + * sees a zero record. Records arrive in order and leave no gaps. So + * @offset is the sum of the @len of all records before it. + * + * The last record is a COREDUMP_RECORD_END record. It is followed by + * nothing. Its @len is zero. Its @offset is the size of the coredump. + * The kernel only sends it once it has written the whole coredump. A + * server that hits end-of-file without having seen an end record must + * treat the coredump as incomplete. + * + * The @size member is set to the size of struct coredump_record_header + * the kernel knows and lets the header grow later. It comes first so it + * can be peeked. Userspace must consume @size bytes and discard + * anything beyond what it knows. It must refuse a @size smaller than + * COREDUMP_RECORD_HEADER_SIZE_VER0. @size covers the header alone. + * @offset and @len count coredump bytes. + * + * The @flags member carries modifiers that change how the record is to + * be interpreted. No flag is defined yet. Userspace must refuse a + * record carrying a flag or a type it doesn't know. Every new record + * type is raised in coredump_req->mask as a feature of its own. A + * server only ever sees the types it asked for. + * + * COREDUMP_RECORDS must be combined with COREDUMP_KERNEL, and + * COREDUMP_SPARSE with COREDUMP_RECORDS. + */ +struct coredump_record_header { + __u32 size; + __u32 type; + __u64 flags; + __u64 offset; + __u64 len; +}; + +enum { + COREDUMP_RECORD_HEADER_SIZE_VER0 = 32U, /* size of first published struct */ +}; + #endif /* _UAPI_LINUX_COREDUMP_H */ diff --git a/tools/testing/selftests/Makefile b/tools/testing/selftests/Makefile index 76279eea7749..79a00e9ee46d 100644 --- a/tools/testing/selftests/Makefile +++ b/tools/testing/selftests/Makefile @@ -39,6 +39,7 @@ TARGETS += filesystems/epoll TARGETS += filesystems/eventfd TARGETS += filesystems/failfs TARGETS += filesystems/fat +TARGETS += filesystems/file_stressor TARGETS += filesystems/openat2 TARGETS += filesystems/open_tree_ns TARGETS += filesystems/overlayfs @@ -51,6 +52,8 @@ TARGETS += filesystems/empty_mntns TARGETS += filesystems/fsmount_ns TARGETS += filesystems/fscontext_ns TARGETS += filesystems/xattr +TARGETS += filesystems/mntns_unbindable +TARGETS += filesystems/umount_propagation TARGETS += firmware TARGETS += fpu TARGETS += ftrace diff --git a/tools/testing/selftests/clone3/clone3_clear_sighand.c b/tools/testing/selftests/clone3/clone3_clear_sighand.c index de0c9d62015d..0ad62b80ada2 100644 --- a/tools/testing/selftests/clone3/clone3_clear_sighand.c +++ b/tools/testing/selftests/clone3/clone3_clear_sighand.c @@ -50,12 +50,12 @@ static void test_clone3_clear_sighand(void) * Check that CLONE_CLEAR_SIGHAND and CLONE_SIGHAND are mutually * exclusive. */ - args.flags |= CLONE_CLEAR_SIGHAND | CLONE_SIGHAND; + args.flags |= CLONE_VM | CLONE_CLEAR_SIGHAND | CLONE_SIGHAND; args.exit_signal = SIGCHLD; pid = sys_clone3(&args, sizeof(args)); - if (pid > 0) + if (pid != -1 || errno != EINVAL) ksft_exit_fail_msg( - "clone3(CLONE_CLEAR_SIGHAND | CLONE_SIGHAND) succeeded\n"); + "clone3(CLONE_CLEAR_SIGHAND | CLONE_SIGHAND) did not fail with EINVAL\n"); act.sa_handler = nop_handler; ret = sigemptyset(&act.sa_mask); diff --git a/tools/testing/selftests/core/close_range_test.c b/tools/testing/selftests/core/close_range_test.c index f14eca63f20c..20ecb65e529b 100644 --- a/tools/testing/selftests/core/close_range_test.c +++ b/tools/testing/selftests/core/close_range_test.c @@ -36,6 +36,14 @@ static inline int sys_close_range(unsigned int fd, unsigned int max_fd, return syscall(__NR_close_range, fd, max_fd, flags); } +static void clear_cloexec(const int *fds, size_t n) +{ + size_t i; + + for (i = 0; i < n; i++) + fcntl(fds[i], F_SETFD, 0); +} + TEST(core_close_range) { int i, ret; @@ -236,6 +244,50 @@ TEST(close_range_unshare_capped) EXPECT_EQ(0, WEXITSTATUS(status)); } +TEST(close_range_unshare_hole) +{ + int i, status; + pid_t pid; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + /* Fill the first two words of the table. */ + for (i = 3; i < 128; i++) + ASSERT_GE(dup2(0, i), 0); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + /* Punch a hole into the second word, behind a full first one. */ + if (sys_close_range(70, 80, CLOSE_RANGE_UNSHARE)) + exit(EXIT_FAILURE); + + for (i = 3; i < 128; i++) { + bool closed = i >= 70 && i <= 80; + + if (closed == (fcntl(i, F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* A stale full bit on word 1 would hand out 128, not 70. */ + if (dup(0) != 70) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 3; i < 128; i++) + EXPECT_NE(-1, fcntl(i, F_GETFD)); +} + TEST(close_range_cloexec) { int i, ret; @@ -593,6 +645,905 @@ TEST(close_range_cloexec_unshare_syzbot) EXPECT_EQ(close(fd3), 0); } +TEST(close_range_except) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + /* The bounds are checked before the range is turned around. */ + EXPECT_EQ(-1, sys_close_range(open_fds[20], open_fds[10], + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); + + /* Everything above open_fds[50] goes. */ + ASSERT_EQ(0, sys_close_range(0, open_fds[50], CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i <= 50, fcntl(open_fds[i], F_GETFD) != -1); + + /* A window in the middle takes stdio with it, so do that in a fork. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i <= 50; i++) { + bool kept = i >= 10 && i <= 20; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The fork had a table of its own. */ + for (i = 0; i <= 50; i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_except_cloexec) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool inside = i >= 10 && i <= 20; + int flags = fcntl(open_fds[i], F_GETFD); + + EXPECT_NE(-1, flags); + EXPECT_EQ(inside ? 0 : FD_CLOEXEC, flags & FD_CLOEXEC); + } + + /* stdio sits outside of the window too. */ + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* A window that starts at 0 marks only what lies above it. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0)); + ASSERT_EQ(0, sys_close_range(0, open_fds[20], + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i <= 20 ? 0 : FD_CLOEXEC, + fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(0, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* One at the top marks only what lies below it, stdio included. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, sys_close_range(open_fds[80], UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(i < 80 ? FD_CLOEXEC : 0, + fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); + + /* One that cannot hold a descriptor marks everything. */ + clear_cloexec(open_fds, ARRAY_SIZE(open_fds)); + ASSERT_EQ(0, fcntl(STDERR_FILENO, F_SETFD, 0)); + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(FD_CLOEXEC, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + EXPECT_EQ(FD_CLOEXEC, fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC); +} + +TEST(close_range_except_cloexec_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything marks nothing. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool inside = i >= 10 && i <= 20; + int flags = fcntl(open_fds[i], F_GETFD); + + if (flags == -1) + exit(EXIT_FAILURE); + if ((flags & FD_CLOEXEC) != (inside ? 0 : FD_CLOEXEC)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(0, fcntl(open_fds[i], F_GETFD) & FD_CLOEXEC); +} + +TEST(close_range_except_bounds) +{ + int i, c, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + struct { + unsigned int fd, max_fd, flags; + } cases[] = { + /* A window at the top drops everything below it. */ + { open_fds[50], UINT_MAX, CLOSE_RANGE_EXCEPT }, + /* One that cannot hold a descriptor keeps nothing. */ + { UINT_MAX, UINT_MAX, CLOSE_RANGE_EXCEPT }, + /* One of a single descriptor keeps just that. */ + { open_fds[30], open_fds[30], CLOSE_RANGE_EXCEPT }, + /* The unshare form on a table that is not shared acts in place. */ + { open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT }, + }; + + /* Each of them takes stdio with it, so do that in a fork. */ + for (c = 0; c < ARRAY_SIZE(cases); c++) { + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(cases[c].fd, cases[c].max_fd, + cases[c].flags); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + unsigned int fd = open_fds[i]; + bool kept = fd >= cases[c].fd && + fd <= cases[c].max_fd; + + if (kept != (fcntl(fd, F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + } + + /* Each fork had a table of its own. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_except_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[200]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + /* Odd slots are close-on-exec, which makes no difference here. */ + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_EXCEPT"); + } + ASSERT_EQ(0, ret); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + /* The window sits near the top, so the clone is sized off its end. */ + ret = sys_close_range(open_fds[150], open_fds[160], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = i >= 150 && i <= 160; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + /* What was left behind is handed out again, from the bottom. */ + if (dup(open_fds[150]) != 0) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window at the bottom keeps just that, stdio included. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(0, open_fds[10], + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + if ((i <= 10) != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + /* The first slot left behind is the next one handed out. */ + if (dup(0) != open_fds[10] + 1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window that cannot hold a descriptor keeps nothing. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_UNSHARE | CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if (fcntl(open_fds[i], F_GETFD) != -1) + exit(EXIT_FAILURE); + + if (fcntl(STDERR_FILENO, F_GETFD) != -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + /* Odd slots are close-on-exec, even ones are not. */ + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + EXPECT_EQ(!closed, fcntl(open_fds[i], F_GETFD) != -1); + } + + /* A range above the table closes nothing. */ + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + EXPECT_EQ(!closed, fcntl(open_fds[i], F_GETFD) != -1); + } + + /* Do what an exec would do to the rest. */ + ASSERT_EQ(0, sys_close_range(0, UINT_MAX, CLOSE_RANGE_CLOEXEC_ONLY)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(!(i % 2), fcntl(open_fds[i], F_GETFD) != -1); +} + +TEST(close_range_cloexec_only_except) +{ + int i, ret; + int open_fds[101]; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = !(i % 2) || (i >= 10 && i <= 20); + int flags = i % 2 ? FD_CLOEXEC : 0; + + /* The kept ones keep their flag, so exec still drops them. */ + EXPECT_EQ(kept ? flags : -1, fcntl(open_fds[i], F_GETFD)); + } + + /* A range that cannot hold an open descriptor keeps nothing. */ + ASSERT_EQ(0, sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_EQ(!(i % 2), fcntl(open_fds[i], F_GETFD) != -1); +} + +TEST(close_range_cloexec_only_except_bounds) +{ + int i, c, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + struct { + unsigned int fd, max_fd; + } cases[] = { + /* A window at the top keeps the marked ones in it. */ + { open_fds[80], UINT_MAX }, + /* One at the bottom keeps the marked ones in it. */ + { 0, open_fds[20] }, + /* One that cannot hold a descriptor keeps none of them. */ + { UINT_MAX, UINT_MAX }, + }; + + for (c = 0; c < ARRAY_SIZE(cases); c++) { + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(cases[c].fd, cases[c].max_fd, + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + unsigned int fd = open_fds[i]; + bool kept = !(i % 2) || (fd >= cases[c].fd && + fd <= cases[c].max_fd); + int flags = i % 2 ? FD_CLOEXEC : 0; + + if (fcntl(fd, F_GETFD) != (kept ? flags : -1)) + exit(EXIT_FAILURE); + } + + /* stdio is neither marked nor gone. */ + if (fcntl(STDERR_FILENO, F_GETFD) & FD_CLOEXEC) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + } + + /* Each fork had a table of its own. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + ASSERT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 10 && i <= 20; + + if (closed == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A range at the top keeps the descriptors without the flag in it. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[50], UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool closed = i % 2 && i >= 50; + + if (closed == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* The first slot left behind is the next one handed out. */ + if (dup(0) != open_fds[51]) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_except_unshare) +{ + int i, ret, status; + pid_t pid; + int open_fds[101]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | (i % 2 ? O_CLOEXEC : 0)); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + /* A range covering everything keeps everything. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + ASSERT_NE(-1, fcntl(open_fds[i], F_GETFD)); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[10], open_fds[20], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = !(i % 2) || (i >= 10 && i <= 20); + int flags = i % 2 ? FD_CLOEXEC : 0; + + if (fcntl(open_fds[i], F_GETFD) != (kept ? flags : -1)) + exit(EXIT_FAILURE); + } + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* A window that cannot hold a descriptor keeps none of the marked. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(UINT_MAX, UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if ((i % 2) == (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* One that covers everything keeps everything, in a clone too. */ + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + if (fcntl(open_fds[i], F_GETFD) != (i % 2 ? FD_CLOEXEC : 0)) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); + + /* The shared table the child unshared from is untouched. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) + EXPECT_NE(-1, fcntl(open_fds[i], F_GETFD)); +} + +TEST(close_range_cloexec_only_except_unshare_sizing) +{ + int i, ret, status; + pid_t pid; + int open_fds[200]; + struct __clone_args args = { + .flags = CLONE_FILES, + .exit_signal = SIGCHLD, + }; + + /* All close-on-exec, so the kept range alone sizes the clone. */ + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + int fd; + + fd = open("/dev/null", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0) { + if (errno == ENOENT) + SKIP(return, "Skipping test since /dev/null does not exist"); + } + + open_fds[i] = fd; + } + + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + pid = sys_clone3(&args, sizeof(args)); + ASSERT_GE(pid, 0); + + if (pid == 0) { + ret = sys_close_range(open_fds[150], open_fds[160], + CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT); + if (ret) + exit(EXIT_FAILURE); + + for (i = 0; i < ARRAY_SIZE(open_fds); i++) { + bool kept = i >= 150 && i <= 160; + + if (kept != (fcntl(open_fds[i], F_GETFD) != -1)) + exit(EXIT_FAILURE); + } + + /* Nothing set close-on-exec on stdio. */ + if (fcntl(STDERR_FILENO, F_GETFD) == -1) + exit(EXIT_FAILURE); + + exit(EXIT_SUCCESS); + } + + EXPECT_EQ(waitpid(pid, &status, 0), pid); + EXPECT_EQ(true, WIFEXITED(status)); + EXPECT_EQ(0, WEXITSTATUS(status)); +} + +TEST(close_range_cloexec_only_einval) +{ + int ret; + + /* A range covering everything keeps everything, so this only probes. */ + ret = sys_close_range(0, UINT_MAX, + CLOSE_RANGE_CLOEXEC_ONLY | CLOSE_RANGE_EXCEPT); + if (ret < 0) { + if (errno == ENOSYS) + SKIP(return, "close_range() syscall not supported"); + if (errno == EINVAL) + SKIP(return, "close_range() doesn't support CLOSE_RANGE_CLOEXEC_ONLY"); + } + ASSERT_EQ(0, ret); + + EXPECT_EQ(-1, sys_close_range(3, UINT_MAX, CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY)); + EXPECT_EQ(EINVAL, errno); + + /* The other flags do not make the pair acceptable. */ + EXPECT_EQ(-1, sys_close_range(3, UINT_MAX, CLOSE_RANGE_UNSHARE | + CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); + + /* The bounds are checked with the new flag too. */ + EXPECT_EQ(-1, sys_close_range(4, 3, CLOSE_RANGE_CLOEXEC_ONLY | + CLOSE_RANGE_EXCEPT)); + EXPECT_EQ(EINVAL, errno); +} + TEST(close_range_bitmap_corruption) { pid_t pid; diff --git a/tools/testing/selftests/coredump/.gitignore b/tools/testing/selftests/coredump/.gitignore index 097f52db0be9..e32f6e9006f6 100644 --- a/tools/testing/selftests/coredump/.gitignore +++ b/tools/testing/selftests/coredump/.gitignore @@ -2,3 +2,5 @@ stackdump_test coredump_socket_test coredump_socket_protocol_test +coredump_signal_test +coredump_worker_test diff --git a/tools/testing/selftests/coredump/Makefile b/tools/testing/selftests/coredump/Makefile index dece1a31d561..c4cdab7b7d0b 100644 --- a/tools/testing/selftests/coredump/Makefile +++ b/tools/testing/selftests/coredump/Makefile @@ -3,7 +3,12 @@ CFLAGS += -Wall -O0 -g $(KHDR_INCLUDES) $(TOOLS_INCLUDES) TEST_GEN_PROGS := stackdump_test \ coredump_socket_test \ - coredump_socket_protocol_test + coredump_socket_protocol_test \ + coredump_notify_signal_test \ + coredump_signal_test \ + coredump_worker_test +# Spawned by the kernel as the |helper, not a test of its own. +TEST_GEN_FILES := coredump_notify_signal_helper TEST_FILES := stackdump include ../lib.mk @@ -11,3 +16,7 @@ include ../lib.mk $(OUTPUT)/stackdump_test: coredump_test_helpers.c $(OUTPUT)/coredump_socket_test: coredump_test_helpers.c $(OUTPUT)/coredump_socket_protocol_test: coredump_test_helpers.c +$(OUTPUT)/coredump_notify_signal_test: coredump_test_helpers.c +$(OUTPUT)/coredump_notify_signal_helper: coredump_test_helpers.c +$(OUTPUT)/coredump_signal_test: coredump_test_helpers.c +$(OUTPUT)/coredump_worker_test: coredump_test_helpers.c diff --git a/tools/testing/selftests/coredump/coredump_notify_signal.h b/tools/testing/selftests/coredump/coredump_notify_signal.h new file mode 100644 index 000000000000..92868a43425c --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_notify_signal.h @@ -0,0 +1,29 @@ +/* SPDX-License-Identifier: GPL-2.0 */ + +#ifndef __COREDUMP_NOTIFY_SIGNAL_H +#define __COREDUMP_NOTIFY_SIGNAL_H + +#include <stdbool.h> +#include <sys/types.h> + +/* + * Define a bunch of constants we need. We create a situation where the + * NT_FILE note blows past 200K. That's way beyond the default 64K + * pipe ring and past the ~36K an af_unix skb holds. So we force a write + * to come back short. + */ +#define NOTIFY_SIGNAL_MAP_COUNT 4000 +#define NOTIFY_SIGNAL_ANON_BYTES (4UL << 20) +#define NOTIFY_SIGNAL_STALL_US 200000 +#define NOTIFY_SIGNAL_MAPFILE "/tmp/coredump.notify_signal.mapfile" +#define NOTIFY_SIGNAL_TRIGGER "/tmp/coredump.notify_signal.trigger" +#define NOTIFY_SIGNAL_CORE_FILE "/tmp/coredump.notify_signal.core" +#define NOTIFY_SIGNAL_CORE_TMPFILE "/tmp/coredump.notify_signal.core.tmp" +#define NOTIFY_SIGNAL_SOCKET "/tmp/coredump.notify_signal.socket" + +void crashing_child_notify_signal(void); +bool coredump_io_uring_available(void); +ssize_t recv_coredump_notify_signal(int fd, int fd_out, bool arm); +long long coredump_expected_size(const char *path); + +#endif /* __COREDUMP_NOTIFY_SIGNAL_H */ diff --git a/tools/testing/selftests/coredump/coredump_notify_signal_helper.c b/tools/testing/selftests/coredump/coredump_notify_signal_helper.c new file mode 100644 index 000000000000..849f5c1ea736 --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_notify_signal_helper.c @@ -0,0 +1,46 @@ +// SPDX-License-Identifier: GPL-2.0 + +/* + * The |helper half of coredump_notify_signal_test. The kernel spawns this + * with the coredump on stdin, so it cannot be part of the test binary. + * It saves the dump and, once the notes have started, trips the fifo the + * crashing task is polling so TIF_NOTIFY_SIGNAL is raised while the note + * write is in flight. + */ + +#include <fcntl.h> +#include <stdio.h> +#include <stdlib.h> +#include <unistd.h> + +#include "coredump_notify_signal.h" + +int main(int argc, char *argv[]) +{ + int fd_core_file; + ssize_t ret; + + fd_core_file = open(NOTIFY_SIGNAL_CORE_TMPFILE, + O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, 0600); + if (fd_core_file < 0) { + fprintf(stderr, "%s: open failed: %m\n", argv[0]); + return EXIT_FAILURE; + } + + ret = recv_coredump_notify_signal(STDIN_FILENO, fd_core_file, true); + close(fd_core_file); + if (ret < 0) + goto err; + + /* The test polls for this name, so only create it once it is whole. */ + if (rename(NOTIFY_SIGNAL_CORE_TMPFILE, NOTIFY_SIGNAL_CORE_FILE)) { + fprintf(stderr, "%s: rename failed: %m\n", argv[0]); + goto err; + } + + return EXIT_SUCCESS; + +err: + unlink(NOTIFY_SIGNAL_CORE_TMPFILE); + return EXIT_FAILURE; +} diff --git a/tools/testing/selftests/coredump/coredump_notify_signal_test.c b/tools/testing/selftests/coredump/coredump_notify_signal_test.c new file mode 100644 index 000000000000..4a98ab141c41 --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_notify_signal_test.c @@ -0,0 +1,245 @@ +// SPDX-License-Identifier: GPL-2.0 + +/* + * A coredump is meant to be interrupted by SIGKILL and by the freezer and + * by nothing else. dump_interrupted() says so, but the blocking waits + * underneath it test signal_pending(), which is also true for + * TIF_NOTIFY_SIGNAL. A crashing task that has an io_uring completion land + * on it mid-dump therefore keeps dumping while every wait it enters bails + * out at once, and the dump is silently cut short. Nothing reports it: + * binfmt_elf sets has_dumped before it writes anything, so WCOREDUMP() + * says the dump worked. + * + * A coredump note is the only dump_emit() that exceeds what the transport + * takes in one go, so it is the one write that is certain to block. Both + * tests crash a child holding enough file backed mappings for its NT_FILE + * note to run past that, arm an io_uring poll on it, and trip the poll + * while the note is being written. What comes out has to be the whole + * dump. + */ + +#include <fcntl.h> +#include <limits.h> +#include <sys/socket.h> +#include <sys/stat.h> +#include <sys/un.h> +#include <sys/wait.h> +#include <unistd.h> + +#include "coredump_test.h" + +FIXTURE_SETUP(coredump) +{ + FILE *file; + int ret; + + self->pid_coredump_server = -ESRCH; + self->fd_tmpfs_detached = -1; + file = fopen("/proc/sys/kernel/core_pattern", "r"); + ASSERT_NE(NULL, file); + + ret = fread(self->original_core_pattern, 1, + sizeof(self->original_core_pattern), file); + ASSERT_TRUE(ret || feof(file)); + ASSERT_LT(ret, sizeof(self->original_core_pattern)); + + self->original_core_pattern[ret] = '\0'; + self->fd_tmpfs_detached = create_detached_tmpfs(); + ASSERT_GE(self->fd_tmpfs_detached, 0); + + ret = fclose(file); + ASSERT_EQ(0, ret); + + /* A stale core file from a killed previous run would fake a pass. */ + unlink(NOTIFY_SIGNAL_CORE_FILE); + /* And a stale socket would fail the server's bind. */ + unlink(NOTIFY_SIGNAL_SOCKET); + unlink(NOTIFY_SIGNAL_TRIGGER); + ASSERT_EQ(mkfifo(NOTIFY_SIGNAL_TRIGGER, 0600), 0); +} + +FIXTURE_TEARDOWN(coredump) +{ + const char *reason; + FILE *file; + int ret, status; + + if (self->pid_coredump_server > 0) { + kill(self->pid_coredump_server, SIGTERM); + waitpid(self->pid_coredump_server, &status, 0); + } + unlink(NOTIFY_SIGNAL_CORE_FILE); + unlink(NOTIFY_SIGNAL_CORE_TMPFILE); + unlink(NOTIFY_SIGNAL_SOCKET); + unlink(NOTIFY_SIGNAL_TRIGGER); + unlink(NOTIFY_SIGNAL_MAPFILE); + + file = fopen("/proc/sys/kernel/core_pattern", "w"); + if (!file) { + reason = "Unable to open core_pattern"; + goto fail; + } + + ret = fprintf(file, "%s", self->original_core_pattern); + if (ret < 0) { + reason = "Unable to write to core_pattern"; + goto fail; + } + + ret = fclose(file); + if (ret) { + reason = "Unable to close core_pattern"; + goto fail; + } + + if (self->fd_tmpfs_detached >= 0) { + ret = close(self->fd_tmpfs_detached); + if (ret < 0) { + reason = "Unable to close detached tmpfs"; + goto fail; + } + self->fd_tmpfs_detached = -1; + } + + return; +fail: + /* This should never happen */ + fprintf(stderr, "Failed to cleanup coredump test: %s\n", reason); +} + +/* Check that what the helper or the server saved is the whole dump. */ +static void check_whole_coredump(struct __test_metadata *const _metadata) +{ + long long expected; + struct stat st; + + expected = coredump_expected_size(NOTIFY_SIGNAL_CORE_FILE); + ASSERT_GT(expected, 0); + ASSERT_EQ(stat(NOTIFY_SIGNAL_CORE_FILE, &st), 0); + ASSERT_EQ((long long)st.st_size, expected); +} + +/* + * The dump goes to a |helper, so the reader is a separate program the + * kernel spawns. It saves what it received to NOTIFY_SIGNAL_CORE_FILE. + */ +TEST_F(coredump, notify_signal_pipe) +{ + char pattern[PATH_MAX], helper[PATH_MAX], *p; + struct stat st; + int status, i; + pid_t pid; + ssize_t n; + + if (!coredump_io_uring_available()) + SKIP(return, "io_uring not available"); + + n = readlink("/proc/self/exe", helper, sizeof(helper) - 1); + ASSERT_GT(n, 0); + helper[n] = '\0'; + p = strstr(helper, "coredump_notify_signal_test"); + ASSERT_NE(p, NULL); + ASSERT_LE((size_t)(p - helper) + sizeof("coredump_notify_signal_helper"), + sizeof(helper)); + strcpy(p, "coredump_notify_signal_helper"); + if (access(helper, X_OK)) + SKIP(return, "coredump_notify_signal_helper not built"); + + ASSERT_LT(snprintf(pattern, sizeof(pattern), "|%s", helper), + (int)sizeof(pattern)); + ASSERT_TRUE(set_core_pattern(pattern)); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_notify_signal(); + + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFSIGNALED(status)); + + /* The kernel does not wait for the helper, so poll for it. */ + for (i = 0; i < 100; i++) { + if (!stat(NOTIFY_SIGNAL_CORE_FILE, &st) && st.st_size) + break; + usleep(100000); + } + + check_whole_coredump(_metadata); +} + +/* The same thing with the dump going to a coredump socket. */ +TEST_F(coredump, notify_signal_socket) +{ + pid_t pid, pid_coredump_server; + int ipc_sockets[2], status; + char pattern[PATH_MAX]; + char c; + + if (!coredump_io_uring_available()) + SKIP(return, "io_uring not available"); + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, + ipc_sockets), 0); + ASSERT_LT(snprintf(pattern, sizeof(pattern), "@%s", + NOTIFY_SIGNAL_SOCKET), (int)sizeof(pattern)); + ASSERT_TRUE(set_core_pattern(pattern)); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_core_file = -1; + int exit_code = EXIT_FAILURE; + + close(ipc_sockets[0]); + + fd_server = create_and_listen_unix_socket(NOTIFY_SIGNAL_SOCKET); + if (fd_server < 0) + goto out; + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_core_file = open(NOTIFY_SIGNAL_CORE_FILE, + O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC, + 0600); + if (fd_core_file < 0) + goto out; + + if (recv_coredump_notify_signal(fd_coredump, fd_core_file, + true) < 0) + goto out; + + exit_code = EXIT_SUCCESS; +out: + if (fd_core_file >= 0) + close(fd_core_file); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_notify_signal(); + + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFSIGNALED(status)); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + check_whole_coredump(_metadata); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_signal_test.c b/tools/testing/selftests/coredump/coredump_signal_test.c new file mode 100644 index 000000000000..fdf48b144781 --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_signal_test.c @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: GPL-2.0 + +#include <fcntl.h> +#include <pthread.h> +#include <signal.h> +#include <sys/mman.h> +#include <sys/socket.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/un.h> +#include <unistd.h> + +#include "coredump_test.h" + +/* Big enough to fill the socket buffer many times over. */ +#define CRASH_MAPPING_SIZE (32 * 1024 * 1024) + +FIXTURE_SETUP(coredump) +{ + FILE *file; + int ret; + + self->pid_coredump_server = -ESRCH; + self->fd_tmpfs_detached = -1; + file = fopen("/proc/sys/kernel/core_pattern", "r"); + ASSERT_NE(NULL, file); + + ret = fread(self->original_core_pattern, 1, sizeof(self->original_core_pattern), file); + ASSERT_TRUE(ret || feof(file)); + ASSERT_LT(ret, sizeof(self->original_core_pattern)); + + self->original_core_pattern[ret] = '\0'; + + ret = fclose(file); + ASSERT_EQ(0, ret); +} + +FIXTURE_TEARDOWN(coredump) +{ + const char *reason; + FILE *file; + int ret, status; + + if (self->pid_coredump_server > 0) { + kill(self->pid_coredump_server, SIGTERM); + waitpid(self->pid_coredump_server, &status, 0); + } + unlink("/tmp/coredump.file"); + unlink("/tmp/coredump.socket"); + + file = fopen("/proc/sys/kernel/core_pattern", "w"); + if (!file) { + reason = "Unable to open core_pattern"; + goto fail; + } + + ret = fprintf(file, "%s", self->original_core_pattern); + if (ret < 0) { + reason = "Unable to write to core_pattern"; + goto fail; + } + + ret = fclose(file); + if (ret) { + reason = "Unable to close core_pattern"; + goto fail; + } + + return; +fail: + /* This should never happen */ + fprintf(stderr, "Failed to cleanup coredump test: %s\n", reason); +} + +static volatile int waiter_ready; + +static void usr1_handler(int sig) +{ +} + +/* Sleeps in sigtimedwait() with SIGUSR1 unblocked only inside the kernel. */ +static void *sigwaiter(void *arg) +{ + sigset_t set; + siginfo_t info; + + sigemptyset(&set); + sigaddset(&set, SIGUSR1); + __atomic_store_n(&waiter_ready, 1, __ATOMIC_RELEASE); + for (;;) + sigtimedwait(&set, &info, NULL); + return NULL; +} + +/* + * Crash with a shared SIGUSR1 still queued for this thread. The vfork() + * child queues it while we sleep killably and then the SIGSEGV that is + * dequeued first. The zap wakes the sibling out of sigtimedwait() and its + * mask restore retargets SIGUSR1 to the dumper. + */ +static void crashing_child_retarget(bool queue_shared) +{ + sigset_t all, old; + pthread_t thread; + pid_t pid, tid; + char *p; + + p = mmap(NULL, CRASH_MAPPING_SIZE, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (p == MAP_FAILED) + _exit(EXIT_FAILURE); + memset(p, 0x5a, CRASH_MAPPING_SIZE); + + signal(SIGUSR1, usr1_handler); + sigfillset(&all); + pthread_sigmask(SIG_BLOCK, &all, &old); + /* One waiter only, retarget stops at the first thread not blocking it. */ + if (pthread_create(&thread, NULL, sigwaiter, NULL)) + _exit(EXIT_FAILURE); + pthread_sigmask(SIG_SETMASK, &old, NULL); + while (!__atomic_load_n(&waiter_ready, __ATOMIC_ACQUIRE)) + usleep(1000); + usleep(50 * 1000); + + pid = getpid(); + tid = syscall(SYS_gettid); + if (vfork() == 0) { + if (queue_shared) + syscall(SYS_kill, pid, SIGUSR1); + syscall(SYS_tgkill, pid, tid, SIGSEGV); + syscall(SYS_exit, 0); + } + + /* Not reached, the pending SIGSEGV dumps core. */ + for (;;) + pause(); +} + +/* + * Dump the crashing child into /tmp/coredump.file through a server that + * holds the read back so the dumper blocks on the full socket buffer. + */ +static void run_coredump(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, bool queue_shared) +{ + pid_t pid, pid_coredump_server; + int ipc_sockets[2]; + int status; + char c; + + unlink("/tmp/coredump.file"); + unlink("/tmp/coredump.socket"); + ASSERT_TRUE(set_core_pattern("@/tmp/coredump.socket")); + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_core_file = -1; + int exit_code = EXIT_FAILURE; + + close(ipc_sockets[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) { + fprintf(stderr, "%s: accept4 failed: %m\n", __func__); + goto out; + } + + /* Let the dumper run into the full socket buffer first. */ + sleep(1); + + fd_core_file = creat("/tmp/coredump.file", 0644); + if (fd_core_file < 0) { + fprintf(stderr, "%s: creat failed: %m\n", __func__); + goto out; + } + + if (recv_coredump_bytes(fd_coredump, fd_core_file) < 0) + goto out; + + exit_code = EXIT_SUCCESS; +out: + if (fd_core_file >= 0) + close(fd_core_file); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_retarget(queue_shared); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_EQ(WTERMSIG(status), SIGSEGV); + ASSERT_TRUE(WCOREDUMP(status)); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +static void check_coredump_complete(struct __test_metadata *const _metadata) +{ + int fd; + + fd = open("/tmp/coredump.file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0); + ASSERT_TRUE(check_coredump_extent(fd)); + close(fd); +} + +TEST_F(coredump, retarget_shared_pending) +{ + run_coredump(_metadata, self, false); + check_coredump_complete(_metadata); + + run_coredump(_metadata, self, true); + check_coredump_complete(_metadata); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c index d9fa6239b5a9..f5c9bad87546 100644 --- a/tools/testing/selftests/coredump/coredump_socket_protocol_test.c +++ b/tools/testing/selftests/coredump/coredump_socket_protocol_test.c @@ -151,9 +151,7 @@ TEST_F(coredump, socket_request_kernel) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_kernel: check_coredump_req failed\n"); goto out; } @@ -301,9 +299,7 @@ TEST_F(coredump, socket_request_userspace) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_userspace: check_coredump_req failed\n"); goto out; } @@ -441,9 +437,7 @@ TEST_F(coredump, socket_request_reject) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_request_reject: check_coredump_req failed\n"); goto out; } @@ -514,93 +508,89 @@ out: wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } -TEST_F(coredump, socket_request_invalid_flag_combination) +/* An ack the kernel must refuse and how. */ +struct refused_ack { + /* The ack, and how many bytes of it the server sends before it hangs up. */ + struct coredump_ack ack; + size_t bytes; + /* The marker the kernel answers with, or none if @no_marker. */ + enum coredump_mark mark; + bool no_marker; +}; + +/* Send @refused, expect the kernel to refuse it and hang up. */ +static void check_refused_ack(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, + const struct refused_ack *refused) { - int pidfd, ret, status; + int pidfd, status; pid_t pid, pid_coredump_server; struct pidfd_info info = {}; int ipc_sockets[2]; char c; + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - pid_coredump_server = fork(); ASSERT_GE(pid_coredump_server, 0); if (pid_coredump_server == 0) { - struct coredump_req req = {}; int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; close(ipc_sockets[0]); fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: create_and_listen_unix_socket failed: %m\n"); + if (fd_server < 0) goto out; - } - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: write_nointr to ipc socket failed: %m\n"); + if (write_nointr(ipc_sockets[1], "1", 1) < 0) goto out; - } close(ipc_sockets[1]); fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: accept4 failed: %m\n"); + if (fd_coredump < 0) goto out; - } fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_flag_combination: get_peer_pidfd failed\n"); + if (fd_peer_pidfd < 0) goto out; - } - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_flag_combination: get_pidfd_info failed\n"); + /* The task shows as dumping while it waits for the ack. */ + if (!get_pidfd_info(fd_peer_pidfd, &info)) goto out; - } - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_flag_combination: PIDFD_INFO_COREDUMP not set in mask\n"); + if (!(info.mask & PIDFD_INFO_COREDUMP) || + !(info.coredump_mask & PIDFD_COREDUMPED)) { + fprintf(stderr, "Peer isn't marked as dumping\n"); goto out; } - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_flag_combination: PIDFD_COREDUMPED not set in coredump_mask\n"); + if (!read_coredump_req(fd_coredump, &req)) goto out; - } - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_flag_combination: read_coredump_req failed\n"); + if (!check_coredump_req(&req)) goto out; - } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { - fprintf(stderr, "socket_request_invalid_flag_combination: check_coredump_req failed\n"); + if (!send_coredump_ack_bytes(fd_coredump, &refused->ack, + refused->bytes)) goto out; - } - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_KERNEL | COREDUMP_REJECT | COREDUMP_WAIT, 0)) { - fprintf(stderr, "socket_request_invalid_flag_combination: send_coredump_ack failed\n"); + /* Nothing more to say. A server that died looks the same. */ + if (shutdown(fd_coredump, SHUT_WR)) goto out; - } - if (!read_marker(fd_coredump, COREDUMP_MARK_CONFLICTING)) { - fprintf(stderr, "socket_request_invalid_flag_combination: read_marker COREDUMP_MARK_CONFLICTING failed\n"); + if (!refused->no_marker && + !read_marker(fd_coredump, refused->mark)) + goto out; + + /* The kernel hangs up after a refusal, marker or not. */ + if (!read_hangup(fd_coredump)) goto out; - } exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_flag_combination: completed successfully\n"); out: if (fd_peer_pidfd >= 0) close(fd_peer_pidfd); @@ -635,368 +625,72 @@ out: wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } -TEST_F(coredump, socket_request_unknown_flag) +/* Ack @ack_mask, expect the kernel to refuse it as conflicting. */ +static void check_conflicting_ack(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, __u64 ack_mask) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_unknown_flag: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_unknown_flag: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_unknown_flag: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_unknown_flag: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_unknown_flag: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_unknown_flag: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_unknown_flag: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_unknown_flag: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { - fprintf(stderr, "socket_request_unknown_flag: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, (1ULL << 63), 0)) { - fprintf(stderr, "socket_request_unknown_flag: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_UNSUPPORTED)) { - fprintf(stderr, "socket_request_unknown_flag: read_marker COREDUMP_MARK_UNSUPPORTED failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_unknown_flag: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = ack_mask, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_CONFLICTING, + }; + + check_refused_ack(_metadata, self, &refused); +} - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); +/* More than one of KERNEL, USERSPACE and REJECT. */ +TEST_F(coredump, socket_request_invalid_flag_combination) +{ + check_conflicting_ack(_metadata, self, + COREDUMP_KERNEL | COREDUMP_REJECT | COREDUMP_WAIT); +} - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +/* A flag the kernel didn't advertise in coredump_req->mask. */ +TEST_F(coredump, socket_request_unknown_flag) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = 1ULL << 63, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); } +/* An ack smaller than the first published struct. */ TEST_F(coredump, socket_request_invalid_size_small) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_size_small: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_size_small: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_size_small: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_size_small: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_size_small: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_size_small: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_size_small: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_size_small: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { - fprintf(stderr, "socket_request_invalid_size_small: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_REJECT | COREDUMP_WAIT, - COREDUMP_ACK_SIZE_VER0 / 2)) { - fprintf(stderr, "socket_request_invalid_size_small: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_MINSIZE)) { - fprintf(stderr, "socket_request_invalid_size_small: read_marker COREDUMP_MARK_MINSIZE failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_size_small: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); - - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); - - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0 / 2, + .mask = COREDUMP_REJECT | COREDUMP_WAIT, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 / 2, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); } +/* An ack bigger than the kernel said it accepts. */ TEST_F(coredump, socket_request_invalid_size_large) { - int pidfd, ret, status; - pid_t pid, pid_coredump_server; - struct pidfd_info info = {}; - int ipc_sockets[2]; - char c; - - ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); - - ret = socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets); - ASSERT_EQ(ret, 0); - - pid_coredump_server = fork(); - ASSERT_GE(pid_coredump_server, 0); - if (pid_coredump_server == 0) { - struct coredump_req req = {}; - int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; - int exit_code = EXIT_FAILURE; - - close(ipc_sockets[0]); - - fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); - if (fd_server < 0) { - fprintf(stderr, "socket_request_invalid_size_large: create_and_listen_unix_socket failed: %m\n"); - goto out; - } - - if (write_nointr(ipc_sockets[1], "1", 1) < 0) { - fprintf(stderr, "socket_request_invalid_size_large: write_nointr to ipc socket failed: %m\n"); - goto out; - } - - close(ipc_sockets[1]); - - fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); - if (fd_coredump < 0) { - fprintf(stderr, "socket_request_invalid_size_large: accept4 failed: %m\n"); - goto out; - } - - fd_peer_pidfd = get_peer_pidfd(fd_coredump); - if (fd_peer_pidfd < 0) { - fprintf(stderr, "socket_request_invalid_size_large: get_peer_pidfd failed\n"); - goto out; - } - - if (!get_pidfd_info(fd_peer_pidfd, &info)) { - fprintf(stderr, "socket_request_invalid_size_large: get_pidfd_info failed\n"); - goto out; - } - - if (!(info.mask & PIDFD_INFO_COREDUMP)) { - fprintf(stderr, "socket_request_invalid_size_large: PIDFD_INFO_COREDUMP not set in mask\n"); - goto out; - } - - if (!(info.coredump_mask & PIDFD_COREDUMPED)) { - fprintf(stderr, "socket_request_invalid_size_large: PIDFD_COREDUMPED not set in coredump_mask\n"); - goto out; - } - - if (!read_coredump_req(fd_coredump, &req)) { - fprintf(stderr, "socket_request_invalid_size_large: read_coredump_req failed\n"); - goto out; - } - - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { - fprintf(stderr, "socket_request_invalid_size_large: check_coredump_req failed\n"); - goto out; - } - - if (!send_coredump_ack(fd_coredump, &req, - COREDUMP_REJECT | COREDUMP_WAIT, - COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE)) { - fprintf(stderr, "socket_request_invalid_size_large: send_coredump_ack failed\n"); - goto out; - } - - if (!read_marker(fd_coredump, COREDUMP_MARK_MAXSIZE)) { - fprintf(stderr, "socket_request_invalid_size_large: read_marker COREDUMP_MARK_MAXSIZE failed\n"); - goto out; - } - - exit_code = EXIT_SUCCESS; - fprintf(stderr, "socket_request_invalid_size_large: completed successfully\n"); -out: - if (fd_peer_pidfd >= 0) - close(fd_peer_pidfd); - if (fd_coredump >= 0) - close(fd_coredump); - if (fd_server >= 0) - close(fd_server); - _exit(exit_code); - } - self->pid_coredump_server = pid_coredump_server; - - EXPECT_EQ(close(ipc_sockets[1]), 0); - ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); - EXPECT_EQ(close(ipc_sockets[0]), 0); - - pid = fork(); - ASSERT_GE(pid, 0); - if (pid == 0) - crashing_child(); - - pidfd = sys_pidfd_open(pid, 0); - ASSERT_GE(pidfd, 0); - - waitpid(pid, &status, 0); - ASSERT_TRUE(WIFSIGNALED(status)); - ASSERT_FALSE(WCOREDUMP(status)); - - ASSERT_TRUE(get_pidfd_info(pidfd, &info)); - ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); - ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); - - wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE, + .mask = COREDUMP_REJECT | COREDUMP_WAIT, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 + PAGE_SIZE, + .mark = COREDUMP_MARK_MAXSIZE, + }; + + check_refused_ack(_metadata, self, &refused); } /* @@ -1355,9 +1049,7 @@ TEST_F_TIMEOUT(coredump, socket_multiple_crashing_coredumps, 500) goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "check_coredump_req failed for fd %d\n", fd_coredump); goto out; } @@ -1509,9 +1201,7 @@ TEST_F_TIMEOUT(coredump, socket_multiple_crashing_coredumps_epoll_workers, 500) fprintf(stderr, "socket_multiple_crashing_coredumps_epoll_workers: read_coredump_req failed\n"); goto out; } - if (!check_coredump_req(&req, COREDUMP_ACK_SIZE_VER0, - COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT)) { + if (!check_coredump_req(&req)) { fprintf(stderr, "socket_multiple_crashing_coredumps_epoll_workers: check_coredump_req failed\n"); goto out; } @@ -1591,4 +1281,1335 @@ out: wait_and_check_coredump_server(pid_coredump_server, _metadata, self); } +/* + * Reassemble a record stream and check that what comes out is an ELF + * core file. The records themselves are validated by recv_coredump_records(). + */ +TEST_F(coredump, socket_request_sparse_reassemble) +{ + int fd_core_file, pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + + close(ipc_sockets[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + fd_file = creat("/tmp/coredump.file", 0644); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (recv_coredump_records(fd_coredump, fd_file, NULL, NULL, -1) < 0) + goto out; + + exit_code = EXIT_SUCCESS; +out: + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child(); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + /* What the records reassemble into has to be an ELF core file. */ + fd_core_file = open("/tmp/coredump.file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd_core_file, 0); + ASSERT_TRUE(is_elf_core(fd_core_file)); + EXPECT_EQ(close(fd_core_file), 0); +} + +/* + * Crash a child with a mostly-unpopulated mapping and reassemble its + * record stream, reporting what crossed the socket and the coredump + * size the records describe. With @kill_peer the server kills the task + * once the coredump is under way so the kernel has to cut it short. + */ +static void check_record_dump(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, __u64 ack_mask, + bool kill_peer, ssize_t *received, + off_t *coredump_size) +{ + bool truncated = false; + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int pipefds[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + bool is_truncated = false; + off_t size = 0; + ssize_t ret; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* + * The reassembled coredump is bigger than the mapping the + * child made, so keep it on the detached tmpfs and sparse. + */ + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, ack_mask, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + ret = recv_coredump_records(fd_coredump, fd_file, &size, &is_truncated, + kill_peer ? fd_peer_pidfd : -1); + if (ret < 0) + goto out; + + if (write_nointr(pipefds[1], &ret, sizeof(ret)) != sizeof(ret)) + goto out; + if (write_nointr(pipefds[1], &size, sizeof(size)) != sizeof(size)) + goto out; + if (write_nointr(pipefds[1], &is_truncated, + sizeof(is_truncated)) != sizeof(is_truncated)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(SPARSE_MAPPING_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + + ASSERT_EQ(read_nointr(pipefds[0], received, sizeof(*received)), + sizeof(*received)); + ASSERT_EQ(read_nointr(pipefds[0], coredump_size, sizeof(*coredump_size)), + sizeof(*coredump_size)); + ASSERT_EQ(read_nointr(pipefds[0], &truncated, sizeof(truncated)), + sizeof(truncated)); + EXPECT_EQ(close(pipefds[0]), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + if (kill_peer) { + /* The kernel gave up partway, so no end record closed the stream. */ + ASSERT_TRUE(truncated); + ASSERT_FALSE(WCOREDUMP(status)); + ASSERT_LT(*coredump_size, (off_t)SPARSE_MAPPING_SIZE); + return; + } + + ASSERT_FALSE(truncated); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + /* The mapping is in the coredump, holes included. */ + ASSERT_GT(*coredump_size, (off_t)SPARSE_MAPPING_SIZE); +} + +/* + * A mapping that has been written to is dumped whole, including the parts + * of it that were never faulted in. With COREDUMP_SPARSE the holes stay + * off the wire. + */ +TEST_F(coredump, socket_request_sparse_hole) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, + false, &received, &coredump_size); + + /* The holes didn't have to go over the socket. */ + ASSERT_LT(received, coredump_size / 8); +} + +/* + * COREDUMP_RECORDS alone splits the stream into records but elides + * nothing: the holes cross the socket as data records. + */ +TEST_F(coredump, socket_request_records_hole) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_WAIT, + false, &received, &coredump_size); + + /* Records alone elide nothing, so everything crossed the socket. */ + ASSERT_GT(received, coredump_size); +} + +/* + * A coredump the kernel gives up on halfway still ends in an end record, + * and that record says the coredump is incomplete. COREDUMP_SPARSE is left + * out on purpose: the holes have to cross the socket so the coredump is + * far larger than the socket buffer and the kernel is still writing it + * when the kill lands. + */ +TEST_F(coredump, socket_request_records_truncated) +{ + off_t coredump_size = 0; + ssize_t received = 0; + + check_record_dump(_metadata, self, + COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_WAIT, + true, &received, &coredump_size); + + /* The end record crossed the socket even though the task was killed. */ + ASSERT_GT(received, 0); +} + +/* + * A coredump server that uploads to a blob store can't upload a sparse + * file. It doesn't have to: it streams the data records into the object + * as they arrive, leaves the holes out, and uploads the corrected + * program header table last. What it ends up with is an ordinary ELF + * core file that describes the same memory as the coredump the records + * came from, minus the holes. + */ +TEST_F(coredump, socket_request_sparse_blob_upload) +{ + int fd_core_file, pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + off_t coredump_size = 0; + ssize_t received = 0; + int ipc_sockets[2]; + int pipefds[2]; + struct stat st; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_object = -1, fd_reference = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + off_t size = 0; + ssize_t ret; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* The object is a plain file. It never sees a hole. */ + fd_object = open("/tmp/coredump.file", + O_RDWR | O_CREAT | O_TRUNC | O_CLOEXEC, 0600); + if (fd_object < 0) + goto out; + + /* + * The coredump with its holes still in it is bigger than + * the mapping the child made, so keep it on the detached + * tmpfs and sparse. + */ + fd_reference = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_reference < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + if (!send_coredump_ack(fd_coredump, &req, + COREDUMP_KERNEL | COREDUMP_RECORDS | + COREDUMP_SPARSE | COREDUMP_WAIT, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + ret = recv_coredump_compact(fd_coredump, fd_object, + fd_reference, &size); + if (ret < 0) + goto out; + + if (check_compact_coredump(fd_object, fd_reference)) + goto out; + + if (write_nointr(pipefds[1], &ret, sizeof(ret)) != sizeof(ret)) + goto out; + if (write_nointr(pipefds[1], &size, sizeof(size)) != sizeof(size)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_reference >= 0) + close(fd_reference); + if (fd_object >= 0) + close(fd_object); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(SPARSE_MAPPING_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_EQ(read_nointr(pipefds[0], &received, sizeof(received)), + sizeof(received)); + ASSERT_EQ(read_nointr(pipefds[0], &coredump_size, sizeof(coredump_size)), + sizeof(coredump_size)); + EXPECT_EQ(close(pipefds[0]), 0); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); + + /* The mapping is in the coredump, holes included. */ + ASSERT_GT(coredump_size, (off_t)SPARSE_MAPPING_SIZE); + + /* The object isn't sparse and doesn't carry them. */ + ASSERT_EQ(stat("/tmp/coredump.file", &st), 0); + ASSERT_LT(st.st_size, coredump_size / 8); + + /* And a debugger still sees an ordinary ELF core file. */ + fd_core_file = open("/tmp/coredump.file", O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd_core_file, 0); + ASSERT_TRUE(is_elf_core(fd_core_file)); + EXPECT_EQ(close(fd_core_file), 0); +} + +/* COREDUMP_RECORDS applies to a coredump the kernel writes, nothing else. */ +TEST_F(coredump, socket_request_records_without_kernel) +{ + check_conflicting_ack(_metadata, self, COREDUMP_USERSPACE | COREDUMP_RECORDS); +} + +/* A zero record can't exist outside a record stream. */ +TEST_F(coredump, socket_request_sparse_without_records) +{ + check_conflicting_ack(_metadata, self, COREDUMP_KERNEL | COREDUMP_SPARSE); +} + +/* What the server reports back about the coredump it decided to take. */ +struct stream_choice { + bool sparse; + ssize_t received; + off_t size; + ssize_t vm_size; +}; + +/* + * The kernel blocks in the coredump request until the ack arrives, so a + * coredump server gets to look at the task before it commits to a + * stream. Take the record stream only for a task whose mappings are + * worth it and the plain byte stream for everything else. + */ +static void check_stream_choice(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, bool big, + struct stream_choice *choice) +{ + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int pipefds[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(pipefds), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + struct stream_choice got = {}; + __u64 mask; + + close(ipc_sockets[0]); + close(pipefds[0]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + /* + * The reassembled coredump is bigger than the mapping the + * child made, so keep it on the detached tmpfs and sparse. + */ + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req(fd_coredump, &req)) + goto out; + + if (!check_coredump_req(&req)) + goto out; + + /* + * Nothing is on the wire yet and the kernel is waiting for + * the ack, so there is all the time in the world to look at + * the task and decide what to ask it for. + */ + got.vm_size = peer_vm_size(fd_peer_pidfd); + if (got.vm_size < 0) + goto out; + got.sparse = got.vm_size >= SPARSE_STREAM_THRESHOLD; + + fprintf(stderr, "Peer maps %zd bytes, asking for %s\n", + got.vm_size, + got.sparse ? "a sparse record stream" : "a byte stream"); + + mask = COREDUMP_KERNEL | COREDUMP_WAIT; + if (got.sparse) + mask |= COREDUMP_RECORDS | COREDUMP_SPARSE; + + if (!send_coredump_ack(fd_coredump, &req, mask, 0)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (got.sparse) { + got.received = recv_coredump_records(fd_coredump, fd_file, + &got.size, NULL, -1); + } else { + got.received = recv_coredump_bytes(fd_coredump, fd_file); + got.size = got.received; + } + if (got.received < 0) + goto out; + + /* Either way a debugger has to see an ordinary core file. */ + if (!is_elf_core(fd_file)) + goto out; + + if (write_nointr(pipefds[1], &got, sizeof(got)) != sizeof(got)) + goto out; + + exit_code = EXIT_SUCCESS; +out: + close(pipefds[1]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(pipefds[1]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_sparse(big ? SPARSE_MAPPING_SIZE : PAGE_SIZE); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_EQ(read_nointr(pipefds[0], choice, sizeof(*choice)), + sizeof(*choice)); + EXPECT_EQ(close(pipefds[0]), 0); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +/* A task with little mapped isn't worth a record stream. */ +TEST_F(coredump, socket_request_stream_choice_small) +{ + struct stream_choice choice = {}; + + check_stream_choice(_metadata, self, false, &choice); + + ASSERT_LT(choice.vm_size, (ssize_t)SPARSE_STREAM_THRESHOLD); + ASSERT_FALSE(choice.sparse); + ASSERT_GT(choice.received, 0); +} + +/* A task sitting on a big mapping is. */ +TEST_F(coredump, socket_request_stream_choice_large) +{ + struct stream_choice choice = {}; + + check_stream_choice(_metadata, self, true, &choice); + + ASSERT_GE(choice.vm_size, (ssize_t)SPARSE_STREAM_THRESHOLD); + ASSERT_TRUE(choice.sparse); + ASSERT_GT(choice.size, (off_t)SPARSE_MAPPING_SIZE); + + /* The holes didn't have to go over the socket. */ + ASSERT_LT(choice.received, choice.size / 8); +} + +/* What a coredump server was built with. */ +struct server_build { + /* sizeof(struct coredump_req) and sizeof(struct coredump_ack) back then. */ + size_t req_size; + size_t ack_size; + /* The features it raises if the kernel offers them. */ + __u64 wants; + /* Its policy: what it drops from and adds to the task's selection. */ + __u64 drop; + __u64 add; +}; + +/* A server from when the structs were first published: kernel-written dumps. */ +static const struct server_build server_build_ver0 = { + .req_size = COREDUMP_REQ_SIZE_VER0, + .ack_size = COREDUMP_ACK_SIZE_VER0, + .wants = COREDUMP_KERNEL, +}; + +/* A server built against this header: no shared memory, always the ELF headers. */ +static const struct server_build server_build_ver1 = { + .req_size = sizeof(struct coredump_req), + .ack_size = sizeof(struct coredump_ack), + .wants = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .drop = COREDUMP_MEMORY_ANON_SHARED | COREDUMP_MEMORY_FILE_SHARED, + .add = COREDUMP_MEMORY_ELF_HEADERS, +}; + +/* + * Build the ack the way a server does: from what the kernel offers, what + * this build implements, and what fits in the ack the kernel accepts. + * Fields the build never read are zero and never consulted. + */ +static void negotiate(const struct coredump_req *req, + const struct server_build *build, + struct coredump_ack *ack) +{ + __u64 offered = req->mask & build->wants; + + memset(ack, 0, sizeof(*ack)); + ack->size = build->ack_size < req->size_ack ? build->ack_size : req->size_ack; + /* These builds only ever have the kernel write the coredump. */ + ack->mask = COREDUMP_KERNEL; + + /* Sparse needs records, records need the kernel to write. */ + if (offered & COREDUMP_RECORDS) { + ack->mask |= COREDUMP_RECORDS; + if (offered & COREDUMP_SPARSE) + ack->mask |= COREDUMP_SPARSE; + } + + /* The memory types need an ack that carries them. */ + if ((offered & COREDUMP_MEMORY_TYPES) && ack->size >= COREDUMP_ACK_SIZE_VER1) { + ack->mask |= COREDUMP_MEMORY_TYPES; + /* Start from the task's selection; only advertised types pass. */ + ack->memory_types = (req->memory_types & ~build->drop) | build->add; + ack->memory_types &= req->memory_types_mask; + } +} + +/* What a memory types test asks of the kernel and what it expects back. */ +struct memory_choice { + /* Memory types the crashing child selects, or FILTER_TASK_INHERIT. */ + __u64 task_filter; + /* Negotiate the ack as this server build, NULL to send it as given. */ + const struct server_build *build; + /* The ack, or what the negotiation must arrive at. */ + __u64 mask; + __u64 memory_types; + size_t size_ack; + /* The shared mapping is in the coredump with all of its memory. */ + bool shared_dumped; + /* No memory at all. Pull a page from /proc/<pid>/mem instead. */ + bool skeleton; +}; + +/* A skeleton still carries the vdso and friends, nothing bigger. */ +#define SKELETON_DATA_PAGES 16 + +/* + * The crashing child maps shared anonymous memory and tells the server + * where. The server acks with @choice and checks whether that mapping's + * segment in the coredump carries its memory. + */ +static void check_memory_dump(struct __test_metadata *const _metadata, + FIXTURE_DATA(coredump) *self, + const struct memory_choice *choice) +{ + int pidfd, status; + pid_t pid, pid_coredump_server; + struct pidfd_info info = {}; + int ipc_sockets[2]; + int addr_pipe[2]; + char c; + + ASSERT_EQ(socketpair(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0, ipc_sockets), 0); + ASSERT_EQ(pipe(addr_pipe), 0); + ASSERT_TRUE(set_core_pattern("@@/tmp/coredump.socket")); + + pid_coredump_server = fork(); + ASSERT_GE(pid_coredump_server, 0); + if (pid_coredump_server == 0) { + int fd_server = -1, fd_coredump = -1, fd_peer_pidfd = -1; + int fd_file = -1; + int exit_code = EXIT_FAILURE; + struct coredump_req req = {}; + struct coredump_ack ack = { + .size = choice->size_ack, + .mask = choice->mask, + .memory_types = choice->memory_types, + }; + /* How much of the request this server reads. */ + size_t req_size = choice->build ? choice->build->req_size : sizeof(req); + __u64 task_filter; + ElfW(Phdr) segment; + ssize_t received; + off_t size; + char *addr; + + close(ipc_sockets[0]); + close(addr_pipe[1]); + + fd_server = create_and_listen_unix_socket("/tmp/coredump.socket"); + if (fd_server < 0) + goto out; + + if (write_nointr(ipc_sockets[1], "1", 1) < 0) + goto out; + + close(ipc_sockets[1]); + + fd_coredump = accept4(fd_server, NULL, NULL, SOCK_CLOEXEC); + if (fd_coredump < 0) + goto out; + + fd_peer_pidfd = get_peer_pidfd(fd_coredump); + if (fd_peer_pidfd < 0) + goto out; + + fd_file = open_coredump_tmpfile(self->fd_tmpfs_detached); + if (fd_file < 0) + goto out; + + if (!read_coredump_req_sized(fd_coredump, &req, req_size)) + goto out; + + if (!peer_coredump_filter(fd_peer_pidfd, &task_filter)) + goto out; + + /* A build from before the memory types never read that far. */ + if (req_size >= COREDUMP_REQ_SIZE_VER1) { + if (!check_coredump_req(&req)) + goto out; + + /* The request reports the memory types the task selected. */ + if (req.memory_types != task_filter) { + fprintf(stderr, "Request reports 0x%llx, task selected 0x%llx\n", + (unsigned long long)req.memory_types, + (unsigned long long)task_filter); + goto out; + } + } + + if (choice->task_filter != FILTER_TASK_INHERIT && + task_filter != choice->task_filter) { + fprintf(stderr, "Task selected 0x%llx, child asked for 0x%llx\n", + (unsigned long long)task_filter, + (unsigned long long)choice->task_filter); + goto out; + } + + /* The child sent the address of its mapping before it crashed. */ + if (read_nointr(addr_pipe[0], &addr, sizeof(addr)) != sizeof(addr)) + goto out; + + /* A server build negotiates its ack and must arrive at the choice. */ + if (choice->build) { + negotiate(&req, choice->build, &ack); + + if (ack.size != choice->size_ack || ack.mask != choice->mask || + ack.memory_types != choice->memory_types) { + fprintf(stderr, + "Negotiated %u bytes, mask 0x%llx, types 0x%llx\n", + ack.size, (unsigned long long)ack.mask, + (unsigned long long)ack.memory_types); + goto out; + } + } + + if (!send_coredump_ack_types(fd_coredump, &req, ack.mask, + ack.memory_types, ack.size)) + goto out; + + if (!read_marker(fd_coredump, COREDUMP_MARK_REQACK)) + goto out; + + if (ack.mask & COREDUMP_RECORDS) + received = recv_coredump_records(fd_coredump, fd_file, + &size, NULL, -1); + else + received = recv_coredump_bytes(fd_coredump, fd_file); + if (received < 0) + goto out; + + if (!is_elf_core(fd_file)) + goto out; + + /* A dump ending in holes or empty segments must still be whole. */ + if (!check_coredump_extent(fd_file)) + goto out; + + if (!find_coredump_segment(fd_file, (__u64)(uintptr_t)addr, &segment)) + goto out; + + if (segment.p_memsz != MEMORY_MAPPING_SIZE) { + fprintf(stderr, "Segment spans %llu bytes, the mapping %u\n", + (unsigned long long)segment.p_memsz, + MEMORY_MAPPING_SIZE); + goto out; + } + + if (segment.p_filesz != (choice->shared_dumped ? segment.p_memsz : 0)) { + fprintf(stderr, "Segment carries %llu bytes, expected %s of them\n", + (unsigned long long)segment.p_filesz, + choice->shared_dumped ? "all" : "none"); + goto out; + } + + if (choice->skeleton) { + __u64 data, notes, data_max; + char buf[PAGE_SIZE]; + + if (!sum_coredump_segments(fd_file, &data, ¬es)) + goto out; + + data_max = SKELETON_DATA_PAGES * sysconf(_SC_PAGESIZE); + if (!notes || data > data_max) { + fprintf(stderr, "Skeleton has %llu note and %llu memory bytes\n", + (unsigned long long)notes, + (unsigned long long)data); + goto out; + } + + /* The task is parked in COREDUMP_WAIT with its memory. */ + if (peer_read_mem(fd_peer_pidfd, (__u64)(uintptr_t)addr, + buf, sizeof(buf)) != sizeof(buf)) + goto out; + + if (buf[0] != 'x') { + fprintf(stderr, "Pulled memory lacks the child's mark\n"); + goto out; + } + + fprintf(stderr, "Skeleton of %zd bytes, pulled %zu bytes of memory\n", + received, sizeof(buf)); + } + + exit_code = EXIT_SUCCESS; +out: + close(addr_pipe[0]); + if (fd_file >= 0) + close(fd_file); + if (fd_peer_pidfd >= 0) + close(fd_peer_pidfd); + if (fd_coredump >= 0) + close(fd_coredump); + if (fd_server >= 0) + close(fd_server); + _exit(exit_code); + } + self->pid_coredump_server = pid_coredump_server; + + EXPECT_EQ(close(ipc_sockets[1]), 0); + EXPECT_EQ(close(addr_pipe[0]), 0); + ASSERT_EQ(read_nointr(ipc_sockets[0], &c, 1), 1); + EXPECT_EQ(close(ipc_sockets[0]), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) + crashing_child_memory(choice->task_filter, addr_pipe[1]); + EXPECT_EQ(close(addr_pipe[1]), 0); + + pidfd = sys_pidfd_open(pid, 0); + ASSERT_GE(pidfd, 0); + + waitpid(pid, &status, 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_TRUE(WCOREDUMP(status)); + + ASSERT_TRUE(get_pidfd_info(pidfd, &info)); + ASSERT_GT((info.mask & PIDFD_INFO_COREDUMP), 0); + ASSERT_GT((info.coredump_mask & PIDFD_COREDUMPED), 0); + + wait_and_check_coredump_server(pid_coredump_server, _metadata, self); +} + +/* Without COREDUMP_MEMORY_TYPES the task's own selection decides. */ +TEST_F(coredump, socket_request_memory_types_task_includes) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +TEST_F(coredump, socket_request_memory_types_task_excludes) +{ + struct memory_choice choice = { + .task_filter = 0, + .mask = COREDUMP_KERNEL, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The server drops a memory type the task would have dumped. */ +TEST_F(coredump, socket_request_memory_types_restricts) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The server adds a memory type the task had excluded. */ +TEST_F(coredump, socket_request_memory_types_widens) +{ + struct memory_choice choice = { + .task_filter = 0, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The memory types decide what goes into a record stream just the same. */ +TEST_F(coredump, socket_request_memory_types_records) +{ + struct memory_choice choice = { + .task_filter = FILTER_TASK_INHERIT, + .mask = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* + * An empty selection leaves a skeleton: every program header and every note + * but no memory. A server that wants to pick the memory itself reads it + * from /proc/<pid>/mem while the task waits for it to finish. + */ +TEST_F(coredump, socket_request_memory_types_skeleton) +{ + struct memory_choice choice = { + .task_filter = FILTER_TASK_INHERIT, + .mask = COREDUMP_KERNEL | COREDUMP_WAIT | COREDUMP_MEMORY_TYPES, + .memory_types = 0, + .shared_dumped = false, + .skeleton = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* A memory type the kernel didn't advertise in memory_types_mask. */ +TEST_F(coredump, socket_request_memory_types_unknown_bit) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + .memory_types = 1ULL << 63, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* The memory types must be zero unless COREDUMP_MEMORY_TYPES is raised. */ +TEST_F(coredump, socket_request_memory_types_stale_field) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_KERNEL, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* COREDUMP_MEMORY_TYPES needs an ack that has the memory types. */ +TEST_F(coredump, socket_request_memory_types_short_ack) +{ + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + }, + .bytes = COREDUMP_ACK_SIZE_VER0, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* The memory types select what the kernel writes, nothing else. */ +TEST_F(coredump, socket_request_memory_types_without_kernel) +{ + check_conflicting_ack(_metadata, self, COREDUMP_USERSPACE | COREDUMP_MEMORY_TYPES); +} + +/* + * A server built with the first structs reads the request it knows, + * discards the rest and acks with the ack it knows. It raises nothing + * it wasn't built for and the kernel dumps what the task selected. + */ +TEST_F(coredump, socket_request_negotiate_ver0) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .build = &server_build_ver0, + .mask = COREDUMP_KERNEL, + .memory_types = 0, + .size_ack = COREDUMP_ACK_SIZE_VER0, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* + * A server built against this header takes every feature the kernel + * offers, drops shared memory from what the task selected and adds the + * ELF headers. + */ +TEST_F(coredump, socket_request_negotiate_ver1) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .build = &server_build_ver1, + .mask = COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ELF_HEADERS, + .size_ack = COREDUMP_ACK_SIZE_VER1, + .shared_dumped = false, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* An ack that picks none of KERNEL, USERSPACE and REJECT. */ +TEST_F(coredump, socket_request_no_mode) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .mask = COREDUMP_WAIT, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_CONFLICTING, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* @spare must be zero, like every field that isn't in use. */ +TEST_F(coredump, socket_request_spare) +{ + struct refused_ack refused = { + .ack = { + .size = sizeof(struct coredump_ack), + .spare = 1, + .mask = COREDUMP_KERNEL, + }, + .bytes = sizeof(struct coredump_ack), + .mark = COREDUMP_MARK_UNSUPPORTED, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* An ack size is a byte count. One that ends inside a field is valid. */ +#define ACK_SIZE_BETWEEN (COREDUMP_ACK_SIZE_VER0 + sizeof(__u32)) + +/* Any size from VER0 up to what the kernel accepts works without memory types. */ +TEST_F(coredump, socket_request_ack_size_between) +{ + struct memory_choice choice = { + .task_filter = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .mask = COREDUMP_KERNEL, + .size_ack = ACK_SIZE_BETWEEN, + .shared_dumped = true, + }; + + check_memory_dump(_metadata, self, &choice); +} + +/* The memory types need the whole field, not the part that happens to fit. */ +TEST_F(coredump, socket_request_memory_types_ack_size_between) +{ + struct refused_ack refused = { + .ack = { + .size = ACK_SIZE_BETWEEN, + .mask = COREDUMP_KERNEL | COREDUMP_MEMORY_TYPES, + }, + .bytes = ACK_SIZE_BETWEEN, + .mark = COREDUMP_MARK_MINSIZE, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* A server that hangs up without acking gets no marker and no coredump. */ +TEST_F(coredump, socket_request_server_hangs_up) +{ + struct refused_ack refused = { + .bytes = 0, + .no_marker = true, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* A server that hangs up in the middle of its ack looks the same. */ +TEST_F(coredump, socket_request_ack_truncated) +{ + struct refused_ack refused = { + .ack = { + .size = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL, + }, + .bytes = COREDUMP_ACK_SIZE_VER0 / 2, + .no_marker = true, + }; + + check_refused_ack(_metadata, self, &refused); +} + +/* + * The kernels a server built against this header can't meet here: + * negotiate() against their requests, no coredump involved. + */ + +/* The request of a kernel with the first structs and features. */ +static const struct coredump_req req_ver0 = { + .size = COREDUMP_REQ_SIZE_VER0, + .size_ack = COREDUMP_ACK_SIZE_VER0, + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT, +}; + +/* The request of this kernel. */ +static const struct coredump_req req_ver1 = { + .size = COREDUMP_REQ_SIZE_VER1, + .size_ack = COREDUMP_ACK_SIZE_VER1, + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .memory_types = COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ANON_SHARED, + .memory_types_mask = TEST_MEMORY_ALL, +}; + +/* A kernel with the first structs gets the first ack and nothing newer. */ +TEST(negotiate_ver0_kernel) +{ + struct coredump_ack ack; + + negotiate(&req_ver0, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL); + ASSERT_EQ(ack.memory_types, 0); +} + +/* A kernel with records and sparse but the first structs: both, no types. */ +TEST(negotiate_sparse_kernel) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_RECORDS | COREDUMP_SPARSE; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE); + ASSERT_EQ(ack.memory_types, 0); +} + +/* Records without sparse: sparse isn't raised on its own. */ +TEST(negotiate_records_without_sparse) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_RECORDS; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS); +} + +/* + * A feature whose ack field lies past what the kernel accepts can't be + * raised. No kernel offers the memory types without the room for them, so a + * request that does stands in for a feature newer than this header. + */ +TEST(negotiate_types_need_room) +{ + struct coredump_req req = req_ver0; + struct coredump_ack ack; + + req.mask |= COREDUMP_MEMORY_TYPES; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER0); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL); + ASSERT_EQ(ack.memory_types, 0); +} + +/* This kernel: the policy applied to the task's selection. */ +TEST(negotiate_ver1_kernel) +{ + struct coredump_ack ack; + + negotiate(&req_ver1, &server_build_ver1, &ack); + ASSERT_EQ(ack.size, COREDUMP_ACK_SIZE_VER1); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES); + ASSERT_EQ(ack.memory_types, COREDUMP_MEMORY_ANON_PRIVATE | + COREDUMP_MEMORY_ELF_HEADERS); +} + +/* A kernel that doesn't know a type the policy adds isn't asked for it. */ +TEST(negotiate_unknown_type) +{ + struct coredump_req req = req_ver1; + struct coredump_ack ack; + + req.memory_types_mask &= ~(__u64)COREDUMP_MEMORY_ELF_HEADERS; + negotiate(&req, &server_build_ver1, &ack); + ASSERT_EQ(ack.mask, COREDUMP_KERNEL | COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES); + ASSERT_EQ(ack.memory_types, COREDUMP_MEMORY_ANON_PRIVATE); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/coredump/coredump_test.h b/tools/testing/selftests/coredump/coredump_test.h index ed47f01fa53c..1c154f513700 100644 --- a/tools/testing/selftests/coredump/coredump_test.h +++ b/tools/testing/selftests/coredump/coredump_test.h @@ -3,18 +3,10 @@ #ifndef __COREDUMP_TEST_H #define __COREDUMP_TEST_H -#include <stdbool.h> -#include <sys/types.h> -#include <linux/coredump.h> - #include "../kselftest_harness.h" -#include "../pidfd/pidfd.h" - -#ifndef PAGE_SIZE -#define PAGE_SIZE 4096 -#endif +#include "coredump_notify_signal.h" -#define NUM_THREAD_SPAWN 128 +#include "coredump_test_helpers.h" /* Coredump fixture */ FIXTURE(coredump) @@ -24,15 +16,6 @@ FIXTURE(coredump) int fd_tmpfs_detached; }; -/* Shared helper function declarations */ -void *do_nothing(void *arg); -void crashing_child(void); -int create_detached_tmpfs(void); -int create_and_listen_unix_socket(const char *path); -bool set_core_pattern(const char *pattern); -int get_peer_pidfd(int fd); -bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); - /* Inline helper that uses harness types */ static inline void wait_and_check_coredump_server(pid_t pid_coredump_server, struct __test_metadata *const _metadata, @@ -45,15 +28,4 @@ static inline void wait_and_check_coredump_server(pid_t pid_coredump_server, ASSERT_EQ(WEXITSTATUS(status), 0); } -/* Protocol helper function declarations */ -ssize_t recv_marker(int fd); -bool read_marker(int fd, enum coredump_mark mark); -bool read_coredump_req(int fd, struct coredump_req *req); -bool send_coredump_ack(int fd, const struct coredump_req *req, - __u64 mask, size_t size_ack); -bool check_coredump_req(const struct coredump_req *req, size_t min_size, - __u64 required_mask); -int open_coredump_tmpfile(int fd_tmpfs_detached); -void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); - #endif /* __COREDUMP_TEST_H */ diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.c b/tools/testing/selftests/coredump/coredump_test_helpers.c index 2a20faf9cb0a..91607e09127f 100644 --- a/tools/testing/selftests/coredump/coredump_test_helpers.c +++ b/tools/testing/selftests/coredump/coredump_test_helpers.c @@ -1,11 +1,18 @@ // SPDX-License-Identifier: GPL-2.0 #include <assert.h> +#include <elf.h> +#include <endian.h> #include <errno.h> #include <fcntl.h> #include <limits.h> +#include <link.h> +#include <linux/stddef.h> +#include <linux/io_uring.h> +#include <linux/swab.h> #include <linux/coredump.h> #include <linux/fs.h> +#include <poll.h> #include <pthread.h> #include <stdbool.h> #include <stdio.h> @@ -13,31 +20,26 @@ #include <string.h> #include <sys/epoll.h> #include <sys/ioctl.h> +#include <sys/mman.h> #include <sys/socket.h> +#include <sys/stat.h> +#include <sys/syscall.h> #include <sys/types.h> #include <sys/un.h> #include <sys/wait.h> #include <unistd.h> #include "../filesystems/wrappers.h" -#include "../pidfd/pidfd.h" +#include "coredump_notify_signal.h" -/* Forward declarations to avoid including harness header */ -struct __test_metadata; +#include "coredump_test_helpers.h" -/* Match the fixture definition from coredump_test.h */ -struct _fixture_coredump_data { - char original_core_pattern[256]; - pid_t pid_coredump_server; - int fd_tmpfs_detached; -}; - -#ifndef PAGE_SIZE -#define PAGE_SIZE 4096 +#if __ELF_NATIVE_CLASS == 64 +#define COREDUMP_ELFCLASS ELFCLASS64 +#else +#define COREDUMP_ELFCLASS ELFCLASS32 #endif -#define NUM_THREAD_SPAWN 128 - void *do_nothing(void *arg) { (void)arg; @@ -59,6 +61,1193 @@ void crashing_child(void) i = *(volatile int *)NULL; } +void crashing_child_sparse(size_t size) +{ + char *p; + + /* + * Touch the first and the last page. This will cause the whole mapping + * to be dumped because it has been written to. Everything between + * those two pages is a hole though. + */ + p = mmap(NULL, size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0); + if (p != MAP_FAILED) { + p[0] = 'x'; + p[size - 1] = 'x'; + } + + /* crash on purpose */ + *(volatile int *)NULL = 0; +} + +/* Select @types through the caller's own /proc/self/coredump_filter. */ +static bool set_coredump_filter(__u64 types) +{ + char buf[32]; + int fd, len; + bool ok; + + fd = open("/proc/self/coredump_filter", O_WRONLY | O_CLOEXEC); + if (fd < 0) + return false; + + len = snprintf(buf, sizeof(buf), "0x%llx", (unsigned long long)types); + ok = write_nointr(fd, buf, len) == len; + close(fd); + return ok; +} + +/* + * Map shared anonymous memory, touch it, tell the server where it is and + * crash. A @task_filter other than FILTER_TASK_INHERIT is selected first. + */ +void crashing_child_memory(__u64 task_filter, int fd_addr) +{ + char *p; + + if (task_filter != FILTER_TASK_INHERIT && !set_coredump_filter(task_filter)) + _exit(EXIT_FAILURE); + + p = mmap(NULL, MEMORY_MAPPING_SIZE, PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_ANONYMOUS, -1, 0); + if (p == MAP_FAILED) + _exit(EXIT_FAILURE); + p[0] = 'x'; + + if (write_nointr(fd_addr, &p, sizeof(p)) != sizeof(p)) + _exit(EXIT_FAILURE); + close(fd_addr); + + /* crash on purpose */ + *(volatile int *)NULL = 0; +} + +/* Sink a reassembled record stream is handed to, record by record. */ +struct coredump_record_sink { + /* @len bytes of coredump data that belong at @offset. */ + int (*data)(void *ctx, const void *buf, size_t len, __u64 offset); + /* @len zero bytes that belong at @offset. */ + int (*zero)(void *ctx, __u64 offset, __u64 len); + void *ctx; +}; + +/* Read @len bytes off the socket and hand them to @sink, if there is one. */ +static ssize_t recv_record_bytes(int fd_coredump, __u64 len, + const struct coredump_record_sink *sink, + __u64 offset) +{ + ssize_t received = 0; + + while (len) { + char buffer[PAGE_SIZE]; + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + ssize_t ret; + + ret = recv(fd_coredump, buffer, chunk, MSG_WAITALL); + if (ret <= 0) { + fprintf(stderr, "%s: short read %zd: %m\n", + __func__, ret); + return -1; + } + + if (sink && sink->data(sink->ctx, buffer, ret, offset + received)) + return -1; + + received += ret; + len -= ret; + } + + return received; +} + +/* Put the data where the records say it goes and leave the holes alone. */ +static int file_sink_data(void *ctx, const void *buf, size_t len, __u64 offset) +{ + int fd = *(int *)ctx; + + if (pwrite(fd, buf, len, offset) != (ssize_t)len) { + fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + return -1; + } + + return 0; +} + +static int file_sink_zero(void *ctx, __u64 offset, __u64 len) +{ + /* Nothing has to be written for a hole. */ + return 0; +} + +/* + * Read a coredump strea and funnel it into @sink. Allow to pass in a + * @fd_peer_pidfd to simulate coredump truncation by killing it after having + * received a coredump record. + */ +static ssize_t __recv_coredump_records(int fd_coredump, + const struct coredump_record_sink *sink, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd) +{ + ssize_t received = 0; + off_t size = 0; + bool is_truncated = false; + bool ended = false; + char trailing; + + while (!ended) { + struct coredump_record_header record = {}; + size_t known_size; + ssize_t ret; + + /* Peek the header size the way read_coredump_req() does. */ + ret = recv(fd_coredump, &record, sizeof(record.size), + MSG_PEEK | MSG_WAITALL); + if (ret == 0) { + /* Nothing closed the stream, so the coredump was cut short. */ + if (truncated) { + is_truncated = true; + break; + } + fprintf(stderr, "%s: stream ended without an end record\n", + __func__); + return -1; + } + if (ret != sizeof(record.size)) { + fprintf(stderr, "%s: short record peek %zd: %m\n", + __func__, ret); + return -1; + } + + if (record.size < COREDUMP_RECORD_HEADER_SIZE_VER0) { + fprintf(stderr, "%s: header size %u below minimum %u\n", + __func__, record.size, + COREDUMP_RECORD_HEADER_SIZE_VER0); + return -1; + } + + /* Consume as much of the header as we know about. */ + known_size = record.size < sizeof(record) ? record.size : sizeof(record); + ret = recv(fd_coredump, &record, known_size, MSG_WAITALL); + if (ret != (ssize_t)known_size) { + fprintf(stderr, "%s: short record read %zd: %m\n", + __func__, ret); + return -1; + } + received += ret; + + /* + * A flag changes what the record means, so refuse one we + * don't know rather than guess. + */ + if (record.flags) { + fprintf(stderr, "%s: unknown header flags 0x%llx\n", + __func__, (unsigned long long)record.flags); + return -1; + } + + /* Discard any part of the header we have no use for. */ + ret = recv_record_bytes(fd_coredump, record.size - known_size, + NULL, 0); + if (ret < 0) + return -1; + received += ret; + + /* Records are sent in order and they don't leave gaps. */ + if (record.offset != (__u64)size) { + fprintf(stderr, "%s: record at %llu, expected %llu\n", + __func__, (unsigned long long)record.offset, + (unsigned long long)size); + return -1; + } + + switch (record.type) { + case COREDUMP_RECORD_ZERO: + /* A hole. It comes with no data and needs none. */ + if (sink->zero(sink->ctx, record.offset, record.len)) + return -1; + break; + case COREDUMP_RECORD_DATA: + ret = recv_record_bytes(fd_coredump, record.len, sink, + record.offset); + if (ret < 0) + return -1; + received += ret; + if (fd_peer_pidfd >= 0) { + if (sys_pidfd_send_signal(fd_peer_pidfd, SIGKILL, + NULL, 0)) { + fprintf(stderr, "%s: kill failed: %m\n", + __func__); + return -1; + } + fd_peer_pidfd = -1; + } + break; + case COREDUMP_RECORD_END: + /* The coredump ends here and nothing follows it. */ + if (record.len) { + fprintf(stderr, "%s: end record covers %llu bytes\n", + __func__, + (unsigned long long)record.len); + return -1; + } + ended = true; + break; + default: + fprintf(stderr, "%s: unknown record type %u\n", + __func__, record.type); + return -1; + } + + size += record.len; + } + + /* The end record is the last thing on the wire. */ + if (recv(fd_coredump, &trailing, sizeof(trailing), MSG_DONTWAIT) > 0) { + fprintf(stderr, "%s: data after the end record\n", __func__); + return -1; + } + + if (truncated) + *truncated = is_truncated; + + *coredump_size = size; + + fprintf(stderr, "Received %zd bytes for a %s coredump of %llu bytes\n", + received, is_truncated ? "truncated" : "complete", + (unsigned long long)size); + return received; +} + +/* Reassemble a record stream into the coredump it describes. */ +ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd) +{ + struct coredump_record_sink sink = { + .data = file_sink_data, + .zero = file_sink_zero, + .ctx = &fd_core_file, + }; + ssize_t received; + off_t size = 0; + + received = __recv_coredump_records(fd_coredump, &sink, &size, truncated, + fd_peer_pidfd); + if (received < 0) + return -1; + + /* + * Nothing is written for a hole, so grow the file to the size the + * records describe in case the coredump ended in one. + */ + if (ftruncate(fd_core_file, size) < 0) { + fprintf(stderr, "%s: ftruncate to %llu failed: %m\n", + __func__, (unsigned long long)size); + return -1; + } + + if (coredump_size) + *coredump_size = size; + + return received; +} + +/* The ELF header of a native core file. */ +static bool is_core_ehdr(const ElfW(Ehdr) *ehdr) +{ + return !memcmp(ehdr->e_ident, ELFMAG, SELFMAG) && + ehdr->e_ident[EI_CLASS] == COREDUMP_ELFCLASS && + ehdr->e_type == ET_CORE; +} + +/* Whatever the server ends up with has to be an ELF core file. */ +bool is_elf_core(int fd) +{ + ElfW(Ehdr) ehdr; + + if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr)) { + fprintf(stderr, "%s: short read: %m\n", __func__); + return false; + } + + if (!is_core_ehdr(&ehdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return false; + } + + return true; +} + +/* + * A coredump server that uploads to a blob store can't upload a sparse + * file and can't seek in the object it is uploading. It streams the data + * records into the object as they arrive, remembers the holes it left + * out, and uploads the program header table that describes the result + * last. What comes out is an ordinary ELF core file without the holes. + */ + +/* A run of the coredump the object doesn't carry. */ +struct compact_hole { + __u64 offset; + __u64 len; +}; + +/* A program header of the object and where its bytes sat in the coredump. */ +struct compact_piece { + ElfW(Phdr) phdr; + __u64 src; +}; + +struct compact_ctx { + int fd_body; /* the object's payload, append only */ + int fd_reference; /* the coredump with its holes, for the test */ + unsigned char *head; /* everything ahead of the segment data */ + size_t head_len; + size_t head_cap; + __u64 data_offset; /* where the segment data starts, 0 while unknown */ + struct compact_hole *holes; + size_t nr_holes; + size_t holes_cap; + __u64 body_len; +}; + +/* Write @len bytes out, short writes and all. */ +static int compact_write(int fd, const void *buf, size_t len) +{ + const unsigned char *pos = buf; + + while (len) { + ssize_t ret = write(fd, pos, len); + + if (ret <= 0) { + fprintf(stderr, "%s: write failed: %m\n", __func__); + return -1; + } + + pos += ret; + len -= ret; + } + + return 0; +} + +/* Keep @len bytes of the head, or @len zeroes if @buf is NULL. */ +static int compact_head_append(struct compact_ctx *ctx, const void *buf, + size_t len) +{ + if (ctx->head_len + len > ctx->head_cap) { + size_t cap = ctx->head_cap ? ctx->head_cap : PAGE_SIZE; + unsigned char *head; + + while (cap < ctx->head_len + len) + cap *= 2; + + head = realloc(ctx->head, cap); + if (!head) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + ctx->head = head; + ctx->head_cap = cap; + } + + if (buf) + memcpy(ctx->head + ctx->head_len, buf, len); + else + memset(ctx->head + ctx->head_len, 0, len); + ctx->head_len += len; + + return 0; +} + +/* Remember a hole so the program header table can account for it later. */ +static int compact_keep_hole(struct compact_ctx *ctx, __u64 offset, __u64 len) +{ + if (ctx->nr_holes == ctx->holes_cap) { + size_t cap = ctx->holes_cap ? ctx->holes_cap * 2 : 64; + struct compact_hole *holes; + + holes = realloc(ctx->holes, cap * sizeof(*holes)); + if (!holes) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + ctx->holes = holes; + ctx->holes_cap = cap; + } + + ctx->holes[ctx->nr_holes].offset = offset; + ctx->holes[ctx->nr_holes].len = len; + ctx->nr_holes++; + + return 0; +} + +/* The segment data starts where the first PT_LOAD points. */ +static int compact_probe(struct compact_ctx *ctx) +{ + const ElfW(Ehdr) *ehdr = (const ElfW(Ehdr) *)ctx->head; + const ElfW(Phdr) *phdr; + size_t i; + + if (ctx->data_offset || ctx->head_len < sizeof(*ehdr)) + return 0; + + if (!is_core_ehdr(ehdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return -1; + } + + if (ehdr->e_phoff != sizeof(*ehdr) || + ehdr->e_phentsize != sizeof(ElfW(Phdr)) || + ehdr->e_phnum == 0 || ehdr->e_phnum == PN_XNUM) { + fprintf(stderr, "%s: unhandled program header table\n", __func__); + return -1; + } + + if (ctx->head_len < ehdr->e_phoff + + (size_t)ehdr->e_phnum * ehdr->e_phentsize) + return 0; + + phdr = (const ElfW(Phdr) *)(ctx->head + ehdr->e_phoff); + for (i = 0; i < ehdr->e_phnum; i++) { + if (phdr[i].p_type != PT_LOAD) + continue; + if (!ctx->data_offset || phdr[i].p_offset < ctx->data_offset) + ctx->data_offset = phdr[i].p_offset; + } + + if (!ctx->data_offset) { + fprintf(stderr, "%s: coredump without a single segment\n", + __func__); + return -1; + } + + return 0; +} + +/* + * Take whatever of [@offset, @offset + @len) still belongs to the head. + * @buf is NULL for a hole. Returns how much was taken. + */ +static ssize_t compact_head_take(struct compact_ctx *ctx, const void *buf, + __u64 offset, __u64 len) +{ + __u64 chunk; + + if (!len || (ctx->data_offset && offset >= ctx->data_offset)) + return 0; + + chunk = len; + if (ctx->data_offset && offset + chunk > ctx->data_offset) + chunk = ctx->data_offset - offset; + + if (offset != ctx->head_len) { + fprintf(stderr, "%s: head has a gap at %llu\n", __func__, + (unsigned long long)offset); + return -1; + } + + if (compact_head_append(ctx, buf, chunk)) + return -1; + + return chunk; +} + +static int compact_data(void *arg, const void *buf, size_t len, __u64 offset) +{ + struct compact_ctx *ctx = arg; + const unsigned char *pos = buf; + ssize_t head; + + /* Only the test needs a coredump with the holes still in it. */ + if (pwrite(ctx->fd_reference, pos, len, offset) != (ssize_t)len) { + fprintf(stderr, "%s: pwrite failed: %m\n", __func__); + return -1; + } + + /* The head has to be rewritten at the end, so hold on to it. */ + head = compact_head_take(ctx, pos, offset, len); + if (head < 0) + return -1; + if (head && compact_probe(ctx)) + return -1; + + pos += head; + len -= head; + if (!len) + return 0; + + /* Everything else goes into the object as it arrives. */ + if (compact_write(ctx->fd_body, pos, len)) + return -1; + ctx->body_len += len; + + return 0; +} + +static int compact_zero(void *arg, __u64 offset, __u64 len) +{ + struct compact_ctx *ctx = arg; + ssize_t head; + + /* A hole in the head is alignment padding. Write it out. */ + head = compact_head_take(ctx, NULL, offset, len); + if (head < 0) + return -1; + + offset += head; + len -= head; + if (!len) + return 0; + + /* This is what the object doesn't have to carry. */ + return compact_keep_hole(ctx, offset, len); +} + +/* Where @offset ends up in the object once the holes ahead of it are gone. */ +static __u64 compact_offset(const struct compact_ctx *ctx, __u64 body_start, + __u64 offset) +{ + __u64 elided = 0; + size_t i; + + for (i = 0; i < ctx->nr_holes; i++) { + __u64 len = ctx->holes[i].len; + + if (ctx->holes[i].offset >= offset) + break; + if (ctx->holes[i].offset + len > offset) + len = offset - ctx->holes[i].offset; + elided += len; + } + + return body_start + (offset - ctx->data_offset) - elided; +} + +/* A run of segment data that made it into the object. */ +static void compact_add_data(struct compact_piece *pieces, size_t *nr, + const ElfW(Phdr) *phdr, __u64 start, __u64 end) +{ + struct compact_piece *piece = &pieces[(*nr)++]; + + piece->phdr = *phdr; + piece->phdr.p_vaddr = phdr->p_vaddr + (start - phdr->p_offset); + piece->phdr.p_paddr = 0; + piece->phdr.p_filesz = end - start; + piece->phdr.p_memsz = end - start; + piece->src = start; +} + +/* + * A run of @len bytes the object doesn't carry. It grows the piece in + * front of it if this segment already has one, because everything a + * segment covers past p_filesz is zeroes anyway. + */ +static void compact_add_zero(struct compact_piece *pieces, size_t *nr, + size_t first, const ElfW(Phdr) *phdr, __u64 vaddr, + __u64 len) +{ + struct compact_piece *piece; + + if (*nr > first) { + pieces[*nr - 1].phdr.p_memsz += len; + return; + } + + piece = &pieces[(*nr)++]; + piece->phdr = *phdr; + piece->phdr.p_vaddr = vaddr; + piece->phdr.p_paddr = 0; + piece->phdr.p_filesz = 0; + piece->phdr.p_memsz = len; + piece->src = 0; +} + +/* Split the segments at the holes and write out what the object became. */ +static int compact_build(struct compact_ctx *ctx, int fd_object) +{ + __u64 note_offset = 0, note_len = 0, note_new; + __u64 align = 0, head_len, body_start, pos; + size_t nr_old, nr_new = 0, note_piece = 0, i; + struct compact_piece *pieces; + char buffer[PAGE_SIZE]; + const ElfW(Phdr) *old; + ElfW(Ehdr) ehdr; + int ret = -1; + + if (!ctx->data_offset) { + fprintf(stderr, "%s: coredump without segment data\n", __func__); + return -1; + } + + memcpy(&ehdr, ctx->head, sizeof(ehdr)); + if (ehdr.e_shoff) { + fprintf(stderr, "%s: section headers are not handled\n", + __func__); + return -1; + } + + old = (const ElfW(Phdr) *)(ctx->head + ehdr.e_phoff); + nr_old = ehdr.e_phnum; + + pieces = calloc(nr_old + 2 * ctx->nr_holes + 1, sizeof(*pieces)); + if (!pieces) { + fprintf(stderr, "%s: out of memory\n", __func__); + return -1; + } + + for (i = 0; i < nr_old; i++) { + ElfW(Phdr) phdr = old[i]; + __u64 end = phdr.p_offset + phdr.p_filesz; + __u64 cur = phdr.p_offset; + size_t first = nr_new, h; + + /* The notes move because the table in front of them grows. */ + if (phdr.p_type == PT_NOTE) { + if (note_len) { + fprintf(stderr, "%s: more than one note segment\n", + __func__); + goto out; + } + note_offset = phdr.p_offset; + note_len = phdr.p_filesz; + note_piece = nr_new; + pieces[nr_new].phdr = phdr; + pieces[nr_new++].src = 0; + continue; + } + + if (phdr.p_type != PT_LOAD) { + if (phdr.p_filesz && phdr.p_offset < ctx->data_offset) { + fprintf(stderr, "%s: segment %zu is in the head\n", + __func__, i); + goto out; + } + pieces[nr_new].phdr = phdr; + pieces[nr_new++].src = phdr.p_offset; + continue; + } + + if (!align) + align = phdr.p_align; + + for (h = 0; h < ctx->nr_holes && cur < end; h++) { + __u64 start = ctx->holes[h].offset; + __u64 stop = start + ctx->holes[h].len; + + if (stop <= cur) + continue; + if (start >= end) + break; + + /* A hole can span more than this one segment. */ + if (start < cur) + start = cur; + if (stop > end) + stop = end; + + if (start > cur) { + compact_add_data(pieces, &nr_new, &phdr, cur, + start); + cur = start; + } + compact_add_zero(pieces, &nr_new, first, &phdr, + phdr.p_vaddr + (cur - phdr.p_offset), + stop - cur); + cur = stop; + } + + if (cur < end) + compact_add_data(pieces, &nr_new, &phdr, cur, end); + + /* Whatever the kernel didn't dump of this mapping. */ + if (phdr.p_memsz > phdr.p_filesz) + compact_add_zero(pieces, &nr_new, first, &phdr, + phdr.p_vaddr + phdr.p_filesz, + phdr.p_memsz - phdr.p_filesz); + } + + if (!note_len || note_offset + note_len > ctx->head_len) { + fprintf(stderr, "%s: notes aren't where they should be\n", + __func__); + goto out; + } + + if (nr_new >= PN_XNUM) { + fprintf(stderr, "%s: %zu program headers don't fit\n", __func__, + nr_new); + goto out; + } + + if (!align || (align & (align - 1))) + align = sysconf(_SC_PAGESIZE); + + note_new = sizeof(ehdr) + (__u64)nr_new * sizeof(ElfW(Phdr)); + head_len = note_new + note_len; + body_start = (head_len + align - 1) & ~(align - 1); + + for (i = 0; i < nr_new; i++) { + struct compact_piece *piece = &pieces[i]; + + if (i == note_piece) + piece->phdr.p_offset = note_new; + else if (piece->phdr.p_filesz) + piece->phdr.p_offset = compact_offset(ctx, body_start, + piece->src); + else + piece->phdr.p_offset = 0; + } + + /* Only now is the head known. That's why it is uploaded last. */ + ehdr.e_phnum = nr_new; + if (compact_write(fd_object, &ehdr, sizeof(ehdr))) + goto out; + + for (i = 0; i < nr_new; i++) + if (compact_write(fd_object, &pieces[i].phdr, + sizeof(pieces[i].phdr))) + goto out; + + if (compact_write(fd_object, ctx->head + note_offset, note_len)) + goto out; + + /* Keep the segments aligned the way a debugger expects them. */ + memset(buffer, 0, sizeof(buffer)); + for (pos = head_len; pos < body_start; ) { + __u64 chunk = body_start - pos; + + if (chunk > sizeof(buffer)) + chunk = sizeof(buffer); + if (compact_write(fd_object, buffer, chunk)) + goto out; + pos += chunk; + } + + /* Putting the parts together is the blob store's job. Do it here. */ + for (pos = 0; pos < ctx->body_len; ) { + ssize_t chunk = pread(ctx->fd_body, buffer, sizeof(buffer), pos); + + if (chunk <= 0) { + fprintf(stderr, "%s: short read %zd: %m\n", __func__, + chunk); + goto out; + } + if (compact_write(fd_object, buffer, chunk)) + goto out; + pos += chunk; + } + + fprintf(stderr, "Object is %llu bytes in %zu program headers, %zu holes left out\n", + (unsigned long long)(body_start + ctx->body_len), nr_new, + ctx->nr_holes); + ret = 0; +out: + free(pieces); + return ret; +} + +/* + * Reassemble a record stream into an ELF core file that has no holes in + * it, the way a coredump server that uploads to a blob store has to. If + * @fd_reference is valid it gets the coredump the records describe, + * holes and all, so the test can compare the two. + */ +ssize_t recv_coredump_compact(int fd_coredump, int fd_object, int fd_reference, + off_t *coredump_size) +{ + struct compact_ctx ctx = { + .fd_body = -1, + .fd_reference = fd_reference, + }; + struct coredump_record_sink sink = { + .data = compact_data, + .zero = compact_zero, + .ctx = &ctx, + }; + ssize_t received; + off_t size = 0; + FILE *body; + + body = tmpfile(); + if (!body) { + fprintf(stderr, "%s: tmpfile failed: %m\n", __func__); + return -1; + } + ctx.fd_body = fileno(body); + + /* An upload is appended to. Make sure nothing here can seek. */ + if (fcntl(ctx.fd_body, F_SETFL, O_APPEND)) { + fprintf(stderr, "%s: F_SETFL failed: %m\n", __func__); + received = -1; + goto out; + } + + received = __recv_coredump_records(fd_coredump, &sink, &size, NULL, -1); + if (received < 0) + goto out; + + /* + * Nothing is written for a hole, so grow the reference to the size + * the records describe in case the coredump ended in one. + */ + if (ftruncate(fd_reference, size) < 0) { + fprintf(stderr, "%s: ftruncate to %llu failed: %m\n", + __func__, (unsigned long long)size); + received = -1; + goto out; + } + + if (compact_build(&ctx, fd_object)) { + received = -1; + goto out; + } + + if (coredump_size) + *coredump_size = size; +out: + fclose(body); + free(ctx.head); + free(ctx.holes); + return received; +} + +/* Read the ELF header and the program header table of @fd. */ +static ElfW(Phdr) *read_phdrs(int fd, size_t *nr) +{ + ElfW(Ehdr) ehdr; + ElfW(Phdr) *phdr; + size_t size; + + if (pread(fd, &ehdr, sizeof(ehdr), 0) != sizeof(ehdr)) { + fprintf(stderr, "%s: no ELF header: %m\n", __func__); + return NULL; + } + + if (!is_core_ehdr(&ehdr) || !ehdr.e_phnum || + ehdr.e_phentsize != sizeof(*phdr)) { + fprintf(stderr, "%s: not an ELF core file\n", __func__); + return NULL; + } + + size = (size_t)ehdr.e_phnum * ehdr.e_phentsize; + phdr = malloc(size); + if (!phdr) { + fprintf(stderr, "%s: out of memory\n", __func__); + return NULL; + } + + if (pread(fd, phdr, size, ehdr.e_phoff) != (ssize_t)size) { + fprintf(stderr, "%s: short program header table: %m\n", __func__); + free(phdr); + return NULL; + } + + *nr = ehdr.e_phnum; + return phdr; +} + +/* The segment @vaddr falls into. */ +static const ElfW(Phdr) *find_segment(const ElfW(Phdr) *phdr, size_t nr, + __u64 vaddr) +{ + size_t i; + + for (i = 0; i < nr; i++) { + if (phdr[i].p_type != PT_LOAD) + continue; + if (vaddr >= phdr[i].p_vaddr && + vaddr < phdr[i].p_vaddr + phdr[i].p_memsz) + return &phdr[i]; + } + + return NULL; +} + +/* The PT_LOAD segment @vaddr falls into. */ +bool find_coredump_segment(int fd, __u64 vaddr, ElfW(Phdr) *segment) +{ + const ElfW(Phdr) *found; + ElfW(Phdr) *phdr; + size_t nr; + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + found = find_segment(phdr, nr, vaddr); + if (found) + *segment = *found; + else + fprintf(stderr, "%s: no segment for 0x%llx\n", __func__, + (unsigned long long)vaddr); + + free(phdr); + return found; +} + +/* How many bytes the PT_LOAD and the PT_NOTE segments of @fd carry. */ +bool sum_coredump_segments(int fd, __u64 *data, __u64 *notes) +{ + ElfW(Phdr) *phdr; + size_t nr, i; + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + *data = 0; + *notes = 0; + for (i = 0; i < nr; i++) { + if (phdr[i].p_type == PT_LOAD) + *data += phdr[i].p_filesz; + else if (phdr[i].p_type == PT_NOTE) + *notes += phdr[i].p_filesz; + } + + free(phdr); + return true; +} + +/* The coredump in @fd is at least as long as every segment it declares. */ +bool check_coredump_extent(int fd) +{ + ElfW(Phdr) *phdr; + struct stat st; + size_t nr, i; + bool ok = true; + + if (fstat(fd, &st)) { + fprintf(stderr, "%s: fstat: %m\n", __func__); + return false; + } + + phdr = read_phdrs(fd, &nr); + if (!phdr) + return false; + + for (i = 0; i < nr; i++) { + if (phdr[i].p_offset + phdr[i].p_filesz <= (__u64)st.st_size) + continue; + fprintf(stderr, "%s: segment %zu ends at %llu, the coredump at %llu\n", + __func__, i, + (unsigned long long)(phdr[i].p_offset + phdr[i].p_filesz), + (unsigned long long)st.st_size); + ok = false; + } + + free(phdr); + return ok; +} + +/* The next stretch of memory the segments cover, split ones merged back. */ +static bool next_range(const ElfW(Phdr) *phdr, size_t nr, size_t *i, + __u64 *start, __u64 *end) +{ + while (*i < nr && phdr[*i].p_type != PT_LOAD) + (*i)++; + + if (*i >= nr) + return false; + + *start = phdr[*i].p_vaddr; + *end = phdr[*i].p_vaddr + phdr[*i].p_memsz; + (*i)++; + + while (*i < nr) { + if (phdr[*i].p_type != PT_LOAD) { + (*i)++; + continue; + } + if (phdr[*i].p_vaddr != *end) + break; + *end = phdr[*i].p_vaddr + phdr[*i].p_memsz; + (*i)++; + } + + return true; +} + +/* Compare @len bytes at @offset against @len bytes at @offset_ref. */ +static int compare_range(int fd, __u64 offset, int fd_ref, __u64 offset_ref, + __u64 len) +{ + char buffer[PAGE_SIZE], buffer_ref[PAGE_SIZE]; + + while (len) { + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + + if (pread(fd, buffer, chunk, offset) != (ssize_t)chunk || + pread(fd_ref, buffer_ref, chunk, offset_ref) != (ssize_t)chunk) { + fprintf(stderr, "%s: short read at %llu: %m\n", + __func__, (unsigned long long)offset); + return -1; + } + + if (memcmp(buffer, buffer_ref, chunk)) { + fprintf(stderr, "%s: %llu differs from %llu\n", __func__, + (unsigned long long)offset, + (unsigned long long)offset_ref); + return -1; + } + + offset += chunk; + offset_ref += chunk; + len -= chunk; + } + + return 0; +} + +/* The @len bytes at @offset the object left out have to have been zeroes. */ +static int check_zero_range(int fd, __u64 offset, __u64 len) +{ + static const char zeroes[PAGE_SIZE]; + char buffer[PAGE_SIZE]; + + while (len) { + size_t chunk = len < sizeof(buffer) ? len : sizeof(buffer); + + if (pread(fd, buffer, chunk, offset) != (ssize_t)chunk) { + fprintf(stderr, "%s: short read at %llu: %m\n", + __func__, (unsigned long long)offset); + return -1; + } + + if (memcmp(buffer, zeroes, chunk)) { + fprintf(stderr, "%s: %llu isn't a hole\n", __func__, + (unsigned long long)offset); + return -1; + } + + offset += chunk; + len -= chunk; + } + + return 0; +} + +/* + * The object has to describe the same memory as the coredump it was built + * from, and it has to describe it correctly. + */ +int check_compact_coredump(int fd_object, int fd_reference) +{ + ElfW(Phdr) *object = NULL, *reference = NULL; + size_t nr_object, nr_reference, i; + size_t io = 0, ir = 0; + int ret = -1; + + object = read_phdrs(fd_object, &nr_object); + reference = read_phdrs(fd_reference, &nr_reference); + if (!object || !reference) + goto out; + + /* Nothing may have been dropped and nothing may have been added. */ + for (;;) { + __u64 start = 0, end = 0, start_ref = 0, end_ref = 0; + bool has, has_ref; + + has = next_range(object, nr_object, &io, &start, &end); + has_ref = next_range(reference, nr_reference, &ir, &start_ref, + &end_ref); + if (!has && !has_ref) + break; + + if (has != has_ref || start != start_ref || end != end_ref) { + fprintf(stderr, "%s: object covers 0x%llx-0x%llx, coredump 0x%llx-0x%llx\n", + __func__, (unsigned long long)start, + (unsigned long long)end, + (unsigned long long)start_ref, + (unsigned long long)end_ref); + goto out; + } + } + + for (i = 0; i < nr_object; i++) { + const ElfW(Phdr) *segment; + __u64 offset, dumped; + + if (object[i].p_type != PT_LOAD || !object[i].p_memsz) + continue; + + segment = find_segment(reference, nr_reference, + object[i].p_vaddr); + if (!segment) { + fprintf(stderr, "%s: 0x%llx isn't in the coredump\n", + __func__, + (unsigned long long)object[i].p_vaddr); + goto out; + } + + offset = object[i].p_vaddr - segment->p_vaddr; + dumped = offset < segment->p_filesz ? + segment->p_filesz - offset : 0; + + /* What the object carries is what the coredump had. */ + if (object[i].p_filesz > dumped) { + fprintf(stderr, "%s: object carries %llu bytes the coredump doesn't have\n", + __func__, + (unsigned long long)(object[i].p_filesz - dumped)); + goto out; + } + + if (compare_range(fd_object, object[i].p_offset, fd_reference, + segment->p_offset + offset, + object[i].p_filesz)) + goto out; + + /* And what it left out was a hole. */ + if (object[i].p_memsz > object[i].p_filesz && + dumped > object[i].p_filesz) { + __u64 left_out = dumped - object[i].p_filesz; + + if (left_out > object[i].p_memsz - object[i].p_filesz) + left_out = object[i].p_memsz - object[i].p_filesz; + + if (check_zero_range(fd_reference, + segment->p_offset + offset + + object[i].p_filesz, left_out)) + goto out; + } + } + + ret = 0; +out: + free(object); + free(reference); + return ret; +} + +/* Read a plain coredump byte stream to end-of-file. */ +ssize_t recv_coredump_bytes(int fd_coredump, int fd_core_file) +{ + ssize_t received = 0; + + for (;;) { + char buffer[PAGE_SIZE]; + ssize_t ret = read_nointr(fd_coredump, buffer, sizeof(buffer)); + + if (ret < 0) { + fprintf(stderr, "%s: read failed: %m\n", __func__); + return -1; + } + if (ret == 0) + break; + + if (write_nointr(fd_core_file, buffer, ret) != ret) { + fprintf(stderr, "%s: write failed: %m\n", __func__); + return -1; + } + received += ret; + } + + fprintf(stderr, "Received %zd bytes of coredump\n", received); + return received; +} + int create_detached_tmpfs(void) { int fd_context, fd_tmpfs; @@ -101,6 +1290,7 @@ int create_and_listen_unix_socket(const char *path) return fd; out: + fprintf(stderr, "%s: %s: %m\n", __func__, path); if (fd >= 0) close(fd); return -1; @@ -153,8 +1343,95 @@ bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info) return true; } +/* + * How much the peer has mapped. The task is parked in the coredump + * handshake, so its mm is still there to be looked at. + */ +ssize_t peer_vm_size(int fd_peer_pidfd) +{ + struct pidfd_info info = {}; + unsigned long pages; + char path[64]; + FILE *f; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return -1; + + snprintf(path, sizeof(path), "/proc/%d/statm", info.pid); + f = fopen(path, "r"); + if (!f) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return -1; + } + + if (fscanf(f, "%lu", &pages) != 1) { + fprintf(stderr, "%s: %s: no size\n", __func__, path); + fclose(f); + return -1; + } + fclose(f); + + return (ssize_t)pages * sysconf(_SC_PAGESIZE); +} + /* Protocol helper functions */ +/* The peer's /proc/<pid>/coredump_filter, which is in memory types. */ +bool peer_coredump_filter(int fd_peer_pidfd, __u64 *memory_types) +{ + struct pidfd_info info = {}; + unsigned long value; + char path[64]; + FILE *f; + int ret; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return false; + + snprintf(path, sizeof(path), "/proc/%d/coredump_filter", info.pid); + f = fopen(path, "r"); + if (!f) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return false; + } + + ret = fscanf(f, "%lx", &value); + fclose(f); + if (ret != 1) { + fprintf(stderr, "%s: %s: no value\n", __func__, path); + return false; + } + + *memory_types = value; + return true; +} + +/* Read @len bytes at @addr from the peer's /proc/<pid>/mem. */ +ssize_t peer_read_mem(int fd_peer_pidfd, __u64 addr, void *buf, size_t len) +{ + struct pidfd_info info = {}; + char path[64]; + ssize_t ret; + int fd; + + if (!get_pidfd_info(fd_peer_pidfd, &info)) + return -1; + + snprintf(path, sizeof(path), "/proc/%d/mem", info.pid); + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) { + fprintf(stderr, "%s: %s: %m\n", __func__, path); + return -1; + } + + ret = pread(fd, buf, len, addr); + if (ret < 0) + fprintf(stderr, "%s: %s at 0x%llx: %m\n", __func__, path, + (unsigned long long)addr); + close(fd); + return ret; +} + ssize_t recv_marker(int fd) { enum coredump_mark mark = COREDUMP_MARK_REQACK; @@ -197,10 +1474,34 @@ bool read_marker(int fd, enum coredump_mark mark) return ret == mark; } -bool read_coredump_req(int fd, struct coredump_req *req) +/* + * The kernel hung up without sending anything more: end of stream, or a + * reset if it refused the ack on its peeked size and never read it. + */ +bool read_hangup(int fd) { ssize_t ret; - size_t field_size, user_size, ack_size, kernel_size, remaining_size; + char c; + + ret = recv(fd, &c, sizeof(c), MSG_WAITALL); + if (ret == 0) { + fprintf(stderr, "Kernel closed the connection\n"); + return true; + } + if (ret < 0 && errno == ECONNRESET) { + fprintf(stderr, "Kernel closed the connection with the ack unread\n"); + return true; + } + + fprintf(stderr, "%s: expected a hangup, got %zd: %m\n", __func__, ret); + return false; +} + +/* Read the request as a server built with a @user_size byte struct does. */ +bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size) +{ + ssize_t ret; + size_t field_size, known_size, kernel_size, remaining_size; memset(req, 0, sizeof(*req)); field_size = sizeof(req->size); @@ -208,37 +1509,36 @@ bool read_coredump_req(int fd, struct coredump_req *req) /* Peek the size of the coredump request. */ ret = recv(fd, req, field_size, MSG_PEEK | MSG_WAITALL); if (ret != field_size) { - fprintf(stderr, "read_coredump_req: peek failed (got %zd, expected %zu): %m\n", + fprintf(stderr, "%s: peek failed (got %zd, expected %zu): %m\n", __func__, ret, field_size); return false; } kernel_size = req->size; - if (kernel_size < COREDUMP_ACK_SIZE_VER0) { - fprintf(stderr, "read_coredump_req: kernel_size %zu < min %d\n", - kernel_size, COREDUMP_ACK_SIZE_VER0); + if (kernel_size < COREDUMP_REQ_SIZE_VER0) { + fprintf(stderr, "%s: kernel_size %zu < min %d\n", __func__, + kernel_size, COREDUMP_REQ_SIZE_VER0); return false; } if (kernel_size >= PAGE_SIZE) { - fprintf(stderr, "read_coredump_req: kernel_size %zu >= PAGE_SIZE %d\n", + fprintf(stderr, "%s: kernel_size %zu >= PAGE_SIZE %d\n", __func__, kernel_size, PAGE_SIZE); return false; } - /* Use the minimum of user and kernel size to read the full request. */ - user_size = sizeof(struct coredump_req); - ack_size = user_size < kernel_size ? user_size : kernel_size; - ret = recv(fd, req, ack_size, MSG_WAITALL); - if (ret != ack_size) + /* Consume as much of the request as we know about. */ + known_size = user_size < kernel_size ? user_size : kernel_size; + ret = recv(fd, req, known_size, MSG_WAITALL); + if (ret != known_size) return false; fprintf(stderr, "Read coredump request with size %u and mask 0x%llx\n", req->size, (unsigned long long)req->mask); - if (user_size > kernel_size) - remaining_size = user_size - kernel_size; - else + if (kernel_size > user_size) remaining_size = kernel_size - user_size; + else + remaining_size = 0; if (PAGE_SIZE <= remaining_size) return false; @@ -250,7 +1550,7 @@ bool read_coredump_req(int fd, struct coredump_req *req) if (remaining_size) { char buffer[PAGE_SIZE]; - ret = recv(fd, buffer, sizeof(buffer), MSG_WAITALL); + ret = recv(fd, buffer, remaining_size, MSG_WAITALL); if (ret != remaining_size) return false; fprintf(stderr, "Discarded %zu bytes of data after coredump request\n", remaining_size); @@ -259,8 +1559,13 @@ bool read_coredump_req(int fd, struct coredump_req *req) return true; } -bool send_coredump_ack(int fd, const struct coredump_req *req, - __u64 mask, size_t size_ack) +bool read_coredump_req(int fd, struct coredump_req *req) +{ + return read_coredump_req_sized(fd, req, sizeof(*req)); +} + +/* Send @len bytes of @ack as they are, more than the struct if asked to. */ +bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len) { ssize_t ret; /* @@ -272,30 +1577,74 @@ bool send_coredump_ack(int fd, const struct coredump_req *req, char buffer[PAGE_SIZE]; } large_ack = {}; + if (len > sizeof(large_ack)) + return false; + + large_ack.ack = *ack; + ret = send(fd, &large_ack, len, MSG_NOSIGNAL); + if (ret != len) { + fprintf(stderr, "%s: short send %zd: %m\n", __func__, ret); + return false; + } + + fprintf(stderr, "Sent %zu bytes of coredump ack: size %u, mask 0x%llx, types 0x%llx\n", + len, ack->size, (unsigned long long)ack->mask, + (unsigned long long)ack->memory_types); + return true; +} + +bool send_coredump_ack_types(int fd, const struct coredump_req *req, + __u64 mask, __u64 memory_types, size_t size_ack) +{ + struct coredump_ack ack = { + .mask = mask, + .memory_types = memory_types, + }; + if (!size_ack) size_ack = sizeof(struct coredump_ack) < req->size_ack ? sizeof(struct coredump_ack) : req->size_ack; - large_ack.ack.mask = mask; - large_ack.ack.size = size_ack; - ret = send(fd, &large_ack, size_ack, MSG_NOSIGNAL); - if (ret != size_ack) - return false; + ack.size = size_ack; + return send_coredump_ack_bytes(fd, &ack, size_ack); +} - fprintf(stderr, "Sent coredump ack with size %zu and mask 0x%llx\n", - size_ack, (unsigned long long)mask); - return true; +bool send_coredump_ack(int fd, const struct coredump_req *req, + __u64 mask, size_t size_ack) +{ + return send_coredump_ack_types(fd, req, mask, 0, size_ack); } -bool check_coredump_req(const struct coredump_req *req, size_t min_size, - __u64 required_mask) +/* Every option the kernel is expected to advertise in coredump_req->mask. */ +#define TEST_REQ_MASK_ALL \ + (COREDUMP_KERNEL | COREDUMP_USERSPACE | \ + COREDUMP_REJECT | COREDUMP_WAIT | \ + COREDUMP_RECORDS | COREDUMP_SPARSE | COREDUMP_MEMORY_TYPES) + +bool check_coredump_req(const struct coredump_req *req) { - if (req->size < min_size) + if (req->size != COREDUMP_REQ_SIZE_VER1) { + fprintf(stderr, "%s: size %u, expected %d\n", + __func__, req->size, COREDUMP_REQ_SIZE_VER1); return false; - if ((req->mask & required_mask) != required_mask) + } + if (req->size_ack != COREDUMP_ACK_SIZE_VER1) { + fprintf(stderr, "%s: size_ack %u, expected %d\n", + __func__, req->size_ack, COREDUMP_ACK_SIZE_VER1); + return false; + } + if (req->mask != TEST_REQ_MASK_ALL) { + fprintf(stderr, "%s: mask 0x%llx, expected 0x%llx\n", + __func__, (unsigned long long)req->mask, + (unsigned long long)TEST_REQ_MASK_ALL); return false; - if (req->mask & ~required_mask) + } + if (req->memory_types_mask != TEST_MEMORY_ALL) { + fprintf(stderr, "%s: memory_types_mask 0x%llx, expected 0x%llx\n", + __func__, (unsigned long long)req->memory_types_mask, + (unsigned long long)TEST_MEMORY_ALL); return false; + } return true; } @@ -381,3 +1730,304 @@ out: close(fd_coredump); _exit(exit_code); } + +/* + * TIF_NOTIFY_SIGNAL coredump helpers. + * + * __dump_emit() takes anything short of a full write as the end of the + * dump, so every emit that blocks on a full transport can lose the rest + * of it. The NT_FILE note is the one emit that is certain to block, + * because it is the only one larger than the transport, so these helpers + * build a note large enough for that and then raise TIF_NOTIFY_SIGNAL on + * the dumping task while that write is in flight. + */ + +static int io_uring_setup_raw(unsigned int entries, struct io_uring_params *p) +{ + return syscall(__NR_io_uring_setup, entries, p); +} + +/* io_uring reads poll32_events back through swahw32() on big endian. */ +static __u32 notify_poll_mask(__u32 events) +{ +#if defined(__BYTE_ORDER) && __BYTE_ORDER == __BIG_ENDIAN + return __swahw32(events); +#else + return events; +#endif +} + +bool coredump_io_uring_available(void) +{ + struct io_uring_params p = {}; + int fd; + + fd = io_uring_setup_raw(1, &p); + if (fd < 0) + return false; + close(fd); + return true; +} + +/* + * Arm a poll on @fd. io_uring leaves ctx->notify_method at TWA_SIGNAL + * unless the ring asks for SQPOLL or COOP_TASKRUN, so the completion runs + * set_notify_signal() against the task that submitted it. That is us, and + * we are about to become the coredumping task. + */ +static int arm_poll_notify(int trigger_fd) +{ + struct io_uring_params p = {}; + unsigned int *sq_tail, *sq_array; + struct io_uring_sqe *sqes; + size_t sqring_sz; + void *sq; + int ring; + + ring = io_uring_setup_raw(8, &p); + if (ring < 0) + return -1; + + sqring_sz = p.sq_off.array + p.sq_entries * sizeof(unsigned int); + sq = mmap(NULL, sqring_sz, PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_POPULATE, ring, IORING_OFF_SQ_RING); + if (sq == MAP_FAILED) + return -1; + + sqes = mmap(NULL, p.sq_entries * sizeof(*sqes), PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_POPULATE, ring, IORING_OFF_SQES); + if (sqes == MAP_FAILED) + return -1; + + sq_tail = (unsigned int *)((char *)sq + p.sq_off.tail); + sq_array = (unsigned int *)((char *)sq + p.sq_off.array); + + memset(&sqes[0], 0, sizeof(sqes[0])); + sqes[0].opcode = IORING_OP_POLL_ADD; + sqes[0].fd = trigger_fd; + sqes[0].poll32_events = notify_poll_mask(POLLIN); + + sq_array[0] = 0; + __atomic_store_n(sq_tail, 1, __ATOMIC_RELEASE); + + if (syscall(__NR_io_uring_enter, ring, 1, 0, 0, NULL, 0) < 0) + return -1; + + /* Deliberately leaked, we are about to crash. */ + return 0; +} + +/* + * Adjacent mappings with identical flags and contiguous file offsets are + * merged into one VMA, which would collapse NT_FILE back to nothing, so + * alternate the protection to keep every mapping an entry of its own. + * Not PROT_EXEC, /tmp is often mounted noexec. Nothing is ever written + * through these so they get no anon_vma and stay out of the dump itself. + */ +static int make_file_mappings(void) +{ + long pgsz = sysconf(_SC_PAGESIZE); + int fd, i; + + fd = open(NOTIFY_SIGNAL_MAPFILE, + O_RDWR | O_CREAT | O_TRUNC | O_CLOEXEC, 0600); + if (fd < 0) + return 0; + if (ftruncate(fd, (off_t)NOTIFY_SIGNAL_MAP_COUNT * pgsz)) { + close(fd); + return 0; + } + + for (i = 0; i < NOTIFY_SIGNAL_MAP_COUNT; i++) { + int prot = (i & 1) ? PROT_READ : (PROT_READ | PROT_WRITE); + + if (mmap(NULL, pgsz, prot, MAP_PRIVATE, fd, + (off_t)i * pgsz) == MAP_FAILED) + break; + } + close(fd); + return i; +} + +void crashing_child_notify_signal(void) +{ + long pgsz = sysconf(_SC_PAGESIZE); + unsigned char *p; + int trigger_fd; + unsigned long off; + + /* Open the read side first so the reader's open() cannot block. */ + trigger_fd = open(NOTIFY_SIGNAL_TRIGGER, O_RDONLY | O_NONBLOCK | O_CLOEXEC); + + /* Exit rather than crash: a dump without the poll armed proves nothing. */ + if (trigger_fd < 0) + _exit(EXIT_FAILURE); + + if (make_file_mappings() < NOTIFY_SIGNAL_MAP_COUNT) + _exit(EXIT_FAILURE); + + p = mmap(NULL, NOTIFY_SIGNAL_ANON_BYTES, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); + if (p == MAP_FAILED) + _exit(EXIT_FAILURE); + for (off = 0; off < NOTIFY_SIGNAL_ANON_BYTES; off += pgsz) + p[off] = 1; + + if (arm_poll_notify(trigger_fd)) + _exit(EXIT_FAILURE); + + /* crash on purpose */ + *(volatile int *)NULL = 0; + _exit(EXIT_FAILURE); +} + +static int pull_trigger(void) +{ + int fd; + + fd = open(NOTIFY_SIGNAL_TRIGGER, O_WRONLY | O_NONBLOCK | O_CLOEXEC); + if (fd < 0) { + fprintf(stderr, "%s: open failed: %m\n", __func__); + return -1; + } + if (write(fd, "x", 1) != 1) { + fprintf(stderr, "%s: write failed: %m\n", __func__); + close(fd); + return -1; + } + close(fd); + return 0; +} + +/* + * phdr[0] is the PT_NOTE entry: elf_core_dump() emits it right after the + * ELF header. Its p_offset is where the notes begin. + */ +static long long note_offset(const unsigned char *hdr) +{ + ElfW(Phdr) ph; + ElfW(Ehdr) eh; + + memcpy(&eh, hdr, sizeof(eh)); + if (memcmp(eh.e_ident, ELFMAG, SELFMAG) || eh.e_type != ET_CORE) + return -1; + memcpy(&ph, hdr + sizeof(eh), sizeof(ph)); + if (ph.p_type != PT_NOTE) + return -1; + return (long long)ph.p_offset; +} + +/* + * Drain a coredump off @fd, counting what arrives. Once the notes have + * started the kernel is inside the one big note write, so poke the fifo + * the crashing task is polling and stop reading, which keeps the + * transport full and the write blocked with a partial count when the + * wakeup lands. Then carry on to end of file. + * + * What arrives is written to @fd_out when that is not negative. + * Returns the number of bytes received, or -1. Failing to trip the fifo + * is an error too: a dump that was never interrupted proves nothing. + */ +ssize_t recv_coredump_notify_signal(int fd, int fd_out, bool arm) +{ + unsigned char hdr[sizeof(ElfW(Ehdr)) + sizeof(ElfW(Phdr))]; + static char buf[64 << 10]; + long pgsz = sysconf(_SC_PAGESIZE); + long long note_off = 0; + size_t hdrlen = 0; + ssize_t total = 0; + bool armed = false; + + for (;;) { + ssize_t n = read(fd, buf, sizeof(buf)); + + if (n < 0) { + if (errno == EINTR) + continue; + return -1; + } + if (n == 0) + break; + if (fd_out >= 0 && write(fd_out, buf, n) != n) + return -1; + + if (hdrlen < sizeof(hdr)) { + size_t want = sizeof(hdr) - hdrlen; + + if (want > (size_t)n) + want = (size_t)n; + memcpy(hdr + hdrlen, buf, want); + hdrlen += want; + if (hdrlen == sizeof(hdr)) + note_off = note_offset(hdr); + } + + total += n; + + if (arm && !armed && note_off > 0 && + total > note_off + (long long)pgsz) { + if (pull_trigger()) + return -1; + armed = true; + usleep(NOTIFY_SIGNAL_STALL_US); + continue; + } + } + + if (arm && !armed) + return -1; + + return total; +} + +/* + * How large the dump was meant to be. The ELF header and the program + * headers are the first thing emitted, so even a truncated dump says how + * far it should have run: the end is max(p_offset + p_filesz). + * + * That end is exact even when the last segment ends in a hole: + * coredump_write() flushes the pending cprm->to_skip with a final one + * byte emit and __dump_skip() writes zeroes for transports that cannot + * seek, so a whole dump carries every byte the headers promise. + */ +long long coredump_expected_size(const char *path) +{ + ElfW(Phdr) *phdr = NULL; + long long expected = 0; + unsigned int nphdr, i; + ElfW(Ehdr) eh; + int fd; + + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) + return -1; + if (read(fd, &eh, sizeof(eh)) != sizeof(eh)) + goto err; + if (memcmp(eh.e_ident, ELFMAG, SELFMAG) || eh.e_type != ET_CORE) + goto err; + if (!eh.e_phnum || eh.e_phentsize != sizeof(*phdr)) + goto err; + + nphdr = eh.e_phnum; + phdr = calloc(nphdr, sizeof(*phdr)); + if (!phdr) + goto err; + if (pread(fd, phdr, (size_t)nphdr * sizeof(*phdr), (off_t)eh.e_phoff) != + (ssize_t)((size_t)nphdr * sizeof(*phdr))) + goto err; + + for (i = 0; i < nphdr; i++) { + long long end = (long long)phdr[i].p_offset + + (long long)phdr[i].p_filesz; + if (end > expected) + expected = end; + } + + free(phdr); + close(fd); + return expected; +err: + free(phdr); + close(fd); + return -1; +} diff --git a/tools/testing/selftests/coredump/coredump_test_helpers.h b/tools/testing/selftests/coredump/coredump_test_helpers.h new file mode 100644 index 000000000000..3f2f87837558 --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_test_helpers.h @@ -0,0 +1,79 @@ +/* SPDX-License-Identifier: GPL-2.0 */ + +#ifndef __COREDUMP_TEST_HELPERS_H +#define __COREDUMP_TEST_HELPERS_H + +#include <link.h> +#include <stdbool.h> +#include <sys/types.h> +#include <linux/coredump.h> + +#include "../pidfd/pidfd.h" + +#ifndef PAGE_SIZE +#define PAGE_SIZE 4096 +#endif + +#define NUM_THREAD_SPAWN 128 + +/* Size of the mostly unpopulated mapping the sparse coredump test maps. */ +#define SPARSE_MAPPING_SIZE (256 * 1024 * 1024) + +/* A task mapping at least this much is worth a record stream. */ +#define SPARSE_STREAM_THRESHOLD (SPARSE_MAPPING_SIZE / 2) + +/* Size of the shared anonymous mapping the memory types tests map. */ +#define MEMORY_MAPPING_SIZE (4 * 1024 * 1024) + +/* Leave the coredump_filter the crashing child inherited alone. */ +#define FILTER_TASK_INHERIT ((__u64)-1) + +/* Every memory type the kernel is expected to advertise. */ +#define TEST_MEMORY_ALL \ + (COREDUMP_MEMORY_ANON_PRIVATE | COREDUMP_MEMORY_ANON_SHARED | \ + COREDUMP_MEMORY_FILE_PRIVATE | COREDUMP_MEMORY_FILE_SHARED | \ + COREDUMP_MEMORY_ELF_HEADERS | \ + COREDUMP_MEMORY_HUGETLB_PRIVATE | COREDUMP_MEMORY_HUGETLB_SHARED | \ + COREDUMP_MEMORY_DAX_PRIVATE | COREDUMP_MEMORY_DAX_SHARED) + +/* Shared helper function declarations */ +void *do_nothing(void *arg); +void crashing_child(void); +void crashing_child_sparse(size_t size); +void crashing_child_memory(__u64 task_filter, int fd_addr); +bool find_coredump_segment(int fd, __u64 vaddr, ElfW(Phdr) *segment); +bool sum_coredump_segments(int fd, __u64 *data, __u64 *notes); +bool check_coredump_extent(int fd); +bool peer_coredump_filter(int fd_peer_pidfd, __u64 *memory_types); +ssize_t peer_read_mem(int fd_peer_pidfd, __u64 addr, void *buf, size_t len); +ssize_t recv_coredump_records(int fd_coredump, int fd_core_file, + off_t *coredump_size, bool *truncated, + int fd_peer_pidfd); +ssize_t recv_coredump_compact(int fd_coredump, int fd_object, int fd_reference, + off_t *coredump_size); +ssize_t recv_coredump_bytes(int fd_coredump, int fd_core_file); +ssize_t peer_vm_size(int fd_peer_pidfd); +bool is_elf_core(int fd); +int check_compact_coredump(int fd_object, int fd_reference); +int create_detached_tmpfs(void); +int create_and_listen_unix_socket(const char *path); +bool set_core_pattern(const char *pattern); +int get_peer_pidfd(int fd); +bool get_pidfd_info(int fd_peer_pidfd, struct pidfd_info *info); + +/* Protocol helper function declarations */ +ssize_t recv_marker(int fd); +bool read_marker(int fd, enum coredump_mark mark); +bool read_hangup(int fd); +bool read_coredump_req(int fd, struct coredump_req *req); +bool read_coredump_req_sized(int fd, struct coredump_req *req, size_t user_size); +bool send_coredump_ack(int fd, const struct coredump_req *req, + __u64 mask, size_t size_ack); +bool send_coredump_ack_types(int fd, const struct coredump_req *req, + __u64 mask, __u64 memory_types, size_t size_ack); +bool send_coredump_ack_bytes(int fd, const struct coredump_ack *ack, size_t len); +bool check_coredump_req(const struct coredump_req *req); +int open_coredump_tmpfile(int fd_tmpfs_detached); +void process_coredump_worker(int fd_coredump, int fd_peer_pidfd, int fd_core_file); + +#endif /* __COREDUMP_TEST_HELPERS_H */ diff --git a/tools/testing/selftests/coredump/coredump_worker_test.c b/tools/testing/selftests/coredump/coredump_worker_test.c new file mode 100644 index 000000000000..9a4270b65a6e --- /dev/null +++ b/tools/testing/selftests/coredump/coredump_worker_test.c @@ -0,0 +1,447 @@ +// SPDX-License-Identifier: GPL-2.0 + +/* + * A user worker as the coredumping thread. + * + * io-wq workers and SQPOLL threads are threads of the process that never + * return to userspace. They block every signal but SIGKILL and SIGSTOP, + * but a tracer can replace that mask with PTRACE_SETSIGMASK and inject a + * coredump signal. get_signal() then runs vfs_coredump() in the worker. + * The worker's own exit bookkeeping runs only after the dump, so a + * zapped sibling that waits for it in its exit path deadlocks with the + * dumper and the whole thread group is stuck in D state. + * + * Inject SIGSEGV into a chosen thread and require that the thread group + * is gone in bounded time, either because the dump completed or because + * SIGKILL still works. A failure leaves the stuck process behind. + */ +#include <ctype.h> +#include <errno.h> +#include <dirent.h> +#include <fcntl.h> +#include <sys/mman.h> +#include <sys/ptrace.h> +#include <sys/resource.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> +#include <unistd.h> +#include <linux/io_uring.h> + +#include "coredump_test.h" + +#ifndef PTRACE_SETSIGMASK +#define PTRACE_SETSIGMASK 0x420b +#endif + +/* The dump of the tiny child takes well under a second. */ +#define EXIT_TIMEOUT_MS 5000 + +FIXTURE_SETUP(coredump) +{ + FILE *file; + int ret; + + self->pid_coredump_server = -ESRCH; + self->fd_tmpfs_detached = -1; + file = fopen("/proc/sys/kernel/core_pattern", "r"); + ASSERT_NE(NULL, file); + + ret = fread(self->original_core_pattern, 1, sizeof(self->original_core_pattern), file); + ASSERT_TRUE(ret || feof(file)); + ASSERT_LT(ret, sizeof(self->original_core_pattern)); + + self->original_core_pattern[ret] = '\0'; + + ret = fclose(file); + ASSERT_EQ(0, ret); +} + +FIXTURE_TEARDOWN(coredump) +{ + const char *reason; + FILE *file; + int ret; + + file = fopen("/proc/sys/kernel/core_pattern", "w"); + if (!file) { + reason = "Unable to open core_pattern"; + goto fail; + } + + ret = fprintf(file, "%s", self->original_core_pattern); + if (ret < 0) { + reason = "Unable to write to core_pattern"; + goto fail; + } + + ret = fclose(file); + if (ret) { + reason = "Unable to close core_pattern"; + goto fail; + } + + return; +fail: + /* This should never happen */ + fprintf(stderr, "Failed to cleanup coredump test: %s\n", reason); +} + +/* A raw ring, no liburing. */ +struct uring { + int fd; + struct io_uring_params params; + void *sq; + size_t sq_len; + struct io_uring_sqe *sqes; + size_t sqes_len; + unsigned int *sq_tail, *sq_mask, *sq_array; + unsigned int *cq_head, *cq_tail, *cq_mask; + struct io_uring_cqe *cqes; +}; + +static int uring_setup(struct uring *r, unsigned int flags) +{ + size_t cq_len; + + memset(r, 0, sizeof(*r)); + r->params.flags = flags; + if (flags & IORING_SETUP_SQPOLL) + r->params.sq_thread_idle = 2000; + r->fd = syscall(__NR_io_uring_setup, 8, &r->params); + if (r->fd < 0) + return -1; + if (!(r->params.features & IORING_FEAT_SINGLE_MMAP)) + return -1; + + r->sq_len = r->params.sq_off.array + r->params.sq_entries * sizeof(unsigned int); + cq_len = r->params.cq_off.cqes + r->params.cq_entries * sizeof(struct io_uring_cqe); + if (cq_len > r->sq_len) + r->sq_len = cq_len; + r->sq = mmap(NULL, r->sq_len, PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_POPULATE, r->fd, IORING_OFF_SQ_RING); + if (r->sq == MAP_FAILED) + return -1; + r->sqes_len = r->params.sq_entries * sizeof(struct io_uring_sqe); + r->sqes = mmap(NULL, r->sqes_len, PROT_READ | PROT_WRITE, + MAP_SHARED | MAP_POPULATE, r->fd, IORING_OFF_SQES); + if (r->sqes == MAP_FAILED) + return -1; + + r->sq_tail = r->sq + r->params.sq_off.tail; + r->sq_mask = r->sq + r->params.sq_off.ring_mask; + r->sq_array = r->sq + r->params.sq_off.array; + r->cq_head = r->sq + r->params.cq_off.head; + r->cq_tail = r->sq + r->params.cq_off.tail; + r->cq_mask = r->sq + r->params.cq_off.ring_mask; + r->cqes = r->sq + r->params.cq_off.cqes; + return 0; +} + +/* Submit one sqe, wait for its completion and return the result. */ +static int uring_submit_wait(struct uring *r, const struct io_uring_sqe *sqe) +{ + unsigned int tail = *r->sq_tail, idx = tail & *r->sq_mask; + unsigned int flags = IORING_ENTER_GETEVENTS; + int i; + + r->sqes[idx] = *sqe; + r->sq_array[idx] = idx; + __atomic_store_n(r->sq_tail, tail + 1, __ATOMIC_RELEASE); + + if (r->params.flags & IORING_SETUP_SQPOLL) + flags |= IORING_ENTER_SQ_WAKEUP; + + for (i = 0; i < 100; i++) { + if (syscall(__NR_io_uring_enter, r->fd, 1, 1, flags, NULL, 0) < 0 && + errno != EINTR) + return -1; + if (__atomic_load_n(r->cq_tail, __ATOMIC_ACQUIRE) != *r->cq_head) { + unsigned int head = *r->cq_head; + int res = r->cqes[head & *r->cq_mask].res; + + __atomic_store_n(r->cq_head, head + 1, __ATOMIC_RELEASE); + return res; + } + flags &= ~IORING_ENTER_SQ_WAKEUP; + usleep(10 * 1000); + } + return -1; +} + +static bool uring_available(unsigned int flags) +{ + struct io_uring_params params = { .flags = flags }; + int fd; + + fd = syscall(__NR_io_uring_setup, 2, ¶ms); + if (fd < 0) + return false; + close(fd); + return true; +} + +/* + * Keep a ring, an idle io-wq worker and with SQPOLL an SQPOLL thread + * alive. The last worker of a ring never exits on its idle timeout. + */ +static void worker_child(bool sqpoll, int fd_ipc) +{ + struct rlimit rl = { RLIM_INFINITY, RLIM_INFINITY }; + struct io_uring_sqe sqe = {}; + static char buf[64]; + struct uring ring; + int memfd; + + if (setrlimit(RLIMIT_CORE, &rl)) + _exit(EXIT_FAILURE); + + memfd = memfd_create("coredump_worker", 0); + if (memfd < 0 || write(memfd, "hello", 5) != 5) + _exit(EXIT_FAILURE); + + if (uring_setup(&ring, sqpoll ? IORING_SETUP_SQPOLL : 0)) + _exit(EXIT_FAILURE); + + /* IOSQE_ASYNC forces the read through io-wq so a worker appears. */ + sqe.opcode = IORING_OP_READ; + sqe.fd = memfd; + sqe.addr = (__u64)(uintptr_t)buf; + sqe.len = sizeof(buf); + sqe.flags = IOSQE_ASYNC; + if (uring_submit_wait(&ring, &sqe) != 5) + _exit(EXIT_FAILURE); + + if (write_nointr(fd_ipc, "1", 1) != 1) + _exit(EXIT_FAILURE); + close(fd_ipc); + + for (;;) + pause(); +} + +/* Find the thread of @pid whose comm starts with @prefix. */ +static pid_t find_thread(pid_t pid, const char *prefix) +{ + char path[64], comm[64]; + pid_t tid = -1; + struct dirent *de; + ssize_t bytes; + DIR *dir; + int fd; + + snprintf(path, sizeof(path), "/proc/%d/task", pid); + dir = opendir(path); + if (!dir) + return -1; + + while (tid < 0 && (de = readdir(dir))) { + if (!isdigit(de->d_name[0])) + continue; + snprintf(path, sizeof(path), "/proc/%d/task/%s/comm", pid, de->d_name); + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) + continue; + bytes = read(fd, comm, sizeof(comm) - 1); + close(fd); + if (bytes <= 0) + continue; + comm[bytes] = '\0'; + if (!strncmp(comm, prefix, strlen(prefix))) + tid = atoi(de->d_name); + } + closedir(dir); + return tid; +} + +/* + * Attach, stop the thread with SIGSTOP, drop the signal mask that + * copy_process() gave it and resume it with SIGSEGV. Returns 1 when the + * mask was changed, 0 when PTRACE_SETSIGMASK was refused (a user worker + * keeps its mask and the SIGSEGV stays pending), -1 on any other failure. + */ +static int inject_coredump_signal(pid_t pid, pid_t tid) +{ + __u64 mask = 0; + int status, ret = 1; + + if (ptrace(PTRACE_SEIZE, tid, NULL, NULL)) + return -1; + if (syscall(SYS_tgkill, pid, tid, SIGSTOP)) + return -1; + if (waitpid(tid, &status, __WALL) != tid) + return -1; + if (!WIFSTOPPED(status) || WSTOPSIG(status) != SIGSTOP) + return -1; + if (ptrace(PTRACE_SETSIGMASK, tid, sizeof(mask), &mask)) { + if (errno != EPERM) + return -1; + ret = 0; + } + if (ptrace(PTRACE_DETACH, tid, NULL, (void *)(long)SIGSEGV)) + return -1; + return ret; +} + +/* Reap @pid within @timeout_ms, -1 when it is still there. */ +static int wait_exit(pid_t pid, int *status, int timeout_ms) +{ + int i; + + for (i = 0; i < timeout_ms / 10; i++) { + pid_t ret = waitpid(pid, status, WNOHANG); + + if (ret == pid) + return 0; + if (ret < 0) + return -1; + usleep(10 * 1000); + } + return -1; +} + +static void log_threads(struct __test_metadata *const _metadata, pid_t pid) +{ + char path[64], line[256], comm[64] = {}; + struct dirent *de; + DIR *dir; + FILE *f; + + snprintf(path, sizeof(path), "/proc/%d/task", pid); + dir = opendir(path); + if (!dir) + return; + while ((de = readdir(dir))) { + if (!isdigit(de->d_name[0])) + continue; + snprintf(path, sizeof(path), "/proc/%d/task/%s/status", pid, de->d_name); + f = fopen(path, "r"); + if (!f) + continue; + while (fgets(line, sizeof(line), f)) { + line[strcspn(line, "\n")] = '\0'; + if (!strncmp(line, "Name:", 5)) + snprintf(comm, sizeof(comm), "%s", line + 6); + else if (!strncmp(line, "State:", 6)) + TH_LOG("tid %s (%s) %s", de->d_name, comm, line + 7); + } + fclose(f); + } + closedir(dir); +} + +enum dumper { + DUMPER_MAIN, + DUMPER_WORKER, + DUMPER_SQPOLL, +}; + +static void run_dumper(struct __test_metadata *const _metadata, bool sqpoll, + enum dumper dumper) +{ + bool killed = false; + char path[64], c; + int ipc[2], status, fd, ret; + pid_t pid, tid; + + ASSERT_TRUE(set_core_pattern("/tmp/coredump.file.%p")); + ASSERT_EQ(pipe2(ipc, O_CLOEXEC), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(ipc[0]); + worker_child(sqpoll, ipc[1]); + } + close(ipc[1]); + ASSERT_EQ(read_nointr(ipc[0], &c, 1), 1); + close(ipc[0]); + + switch (dumper) { + case DUMPER_MAIN: + tid = pid; + break; + case DUMPER_WORKER: + tid = find_thread(pid, "iou-wrk-"); + break; + case DUMPER_SQPOLL: + tid = find_thread(pid, "iou-sqp-"); + break; + } + ASSERT_GT(tid, 0); + ret = inject_coredump_signal(pid, tid); + ASSERT_GE(ret, 0); + if (!ret) { + /* The signal sits on the worker, the group must be untouched. */ + ASSERT_NE(dumper, DUMPER_MAIN); + TH_LOG("PTRACE_SETSIGMASK refused for tid %d, the SIGSEGV stays pending", tid); + ASSERT_EQ(wait_exit(pid, &status, 1000), -1); + kill(pid, SIGKILL); + ASSERT_EQ(wait_exit(pid, &status, EXIT_TIMEOUT_MS), 0); + ASSERT_TRUE(WIFSIGNALED(status)); + ASSERT_EQ(WTERMSIG(status), SIGKILL); + return; + } + + if (wait_exit(pid, &status, EXIT_TIMEOUT_MS)) { + /* No dump. Whatever happened, SIGKILL must still work. */ + log_threads(_metadata, pid); + kill(pid, SIGKILL); + killed = true; + ASSERT_EQ(wait_exit(pid, &status, EXIT_TIMEOUT_MS), 0) { + TH_LOG("thread group %d is stuck after SIGSEGV to tid %d", + pid, tid); + } + } + + ASSERT_TRUE(WIFSIGNALED(status)); + if (killed) { + TH_LOG("tid %d did not dump, the group was killed instead", tid); + ASSERT_EQ(WTERMSIG(status), SIGKILL); + return; + } + ASSERT_EQ(WTERMSIG(status), SIGSEGV); + ASSERT_TRUE(WCOREDUMP(status)); + + snprintf(path, sizeof(path), "/tmp/coredump.file.%d", pid); + fd = open(path, O_RDONLY | O_CLOEXEC); + unlink(path); + ASSERT_GE(fd, 0); + ASSERT_TRUE(check_coredump_extent(fd)); + close(fd); +} + +/* The mechanics: an injected SIGSEGV into a normal thread dumps core. */ +TEST_F(coredump, main_thread_dumper) +{ + if (!uring_available(0)) + SKIP(return, "io_uring is not available"); + run_dumper(_metadata, false, DUMPER_MAIN); +} + +TEST_F(coredump, plain_worker_dumper) +{ + if (!uring_available(0)) + SKIP(return, "io_uring is not available"); + run_dumper(_metadata, false, DUMPER_WORKER); +} + +TEST_F(coredump, sqpoll_thread_dumper) +{ + if (!uring_available(IORING_SETUP_SQPOLL)) + SKIP(return, "io_uring SQPOLL is not available"); + run_dumper(_metadata, true, DUMPER_SQPOLL); +} + +/* + * The SQPOLL thread leaves its loop on the zap and waits for its io-wq + * workers to exit before it parks. The dumping worker never does. + */ +TEST_F(coredump, sqpoll_worker_dumper) +{ + if (!uring_available(IORING_SETUP_SQPOLL)) + SKIP(return, "io_uring SQPOLL is not available"); + run_dumper(_metadata, true, DUMPER_WORKER); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/exec/Makefile b/tools/testing/selftests/exec/Makefile index b640af8f02b5..2220ed345e92 100644 --- a/tools/testing/selftests/exec/Makefile +++ b/tools/testing/selftests/exec/Makefile @@ -45,6 +45,10 @@ TEST_GEN_FILES += binfmt_transparent_interp TEST_GEN_PROGS += binfmt_misc_loader TEST_GEN_FILES += binfmt_loader_payload binfmt_loader_payload_static +# Only ASCII punctuation delimits the fields of a register string, so a new +# flag character cannot change which strings register. No bpf toolchain. +TEST_GEN_PROGS += binfmt_misc_delim + # binfmt_misc bpf-backed ('B') handler test: a libbpf harness plus its # struct_ops objects and the test interpreter/app it routes between. Only # built when clang, bpftool, the vmlinux BTF and libbpf are all present diff --git a/tools/testing/selftests/exec/binfmt_misc_delim.c b/tools/testing/selftests/exec/binfmt_misc_delim.c new file mode 100644 index 000000000000..ffc17cb78545 --- /dev/null +++ b/tools/testing/selftests/exec/binfmt_misc_delim.c @@ -0,0 +1,127 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Test which characters may delimit the fields of a register string. + */ +#define _GNU_SOURCE +#include <stdio.h> +#include <stdlib.h> + +#include "binfmt_misc_common.h" +#include "kselftest_harness.h" + +#define ENTRY "bmdelim" +/* Shares no character with the sets below, or a refusal proves nothing. */ +#define MAGIC "bmmagic" +#define INTERP "/bin/true" + +/* + * ASCII punctuation without '\' and '/'. The backslash is refused because + * it would cut a magic that uses \x to escape short. '/' is accepted + * but cannot delimit a rule that names an absolute interpreter. + */ +#define PUNCTUATION "!\"#$%&'()*+,-.:;<=>?@[]^_`{|}~" + +/* 'M', 'E' and 'B' name types, 'P' through 'D' are the flags. */ +#define LETTERS "MEBPOCFTLDqz" +#define DIGITS "0157" +#define WHITESPACE " \t\n" +#define CONTROL "\001\033\177" +#define NON_ASCII "\200\244\377" + +/* ':bmdelim:E::bmmagic::/bin/true:' with @del in place of every ':'. */ +static int register_with(char del) +{ + char rule[128]; + + snprintf(rule, sizeof(rule), "%c%s%cE%c%c%s%c%c%s%c", del, ENTRY, del, + del, del, MAGIC, del, del, INTERP, del); + return write_reg(rule); +} + +/* No character of @set may delimit a register string. */ +static void expect_refused(struct __test_metadata *_metadata, const char *set) +{ + const char *d; + + for (d = set; *d; d++) { + int rc = register_with(*d); + + EXPECT_EQ(rc, -1) + TH_LOG("%#x delimited a register string", + (unsigned char)*d); + if (rc == 0) { + unregister(ENTRY); + continue; + } + EXPECT_EQ(errno, EINVAL); + } +} + +FIXTURE(delim) { +}; + +FIXTURE_SETUP(delim) +{ + if (getuid() != 0) + SKIP(return, "test must be run as root"); + if (!binfmt_misc_available()) + SKIP(return, "no binfmt_misc"); + + /* A kernel without the allow-list takes any character but a flag. */ + if (register_with('q') == 0) { + unregister(ENTRY); + SKIP(return, "kernel without the delimiter allow-list"); + } +} + +FIXTURE_TEARDOWN(delim) +{ + unregister(ENTRY); +} + +/* Punctuation delimits, which is all anything deployed ever uses. */ +TEST_F(delim, punctuation_accepted) +{ + const char *d; + + for (d = PUNCTUATION; *d; d++) { + EXPECT_EQ(register_with(*d), 0) + TH_LOG("'%c' refused with errno %d", *d, errno); + unregister(ENTRY); + } +} + +/* Letters name the types and the flags, so none of them can delimit. */ +TEST_F(delim, letters_refused) +{ + expect_refused(_metadata, LETTERS); +} + +/* The offset field is written in digits. */ +TEST_F(delim, digits_refused) +{ + expect_refused(_metadata, DIGITS); +} + +TEST_F(delim, whitespace_refused) +{ + expect_refused(_metadata, WHITESPACE); +} + +TEST_F(delim, control_refused) +{ + expect_refused(_metadata, CONTROL); +} + +TEST_F(delim, non_ascii_refused) +{ + expect_refused(_metadata, NON_ASCII); +} + +/* The escape character would cut every magic that uses one short. */ +TEST_F(delim, backslash_refused) +{ + expect_refused(_metadata, "\\"); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore index 9eb185fb2f9d..57f5bbdbedff 100644 --- a/tools/testing/selftests/filesystems/.gitignore +++ b/tools/testing/selftests/filesystems/.gitignore @@ -2,7 +2,6 @@ dnotify_test devpts_pts fclog -file_stressor anon_inode_test kernfs_test idmapped_tmpfile diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile index 03be337c1f35..bc4bfb677589 100644 --- a/tools/testing/selftests/filesystems/Makefile +++ b/tools/testing/selftests/filesystems/Makefile @@ -1,7 +1,7 @@ # SPDX-License-Identifier: GPL-2.0 CFLAGS += $(KHDR_INCLUDES) -TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test +TEST_GEN_PROGS := devpts_pts anon_inode_test kernfs_test fclog ustat_test TEST_GEN_PROGS += idmapped_tmpfile TEST_GEN_PROGS_EXTENDED := dnotify_test diff --git a/tools/testing/selftests/filesystems/config b/tools/testing/selftests/filesystems/config new file mode 100644 index 000000000000..9f45bc493a30 --- /dev/null +++ b/tools/testing/selftests/filesystems/config @@ -0,0 +1,8 @@ +CONFIG_CGROUPS=y +CONFIG_CGROUP_PIDS=y +CONFIG_FHANDLE=y +CONFIG_NAMESPACES=y +CONFIG_NET=y +CONFIG_NET_NS=y +CONFIG_INET=y +CONFIG_SYSFS=y diff --git a/tools/testing/selftests/filesystems/file_stressor/.gitignore b/tools/testing/selftests/filesystems/file_stressor/.gitignore new file mode 100644 index 000000000000..1eb3f40077d3 --- /dev/null +++ b/tools/testing/selftests/filesystems/file_stressor/.gitignore @@ -0,0 +1,2 @@ +# SPDX-License-Identifier: GPL-2.0-only +file_stressor diff --git a/tools/testing/selftests/filesystems/file_stressor/Makefile b/tools/testing/selftests/filesystems/file_stressor/Makefile new file mode 100644 index 000000000000..88c8231ac144 --- /dev/null +++ b/tools/testing/selftests/filesystems/file_stressor/Makefile @@ -0,0 +1,6 @@ +# SPDX-License-Identifier: GPL-2.0 + +CFLAGS += $(KHDR_INCLUDES) +TEST_GEN_PROGS := file_stressor + +include ../../lib.mk diff --git a/tools/testing/selftests/filesystems/file_stressor.c b/tools/testing/selftests/filesystems/file_stressor/file_stressor.c index 141badd671a9..141badd671a9 100644 --- a/tools/testing/selftests/filesystems/file_stressor.c +++ b/tools/testing/selftests/filesystems/file_stressor/file_stressor.c diff --git a/tools/testing/selftests/filesystems/file_stressor/settings b/tools/testing/selftests/filesystems/file_stressor/settings new file mode 100644 index 000000000000..b675ca93f936 --- /dev/null +++ b/tools/testing/selftests/filesystems/file_stressor/settings @@ -0,0 +1,3 @@ +# Timeout for file_stressor test +# The test runs for 900 seconds (15 minutes) plus setup/teardown time +timeout=1800 diff --git a/tools/testing/selftests/filesystems/fscontext_ns/.gitignore b/tools/testing/selftests/filesystems/fscontext_ns/.gitignore new file mode 100644 index 000000000000..a632905257a3 --- /dev/null +++ b/tools/testing/selftests/filesystems/fscontext_ns/.gitignore @@ -0,0 +1,2 @@ +# SPDX-License-Identifier: GPL-2.0-only +fscontext_ns_test diff --git a/tools/testing/selftests/filesystems/kernfs_test.c b/tools/testing/selftests/filesystems/kernfs_test.c index 84c2b910a60d..f4bcf3caf1d5 100644 --- a/tools/testing/selftests/filesystems/kernfs_test.c +++ b/tools/testing/selftests/filesystems/kernfs_test.c @@ -2,9 +2,23 @@ #define _GNU_SOURCE #define __SANE_USERSPACE_TYPES__ +#include <dirent.h> +#include <errno.h> #include <fcntl.h> +#include <limits.h> +#include <net/if.h> +#include <sched.h> +#include <signal.h> #include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <time.h> +#include <unistd.h> +#include <sys/ioctl.h> +#include <sys/mount.h> +#include <sys/socket.h> #include <sys/stat.h> +#include <sys/syscall.h> #include <sys/xattr.h> #include "kselftest_harness.h" @@ -12,12 +26,24 @@ TEST(kernfs_listxattr) { + ssize_t len; int fd; - /* Read-only file that can never have any extended attributes set. */ + /* Read-only file that can never have any extended attributes set. + * However, on systems with SELinux enabled, security.selinux xattr + * may be present. Skip the content check if any xattrs are found. + */ fd = open("/sys/kernel/warn_count", O_RDONLY | O_CLOEXEC); ASSERT_GE(fd, 0); - ASSERT_EQ(flistxattr(fd, NULL, 0), 0); + + len = flistxattr(fd, NULL, 0); + ASSERT_GE(len, 0); + + if (len > 0) { + close(fd); + SKIP(return, "xattrs present on /sys/kernel/warn_count, skipping xattr content check"); + } + EXPECT_EQ(close(fd), 0); } @@ -34,5 +60,1176 @@ TEST(kernfs_getxattr) EXPECT_EQ(close(fd), 0); } -TEST_HARNESS_MAIN +/* + * Exercise the kernfs dentry cache: lookup, revalidation of positive and + * negative dentries, readdir and namespace tagging. + * + * These drive kernfs from kernel context rather than VFS create/unlink, + * which is what ->d_revalidate() exists for: writing cgroup.subtree_control + * adds and removes files in every child cgroup with no VFS operation + * touching those names. + */ + +#define CG_SCRATCH "kernfs_selftest" +#define TEST_IFNAME "kfstest0" + +/* + * Controllers that add a file to each child cgroup when enabled. The probe + * file must be owned by the controller: cgroup_base_files[] entries such as + * cpu.stat exist in every cgroup regardless, and every file the cpu + * controller does own is behind a Kconfig symbol, so cpu is not usable here. + */ +static const struct { + const char *name; + const char *probe_file; +} controllers[] = { + { "memory", "memory.current" }, + { "pids", "pids.current" }, +}; + +static int find_cgroup2_root(char *buf, size_t len) +{ + char line[PATH_MAX * 2]; + FILE *f; + int ret = -1; + + f = fopen("/proc/self/mounts", "re"); + if (!f) + return -1; + + while (fgets(line, sizeof(line), f)) { + char mnt[PATH_MAX], type[64]; + + /* Octal escaping can expand a path fourfold; bound both %s. */ + if (sscanf(line, "%*s %4095s %63s", mnt, type) != 2) + continue; + if (strcmp(type, "cgroup2")) + continue; + if (strlen(mnt) >= len) + break; + strcpy(buf, mnt); + ret = 0; + break; + } + + fclose(f); + return ret; +} + +static int write_file(const char *path, const char *val) +{ + ssize_t len = strlen(val); + int fd, ret; + + fd = open(path, O_WRONLY | O_CLOEXEC); + if (fd < 0) + return -1; + ret = write(fd, val, len) == len ? 0 : -1; + close(fd); + return ret; +} + +static bool file_has_word(const char *path, const char *word) +{ + char buf[4096], *tok, *save; + bool found = false; + ssize_t n; + int fd; + + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) + return false; + n = read(fd, buf, sizeof(buf) - 1); + close(fd); + if (n < 0) + return false; + buf[n] = '\0'; + + for (tok = strtok_r(buf, "\n ", &save); tok; + tok = strtok_r(NULL, "\n ", &save)) { + if (!strcmp(tok, word)) { + found = true; + break; + } + } + return found; +} + +static bool path_is_mounted(const char *path) +{ + char line[PATH_MAX * 2]; + bool found = false; + FILE *f; + + f = fopen("/proc/self/mounts", "re"); + if (!f) + return false; + while (fgets(line, sizeof(line), f)) { + char mnt[PATH_MAX]; + + if (sscanf(line, "%*s %4095s", mnt) != 1) + continue; + if (!strcmp(mnt, path)) { + found = true; + break; + } + } + fclose(f); + return found; +} + +/* Shared by the stress tests below. */ +static bool stress_deadline(const struct timespec *end) +{ + struct timespec now; + + clock_gettime(CLOCK_MONOTONIC, &now); + return now.tv_sec > end->tv_sec || + (now.tv_sec == end->tv_sec && now.tv_nsec >= end->tv_nsec); +} + +FIXTURE(kernfs_cgroup) +{ + char scratch[PATH_MAX]; /* <cg2>/kernfs_selftest.<pid> */ + char child[PATH_MAX]; /* <scratch>/child */ + char probe[PATH_MAX]; /* child's controller file */ + char scratch_sc[PATH_MAX]; /* scratch's cgroup.subtree_control */ + char root_sc[PATH_MAX]; /* root's cgroup.subtree_control */ + char enable[32]; /* "+<controller>" */ + char disable[32]; /* "-<controller>" */ + char mnt[PATH_MAX]; /* our own mount, if we made one */ + bool mounted; + bool enabled_at_root; +}; + +/* A cgroup stays busy briefly after its last task exits. */ +static void rmdir_retry(const char *path) +{ + int i; + + for (i = 0; i < 500; i++) { + if (!rmdir(path) || errno != EBUSY) + return; + usleep(10000); + } +} + +/* + * Undo whatever SETUP managed to do. The harness skips TEARDOWN after a + * failed or skipped SETUP, so SETUP must call this before returning early. + */ +static void kernfs_cgroup_undo(FIXTURE_DATA(kernfs_cgroup) *self) +{ + rmdir_retry(self->child); + rmdir(self->scratch); + if (self->enabled_at_root) + write_file(self->root_sc, self->disable); + if (self->mounted) { + umount2(self->mnt, MNT_DETACH); + rmdir(self->mnt); + } + self->enabled_at_root = false; + self->mounted = false; +} + +FIXTURE_SETUP(kernfs_cgroup) +{ + char root[PATH_MAX], ctl[PATH_MAX]; + const char *probe_file = NULL; + size_t i; + + if (geteuid()) + SKIP(return, "test needs to run as root"); + + /* + * A private mount namespace stops our mounts leaking, but does not + * isolate the cgroup hierarchy: cgroup2 has one default hierarchy + * however many times it is mounted. The scratch cgroups live in the + * host's and must be removed, not discarded with the namespace. + */ + if (unshare(CLONE_NEWNS)) + SKIP(return, "unshare(CLONE_NEWNS): %s", strerror(errno)); + if (mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL)) + SKIP(return, "make / private: %s", strerror(errno)); + + /* Use an existing cgroup2 mount if there is one, else make our own. */ + if (find_cgroup2_root(root, sizeof(root))) { + strcpy(self->mnt, "/tmp/kernfs_selftest_cg2.XXXXXX"); + if (!mkdtemp(self->mnt)) + SKIP(return, "mkdtemp: %s", strerror(errno)); + if (mount("none", self->mnt, "cgroup2", 0, NULL)) { + rmdir(self->mnt); + SKIP(return, "mount cgroup2: %s", strerror(errno)); + } + self->mounted = true; + strcpy(root, self->mnt); + } + + snprintf(self->root_sc, sizeof(self->root_sc), + "%s/cgroup.subtree_control", root); + snprintf(ctl, sizeof(ctl), "%s/cgroup.controllers", root); + + /* Named after our pid so we cannot collide with anything else. */ + snprintf(self->scratch, sizeof(self->scratch), "%s/%s.%d", root, + CG_SCRATCH, getpid()); + snprintf(self->child, sizeof(self->child), "%s/child", self->scratch); + snprintf(self->scratch_sc, sizeof(self->scratch_sc), + "%s/cgroup.subtree_control", self->scratch); + + for (i = 0; i < ARRAY_SIZE(controllers); i++) { + if (!file_has_word(ctl, controllers[i].name)) + continue; + snprintf(self->enable, sizeof(self->enable), "+%s", + controllers[i].name); + snprintf(self->disable, sizeof(self->disable), "-%s", + controllers[i].name); + probe_file = controllers[i].probe_file; + + /* + * A controller must be in the root's subtree_control before + * it appears in our scratch cgroup. Note if we enabled it, + * so it can be put back. + */ + self->enabled_at_root = !file_has_word(self->root_sc, + controllers[i].name); + if (self->enabled_at_root && + write_file(self->root_sc, self->enable)) { + self->enabled_at_root = false; + probe_file = NULL; + continue; + } + break; + } + if (!probe_file) { + kernfs_cgroup_undo(self); + SKIP(return, "no usable cgroup2 controller"); + } + + snprintf(self->probe, sizeof(self->probe), "%s/%s", self->child, + probe_file); + + /* + * Only an unusable environment may skip. A scratch cgroup named + * after our own pid should always be creatable, so failing to make + * one is a result -- skipping would let a broken kernel look green. + */ + if (mkdir(self->scratch, 0755)) { + int err = errno; + + kernfs_cgroup_undo(self); + if (err == EROFS || err == EACCES || err == EPERM) + SKIP(return, "mkdir %s: %s", self->scratch, + strerror(err)); + ASSERT_EQ(err, 0) TH_LOG("mkdir %s: %s", self->scratch, + strerror(err)); + } + if (mkdir(self->child, 0755)) { + int err = errno; + + kernfs_cgroup_undo(self); + ASSERT_EQ(err, 0) TH_LOG("mkdir %s: %s", self->child, + strerror(err)); + } + + /* + * The tests below need the probe file to appear and disappear with + * the controller, so it must be absent now, before anything enables + * it in the scratch cgroup. Skip rather than fail: a probe file that + * is already there means the table names one the controller does not + * own, not that the kernel is broken. + */ + if (!access(self->probe, F_OK)) { + kernfs_cgroup_undo(self); + SKIP(return, "%s is not owned by the %s controller", + probe_file, self->enable + 1); + } +} + +FIXTURE_TEARDOWN(kernfs_cgroup) +{ + write_file(self->scratch_sc, self->disable); + kernfs_cgroup_undo(self); +} + +/* + * Walking already-cached dentries must not invalidate them. Spurious + * invalidation is not merely slow: d_invalidate() calls detach_mounts(), so + * an unrelated lookup would silently tear down any mount below. + */ +TEST_F(kernfs_cgroup, path_walk_does_not_invalidate) +{ + char src[] = "/tmp/kernfs_selftest_bind.XXXXXX"; + char sub[PATH_MAX], probe[PATH_MAX]; + int i; + + snprintf(sub, sizeof(sub), "%s/sub", self->child); + ASSERT_EQ(mkdir(sub, 0755), 0); + + if (!mkdtemp(src)) { + rmdir(sub); + SKIP(return, "mkdtemp: %s", strerror(errno)); + } + if (mount(src, sub, NULL, MS_BIND, NULL)) { + int err = errno; + + rmdir(sub); + rmdir(src); + SKIP(return, "bind mount onto a cgroup dir: %s", strerror(err)); + } + ASSERT_TRUE(path_is_mounted(sub)); + + /* Walk a sibling path through the same directory, repeatedly. */ + snprintf(probe, sizeof(probe), "%s/cgroup.procs", self->child); + for (i = 0; i < 8; i++) { + int fd = open(probe, O_RDONLY | O_CLOEXEC); + + if (fd >= 0) + close(fd); + } + + EXPECT_TRUE(path_is_mounted(sub)); + + umount2(sub, MNT_DETACH); + rmdir(sub); + rmdir(src); +} + +/* + * A cached negative dentry must be invalidated when the kernel creates the + * name behind the dcache's back. That is what kernfs_dir_changed() and + * kernfs_elem_dir::rev are for. + */ +TEST_F(kernfs_cgroup, negative_dentry_invalidated_by_kernel_create) +{ + struct stat st; + + /* Caches a negative dentry for the probe file. */ + ASSERT_EQ(stat(self->probe, &st), -1); + ASSERT_EQ(errno, ENOENT); + + /* The kernel now creates it, with no VFS operation on that name. */ + ASSERT_EQ(write_file(self->scratch_sc, self->enable), 0); + + EXPECT_EQ(stat(self->probe, &st), 0); +} + +/* The mirror image: a cached positive dentry must go when the node does. */ +TEST_F(kernfs_cgroup, positive_dentry_invalidated_by_kernel_remove) +{ + struct stat st; + + ASSERT_EQ(write_file(self->scratch_sc, self->enable), 0); + /* Caches a positive dentry. */ + ASSERT_EQ(stat(self->probe, &st), 0); + + ASSERT_EQ(write_file(self->scratch_sc, self->disable), 0); + + ASSERT_EQ(stat(self->probe, &st), -1); + EXPECT_EQ(errno, ENOENT); +} + +/* Opening a removed node fails; it never returns stale content. */ +TEST_F(kernfs_cgroup, open_after_rmdir_fails) +{ + char path[PATH_MAX]; + char buf[64]; + int fd; + + snprintf(path, sizeof(path), "%s/cgroup.procs", self->child); + + fd = open(path, O_RDONLY | O_CLOEXEC); + ASSERT_GE(fd, 0); + + ASSERT_EQ(rmdir(self->child), 0); + + /* Lookup by path must fail. */ + EXPECT_EQ(open(path, O_RDONLY | O_CLOEXEC), -1); + EXPECT_EQ(errno, ENOENT); + + /* + * An fd held across removal must fail rather than return stale + * content. rmdir() deactivates the node before it returns, so + * kernfs_seq_start() fails to get an active reference. + */ + EXPECT_LT(read(fd, buf, sizeof(buf)), 0); + EXPECT_EQ(errno, ENODEV); + EXPECT_EQ(close(fd), 0); + + ASSERT_EQ(mkdir(self->child, 0755), 0); +} + +/* readdir returns every entry exactly once. */ +TEST_F(kernfs_cgroup, readdir_no_duplicates) +{ + char names[512][NAME_MAX + 1]; + struct dirent *de; + int n = 0, i, j; + DIR *d; + + ASSERT_EQ(write_file(self->scratch_sc, self->enable), 0); + + d = opendir(self->child); + ASSERT_NE(d, NULL); + while ((de = readdir(d))) { + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + ASSERT_LT(n, (int)ARRAY_SIZE(names)); + strncpy(names[n], de->d_name, NAME_MAX); + names[n][NAME_MAX] = '\0'; + n++; + } + closedir(d); + + ASSERT_GT(n, 0); + for (i = 0; i < n; i++) + for (j = i + 1; j < n; j++) + EXPECT_STRNE(names[i], names[j]); +} + +#define RESUME_DIRS 24 + +/* + * Resuming at an entry that has gone must carry on after it, never before. + * Take a cookie for every entry, then remove each one, seek to its cookie + * and read the rest; nothing already reported may come back. + */ +TEST_F(kernfs_cgroup, readdir_resume_at_removed_entry) +{ + /* The cgroup's own control files are listed alongside ours. */ + char names[128][NAME_MAX + 1]; + long pos[128]; + char path[PATH_MAX]; + struct dirent *de; + int n = 0, i, j; + DIR *d; + + for (i = 0; i < RESUME_DIRS; i++) { + snprintf(path, sizeof(path), "%s/e%02d", self->scratch, i); + ASSERT_EQ(mkdir(path, 0755), 0); + } + + /* Record the cookie before reading each entry, with its name. */ + d = opendir(self->scratch); + ASSERT_NE(d, NULL); + while (1) { + long here = telldir(d); + + de = readdir(d); + if (!de) + break; + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + ASSERT_LT(n, (int)ARRAY_SIZE(pos)); + pos[n] = here; + strncpy(names[n], de->d_name, NAME_MAX); + names[n][NAME_MAX] = '\0'; + n++; + } + closedir(d); + ASSERT_GT(n, 1); + + for (i = 0; i < n; i++) { + /* Only the directories we made can be removed and put back. */ + if (strncmp(names[i], "e", 1)) + continue; + + snprintf(path, sizeof(path), "%s/%s", self->scratch, names[i]); + ASSERT_EQ(rmdir(path), 0); + + /* Reopen so the seek has to reach the kernel. */ + d = opendir(self->scratch); + ASSERT_NE(d, NULL); + seekdir(d, pos[i]); + while ((de = readdir(d))) { + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + for (j = 0; j < i; j++) + ASSERT_STRNE(de->d_name, names[j]) + TH_LOG("resuming at %s (gone) went back to %s", + names[i], names[j]); + } + closedir(d); + + ASSERT_EQ(mkdir(path, 0755), 0); + } + + for (i = 0; i < RESUME_DIRS; i++) { + snprintf(path, sizeof(path), "%s/e%02d", self->scratch, i); + EXPECT_EQ(rmdir(path), 0); + } +} + +#define CHURN_ROUNDS 400 +#define CHURN_BUFSZ 512 /* small, so a listing takes several calls */ + +/* + * The files appear at the end of the enabling write and go at the start of + * the disabling one, so the window where they exist is the short one. + */ +#define CHURN_DWELL_ON 2000 +#define CHURN_DWELL_OFF 200 + +struct kernfs_dirent64 { + unsigned long long d_ino; + long long d_off; + unsigned short d_reclen; + unsigned char d_type; + char d_name[]; +}; + +/* + * The same resume, but inside one getdents(2) call. rmdir(2) cannot reach + * that window because iterate_dir() holds the listed directory's i_rwsem + * for the whole listing; cgroup.subtree_control can, having no VFS + * operation on the names it adds and removes. The files that are not the + * controller's stay throughout, so each must appear exactly once. + * + * A stress test: it has not been seen to catch the ordering bug, and is + * here to keep the unlocked window under load for lockdep and KASAN. + */ +TEST_F(kernfs_cgroup, readdir_resume_vs_internal_remove) +{ + char buf[CHURN_BUFSZ] __attribute__((aligned(8))); + char stable[128][NAME_MAX + 1]; + int nstable = 0, i, r; + int withctl = 0, without = 0; + int seen[128], status; + pid_t churner; + DIR *d; + + /* With the controller off, whatever is left is what must persist. */ + ASSERT_EQ(write_file(self->scratch_sc, self->disable), 0); + d = opendir(self->child); + ASSERT_NE(d, NULL); + for (;;) { + struct dirent *de = readdir(d); + + if (!de) + break; + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + ASSERT_LT(nstable, (int)ARRAY_SIZE(stable)); + strncpy(stable[nstable], de->d_name, NAME_MAX); + stable[nstable][NAME_MAX] = '\0'; + nstable++; + } + closedir(d); + ASSERT_GT(nstable, 0); + + churner = fork(); + ASSERT_GE(churner, 0); + if (churner == 0) { + for (;;) { + if (write_file(self->scratch_sc, self->enable)) + _exit(10); + usleep(CHURN_DWELL_ON); + if (write_file(self->scratch_sc, self->disable)) + _exit(11); + usleep(CHURN_DWELL_OFF); + } + } + + for (r = 0; r < CHURN_ROUNDS; r++) { + int fd = open(self->child, O_RDONLY | O_DIRECTORY); + int extra = 0; + int n; + ASSERT_GE(fd, 0); + memset(seen, 0, sizeof(seen)); + + while ((n = syscall(SYS_getdents64, fd, buf, sizeof(buf))) > 0) { + int off = 0; + + while (off < n) { + struct kernfs_dirent64 *de = (void *)(buf + off); + bool known = false; + + off += de->d_reclen; + for (i = 0; i < nstable; i++) + if (!strcmp(de->d_name, stable[i])) { + seen[i]++; + known = true; + } + if (!known && strcmp(de->d_name, ".") && + strcmp(de->d_name, "..")) + extra++; + } + } + ASSERT_GE(n, 0); + EXPECT_EQ(close(fd), 0); + + if (extra) + withctl++; + else + without++; + + for (i = 0; i < nstable; i++) + ASSERT_EQ(seen[i], 1) + TH_LOG("round %d: %s seen %d times", + r, stable[i], seen[i]); + } + + /* The churn must have been running, or the listings prove nothing. */ + EXPECT_EQ(kill(churner, SIGKILL), 0); + ASSERT_EQ(waitpid(churner, &status, 0), churner); + ASSERT_TRUE(WIFSIGNALED(status) && WTERMSIG(status) == SIGKILL) + TH_LOG("churner exited on its own: status %d", status); + + /* + * They also have to have overlapped it. How much depends on the + * machine, so say the race could not be arranged rather than fail. + */ + if (!withctl || !without) + SKIP(return, "listings did not span the churn: %d with, %d without", + withctl, without); +} + +/* + * A telldir() cookie must resolve back to the same entry after seekdir(). + * kernfs encodes the cookie as the node's name hash, so this covers + * kernfs_dir_pos() as well as plain iteration. + */ +TEST_F(kernfs_cgroup, readdir_seekdir_roundtrip) +{ + char names[512][NAME_MAX + 1]; + struct dirent *de; + long pos[512]; + int n = 0, i; + DIR *d; + + ASSERT_EQ(write_file(self->scratch_sc, self->enable), 0); + + d = opendir(self->child); + ASSERT_NE(d, NULL); + + /* Record the cookie *before* reading each entry, with its name. */ + while (1) { + long here = telldir(d); + + de = readdir(d); + if (!de) + break; + if (!strcmp(de->d_name, ".") || !strcmp(de->d_name, "..")) + continue; + ASSERT_LT(n, (int)ARRAY_SIZE(pos)); + pos[n] = here; + strncpy(names[n], de->d_name, NAME_MAX); + names[n][NAME_MAX] = '\0'; + n++; + } + ASSERT_GT(n, 0); + + /* Seeking back to a cookie must land on the entry it was taken at. */ + for (i = 0; i < n; i++) { + seekdir(d, pos[i]); + de = readdir(d); + ASSERT_NE(de, NULL); + EXPECT_STREQ(de->d_name, names[i]); + } + + closedir(d); +} + +#define STRESS_SECS 2 +#define STRESS_DIRS 4 +#define STRESS_READERS 4 + +/* + * Hammer lookup against creation and removal. Revalidation holds no lock + * against the writers, so what makes it safe is that every answer it can + * give is one the caller already handles: a reader must only ever see + * success or an errno meaning "it went away", never garbage or a hang. + */ +TEST_F(kernfs_cgroup, lookup_vs_create_remove_stress) +{ + pid_t pids[STRESS_DIRS + STRESS_READERS]; + struct timespec end; + int i, status, n = 0; + + clock_gettime(CLOCK_MONOTONIC, &end); + end.tv_sec += STRESS_SECS; + + for (i = 0; i < STRESS_DIRS; i++) { + pid_t pid = fork(); + + ASSERT_GE(pid, 0); + if (pid == 0) { + char dir[PATH_MAX]; + + snprintf(dir, sizeof(dir), "%s/s%d", self->scratch, i); + while (!stress_deadline(&end)) { + if (mkdir(dir, 0755) && errno != EEXIST) + _exit(10); + if (rmdir(dir) && errno != ENOENT && + errno != EBUSY) + _exit(11); + } + _exit(0); + } + pids[n++] = pid; + } + + for (i = 0; i < STRESS_READERS; i++) { + pid_t pid = fork(); + + ASSERT_GE(pid, 0); + if (pid == 0) { + /* Start each reader on a different directory. */ + unsigned int seq = i; + + while (!stress_deadline(&end)) { + int which = seq++ % STRESS_DIRS; + char path[PATH_MAX]; + struct stat st; + int fd; + + snprintf(path, sizeof(path), + "%s/s%d/cgroup.procs", + self->scratch, which); + + if (stat(path, &st) && errno != ENOENT && + errno != ENODEV) + _exit(20); + + fd = open(path, O_RDONLY | O_CLOEXEC); + if (fd < 0) { + if (errno != ENOENT && errno != ENODEV) + _exit(21); + } else { + close(fd); + } + + if (access(path, F_OK) && errno != ENOENT && + errno != ENODEV) + _exit(22); + } + _exit(0); + } + pids[n++] = pid; + } + + for (i = 0; i < n; i++) { + ASSERT_EQ(waitpid(pids[i], &status, 0), pids[i]); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(WEXITSTATUS(status), 0); + } + + for (i = 0; i < STRESS_DIRS; i++) { + char dir[PATH_MAX]; + + snprintf(dir, sizeof(dir), "%s/s%d", self->scratch, i); + rmdir_retry(dir); + } +} + +struct kernfs_handle { + struct file_handle h; + unsigned char buf[MAX_HANDLE_SZ]; +}; + +static int kernfs_encode(const char *path, struct kernfs_handle *fh) +{ + int mount_id; + + memset(fh, 0, sizeof(*fh)); + fh->h.handle_bytes = sizeof(fh->buf); + return name_to_handle_at(AT_FDCWD, path, &fh->h, &mount_id, 0); +} + +/* + * Skip only where file handles do not work at all. ENOENT must still + * fail: mkdir leaves a negative dentry cached, so the name resolves only + * after ->d_revalidate() drops it. The encode tests revalidation too. + */ +static bool fh_unsupported(int err) +{ + return err == EOPNOTSUPP || err == EPERM || err == ENOSYS; +} + +/* + * Decoding a file needs CAP_DAC_READ_SEARCH in the initial user + * namespace. Probe once so the tests skip instead of fail. + */ +static bool fh_can_decode(int mfd, struct kernfs_handle *fh) +{ + int fd = open_by_handle_at(mfd, &fh->h, O_PATH); + + if (fd < 0) + return errno != EPERM; + close(fd); + return true; +} + +/* + * A file handle reaches a node without a lookup through its parent. A + * live node must decode. A removed one must not, because + * kernfs_find_and_get_node_by_id() refuses inactive nodes. + * + * Use O_PATH: opening a removed node fails with ENODEV, which would hide + * what is being tested. + */ +TEST_F(kernfs_cgroup, exportfs_decode_and_stale) +{ + char victim[PATH_MAX], procs[PATH_MAX]; + struct kernfs_handle fh; + struct stat st; + int mfd, fd; + + snprintf(victim, sizeof(victim), "%s/fh", self->scratch); + snprintf(procs, sizeof(procs), "%s/cgroup.procs", victim); + ASSERT_EQ(mkdir(victim, 0755), 0); + + /* Any fd on the filesystem identifies it to open_by_handle_at(). */ + mfd = open(self->scratch, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(mfd, 0); + + if (kernfs_encode(procs, &fh)) { + int err = errno; + + close(mfd); + rmdir(victim); + ASSERT_TRUE(fh_unsupported(err)) + TH_LOG("name_to_handle_at: %s", strerror(err)); + SKIP(return, "name_to_handle_at: %s", strerror(err)); + } + + if (!fh_can_decode(mfd, &fh)) { + close(mfd); + rmdir(victim); + SKIP(return, "open_by_handle_at: no CAP_DAC_READ_SEARCH"); + } + + fd = open_by_handle_at(mfd, &fh.h, O_PATH); + ASSERT_GE(fd, 0); + EXPECT_EQ(fstat(fd, &st), 0); + EXPECT_EQ(st.st_nlink, 1); + EXPECT_EQ(close(fd), 0); + + ASSERT_EQ(rmdir(victim), 0); + + fd = open_by_handle_at(mfd, &fh.h, O_PATH); + EXPECT_LT(fd, 0); + if (fd >= 0) + close(fd); + else + EXPECT_EQ(errno, ESTALE); + + EXPECT_EQ(close(mfd), 0); +} + +#define FH_STRESS_SECS 2 +#define FH_DECODE_CAP 10000 + +/* + * Decode file handles while the node is being removed. A decode must + * answer with a usable handle or ESTALE, never garbage and never a hang. + * + * The link count is checked too. An inode that reaches the inode hash + * after the removal cleared link counts keeps the 1 it was born with, so + * it never gets an IN_DELETE_SELF. This has not been seen to fire: it + * needs the decode to stall between the lookup by id and the hash insert, + * and nothing there blocks. It is kept because it is cheap and only + * looks once the directory is gone, so it cannot fail falsely. + */ +TEST_F(kernfs_cgroup, exportfs_decode_vs_rmdir_stress) +{ + int mfd, bad = 0, rounds = 0; + struct timespec end; + + mfd = open(self->scratch, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + ASSERT_GE(mfd, 0); + + clock_gettime(CLOCK_MONOTONIC, &end); + end.tv_sec += FH_STRESS_SECS; + + while (!stress_deadline(&end)) { + char victim[PATH_MAX], procs[PATH_MAX]; + int last = -1, fd, i; + struct kernfs_handle fh; + struct stat st; + pid_t pid; + + snprintf(victim, sizeof(victim), "%s/fh%d", self->scratch, + rounds++); + snprintf(procs, sizeof(procs), "%s/cgroup.procs", victim); + if (mkdir(victim, 0755)) + break; + if (kernfs_encode(procs, &fh)) { + int err = errno; + + rmdir(victim); + ASSERT_TRUE(fh_unsupported(err)) + TH_LOG("name_to_handle_at: %s", strerror(err)); + SKIP(goto out, "name_to_handle_at: %s", strerror(err)); + } + if (rounds == 1 && !fh_can_decode(mfd, &fh)) { + rmdir(victim); + SKIP(goto out, + "open_by_handle_at: no CAP_DAC_READ_SEARCH"); + } + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + rmdir_retry(victim); + _exit(0); + } + + /* + * Decode until the removal deactivates the node. Keep the + * last one that worked: it ran closest to the removal. + */ + for (i = 0; i < FH_DECODE_CAP; i++) { + fd = open_by_handle_at(mfd, &fh.h, O_PATH); + if (fd < 0) + break; + if (last >= 0) + close(last); + last = fd; + } + ASSERT_EQ(waitpid(pid, NULL, 0), pid); + + if (last >= 0) { + if (access(victim, F_OK) && errno == ENOENT && + !fstat(last, &st) && st.st_nlink != 0) + bad++; + close(last); + } + rmdir(victim); + } + + EXPECT_EQ(bad, 0) + TH_LOG("%d of %d rounds decoded a removed node whose inode kept its link count", + bad, rounds); +out: + close(mfd); +} + +/* + * sysfs is namespace tagged (KERNFS_NS) and supports rename; cgroup2 does + * neither. Run in a private netns with its own sysfs so the host is + * untouched. + */ +FIXTURE(kernfs_netns) +{ + char mnt[PATH_MAX]; + char net[PATH_MAX]; + bool mounted; +}; + +FIXTURE_SETUP(kernfs_netns) +{ + if (geteuid()) + SKIP(return, "test needs to run as root"); + + if (unshare(CLONE_NEWNS | CLONE_NEWNET)) + SKIP(return, "unshare(CLONE_NEWNS|CLONE_NEWNET): %s", + strerror(errno)); + + /* Don't let our sysfs mount escape into the parent namespace. */ + ASSERT_EQ(mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + strcpy(self->mnt, "/tmp/kernfs_selftest_sysfs.XXXXXX"); + if (!mkdtemp(self->mnt)) + SKIP(return, "mkdtemp: %s", strerror(errno)); + + if (mount("none", self->mnt, "sysfs", 0, NULL)) { + rmdir(self->mnt); + SKIP(return, "mount sysfs: %s", strerror(errno)); + } + self->mounted = true; + + snprintf(self->net, sizeof(self->net), "%s/class/net", self->mnt); +} + +FIXTURE_TEARDOWN(kernfs_netns) +{ + if (self->mounted) + umount2(self->mnt, MNT_DETACH); + rmdir(self->mnt); +} + +/* + * sysfs must show this network namespace's interfaces, not the parent's. + * + * Do not assume a fresh netns contains only "lo": fallback tunnel devices + * (tunl0, sit0, gre0, ...) are created in every namespace unless + * net.core.fb_tunnels_only_for_init_net is set, so which names appear + * depends on the modules the host has. Check the set instead -- + * if_nametoindex() resolves in the current netns, so every name sysfs shows + * must resolve there, and the counts must agree. + * + * Count only symlinks. Not every entry is a device: bonding adds a + * bonding_masters attribute to /sys/class/net in every namespace. + */ +TEST_F(kernfs_netns, ns_tag_isolates_class_net) +{ + struct if_nameindex *idx, *i; + bool found_lo = false; + int n = 0, want = 0; + struct dirent *de; + DIR *d; + + d = opendir(self->net); + ASSERT_NE(d, NULL); + while ((de = readdir(d))) { + if (de->d_type != DT_LNK) + continue; + EXPECT_NE(if_nametoindex(de->d_name), 0u) + TH_LOG("%s is not in this netns", de->d_name); + if (!strcmp(de->d_name, "lo")) + found_lo = true; + n++; + } + closedir(d); + + idx = if_nameindex(); + ASSERT_NE(idx, NULL); + for (i = idx; i->if_index; i++) + want++; + if_freenameindex(idx); + + EXPECT_TRUE(found_lo); + EXPECT_EQ(n, want); +} + +/* + * After a rename the old name must stop resolving and the new one must + * start, even though both dentries are already cached. + */ +TEST_F(kernfs_netns, rename_is_revalidated) +{ + char old_path[PATH_MAX], new_path[PATH_MAX]; + struct ifreq ifr = {}; + struct stat st; + int sk; + + snprintf(old_path, sizeof(old_path), "%s/lo", self->net); + snprintf(new_path, sizeof(new_path), "%s/%s", self->net, TEST_IFNAME); + + /* Warm both dentries: one positive, one negative. */ + ASSERT_EQ(stat(old_path, &st), 0); + ASSERT_EQ(stat(new_path, &st), -1); + ASSERT_EQ(errno, ENOENT); + + sk = socket(AF_INET, SOCK_DGRAM | SOCK_CLOEXEC, 0); + ASSERT_GE(sk, 0); + strcpy(ifr.ifr_name, "lo"); + strcpy(ifr.ifr_newname, TEST_IFNAME); + if (ioctl(sk, SIOCSIFNAME, &ifr)) { + close(sk); + SKIP(return, "SIOCSIFNAME: %s", strerror(errno)); + } + close(sk); + + EXPECT_EQ(stat(old_path, &st), -1); + EXPECT_EQ(errno, ENOENT); + EXPECT_EQ(stat(new_path, &st), 0); +} + +static int netdev_rename(const char *from, const char *to) +{ + struct ifreq ifr = {}; + int sk, ret; + + sk = socket(AF_INET, SOCK_DGRAM | SOCK_CLOEXEC, 0); + if (sk < 0) + return -1; + strncpy(ifr.ifr_name, from, IFNAMSIZ - 1); + strncpy(ifr.ifr_newname, to, IFNAMSIZ - 1); + ret = ioctl(sk, SIOCSIFNAME, &ifr); + close(sk); + return ret; +} + +/* + * Bounded by a count, not by time: every rename is logged and not rate + * limited, so a timed loop would flood the kernel log. + */ +#define RENAME_FLIPS 200 +#define RENAME_READERS 4 + +/* + * Rename an interface while other tasks look up the names it moves + * between. This renames its /sys/class/net entry through + * kernfs_rename_ns() with the parent unchanged. + * + * The renamer checks what is certain: SIOCSIFNAME returns once the rename + * is done and nothing else renames here, so the new name must resolve and + * the old must not. The readers cannot check that, because the name can + * move between their two lstat() calls. They only check that a lookup + * returns success or ENOENT, and keep the lock busy while renames run. + * + * lstat() not stat(): /sys/class/net/<dev> is a symlink and is renamed + * before the directory it points at, so the two are not atomic. + */ +TEST_F(kernfs_netns, rename_vs_lookup_stress) +{ + char old_path[PATH_MAX], new_path[PATH_MAX]; + pid_t pids[RENAME_READERS]; + int i, status, n = 0, bad = 0; + struct stat st; + int done[2]; + + snprintf(old_path, sizeof(old_path), "%s/lo", self->net); + snprintf(new_path, sizeof(new_path), "%s/%s", self->net, TEST_IFNAME); + + if (netdev_rename("lo", TEST_IFNAME)) + SKIP(return, "SIOCSIFNAME: %s", strerror(errno)); + if (netdev_rename(TEST_IFNAME, "lo")) + SKIP(return, "SIOCSIFNAME back: %s", strerror(errno)); + + /* Readers run until the renamer closes the write end. */ + ASSERT_EQ(pipe2(done, O_NONBLOCK | O_CLOEXEC), 0); + + for (i = 0; i < RENAME_READERS; i++) { + pid_t pid = fork(); + + ASSERT_GE(pid, 0); + if (pid == 0) { + struct stat rst; + char c; + + close(done[1]); + while (read(done[0], &c, 1) < 0 && errno == EAGAIN) { + if (lstat(old_path, &rst) && errno != ENOENT) + _exit(20); + if (lstat(new_path, &rst) && errno != ENOENT) + _exit(21); + } + _exit(0); + } + pids[n++] = pid; + } + close(done[0]); + + for (i = 0; i < RENAME_FLIPS; i++) { + if (netdev_rename("lo", TEST_IFNAME)) + break; + if (lstat(new_path, &st) || !lstat(old_path, &st)) { + bad++; + break; + } + if (netdev_rename(TEST_IFNAME, "lo")) + break; + if (lstat(old_path, &st) || !lstat(new_path, &st)) { + bad++; + break; + } + } + close(done[1]); + + for (i = 0; i < n; i++) { + ASSERT_EQ(waitpid(pids[i], &status, 0), pids[i]); + ASSERT_TRUE(WIFEXITED(status)); + EXPECT_EQ(WEXITSTATUS(status), 0); + } + + EXPECT_EQ(bad, 0) + TH_LOG("a completed rename left the wrong name resolving"); + + /* Leave the interface as the fixture found it. */ + netdev_rename(TEST_IFNAME, "lo"); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/mntns_unbindable/Makefile b/tools/testing/selftests/filesystems/mntns_unbindable/Makefile new file mode 100644 index 000000000000..33a311c5bd72 --- /dev/null +++ b/tools/testing/selftests/filesystems/mntns_unbindable/Makefile @@ -0,0 +1,6 @@ +# SPDX-License-Identifier: GPL-2.0 +TEST_GEN_PROGS := mntns_unbindable_test + +CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) + +include ../../lib.mk diff --git a/tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c b/tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c new file mode 100644 index 000000000000..9aebc37cf74d --- /dev/null +++ b/tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c @@ -0,0 +1,227 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * An unbindable mount stays unbindable in a cloned mount namespace. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> +#include <unistd.h> + +#include "../../kselftest_harness.h" + +#ifndef OPEN_TREE_CLONE +#define OPEN_TREE_CLONE 1 +#endif +#ifndef OPEN_TREE_CLOEXEC +#define OPEN_TREE_CLOEXEC O_CLOEXEC +#endif + +static int sys_open_tree(int dfd, const char *filename, unsigned int flags) +{ + return syscall(__NR_open_tree, dfd, filename, flags); +} + +/* Child exit codes. */ +enum { + CHILD_OK, /* the operation failed with EINVAL as it must */ + CHILD_ALLOWED, /* the operation succeeded: the flag was lost */ + CHILD_UNSHARE, /* unshare(CLONE_NEWNS) failed */ + CHILD_ERRNO, /* the operation failed with some other errno */ + CHILD_MOUNTINFO, /* the mount was not found in mountinfo */ +}; + +FIXTURE(mntns_unbindable) +{ + char base[64]; + char src[80]; + char dst[80]; + bool mounted; +}; + +FIXTURE_SETUP(mntns_unbindable) +{ + self->mounted = false; + + if (geteuid() != 0) + SKIP(return, "test requires CAP_SYS_ADMIN"); + + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + snprintf(self->base, sizeof(self->base), "/tmp/mntns_unbindable.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0); + self->mounted = true; + + snprintf(self->src, sizeof(self->src), "%s/src", self->base); + snprintf(self->dst, sizeof(self->dst), "%s/dst", self->base); + ASSERT_EQ(mkdir(self->src, 0755), 0); + ASSERT_EQ(mkdir(self->dst, 0755), 0); + + ASSERT_EQ(mount("tmpfs", self->src, "tmpfs", 0, NULL), 0); + ASSERT_EQ(mount(NULL, self->src, NULL, MS_UNBINDABLE, NULL), 0); +} + +FIXTURE_TEARDOWN(mntns_unbindable) +{ + if (self->mounted) + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +static int classify(int ret, int err) +{ + if (ret >= 0) + return CHILD_ALLOWED; + return err == EINVAL ? CHILD_OK : CHILD_ERRNO; +} + +/* Is the mount on @mountpoint marked unbindable in /proc/self/mountinfo? */ +static int mountinfo_unbindable(const char *mountpoint) +{ + char line[4096]; + FILE *f; + int ret = CHILD_MOUNTINFO; + + f = fopen("/proc/self/mountinfo", "re"); + if (!f) + return CHILD_ERRNO; + + while (fgets(line, sizeof(line), f)) { + char *fields[6], *p = line, *opt; + int i; + + for (i = 0; i < 6; i++) { + fields[i] = strsep(&p, " "); + if (!fields[i]) + break; + } + if (i < 6 || strcmp(fields[4], mountpoint)) + continue; + + /* the optional fields, up to the "-" separator */ + ret = CHILD_ALLOWED; + while ((opt = strsep(&p, " ")) && strcmp(opt, "-")) { + if (!strcmp(opt, "unbindable")) + ret = CHILD_OK; + } + break; + } + fclose(f); + return ret; +} + +static int run_in_child(int (*fn)(const char *src, const char *dst), + const char *src, const char *dst) +{ + int status; + pid_t pid; + + pid = fork(); + if (pid < 0) + return -1; + if (pid == 0) + _exit(fn(src, dst)); + if (waitpid(pid, &status, 0) != pid || !WIFEXITED(status)) + return -1; + return WEXITSTATUS(status); +} + +static int bind_after_clone(const char *src, const char *dst) +{ + int ret; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + ret = mount(src, dst, NULL, MS_BIND, NULL); + return classify(ret, errno); +} + +static int rbind_after_clone(const char *src, const char *dst) +{ + int ret; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + ret = mount(src, dst, NULL, MS_BIND | MS_REC, NULL); + return classify(ret, errno); +} + +static int open_tree_after_clone(const char *src, const char *dst) +{ + int ret; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + ret = sys_open_tree(AT_FDCWD, src, OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC); + return classify(ret, errno); +} + +static int mountinfo_after_clone(const char *src, const char *dst) +{ + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + return mountinfo_unbindable(src); +} + +static int bind_after_two_clones(const char *src, const char *dst) +{ + int ret; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + ret = mount(src, dst, NULL, MS_BIND, NULL); + return classify(ret, errno); +} + +/* The namespace the mount was made unbindable in. */ +TEST_F(mntns_unbindable, refuses_bind) +{ + int ret = mount(self->src, self->dst, NULL, MS_BIND, NULL); + + ASSERT_EQ(classify(ret, errno), CHILD_OK); + ASSERT_EQ(mountinfo_unbindable(self->src), CHILD_OK); +} + +/* A copy of the namespace must not turn the mount bindable. */ +TEST_F(mntns_unbindable, refuses_bind_after_clone) +{ + ASSERT_EQ(run_in_child(bind_after_clone, self->src, self->dst), CHILD_OK) + TH_LOG("bind of an unbindable mount allowed in a copied mount namespace"); +} + +TEST_F(mntns_unbindable, refuses_rbind_after_clone) +{ + ASSERT_EQ(run_in_child(rbind_after_clone, self->src, self->dst), CHILD_OK) + TH_LOG("rbind of an unbindable mount allowed in a copied mount namespace"); +} + +TEST_F(mntns_unbindable, refuses_open_tree_after_clone) +{ + ASSERT_EQ(run_in_child(open_tree_after_clone, self->src, self->dst), CHILD_OK) + TH_LOG("OPEN_TREE_CLONE of an unbindable mount allowed in a copied mount namespace"); +} + +TEST_F(mntns_unbindable, mountinfo_after_clone) +{ + ASSERT_EQ(run_in_child(mountinfo_after_clone, self->src, self->dst), CHILD_OK) + TH_LOG("mountinfo does not show the mount as unbindable in a copied mount namespace"); +} + +TEST_F(mntns_unbindable, refuses_bind_after_two_clones) +{ + ASSERT_EQ(run_in_child(bind_after_two_clones, self->src, self->dst), CHILD_OK) + TH_LOG("bind of an unbindable mount allowed two mount namespace copies down"); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/filesystems/openat2/openat2_test.c b/tools/testing/selftests/filesystems/openat2/openat2_test.c index 6f5afbe2d8d3..e08c94ce0530 100644 --- a/tools/testing/selftests/filesystems/openat2/openat2_test.c +++ b/tools/testing/selftests/filesystems/openat2/openat2_test.c @@ -23,8 +23,16 @@ * XXX: This is wrong on {mips, parisc, powerpc, sparc}. */ #undef O_LARGEFILE -#ifdef __aarch64__ +#if defined(__aarch64__) || defined(__alpha__) #define O_LARGEFILE 0x20000 +#elif defined(__powerpc__) || defined(__ppc__) +#define O_LARGEFILE 0x10000 +#elif defined(__sparc__) +#define O_LARGEFILE 0x40000 +#elif defined(__mips__) +#define O_LARGEFILE 0x2000 +#elif defined(__parisc__) +#define O_LARGEFILE 0x800 #else #define O_LARGEFILE 0x8000 #endif diff --git a/tools/testing/selftests/filesystems/openat2/resolve_test.c b/tools/testing/selftests/filesystems/openat2/resolve_test.c index eacde59ce158..6216a8546388 100644 --- a/tools/testing/selftests/filesystems/openat2/resolve_test.c +++ b/tools/testing/selftests/filesystems/openat2/resolve_test.c @@ -140,9 +140,12 @@ FIXTURE_SETUP(openat2_resolve) if (!openat2_supported) SKIP(return, "openat2(2) not supported"); - /* Unshare and make /tmp a new directory. */ + /* Unshare and make the mount tree private. */ ASSERT_EQ(unshare(CLONE_NEWNS), 0); - ASSERT_EQ(mount("", "/tmp", "", MS_PRIVATE, ""), 0); + ASSERT_EQ(mount("", "/", "", MS_PRIVATE | MS_REC, ""), 0); + + /* Ensure /tmp is a mountpoint for RESOLVE_NO_XDEV test crossing into /tmp. */ + ASSERT_EQ(mount("/tmp", "/tmp", NULL, MS_BIND, NULL), 0); /* Make the top-level directory. */ ASSERT_NE(mkdtemp(dirname), NULL); diff --git a/tools/testing/selftests/filesystems/statmount/statmount_test.c b/tools/testing/selftests/filesystems/statmount/statmount_test.c index 60c2c544db6a..52544512edc7 100644 --- a/tools/testing/selftests/filesystems/statmount/statmount_test.c +++ b/tools/testing/selftests/filesystems/statmount/statmount_test.c @@ -17,7 +17,7 @@ static const char *const known_fs[] = { "9p", "adfs", "affs", "afs", "aio", "anon_inodefs", "apparmorfs", - "autofs", "bcachefs", "bdev", "befs", "bfs", "binder", "binfmt_misc", + "autofs", "bcachefs", "bdev", "befs", "binder", "binfmt_misc", "bpf", "btrfs", "btrfs_test_fs", "ceph", "cgroup", "cgroup2", "cifs", "coda", "configfs", "cpuset", "cramfs", "cxl", "dax", "debugfs", "devpts", "devtmpfs", "dmabuf", "drm", "ecryptfs", "efivarfs", "efs", diff --git a/tools/testing/selftests/filesystems/umount_propagation/Makefile b/tools/testing/selftests/filesystems/umount_propagation/Makefile new file mode 100644 index 000000000000..fc0a0783018b --- /dev/null +++ b/tools/testing/selftests/filesystems/umount_propagation/Makefile @@ -0,0 +1,6 @@ +# SPDX-License-Identifier: GPL-2.0 +TEST_GEN_PROGS := umount_propagation_test + +CFLAGS += -Wall -O2 -g $(KHDR_INCLUDES) + +include ../../lib.mk diff --git a/tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c b/tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c new file mode 100644 index 000000000000..9e18d54dfb32 --- /dev/null +++ b/tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * A synchronous umount fails with EBUSY when a mount it would pull out by + * propagation is still in use. + */ +#define _GNU_SOURCE +#include <errno.h> +#include <fcntl.h> +#include <sched.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/mount.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/wait.h> +#include <unistd.h> +#include <linux/mount.h> +#include <linux/stat.h> + +#include "../../kselftest_harness.h" + +#ifndef OPEN_TREE_CLONE +#define OPEN_TREE_CLONE 1 +#endif +#ifndef OPEN_TREE_CLOEXEC +#define OPEN_TREE_CLOEXEC O_CLOEXEC +#endif +#ifndef AT_RECURSIVE +#define AT_RECURSIVE 0x8000 +#endif +#ifndef MOVE_MOUNT_F_EMPTY_PATH +#define MOVE_MOUNT_F_EMPTY_PATH 0x00000004 +#endif +#ifndef MOVE_MOUNT_BENEATH +#define MOVE_MOUNT_BENEATH 0x00000200 +#endif +#ifndef STATX_MNT_ID +#define STATX_MNT_ID 0x00001000U +#endif + +static int sys_open_tree(int dfd, const char *filename, unsigned int flags) +{ + return syscall(__NR_open_tree, dfd, filename, flags); +} + +static int sys_move_mount(int from_dfd, const char *from_pathname, + int to_dfd, const char *to_pathname, + unsigned int flags) +{ + return syscall(__NR_move_mount, from_dfd, from_pathname, to_dfd, + to_pathname, flags); +} + +/* Child exit codes. */ +enum { + CHILD_OK, + CHILD_UNSHARE, /* could not set up the slave namespace */ + CHILD_OPEN_TREE, /* open_tree() failed */ + CHILD_MOVE_MOUNT, /* move_mount() failed */ + CHILD_STATX, /* statx() failed */ + CHILD_PIPE, /* the parent went away */ +}; + +/* Messages between parent and child. */ +enum { + MSG_READY = 'r', /* child: the copy is mounted and referenced */ + MSG_CHECK = 'c', /* parent: check that the copy is still attached */ + MSG_ATTACHED = 'a', /* child: it is */ + MSG_DETACHED = 'd', /* child: it is not */ + MSG_CLOSE = 'x', /* parent: drop the reference */ + MSG_CLOSED = 'y', /* child: dropped */ + MSG_EXIT = 'e', /* parent: done */ +}; + +FIXTURE(umount_propagation) +{ + char base[64]; + char victim[80]; + bool mounted; +}; + +FIXTURE_SETUP(umount_propagation) +{ + self->mounted = false; + + if (geteuid() != 0) + SKIP(return, "test requires CAP_SYS_ADMIN"); + + ASSERT_EQ(unshare(CLONE_NEWNS), 0); + ASSERT_EQ(mount("", "/", NULL, MS_REC | MS_PRIVATE, NULL), 0); + + snprintf(self->base, sizeof(self->base), "/tmp/umount_propagation.XXXXXX"); + ASSERT_NE(mkdtemp(self->base), NULL); + ASSERT_EQ(mount("tmpfs", self->base, "tmpfs", 0, NULL), 0); + self->mounted = true; + ASSERT_EQ(mount(NULL, self->base, NULL, MS_SHARED, NULL), 0); + + snprintf(self->victim, sizeof(self->victim), "%s/victim", self->base); + ASSERT_EQ(mkdir(self->victim, 0755), 0); + ASSERT_EQ(mount("tmpfs", self->victim, "tmpfs", 0, NULL), 0); +} + +FIXTURE_TEARDOWN(umount_propagation) +{ + if (self->mounted) + umount2(self->base, MNT_DETACH); + rmdir(self->base); +} + +static int send_msg(int fd, char msg) +{ + return write(fd, &msg, 1) == 1 ? 0 : -1; +} + +static char recv_msg(int fd) +{ + char msg; + + if (read(fd, &msg, 1) != 1) + return 0; + return msg; +} + +/* Is the mount with id @mnt_id attached in this mount namespace? */ +static bool mount_attached(__u64 mnt_id) +{ + char line[4096]; + bool found = false; + FILE *f; + + f = fopen("/proc/self/mountinfo", "re"); + if (!f) + return false; + + while (fgets(line, sizeof(line), f)) { + if (strtoull(line, NULL, 10) == mnt_id) { + found = true; + break; + } + } + fclose(f); + return found; +} + +/* + * The slave namespace: take a detached copy of the shared tree and move it + * beneath the propagated copy of the victim, keeping the open_tree() + * descriptor as a reference on it. + */ +static int slave_child(const char *base, const char *victim, int to_parent, + int from_parent) +{ + struct statx stx; + int fd; + + if (unshare(CLONE_NEWNS)) + return CHILD_UNSHARE; + if (mount("", "/", NULL, MS_REC | MS_SLAVE, NULL)) + return CHILD_UNSHARE; + + fd = sys_open_tree(AT_FDCWD, base, + OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC | AT_RECURSIVE); + if (fd < 0) + return CHILD_OPEN_TREE; + if (sys_move_mount(fd, "", AT_FDCWD, victim, + MOVE_MOUNT_F_EMPTY_PATH | MOVE_MOUNT_BENEATH)) + return CHILD_MOVE_MOUNT; + if (statx(fd, "", AT_EMPTY_PATH, STATX_MNT_ID, &stx)) + return CHILD_STATX; + + if (send_msg(to_parent, MSG_READY) || recv_msg(from_parent) != MSG_CHECK) + return CHILD_PIPE; + if (send_msg(to_parent, mount_attached(stx.stx_mnt_id) ? + MSG_ATTACHED : MSG_DETACHED)) + return CHILD_PIPE; + + if (recv_msg(from_parent) != MSG_CLOSE) + return CHILD_PIPE; + close(fd); + if (send_msg(to_parent, MSG_CLOSED) || recv_msg(from_parent) != MSG_EXIT) + return CHILD_PIPE; + return CHILD_OK; +} + +TEST_F(umount_propagation, busy_copy_pulled_out) +{ + int to_child[2], to_parent[2]; + int status; + pid_t pid; + + ASSERT_EQ(pipe(to_child), 0); + ASSERT_EQ(pipe(to_parent), 0); + + pid = fork(); + ASSERT_GE(pid, 0); + if (pid == 0) { + close(to_child[1]); + close(to_parent[0]); + _exit(slave_child(self->base, self->victim, to_parent[1], + to_child[0])); + } + close(to_child[0]); + close(to_parent[1]); + + ASSERT_EQ(recv_msg(to_parent[0]), MSG_READY); + + /* the copy in the slave namespace is in use */ + ASSERT_EQ(umount2(self->victim, 0), -1); + ASSERT_EQ(errno, EBUSY); + + ASSERT_EQ(send_msg(to_child[1], MSG_CHECK), 0); + ASSERT_EQ(recv_msg(to_parent[0]), MSG_ATTACHED); + + /* and once it is not, the umount goes through */ + ASSERT_EQ(send_msg(to_child[1], MSG_CLOSE), 0); + ASSERT_EQ(recv_msg(to_parent[0]), MSG_CLOSED); + ASSERT_EQ(umount2(self->victim, 0), 0); + + ASSERT_EQ(send_msg(to_child[1], MSG_EXIT), 0); + ASSERT_EQ(waitpid(pid, &status, 0), pid); + ASSERT_TRUE(WIFEXITED(status)); + ASSERT_EQ(WEXITSTATUS(status), CHILD_OK); +} + +TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/move_mount_set_group/move_mount_set_group_test.c b/tools/testing/selftests/move_mount_set_group/move_mount_set_group_test.c index 12434415ec36..9c8fc8c7f62c 100644 --- a/tools/testing/selftests/move_mount_set_group/move_mount_set_group_test.c +++ b/tools/testing/selftests/move_mount_set_group/move_mount_set_group_test.c @@ -146,17 +146,19 @@ static void null_endofword(char *word) *word = '\0'; } -static bool is_shared_mount(const char *path) +/* Does the mount on @path carry the optional field @field in mountinfo? */ +static bool mount_has_field(const char *path, const char *field) { size_t len = 0; char *line = NULL; FILE *f = NULL; + bool found = false; f = fopen("/proc/self/mountinfo", "re"); if (!f) return false; - while (getline(&line, &len, f) != -1) { + while (!found && getline(&line, &len, f) != -1) { char *opts, *target; target = get_field(line, 4); @@ -172,15 +174,29 @@ static bool is_shared_mount(const char *path) if (strcmp(target, path) != 0) continue; - null_endofword(opts); - if (strstr(opts, "shared:")) - return true; + /* the optional fields end at the "-" separator */ + while (opts && *opts != '-') { + char *next = strchr(opts, ' '); + + if (next) + *next++ = '\0'; + if (!strncmp(opts, field, strlen(field))) { + found = true; + break; + } + opts = next; + } } free(line); fclose(f); - return false; + return found; +} + +static bool is_shared_mount(const char *path) +{ + return mount_has_field(path, "shared:"); } /* Attempt to de-conflict with the selftests tree. */ @@ -372,4 +388,50 @@ TEST_F(move_mount_set_group, complex_sharing_copying) ASSERT_EQ(is_shared_mount(SET_GROUP_A), 1); } +#define SET_GROUP_B "/tmp/B" +#define SET_GROUP_C "/tmp/C" + +/* + * An unbindable mount is neither shared nor a slave, so it must not be + * accepted as the target: with a slave source it would end up unbindable + * and a slave at the same time. + */ +TEST_F(move_mount_set_group, unbindable_target) +{ + bool ret; + + ret = move_mount_set_group_supported(); + ASSERT_GE(ret, 0); + if (!ret) + SKIP(return, "move_mount(MOVE_MOUNT_SET_GROUP) is not supported"); + + ASSERT_EQ(mount(NULL, SET_GROUP_A, NULL, MS_SHARED, 0), 0); + + /* B: a slave of A's peer group */ + ASSERT_EQ(mkdir(SET_GROUP_B, 0777), 0); + ASSERT_EQ(mount(SET_GROUP_A, SET_GROUP_B, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, SET_GROUP_B, NULL, MS_SLAVE, 0), 0); + ASSERT_TRUE(mount_has_field(SET_GROUP_B, "master:")); + + /* C: unbindable */ + ASSERT_EQ(mkdir(SET_GROUP_C, 0777), 0); + ASSERT_EQ(mount(SET_GROUP_A, SET_GROUP_C, NULL, MS_BIND, NULL), 0); + ASSERT_EQ(mount(NULL, SET_GROUP_C, NULL, MS_UNBINDABLE, 0), 0); + ASSERT_TRUE(mount_has_field(SET_GROUP_C, "unbindable")); + + /* from a slave */ + ASSERT_EQ(syscall(__NR_move_mount, AT_FDCWD, SET_GROUP_B, + AT_FDCWD, SET_GROUP_C, MOVE_MOUNT_SET_GROUP), -1); + ASSERT_EQ(errno, EINVAL); + ASSERT_FALSE(mount_has_field(SET_GROUP_C, "master:")); + ASSERT_TRUE(mount_has_field(SET_GROUP_C, "unbindable")); + + /* from a shared mount */ + ASSERT_EQ(syscall(__NR_move_mount, AT_FDCWD, SET_GROUP_A, + AT_FDCWD, SET_GROUP_C, MOVE_MOUNT_SET_GROUP), -1); + ASSERT_EQ(errno, EINVAL); + ASSERT_FALSE(mount_has_field(SET_GROUP_C, "shared:")); + ASSERT_TRUE(mount_has_field(SET_GROUP_C, "unbindable")); +} + TEST_HARNESS_MAIN diff --git a/tools/testing/selftests/pidfd/pidfd_open_test.c b/tools/testing/selftests/pidfd/pidfd_open_test.c index 318e6f09c8e0..c6698b7cdb47 100644 --- a/tools/testing/selftests/pidfd/pidfd_open_test.c +++ b/tools/testing/selftests/pidfd/pidfd_open_test.c @@ -168,7 +168,7 @@ int main(int argc, char **argv) } if (info.ppid != getppid()) { ksft_print_msg("ppid %d does not match ppid from ioctl %d\n", - pid, info.pid); + getppid(), info.ppid); goto on_error; } if (info.ruid != getuid()) { diff --git a/virt/kvm/guest_memfd.c b/virt/kvm/guest_memfd.c index 625e62e1a031..5c084d5393a6 100644 --- a/virt/kvm/guest_memfd.c +++ b/virt/kvm/guest_memfd.c @@ -506,7 +506,7 @@ static const struct address_space_operations kvm_gmem_aops = { #endif }; -static int kvm_gmem_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int kvm_gmem_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return -EINVAL; |
