summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorMark Brown <broonie@kernel.org>2026-07-26 21:52:04 +0100
committerMark Brown <broonie@kernel.org>2026-07-26 21:52:04 +0100
commitc7b3ca61f139b647b0d6313b82118a8273750995 (patch)
treef0409329e9d795bf4c7a23d7e4a870a69078a8c0
parent5dd91aac4216c19e9eeea6e0c1a04bcb3e43a876 (diff)
parentbc829bbcf423192471f17b725e8536aed2714188 (diff)
downloadlinux-next-c7b3ca61f139b647b0d6313b82118a8273750995.tar.gz
linux-next-c7b3ca61f139b647b0d6313b82118a8273750995.zip
Merge branch 'fs-next' of linux-next
-rw-r--r--Documentation/admin-guide/binfmt-misc.rst67
-rw-r--r--Documentation/admin-guide/ext4.rst8
-rw-r--r--Documentation/block/inline-encryption.rst43
-rw-r--r--Documentation/filesystems/ext4/group_descr.rst9
-rw-r--r--Documentation/filesystems/f2fs.rst10
-rw-r--r--Documentation/filesystems/fscrypt.rst154
-rw-r--r--Documentation/filesystems/locking.rst2
-rw-r--r--Documentation/filesystems/overlayfs.rst16
-rw-r--r--Documentation/filesystems/porting.rst10
-rw-r--r--Documentation/filesystems/vfs.rst2
-rw-r--r--Documentation/sunrpc/xdr/nfs4_1.x258
-rw-r--r--MAINTAINERS11
-rw-r--r--arch/loongarch/configs/loongson32_defconfig1
-rw-r--r--arch/loongarch/configs/loongson64_defconfig1
-rw-r--r--block/bdev.c125
-rw-r--r--block/blk-crypto-fallback.c3
-rw-r--r--block/blk-crypto-internal.h3
-rw-r--r--block/blk-crypto-profile.c22
-rw-r--r--block/blk-crypto.c46
-rw-r--r--drivers/base/devtmpfs.c2
-rw-r--r--drivers/block/rnbd/rnbd-srv.c4
-rw-r--r--drivers/char/misc_minor_kunit.c25
-rw-r--r--drivers/crypto/ccp/sev-dev.c12
-rw-r--r--drivers/md/dm-inlinecrypt.c3
-rw-r--r--drivers/mtd/ubi/ubi.h9
-rw-r--r--drivers/target/target_core_alua.c11
-rw-r--r--drivers/target/target_core_pr.c4
-rw-r--r--fs/9p/vfs_inode.c5
-rw-r--r--fs/9p/vfs_inode_dotl.c4
-rw-r--r--fs/Kconfig1
-rw-r--r--fs/Kconfig.binfmt14
-rw-r--r--fs/Makefile2
-rw-r--r--fs/affs/affs.h2
-rw-r--r--fs/affs/namei.c4
-rw-r--r--fs/affs/super.c2
-rw-r--r--fs/afs/dir.c6
-rw-r--r--fs/autofs/root.c2
-rw-r--r--fs/bad_inode.c2
-rw-r--r--fs/bfs/dir.c2
-rw-r--r--fs/binfmt_misc.c1047
-rw-r--r--fs/binfmt_misc_bpf.c354
-rw-r--r--fs/bpf_fs_kfuncs.c65
-rw-r--r--fs/btrfs/Kconfig1
-rw-r--r--fs/btrfs/block-group.c13
-rw-r--r--fs/btrfs/block-group.h4
-rw-r--r--fs/btrfs/compression.c14
-rw-r--r--fs/btrfs/defrag.c61
-rw-r--r--fs/btrfs/dev-replace.c65
-rw-r--r--fs/btrfs/direct-io.c212
-rw-r--r--fs/btrfs/disk-io.c47
-rw-r--r--fs/btrfs/extent-tree.c2
-rw-r--r--fs/btrfs/extent_io.c178
-rw-r--r--fs/btrfs/extent_io.h5
-rw-r--r--fs/btrfs/fiemap.c2
-rw-r--r--fs/btrfs/file.c114
-rw-r--r--fs/btrfs/fs.c19
-rw-r--r--fs/btrfs/fs.h6
-rw-r--r--fs/btrfs/inode.c59
-rw-r--r--fs/btrfs/ioctl.c138
-rw-r--r--fs/btrfs/raid56.c18
-rw-r--r--fs/btrfs/raid56.h2
-rw-r--r--fs/btrfs/reflink.c6
-rw-r--r--fs/btrfs/relocation.c95
-rw-r--r--fs/btrfs/scrub.c232
-rw-r--r--fs/btrfs/send.c15
-rw-r--r--fs/btrfs/space-info.c2
-rw-r--r--fs/btrfs/super.c35
-rw-r--r--fs/btrfs/transaction.c20
-rw-r--r--fs/btrfs/transaction.h25
-rw-r--r--fs/btrfs/volumes.c180
-rw-r--r--fs/btrfs/volumes.h16
-rw-r--r--fs/buffer.c48
-rw-r--r--fs/ceph/dir.c3
-rw-r--r--fs/coda/dir.c9
-rw-r--r--fs/coredump.c11
-rw-r--r--fs/cramfs/inode.c2
-rw-r--r--fs/crypto/Kconfig8
-rw-r--r--fs/crypto/Makefile3
-rw-r--r--fs/crypto/bio.c216
-rw-r--r--fs/crypto/block.c (renamed from fs/crypto/inline_crypt.c)279
-rw-r--r--fs/crypto/crypto.c180
-rw-r--r--fs/crypto/fscrypt_private.h33
-rw-r--r--fs/crypto/keyring.c21
-rw-r--r--fs/crypto/keysetup.c63
-rw-r--r--fs/crypto/keysetup_v1.c2
-rw-r--r--fs/crypto/policy.c34
-rw-r--r--fs/ecryptfs/crypto.c2
-rw-r--r--fs/ecryptfs/ecryptfs_kernel.h15
-rw-r--r--fs/ecryptfs/inode.c2
-rw-r--r--fs/ecryptfs/keystore.c45
-rw-r--r--fs/ecryptfs/messaging.c11
-rw-r--r--fs/ecryptfs/miscdev.c5
-rw-r--r--fs/ecryptfs/mmap.c30
-rw-r--r--fs/ecryptfs/super.c7
-rw-r--r--fs/efivarfs/inode.c2
-rw-r--r--fs/efs/Kconfig16
-rw-r--r--fs/efs/Makefile8
-rw-r--r--fs/efs/dir.c105
-rw-r--r--fs/efs/efs.h144
-rw-r--r--fs/efs/file.c42
-rw-r--r--fs/efs/inode.c315
-rw-r--r--fs/efs/namei.c120
-rw-r--r--fs/efs/super.c368
-rw-r--r--fs/efs/symlink.c50
-rw-r--r--fs/erofs/Kconfig15
-rw-r--r--fs/erofs/decompressor_lzma.c3
-rw-r--r--fs/erofs/internal.h4
-rw-r--r--fs/erofs/ishare.c60
-rw-r--r--fs/erofs/super.c108
-rw-r--r--fs/exec.c2
-rw-r--r--fs/exfat/exfat_fs.h2
-rw-r--r--fs/exfat/file.c180
-rw-r--r--fs/exfat/iomap.c11
-rw-r--r--fs/exfat/namei.c2
-rw-r--r--fs/ext2/namei.c4
-rw-r--r--fs/ext4/crypto.c2
-rw-r--r--fs/ext4/dir.c20
-rw-r--r--fs/ext4/ext4.h41
-rw-r--r--fs/ext4/ext4_jbd2.c36
-rw-r--r--fs/ext4/ext4_jbd2.h6
-rw-r--r--fs/ext4/extents-test.c9
-rw-r--r--fs/ext4/extents.c39
-rw-r--r--fs/ext4/fast_commit.c37
-rw-r--r--fs/ext4/file.c148
-rw-r--r--fs/ext4/inline.c38
-rw-r--r--fs/ext4/inode.c242
-rw-r--r--fs/ext4/mballoc-test.c9
-rw-r--r--fs/ext4/mballoc.c21
-rw-r--r--fs/ext4/migrate.c3
-rw-r--r--fs/ext4/namei.c6
-rw-r--r--fs/ext4/orphan.c15
-rw-r--r--fs/ext4/page-io.c74
-rw-r--r--fs/ext4/readpage.c139
-rw-r--r--fs/ext4/super.c48
-rw-r--r--fs/ext4/xattr.c157
-rw-r--r--fs/ext4/xattr.h9
-rw-r--r--fs/f2fs/compress.c31
-rw-r--r--fs/f2fs/data.c100
-rw-r--r--fs/f2fs/debug.c4
-rw-r--r--fs/f2fs/f2fs.h31
-rw-r--r--fs/f2fs/file.c34
-rw-r--r--fs/f2fs/gc.c36
-rw-r--r--fs/f2fs/gc.h27
-rw-r--r--fs/f2fs/namei.c4
-rw-r--r--fs/f2fs/segment.c11
-rw-r--r--fs/f2fs/super.c14
-rw-r--r--fs/f2fs/sysfs.c22
-rw-r--r--fs/fat/namei_msdos.c5
-rw-r--r--fs/fat/namei_vfat.c2
-rw-r--r--fs/fs_struct.c103
-rw-r--r--fs/fuse/dir.c10
-rw-r--r--fs/gfs2/inode.c7
-rw-r--r--fs/hfs/dir.c4
-rw-r--r--fs/hfs/super.c2
-rw-r--r--fs/hfsplus/dir.c4
-rw-r--r--fs/hfsplus/super.c2
-rw-r--r--fs/hostfs/hostfs_kern.c2
-rw-r--r--fs/hpfs/namei.c6
-rw-r--r--fs/hugetlbfs/inode.c4
-rw-r--r--fs/inode.c23
-rw-r--r--fs/internal.h1
-rw-r--r--fs/iomap/buffered-io.c6
-rw-r--r--fs/iomap/direct-io.c293
-rw-r--r--fs/iomap/ioend.c21
-rw-r--r--fs/isofs/compress.c6
-rw-r--r--fs/jbd2/checkpoint.c28
-rw-r--r--fs/jbd2/journal.c1
-rw-r--r--fs/jffs2/dir.c6
-rw-r--r--fs/jffs2/wbuf.c2
-rw-r--r--fs/jfs/namei.c4
-rw-r--r--fs/kernel_read_file.c9
-rw-r--r--fs/lockd/lockd.h2
-rw-r--r--fs/lockd/svc.c4
-rw-r--r--fs/lockd/svc4proc.c7
-rw-r--r--fs/lockd/svclock.c42
-rw-r--r--fs/lockd/svcproc.c4
-rw-r--r--fs/lockd/svcsubs.c109
-rw-r--r--fs/minix/namei.c4
-rw-r--r--fs/namei.c30
-rw-r--r--fs/namespace.c25
-rw-r--r--fs/nfs/blocklayout/dev.c15
-rw-r--r--fs/nfs/callback.c4
-rw-r--r--fs/nfs/dir.c6
-rw-r--r--fs/nfs/internal.h2
-rw-r--r--fs/nfs/nfs4proc.c17
-rw-r--r--fs/nfs_common/nfslocalio.c16
-rw-r--r--fs/nfsd/filecache.c265
-rw-r--r--fs/nfsd/filecache.h3
-rw-r--r--fs/nfsd/flexfilelayoutxdr.c20
-rw-r--r--fs/nfsd/localio.c8
-rw-r--r--fs/nfsd/lockd.c6
-rw-r--r--fs/nfsd/netns.h29
-rw-r--r--fs/nfsd/nfs2acl.c21
-rw-r--r--fs/nfsd/nfs3acl.c17
-rw-r--r--fs/nfsd/nfs3proc.c42
-rw-r--r--fs/nfsd/nfs3xdr.c2
-rw-r--r--fs/nfsd/nfs4callback.c207
-rw-r--r--fs/nfsd/nfs4layouts.c45
-rw-r--r--fs/nfsd/nfs4proc.c115
-rw-r--r--fs/nfsd/nfs4recover.c48
-rw-r--r--fs/nfsd/nfs4state.c938
-rw-r--r--fs/nfsd/nfs4xdr.c396
-rw-r--r--fs/nfsd/nfs4xdr_gen.c600
-rw-r--r--fs/nfsd/nfs4xdr_gen.h17
-rw-r--r--fs/nfsd/nfscache.c4
-rw-r--r--fs/nfsd/nfsctl.c160
-rw-r--r--fs/nfsd/nfsd.h12
-rw-r--r--fs/nfsd/nfsfh.c42
-rw-r--r--fs/nfsd/nfsfh.h3
-rw-r--r--fs/nfsd/nfsproc.c9
-rw-r--r--fs/nfsd/nfssvc.c61
-rw-r--r--fs/nfsd/nfsxdr.c32
-rw-r--r--fs/nfsd/state.h88
-rw-r--r--fs/nfsd/trace.h42
-rw-r--r--fs/nfsd/vfs.c61
-rw-r--r--fs/nfsd/xdr4.h5
-rw-r--r--fs/nfsd/xdr4cb.h12
-rw-r--r--fs/nilfs2/namei.c4
-rw-r--r--fs/notify/fanotify/fanotify.c1
-rw-r--r--fs/ntfs/attrib.c15
-rw-r--r--fs/ntfs/compress.c387
-rw-r--r--fs/ntfs/dir.c5
-rw-r--r--fs/ntfs/ea.c188
-rw-r--r--fs/ntfs/file.c82
-rw-r--r--fs/ntfs/index.c11
-rw-r--r--fs/ntfs/inode.c6
-rw-r--r--fs/ntfs/inode.h2
-rw-r--r--fs/ntfs/iomap.c34
-rw-r--r--fs/ntfs/mft.c3
-rw-r--r--fs/ntfs/namei.c13
-rw-r--r--fs/ntfs/reparse.c6
-rw-r--r--fs/ntfs/runlist.c50
-rw-r--r--fs/ntfs3/attrib.c2
-rw-r--r--fs/ntfs3/dir.c41
-rw-r--r--fs/ntfs3/file.c2
-rw-r--r--fs/ntfs3/frecord.c31
-rw-r--r--fs/ntfs3/fslog.c21
-rw-r--r--fs/ntfs3/fsntfs.c8
-rw-r--r--fs/ntfs3/index.c7
-rw-r--r--fs/ntfs3/inode.c18
-rw-r--r--fs/ntfs3/lznt.c4
-rw-r--r--fs/ntfs3/namei.c13
-rw-r--r--fs/ntfs3/ntfs_fs.h6
-rw-r--r--fs/ntfs3/run.c9
-rw-r--r--fs/ntfs3/super.c11
-rw-r--r--fs/ntfs3/xattr.c1
-rw-r--r--fs/nullfs.c14
-rw-r--r--fs/ocfs2/dlmfs/dlmfs.c5
-rw-r--r--fs/ocfs2/namei.c5
-rw-r--r--fs/ocfs2/super.c1
-rw-r--r--fs/omfs/dir.c4
-rw-r--r--fs/orangefs/namei.c5
-rw-r--r--fs/overlayfs/dir.c18
-rw-r--r--fs/overlayfs/inode.c26
-rw-r--r--fs/overlayfs/overlayfs.h1
-rw-r--r--fs/overlayfs/super.c2
-rw-r--r--fs/overlayfs/xattrs.c1
-rw-r--r--fs/posix_acl.c2
-rw-r--r--fs/proc/array.c4
-rw-r--r--fs/proc/base.c8
-rw-r--r--fs/proc_namespace.c4
-rw-r--r--fs/ramfs/inode.c4
-rw-r--r--fs/romfs/super.c10
-rw-r--r--fs/smb/client/cifs_swn.c6
-rw-r--r--fs/smb/client/cifsacl.c3
-rw-r--r--fs/smb/client/cifsfs.h2
-rw-r--r--fs/smb/client/cifssmb.c12
-rw-r--r--fs/smb/client/dir.c2
-rw-r--r--fs/smb/client/inode.c7
-rw-r--r--fs/smb/client/smb1maperror.c6
-rw-r--r--fs/smb/server/ksmbd_netlink.h1
-rw-r--r--fs/smb/server/mgmt/share_config.c4
-rw-r--r--fs/smb/server/smb2pdu.c169
-rw-r--r--fs/smb/server/smbacl.c111
-rw-r--r--fs/smb/server/smbacl.h3
-rw-r--r--fs/smb/server/vfs.c17
-rw-r--r--fs/super.c639
-rw-r--r--fs/ubifs/dir.c4
-rw-r--r--fs/udf/balloc.c12
-rw-r--r--fs/udf/inode.c26
-rw-r--r--fs/udf/namei.c4
-rw-r--r--fs/udf/partition.c2
-rw-r--r--fs/udf/super.c23
-rw-r--r--fs/udf/symlink.c2
-rw-r--r--fs/udf/truncate.c2
-rw-r--r--fs/udf/udfdecl.h3
-rw-r--r--fs/ufs/namei.c5
-rw-r--r--fs/ufs/super.c2
-rw-r--r--fs/vboxsf/dir.c4
-rw-r--r--fs/xfs/libxfs/xfs_da_btree.c3
-rw-r--r--fs/xfs/libxfs/xfs_dquot_buf.c39
-rw-r--r--fs/xfs/libxfs/xfs_metadir.c64
-rw-r--r--fs/xfs/libxfs/xfs_metadir.h8
-rw-r--r--fs/xfs/libxfs/xfs_rtgroup.c58
-rw-r--r--fs/xfs/libxfs/xfs_rtrefcount_btree.c4
-rw-r--r--fs/xfs/scrub/agheader.c2
-rw-r--r--fs/xfs/scrub/bmap.c5
-rw-r--r--fs/xfs/scrub/inode_repair.c2
-rw-r--r--fs/xfs/scrub/nlinks_repair.c7
-rw-r--r--fs/xfs/scrub/rtbitmap_repair.c46
-rw-r--r--fs/xfs/scrub/rtsummary.c2
-rw-r--r--fs/xfs/xfs_bmap_util.c42
-rw-r--r--fs/xfs/xfs_bmap_util.h7
-rw-r--r--fs/xfs/xfs_buf.c4
-rw-r--r--fs/xfs/xfs_buf_item_recover.c57
-rw-r--r--fs/xfs/xfs_file.c92
-rw-r--r--fs/xfs/xfs_iops.c7
-rw-r--r--fs/xfs/xfs_mount.h7
-rw-r--r--fs/xfs/xfs_rtalloc.c2
-rw-r--r--fs/xfs/xfs_super.c63
-rw-r--r--include/linux/binfmt_misc.h71
-rw-r--r--include/linux/binfmts.h12
-rw-r--r--include/linux/blk-crypto.h15
-rw-r--r--include/linux/blk_types.h2
-rw-r--r--include/linux/blkdev.h12
-rw-r--r--include/linux/efs_vh.h54
-rw-r--r--include/linux/fs.h19
-rw-r--r--include/linux/fs/super.h8
-rw-r--r--include/linux/fs/super_types.h6
-rw-r--r--include/linux/fs_struct.h34
-rw-r--r--include/linux/fscrypt.h107
-rw-r--r--include/linux/init_task.h1
-rw-r--r--include/linux/jbd2.h11
-rw-r--r--include/linux/lockd/bind.h12
-rw-r--r--include/linux/net.h1
-rw-r--r--include/linux/nfs4.h127
-rw-r--r--include/linux/sched.h1
-rw-r--r--include/linux/sched/task.h1
-rw-r--r--include/linux/sunrpc/bc_xprt.h5
-rw-r--r--include/linux/sunrpc/svc.h1
-rw-r--r--include/linux/sunrpc/svc_rdma_pcl.h4
-rw-r--r--include/linux/sunrpc/xdrgen/nfs4_1.h290
-rw-r--r--include/linux/types.h2
-rw-r--r--include/uapi/asm-generic/errno.h2
-rw-r--r--include/uapi/linux/efs_fs_sb.h63
-rw-r--r--include/uapi/linux/fscrypt.h1
-rw-r--r--include/uapi/linux/nfs4.h2
-rw-r--r--init/init_task.c1
-rw-r--r--init/initramfs.c14
-rw-r--r--init/initramfs_test.c222
-rw-r--r--init/main.c10
-rw-r--r--ipc/mqueue.c7
-rw-r--r--kernel/exit.c2
-rw-r--r--kernel/fork.c53
-rw-r--r--kernel/kcmp.c2
-rw-r--r--kernel/umh.c6
-rw-r--r--kernel/user.c4
-rw-r--r--mm/shmem.c2
-rw-r--r--net/socket.c25
-rw-r--r--net/sunrpc/auth_gss/auth_gss.c6
-rw-r--r--net/sunrpc/auth_gss/gss_krb5_unseal.c3
-rw-r--r--net/sunrpc/auth_gss/gss_krb5_wrap.c13
-rw-r--r--net/sunrpc/auth_gss/gss_rpc_upcall.c6
-rw-r--r--net/sunrpc/auth_gss/gss_rpc_upcall.h1
-rw-r--r--net/sunrpc/auth_gss/gss_rpc_xdr.c15
-rw-r--r--net/sunrpc/auth_gss/svcauth_gss.c8
-rw-r--r--net/sunrpc/backchannel_rqst.c38
-rw-r--r--net/sunrpc/cache.c7
-rw-r--r--net/sunrpc/sunrpc_syms.c1
-rw-r--r--net/sunrpc/svc.c81
-rw-r--r--net/sunrpc/svcauth_unix.c4
-rw-r--r--net/sunrpc/xdr.c2
-rw-r--r--net/sunrpc/xprtrdma/ib_client.c40
-rw-r--r--net/sunrpc/xprtrdma/svc_rdma_pcl.c63
-rw-r--r--net/sunrpc/xprtrdma/svc_rdma_recvfrom.c24
-rw-r--r--net/sunrpc/xprtrdma/svc_rdma_rw.c52
-rw-r--r--net/sunrpc/xprtrdma/svc_rdma_sendto.c47
-rw-r--r--net/sunrpc/xprtrdma/svc_rdma_transport.c69
-rw-r--r--net/unix/af_unix.c17
-rw-r--r--tools/include/uapi/linux/fscrypt.h1
-rw-r--r--tools/testing/selftests/bpf/bpf_experimental.h3
-rw-r--r--tools/testing/selftests/bpf/prog_tests/sock_xattr.c67
-rw-r--r--tools/testing/selftests/bpf/progs/sock_read_xattr.c54
-rw-r--r--tools/testing/selftests/exec/.gitignore5
-rw-r--r--tools/testing/selftests/exec/Makefile48
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_app.c12
-rw-r--r--tools/testing/selftests/exec/binfmt_bpf_interp.c15
-rw-r--r--tools/testing/selftests/exec/binfmt_misc_bpf.c277
-rw-r--r--tools/testing/selftests/exec/bpf_interp.bpf.c61
-rw-r--r--tools/testing/selftests/exec/nix_origin.bpf.c224
-rw-r--r--tools/testing/selftests/filesystems/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/.gitignore1
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/Makefile2
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c501
-rw-r--r--tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c16
-rw-r--r--tools/testing/selftests/filesystems/statmount/statmount_test.c5
-rw-r--r--tools/testing/selftests/filesystems/ustat_test.c135
-rw-r--r--tools/testing/selftests/proc/proc-pidns.c1
389 files changed, 11699 insertions, 6238 deletions
diff --git a/Documentation/admin-guide/binfmt-misc.rst b/Documentation/admin-guide/binfmt-misc.rst
index c0a34fbf8022..7f42abf9cfba 100644
--- a/Documentation/admin-guide/binfmt-misc.rst
+++ b/Documentation/admin-guide/binfmt-misc.rst
@@ -26,7 +26,8 @@ Here is what the fields mean:
name below ``/proc/sys/fs/binfmt_misc``; cannot contain slashes ``/`` for
obvious reasons.
- ``type``
- is the type of recognition. Give ``M`` for magic and ``E`` for extension.
+ is the type of recognition. Give ``M`` for magic, ``E`` for extension and
+ ``B`` for a bpf-backed handler (see below).
- ``offset``
is the offset of the magic/mask in the file, counted in bytes. This
defaults to 0 if you omit it (i.e. you write ``:name:type::magic...``).
@@ -48,7 +49,8 @@ Here is what the fields mean:
filename extension matching.
- ``interpreter``
is the program that should be invoked with the binary as first
- argument (specify the full path)
+ argument (specify the full path). For ``B`` entries this field
+ carries the name of the bpf handler instead (see below).
- ``flags``
is an optional field that controls several aspects of the invocation
of the interpreter. It is a string of capital letters, each controls a
@@ -97,6 +99,64 @@ There are some restrictions:
offset+size(magic) has to be less than 128
- the interpreter string may not exceed 127 characters
+
+bpf-backed handlers
+-------------------
+
+With ``CONFIG_BINFMT_MISC_BPF`` both the matching and the interpreter
+selection can be delegated to bpf programs. A handler is an instance of the
+``binfmt_misc_ops`` struct_ops with a ``match`` and a ``load`` program and a
+``name``. Once the struct_ops map is registered the handler can be activated
+with a ``B`` entry that references it by name in the ``interpreter`` field
+and carries neither offset, magic, nor mask::
+
+ echo ':qemu:B::::my_handler:' > register
+
+Both programs receive the ``linux_binprm`` of the binary and both can
+sleep. The ``match`` program decides whether the handler applies: it is
+consulted during the entry walk exactly like magic and extension matching,
+in the same registration order with the same first-match-wins semantics.
+Unlike static matching it is not limited to the prefetched first bytes of
+the file in ``bprm->buf``: it can read the file, e.g. to parse ELF program
+headers whose data sits at arbitrary offsets. It only decides, though: the
+selection kfuncs below are rejected in it. The ``load`` program of the
+matched handler then selects the interpreter: it can equally read the file
+and derive the interpreter from the binary's location. It selects the
+interpreter by calling the ``bpf_binprm_set_interp()`` kfunc with an
+absolute path and returning ``0``. A match is committed: a failing
+``load`` fails the exec with its error instead of falling through to later
+entries; ``-ENOEXEC`` lets the remaining binary formats have a go. The
+interpreter is opened with the credentials of the task doing the exec,
+exactly as a statically registered interpreter would be.
+
+The ``load`` program can also pass a single argument to the interpreter with
+the ``bpf_binprm_set_interp_arg()`` kfunc. It is inserted between the
+interpreter and the binary, exactly like the optional argument of a ``#!``
+interpreter line, e.g. for a handler that resolves ``$ORIGIN`` in a script's
+``#!`` path and needs to preserve the argument that followed it.
+
+The invocation flags a static entry fixes at registration - ``P``, ``C``
+and ``O`` - are per-exec choices for a bpf handler, made by the ``load``
+program with the ``bpf_binprm_set_flags()`` kfunc, so a single handler can
+decide them differently for each binary it handles:
+
+- ``BPF_BINPRM_PRESERVE_ARGV0`` keeps the caller's ``argv[0]`` (the ``P``
+ flag).
+- ``BPF_BINPRM_CREDENTIALS`` computes credentials from the binary (the ``C``
+ flag), bounded to user namespaces that map the binary's owner just like
+ any other setuid exec.
+- ``BPF_BINPRM_EXECFD`` opens the binary on the interpreter's behalf and
+ passes it through the ``AT_EXECFD`` aux vector entry (the ``O`` flag), so
+ the interpreter can run binaries it could not open by path.
+
+Because these are program choices, a ``B`` entry carries no flags in the
+register string; ``F`` (pre-open a fixed interpreter) has no meaning for it.
+
+Handlers are looked up in the user namespace the struct_ops map was
+registered in, falling back to ancestor namespaces, mirroring how
+binfmt_misc instances themselves are looked up. The entry keeps the handler
+alive; deleting the struct_ops map only prevents new activations.
+
To use binfmt_misc you have to mount it first. You can mount it with
``mount -t binfmt_misc none /proc/sys/fs/binfmt_misc`` command, or you can add
a line ``none /proc/sys/fs/binfmt_misc binfmt_misc defaults 0 0`` to your
@@ -133,7 +193,8 @@ or 1 (to enable) to ``/proc/sys/fs/binfmt_misc/status`` or
Catting the file tells you the current status of ``binfmt_misc/the_entry``.
You can remove one entry or all entries by echoing -1 to ``/proc/.../the_name``
-or ``/proc/sys/fs/binfmt_misc/status``.
+or ``/proc/sys/fs/binfmt_misc/status``. A single entry can also be removed
+by simply unlinking (``rm``) ``/proc/.../the_name``.
Hints
diff --git a/Documentation/admin-guide/ext4.rst b/Documentation/admin-guide/ext4.rst
index ac0c709ea9e7..742a48e6fc0c 100644
--- a/Documentation/admin-guide/ext4.rst
+++ b/Documentation/admin-guide/ext4.rst
@@ -385,11 +385,9 @@ When mounting an ext4 filesystem, the following option are accepted:
incompatible with data=journal.
inlinecrypt
- When possible, encrypt/decrypt the contents of encrypted files using the
- blk-crypto framework rather than filesystem-layer encryption. This
- allows the use of inline encryption hardware. The on-disk format is
- unaffected. For more details, see
- Documentation/block/inline-encryption.rst.
+ When possible, encrypt/decrypt the contents of encrypted files using
+ inline encryption hardware rather than the CPU. For more details, see
+ Documentation/filesystems/fscrypt.rst.
Data Mode
=========
diff --git a/Documentation/block/inline-encryption.rst b/Documentation/block/inline-encryption.rst
index cae23949a626..0df964507f76 100644
--- a/Documentation/block/inline-encryption.rst
+++ b/Documentation/block/inline-encryption.rst
@@ -37,12 +37,12 @@ initialization vector for each sector, and can be tested for correctness.
Objective
=========
-We want to support inline encryption in the kernel. To make testing easier, we
-also want support for falling back to the kernel crypto API when actual inline
-encryption hardware is absent. We also want inline encryption to work with
-layered devices like device-mapper and loopback (i.e. we want to be able to use
-the inline encryption hardware of the underlying devices if present, or else
-fall back to crypto API en/decryption).
+We want to support inline encryption hardware in the kernel. The API for using
+such hardware should also support a fallback to the CPU, so that users only need
+to use a single API and more of the code can be tested without actual hardware.
+We also want inline encryption to work with layered devices like device-mapper
+and loopback (i.e. we want to be able to use the inline encryption hardware of
+the underlying devices if present, or else fall back to the CPU).
Constraints and notes
=====================
@@ -185,20 +185,12 @@ blk-crypto-fallback is optional and is controlled by the
API presented to users of the block layer
=========================================
-``blk_crypto_config_supported()`` allows users to check ahead of time whether
-inline encryption with particular crypto settings will work on a particular
-block_device -- either via hardware or via blk-crypto-fallback. This function
-takes in a ``struct blk_crypto_config`` which is like blk_crypto_key, but omits
-the actual bytes of the key and instead just contains the algorithm, data unit
-size, etc. This function can be useful if blk-crypto-fallback is disabled.
-
``blk_crypto_init_key()`` allows users to initialize a blk_crypto_key.
Users must call ``blk_crypto_start_using_key()`` before actually starting to use
-a blk_crypto_key on a block_device (even if ``blk_crypto_config_supported()``
-was called earlier). This is needed to initialize blk-crypto-fallback if it
-will be needed. This must not be called from the data path, as this may have to
-allocate resources, which may deadlock in that case.
+a blk_crypto_key on a block_device. This is needed to initialize
+blk-crypto-fallback if it will be needed. This must not be called from the data
+path, as this may have to allocate resources, which may deadlock in that case.
Next, to attach an encryption context to a bio, users should call
``bio_crypt_set_ctx()``. This function allocates a bio_crypt_ctx and attaches
@@ -220,16 +212,15 @@ any kernel data structures it may be linked into.
In summary, for users of the block layer, the lifecycle of a blk_crypto_key is
as follows:
-1. ``blk_crypto_config_supported()`` (optional)
-2. ``blk_crypto_init_key()``
-3. ``blk_crypto_start_using_key()``
-4. ``bio_crypt_set_ctx()`` (potentially many times)
-5. ``blk_crypto_evict_key()`` (after all I/O has completed)
-6. Zeroize the blk_crypto_key (this has no dedicated function)
+1. ``blk_crypto_init_key()``
+2. ``blk_crypto_start_using_key()``
+3. ``bio_crypt_set_ctx()`` (potentially many times)
+4. ``blk_crypto_evict_key()`` (after all I/O has completed)
+5. Zeroize the blk_crypto_key (this has no dedicated function)
If a blk_crypto_key is being used on multiple block_devices, then
-``blk_crypto_config_supported()`` (if used), ``blk_crypto_start_using_key()``,
-and ``blk_crypto_evict_key()`` must be called on each block_device.
+``blk_crypto_start_using_key()`` and ``blk_crypto_evict_key()`` must be called
+on each block_device.
API presented to device drivers
===============================
@@ -304,7 +295,7 @@ hardware implementations might not implement both features together correctly,
and disallow the combination for now. Whenever a device supports integrity, the
kernel will pretend that the device does not support hardware inline encryption
(by setting the blk_crypto_profile in the request_queue of the device to NULL).
-When the crypto API fallback is enabled, this means that all bios with and
+When the crypto API fallback is enabled, this means that all bios with an
encryption context will use the fallback, and IO will complete as usual. When
the fallback is disabled, a bio with an encryption context will be failed.
diff --git a/Documentation/filesystems/ext4/group_descr.rst b/Documentation/filesystems/ext4/group_descr.rst
index 392ec44f8fb0..9a0c2d92c2ef 100644
--- a/Documentation/filesystems/ext4/group_descr.rst
+++ b/Documentation/filesystems/ext4/group_descr.rst
@@ -20,11 +20,10 @@ group of the flex group.
If the meta_bg feature flag is set, then several block groups are
grouped together into a meta group. Note that in the meta_bg case,
-however, the first and last two block groups within the larger meta
-group contain only group descriptors for the groups inside the meta
-group.
-
-flex_bg and meta_bg do not appear to be mutually exclusive features.
+however, the superblock and a single block group descriptor block is
+placed at the beginning of the first, second, and last block groups in a
+meta-block group. The flex_bg and meta_bg features are not mutually
+exclusive.
In ext2, ext3, and ext4 (when the 64bit feature is not enabled), the
block group descriptor was only 32 bytes long and therefore ends at
diff --git a/Documentation/filesystems/f2fs.rst b/Documentation/filesystems/f2fs.rst
index 8c4a14ae444f..b45d7a687625 100644
--- a/Documentation/filesystems/f2fs.rst
+++ b/Documentation/filesystems/f2fs.rst
@@ -351,12 +351,10 @@ compress_mode=%s Control file compression mode. This supports "fs" and "user"
compress_cache Support to use address space of a filesystem managed inode to
cache compressed block, in order to improve cache hit ratio of
random read.
-inlinecrypt When possible, encrypt/decrypt the contents of encrypted
- files using the blk-crypto framework rather than
- filesystem-layer encryption. This allows the use of
- inline encryption hardware. The on-disk format is
- unaffected. For more details, see
- Documentation/block/inline-encryption.rst.
+inlinecrypt When possible, encrypt/decrypt the contents of
+ encrypted files using inline encryption hardware rather
+ than the CPU. For more details, see
+ Documentation/filesystems/fscrypt.rst.
atgc Enable age-threshold garbage collection, it provides high
effectiveness and efficiency on background GC.
discard_unit=%s Control discard unit, the argument can be "block", "segment"
diff --git a/Documentation/filesystems/fscrypt.rst b/Documentation/filesystems/fscrypt.rst
index c0dd35f1af12..f309337fe110 100644
--- a/Documentation/filesystems/fscrypt.rst
+++ b/Documentation/filesystems/fscrypt.rst
@@ -188,8 +188,8 @@ attacks:
- Non-root users cannot securely remove encryption keys.
All the above problems are fixed with v2 encryption policies. For
-this reason among others, it is recommended to use v2 encryption
-policies on all new encrypted directories.
+this reason among others, v1 encryption policies are deprecated. Use
+v2 encryption policies on all new encrypted directories.
Key hierarchy
=============
@@ -305,7 +305,8 @@ included in the IV. Moreover:
- For v2 encryption policies, the encryption is done with a per-mode
key derived using the KDF. Users may use the same master key for
- other v2 encryption policies.
+ other v2 encryption policies. However, using a distinct master key
+ for each policy is still the best practice and normal usage.
IV_INO_LBLK_64 policies
-----------------------
@@ -336,6 +337,9 @@ per I/O request and may have only a small number of keyslots. This
format results in some level of IV reuse, so it should only be used
when necessary due to hardware limitations.
+IV_INO_LBLK_32 is supported only when the filesystem block size is
+equal to the page size.
+
Key identifiers
---------------
@@ -601,7 +605,9 @@ This structure must be initialized as follows:
struct fscrypt_policy_v1 is used or FSCRYPT_POLICY_V2 (2) if
struct fscrypt_policy_v2 is used. (Note: we refer to the original
policy version as "v1", though its version code is really 0.)
- For new encrypted directories, use v2 policies.
+ For new encrypted directories, use v2 policies, which are supported
+ since Linux v5.4. v1 policies are deprecated and have several
+ usability and security problems.
- ``contents_encryption_mode`` and ``filenames_encryption_mode`` must
be set to constants from ``<linux/fscrypt.h>`` which identify the
@@ -736,17 +742,6 @@ FS_IOC_SET_ENCRYPTION_POLICY can fail with the following errors:
Getting an encryption policy
----------------------------
-Two ioctls are available to get a file's encryption policy:
-
-- `FS_IOC_GET_ENCRYPTION_POLICY_EX`_
-- `FS_IOC_GET_ENCRYPTION_POLICY`_
-
-The extended (_EX) version of the ioctl is more general and is
-recommended to use when possible. However, on older kernels only the
-original ioctl is available. Applications should try the extended
-version, and if it fails with ENOTTY fall back to the original
-version.
-
FS_IOC_GET_ENCRYPTION_POLICY_EX
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
@@ -780,7 +775,6 @@ FS_IOC_GET_ENCRYPTION_POLICY_EX can fail with the following errors:
- ``ENODATA``: the file is not encrypted
- ``ENOTTY``: this type of filesystem does not implement encryption,
or this kernel is too old to support FS_IOC_GET_ENCRYPTION_POLICY_EX
- (try FS_IOC_GET_ENCRYPTION_POLICY instead)
- ``EOPNOTSUPP``: the kernel was not configured with encryption
support for this filesystem, or the filesystem superblock has not
had encryption enabled on it
@@ -796,12 +790,13 @@ check for STATX_ATTR_ENCRYPTED in stx_attributes.
FS_IOC_GET_ENCRYPTION_POLICY
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
-The FS_IOC_GET_ENCRYPTION_POLICY ioctl can also retrieve the
-encryption policy, if any, for a directory or regular file. However,
-unlike `FS_IOC_GET_ENCRYPTION_POLICY_EX`_,
-FS_IOC_GET_ENCRYPTION_POLICY only supports the original policy
-version. It takes in a pointer directly to struct fscrypt_policy_v1
-rather than struct fscrypt_get_policy_ex_arg.
+The FS_IOC_GET_ENCRYPTION_POLICY ioctl is deprecated. It supports
+only v1 encryption policies, which themselves are deprecated. Use
+`FS_IOC_GET_ENCRYPTION_POLICY_EX`_ instead.
+
+FS_IOC_GET_ENCRYPTION_POLICY retrieves the encryption policy for a
+directory or regular file, but only if it uses a v1 policy. It takes
+in a pointer directly to struct fscrypt_policy_v1.
The error codes for FS_IOC_GET_ENCRYPTION_POLICY are the same as those
for FS_IOC_GET_ENCRYPTION_POLICY_EX, except that
@@ -884,6 +879,10 @@ as follows:
To add this type of key, the calling process must have the
CAP_SYS_ADMIN capability in the initial user namespace.
+ (Note that v1 encryption policies are deprecated. The ability to
+ add a key for v1 encryption policies remains only for compatibility
+ with existing encrypted directories.)
+
Alternatively, if the key is being added for use by v2 encryption
policies, then ``key_spec.type`` must contain
FSCRYPT_KEY_SPEC_TYPE_IDENTIFIER, and ``key_spec.u.identifier`` is
@@ -1315,32 +1314,20 @@ Inline encryption support
Many newer systems (especially mobile SoCs) have *inline encryption
hardware* that can encrypt/decrypt data while it is on its way to/from
-the storage device. Linux supports inline encryption through a set of
-extensions to the block layer called *blk-crypto*. blk-crypto allows
-filesystems to attach encryption contexts to bios (I/O requests) to
-specify how the data will be encrypted or decrypted in-line. For more
-information about blk-crypto, see
-:ref:`Documentation/block/inline-encryption.rst <inline_encryption>`.
+the storage device.
On supported filesystems (currently ext4 and f2fs), fscrypt can use
-blk-crypto instead of the kernel crypto API to encrypt/decrypt file
-contents. To enable this, set CONFIG_FS_ENCRYPTION_INLINE_CRYPT=y in
-the kernel configuration, and specify the "inlinecrypt" mount option
-when mounting the filesystem.
-
-Note that the "inlinecrypt" mount option just specifies to use inline
-encryption when possible; it doesn't force its use. fscrypt will
-still fall back to using the kernel crypto API on files where the
-inline encryption hardware doesn't have the needed crypto capabilities
-(e.g. support for the needed encryption algorithm and data unit size)
-and where blk-crypto-fallback is unusable. (For blk-crypto-fallback
-to be usable, it must be enabled in the kernel configuration with
-CONFIG_BLK_INLINE_ENCRYPTION_FALLBACK=y, and the file must be
-protected by a raw key rather than a hardware-wrapped key.)
-
-Currently fscrypt always uses the filesystem block size (which is
-usually 4096 bytes) as the data unit size. Therefore, it can only use
-inline encryption hardware that supports that data unit size.
+inline encryption hardware instead of the CPU to encrypt/decrypt file
+contents. To enable this, specify the "inlinecrypt" mount option when
+mounting the filesystem.
+
+This causes the filesystem to use inline encryption hardware whenever
+possible, falling back to the CPU only if such hardware is absent or
+doesn't provide the needed crypto capabilities.
+
+For more information about the kernel's support for inline encryption
+hardware, see :ref:`Documentation/block/inline-encryption.rst
+<inline_encryption>`.
Inline encryption doesn't affect the ciphertext or other aspects of
the on-disk format, so users may freely switch back and forth between
@@ -1422,10 +1409,8 @@ For direct I/O on an encrypted file to work, the following conditions
must be met (in addition to the conditions for direct I/O on an
unencrypted file):
-* The file must be using inline encryption. Usually this means that
- the filesystem must be mounted with ``-o inlinecrypt`` and inline
- encryption hardware must be present. However, a software fallback
- is also available. For details, see `Inline encryption support`_.
+* The filesystem must be block-based. (Before Linux v7.3, the
+ filesystem also needed to be mounted with ``-o inlinecrypt``.)
* The I/O request must be fully aligned to the filesystem block size.
This means that the file position the I/O is targeting, the lengths
@@ -1486,25 +1471,43 @@ keys`_ and `DIRECT_KEY policies`_.
Data path changes
-----------------
-When inline encryption is used, filesystems just need to associate
-encryption contexts with bios to specify how the block layer or the
-inline encryption hardware will encrypt/decrypt the file contents.
-
-When inline encryption isn't used, filesystems must encrypt/decrypt
-the file contents themselves, as described below:
-
-For the read path (->read_folio()) of regular files, filesystems can
-read the ciphertext into the page cache and decrypt it in-place. The
-folio lock must be held until decryption has finished, to prevent the
-folio from becoming visible to userspace prematurely.
-
-For the write path (->writepages()) of regular files, filesystems
-cannot encrypt data in-place in the page cache, since the cached
-plaintext must be preserved. Instead, filesystems must encrypt into a
-temporary buffer or "bounce page", then write out the temporary
-buffer. Some filesystems, such as UBIFS, already use temporary
-buffers regardless of encryption. Other filesystems, such as ext4 and
-F2FS, have to allocate bounce pages specially for encryption.
+The block-based filesystems that support fscrypt, such as ext4 and
+f2fs, use blk-crypto (:ref:`inline_encryption`) to implement file
+contents encryption and decryption. With blk-crypto, the filesystem
+assigns an encryption context to each I/O request it issues to the
+contents of an encrypted file. The encryption (for writes) or
+decryption (for reads) is handled by the block layer transparently to
+the filesystem, using either the CPU or inline encryption hardware.
+
+Non-block-based filesystems can't use blk-crypto, so they make the
+calls to the cryptographic algorithms at the filesystem layer instead.
+
+Regardless of the layer in which they occur (blk-crypto-fallback or the
+filesystem), for CPU-based encryption and decryption of file contents:
+
+- For reads, the ciphertext data is read from the storage backend
+ (block device, network, UBI device, etc.) into the destination
+ buffers, then decrypted in-place. The destination buffers are
+ pagecache folios for buffered reads, or application-provided buffers
+ for direct reads. In either case, the filesystem reports success
+ only after decryption has successfully completed.
+
+- For writes, the plaintext data is encrypted from the source buffers
+ (which cannot be modified) into bounce buffers. Then, the
+ ciphertext in the bounce buffers is written to the storage backend.
+
+ The source buffers are usually pagecache folios for buffered writes,
+ or application-provided buffers for direct writes. There are also
+ some cases (all files on UBIFS, and compressed files on f2fs) where
+ the filesystem already uses bounce buffers for writes for other
+ reasons; in these cases the source plaintext data is already in
+ bounce buffers. UBIFS optimizes this case by encrypting the data
+ in-place in its existing bounce buffers.
+
+When inline encryption hardware is used instead of the CPU, reads from
+the storage backend logically return plaintext data, and writes accept
+plaintext data. In that case the flow is simplified: there's no
+scheduling of decryption work, and no bounce buffers are used.
Filename hashing and encoding
-----------------------------
@@ -1552,14 +1555,11 @@ Tests
To test fscrypt, use xfstests, which is Linux's de facto standard
filesystem test suite. First, run all the tests in the "encrypt"
-group on the relevant filesystem(s). One can also run the tests
-with the 'inlinecrypt' mount option to test the implementation for
-inline encryption support. For example, to test ext4 and
+group on the relevant filesystem(s). For example, to test ext4 and
f2fs encryption using `kvm-xfstests
<https://github.com/tytso/xfstests-bld/blob/master/Documentation/kvm-quickstart.md>`_::
kvm-xfstests -c ext4,f2fs -g encrypt
- kvm-xfstests -c ext4,f2fs -g encrypt -m inlinecrypt
UBIFS encryption can also be tested this way, but it should be done in
a separate command, and it takes some time for kvm-xfstests to set up
@@ -1581,7 +1581,6 @@ This tests the encrypted I/O paths more thoroughly. To do this with
kvm-xfstests, use the "encrypt" filesystem configuration::
kvm-xfstests -c ext4/encrypt,f2fs/encrypt -g auto
- kvm-xfstests -c ext4/encrypt,f2fs/encrypt -g auto -m inlinecrypt
Because this runs many more tests than "-g encrypt" does, it takes
much longer to run; so also consider using `gce-xfstests
@@ -1589,4 +1588,9 @@ much longer to run; so also consider using `gce-xfstests
instead of kvm-xfstests::
gce-xfstests -c ext4/encrypt,f2fs/encrypt -g auto
- gce-xfstests -c ext4/encrypt,f2fs/encrypt -g auto -m inlinecrypt
+
+To test inline encryption hardware on a platform that supports such
+hardware, run xfstests directly with the ``inlinecrypt`` mount option
+enabled. For example::
+
+ EXT_MOUNT_OPTIONS="-o inlinecrypt" ./check -g encrypt
diff --git a/Documentation/filesystems/locking.rst b/Documentation/filesystems/locking.rst
index 1a50d41a39a1..5e808c1ebc68 100644
--- a/Documentation/filesystems/locking.rst
+++ b/Documentation/filesystems/locking.rst
@@ -61,7 +61,7 @@ inode_operations
prototypes::
- int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t, bool);
+ int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t);
struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int);
int (*link) (struct dentry *,struct inode *,struct dentry *);
int (*unlink) (struct inode *,struct dentry *);
diff --git a/Documentation/filesystems/overlayfs.rst b/Documentation/filesystems/overlayfs.rst
index eb846518e6ac..1a29a5afabd7 100644
--- a/Documentation/filesystems/overlayfs.rst
+++ b/Documentation/filesystems/overlayfs.rst
@@ -347,6 +347,22 @@ The resulting access permissions should be the same. The difference is in
the time of copy (on-demand vs. up-front).
+Idmapped mounts
+---------------
+
+The overlay mount itself can be turned into an idmapped mount by applying an
+idmapping to it with mount_setattr(2) and MOUNT_ATTR_IDMAP, just like for
+other filesystems that support idmapped mounts.
+
+The mount idmapping only changes how ownership and permissions of the overlay
+inodes are presented to and interpreted for the caller. It does not change
+how overlayfs accesses the underlying layers: those are still accessed with
+the stashed mounter's credentials through their own mounts, which may
+themselves be idmapped. The overlay mount idmapping and any layer idmapping
+compose, an underlying id is first mapped according to the relevant layer
+idmapping and then according to the overlay mount idmapping.
+
+
Multiple lower layers
---------------------
diff --git a/Documentation/filesystems/porting.rst b/Documentation/filesystems/porting.rst
index d13f0a23c882..60880eb0c49d 100644
--- a/Documentation/filesystems/porting.rst
+++ b/Documentation/filesystems/porting.rst
@@ -1173,7 +1173,7 @@ these conditions don't require explicit checks:
- if LOOKUP_CREATE is NOT given, then the dentry won't be negative,
ERR_PTR(-ENOENT) is returned instead
- if LOOKUP_EXCL IS given, then the dentry won't be positive,
- ERR_PTR(-EEXIST) is rreturned instread
+ ERR_PTR(-EEXIST) is returned instead
LOOKUP_EXCL now means "target must not exist". It can be combined with
LOOK_CREATE or LOOKUP_RENAME_TARGET.
@@ -1401,3 +1401,11 @@ as with d_dispose_if_unused() these are not trivial; with this variant
of API it's more explicit, since grabbing ->d_lock is caller-side, but
d_dispose_if_unused() had all the same issues. It's a low-level primitive;
use only if you have no alternative.
+
+---
+
+**mandatory**
+
+The .create inode_operation no longer receives the 'excl' arg. It must
+always assume the file does not already exist. If the filesystem needs
+to be involved in non-exclusive create, it should provide atomic_open.
diff --git a/Documentation/filesystems/vfs.rst b/Documentation/filesystems/vfs.rst
index e7677423a20f..c2886fa50e51 100644
--- a/Documentation/filesystems/vfs.rst
+++ b/Documentation/filesystems/vfs.rst
@@ -415,7 +415,7 @@ As of kernel 2.6.22, the following members are defined:
.. code-block:: c
struct inode_operations {
- int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, umode_t, bool);
+ int (*create) (struct mnt_idmap *, struct inode *,struct dentry *, umode_t);
struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int);
int (*link) (struct dentry *,struct inode *,struct dentry *);
int (*unlink) (struct inode *,struct dentry *);
diff --git a/Documentation/sunrpc/xdr/nfs4_1.x b/Documentation/sunrpc/xdr/nfs4_1.x
index 5b45547b2ebc..e66f396ae659 100644
--- a/Documentation/sunrpc/xdr/nfs4_1.x
+++ b/Documentation/sunrpc/xdr/nfs4_1.x
@@ -45,19 +45,162 @@ pragma header nfs4;
/*
* Basic typedefs for RFC 1832 data type definitions
*/
-typedef hyper int64_t;
-typedef unsigned int uint32_t;
+typedef int int32_t;
+typedef unsigned int uint32_t;
+typedef hyper int64_t;
+typedef unsigned hyper uint64_t;
+
+const NFS4_VERIFIER_SIZE = 8;
+const NFS4_FHSIZE = 128;
+
+enum nfsstat4 {
+ NFS4_OK = 0, /* everything is okay */
+ NFS4ERR_PERM = 1, /* caller not privileged */
+ NFS4ERR_NOENT = 2, /* no such file/directory */
+ NFS4ERR_IO = 5, /* hard I/O error */
+ NFS4ERR_NXIO = 6, /* no such device */
+ NFS4ERR_ACCESS = 13, /* access denied */
+ NFS4ERR_EXIST = 17, /* file already exists */
+ NFS4ERR_XDEV = 18, /* different filesystems */
+
+ /*
+ * Please do not allocate value 19; it was used in NFSv3
+ * and we do not want a value in NFSv3 to have a different
+ * meaning in NFSv4.x.
+ */
+
+ NFS4ERR_NOTDIR = 20, /* should be a directory */
+ NFS4ERR_ISDIR = 21, /* should not be directory */
+ NFS4ERR_INVAL = 22, /* invalid argument */
+ NFS4ERR_FBIG = 27, /* file exceeds server max */
+ NFS4ERR_NOSPC = 28, /* no space on filesystem */
+ NFS4ERR_ROFS = 30, /* read-only filesystem */
+ NFS4ERR_MLINK = 31, /* too many hard links */
+ NFS4ERR_NAMETOOLONG = 63, /* name exceeds server max */
+ NFS4ERR_NOTEMPTY = 66, /* directory not empty */
+ NFS4ERR_DQUOT = 69, /* hard quota limit reached*/
+ NFS4ERR_STALE = 70, /* file no longer exists */
+ NFS4ERR_BADHANDLE = 10001,/* Illegal filehandle */
+ NFS4ERR_BAD_COOKIE = 10003,/* READDIR cookie is stale */
+ NFS4ERR_NOTSUPP = 10004,/* operation not supported */
+ NFS4ERR_TOOSMALL = 10005,/* response limit exceeded */
+ NFS4ERR_SERVERFAULT = 10006,/* undefined server error */
+ NFS4ERR_BADTYPE = 10007,/* type invalid for CREATE */
+ NFS4ERR_DELAY = 10008,/* file "busy" - retry */
+ NFS4ERR_SAME = 10009,/* nverify says attrs same */
+ NFS4ERR_DENIED = 10010,/* lock unavailable */
+ NFS4ERR_EXPIRED = 10011,/* lock lease expired */
+ NFS4ERR_LOCKED = 10012,/* I/O failed due to lock */
+ NFS4ERR_GRACE = 10013,/* in grace period */
+ NFS4ERR_FHEXPIRED = 10014,/* filehandle expired */
+ NFS4ERR_SHARE_DENIED = 10015,/* share reserve denied */
+ NFS4ERR_WRONGSEC = 10016,/* wrong security flavor */
+ NFS4ERR_CLID_INUSE = 10017,/* clientid in use */
+
+ /* NFS4ERR_RESOURCE is not a valid error in NFSv4.1 */
+ NFS4ERR_RESOURCE = 10018,/* resource exhaustion */
+
+ NFS4ERR_MOVED = 10019,/* filesystem relocated */
+ NFS4ERR_NOFILEHANDLE = 10020,/* current FH is not set */
+ NFS4ERR_MINOR_VERS_MISMATCH= 10021,/* minor vers not supp */
+ NFS4ERR_STALE_CLIENTID = 10022,/* server has rebooted */
+ NFS4ERR_STALE_STATEID = 10023,/* server has rebooted */
+ NFS4ERR_OLD_STATEID = 10024,/* state is out of sync */
+ NFS4ERR_BAD_STATEID = 10025,/* incorrect stateid */
+ NFS4ERR_BAD_SEQID = 10026,/* request is out of seq. */
+ NFS4ERR_NOT_SAME = 10027,/* verify - attrs not same */
+ NFS4ERR_LOCK_RANGE = 10028,/* overlapping lock range */
+ NFS4ERR_SYMLINK = 10029,/* should be file/directory*/
+ NFS4ERR_RESTOREFH = 10030,/* no saved filehandle */
+ NFS4ERR_LEASE_MOVED = 10031,/* some filesystem moved */
+ NFS4ERR_ATTRNOTSUPP = 10032,/* recommended attr not sup*/
+ NFS4ERR_NO_GRACE = 10033,/* reclaim outside of grace*/
+ NFS4ERR_RECLAIM_BAD = 10034,/* reclaim error at server */
+ NFS4ERR_RECLAIM_CONFLICT= 10035,/* conflict on reclaim */
+ NFS4ERR_BADXDR = 10036,/* XDR decode failed */
+ NFS4ERR_LOCKS_HELD = 10037,/* file locks held at CLOSE*/
+ NFS4ERR_OPENMODE = 10038,/* conflict in OPEN and I/O*/
+ NFS4ERR_BADOWNER = 10039,/* owner translation bad */
+ NFS4ERR_BADCHAR = 10040,/* utf-8 char not supported*/
+ NFS4ERR_BADNAME = 10041,/* name not supported */
+ NFS4ERR_BAD_RANGE = 10042,/* lock range not supported*/
+ NFS4ERR_LOCK_NOTSUPP = 10043,/* no atomic up/downgrade */
+ NFS4ERR_OP_ILLEGAL = 10044,/* undefined operation */
+ NFS4ERR_DEADLOCK = 10045,/* file locking deadlock */
+ NFS4ERR_FILE_OPEN = 10046,/* open file blocks op. */
+ NFS4ERR_ADMIN_REVOKED = 10047,/* lockowner state revoked */
+ NFS4ERR_CB_PATH_DOWN = 10048,/* callback path down */
+
+ /* NFSv4.1 errors start here. */
+
+ NFS4ERR_BADIOMODE = 10049,
+ NFS4ERR_BADLAYOUT = 10050,
+ NFS4ERR_BAD_SESSION_DIGEST = 10051,
+ NFS4ERR_BADSESSION = 10052,
+ NFS4ERR_BADSLOT = 10053,
+ NFS4ERR_COMPLETE_ALREADY = 10054,
+ NFS4ERR_CONN_NOT_BOUND_TO_SESSION = 10055,
+ NFS4ERR_DELEG_ALREADY_WANTED = 10056,
+ NFS4ERR_BACK_CHAN_BUSY = 10057,/*backchan reqs outstanding*/
+ NFS4ERR_LAYOUTTRYLATER = 10058,
+ NFS4ERR_LAYOUTUNAVAILABLE = 10059,
+ NFS4ERR_NOMATCHING_LAYOUT = 10060,
+ NFS4ERR_RECALLCONFLICT = 10061,
+ NFS4ERR_UNKNOWN_LAYOUTTYPE = 10062,
+ NFS4ERR_SEQ_MISORDERED = 10063,/* unexpected seq.ID in req*/
+ NFS4ERR_SEQUENCE_POS = 10064,/* [CB_]SEQ. op not 1st op */
+ NFS4ERR_REQ_TOO_BIG = 10065,/* request too big */
+ NFS4ERR_REP_TOO_BIG = 10066,/* reply too big */
+ NFS4ERR_REP_TOO_BIG_TO_CACHE =10067,/* rep. not all cached*/
+ NFS4ERR_RETRY_UNCACHED_REP =10068,/* retry & rep. uncached*/
+ NFS4ERR_UNSAFE_COMPOUND =10069,/* retry/recovery too hard */
+ NFS4ERR_TOO_MANY_OPS = 10070,/*too many ops in [CB_]COMP*/
+ NFS4ERR_OP_NOT_IN_SESSION =10071,/* op needs [CB_]SEQ. op */
+ NFS4ERR_HASH_ALG_UNSUPP = 10072, /* hash alg. not supp. */
+ /* Error 10073 is unused. */
+ NFS4ERR_CLIENTID_BUSY = 10074,/* clientid has state */
+ NFS4ERR_PNFS_IO_HOLE = 10075,/* IO to _SPARSE file hole */
+ NFS4ERR_SEQ_FALSE_RETRY= 10076,/* Retry != original req. */
+ NFS4ERR_BAD_HIGH_SLOT = 10077,/* req has bad highest_slot*/
+ NFS4ERR_DEADSESSION = 10078,/*new req sent to dead sess*/
+ NFS4ERR_ENCR_ALG_UNSUPP= 10079,/* encr alg. not supp. */
+ NFS4ERR_PNFS_NO_LAYOUT = 10080,/* I/O without a layout */
+ NFS4ERR_NOT_ONLY_OP = 10081,/* addl ops not allowed */
+ NFS4ERR_WRONG_CRED = 10082,/* op done by wrong cred */
+ NFS4ERR_WRONG_TYPE = 10083,/* op on wrong type object */
+ NFS4ERR_DIRDELEG_UNAVAIL=10084,/* delegation not avail. */
+ NFS4ERR_REJECT_DELEG = 10085,/* cb rejected delegation */
+ NFS4ERR_RETURNCONFLICT = 10086,/* layout get before return*/
+ NFS4ERR_DELEG_REVOKED = 10087, /* deleg./layout revoked */
+ NFS4ERR_PARTNER_NOTSUPP = 10088,
+ NFS4ERR_PARTNER_NO_AUTH = 10089,
+ NFS4ERR_UNION_NOTSUPP = 10090,
+ NFS4ERR_OFFLOAD_DENIED = 10091,
+ NFS4ERR_WRONG_LFS = 10092,
+ NFS4ERR_BADLABEL = 10093,
+ NFS4ERR_OFFLOAD_NO_REQS = 10094,
+ NFS4ERR_NOXATTR = 10095,
+ NFS4ERR_XATTR2BIG = 10096
+};
/*
* Basic data types
*/
+typedef opaque attrlist4<>;
typedef uint32_t bitmap4<>;
+typedef opaque verifier4[NFS4_VERIFIER_SIZE];
+typedef uint64_t nfs_cookie4;
+typedef opaque nfs_fh4<NFS4_FHSIZE>;
typedef opaque utf8string<>;
typedef utf8string utf8str_cis;
typedef utf8string utf8str_cs;
typedef utf8string utf8str_mixed;
+typedef utf8str_cs component4;
+typedef utf8str_cs linktext4;
+typedef component4 pathname4<>;
+
/*
* Timeval
*/
@@ -66,6 +209,21 @@ struct nfstime4 {
uint32_t nseconds;
};
+/*
+ * File attribute container
+ */
+struct fattr4 {
+ bitmap4 attrmask;
+ attrlist4 attr_vals;
+};
+
+/*
+ * Stateid
+ */
+struct stateid4 {
+ uint32_t seqid;
+ opaque other[12];
+};
/*
* The following content was extracted from draft-ietf-nfsv4-delstid
@@ -245,3 +403,99 @@ const FATTR4_ACL_TRUEFORM = 89;
const FATTR4_ACL_TRUEFORM_SCOPE = 90;
const FATTR4_POSIX_DEFAULT_ACL = 91;
const FATTR4_POSIX_ACCESS_ACL = 92;
+
+/*
+ * Directory notification types.
+ */
+enum notify_type4 {
+ NOTIFY4_CHANGE_CHILD_ATTRS = 0,
+ NOTIFY4_CHANGE_DIR_ATTRS = 1,
+ NOTIFY4_REMOVE_ENTRY = 2,
+ NOTIFY4_ADD_ENTRY = 3,
+ NOTIFY4_RENAME_ENTRY = 4,
+ NOTIFY4_CHANGE_COOKIE_VERIFIER = 5,
+ /* Proposed in RFC8881bis */
+ NOTIFY4_GFLAG_EXTEND = 6,
+ NOTIFY4_AUFLAG_VALID = 7,
+ NOTIFY4_AUFLAG_USER = 8,
+ NOTIFY4_AUFLAG_GROUP = 9,
+ NOTIFY4_AUFLAG_OTHER = 10,
+ NOTIFY4_CHANGE_AUTH = 11,
+ NOTIFY4_CFLAG_ORDER = 12,
+ NOTIFY4_AUFLAG_GANOW = 13,
+ NOTIFY4_AUFLAG_GALATER = 14,
+ NOTIFY4_CHANGE_GA = 15,
+ NOTIFY4_CHANGE_AMASK = 16
+};
+
+/* Changed entry information. */
+struct notify_entry4 {
+ component4 ne_file;
+ fattr4 ne_attrs;
+};
+
+/* Previous entry information */
+struct prev_entry4 {
+ notify_entry4 pe_prev_entry;
+ /* what READDIR returned for this entry */
+ nfs_cookie4 pe_prev_entry_cookie;
+};
+
+struct notify_remove4 {
+ notify_entry4 nrm_old_entry;
+ nfs_cookie4 nrm_old_entry_cookie;
+};
+pragma public notify_remove4;
+
+struct notify_add4 {
+ /*
+ * Information on object
+ * possibly renamed over.
+ */
+ notify_remove4 nad_old_entry<1>;
+ notify_entry4 nad_new_entry;
+ /* what READDIR would have returned for this entry */
+ nfs_cookie4 nad_new_entry_cookie<1>;
+ prev_entry4 nad_prev_entry<1>;
+ bool nad_last_entry;
+};
+pragma public notify_add4;
+
+struct notify_attr4 {
+ notify_entry4 na_changed_entry;
+};
+pragma public notify_attr4;
+
+struct notify_rename4 {
+ notify_remove4 nrn_old_entry;
+ notify_add4 nrn_new_entry;
+};
+pragma public notify_rename4;
+
+struct notify_verifier4 {
+ verifier4 nv_old_cookieverf;
+ verifier4 nv_new_cookieverf;
+};
+
+/*
+ * Objects of type notify_<>4 and
+ * notify_device_<>4 are encoded in this.
+ */
+typedef opaque notifylist4<>;
+
+struct notify4 {
+ /* composed from notify_type4 or notify_deviceid_type4 */
+ bitmap4 notify_mask;
+ notifylist4 notify_vals;
+};
+
+struct CB_NOTIFY4args {
+ stateid4 cna_stateid;
+ nfs_fh4 cna_fh;
+ notify4 cna_changes<>;
+};
+pragma public CB_NOTIFY4args;
+
+struct CB_NOTIFY4res {
+ nfsstat4 cnr_status;
+};
diff --git a/MAINTAINERS b/MAINTAINERS
index ace2a06f3395..3e0d57bdf66e 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -9503,11 +9503,6 @@ L: linux-fbdev@vger.kernel.org
S: Maintained
F: drivers/video/fbdev/efifb.c
-EFS FILESYSTEM
-S: Orphan
-W: http://aeschi.ch.eu.org/efs/
-F: fs/efs/
-
EHEA (IBM pSeries eHEA 10Gb ethernet adapter) DRIVER
L: netdev@vger.kernel.org
S: Orphan
@@ -9626,7 +9621,7 @@ M: Chao Yu <chao@kernel.org>
R: Yue Hu <zbestahu@gmail.com>
R: Jeffle Xu <jefflexu@linux.alibaba.com>
R: Sandeep Dhavale <dhavale@google.com>
-R: Hongbo Li <lihongbo22@huawei.com>
+R: Hongbo Li <hongbohbli@tencent.com>
R: Chunhai Guo <guochunhai@vivo.com>
L: linux-erofs@lists.ozlabs.org
S: Maintained
@@ -9768,7 +9763,7 @@ EXFAT FILE SYSTEM
M: Namjae Jeon <linkinjeon@kernel.org>
M: Sungjong Seo <sj1557.seo@samsung.com>
R: Yuezhang Mo <yuezhang.mo@sony.com>
-L: linux-fsdevel@vger.kernel.org
+L: exfat@lists.linux.dev
S: Maintained
T: git git://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git
F: fs/exfat/
@@ -19296,7 +19291,7 @@ F: drivers/ntb/hw/intel/
NTFS FILESYSTEM
M: Namjae Jeon <linkinjeon@kernel.org>
M: Hyunchul Lee <hyc.lee@gmail.com>
-L: linux-fsdevel@vger.kernel.org
+L: ntfs@lists.linux.dev
S: Maintained
T: git git://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git
F: Documentation/filesystems/ntfs.rst
diff --git a/arch/loongarch/configs/loongson32_defconfig b/arch/loongarch/configs/loongson32_defconfig
index 7c8f01513ed2..6bf2867dbdc6 100644
--- a/arch/loongarch/configs/loongson32_defconfig
+++ b/arch/loongarch/configs/loongson32_defconfig
@@ -969,7 +969,6 @@ CONFIG_F2FS_FS_SECURITY=y
CONFIG_F2FS_CHECK_FS=y
CONFIG_F2FS_FS_COMPRESSION=y
CONFIG_FS_ENCRYPTION=y
-CONFIG_FS_ENCRYPTION_INLINE_CRYPT=y
CONFIG_FS_VERITY=y
CONFIG_FANOTIFY=y
CONFIG_FANOTIFY_ACCESS_PERMISSIONS=y
diff --git a/arch/loongarch/configs/loongson64_defconfig b/arch/loongarch/configs/loongson64_defconfig
index 8e3906d3bd70..def104c9d405 100644
--- a/arch/loongarch/configs/loongson64_defconfig
+++ b/arch/loongarch/configs/loongson64_defconfig
@@ -1000,7 +1000,6 @@ CONFIG_F2FS_FS_SECURITY=y
CONFIG_F2FS_CHECK_FS=y
CONFIG_F2FS_FS_COMPRESSION=y
CONFIG_FS_ENCRYPTION=y
-CONFIG_FS_ENCRYPTION_INLINE_CRYPT=y
CONFIG_FS_VERITY=y
CONFIG_FANOTIFY=y
CONFIG_FANOTIFY_ACCESS_PERMISSIONS=y
diff --git a/block/bdev.c b/block/bdev.c
index 85ce57bd2ae4..797d7f0ef609 100644
--- a/block/bdev.c
+++ b/block/bdev.c
@@ -304,7 +304,12 @@ int bdev_freeze(struct block_device *bdev)
mutex_lock(&bdev->bd_fsfreeze_mutex);
- if (atomic_inc_return(&bdev->bd_fsfreeze_count) > 1) {
+ /* A device being removed from its filesystem refuses freezes. */
+ if (!atomic_inc_unless_negative(&bdev->bd_fsfreeze_count)) {
+ mutex_unlock(&bdev->bd_fsfreeze_mutex);
+ return -EBUSY;
+ }
+ if (atomic_read(&bdev->bd_fsfreeze_count) > 1) {
mutex_unlock(&bdev->bd_fsfreeze_mutex);
return 0;
}
@@ -340,18 +345,18 @@ int bdev_thaw(struct block_device *bdev)
mutex_lock(&bdev->bd_fsfreeze_mutex);
- /*
- * If this returns < 0 it means that @bd_fsfreeze_count was
- * already 0 and no decrement was performed.
- */
- nr_freeze = atomic_dec_if_positive(&bdev->bd_fsfreeze_count);
- if (nr_freeze < 0)
+ /* <= 0: not frozen (0) or a freeze deny is held (< 0); leave it. */
+ nr_freeze = atomic_read(&bdev->bd_fsfreeze_count);
+ if (nr_freeze <= 0)
goto out;
error = 0;
- if (nr_freeze > 0)
+ if (nr_freeze > 1) {
+ atomic_dec(&bdev->bd_fsfreeze_count);
goto out;
+ }
+ /* Keep the count positive across the thaw so a deny is refused. */
mutex_lock(&bdev->bd_holder_lock);
if (bdev->bd_holder_ops && bdev->bd_holder_ops->thaw) {
error = bdev->bd_holder_ops->thaw(bdev);
@@ -360,14 +365,52 @@ int bdev_thaw(struct block_device *bdev)
mutex_unlock(&bdev->bd_holder_lock);
}
- if (error)
- atomic_inc(&bdev->bd_fsfreeze_count);
+ if (!error)
+ atomic_dec(&bdev->bd_fsfreeze_count);
out:
mutex_unlock(&bdev->bd_fsfreeze_mutex);
return error;
}
EXPORT_SYMBOL(bdev_thaw);
+/**
+ * bdev_deny_freeze - make a block device unfreezable
+ * @bdev: block device
+ *
+ * Reserve @bdev against bdev_freeze() the way deny_write_access() reserves a
+ * file against writers. bd_fsfreeze_count is sign-encoded: > 0 counts active
+ * freezes, < 0 counts deniers, so a deny succeeds only while no freeze is in
+ * progress. While held, bdev_freeze() returns -EBUSY. Pair with
+ * bdev_allow_freeze().
+ *
+ * A filesystem removing, adding or replacing a member device denies freezes on
+ * it for the duration, so a claim a freeze walk might act on is never torn down
+ * behind the freezer's back. The deny is device-scoped, not (device,
+ * superblock)-scoped: a device shared by several superblocks is refused for all
+ * of them. No in-tree filesystem removes a shared claim from a live superblock.
+ *
+ * Return: 0, or -EBUSY if the device is currently frozen.
+ */
+int bdev_deny_freeze(struct block_device *bdev)
+{
+ return atomic_dec_unless_positive(&bdev->bd_fsfreeze_count) ? 0 : -EBUSY;
+}
+EXPORT_SYMBOL_GPL(bdev_deny_freeze);
+
+/**
+ * bdev_allow_freeze - allow freezing a block device again
+ * @bdev: block device
+ *
+ * Undo one bdev_deny_freeze().
+ */
+void bdev_allow_freeze(struct block_device *bdev)
+{
+ /* A deny must be held, i.e. the count must be negative. */
+ WARN_ON_ONCE(atomic_read(&bdev->bd_fsfreeze_count) >= 0);
+ atomic_inc(&bdev->bd_fsfreeze_count);
+}
+EXPORT_SYMBOL_GPL(bdev_allow_freeze);
+
/*
* pseudo-fs
*/
@@ -1153,6 +1196,39 @@ put_no_open:
}
/**
+ * bdev_yield_claim - give up the holder claim on an open block device
+ * @bdev_file: open block device
+ *
+ * Yield the holder and any write access for @bdev_file without closing it, so
+ * the caller can still act on the device - e.g. bdev_allow_freeze() it - before
+ * the final bdev_fput(). bdev_fput() yields too, so calling it afterwards is
+ * safe.
+ */
+void bdev_yield_claim(struct file *bdev_file)
+{
+ struct block_device *bdev;
+ struct gendisk *disk;
+
+ if (!bdev_file->private_data)
+ return;
+
+ bdev = file_bdev(bdev_file);
+ disk = bdev->bd_disk;
+
+ mutex_lock(&disk->open_mutex);
+ bdev_yield_write_access(bdev_file);
+ bd_yield_claim(bdev_file);
+ /*
+ * Tell release we already gave up our hold on the
+ * device and if write restrictions are available that
+ * we already gave up write access to the device.
+ */
+ bdev_file->private_data = BDEV_I(bdev_file->f_mapping->host);
+ mutex_unlock(&disk->open_mutex);
+}
+EXPORT_SYMBOL_GPL(bdev_yield_claim);
+
+/**
* bdev_fput - yield claim to the block device and put the file
* @bdev_file: open block device
*
@@ -1165,22 +1241,7 @@ void bdev_fput(struct file *bdev_file)
if (WARN_ON_ONCE(bdev_file->f_op != &def_blk_fops))
return;
- if (bdev_file->private_data) {
- struct block_device *bdev = file_bdev(bdev_file);
- struct gendisk *disk = bdev->bd_disk;
-
- mutex_lock(&disk->open_mutex);
- bdev_yield_write_access(bdev_file);
- bd_yield_claim(bdev_file);
- /*
- * Tell release we already gave up our hold on the
- * device and if write restrictions are available that
- * we already gave up write access to the device.
- */
- bdev_file->private_data = BDEV_I(bdev_file->f_mapping->host);
- mutex_unlock(&disk->open_mutex);
- }
-
+ bdev_yield_claim(bdev_file);
fput(bdev_file);
}
EXPORT_SYMBOL(bdev_fput);
@@ -1217,6 +1278,18 @@ int lookup_bdev(const char *pathname, dev_t *dev)
if (!may_open_dev(&path))
goto out_path_put;
+ /*
+ * Reject a block device inode with i_rdev == 0. A dev_t of 0 is
+ * never valid for a block device: no real block device driver
+ * registers major 0. Fake block device inodes (e.g. fuse with
+ * rootmode=S_IFBLK) can expose i_rdev == 0, and letting that
+ * propagate would confuse superblock lookup and trigger warnings
+ * in the device-to-superblock table (super_dev_register).
+ */
+ error = -ENODEV;
+ if (!inode->i_rdev)
+ goto out_path_put;
+
*dev = inode->i_rdev;
error = 0;
out_path_put:
diff --git a/block/blk-crypto-fallback.c b/block/blk-crypto-fallback.c
index 2a5c52ab74b4..2a8f40a65158 100644
--- a/block/blk-crypto-fallback.c
+++ b/block/blk-crypto-fallback.c
@@ -496,8 +496,7 @@ bool blk_crypto_fallback_bio_prep(struct bio *bio)
return false;
}
- if (!__blk_crypto_cfg_supported(blk_crypto_fallback_profile,
- &bc->bc_key->crypto_cfg)) {
+ if (bc->bc_key->crypto_cfg.key_type != BLK_CRYPTO_KEY_TYPE_RAW) {
bio_endio_status(bio, BLK_STS_NOTSUPP);
return false;
}
diff --git a/block/blk-crypto-internal.h b/block/blk-crypto-internal.h
index 742694213529..2c7a0446572a 100644
--- a/block/blk-crypto-internal.h
+++ b/block/blk-crypto-internal.h
@@ -80,9 +80,6 @@ void blk_crypto_put_keyslot(struct blk_crypto_keyslot *slot);
int __blk_crypto_evict_key(struct blk_crypto_profile *profile,
const struct blk_crypto_key *key);
-bool __blk_crypto_cfg_supported(struct blk_crypto_profile *profile,
- const struct blk_crypto_config *cfg);
-
int blk_crypto_ioctl(struct block_device *bdev, unsigned int cmd,
void __user *argp);
diff --git a/block/blk-crypto-profile.c b/block/blk-crypto-profile.c
index cf447ba4a66e..53126c091b0b 100644
--- a/block/blk-crypto-profile.c
+++ b/block/blk-crypto-profile.c
@@ -335,28 +335,6 @@ void blk_crypto_put_keyslot(struct blk_crypto_keyslot *slot)
}
}
-/**
- * __blk_crypto_cfg_supported() - Check whether the given crypto profile
- * supports the given crypto configuration.
- * @profile: the crypto profile to check
- * @cfg: the crypto configuration to check for
- *
- * Return: %true if @profile supports the given @cfg.
- */
-bool __blk_crypto_cfg_supported(struct blk_crypto_profile *profile,
- const struct blk_crypto_config *cfg)
-{
- if (!profile)
- return false;
- if (!(profile->modes_supported[cfg->crypto_mode] & cfg->data_unit_size))
- return false;
- if (profile->max_dun_bytes_supported < cfg->dun_bytes)
- return false;
- if (!(profile->key_types_supported & cfg->key_type))
- return false;
- return true;
-}
-
/*
* This is an internal function that evicts a key from an inline encryption
* device that can be either a real device or the blk-crypto-fallback "device".
diff --git a/block/blk-crypto.c b/block/blk-crypto.c
index 15e25e41b166..bc3a9f59574b 100644
--- a/block/blk-crypto.c
+++ b/block/blk-crypto.c
@@ -300,6 +300,7 @@ int __blk_crypto_rq_bio_prep(struct request *rq, struct bio *bio,
* @dun_bytes: number of bytes that will be used to specify the DUN when this
* key is used
* @data_unit_size: the data unit size to use for en/decryption
+ * @flags: BLK_CRYPTO_CFG_* flags
*
* Return: 0 on success, -errno on failure. The caller is responsible for
* zeroizing both blk_key and key_bytes when done with them.
@@ -309,7 +310,7 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
enum blk_crypto_key_type key_type,
enum blk_crypto_mode_num crypto_mode,
unsigned int dun_bytes,
- unsigned int data_unit_size)
+ unsigned int data_unit_size, int flags)
{
const struct blk_crypto_mode *mode;
@@ -318,6 +319,9 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
if (crypto_mode >= ARRAY_SIZE(blk_crypto_modes))
return -EINVAL;
+ if (flags & ~BLK_CRYPTO_CFG_ALLOW_HW)
+ return -EINVAL;
+
mode = &blk_crypto_modes[crypto_mode];
switch (key_type) {
case BLK_CRYPTO_KEY_TYPE_RAW:
@@ -328,6 +332,8 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
if (key_size < mode->security_strength ||
key_size > BLK_CRYPTO_MAX_HW_WRAPPED_KEY_SIZE)
return -EINVAL;
+ if (!(flags & BLK_CRYPTO_CFG_ALLOW_HW))
+ return -EINVAL;
break;
default:
return -EINVAL;
@@ -343,6 +349,7 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
blk_key->crypto_cfg.dun_bytes = dun_bytes;
blk_key->crypto_cfg.data_unit_size = data_unit_size;
blk_key->crypto_cfg.key_type = key_type;
+ blk_key->crypto_cfg.flags = flags;
blk_key->data_unit_size_bits = ilog2(data_unit_size);
blk_key->size = key_size;
memcpy(blk_key->bytes, key_bytes, key_size);
@@ -351,25 +358,32 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
}
EXPORT_SYMBOL_GPL(blk_crypto_init_key);
+/**
+ * blk_crypto_config_supported_natively() - Check whether a block device
+ * supports hardware inline encryption
+ * with the given configuration.
+ * @bdev: the block device
+ * @cfg: the crypto configuration to check for
+ *
+ * Return: %true if @bdev supports hardware inline encryption with @cfg.
+ */
bool blk_crypto_config_supported_natively(struct block_device *bdev,
const struct blk_crypto_config *cfg)
{
- return __blk_crypto_cfg_supported(bdev_get_queue(bdev)->crypto_profile,
- cfg);
-}
+ struct blk_crypto_profile *profile =
+ bdev_get_queue(bdev)->crypto_profile;
-/*
- * Check if bios with @cfg can be en/decrypted by blk-crypto (i.e. either the
- * block_device it's submitted to supports inline crypto, or the
- * blk-crypto-fallback is enabled and supports the cfg).
- */
-bool blk_crypto_config_supported(struct block_device *bdev,
- const struct blk_crypto_config *cfg)
-{
- if (IS_ENABLED(CONFIG_BLK_INLINE_ENCRYPTION_FALLBACK) &&
- cfg->key_type == BLK_CRYPTO_KEY_TYPE_RAW)
- return true;
- return blk_crypto_config_supported_natively(bdev, cfg);
+ if (!profile)
+ return false;
+ if (!(cfg->flags & BLK_CRYPTO_CFG_ALLOW_HW))
+ return false;
+ if (!(profile->modes_supported[cfg->crypto_mode] & cfg->data_unit_size))
+ return false;
+ if (profile->max_dun_bytes_supported < cfg->dun_bytes)
+ return false;
+ if (!(profile->key_types_supported & cfg->key_type))
+ return false;
+ return true;
}
/**
diff --git a/drivers/base/devtmpfs.c b/drivers/base/devtmpfs.c
index b1c4ceb65026..aef0fcc6aba1 100644
--- a/drivers/base/devtmpfs.c
+++ b/drivers/base/devtmpfs.c
@@ -413,7 +413,7 @@ static noinline int __init devtmpfs_setup(void *p)
{
int err;
- err = ksys_unshare(CLONE_NEWNS);
+ err = ksys_unshare(UNSHARE_EMPTY_MNTNS);
if (err)
goto out;
err = init_mount("devtmpfs", "/", "devtmpfs", DEVTMPFS_MFLAGS, NULL);
diff --git a/drivers/block/rnbd/rnbd-srv.c b/drivers/block/rnbd/rnbd-srv.c
index 10e8c438bb43..79c9a5fb418f 100644
--- a/drivers/block/rnbd/rnbd-srv.c
+++ b/drivers/block/rnbd/rnbd-srv.c
@@ -11,6 +11,7 @@
#include <linux/module.h>
#include <linux/blkdev.h>
+#include <linux/fs_struct.h>
#include "rnbd-srv.h"
#include "rnbd-srv-trace.h"
@@ -734,7 +735,8 @@ static int process_msg_open(struct rnbd_srv_session *srv_sess,
goto reject;
}
- bdev_file = bdev_file_open_by_path(full_path, open_flags, NULL, NULL);
+ scoped_with_init_fs()
+ bdev_file = bdev_file_open_by_path(full_path, open_flags, NULL, NULL);
if (IS_ERR(bdev_file)) {
ret = PTR_ERR(bdev_file);
pr_err("Opening device '%s' on session %s failed, failed to open the block device, err: %pe\n",
diff --git a/drivers/char/misc_minor_kunit.c b/drivers/char/misc_minor_kunit.c
index e930c78e1ef9..e85210cbb640 100644
--- a/drivers/char/misc_minor_kunit.c
+++ b/drivers/char/misc_minor_kunit.c
@@ -5,6 +5,7 @@
#include <linux/miscdevice.h>
#include <linux/fs.h>
#include <linux/file.h>
+#include <linux/fs_struct.h>
#include <linux/init_syscalls.h>
/* static minor (LCD_MINOR) */
@@ -160,18 +161,22 @@ static void __init miscdev_test_can_open(struct kunit *test, struct miscdevice *
char *devname;
devname = kasprintf(GFP_KERNEL, "/dev/%s", misc->name);
- ret = init_mknod(devname, S_IFCHR | 0600,
- new_encode_dev(MKDEV(MISC_MAJOR, misc->minor)));
- if (ret != 0)
- KUNIT_FAIL(test, "failed to create node\n");
- filp = filp_open(devname, O_RDONLY, 0);
- if (IS_ERR(filp))
- KUNIT_FAIL(test, "failed to open misc device: %ld\n", PTR_ERR(filp));
- else
- fput(filp);
+ /* Tests run in a nullfs kthread; borrow the init fs to resolve /dev. */
+ scoped_with_init_fs() {
+ ret = init_mknod(devname, S_IFCHR | 0600,
+ new_encode_dev(MKDEV(MISC_MAJOR, misc->minor)));
+ if (ret != 0)
+ KUNIT_FAIL(test, "failed to create node\n");
+
+ filp = filp_open(devname, O_RDONLY, 0);
+ if (IS_ERR(filp))
+ KUNIT_FAIL(test, "failed to open misc device: %ld\n", PTR_ERR(filp));
+ else
+ fput(filp);
- init_unlink(devname);
+ init_unlink(devname);
+ }
kfree(devname);
}
diff --git a/drivers/crypto/ccp/sev-dev.c b/drivers/crypto/ccp/sev-dev.c
index ca473ca198b8..0f72352030ba 100644
--- a/drivers/crypto/ccp/sev-dev.c
+++ b/drivers/crypto/ccp/sev-dev.c
@@ -260,20 +260,16 @@ static int sev_cmd_buffer_len(int cmd)
static struct file *open_file_as_root(const char *filename, int flags, umode_t mode)
{
- struct path root __free(path_put) = {};
-
- task_lock(&init_task);
- get_fs_root(init_task.fs, &root);
- task_unlock(&init_task);
-
CLASS(prepare_creds, cred)();
if (!cred)
return ERR_PTR(-ENOMEM);
cred->fsuid = GLOBAL_ROOT_UID;
- scoped_with_creds(cred)
- return file_open_root(&root, filename, flags, mode);
+ scoped_with_init_fs() {
+ scoped_with_creds(cred)
+ return filp_open(filename, flags, mode);
+ }
}
static int sev_read_init_ex_file(void)
diff --git a/drivers/md/dm-inlinecrypt.c b/drivers/md/dm-inlinecrypt.c
index 41293c18d10f..f50970db0f94 100644
--- a/drivers/md/dm-inlinecrypt.c
+++ b/drivers/md/dm-inlinecrypt.c
@@ -408,7 +408,8 @@ static int inlinecrypt_ctr(struct dm_target *ti, unsigned int argc, char **argv)
err = blk_crypto_init_key(&ctx->key, key_bytes, ctx->key_size,
ctx->key_type, cipher->mode_num,
- dun_bytes, ctx->sector_size);
+ dun_bytes, ctx->sector_size,
+ BLK_CRYPTO_CFG_ALLOW_HW);
if (err) {
ti->error = "Error initializing blk-crypto key";
goto bad;
diff --git a/drivers/mtd/ubi/ubi.h b/drivers/mtd/ubi/ubi.h
index af466cd83ae0..95548d5aa930 100644
--- a/drivers/mtd/ubi/ubi.h
+++ b/drivers/mtd/ubi/ubi.h
@@ -160,6 +160,7 @@ struct ubi_vid_io_buf {
* struct ubi_wl_entry - wear-leveling entry.
* @u.rb: link in the corresponding (free/used) RB-tree
* @u.list: link in the protection queue
+ * @u: union of rb_node and list_head
* @ec: erase counter
* @pnum: physical eraseblock number
*
@@ -282,6 +283,7 @@ struct ubi_eba_leb_desc {
* @writers: number of users holding this volume in read-write mode
* @exclusive: whether somebody holds this volume in exclusive mode
* @metaonly: whether somebody is altering only meta data of this volume
+ * @is_dead: prevents taking a reference to the volume
*
* @reserved_pebs: how many physical eraseblocks are reserved for this volume
* @vol_type: volume type (%UBI_DYNAMIC_VOLUME or %UBI_STATIC_VOLUME)
@@ -449,6 +451,7 @@ struct ubi_debug_info {
* @vol->eba_tbl.
* @ref_count: count of references on the UBI device
* @image_seq: image sequence number recorded on EC headers
+ * @is_dead: prevents taking a reference to the volume
*
* @rsvd_pebs: count of reserved physical eraseblocks
* @avail_pebs: count of available physical eraseblocks
@@ -741,7 +744,7 @@ struct ubi_ainf_volume {
* @highest_vol_id: highest volume ID
* @is_empty: flag indicating whether the MTD device is empty or not
* @force_full_scan: flag indicating whether we need to do a full scan and drop
- all existing Fastmap data structures
+ * all existing Fastmap data structures
* @min_ec: lowest erase counter value
* @max_ec: highest erase counter value
* @max_sqnum: highest sequence number value
@@ -750,7 +753,7 @@ struct ubi_ainf_volume {
* @ec_count: a temporary variable used when calculating @mean_ec
* @aeb_slab_cache: slab cache for &struct ubi_ainf_peb objects
* @ech: temporary EC header. Only available during scan
- * @vidh: temporary VID buffer. Only available during scan
+ * @vidb: temporary VID buffer. Only available during scan
*
* This data structure contains the result of attaching an MTD device and may
* be used by other UBI sub-systems to build final UBI data structures, further
@@ -1090,7 +1093,7 @@ static inline void ubi_init_vid_buf(const struct ubi_device *ubi,
}
/**
- * ubi_init_vid_buf - Allocate a VID buffer
+ * ubi_alloc_vid_buf - Allocate a VID buffer
* @ubi: the UBI device
* @gfp_flags: GFP flags to use for the allocation
*/
diff --git a/drivers/target/target_core_alua.c b/drivers/target/target_core_alua.c
index 10250aca5a81..140154d93c43 100644
--- a/drivers/target/target_core_alua.c
+++ b/drivers/target/target_core_alua.c
@@ -18,6 +18,8 @@
#include <linux/fcntl.h>
#include <linux/file.h>
#include <linux/fs.h>
+#include <linux/fs_struct.h>
+#include <linux/kthread.h>
#include <scsi/scsi_proto.h>
#include <linux/unaligned.h>
@@ -856,10 +858,17 @@ static int core_alua_write_tpg_metadata(
unsigned char *md_buf,
u32 md_buf_len)
{
- struct file *file = filp_open(path, O_RDWR | O_CREAT | O_TRUNC, 0600);
+ struct file *file;
loff_t pos = 0;
int ret;
+ if (tsk_is_kthread(current)) {
+ scoped_with_init_fs()
+ file = filp_open(path, O_RDWR | O_CREAT | O_TRUNC, 0600);
+ } else {
+ file = filp_open(path, O_RDWR | O_CREAT | O_TRUNC, 0600);
+ }
+
if (IS_ERR(file)) {
pr_err("filp_open(%s) for ALUA metadata failed\n", path);
return -ENODEV;
diff --git a/drivers/target/target_core_pr.c b/drivers/target/target_core_pr.c
index 1a77b4bb62b0..25b1bcacc0c8 100644
--- a/drivers/target/target_core_pr.c
+++ b/drivers/target/target_core_pr.c
@@ -18,6 +18,7 @@
#include <linux/file.h>
#include <linux/fcntl.h>
#include <linux/fs.h>
+#include <linux/fs_struct.h>
#include <scsi/scsi_proto.h>
#include <linux/unaligned.h>
@@ -1969,7 +1970,8 @@ static int __core_scsi3_write_aptpl_to_file(
if (!path)
return -ENOMEM;
- file = filp_open(path, flags, 0600);
+ scoped_with_init_fs()
+ file = filp_open(path, flags, 0600);
if (IS_ERR(file)) {
pr_err("filp_open(%s) for APTPL metadata"
" failed\n", path);
diff --git a/fs/9p/vfs_inode.c b/fs/9p/vfs_inode.c
index 5783d0336f96..3829554ca369 100644
--- a/fs/9p/vfs_inode.c
+++ b/fs/9p/vfs_inode.c
@@ -645,7 +645,6 @@ error:
* @dir: The parent directory
* @dentry: The name of file to be created
* @mode: The UNIX file mode to set
- * @excl: True if the file must not yet exist
*
* open(.., O_CREAT) is handled in v9fs_vfs_atomic_open(). This is only called
* for mknod(2).
@@ -654,7 +653,7 @@ error:
static int
v9fs_vfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct v9fs_session_info *v9ses = v9fs_inode2v9ses(dir);
u32 perm = unixmode2p9mode(v9ses, mode);
@@ -689,7 +688,7 @@ static struct dentry *v9fs_vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
p9_debug(P9_DEBUG_VFS, "name %pd\n", dentry);
v9ses = v9fs_inode2v9ses(dir);
- perm = unixmode2p9mode(v9ses, mode | S_IFDIR);
+ perm = unixmode2p9mode(v9ses, mode);
fid = v9fs_create(v9ses, dir, dentry, NULL, perm, P9_OREAD);
if (IS_ERR(fid))
return ERR_CAST(fid);
diff --git a/fs/9p/vfs_inode_dotl.c b/fs/9p/vfs_inode_dotl.c
index f7396d20cb6c..116b29e95f21 100644
--- a/fs/9p/vfs_inode_dotl.c
+++ b/fs/9p/vfs_inode_dotl.c
@@ -213,12 +213,11 @@ int v9fs_open_to_dotl_flags(int flags)
* @dir: directory inode that is being created
* @dentry: dentry that is being deleted
* @omode: create permissions
- * @excl: True if the file must not yet exist
*
*/
static int
v9fs_vfs_create_dotl(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t omode, bool excl)
+ struct dentry *dentry, umode_t omode)
{
return v9fs_vfs_mknod_dotl(idmap, dir, dentry, omode, 0);
}
@@ -362,7 +361,6 @@ static struct dentry *v9fs_vfs_mkdir_dotl(struct mnt_idmap *idmap,
p9_debug(P9_DEBUG_VFS, "name %pd\n", dentry);
v9ses = v9fs_inode2v9ses(dir);
- omode |= S_IFDIR;
if (dir->i_mode & S_ISGID)
omode |= S_ISGID;
diff --git a/fs/Kconfig b/fs/Kconfig
index cf6ae64776e6..64ff193fb42c 100644
--- a/fs/Kconfig
+++ b/fs/Kconfig
@@ -315,7 +315,6 @@ source "fs/hfs/Kconfig"
source "fs/hfsplus/Kconfig"
source "fs/befs/Kconfig"
source "fs/bfs/Kconfig"
-source "fs/efs/Kconfig"
source "fs/jffs2/Kconfig"
# UBIFS File system configuration
source "fs/ubifs/Kconfig"
diff --git a/fs/Kconfig.binfmt b/fs/Kconfig.binfmt
index 1949e25c7741..daeac4889d03 100644
--- a/fs/Kconfig.binfmt
+++ b/fs/Kconfig.binfmt
@@ -168,6 +168,20 @@ config BINFMT_MISC
you have use for it; the module is called binfmt_misc. If you
don't know what to answer at this point, say Y.
+config BINFMT_MISC_BPF
+ bool "BPF-selected interpreters for misc binaries"
+ depends on BINFMT_MISC=y
+ depends on BPF_SYSCALL && BPF_JIT && DEBUG_INFO_BTF
+ help
+ Allow binfmt_misc binary type handlers to be implemented as bpf
+ struct_ops programs. Instead of matching a fixed magic and
+ redirecting to a fixed interpreter recorded at registration time
+ such handlers match binaries programmatically and compute the
+ interpreter to use per binary, e.g. relative to the location of
+ the binary itself.
+
+ If you don't know what to answer at this point, say N.
+
config COREDUMP
bool "Enable core dump support" if EXPERT
default y
diff --git a/fs/Makefile b/fs/Makefile
index 89a8a9d207d1..447a1f7c5732 100644
--- a/fs/Makefile
+++ b/fs/Makefile
@@ -33,6 +33,7 @@ obj-$(CONFIG_FS_ENCRYPTION) += crypto/
obj-$(CONFIG_FS_VERITY) += verity/
obj-$(CONFIG_FILE_LOCKING) += locks.o
obj-$(CONFIG_BINFMT_MISC) += binfmt_misc.o
+obj-$(CONFIG_BINFMT_MISC_BPF) += binfmt_misc_bpf.o
obj-$(CONFIG_BINFMT_SCRIPT) += binfmt_script.o
obj-$(CONFIG_BINFMT_ELF) += binfmt_elf.o
obj-$(CONFIG_COMPAT_BINFMT_ELF) += compat_binfmt_elf.o
@@ -92,7 +93,6 @@ obj-$(CONFIG_HPFS_FS) += hpfs/
obj-$(CONFIG_NTFS_FS) += ntfs/
obj-$(CONFIG_NTFS3_FS) += ntfs3/
obj-$(CONFIG_UFS_FS) += ufs/
-obj-$(CONFIG_EFS_FS) += efs/
obj-$(CONFIG_JFFS2_FS) += jffs2/
obj-$(CONFIG_UBIFS_FS) += ubifs/
obj-$(CONFIG_AFFS_FS) += affs/
diff --git a/fs/affs/affs.h b/fs/affs/affs.h
index 44a3f69d275f..d6b3393633f2 100644
--- a/fs/affs/affs.h
+++ b/fs/affs/affs.h
@@ -169,7 +169,7 @@ extern int affs_hash_name(struct super_block *sb, const u8 *name, unsigned int l
extern struct dentry *affs_lookup(struct inode *dir, struct dentry *dentry, unsigned int);
extern int affs_unlink(struct inode *dir, struct dentry *dentry);
extern int affs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool);
+ struct dentry *dentry, umode_t mode);
extern struct dentry *affs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode);
extern int affs_rmdir(struct inode *dir, struct dentry *dentry);
diff --git a/fs/affs/namei.c b/fs/affs/namei.c
index c3c6532da4b0..5311828b19e8 100644
--- a/fs/affs/namei.c
+++ b/fs/affs/namei.c
@@ -243,7 +243,7 @@ affs_unlink(struct inode *dir, struct dentry *dentry)
int
affs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct super_block *sb = dir->i_sb;
struct inode *inode;
@@ -287,7 +287,7 @@ affs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (!inode)
return ERR_PTR(-ENOSPC);
- inode->i_mode = S_IFDIR | mode;
+ inode->i_mode = mode;
affs_mode_to_prot(inode);
inode->i_op = &affs_dir_inode_operations;
diff --git a/fs/affs/super.c b/fs/affs/super.c
index b232251aa7bb..4f331f784db2 100644
--- a/fs/affs/super.c
+++ b/fs/affs/super.c
@@ -88,7 +88,7 @@ void affs_mark_sb_dirty(struct super_block *sb)
spin_lock(&sbi->work_lock);
if (!sbi->work_queued) {
delay = msecs_to_jiffies(dirty_writeback_interval * 10);
- queue_delayed_work(system_long_wq, &sbi->sb_work, delay);
+ queue_delayed_work(system_dfl_long_wq, &sbi->sb_work, delay);
sbi->work_queued = 1;
}
spin_unlock(&sbi->work_lock);
diff --git a/fs/afs/dir.c b/fs/afs/dir.c
index 6df56fe9163f..81565366d937 100644
--- a/fs/afs/dir.c
+++ b/fs/afs/dir.c
@@ -34,7 +34,7 @@ static bool afs_lookup_filldir(struct dir_context *ctx, const char *name, int nl
u64 ino, u32 uniquifier);
#define AFS_LOOKUP ((filldir_t)0x137UL)
static int afs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl);
+ struct dentry *dentry, umode_t mode);
static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode);
static int afs_rmdir(struct inode *dir, struct dentry *dentry);
@@ -1333,7 +1333,7 @@ static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
op->file[0].modification = true;
op->file[0].update_ctime = true;
op->dentry = dentry;
- op->create.mode = S_IFDIR | mode;
+ op->create.mode = mode;
op->create.reason = afs_edit_dir_for_mkdir;
op->mtime = current_time(dir);
op->ops = &afs_mkdir_operation;
@@ -1633,7 +1633,7 @@ static const struct afs_operation_ops afs_create_operation = {
* create a regular file on an AFS filesystem
*/
static int afs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct afs_operation *op;
struct afs_vnode *dvnode = AFS_FS_I(dir);
diff --git a/fs/autofs/root.c b/fs/autofs/root.c
index 186e960f1e23..b36439f4521e 100644
--- a/fs/autofs/root.c
+++ b/fs/autofs/root.c
@@ -741,7 +741,7 @@ static struct dentry *autofs_dir_mkdir(struct mnt_idmap *idmap,
autofs_del_active(dentry);
- inode = autofs_get_inode(dir->i_sb, S_IFDIR | mode);
+ inode = autofs_get_inode(dir->i_sb, mode);
if (!inode)
return ERR_PTR(-ENOMEM);
diff --git a/fs/bad_inode.c b/fs/bad_inode.c
index acf8613f5e36..486c40f73e51 100644
--- a/fs/bad_inode.c
+++ b/fs/bad_inode.c
@@ -29,7 +29,7 @@ static const struct file_operations bad_file_ops =
static int bad_inode_create(struct mnt_idmap *idmap,
struct inode *dir, struct dentry *dentry,
- umode_t mode, bool excl)
+ umode_t mode)
{
return -EIO;
}
diff --git a/fs/bfs/dir.c b/fs/bfs/dir.c
index 5b40ab09a796..4c3b4db08cde 100644
--- a/fs/bfs/dir.c
+++ b/fs/bfs/dir.c
@@ -83,7 +83,7 @@ const struct file_operations bfs_dir_operations = {
};
static int bfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
int err;
struct inode *inode;
diff --git a/fs/binfmt_misc.c b/fs/binfmt_misc.c
index 5de615ca7a75..c06cea00a6d6 100644
--- a/fs/binfmt_misc.c
+++ b/fs/binfmt_misc.c
@@ -10,45 +10,49 @@
#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
-#include <linux/kernel.h>
-#include <linux/module.h>
-#include <linux/hex.h>
-#include <linux/init.h>
-#include <linux/sched/mm.h>
-#include <linux/magic.h>
+#include <linux/binfmt_misc.h>
#include <linux/binfmts.h>
-#include <linux/slab.h>
+#include <linux/bitops.h>
+#include <linux/bits.h>
+#include <linux/bug.h>
+#include <linux/cleanup.h>
+#include <linux/cred.h>
#include <linux/ctype.h>
-#include <linux/string_helpers.h>
#include <linux/file.h>
-#include <linux/pagemap.h>
-#include <linux/namei.h>
-#include <linux/mount.h>
-#include <linux/fs_context.h>
-#include <linux/syscalls.h>
#include <linux/fs.h>
+#include <linux/fs_context.h>
+#include <linux/init.h>
+#include <linux/kstrtox.h>
+#include <linux/magic.h>
+#include <linux/module.h>
+#include <linux/printk.h>
+#include <linux/rculist.h>
+#include <linux/refcount.h>
+#include <linux/seq_file.h>
+#include <linux/slab.h>
+#include <linux/srcu.h>
+#include <linux/string.h>
+#include <linux/string_helpers.h>
#include <linux/uaccess.h>
+#include <linux/user_namespace.h>
-#include "internal.h"
-
-#ifdef DEBUG
-# define USE_DEBUG 1
-#else
-# define USE_DEBUG 0
-#endif
-
-enum {
- VERBOSE_STATUS = 1 /* make it zero to save 400 bytes kernel memory */
+/* Entry status and match type bit numbers. */
+enum binfmt_misc_entry_bits {
+ MISC_FMT_ENABLED_BIT = 0,
+ MISC_FMT_MAGIC_BIT = 1,
+ MISC_FMT_BPF_BIT = 2,
};
-enum {Enabled, Magic};
-#define MISC_FMT_PRESERVE_ARGV0 (1UL << 31)
-#define MISC_FMT_OPEN_BINARY (1UL << 30)
-#define MISC_FMT_CREDENTIALS (1UL << 29)
-#define MISC_FMT_OPEN_FILE (1UL << 28)
+/* Entry behavior flags, fixed at registration time. */
+enum binfmt_misc_entry_flags {
+ MISC_FMT_PRESERVE_ARGV0 = (1U << 31),
+ MISC_FMT_OPEN_BINARY = (1U << 30),
+ MISC_FMT_CREDENTIALS = (1U << 29),
+ MISC_FMT_OPEN_FILE = (1U << 28),
+};
-typedef struct {
- struct list_head list;
+struct binfmt_misc_entry {
+ struct hlist_node node;
unsigned long flags; /* type, status, etc. */
int offset; /* offset of magic */
int size; /* size of magic/mask */
@@ -58,10 +62,12 @@ typedef struct {
char *name;
struct dentry *dentry;
struct file *interp_file;
+ const struct binfmt_misc_ops *bpf_ops; /* bpf-backed handler ('B') */
+ const char *bpf_ops_name;
refcount_t users; /* sync removal with load_misc_binary() */
-} Node;
-
-static struct file_system_type bm_fs_type;
+ struct rcu_head rcu;
+ char buf[]; /* register string, fields point in here */
+};
/*
* Max length of the register string. Determined by:
@@ -74,54 +80,83 @@ static struct file_system_type bm_fs_type;
* - interp: ~50 bytes
* - flags: 5 bytes
* Round that up a bit, and then back off to hold the internal data
- * (like struct Node).
+ * (like struct binfmt_misc_entry).
*/
#define MAX_REGISTER_LENGTH 1920
+/* Trailing delimiter pad so field parsing always terminates at a delimiter. */
+#define MISC_DELIM_PAD 8
+
+/* Protects the entry walk in load_misc_binary(), which may sleep in it. */
+DEFINE_STATIC_SRCU_FAST(bm_entries_srcu);
+
+/* Check if @e's magic matches @bprm's buffer, applying the mask if set. */
+static bool entry_matches_magic(const struct binfmt_misc_entry *e,
+ const struct linux_binprm *bprm)
+{
+ const char *s = bprm->buf + e->offset;
+ int i;
+
+ if (!e->mask)
+ return !memcmp(s, e->magic, e->size);
+
+ for (i = 0; i < e->size; i++)
+ if ((s[i] ^ e->magic[i]) & e->mask[i])
+ return false;
+ return true;
+}
+
+/* Check if @e's registered extension matches @ext, NULL if there is none. */
+static bool entry_matches_extension(const struct binfmt_misc_entry *e,
+ const char *ext)
+{
+ return ext && !strcmp(e->magic, ext);
+}
+
/**
* search_binfmt_handler - search for a binary handler for @bprm
* @misc: handle to binfmt_misc instance
* @bprm: binary for which we are looking for a handler
*
* Search for a binary type handler for @bprm in the list of registered binary
- * type handlers.
+ * type handlers. A 'B' entry's match program decides whether the handler
+ * applies; it may sleep to read the binary. The matched entry is returned
+ * with a reference taken while the walk still held it; a dying entry -
+ * unlinked with its last reference gone - cannot be matched and the walk
+ * moves on.
*
- * Return: binary type list entry on success, NULL on failure
+ * The caller must hold the bm_entries_srcu read lock, which allows an
+ * entry's evaluation to sleep.
+ *
+ * Return: referenced binary type list entry on success, NULL on failure
*/
-static Node *search_binfmt_handler(struct binfmt_misc *misc,
- struct linux_binprm *bprm)
+static struct binfmt_misc_entry *
+search_binfmt_handler(struct binfmt_misc *misc, struct linux_binprm *bprm)
{
- char *p = strrchr(bprm->interp, '.');
- Node *e;
+ char *dot = strrchr(bprm->interp, '.');
+ const char *ext = dot ? dot + 1 : NULL;
+ struct binfmt_misc_entry *e;
/* Walk all the registered handlers. */
- list_for_each_entry(e, &misc->entries, list) {
- char *s;
- int j;
-
+ hlist_for_each_entry_rcu(e, &misc->entries, node,
+ srcu_read_lock_held(&bm_entries_srcu)) {
/* Make sure this one is currently enabled. */
- if (!test_bit(Enabled, &e->flags))
- continue;
-
- /* Do matching based on extension if applicable. */
- if (!test_bit(Magic, &e->flags)) {
- if (p && !strcmp(e->magic, p + 1))
- return e;
+ if (!test_bit(MISC_FMT_ENABLED_BIT, &e->flags))
continue;
- }
- /* Do matching based on magic & mask. */
- s = bprm->buf + e->offset;
- if (e->mask) {
- for (j = 0; j < e->size; j++)
- if ((*s++ ^ e->magic[j]) & e->mask[j])
- break;
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags)) {
+ if (!e->bpf_ops->match(bprm))
+ continue;
+ } else if (test_bit(MISC_FMT_MAGIC_BIT, &e->flags)) {
+ if (!entry_matches_magic(e, bprm))
+ continue;
} else {
- for (j = 0; j < e->size; j++)
- if ((*s++ ^ e->magic[j]))
- break;
+ if (!entry_matches_extension(e, ext))
+ continue;
}
- if (j == e->size)
+
+ /* A dying entry cannot be matched, walk on. */
+ if (refcount_inc_not_zero(&e->users))
return e;
}
@@ -133,157 +168,244 @@ static Node *search_binfmt_handler(struct binfmt_misc *misc,
* @misc: handle to binfmt_misc instance
* @bprm: binary for which we are looking for a handler
*
- * Try to find a binfmt handler for the binary type. If one is found take a
- * reference to protect against removal via bm_{entry,status}_write().
+ * Try to find a binfmt handler for the binary type. If one is found it is
+ * returned with a reference protecting it against removal via
+ * bm_{entry,status}_write().
*
* Return: binary type list entry on success, NULL on failure
*/
-static Node *get_binfmt_handler(struct binfmt_misc *misc,
- struct linux_binprm *bprm)
+static struct binfmt_misc_entry *get_binfmt_handler(struct binfmt_misc *misc,
+ struct linux_binprm *bprm)
+{
+ guard(srcu_fast)(&bm_entries_srcu);
+ return search_binfmt_handler(misc, bprm);
+}
+
+static void bm_entry_free_rcu(struct rcu_head *rcu)
{
- Node *e;
-
- read_lock(&misc->entries_lock);
- e = search_binfmt_handler(misc, bprm);
- if (e)
- refcount_inc(&e->users);
- read_unlock(&misc->entries_lock);
- return e;
+ struct binfmt_misc_entry *e = container_of(rcu, struct binfmt_misc_entry, rcu);
+
+ /* No walker that could sleep in the handler's programs is left. */
+ if (e->bpf_ops)
+ binfmt_misc_put_ops(e->bpf_ops);
+ kfree(e);
}
/**
- * put_binfmt_handler - put binary handler node
- * @e: node to put
+ * put_binfmt_handler - put binary handler entry
+ * @e: entry to put
*
- * Free node syncing with load_misc_binary() and defer final free to
+ * Free entry syncing with load_misc_binary() and defer final free to
* load_misc_binary() in case it is using the binary type handler we were
* requested to remove.
*/
-static void put_binfmt_handler(Node *e)
+static void put_binfmt_handler(struct binfmt_misc_entry *e)
{
if (refcount_dec_and_test(&e->users)) {
- if (e->flags & MISC_FMT_OPEN_FILE)
+ if (e->flags & MISC_FMT_OPEN_FILE) {
+ exe_file_allow_write_access(e->interp_file);
filp_close(e->interp_file, NULL);
- kfree(e);
+ }
+ /* Walkers may still dereference this entry, even sleeping. */
+ call_srcu(&bm_entries_srcu, &e->rcu, bm_entry_free_rcu);
}
}
+DEFINE_FREE(put_binfmt_handler, struct binfmt_misc_entry *, if (_T) put_binfmt_handler(_T))
+
/**
- * load_binfmt_misc - load the binfmt_misc of the caller's user namespace
+ * current_binfmt_misc - get the binfmt_misc instance of the caller's user namespace
*
- * To be called in load_misc_binary() to load the relevant struct binfmt_misc.
- * If a user namespace doesn't have its own binfmt_misc mount it can make use
- * of its ancestor's binfmt_misc handlers. This mimicks the behavior of
- * pre-namespaced binfmt_misc where all registered binfmt_misc handlers where
- * available to all user and user namespaces on the system.
+ * If a user namespace doesn't have its own binfmt_misc mount it uses the
+ * handlers of its closest ancestor with one. This mimics the behavior of
+ * pre-namespaced binfmt_misc where all registered handlers were available
+ * to all users and user namespaces on the system. The init user namespace
+ * instance is statically set up so the fallback is never reached in
+ * practice.
*
* Return: the binfmt_misc instance of the caller's user namespace
*/
-static struct binfmt_misc *load_binfmt_misc(void)
+static struct binfmt_misc *current_binfmt_misc(void)
{
const struct user_namespace *user_ns;
struct binfmt_misc *misc;
- user_ns = current_user_ns();
- while (user_ns) {
+ for (user_ns = current_user_ns(); user_ns; user_ns = user_ns->parent) {
/* Pairs with smp_store_release() in bm_fill_super(). */
misc = smp_load_acquire(&user_ns->binfmt_misc);
if (misc)
return misc;
-
- user_ns = user_ns->parent;
}
return &init_binfmt_misc;
}
+/**
+ * entry_select_interpreter - get the interpreter for the matched @e
+ * @e: matched binary type handler
+ * @bprm: binary that is being executed
+ *
+ * A static entry carries its interpreter path, for a 'B' entry the
+ * handler's load program selects it. The match is committed, so a failing
+ * program fails the exec.
+ *
+ * Return: the interpreter on success, an ERR_PTR on failure
+ */
+static const char *entry_select_interpreter(const struct binfmt_misc_entry *e,
+ struct linux_binprm *bprm)
+{
+ int retval;
+
+ if (!test_bit(MISC_FMT_BPF_BIT, &e->flags))
+ return e->interpreter;
+
+ /* Drop any interpreter or flags a previous chain level staged. */
+ kfree(bprm->bpf_interp);
+ bprm->bpf_interp = NULL;
+ bprm->bpf_flags = 0;
+
+ retval = e->bpf_ops->load(bprm);
+ if (retval) {
+ /* Keep a program-supplied error within errno range. */
+ if (retval > 0 || retval < -MAX_ERRNO)
+ retval = -ENOEXEC;
+ goto drop_staged;
+ }
+
+ /* Selecting an interpreter is part of the contract. */
+ if (!bprm->bpf_interp) {
+ retval = -ENOEXEC;
+ goto drop_staged;
+ }
+
+ return bprm->bpf_interp;
+
+drop_staged:
+ /* A failing load leaves nothing behind for later entries. */
+ kfree(bprm->bpf_interp_arg);
+ bprm->bpf_interp_arg = NULL;
+ bprm->bpf_flags = 0;
+ return ERR_PTR(retval);
+}
+
/*
* the loader itself
*/
static int load_misc_binary(struct linux_binprm *bprm)
{
- Node *fmt;
- struct file *interp_file = NULL;
- int retval = -ENOEXEC;
+ struct binfmt_misc_entry *fmt __free(put_binfmt_handler) = NULL;
+ const char *interpreter;
+ struct file *interp_file;
struct binfmt_misc *misc;
+ bool preserve_argv0, want_execfd, want_creds;
+ int retval;
- misc = load_binfmt_misc();
- if (!misc->enabled)
- return retval;
+ misc = current_binfmt_misc();
+ if (!READ_ONCE(misc->enabled))
+ return -ENOEXEC;
fmt = get_binfmt_handler(misc, bprm);
if (!fmt)
- return retval;
+ return -ENOEXEC;
/* Need to be able to load the file after exec */
- retval = -ENOENT;
if (bprm->interp_flags & BINPRM_FLAGS_PATH_INACCESSIBLE)
- goto ret;
+ return -ENOENT;
+
+ interpreter = entry_select_interpreter(fmt, bprm);
+ if (IS_ERR(interpreter))
+ return PTR_ERR(interpreter);
+
+ /*
+ * The invocation flags are fixed at registration for a static handler
+ * and chosen per exec by the load program, via bpf_binprm_set_flags(),
+ * for a bpf one.
+ */
+ if (test_bit(MISC_FMT_BPF_BIT, &fmt->flags)) {
+ u64 f = bprm->bpf_flags;
- if (fmt->flags & MISC_FMT_PRESERVE_ARGV0) {
+ /* Clear so it can't accumulate into a nested interpreter level. */
+ bprm->bpf_flags = 0;
+
+ preserve_argv0 = f & BPF_BINPRM_PRESERVE_ARGV0;
+ want_creds = f & BPF_BINPRM_CREDENTIALS;
+ want_execfd = f & (BPF_BINPRM_CREDENTIALS | BPF_BINPRM_EXECFD);
+ } else {
+ preserve_argv0 = fmt->flags & MISC_FMT_PRESERVE_ARGV0;
+ want_creds = fmt->flags & MISC_FMT_CREDENTIALS;
+ want_execfd = fmt->flags & MISC_FMT_OPEN_BINARY;
+ }
+
+ /* The entry's own choice - not one accumulated from an earlier level. */
+ if (preserve_argv0) {
bprm->interp_flags |= BINPRM_FLAGS_PRESERVE_ARGV0;
} else {
retval = remove_arg_zero(bprm);
if (retval)
- goto ret;
+ return retval;
}
- /* make argv[1] be the path to the binary */
+ /* make the binary the last argument to the interpreter */
retval = copy_string_kernel(bprm->interp, bprm);
if (retval < 0)
- goto ret;
+ return retval;
bprm->argc++;
+ /*
+ * A single optional argument to the interpreter, inserted between it
+ * and the binary just like the argument of a #! interpreter line.
+ */
+ if (bprm->bpf_interp_arg) {
+ retval = copy_string_kernel(bprm->bpf_interp_arg, bprm);
+ if (retval < 0)
+ return retval;
+ bprm->argc++;
+ /* Consumed - don't let it leak into a nested interpreter's argv. */
+ kfree(bprm->bpf_interp_arg);
+ bprm->bpf_interp_arg = NULL;
+ }
+
/* add the interp as argv[0] */
- retval = copy_string_kernel(fmt->interpreter, bprm);
+ retval = copy_string_kernel(interpreter, bprm);
if (retval < 0)
- goto ret;
+ return retval;
bprm->argc++;
/* Update interp in case binfmt_script needs it. */
- retval = bprm_change_interp(fmt->interpreter, bprm);
+ retval = bprm_change_interp(interpreter, bprm);
if (retval < 0)
- goto ret;
+ return retval;
if (fmt->flags & MISC_FMT_OPEN_FILE) {
interp_file = file_clone_open(fmt->interp_file);
- if (!IS_ERR(interp_file))
- deny_write_access(interp_file);
+ if (!IS_ERR(interp_file)) {
+ int err = exe_file_deny_write_access(interp_file);
+
+ if (err) {
+ fput(interp_file);
+ interp_file = ERR_PTR(err);
+ }
+ }
} else {
- interp_file = open_exec(fmt->interpreter);
+ interp_file = open_exec(interpreter);
}
- retval = PTR_ERR(interp_file);
if (IS_ERR(interp_file))
- goto ret;
+ return PTR_ERR(interp_file);
bprm->interpreter = interp_file;
- if (fmt->flags & MISC_FMT_OPEN_BINARY)
+ if (want_execfd)
bprm->have_execfd = 1;
- if (fmt->flags & MISC_FMT_CREDENTIALS)
+ if (want_creds)
bprm->execfd_creds = 1;
-
- retval = 0;
-ret:
-
- /*
- * If we actually put the node here all concurrent calls to
- * load_misc_binary() will have finished. We also know
- * that for the refcount to be zero someone must have concurently
- * removed the binary type handler from the list and it's our job to
- * free it.
- */
- put_binfmt_handler(fmt);
-
- return retval;
+ return 0;
}
/* Command parsers */
/*
- * parses and copies one argument enclosed in del from *sp to *dp,
- * recognising the \x special.
- * returns pointer to the copied argument or NULL in case of an
- * error (and sets err) or null argument length.
+ * Scan the argument starting at @s up to the delimiter @del, recognising
+ * the \x escape. Terminates the argument with a NUL and returns a pointer
+ * past it or NULL on a malformed escape.
*/
static char *scanarg(char *s, char del)
{
@@ -298,45 +420,143 @@ static char *scanarg(char *s, char del)
return NULL;
}
}
- s[-1] ='\0';
+ s[-1] = '\0';
return s;
}
-static char *check_special_flags(char *sfs, Node *e)
+static char *check_special_flags(char *p, struct binfmt_misc_entry *e)
{
- char *p = sfs;
- int cont = 1;
-
- /* special flags */
- while (cont) {
+ for (;; p++) {
switch (*p) {
case 'P':
pr_debug("register: flag: P (preserve argv0)\n");
- p++;
e->flags |= MISC_FMT_PRESERVE_ARGV0;
break;
case 'O':
pr_debug("register: flag: O (open binary)\n");
- p++;
e->flags |= MISC_FMT_OPEN_BINARY;
break;
case 'C':
pr_debug("register: flag: C (preserve creds)\n");
- p++;
- /* this flags also implies the
- open-binary flag */
- e->flags |= (MISC_FMT_CREDENTIALS |
- MISC_FMT_OPEN_BINARY);
+ /* C implies O */
+ e->flags |= MISC_FMT_CREDENTIALS | MISC_FMT_OPEN_BINARY;
break;
case 'F':
pr_debug("register: flag: F: open interpreter file now\n");
- p++;
e->flags |= MISC_FMT_OPEN_FILE;
break;
default:
- cont = 0;
+ return p;
}
}
+}
+
+/* Parse the 'offset', 'magic' and 'mask' fields of an 'M' entry. */
+static char *parse_magic_fields(struct binfmt_misc_entry *e, char *p, char del)
+{
+ char *s;
+
+ /* Parse the 'offset' field. */
+ s = strchr(p, del);
+ if (!s)
+ return NULL;
+ *s = '\0';
+ if (p != s) {
+ if (kstrtoint(p, 10, &e->offset) || e->offset < 0)
+ return NULL;
+ }
+ p = s + 1;
+ pr_debug("register: offset: %#x\n", e->offset);
+
+ /* Parse the 'magic' field. */
+ e->magic = p;
+ p = scanarg(p, del);
+ if (!p || !e->magic[0])
+ return NULL;
+ print_hex_dump_debug(
+ KBUILD_MODNAME ": register: magic[raw]: ",
+ DUMP_PREFIX_NONE, 16, 1, e->magic, p - e->magic, true);
+
+ /* Parse the 'mask' field. */
+ e->mask = p;
+ p = scanarg(p, del);
+ if (!p)
+ return NULL;
+ if (!e->mask[0]) {
+ e->mask = NULL;
+ pr_debug("register: mask[raw]: none\n");
+ } else {
+ print_hex_dump_debug(
+ KBUILD_MODNAME ": register: mask[raw]: ",
+ DUMP_PREFIX_NONE, 16, 1, e->mask, p - e->mask, true);
+ }
+
+ /*
+ * Decode the magic & mask fields. Note: while we might have accepted
+ * embedded NUL bytes from above, the unescape helpers will stop at
+ * the first one they encounter.
+ */
+ e->size = string_unescape_inplace(e->magic, UNESCAPE_HEX);
+ if (e->mask && string_unescape_inplace(e->mask, UNESCAPE_HEX) != e->size)
+ return NULL;
+ if (e->size > BINPRM_BUF_SIZE || BINPRM_BUF_SIZE - e->size < e->offset)
+ return NULL;
+ pr_debug("register: magic/mask length: %i\n", e->size);
+ print_hex_dump_debug(
+ KBUILD_MODNAME ": register: magic[decoded]: ",
+ DUMP_PREFIX_NONE, 16, 1, e->magic, e->size, true);
+ if (e->mask)
+ print_hex_dump_debug(
+ KBUILD_MODNAME ": register: mask[decoded]: ",
+ DUMP_PREFIX_NONE, 16, 1, e->mask, e->size, true);
+ return p;
+}
+
+/* Parse the 'magic' field of an 'E' entry: the filename extension. */
+static char *parse_extension_fields(struct binfmt_misc_entry *e, char *p,
+ char del)
+{
+ /* Skip the 'offset' field. */
+ p = strchr(p, del);
+ if (!p)
+ return NULL;
+ *p++ = '\0';
+
+ /* Parse the 'magic' field. */
+ e->magic = p;
+ p = strchr(p, del);
+ if (!p)
+ return NULL;
+ *p++ = '\0';
+ if (!e->magic[0] || strchr(e->magic, '/'))
+ return NULL;
+ pr_debug("register: extension: {%s}\n", e->magic);
+
+ /* Skip the 'mask' field. */
+ p = strchr(p, del);
+ if (!p)
+ return NULL;
+ *p++ = '\0';
+ return p;
+}
+
+/*
+ * Parse the fields of a 'B' entry: the 'offset', 'magic' and 'mask' fields
+ * must be empty. The handler name is carried in the 'interpreter' field.
+ */
+static char *parse_bpf_fields(struct binfmt_misc_entry *e, char *p, char del)
+{
+ /* The 'offset' field must be empty. */
+ if (*p++ != del)
+ return NULL;
+
+ /* The 'magic' field must be empty. */
+ if (*p++ != del)
+ return NULL;
+
+ /* The 'mask' field must be empty. */
+ if (*p++ != del)
+ return NULL;
return p;
}
@@ -346,50 +566,52 @@ static char *check_special_flags(char *sfs, Node *e)
* ':name:type:offset:magic:mask:interpreter:flags'
* where the ':' is the IFS, that can be chosen with the first char
*/
-static Node *create_entry(const char __user *buffer, size_t count)
+static struct binfmt_misc_entry *create_entry(const char __user *buffer,
+ size_t count)
{
- Node *e;
- int memsize, err;
+ struct binfmt_misc_entry *e __free(kfree) = NULL;
char *buf, *p;
char del;
pr_debug("register: received %zu bytes\n", count);
/* some sanity checks */
- err = -EINVAL;
if ((count < 11) || (count > MAX_REGISTER_LENGTH))
- goto out;
+ return ERR_PTR(-EINVAL);
- err = -ENOMEM;
- memsize = sizeof(Node) + count + 8;
- e = kmalloc(memsize, GFP_KERNEL_ACCOUNT);
+ e = kmalloc(struct_size(e, buf, count + MISC_DELIM_PAD),
+ GFP_KERNEL_ACCOUNT);
if (!e)
- goto out;
+ return ERR_PTR(-ENOMEM);
- p = buf = (char *)e + sizeof(Node);
+ p = buf = e->buf;
- memset(e, 0, sizeof(Node));
+ memset(e, 0, sizeof(*e));
if (copy_from_user(buf, buffer, count))
- goto efault;
+ return ERR_PTR(-EFAULT);
- del = *p++; /* delimeter */
+ del = *p++; /* delimiter */
pr_debug("register: delim: %#x {%c}\n", del, del);
+ /* A flag-char delimiter runs the flag scan off the buffer. */
+ if (del == 'P' || del == 'O' || del == 'C' || del == 'F')
+ return ERR_PTR(-EINVAL);
+
/* Pad the buffer with the delim to simplify parsing below. */
- memset(buf + count, del, 8);
+ memset(buf + count, del, MISC_DELIM_PAD);
/* Parse the 'name' field. */
e->name = p;
p = strchr(p, del);
if (!p)
- goto einval;
+ return ERR_PTR(-EINVAL);
*p++ = '\0';
if (!e->name[0] ||
!strcmp(e->name, ".") ||
!strcmp(e->name, "..") ||
strchr(e->name, '/'))
- goto einval;
+ return ERR_PTR(-EINVAL);
pr_debug("register: name: {%s}\n", e->name);
@@ -397,159 +619,84 @@ static Node *create_entry(const char __user *buffer, size_t count)
switch (*p++) {
case 'E':
pr_debug("register: type: E (extension)\n");
- e->flags = 1 << Enabled;
+ e->flags = BIT(MISC_FMT_ENABLED_BIT);
break;
case 'M':
pr_debug("register: type: M (magic)\n");
- e->flags = (1 << Enabled) | (1 << Magic);
+ e->flags = BIT(MISC_FMT_ENABLED_BIT) | BIT(MISC_FMT_MAGIC_BIT);
+ break;
+ case 'B':
+ pr_debug("register: type: B (bpf)\n");
+ if (!IS_ENABLED(CONFIG_BINFMT_MISC_BPF))
+ return ERR_PTR(-EINVAL);
+ e->flags = BIT(MISC_FMT_ENABLED_BIT) | BIT(MISC_FMT_BPF_BIT);
break;
default:
- goto einval;
+ return ERR_PTR(-EINVAL);
}
if (*p++ != del)
- goto einval;
-
- if (test_bit(Magic, &e->flags)) {
- /* Handle the 'M' (magic) format. */
- char *s;
-
- /* Parse the 'offset' field. */
- s = strchr(p, del);
- if (!s)
- goto einval;
- *s = '\0';
- if (p != s) {
- int r = kstrtoint(p, 10, &e->offset);
- if (r != 0 || e->offset < 0)
- goto einval;
- }
- p = s;
- if (*p++)
- goto einval;
- pr_debug("register: offset: %#x\n", e->offset);
-
- /* Parse the 'magic' field. */
- e->magic = p;
- p = scanarg(p, del);
- if (!p)
- goto einval;
- if (!e->magic[0])
- goto einval;
- if (USE_DEBUG)
- print_hex_dump_bytes(
- KBUILD_MODNAME ": register: magic[raw]: ",
- DUMP_PREFIX_NONE, e->magic, p - e->magic);
-
- /* Parse the 'mask' field. */
- e->mask = p;
- p = scanarg(p, del);
- if (!p)
- goto einval;
- if (!e->mask[0]) {
- e->mask = NULL;
- pr_debug("register: mask[raw]: none\n");
- } else if (USE_DEBUG)
- print_hex_dump_bytes(
- KBUILD_MODNAME ": register: mask[raw]: ",
- DUMP_PREFIX_NONE, e->mask, p - e->mask);
-
- /*
- * Decode the magic & mask fields.
- * Note: while we might have accepted embedded NUL bytes from
- * above, the unescape helpers here will stop at the first one
- * it encounters.
- */
- e->size = string_unescape_inplace(e->magic, UNESCAPE_HEX);
- if (e->mask &&
- string_unescape_inplace(e->mask, UNESCAPE_HEX) != e->size)
- goto einval;
- if (e->size > BINPRM_BUF_SIZE ||
- BINPRM_BUF_SIZE - e->size < e->offset)
- goto einval;
- pr_debug("register: magic/mask length: %i\n", e->size);
- if (USE_DEBUG) {
- print_hex_dump_bytes(
- KBUILD_MODNAME ": register: magic[decoded]: ",
- DUMP_PREFIX_NONE, e->magic, e->size);
-
- if (e->mask) {
- int i;
- char *masked = kmalloc(e->size, GFP_KERNEL_ACCOUNT);
-
- print_hex_dump_bytes(
- KBUILD_MODNAME ": register: mask[decoded]: ",
- DUMP_PREFIX_NONE, e->mask, e->size);
-
- if (masked) {
- for (i = 0; i < e->size; ++i)
- masked[i] = e->magic[i] & e->mask[i];
- print_hex_dump_bytes(
- KBUILD_MODNAME ": register: magic[masked]: ",
- DUMP_PREFIX_NONE, masked, e->size);
-
- kfree(masked);
- }
- }
- }
- } else {
- /* Handle the 'E' (extension) format. */
-
- /* Skip the 'offset' field. */
- p = strchr(p, del);
- if (!p)
- goto einval;
- *p++ = '\0';
-
- /* Parse the 'magic' field. */
- e->magic = p;
- p = strchr(p, del);
- if (!p)
- goto einval;
- *p++ = '\0';
- if (!e->magic[0] || strchr(e->magic, '/'))
- goto einval;
- pr_debug("register: extension: {%s}\n", e->magic);
-
- /* Skip the 'mask' field. */
- p = strchr(p, del);
- if (!p)
- goto einval;
- *p++ = '\0';
- }
+ return ERR_PTR(-EINVAL);
+
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags))
+ p = parse_bpf_fields(e, p, del);
+ else if (test_bit(MISC_FMT_MAGIC_BIT, &e->flags))
+ p = parse_magic_fields(e, p, del);
+ else
+ p = parse_extension_fields(e, p, del);
+ if (!p)
+ return ERR_PTR(-EINVAL);
/* Parse the 'interpreter' field. */
e->interpreter = p;
p = strchr(p, del);
if (!p)
- goto einval;
+ return ERR_PTR(-EINVAL);
*p++ = '\0';
- if (!e->interpreter[0])
- goto einval;
- pr_debug("register: interpreter: {%s}\n", e->interpreter);
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags)) {
+ /* The 'interpreter' field carries the handler name. */
+ e->bpf_ops_name = e->interpreter;
+ e->interpreter = NULL;
+ if (!e->bpf_ops_name[0])
+ return ERR_PTR(-EINVAL);
+ pr_debug("register: bpf handler: {%s}\n", e->bpf_ops_name);
+ } else if (!e->interpreter[0]) {
+ return ERR_PTR(-EINVAL);
+ } else {
+ pr_debug("register: interpreter: {%s}\n", e->interpreter);
+ }
/* Parse the 'flags' field. */
p = check_special_flags(p, e);
if (*p == '\n')
p++;
if (p != buf + count)
- goto einval;
-
- return e;
+ return ERR_PTR(-EINVAL);
-out:
- return ERR_PTR(err);
+ /*
+ * A bpf handler decides the invocation flags per exec with
+ * bpf_binprm_set_flags() rather than fixing them at registration, so a
+ * 'B' entry carries no flags: 'P', 'C' and 'O' become per-exec choices
+ * and 'F' (pre-open a fixed interpreter) is meaningless for it.
+ */
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags) &&
+ (e->flags & (MISC_FMT_PRESERVE_ARGV0 | MISC_FMT_OPEN_BINARY |
+ MISC_FMT_CREDENTIALS | MISC_FMT_OPEN_FILE)))
+ return ERR_PTR(-EINVAL);
-efault:
- kfree(e);
- return ERR_PTR(-EFAULT);
-einval:
- kfree(e);
- return ERR_PTR(-EINVAL);
+ return no_free_ptr(e);
}
+/* Commands accepted by the /status and /<entry> files. */
+enum bm_command {
+ BM_CMD_IGNORE, /* empty write */
+ BM_CMD_DISABLE, /* "0" */
+ BM_CMD_ENABLE, /* "1" */
+ BM_CMD_REMOVE, /* "-1" */
+};
+
/*
- * Set status of entry/binfmt_misc:
- * '1' enables, '0' disables and '-1' clears entry/binfmt_misc
+ * Parse what userspace wrote to /status or an entry file: '1' enables,
+ * '0' disables and '-1' removes the entry or all entries.
*/
static int parse_command(const char __user *buffer, size_t count)
{
@@ -560,62 +707,69 @@ static int parse_command(const char __user *buffer, size_t count)
if (copy_from_user(s, buffer, count))
return -EFAULT;
if (!count)
- return 0;
+ return BM_CMD_IGNORE;
if (s[count - 1] == '\n')
count--;
if (count == 1 && s[0] == '0')
- return 1;
+ return BM_CMD_DISABLE;
if (count == 1 && s[0] == '1')
- return 2;
+ return BM_CMD_ENABLE;
if (count == 2 && s[0] == '-' && s[1] == '1')
- return 3;
+ return BM_CMD_REMOVE;
return -EINVAL;
}
/* generic stuff */
-static void entry_status(Node *e, char *page)
+static void bm_seq_hex(struct seq_file *m, const u8 *data, int size)
{
- char *dp = page;
- const char *status = "disabled";
+ for (int i = 0; i < size; i++)
+ seq_printf(m, "%02x", data[i]);
+}
- if (test_bit(Enabled, &e->flags))
- status = "enabled";
+static int bm_entry_show(struct seq_file *m, void *unused)
+{
+ struct binfmt_misc_entry *e = m->private;
- if (!VERBOSE_STATUS) {
- sprintf(page, "%s\n", status);
- return;
- }
+ if (test_bit(MISC_FMT_ENABLED_BIT, &e->flags))
+ seq_puts(m, "enabled\n");
+ else
+ seq_puts(m, "disabled\n");
- dp += sprintf(dp, "%s\ninterpreter %s\n", status, e->interpreter);
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags))
+ seq_printf(m, "bpf %s\n", e->bpf_ops->name);
+ else
+ seq_printf(m, "interpreter %s\n", e->interpreter);
/* print the special flags */
- dp += sprintf(dp, "flags: ");
+ seq_puts(m, "flags: ");
if (e->flags & MISC_FMT_PRESERVE_ARGV0)
- *dp++ = 'P';
+ seq_putc(m, 'P');
if (e->flags & MISC_FMT_OPEN_BINARY)
- *dp++ = 'O';
+ seq_putc(m, 'O');
if (e->flags & MISC_FMT_CREDENTIALS)
- *dp++ = 'C';
+ seq_putc(m, 'C');
if (e->flags & MISC_FMT_OPEN_FILE)
- *dp++ = 'F';
- *dp++ = '\n';
+ seq_putc(m, 'F');
+ seq_putc(m, '\n');
- if (!test_bit(Magic, &e->flags)) {
- sprintf(dp, "extension .%s\n", e->magic);
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags)) {
+ /* The program does the matching. */
+ } else if (!test_bit(MISC_FMT_MAGIC_BIT, &e->flags)) {
+ seq_printf(m, "extension .%s\n", e->magic);
} else {
- dp += sprintf(dp, "offset %i\nmagic ", e->offset);
- dp = bin2hex(dp, e->magic, e->size);
+ seq_printf(m, "offset %i\nmagic ", e->offset);
+ bm_seq_hex(m, e->magic, e->size);
if (e->mask) {
- dp += sprintf(dp, "\nmask ");
- dp = bin2hex(dp, e->mask, e->size);
+ seq_puts(m, "\nmask ");
+ bm_seq_hex(m, e->mask, e->size);
}
- *dp++ = '\n';
- *dp = '\0';
+ seq_putc(m, '\n');
}
+ return 0;
}
-static struct inode *bm_get_inode(struct super_block *sb, int mode)
+static struct inode *bm_get_inode(struct super_block *sb, umode_t mode)
{
struct inode *inode = new_inode(sb);
@@ -651,14 +805,14 @@ static struct binfmt_misc *i_binfmt_misc(struct inode *inode)
* entry is removed or the filesystem is unmounted and the super block is
* shutdown.
*
- * If the ->evict call was not caused by a super block shutdown but by a write
- * to remove the entry or all entries via bm_{entry,status}_write() the entry
- * will have already been removed from the list. We keep the list_empty() check
- * to make that explicit.
+ * If the ->evict call was not caused by a super block shutdown but by
+ * removing the entry via bm_{entry,status}_write() or unlink(2) the entry
+ * will have already been removed from the list. We keep the hlist_unhashed()
+ * check to make that explicit.
*/
static void bm_evict_inode(struct inode *inode)
{
- Node *e = inode->i_private;
+ struct binfmt_misc_entry *e = inode->i_private;
clear_inode(inode);
@@ -666,89 +820,132 @@ static void bm_evict_inode(struct inode *inode)
struct binfmt_misc *misc;
misc = i_binfmt_misc(inode);
- write_lock(&misc->entries_lock);
- if (!list_empty(&e->list))
- list_del_init(&e->list);
- write_unlock(&misc->entries_lock);
+ spin_lock(&misc->entries_lock);
+ if (!hlist_unhashed(&e->node))
+ hlist_del_init_rcu(&e->node);
+ spin_unlock(&misc->entries_lock);
put_binfmt_handler(e);
}
}
/**
+ * unlink_binfmt_handler - unhash a binary type handler
+ * @misc: handle to binfmt_misc instance
+ * @e: binary type handler to unhash
+ *
+ * Adding and removing entries via bm_{entry,register,status}_write() and
+ * unlink(2) happens under the exclusively held inode lock of the root
+ * dentry keeping the list stable for writers. load_misc_binary() walks it
+ * concurrently under SRCU. The entries_lock is only held around the actual
+ * unlink to serialize against bm_evict_inode() which unlinks entries
+ * during umount without holding the root inode lock.
+ */
+static void unlink_binfmt_handler(struct binfmt_misc *misc,
+ struct binfmt_misc_entry *e)
+{
+ spin_lock(&misc->entries_lock);
+ hlist_del_init_rcu(&e->node);
+ spin_unlock(&misc->entries_lock);
+}
+
+/**
* remove_binfmt_handler - remove a binary type handler
* @misc: handle to binfmt_misc instance
* @e: binary type handler to remove
*
* Remove a binary type handler from the list of binary type handlers and
- * remove its associated dentry. This is called from
- * binfmt_{entry,status}_write(). In the future, we might want to think about
- * adding a proper ->unlink() method to binfmt_misc instead of forcing caller's
- * to use writes to files in order to delete binary type handlers. But it has
- * worked for so long that it's not a pressing issue.
+ * remove its associated dentry.
*/
-static void remove_binfmt_handler(struct binfmt_misc *misc, Node *e)
+static void remove_binfmt_handler(struct binfmt_misc *misc,
+ struct binfmt_misc_entry *e)
{
- write_lock(&misc->entries_lock);
- list_del_init(&e->list);
- write_unlock(&misc->entries_lock);
+ unlink_binfmt_handler(misc, e);
locked_recursive_removal(e->dentry, NULL);
}
-/* /<entry> */
+/* Remove @e unless it was already removed. */
+static void bm_remove_entry(struct binfmt_misc_entry *e, struct super_block *sb)
+{
+ struct inode *root = d_inode(sb->s_root);
-static ssize_t
-bm_entry_read(struct file *file, char __user *buf, size_t nbytes, loff_t *ppos)
+ inode_lock_nested(root, I_MUTEX_PARENT);
+ if (!hlist_unhashed(&e->node))
+ remove_binfmt_handler(i_binfmt_misc(root), e);
+ inode_unlock(root);
+}
+
+/* Remove all entries of the binfmt_misc instance @misc belonging to @sb. */
+static void bm_remove_all_entries(struct binfmt_misc *misc,
+ struct super_block *sb)
{
- Node *e = file_inode(file)->i_private;
- ssize_t res;
- char *page;
+ struct inode *root = d_inode(sb->s_root);
+ struct binfmt_misc_entry *e;
+ struct hlist_node *next;
+
+ inode_lock_nested(root, I_MUTEX_PARENT);
+ hlist_for_each_entry_safe(e, next, &misc->entries, node)
+ remove_binfmt_handler(misc, e);
+ inode_unlock(root);
+}
- page = kmalloc(PAGE_SIZE, GFP_KERNEL);
- if (!page)
- return -ENOMEM;
+/**
+ * bm_unlink - remove a binary type handler via unlink(2)
+ * @dir: inode of the root directory
+ * @dentry: entry file to remove
+ *
+ * Removing the entry file removes its binary type handler, exactly like
+ * writing -1 to it does. The status and register control files can't be
+ * removed. The VFS calls this with the root inode lock held which
+ * serializes against the write based add and remove paths.
+ */
+static int bm_unlink(struct inode *dir, struct dentry *dentry)
+{
+ struct binfmt_misc_entry *e = d_inode(dentry)->i_private;
+
+ if (!e)
+ return -EPERM;
+
+ unlink_binfmt_handler(i_binfmt_misc(dir), e);
+ return simple_unlink(dir, dentry);
+}
+
+static const struct inode_operations bm_dir_inode_operations = {
+ .lookup = simple_lookup,
+ .unlink = bm_unlink,
+};
+
+/* /<entry> */
- entry_status(e, page);
+static int bm_entry_open(struct inode *inode, struct file *file)
+{
+ int ret;
- res = simple_read_from_buffer(buf, nbytes, ppos, page, strlen(page));
+ ret = single_open(file, bm_entry_show, inode->i_private);
+ if (ret)
+ return ret;
- kfree(page);
- return res;
+ /* seq_open() clears FMODE_PWRITE, bm_entry_write() takes any offset */
+ if (file->f_mode & FMODE_WRITE)
+ file->f_mode |= FMODE_PWRITE;
+ return 0;
}
static ssize_t bm_entry_write(struct file *file, const char __user *buffer,
size_t count, loff_t *ppos)
{
struct inode *inode = file_inode(file);
- Node *e = inode->i_private;
+ struct binfmt_misc_entry *e = inode->i_private;
int res = parse_command(buffer, count);
switch (res) {
- case 1:
- /* Disable this handler. */
- clear_bit(Enabled, &e->flags);
+ case BM_CMD_DISABLE:
+ clear_bit(MISC_FMT_ENABLED_BIT, &e->flags);
break;
- case 2:
- /* Enable this handler. */
- set_bit(Enabled, &e->flags);
+ case BM_CMD_ENABLE:
+ set_bit(MISC_FMT_ENABLED_BIT, &e->flags);
break;
- case 3:
- /* Delete this handler. */
- inode = d_inode(inode->i_sb->s_root);
- inode_lock_nested(inode, I_MUTEX_PARENT);
-
- /*
- * In order to add new element or remove elements from the list
- * via bm_{entry,register,status}_write() inode_lock() on the
- * root inode must be held.
- * The lock is exclusive ensuring that the list can't be
- * modified. Only load_misc_binary() can access but does so
- * read-only. So we only need to take the write lock when we
- * actually remove the entry from the list.
- */
- if (!list_empty(&e->list))
- remove_binfmt_handler(i_binfmt_misc(inode), e);
-
- inode_unlock(inode);
+ case BM_CMD_REMOVE:
+ bm_remove_entry(e, inode->i_sb);
break;
default:
return res;
@@ -758,15 +955,17 @@ static ssize_t bm_entry_write(struct file *file, const char __user *buffer,
}
static const struct file_operations bm_entry_operations = {
- .read = bm_entry_read,
+ .open = bm_entry_open,
+ .read = seq_read,
.write = bm_entry_write,
- .llseek = default_llseek,
+ .llseek = seq_lseek,
+ .release = single_release,
};
/* /register */
/* add to filesystem */
-static int add_entry(Node *e, struct super_block *sb)
+static int add_entry(struct binfmt_misc_entry *e, struct super_block *sb)
{
struct dentry *dentry = simple_start_creating(sb->s_root, e->name);
struct inode *inode;
@@ -788,9 +987,9 @@ static int add_entry(Node *e, struct super_block *sb)
d_make_persistent(dentry, inode);
misc = i_binfmt_misc(inode);
- write_lock(&misc->entries_lock);
- list_add(&e->list, &misc->entries);
- write_unlock(&misc->entries_lock);
+ spin_lock(&misc->entries_lock);
+ hlist_add_head_rcu(&e->node, &misc->entries);
+ spin_unlock(&misc->entries_lock);
simple_done_creating(dentry);
return 0;
}
@@ -798,16 +997,24 @@ static int add_entry(Node *e, struct super_block *sb)
static ssize_t bm_register_write(struct file *file, const char __user *buffer,
size_t count, loff_t *ppos)
{
- Node *e;
+ struct binfmt_misc_entry *e __free(kfree) = NULL;
struct super_block *sb = file_inode(file)->i_sb;
- int err = 0;
struct file *f = NULL;
+ int err;
e = create_entry(buffer, count);
-
if (IS_ERR(e))
return PTR_ERR(e);
+ if (test_bit(MISC_FMT_BPF_BIT, &e->flags)) {
+ e->bpf_ops = binfmt_misc_get_ops(sb->s_user_ns, e->bpf_ops_name);
+ if (!e->bpf_ops) {
+ pr_notice("register: no bpf handler named %s\n",
+ e->bpf_ops_name);
+ return -ENOENT;
+ }
+ }
+
if (e->flags & MISC_FMT_OPEN_FILE) {
/*
* Now that we support unprivileged binfmt_misc mounts make
@@ -821,7 +1028,6 @@ static ssize_t bm_register_write(struct file *file, const char __user *buffer,
if (IS_ERR(f)) {
pr_notice("register: failed to install interpreter file %s\n",
e->interpreter);
- kfree(e);
return PTR_ERR(f);
}
e->interp_file = f;
@@ -833,9 +1039,13 @@ static ssize_t bm_register_write(struct file *file, const char __user *buffer,
exe_file_allow_write_access(f);
filp_close(f, NULL);
}
- kfree(e);
+ if (e->bpf_ops)
+ binfmt_misc_put_ops(e->bpf_ops);
return err;
}
+
+ /* The entry is owned by its inode now. */
+ retain_and_null_ptr(e);
return count;
}
@@ -850,10 +1060,10 @@ static ssize_t
bm_status_read(struct file *file, char __user *buf, size_t nbytes, loff_t *ppos)
{
struct binfmt_misc *misc;
- char *s;
+ const char *s;
misc = i_binfmt_misc(file_inode(file));
- s = misc->enabled ? "enabled\n" : "disabled\n";
+ s = READ_ONCE(misc->enabled) ? "enabled\n" : "disabled\n";
return simple_read_from_buffer(buf, nbytes, ppos, s, strlen(s));
}
@@ -862,37 +1072,17 @@ static ssize_t bm_status_write(struct file *file, const char __user *buffer,
{
struct binfmt_misc *misc;
int res = parse_command(buffer, count);
- Node *e, *next;
- struct inode *inode;
misc = i_binfmt_misc(file_inode(file));
switch (res) {
- case 1:
- /* Disable all handlers. */
- misc->enabled = false;
+ case BM_CMD_DISABLE:
+ WRITE_ONCE(misc->enabled, false);
break;
- case 2:
- /* Enable all handlers. */
- misc->enabled = true;
+ case BM_CMD_ENABLE:
+ WRITE_ONCE(misc->enabled, true);
break;
- case 3:
- /* Delete all handlers. */
- inode = d_inode(file_inode(file)->i_sb->s_root);
- inode_lock_nested(inode, I_MUTEX_PARENT);
-
- /*
- * In order to add new element or remove elements from the list
- * via bm_{entry,register,status}_write() inode_lock() on the
- * root inode must be held.
- * The lock is exclusive ensuring that the list can't be
- * modified. Only load_misc_binary() can access but does so
- * read-only. So we only need to take the write lock when we
- * actually remove the entry from the list.
- */
- list_for_each_entry_safe(e, next, &misc->entries, list)
- remove_binfmt_handler(misc, e);
-
- inode_unlock(inode);
+ case BM_CMD_REMOVE:
+ bm_remove_all_entries(misc, file_inode(file)->i_sb);
break;
default:
return res;
@@ -917,7 +1107,7 @@ static void bm_put_super(struct super_block *sb)
put_user_ns(user_ns);
}
-static const struct super_operations s_ops = {
+static const struct super_operations bm_super_ops = {
.statfs = simple_statfs,
.evict_inode = bm_evict_inode,
.put_super = bm_put_super,
@@ -964,10 +1154,10 @@ static int bm_fill_super(struct super_block *sb, struct fs_context *fc)
if (!misc)
return -ENOMEM;
- INIT_LIST_HEAD(&misc->entries);
- rwlock_init(&misc->entries_lock);
+ INIT_HLIST_HEAD(&misc->entries);
+ spin_lock_init(&misc->entries_lock);
- /* Pairs with smp_load_acquire() in load_binfmt_misc(). */
+ /* Pairs with smp_load_acquire() in current_binfmt_misc(). */
smp_store_release(&user_ns->binfmt_misc, misc);
}
@@ -982,12 +1172,15 @@ static int bm_fill_super(struct super_block *sb, struct fs_context *fc)
* someone mounts binfmt_misc for the first time or again we simply
* reset ->enabled to true.
*/
- misc->enabled = true;
+ WRITE_ONCE(misc->enabled, true);
err = simple_fill_super(sb, BINFMTFS_MAGIC, bm_files);
- if (!err)
- sb->s_op = &s_ops;
- return err;
+ if (err)
+ return err;
+
+ sb->s_op = &bm_super_ops;
+ d_inode(sb->s_root)->i_op = &bm_dir_inode_operations;
+ return 0;
}
static void bm_free(struct fs_context *fc)
@@ -1038,6 +1231,8 @@ static void __exit exit_misc_binfmt(void)
{
unregister_binfmt(&misc_format);
unregister_filesystem(&bm_fs_type);
+ /* Flush pending bm_entry_free_rcu() callbacks before the text goes. */
+ srcu_barrier(&bm_entries_srcu);
}
core_initcall(init_misc_binfmt);
diff --git a/fs/binfmt_misc_bpf.c b/fs/binfmt_misc_bpf.c
new file mode 100644
index 000000000000..65a3b8313fbe
--- /dev/null
+++ b/fs/binfmt_misc_bpf.c
@@ -0,0 +1,354 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * BPF-backed binary type handlers for binfmt_misc.
+ *
+ * A handler is a struct binfmt_misc_ops struct_ops map. Loading and
+ * registering it makes the handler available under its name in the user
+ * namespace it was registered in. A binfmt_misc 'B' entry activates it:
+ *
+ * echo ':entry:B::::<handler-name>:' > <binfmt_misc>/register
+ */
+
+#include <linux/binfmt_misc.h>
+#include <linux/binfmts.h>
+#include <linux/bpf.h>
+#include <linux/bpf_verifier.h>
+#include <linux/btf.h>
+#include <linux/btf_ids.h>
+#include <linux/cred.h>
+#include <linux/init.h>
+#include <linux/limits.h>
+#include <linux/slab.h>
+#include <linux/spinlock.h>
+#include <linux/string.h>
+#include <linux/user_namespace.h>
+
+struct bm_bpf_ops_reg {
+ struct list_head list;
+ const struct binfmt_misc_ops *ops;
+ struct bpf_link *link;
+ struct user_namespace *user_ns;
+};
+
+static DEFINE_SPINLOCK(bm_bpf_ops_lock);
+static LIST_HEAD(bm_bpf_ops_list);
+
+static struct bpf_struct_ops bpf_binfmt_misc_ops;
+
+static struct bm_bpf_ops_reg *bm_bpf_ops_find(const struct user_namespace *user_ns,
+ const char *name)
+{
+ struct bm_bpf_ops_reg *reg;
+
+ lockdep_assert_held(&bm_bpf_ops_lock);
+
+ list_for_each_entry(reg, &bm_bpf_ops_list, list) {
+ if (reg->user_ns == user_ns && !strcmp(reg->ops->name, name))
+ return reg;
+ }
+ return NULL;
+}
+
+/**
+ * binfmt_misc_get_ops - look up a bpf binary type handler by name
+ * @user_ns: user namespace of the binfmt_misc instance
+ * @name: name the handler was registered under
+ *
+ * Search @user_ns and its ancestors for a handler named @name, mirroring
+ * the instance lookup in current_binfmt_misc(). The returned handler stays
+ * callable until binfmt_misc_put_ops() even if the backing struct_ops map
+ * is detached or deleted in the meantime.
+ *
+ * Return: the handler on success, NULL on failure
+ */
+const struct binfmt_misc_ops *binfmt_misc_get_ops(struct user_namespace *user_ns,
+ const char *name)
+{
+ const struct user_namespace *ns;
+ struct bm_bpf_ops_reg *reg;
+
+ guard(spinlock)(&bm_bpf_ops_lock);
+
+ for (ns = user_ns; ns; ns = ns->parent) {
+ reg = bm_bpf_ops_find(ns, name);
+ if (!reg)
+ continue;
+ if (!bpf_struct_ops_get(reg->ops))
+ return NULL;
+ return reg->ops;
+ }
+ return NULL;
+}
+
+void binfmt_misc_put_ops(const struct binfmt_misc_ops *ops)
+{
+ bpf_struct_ops_put(ops);
+}
+
+bool bpf_prog_is_binfmt_misc_ops(const struct bpf_prog *prog)
+{
+ return prog->type == BPF_PROG_TYPE_STRUCT_OPS &&
+ prog->aux->st_ops == &bpf_binfmt_misc_ops;
+}
+
+__bpf_kfunc_start_defs();
+
+/**
+ * bpf_binprm_set_interp - select the interpreter for the current exec
+ * @bprm: binary that is being executed
+ * @path: absolute path to the interpreter
+ * @path__sz: size of the @path buffer, including the terminating NUL
+ *
+ * To be called from the load program of a struct binfmt_misc_ops handler
+ * before returning zero; the verifier rejects the call from any other
+ * program, including the handler's own match program. The path is opened
+ * with the credentials of the task doing the exec after the program
+ * returns.
+ *
+ * Return: 0 on success, a negative errno on failure
+ */
+__bpf_kfunc int bpf_binprm_set_interp(struct linux_binprm *bprm,
+ const char *path, size_t path__sz)
+{
+ size_t len;
+ char *interp;
+
+ if (!path__sz)
+ return -EINVAL;
+ len = strnlen(path, path__sz);
+ if (len == path__sz)
+ return -EINVAL;
+ if (path[0] != '/')
+ return -EINVAL;
+ if (len >= PATH_MAX)
+ return -ENAMETOOLONG;
+
+ interp = kmemdup_nul(path, len, GFP_KERNEL);
+ if (!interp)
+ return -ENOMEM;
+
+ kfree(bprm->bpf_interp);
+ bprm->bpf_interp = interp;
+ return 0;
+}
+
+/**
+ * bpf_binprm_set_interp_arg - set a single argument for the interpreter
+ * @bprm: binary that is being executed
+ * @arg: argument to pass to the interpreter
+ * @arg__sz: size of the @arg buffer, including the terminating NUL
+ *
+ * To be called from the load program of a struct binfmt_misc_ops handler. The
+ * argument is passed to the interpreter ahead of the binary, mirroring the
+ * single optional argument of a #! interpreter line. Calling it again
+ * replaces the argument.
+ *
+ * Return: 0 on success, a negative errno on failure
+ */
+__bpf_kfunc int bpf_binprm_set_interp_arg(struct linux_binprm *bprm,
+ const char *arg, size_t arg__sz)
+{
+ size_t len;
+ char *val;
+
+ if (!arg__sz)
+ return -EINVAL;
+ len = strnlen(arg, arg__sz);
+ if (len == arg__sz)
+ return -EINVAL;
+ if (!len)
+ return -EINVAL;
+
+ val = kmemdup_nul(arg, len, GFP_KERNEL);
+ if (!val)
+ return -ENOMEM;
+
+ kfree(bprm->bpf_interp_arg);
+ bprm->bpf_interp_arg = val;
+ return 0;
+}
+
+/**
+ * bpf_binprm_set_flags - choose the interpreter invocation flags for this exec
+ * @bprm: binary that is being executed
+ * @flags: an OR of enum bpf_binprm_flags values
+ *
+ * To be called from the load program of a struct binfmt_misc_ops handler. It
+ * decides per exec what a static entry fixes at registration with the P, C and
+ * O flags: BPF_BINPRM_PRESERVE_ARGV0 keeps the caller's argv[0],
+ * BPF_BINPRM_CREDENTIALS computes credentials from the binary, and
+ * BPF_BINPRM_EXECFD hands the binary to the interpreter through AT_EXECFD.
+ * Calling it again replaces the flags, passing zero clears them again.
+ *
+ * Return: 0 on success, -EINVAL if @flags contains an unknown bit
+ */
+__bpf_kfunc int bpf_binprm_set_flags(struct linux_binprm *bprm,
+ enum bpf_binprm_flags flags)
+{
+ if (flags & ~(BPF_BINPRM_PRESERVE_ARGV0 | BPF_BINPRM_CREDENTIALS |
+ BPF_BINPRM_EXECFD))
+ return -EINVAL;
+
+ bprm->bpf_flags = flags;
+ return 0;
+}
+
+__bpf_kfunc_end_defs();
+
+BTF_KFUNCS_START(bm_bpf_kfunc_ids)
+BTF_ID_FLAGS(func, bpf_binprm_set_interp, KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_binprm_set_interp_arg, KF_SLEEPABLE)
+BTF_ID_FLAGS(func, bpf_binprm_set_flags, KF_SLEEPABLE)
+BTF_KFUNCS_END(bm_bpf_kfunc_ids)
+
+static int bm_bpf_kfunc_filter(const struct bpf_prog *prog, u32 kfunc_id)
+{
+ if (!btf_id_set8_contains(&bm_bpf_kfunc_ids, kfunc_id))
+ return 0;
+ if (prog->type != BPF_PROG_TYPE_STRUCT_OPS)
+ return -EACCES;
+ /* ->st_ops is unset during the cfg pass; enforced once it is set. */
+ if (!prog->aux->st_ops)
+ return 0;
+ /* Only the load program decides how a binary is run. */
+ if (bpf_prog_is_binfmt_misc_ops(prog) &&
+ prog->aux->attach_st_ops_member_off == offsetof(struct binfmt_misc_ops, load))
+ return 0;
+ return -EACCES;
+}
+
+static const struct btf_kfunc_id_set bm_bpf_kfunc_set = {
+ .owner = THIS_MODULE,
+ .set = &bm_bpf_kfunc_ids,
+ .filter = bm_bpf_kfunc_filter,
+};
+
+static bool bm_bpf_ops__match(struct linux_binprm *bprm)
+{
+ return false;
+}
+
+static int bm_bpf_ops__load(struct linux_binprm *bprm)
+{
+ return 0;
+}
+
+static struct binfmt_misc_ops bm_bpf_ops_stubs = {
+ .match = bm_bpf_ops__match,
+ .load = bm_bpf_ops__load,
+};
+
+static int bm_bpf_init(struct btf *btf)
+{
+ return register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
+ &bm_bpf_kfunc_set);
+}
+
+static int bm_bpf_check_member(const struct btf_type *t,
+ const struct btf_member *member,
+ const struct bpf_prog *prog)
+{
+ u32 moff = __btf_member_bit_offset(t, member) / 8;
+
+ switch (moff) {
+ case offsetof(struct binfmt_misc_ops, match):
+ case offsetof(struct binfmt_misc_ops, load):
+ /* Reliable file reads at exec time require sleeping. */
+ if (!prog->sleepable)
+ return -EINVAL;
+ break;
+ }
+ return 0;
+}
+
+static int bm_bpf_init_member(const struct btf_type *t,
+ const struct btf_member *member,
+ void *kdata, const void *udata)
+{
+ const struct binfmt_misc_ops *uops = udata;
+ struct binfmt_misc_ops *ops = kdata;
+ u32 moff = __btf_member_bit_offset(t, member) / 8;
+
+ switch (moff) {
+ case offsetof(struct binfmt_misc_ops, name):
+ if (bpf_obj_name_cpy(ops->name, uops->name,
+ sizeof(ops->name)) <= 0)
+ return -EINVAL;
+ return 1;
+ }
+ return 0;
+}
+
+static int bm_bpf_validate(void *kdata)
+{
+ struct binfmt_misc_ops *ops = kdata;
+
+ if (!ops->match || !ops->load)
+ return -EINVAL;
+ return 0;
+}
+
+static int bm_bpf_reg(void *kdata, struct bpf_link *link)
+{
+ struct binfmt_misc_ops *ops = kdata;
+ struct bm_bpf_ops_reg *reg;
+
+ reg = kzalloc_obj(*reg, GFP_KERNEL_ACCOUNT);
+ if (!reg)
+ return -ENOMEM;
+
+ reg->ops = ops;
+ reg->link = link;
+ reg->user_ns = get_user_ns(current_user_ns());
+
+ guard(spinlock)(&bm_bpf_ops_lock);
+
+ if (bm_bpf_ops_find(reg->user_ns, ops->name)) {
+ put_user_ns(reg->user_ns);
+ kfree(reg);
+ return -EEXIST;
+ }
+
+ list_add(&reg->list, &bm_bpf_ops_list);
+ return 0;
+}
+
+static void bm_bpf_unreg(void *kdata, struct bpf_link *link)
+{
+ struct bm_bpf_ops_reg *reg;
+
+ guard(spinlock)(&bm_bpf_ops_lock);
+
+ list_for_each_entry(reg, &bm_bpf_ops_list, list) {
+ if (reg->ops == kdata && reg->link == link) {
+ list_del(&reg->list);
+ put_user_ns(reg->user_ns);
+ kfree(reg);
+ return;
+ }
+ }
+}
+
+static const struct bpf_verifier_ops bm_bpf_verifier_ops = {
+ .get_func_proto = bpf_base_func_proto,
+ .is_valid_access = bpf_tracing_btf_ctx_access,
+};
+
+static struct bpf_struct_ops bpf_binfmt_misc_ops = {
+ .verifier_ops = &bm_bpf_verifier_ops,
+ .init = bm_bpf_init,
+ .check_member = bm_bpf_check_member,
+ .init_member = bm_bpf_init_member,
+ .validate = bm_bpf_validate,
+ .reg = bm_bpf_reg,
+ .unreg = bm_bpf_unreg,
+ .cfi_stubs = &bm_bpf_ops_stubs,
+ .name = "binfmt_misc_ops",
+ .owner = THIS_MODULE,
+};
+
+static int __init bm_bpf_struct_ops_init(void)
+{
+ return register_bpf_struct_ops(&bpf_binfmt_misc_ops, binfmt_misc_ops);
+}
+late_initcall(bm_bpf_struct_ops_init);
diff --git a/fs/bpf_fs_kfuncs.c b/fs/bpf_fs_kfuncs.c
index f1863a891db6..6cb877267978 100644
--- a/fs/bpf_fs_kfuncs.c
+++ b/fs/bpf_fs_kfuncs.c
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: GPL-2.0
/* Copyright (c) 2024 Google LLC. */
+#include <linux/binfmt_misc.h>
#include <linux/bpf.h>
#include <linux/bpf_lsm.h>
#include <linux/btf.h>
@@ -11,6 +12,7 @@
#include <linux/file.h>
#include <linux/kernfs.h>
#include <linux/mm.h>
+#include <linux/net.h>
#include <linux/xattr.h>
__bpf_kfunc_start_defs();
@@ -359,6 +361,39 @@ __bpf_kfunc int bpf_cgroup_read_xattr(struct cgroup *cgroup, const char *name__s
}
#endif /* CONFIG_CGROUPS */
+#ifdef CONFIG_NET
+/**
+ * bpf_sock_read_xattr - read xattr of a socket's inode in sockfs
+ * @sock: socket to get xattr from
+ * @name__str: name of the xattr
+ * @value_p: output buffer of the xattr value
+ *
+ * Get xattr *name__str* of *sock* and store the output in *value_p*.
+ *
+ * For security reasons, only *name__str* with prefix "user." is allowed.
+ *
+ * Return: length of the xattr value on success, a negative value on error.
+ */
+__bpf_kfunc int bpf_sock_read_xattr(struct socket *sock, const char *name__str,
+ struct bpf_dynptr *value_p)
+{
+ struct bpf_dynptr_kern *value_ptr = (struct bpf_dynptr_kern *)value_p;
+ u32 value_len;
+ void *value;
+
+ /* Only allow reading "user.*" xattrs */
+ if (strncmp(name__str, XATTR_USER_PREFIX, XATTR_USER_PREFIX_LEN))
+ return -EPERM;
+
+ value_len = __bpf_dynptr_size(value_ptr);
+ value = __bpf_dynptr_data_rw(value_ptr, value_len);
+ if (!value)
+ return -EINVAL;
+
+ return sock_read_xattr(sock, name__str, value, value_len);
+}
+#endif /* CONFIG_NET */
+
/**
* bpf_real_data_inode - get the real inode hosting a file's data
* @file: file to resolve
@@ -390,12 +425,30 @@ BTF_ID_FLAGS(func, bpf_get_file_xattr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_set_dentry_xattr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_remove_dentry_xattr, KF_SLEEPABLE)
BTF_ID_FLAGS(func, bpf_real_data_inode, KF_SLEEPABLE | KF_RET_NULL)
+#ifdef CONFIG_NET
+BTF_ID_FLAGS(func, bpf_sock_read_xattr, KF_RCU)
+#endif
BTF_KFUNCS_END(bpf_fs_kfunc_set_ids)
+/* Side-effecting kfuncs that stay exclusive to LSM programs. */
+BTF_SET_START(bpf_fs_kfunc_lsm_only_ids)
+BTF_ID(func, bpf_set_dentry_xattr)
+BTF_ID(func, bpf_remove_dentry_xattr)
+BTF_SET_END(bpf_fs_kfunc_lsm_only_ids)
+
static int bpf_fs_kfuncs_filter(const struct bpf_prog *prog, u32 kfunc_id)
{
- if (!btf_id_set8_contains(&bpf_fs_kfunc_set_ids, kfunc_id) ||
- prog->type == BPF_PROG_TYPE_LSM)
+ if (!btf_id_set8_contains(&bpf_fs_kfunc_set_ids, kfunc_id))
+ return 0;
+ if (prog->type == BPF_PROG_TYPE_LSM)
+ return 0;
+ if (prog->type != BPF_PROG_TYPE_STRUCT_OPS)
+ return -EACCES;
+ /* ->st_ops is unset during the cfg pass; enforced once it is set. */
+ if (!prog->aux->st_ops)
+ return 0;
+ if (bpf_prog_is_binfmt_misc_ops(prog) &&
+ !btf_id_set_contains(&bpf_fs_kfunc_lsm_only_ids, kfunc_id))
return 0;
return -EACCES;
}
@@ -438,7 +491,13 @@ static const struct btf_kfunc_id_set bpf_fs_kfunc_set = {
static int __init bpf_fs_kfuncs_init(void)
{
- return register_btf_kfunc_id_set(BPF_PROG_TYPE_LSM, &bpf_fs_kfunc_set);
+ int ret;
+
+ ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_LSM, &bpf_fs_kfunc_set);
+ if (ret || !IS_ENABLED(CONFIG_BINFMT_MISC_BPF))
+ return ret;
+ return register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
+ &bpf_fs_kfunc_set);
}
late_initcall(bpf_fs_kfuncs_init);
diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig
index 9de04c37e11a..43dd8ef763b8 100644
--- a/fs/btrfs/Kconfig
+++ b/fs/btrfs/Kconfig
@@ -1,4 +1,5 @@
# SPDX-License-Identifier: GPL-2.0
+# misc-next marker
config BTRFS_FS
tristate "Btrfs filesystem support"
diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c
index 8def7abb728f..830460a40e86 100644
--- a/fs/btrfs/block-group.c
+++ b/fs/btrfs/block-group.c
@@ -2047,6 +2047,11 @@ static int btrfs_reclaim_block_group(struct btrfs_block_group *bg, int *reclaime
trace_btrfs_reclaim_block_group(bg);
ret = btrfs_relocate_chunk(fs_info, bg->start, false);
+ if (btrfs_is_zoned(fs_info) && ret == -EAGAIN) {
+ btrfs_dec_block_group_ro(bg);
+ btrfs_debug(fs_info, "deferring reclaim of chunk %llu", bg->start);
+ return ret;
+ }
if (ret) {
btrfs_dec_block_group_ro(bg);
btrfs_err(fs_info, "error relocating chunk %llu",
@@ -2113,7 +2118,8 @@ void btrfs_reclaim_block_groups(struct btrfs_fs_info *fs_info, unsigned int limi
spin_unlock(&fs_info->unused_bgs_lock);
ret = btrfs_reclaim_block_group(bg, &reclaimed);
- if (ret && !READ_ONCE(space_info->periodic_reclaim))
+ if ((btrfs_is_zoned(fs_info) && ret == -EAGAIN) ||
+ (ret && !READ_ONCE(space_info->periodic_reclaim)))
btrfs_link_bg_list(bg, &retry_list);
btrfs_put_block_group(bg);
@@ -2624,10 +2630,9 @@ static int fill_dummy_bgs(struct btrfs_fs_info *fs_info)
/* Fill dummy cache as FULL */
bg->length = map->chunk_len;
- bg->flags = map->type;
+ bg->flags = map->on_disk_type;
bg->cached = BTRFS_CACHE_FINISHED;
bg->used = map->chunk_len;
- bg->flags = map->type;
bg->space_info = btrfs_find_space_info(fs_info, bg->flags);
ret = btrfs_add_block_group_cache(bg);
/*
@@ -3916,7 +3921,7 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans,
old_val += num_bytes;
cache->used = old_val;
cache->reserved -= num_bytes;
- cache->reclaim_mark = 0;
+ cache->reclaim_mark = false;
space_info->bytes_reserved -= num_bytes;
space_info->bytes_used += num_bytes;
space_info->disk_used += num_bytes * factor;
diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h
index 790c2d467af5..69d56864d4ba 100644
--- a/fs/btrfs/block-group.h
+++ b/fs/btrfs/block-group.h
@@ -263,6 +263,9 @@ struct btrfs_block_group {
enum btrfs_block_group_size_class size_class:8;
+ /* If set, this blockgroup is not used for allocation between two reclaim sweeps. */
+ bool reclaim_mark;
+
/*
* Number of extents in this block group used for swap files.
* All accesses protected by the spinlock 'lock'.
@@ -281,7 +284,6 @@ struct btrfs_block_group {
struct list_head active_bg_list;
struct work_struct zone_finish_work;
struct extent_buffer *last_eb;
- u64 reclaim_mark;
};
static inline u64 btrfs_block_group_end(const struct btrfs_block_group *block_group)
diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c
index ffb6b52863a7..c62b5148d5ac 100644
--- a/fs/btrfs/compression.c
+++ b/fs/btrfs/compression.c
@@ -651,9 +651,9 @@ struct heuristic_ws {
u8 *sample;
u32 sample_size;
/* Buckets store counters for each byte value */
- struct bucket_item *bucket;
+ struct bucket_item bucket[BUCKET_SIZE];
/* Sorting buffer */
- struct bucket_item *bucket_b;
+ struct bucket_item bucket_b[BUCKET_SIZE];
struct list_head list;
};
@@ -664,8 +664,6 @@ static void free_heuristic_ws(struct list_head *ws)
workspace = list_entry(ws, struct heuristic_ws, list);
kvfree(workspace->sample);
- kfree(workspace->bucket);
- kfree(workspace->bucket_b);
kfree(workspace);
}
@@ -681,14 +679,6 @@ static struct list_head *alloc_heuristic_ws(struct btrfs_fs_info *fs_info)
if (!ws->sample)
goto fail;
- ws->bucket = kzalloc_objs(*ws->bucket, BUCKET_SIZE);
- if (!ws->bucket)
- goto fail;
-
- ws->bucket_b = kzalloc_objs(*ws->bucket_b, BUCKET_SIZE);
- if (!ws->bucket_b)
- goto fail;
-
INIT_LIST_HEAD(&ws->list);
return &ws->list;
fail:
diff --git a/fs/btrfs/defrag.c b/fs/btrfs/defrag.c
index f0c6758b7055..6ec5dd760d42 100644
--- a/fs/btrfs/defrag.c
+++ b/fs/btrfs/defrag.c
@@ -1093,7 +1093,7 @@ next:
struct defrag_target_range *tmp;
list_for_each_entry_safe(entry, tmp, target_list, list) {
- list_del_init(&entry->list);
+ list_del(&entry->list);
kfree(entry);
}
}
@@ -1130,20 +1130,15 @@ static_assert(PAGE_ALIGNED(CLUSTER_SIZE));
*
* - Extent bits are locked
*/
-static int defrag_one_locked_target(struct btrfs_inode *inode,
- struct defrag_target_range *target,
- struct folio **folios, int nr_pages,
- struct extent_state **cached_state)
+static void defrag_one_locked_target(struct btrfs_inode *inode,
+ struct defrag_target_range *target,
+ struct folio **folios, int nr_pages,
+ struct extent_state **cached_state)
{
struct btrfs_fs_info *fs_info = inode->root->fs_info;
- struct extent_changeset *data_reserved = NULL;
const u64 start = target->start;
const u64 len = target->len;
- int ret = 0;
- ret = btrfs_delalloc_reserve_space(inode, &data_reserved, start, len);
- if (ret < 0)
- return ret;
btrfs_clear_extent_bit(&inode->io_tree, start, start + len - 1,
EXTENT_DELALLOC | EXTENT_DO_ACCOUNTING |
EXTENT_DEFRAG, cached_state);
@@ -1164,10 +1159,6 @@ static int defrag_one_locked_target(struct btrfs_inode *inode,
continue;
btrfs_folio_clamp_set_dirty(fs_info, folio, start, len);
}
- btrfs_delalloc_release_extents(inode, len);
- extent_changeset_free(data_reserved);
-
- return ret;
}
static int defrag_one_range(struct btrfs_inode *inode, u64 start, u32 len,
@@ -1178,11 +1169,13 @@ static int defrag_one_range(struct btrfs_inode *inode, u64 start, u32 len,
struct defrag_target_range *entry;
struct defrag_target_range *tmp;
LIST_HEAD(target_list);
- struct folio **folios;
+ struct folio AUTO_KFREE(*folios);
const u32 sectorsize = inode->root->fs_info->sectorsize;
u64 cur = start;
const unsigned int nr_pages = ((start + len - 1) >> PAGE_SHIFT) -
(start >> PAGE_SHIFT) + 1;
+ struct extent_changeset *data_reserved = NULL;
+ u64 last_defrag_end = start;
int ret = 0;
ASSERT(nr_pages <= CLUSTER_SIZE / PAGE_SIZE);
@@ -1192,6 +1185,20 @@ static int defrag_one_range(struct btrfs_inode *inode, u64 start, u32 len,
if (!folios)
return -ENOMEM;
+ /*
+ * Reserve delalloc space before locking the range and before locking
+ * and dirtying any folios - otherwise we could deadlock, for example
+ * after defrag of one range we dirty folios and keep them locked when
+ * we move to the next range, so reserving delalloc space right before
+ * each range could trigger flushing of delalloc and deadlock on the
+ * extent lock or trigger a transaction commit with flushoncommit, which
+ * can either deadlock on the lock of a folio made dirty in the previous
+ * range or the extent lock.
+ */
+ ret = btrfs_delalloc_reserve_space(inode, &data_reserved, start, len);
+ if (ret < 0)
+ return ret;
+
/* Prepare all pages */
for (int i = 0; cur < start + len && i < nr_pages; i++) {
folios[i] = defrag_prepare_one_folio(inode, cur >> PAGE_SHIFT);
@@ -1225,15 +1232,12 @@ static int defrag_one_range(struct btrfs_inode *inode, u64 start, u32 len,
if (ret < 0)
goto unlock_extent;
- list_for_each_entry(entry, &target_list, list) {
- ret = defrag_one_locked_target(inode, entry, folios, nr_pages,
- &cached_state);
- if (ret < 0)
- break;
- }
-
list_for_each_entry_safe(entry, tmp, &target_list, list) {
- list_del_init(&entry->list);
+ defrag_one_locked_target(inode, entry, folios, nr_pages, &cached_state);
+ if (entry->start > last_defrag_end)
+ btrfs_delalloc_release_space(inode, data_reserved, last_defrag_end,
+ entry->start - last_defrag_end, true);
+ last_defrag_end = entry->start + entry->len;
kfree(entry);
}
unlock_extent:
@@ -1245,7 +1249,12 @@ free_folios:
folio_unlock(folios[i]);
folio_put(folios[i]);
}
- kfree(folios);
+ btrfs_delalloc_release_extents(inode, len);
+ if (last_defrag_end < start + len)
+ btrfs_delalloc_release_space(inode, data_reserved, last_defrag_end,
+ start + len - last_defrag_end, true);
+ extent_changeset_free(data_reserved);
+
return ret;
}
@@ -1310,10 +1319,8 @@ static int defrag_one_cluster(struct btrfs_inode *inode,
inode->root->fs_info->sectorsize_bits;
}
out:
- list_for_each_entry_safe(entry, tmp, &target_list, list) {
- list_del_init(&entry->list);
+ list_for_each_entry_safe(entry, tmp, &target_list, list)
kfree(entry);
- }
if (ret >= 0)
*last_scanned_ret = max(*last_scanned_ret, start + len);
return ret;
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
index 318ddb790429..dc0834f920c3 100644
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -247,8 +247,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
return -EINVAL;
}
- bdev_file = bdev_file_open_by_path(device_path, BLK_OPEN_WRITE,
- fs_info->sb, &fs_holder_ops);
+ /* Unfreezable for the whole replace; see btrfs_dev_replace_start(). */
+ bdev_file = btrfs_open_device_deny_freeze(device_path, fs_info->sb);
if (IS_ERR(bdev_file)) {
btrfs_err(fs_info, "target device %s is invalid!", device_path);
return PTR_ERR(bdev_file);
@@ -327,7 +327,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info,
return 0;
error:
- bdev_fput(bdev_file);
+ /* Undo the open-time freeze deny. */
+ btrfs_release_device_allow_freeze(bdev_file);
return ret;
}
@@ -624,6 +625,15 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info,
if (ret)
return ret;
+ /* Deny the source before mark, so every 'leave' unwinds both denied. */
+ if (src_device->bdev) {
+ ret = bdev_deny_freeze(src_device->bdev);
+ if (ret) {
+ btrfs_destroy_dev_replace_tgtdev(tgt_device, true);
+ return ret;
+ }
+ }
+
ret = mark_block_group_to_copy(fs_info, src_device);
if (ret)
return ret;
@@ -708,7 +718,9 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info,
return ret;
leave:
- btrfs_destroy_dev_replace_tgtdev(tgt_device);
+ if (src_device->bdev)
+ bdev_allow_freeze(src_device->bdev);
+ btrfs_destroy_dev_replace_tgtdev(tgt_device, true);
return ret;
}
@@ -889,6 +901,7 @@ static int btrfs_dev_replace_finishing(struct btrfs_fs_info *fs_info,
*/
ret = btrfs_start_delalloc_roots(fs_info, LONG_MAX, false);
if (ret) {
+ /* Stays started/resumable; keep both denied. */
mutex_unlock(&dev_replace->lock_finishing_cancel_unmount);
return ret;
}
@@ -902,6 +915,7 @@ static int btrfs_dev_replace_finishing(struct btrfs_fs_info *fs_info,
while (1) {
trans = btrfs_start_transaction(root, 0);
if (IS_ERR(trans)) {
+ /* Stays started/resumable; keep both denied. */
mutex_unlock(&dev_replace->lock_finishing_cancel_unmount);
return PTR_ERR(trans);
}
@@ -954,7 +968,10 @@ error:
mutex_unlock(&fs_devices->device_list_mutex);
btrfs_rm_dev_replace_blocked(fs_info);
if (tgt_device)
- btrfs_destroy_dev_replace_tgtdev(tgt_device);
+ btrfs_destroy_dev_replace_tgtdev(tgt_device, true);
+ /* The source stays a member; re-allow freezing it. */
+ if (src_device->bdev)
+ bdev_allow_freeze(src_device->bdev);
btrfs_rm_dev_replace_unblocked(fs_info);
mutex_unlock(&dev_replace->lock_finishing_cancel_unmount);
@@ -1027,6 +1044,8 @@ error:
mutex_unlock(&dev_replace->lock_finishing_cancel_unmount);
+ /* The target is now a member; the source is freed (allow + release). */
+ bdev_allow_freeze(tgt_device->bdev);
btrfs_rm_dev_replace_free_srcdev(src_device);
return 0;
@@ -1155,8 +1174,9 @@ int btrfs_dev_replace_cancel(struct btrfs_fs_info *fs_info)
btrfs_dev_name(src_device), src_device->devid,
btrfs_dev_name(tgt_device));
+ /* A suspended replace never re-denied freezing; do not allow. */
if (tgt_device)
- btrfs_destroy_dev_replace_tgtdev(tgt_device);
+ btrfs_destroy_dev_replace_tgtdev(tgt_device, false);
break;
default:
up_write(&dev_replace->rwsem);
@@ -1186,6 +1206,11 @@ void btrfs_dev_replace_suspend_for_unmount(struct btrfs_fs_info *fs_info)
dev_replace->time_stopped = ktime_get_real_seconds();
dev_replace->item_needs_writeback = 1;
btrfs_info(fs_info, "suspending dev_replace for unmount");
+ /* Reopened freezable next mount; resume re-denies. */
+ if (dev_replace->srcdev && dev_replace->srcdev->bdev)
+ bdev_allow_freeze(dev_replace->srcdev->bdev);
+ if (dev_replace->tgtdev && dev_replace->tgtdev->bdev)
+ bdev_allow_freeze(dev_replace->tgtdev->bdev);
break;
}
@@ -1198,6 +1223,7 @@ int btrfs_resume_dev_replace_async(struct btrfs_fs_info *fs_info)
{
struct task_struct *task;
struct btrfs_dev_replace *dev_replace = &fs_info->dev_replace;
+ int ret = 0;
down_write(&dev_replace->rwsem);
@@ -1241,8 +1267,33 @@ int btrfs_resume_dev_replace_async(struct btrfs_fs_info *fs_info)
return 0;
}
+ /* Re-deny for the resumed replace; stay suspended if frozen now. */
+ if (dev_replace->srcdev->bdev &&
+ bdev_deny_freeze(dev_replace->srcdev->bdev))
+ goto suspend;
+ if (bdev_deny_freeze(dev_replace->tgtdev->bdev)) {
+ if (dev_replace->srcdev->bdev)
+ bdev_allow_freeze(dev_replace->srcdev->bdev);
+ goto suspend;
+ }
+
task = kthread_run(btrfs_dev_replace_kthread, fs_info, "btrfs-devrepl");
- return PTR_ERR_OR_ZERO(task);
+ if (IS_ERR(task)) {
+ bdev_allow_freeze(dev_replace->tgtdev->bdev);
+ if (dev_replace->srcdev->bdev)
+ bdev_allow_freeze(dev_replace->srcdev->bdev);
+ /* Undo the deny and suspend, but still fail the mount. */
+ ret = PTR_ERR(task);
+ goto suspend;
+ }
+ return 0;
+
+suspend:
+ btrfs_exclop_finish(fs_info);
+ down_write(&dev_replace->rwsem);
+ dev_replace->replace_state = BTRFS_IOCTL_DEV_REPLACE_STATE_SUSPENDED;
+ up_write(&dev_replace->rwsem);
+ return ret;
}
static int btrfs_dev_replace_kthread(void *data)
diff --git a/fs/btrfs/direct-io.c b/fs/btrfs/direct-io.c
index 460326d34143..ed1779ccb4de 100644
--- a/fs/btrfs/direct-io.c
+++ b/fs/btrfs/direct-io.c
@@ -14,7 +14,6 @@
#include "ordered-data.h"
struct btrfs_dio_data {
- ssize_t submitted;
loff_t old_isize;
struct extent_changeset *data_reserved;
struct btrfs_ordered_extent *ordered;
@@ -151,7 +150,7 @@ static struct extent_map *btrfs_create_dio_extent(struct btrfs_inode *inode,
if (type != BTRFS_ORDERED_NOCOW) {
em = btrfs_create_io_em(inode, start, file_extent, type);
if (IS_ERR(em))
- goto out;
+ return em;
}
ordered = btrfs_alloc_ordered_extent(inode, start, file_extent,
@@ -168,7 +167,6 @@ static struct extent_map *btrfs_create_dio_extent(struct btrfs_inode *inode,
ASSERT(!dio_data->ordered);
dio_data->ordered = ordered;
}
- out:
return em;
}
@@ -281,17 +279,24 @@ static int btrfs_get_blocks_direct_write(struct extent_map **map,
em2 = btrfs_create_dio_extent(BTRFS_I(inode), dio_data, start,
&file_extent, type);
btrfs_dec_nocow_writers(bg);
- if (type == BTRFS_ORDERED_PREALLOC) {
- btrfs_free_extent_map(em);
- *map = em2;
- em = em2;
- }
-
if (IS_ERR(em2)) {
ret = PTR_ERR(em2);
+ btrfs_free_extent_map(em);
+ *map = NULL;
goto out;
}
+ /*
+ * True NOCOW writes don't need to create a new extent map,
+ * while PREALLOC writes must replace the existing one.
+ */
+ if (em2) {
+ ASSERT(type == BTRFS_ORDERED_PREALLOC);
+ btrfs_free_extent_map(em);
+ *map = em2;
+ em = em2;
+ }
+
dio_data->nocow_done = true;
} else {
/* Our caller expects us to free the input extent map. */
@@ -619,78 +624,81 @@ static int btrfs_dio_iomap_end(struct inode *inode, loff_t pos, loff_t length,
{
struct iomap_iter *iter = container_of(iomap, struct iomap_iter, iomap);
struct btrfs_dio_data *dio_data = iter->private;
- size_t submitted = dio_data->submitted;
const bool write = !!(flags & IOMAP_WRITE);
int ret = 0;
- if (!write && (iomap->type == IOMAP_HOLE)) {
- /* If reading from a hole, unlock and return */
- btrfs_unlock_dio_extent(&BTRFS_I(inode)->io_tree, pos,
- pos + length - 1, NULL);
+ if (!write) {
+ /*
+ * Hole read, nothing is submitted, thus we have to unlock
+ * the whole range.
+ */
+ if (iomap->type == IOMAP_HOLE) {
+ btrfs_unlock_dio_extent(&BTRFS_I(inode)->io_tree, pos,
+ pos + length - 1, NULL);
+ return 0;
+ }
+ /*
+ * Short read, needs to unlock the remaining range, and
+ * return -ENOTBLK so we can later fault in the pages and retry.
+ */
+ if (written < length) {
+ btrfs_unlock_dio_extent(&BTRFS_I(inode)->io_tree, pos + written,
+ pos + length - 1, NULL);
+ return -ENOTBLK;
+ }
+ /* The full range is submitted, endio will do the unlock. */
return 0;
}
- if (submitted < length) {
- pos += submitted;
- length -= submitted;
- if (write) {
- /*
- * Got a short write and have updated the isize, need to
- * revert the isize change.
- *
- * Normally we need to update isize with extent lock hold,
- * but we're safe due to the following factors:
- *
- * - Only a single writer can be enlarging isize
- * Enlarging isize will take the exclusive inode lock.
- *
- * - Buffered readers need to wait for the OE we're holding
- * Buffered readers will lock extent and wait for OE
- * of the folio range, and since page cache is invalidated
- * the OE wait can not be skipped.
- *
- * So here we are safe to revert the isize before
- * finishing the OE, and no reader of the remaining range
- * can see the enlarged size.
- *
- * TODO: Extend the DIO_LOCKED lifespan for direct writes,
- * and only enlarge isize after a successful write.
- */
- if (dio_data->updated_isize) {
- u64 new_isize;
-
- if (submitted == 0)
- new_isize = dio_data->old_isize;
- else
- new_isize = max(dio_data->old_isize, pos);
- i_size_write(inode, new_isize);
- dio_data->updated_isize = false;
- }
- /*
- * We have a short write, if there is any range
- * that is submitted properly, that part will have
- * its own OE split from the original one.
- *
- * So for the OE at dio_data->ordered, it's the part
- * that is not submitted, and should be marked
- * as fully truncated.
- */
- btrfs_mark_ordered_extent_truncated(dio_data->ordered, 0);
- btrfs_finish_ordered_extent(dio_data->ordered,
- pos, length, true);
- } else {
- btrfs_unlock_dio_extent(&BTRFS_I(inode)->io_tree, pos,
- pos + length - 1, NULL);
+ if (written < length) {
+ /*
+ * Got a short write and have updated the i_size, need to revert
+ * the i_size change.
+ *
+ * Normally we need to update i_size with extent lock held, but
+ * we're safe due to the following factors:
+ *
+ * - Only a single writer can be enlarging i_size
+ * Enlarging i_size will take the exclusive inode lock.
+ *
+ * - Buffered readers need to wait for the OE we're holding
+ * Buffered readers will lock extent and wait for OE
+ * of the folio range, and since page cache is invalidated
+ * the OE wait cannot be skipped.
+ *
+ * So here we are safe to revert the isize before finishing the
+ * OE, and no reader of the remaining range can see the enlarged
+ * size.
+ *
+ * TODO: Extend the DIO_LOCKED lifespan for direct writes,
+ * and only enlarge isize after a successful write.
+ */
+ if (dio_data->updated_isize) {
+ u64 new_isize;
+
+ if (written == 0)
+ new_isize = dio_data->old_isize;
+ else
+ new_isize = max(dio_data->old_isize, pos + written);
+ i_size_write(inode, new_isize);
+ dio_data->updated_isize = false;
}
+ /*
+ * We have a short write, if there is any range that is submitted
+ * properly, that part will have its own OE split from the
+ * original one.
+ *
+ * So for the OE at dio_data->ordered, it's the part that is not
+ * submitted, and should be marked as fully truncated.
+ */
+ btrfs_mark_ordered_extent_truncated(dio_data->ordered, 0);
+ btrfs_finish_ordered_extent(dio_data->ordered,
+ pos + written, length - written, true);
ret = -ENOTBLK;
}
- if (write) {
- btrfs_put_ordered_extent(dio_data->ordered);
- dio_data->ordered = NULL;
- }
-
- if (write)
- extent_changeset_free(dio_data->data_reserved);
+ btrfs_put_ordered_extent(dio_data->ordered);
+ dio_data->ordered = NULL;
+ extent_changeset_free(dio_data->data_reserved);
return ret;
}
@@ -772,8 +780,6 @@ static void btrfs_dio_submit_io(const struct iomap_iter *iter, struct bio *bio,
dip->file_offset = file_offset;
dip->bytes = bio->bi_iter.bi_size;
- dio_data->submitted += bio->bi_iter.bi_size;
-
/*
* Check if we are doing a partial write. If we are, we need to split
* the ordered extent to match the submitted bio. Hang on to the
@@ -817,13 +823,41 @@ static ssize_t btrfs_dio_read(struct kiocb *iocb, struct iov_iter *iter,
IOMAP_DIO_PARTIAL | IOMAP_DIO_FSBLOCK_ALIGNED, &data, done_before);
}
+static bool need_stable_write(struct btrfs_inode *inode)
+{
+ const u64 data_profile = btrfs_data_alloc_profile(inode->root->fs_info) &
+ BTRFS_BLOCK_GROUP_PROFILE_MASK;
+
+ /* Data checksum requires stable buffer. */
+ if (!(inode->flags & BTRFS_INODE_NODATASUM))
+ return true;
+ /*
+ * Any profile with mirror/parity will require stable buffer.
+ * Otherwise the mirror may differ from each other.
+ *
+ * Thus only SINGLE and RAID0 doesn't require stable buffer.
+ */
+ if (data_profile != 0 && data_profile != BTRFS_BLOCK_GROUP_RAID0)
+ return true;
+ return false;
+}
+
static struct iomap_dio *btrfs_dio_write(struct kiocb *iocb, struct iov_iter *iter,
size_t done_before)
{
struct btrfs_dio_data data = { 0 };
+ unsigned int dio_flags = IOMAP_DIO_PARTIAL | IOMAP_DIO_FSBLOCK_ALIGNED;
+
+ if (need_stable_write(BTRFS_I(file_inode(iocb->ki_filp)))) {
+ /* For now no support for BOUNCE and NOWAIT direct write. */
+ if (iocb->ki_flags & IOCB_NOWAIT)
+ return ERR_PTR(-EAGAIN);
+
+ dio_flags |= IOMAP_DIO_BOUNCE;
+ }
return __iomap_dio_rw(iocb, iter, &btrfs_dio_iomap_ops, &btrfs_dio_ops,
- IOMAP_DIO_PARTIAL | IOMAP_DIO_FSBLOCK_ALIGNED, &data, done_before);
+ dio_flags, &data, done_before);
}
static ssize_t check_direct_IO(struct btrfs_fs_info *fs_info,
@@ -852,8 +886,6 @@ ssize_t btrfs_direct_write(struct kiocb *iocb, struct iov_iter *from)
ssize_t ret;
unsigned int ilock_flags = 0;
struct iomap_dio *dio;
- const u64 data_profile = btrfs_data_alloc_profile(fs_info) &
- BTRFS_BLOCK_GROUP_PROFILE_MASK;
if (iocb->ki_flags & IOCB_NOWAIT)
ilock_flags |= BTRFS_ILOCK_TRY;
@@ -867,16 +899,6 @@ ssize_t btrfs_direct_write(struct kiocb *iocb, struct iov_iter *from)
if (iocb->ki_pos + iov_iter_count(from) <= i_size_read(inode) && IS_NOSEC(inode))
ilock_flags |= BTRFS_ILOCK_SHARED;
- /*
- * If our data profile has duplication (either extra mirrors or RAID56),
- * we can not trust the direct IO buffer, the content may change during
- * writeback and cause different contents written to different mirrors.
- *
- * Thus only RAID0 and SINGLE can go true zero-copy direct IO.
- */
- if (data_profile != BTRFS_BLOCK_GROUP_RAID0 && data_profile != 0)
- goto buffered;
-
relock:
ret = btrfs_inode_lock(BTRFS_I(inode), ilock_flags);
if (ret < 0)
@@ -917,22 +939,6 @@ relock:
btrfs_inode_unlock(BTRFS_I(inode), ilock_flags);
goto buffered;
}
- /*
- * We can't control the folios being passed in, applications can write
- * to them while a direct IO write is in progress. This means the
- * content might change after we calculated the data checksum.
- * Therefore we can end up storing a checksum that doesn't match the
- * persisted data.
- *
- * To be extra safe and avoid false data checksum mismatch, if the
- * inode requires data checksum, just fallback to buffered IO.
- * For buffered IO we have full control of page cache and can ensure
- * no one is modifying the content during writeback.
- */
- if (!(BTRFS_I(inode)->flags & BTRFS_INODE_NODATASUM)) {
- btrfs_inode_unlock(BTRFS_I(inode), ilock_flags);
- goto buffered;
- }
/*
* The iov_iter can be mapped to the same file range we are writing to.
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index 22c321719e4f..37fc0d6b960d 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -2390,8 +2390,8 @@ short_read:
int btrfs_validate_super(const struct btrfs_fs_info *fs_info,
const struct btrfs_super_block *sb, int mirror_num)
{
- u64 nodesize = btrfs_super_nodesize(sb);
- u64 sectorsize = btrfs_super_sectorsize(sb);
+ const u32 nodesize = btrfs_super_nodesize(sb);
+ const u32 sectorsize = btrfs_super_sectorsize(sb);
int ret = 0;
const bool ignore_flags = btrfs_test_opt(fs_info, IGNORESUPERFLAGS);
@@ -2433,24 +2433,24 @@ int btrfs_validate_super(const struct btrfs_fs_info *fs_info,
*/
if (unlikely(!is_power_of_2(sectorsize) || sectorsize < BTRFS_MIN_BLOCKSIZE ||
sectorsize > BTRFS_MAX_METADATA_BLOCKSIZE)) {
- btrfs_err(fs_info, "invalid sectorsize %llu", sectorsize);
+ btrfs_err(fs_info, "invalid sectorsize %u", sectorsize);
ret = -EINVAL;
}
if (unlikely(!btrfs_supported_blocksize(sectorsize))) {
btrfs_err(fs_info,
- "sectorsize %llu not yet supported for page size %lu",
+ "sectorsize %u not yet supported for page size %lu",
sectorsize, PAGE_SIZE);
ret = -EINVAL;
}
if (unlikely(!is_power_of_2(nodesize) || nodesize < sectorsize ||
nodesize > BTRFS_MAX_METADATA_BLOCKSIZE)) {
- btrfs_err(fs_info, "invalid nodesize %llu", nodesize);
+ btrfs_err(fs_info, "invalid nodesize %u", nodesize);
ret = -EINVAL;
}
if (unlikely(nodesize != le32_to_cpu(sb->__unused_leafsize))) {
- btrfs_err(fs_info, "invalid leafsize %u, should be %llu",
+ btrfs_err(fs_info, "invalid leafsize %u, should be %u",
le32_to_cpu(sb->__unused_leafsize), nodesize);
ret = -EINVAL;
}
@@ -2900,7 +2900,6 @@ void btrfs_init_fs_info(struct btrfs_fs_info *fs_info)
fs_info->nodesize = 4096;
fs_info->sectorsize = 4096;
fs_info->sectorsize_bits = ilog2(4096);
- fs_info->stripesize = 4096;
/* Default compress algorithm when user does -o compress */
fs_info->compress_type = BTRFS_COMPRESS_ZLIB;
@@ -3309,6 +3308,8 @@ static void invalidate_and_check_btree_folios(struct btrfs_fs_info *fs_info)
*/
rcu_read_lock();
xa_for_each(&fs_info->buffer_tree, index, eb) {
+ unsigned int refs;
+
/* Increase the ref so that the eb won't disappear. */
if (!refcount_inc_not_zero(&eb->refs))
continue;
@@ -3319,16 +3320,26 @@ static void invalidate_and_check_btree_folios(struct btrfs_fs_info *fs_info)
wait_on_bit_io(&eb->bflags, EXTENT_BUFFER_READING,
TASK_UNINTERRUPTIBLE);
/*
+ * We hold the spinlock to make sure above
+ * EXTENT_BUFFER_READING flag is cleared with the held
+ * ref dropped.
+ * Or we can hit a race window and lead to false alerts.
+ */
+ spin_lock(&eb->refs_lock);
+ refs = refcount_read(&eb->refs);
+ spin_unlock(&eb->refs_lock);
+
+ /*
* The refs threshold is 2, one held by us at the beginning
* of the loop, one for the ownership in the buffer tree.
*/
- if (unlikely(refcount_read(&eb->refs) > 2 || extent_buffer_under_io(eb))) {
+ if (unlikely(refs > 2 || extent_buffer_under_io(eb))) {
WARN_ON_ONCE(IS_ENABLED(CONFIG_BTRFS_DEBUG));
btrfs_warn(fs_info,
"unable to release extent buffer %llu owner %llu gen %llu refs %u flags 0x%lx",
eb->start, btrfs_header_owner(eb),
btrfs_header_generation(eb),
- refcount_read(&eb->refs), eb->bflags);
+ refs, eb->bflags);
}
free_extent_buffer(eb);
rcu_read_lock();
@@ -3350,7 +3361,6 @@ int __cold open_ctree(struct super_block *sb, struct btrfs_fs_devices *fs_device
{
u32 sectorsize;
u32 nodesize;
- u32 stripesize;
u64 generation;
u16 csum_type;
struct btrfs_super_block *disk_super;
@@ -3459,7 +3469,6 @@ int __cold open_ctree(struct super_block *sb, struct btrfs_fs_devices *fs_device
/* Set up fs_info before parsing mount options */
nodesize = btrfs_super_nodesize(disk_super);
sectorsize = btrfs_super_sectorsize(disk_super);
- stripesize = sectorsize;
fs_info->dirty_metadata_batch = nodesize * (1 + ilog2(nr_cpu_ids));
fs_info->delalloc_batch = sectorsize * 512 * (1 + ilog2(nr_cpu_ids));
@@ -3470,7 +3479,6 @@ int __cold open_ctree(struct super_block *sb, struct btrfs_fs_devices *fs_device
fs_info->block_min_order = ilog2(round_up(sectorsize, PAGE_SIZE) >> PAGE_SHIFT);
fs_info->block_max_order = calc_block_max_order(fs_info->sectorsize_bits);
fs_info->csums_per_leaf = BTRFS_MAX_ITEM_SIZE(fs_info) / fs_info->csum_size;
- fs_info->stripesize = stripesize;
fs_info->fs_devices->fs_info = fs_info;
if (fs_info->sectorsize > PAGE_SIZE)
@@ -4357,6 +4365,21 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info)
btrfs_cleanup_defrag_inodes(fs_info);
/*
+ * After we entered close_ctree() autodefrag could be running and before
+ * we parked the cleaner kthread, it dirtied folios of some inode.
+ * We don't want to leave any delalloc here, it may be flushed any time
+ * after this point and result in ordered extents that create delayed
+ * iputs after flushed the ordered extent queues further below, run
+ * delayed iputs and set BTRFS_FS_STATE_NO_DELAYED_IPUT. If we are
+ * mounted with flushoncommit, then btrfs_commit_super() called below
+ * will flush delalloc and wait for ordered extents but we end up
+ * getting delayed iputs than are never run. So flush delalloc and wait
+ * for ordered extents.
+ */
+ btrfs_start_delalloc_roots(fs_info, LONG_MAX, false);
+ btrfs_wait_ordered_roots(fs_info, U64_MAX, NULL);
+
+ /*
* Handle the error fs first, as it will flush and wait for all ordered
* extents. This will generate delayed iputs, thus we want to handle
* it first.
diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c
index 624d76e0ca01..235381b31298 100644
--- a/fs/btrfs/extent-tree.c
+++ b/fs/btrfs/extent-tree.c
@@ -4757,7 +4757,7 @@ have_block_group:
/* Checks */
ffe_ctl->search_start = round_up(ffe_ctl->found_offset,
- fs_info->stripesize);
+ fs_info->sectorsize);
/* move on to the next group */
if (ffe_ctl->search_start + ffe_ctl->num_bytes >
diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c
index de5785117a47..d285bc98b0fb 100644
--- a/fs/btrfs/extent_io.c
+++ b/fs/btrfs/extent_io.c
@@ -6,6 +6,7 @@
#include <linux/mm.h>
#include <linux/pagemap.h>
#include <linux/page-flags.h>
+#include <linux/rmap.h>
#include <linux/sched/mm.h>
#include <linux/spinlock.h>
#include <linux/blkdev.h>
@@ -299,6 +300,25 @@ static noinline void unlock_delalloc_folio(const struct inode *inode,
PAGE_UNLOCK);
}
+#ifdef CONFIG_BTRFS_DEBUG
+/*
+ * Writeback must write-protect a folio when locking it for IO, before
+ * anything consumes its data (zeroing, inline copy, compression,
+ * checksumming). If this fails, then an mmap writer would be able to
+ * modify the data concurrently while we need it to be stable.
+ */
+void btrfs_check_folio_write_protected(struct folio *folio)
+{
+ if (folio_mkclean(folio)) {
+ const struct btrfs_inode *inode = BTRFS_I(folio->mapping->host);
+
+ DEBUG_WARN("writable mmap PTEs, root %llu ino %llu pos %llu order %u",
+ btrfs_root_id(inode->root), btrfs_ino(inode), folio_pos(folio),
+ folio_order(folio));
+ }
+}
+#endif
+
static noinline int lock_delalloc_folios(struct inode *inode,
struct folio *locked_folio,
u64 start, u64 end)
@@ -332,6 +352,8 @@ static noinline int lock_delalloc_folios(struct inode *inode,
folio_unlock(folio);
goto out;
}
+ /* Locked for writeback; revoke writable mmap PTEs before using the data. */
+ folio_mkclean(folio);
range_start = max_t(u64, folio_pos(folio), start);
range_len = min_t(u64, folio_next_pos(folio), end + 1) - range_start;
btrfs_folio_set_lock(fs_info, folio, range_start, range_len);
@@ -1780,6 +1802,13 @@ static noinline_for_stack int extent_writepage_io(struct btrfs_inode *inode,
ASSERT(end <= folio_end, "start=%llu len=%u folio_start=%llu folio_size=%zu",
start, len, folio_start, folio_size(folio));
+ /*
+ * We are about to checksum and write out the data, so it must not be
+ * mmap writeable, or we could corrupt the data and end up with invalid
+ * checksums.
+ */
+ btrfs_check_folio_write_protected(folio);
+
/* Truncate the submit bitmap to the current range. */
if (start > folio_start)
bitmap_clear(bio_ctrl->submit_bitmap, 0,
@@ -2590,6 +2619,8 @@ retry:
continue;
}
+ /* Locked for writeback; revoke writable mmap PTEs before using the data. */
+ folio_mkclean(folio);
ret = extent_writepage(folio, bio_ctrl);
if (ret < 0) {
done = true;
@@ -2745,12 +2776,23 @@ void btrfs_readahead(struct readahead_control *rac)
struct fsverity_info *vi = NULL;
lock_extents_for_read(inode, start, end, &cached_state);
+ /* We don't use cached state for a bulk unlock, just free it. */
+ btrfs_free_extent_state(cached_state);
if (start < i_size_read(vfs_inode))
vi = fsverity_get_info(vfs_inode);
- while ((folio = readahead_folio(rac)) != NULL)
- btrfs_do_readpage(folio, &em_cached, &bio_ctrl, vi);
+ while ((folio = readahead_folio(rac)) != NULL) {
+ /*
+ * Read start and end before btrfs_do_readpage(). It unlocks the
+ * folio, so our reference might not be valid after.
+ */
+ const u64 folio_start = folio_pos(folio);
+ const u64 folio_end = folio_start + folio_size(folio) - 1;
- btrfs_unlock_extent(&inode->io_tree, start, end, &cached_state);
+ btrfs_do_readpage(folio, &em_cached, &bio_ctrl, vi);
+ /* Only unlock the range we locked, even if readahead expands. */
+ if (folio_start >= start && folio_end <= end)
+ btrfs_unlock_extent(&inode->io_tree, folio_start, folio_end, NULL);
+ }
if (em_cached)
btrfs_free_extent_map(em_cached);
@@ -2986,46 +3028,70 @@ static inline void btrfs_release_extent_buffer(struct extent_buffer *eb)
}
/*
+ * Claim a slot to track an extent buffer in, evicting the coldest tracked buffer
+ * when the array is full.
+ *
+ * Slots fill in order until the array is full. After that a CLOCK (second
+ * chance) scan advances the hand, clearing one reference bit per step, until
+ * it lands on an unreferenced slot whose buffer is evicted. Clearing a bit per
+ * step bounds the scan to BTRFS_INHIBITED_EBS_SLOTS iterations.
+ */
+static int btrfs_inhibit_claim_slot(struct btrfs_trans_handle *trans)
+{
+ int slot;
+
+ if (trans->nr_inhibited_ebs < BTRFS_INHIBITED_EBS_SLOTS)
+ return trans->nr_inhibited_ebs++;
+
+ while (trans->inhibited_ebs_referenced & (1U << trans->inhibited_ebs_hand)) {
+ trans->inhibited_ebs_referenced &= ~(1U << trans->inhibited_ebs_hand);
+ trans->inhibited_ebs_hand =
+ (trans->inhibited_ebs_hand + 1) % BTRFS_INHIBITED_EBS_SLOTS;
+ }
+ slot = trans->inhibited_ebs_hand;
+ trans->inhibited_ebs_hand = (trans->inhibited_ebs_hand + 1) % BTRFS_INHIBITED_EBS_SLOTS;
+
+ atomic_dec(&trans->inhibited_ebs[slot]->writeback_inhibitors);
+ free_extent_buffer(trans->inhibited_ebs[slot]);
+
+ return slot;
+}
+
+/*
* Inhibit writeback on buffer during transaction.
*
* @trans: transaction handle that will own the inhibitor
* @eb: extent buffer to inhibit writeback on
*
- * Attempt to track this extent buffer in the transaction's inhibited set. If
- * memory allocation fails, the buffer is simply not tracked. It may be written
- * back and need re-COW, which is the original behavior. This is acceptable
- * since inhibiting writeback is an optimization.
+ * Attempt to track this extent buffer in the transaction's inhibited set. When
+ * the set is full the coldest tracked buffer is evicted instead. An untracked
+ * buffer may be written back and need re-COW, which is the original behavior.
+ * This is acceptable since inhibiting writeback is an optimization.
*/
void btrfs_inhibit_eb_writeback(struct btrfs_trans_handle *trans, struct extent_buffer *eb)
{
- unsigned long index = eb->start >> trans->fs_info->nodesize_bits;
- void *old;
+ int slot;
lockdep_assert_held(&eb->lock);
- /* Check if already inhibited by this handle. */
- old = xa_load(&trans->writeback_inhibited_ebs, index);
- if (old == eb)
- return;
- /* Take reference for the xarray entry. */
- refcount_inc(&eb->refs);
-
- old = xa_store(&trans->writeback_inhibited_ebs, index, eb, GFP_NOFS);
- if (xa_is_err(old)) {
- /* Allocation failed, just skip inhibiting this buffer. */
- free_extent_buffer(eb);
- return;
+ /* Already tracked: set its reference bit (second chance) and return. */
+ for (int i = 0; i < trans->nr_inhibited_ebs; i++) {
+ if (trans->inhibited_ebs[i] == eb) {
+ trans->inhibited_ebs_referenced |= 1U << i;
+ return;
+ }
}
- /* Handle replacement of different eb at same index. */
- if (old && old != eb) {
- struct extent_buffer *old_eb = old;
-
- atomic_dec(&old_eb->writeback_inhibitors);
- free_extent_buffer(old_eb);
- }
+ slot = btrfs_inhibit_claim_slot(trans);
+ /*
+ * Pin the eb while the array holds a raw pointer to it; the counter is
+ * what lock_extent_buffer_for_io() checks.
+ */
+ refcount_inc(&eb->refs);
atomic_inc(&eb->writeback_inhibitors);
+ trans->inhibited_ebs[slot] = eb;
+ trans->inhibited_ebs_referenced |= 1U << slot;
}
/*
@@ -3033,14 +3099,13 @@ void btrfs_inhibit_eb_writeback(struct btrfs_trans_handle *trans, struct extent_
*/
void btrfs_uninhibit_all_eb_writeback(struct btrfs_trans_handle *trans)
{
- struct extent_buffer *eb;
- unsigned long index;
-
- xa_for_each(&trans->writeback_inhibited_ebs, index, eb) {
- atomic_dec(&eb->writeback_inhibitors);
- free_extent_buffer(eb);
+ for (int i = 0; i < trans->nr_inhibited_ebs; i++) {
+ atomic_dec(&trans->inhibited_ebs[i]->writeback_inhibitors);
+ free_extent_buffer(trans->inhibited_ebs[i]);
}
- xa_destroy(&trans->writeback_inhibited_ebs);
+ trans->nr_inhibited_ebs = 0;
+ trans->inhibited_ebs_referenced = 0;
+ trans->inhibited_ebs_hand = 0;
}
static struct extent_buffer *__alloc_extent_buffer(struct btrfs_fs_info *fs_info,
@@ -3698,12 +3763,31 @@ static int release_extent_buffer(struct extent_buffer *eb)
return 0;
}
-void free_extent_buffer(struct extent_buffer *eb)
+static void clear_extent_buffer_reading(struct extent_buffer *eb)
+{
+ clear_and_wake_up_bit(EXTENT_BUFFER_READING, &eb->bflags);
+}
+
+static void free_extent_buffer_clear_reading(struct extent_buffer *eb,
+ bool clear_reading)
{
int refs;
+
if (!eb)
return;
+ /*
+ * We want to clear EXTENT_BUFFER_READING flag and decrease refs
+ * in the same critical section.
+ * This will make sure invalidate_and_check_btree_folios() won't
+ * see an eb with EXTENT_BUFFER_READING cleared but refs not yet
+ * decreased.
+ */
+ if (clear_reading) {
+ spin_lock(&eb->refs_lock);
+ clear_extent_buffer_reading(eb);
+ }
+
refs = refcount_read(&eb->refs);
while (1) {
if (test_bit(EXTENT_BUFFER_UNMAPPED, &eb->bflags)) {
@@ -3714,11 +3798,16 @@ void free_extent_buffer(struct extent_buffer *eb)
}
/* Optimization to avoid locking eb->refs_lock. */
- if (atomic_try_cmpxchg(&eb->refs.refs, &refs, refs - 1))
+ if (atomic_try_cmpxchg(&eb->refs.refs, &refs, refs - 1)) {
+ if (clear_reading)
+ spin_unlock(&eb->refs_lock);
return;
+ }
}
- spin_lock(&eb->refs_lock);
+ if (!clear_reading)
+ spin_lock(&eb->refs_lock);
+
if (refcount_read(&eb->refs) == 2 &&
test_bit(EXTENT_BUFFER_STALE, &eb->bflags) &&
!extent_buffer_under_io(eb) &&
@@ -3732,6 +3821,11 @@ void free_extent_buffer(struct extent_buffer *eb)
release_extent_buffer(eb);
}
+void free_extent_buffer(struct extent_buffer *eb)
+{
+ return free_extent_buffer_clear_reading(eb, false);
+}
+
void free_extent_buffer_stale(struct extent_buffer *eb)
{
if (!eb)
@@ -3857,11 +3951,6 @@ void set_extent_buffer_uptodate(struct extent_buffer *eb)
btrfs_meta_folio_set_uptodate(eb->folios[i], eb);
}
-static void clear_extent_buffer_reading(struct extent_buffer *eb)
-{
- clear_and_wake_up_bit(EXTENT_BUFFER_READING, &eb->bflags);
-}
-
static void end_bbio_meta_read(struct btrfs_bio *bbio)
{
struct extent_buffer *eb = bbio->private;
@@ -3885,8 +3974,7 @@ static void end_bbio_meta_read(struct btrfs_bio *bbio)
else
clear_extent_buffer_uptodate(eb);
- clear_extent_buffer_reading(eb);
- free_extent_buffer(eb);
+ free_extent_buffer_clear_reading(eb, true);
bio_put(&bbio->bio);
}
diff --git a/fs/btrfs/extent_io.h b/fs/btrfs/extent_io.h
index 9896e15ddc40..869925337699 100644
--- a/fs/btrfs/extent_io.h
+++ b/fs/btrfs/extent_io.h
@@ -255,6 +255,11 @@ bool try_release_extent_mapping(struct folio *folio, gfp_t mask);
int try_release_extent_buffer(struct folio *folio);
int btrfs_read_folio(struct file *file, struct folio *folio);
+#ifdef CONFIG_BTRFS_DEBUG
+void btrfs_check_folio_write_protected(struct folio *folio);
+#else
+static inline void btrfs_check_folio_write_protected(struct folio *folio) { }
+#endif
void extent_write_locked_range(struct inode *inode, const struct folio *locked_folio,
u64 start, u64 end, struct writeback_control *wbc,
bool pages_dirty);
diff --git a/fs/btrfs/fiemap.c b/fs/btrfs/fiemap.c
index 6263e837093e..ba6a360074c0 100644
--- a/fs/btrfs/fiemap.c
+++ b/fs/btrfs/fiemap.c
@@ -641,7 +641,7 @@ static int extent_fiemap(struct btrfs_inode *inode,
u64 prev_extent_end;
u64 range_start;
u64 range_end;
- const u64 sectorsize = inode->root->fs_info->sectorsize;
+ const u32 sectorsize = inode->root->fs_info->sectorsize;
bool stopped = false;
int ret;
diff --git a/fs/btrfs/file.c b/fs/btrfs/file.c
index a2a2df2df786..818c53c445f8 100644
--- a/fs/btrfs/file.c
+++ b/fs/btrfs/file.c
@@ -875,62 +875,56 @@ again:
/*
* Locks the extent and properly waits for data=ordered extents to finish
- * before allowing the folios to be modified if need.
+ * before allowing the folios to be modified.
*
* Return:
- * 1 - the extent is locked
- * 0 - the extent is not locked, and everything is OK
+ * 0 - the extent is locked
* -EAGAIN - need to prepare the folios again
*/
static noinline int
-lock_and_cleanup_extent_if_need(struct btrfs_inode *inode, struct folio *folio,
- loff_t pos, size_t write_bytes,
- u64 *lockstart, u64 *lockend, bool nowait,
- struct extent_state **cached_state)
+lock_and_cleanup_extent(struct btrfs_inode *inode, struct folio *folio,
+ loff_t pos, size_t write_bytes,
+ u64 *lockstart, u64 *lockend, bool nowait,
+ struct extent_state **cached_state)
{
struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ struct btrfs_ordered_extent *ordered;
u64 start_pos;
u64 last_pos;
- int ret = 0;
start_pos = round_down(pos, fs_info->sectorsize);
last_pos = round_up(pos + write_bytes, fs_info->sectorsize) - 1;
- if (start_pos < inode->vfs_inode.i_size) {
- struct btrfs_ordered_extent *ordered;
-
- if (nowait) {
- if (!btrfs_try_lock_extent(&inode->io_tree, start_pos,
- last_pos, cached_state)) {
- folio_unlock(folio);
- folio_put(folio);
- return -EAGAIN;
- }
- } else {
- btrfs_lock_extent(&inode->io_tree, start_pos, last_pos,
- cached_state);
- }
-
- ordered = btrfs_lookup_ordered_range(inode, start_pos,
- last_pos - start_pos + 1);
- if (ordered &&
- ordered->file_offset + ordered->num_bytes > start_pos &&
- ordered->file_offset <= last_pos) {
- btrfs_unlock_extent(&inode->io_tree, start_pos, last_pos,
- cached_state);
+ if (nowait) {
+ if (!btrfs_try_lock_extent(&inode->io_tree, start_pos,
+ last_pos, cached_state)) {
folio_unlock(folio);
folio_put(folio);
- btrfs_start_ordered_extent(ordered);
- btrfs_put_ordered_extent(ordered);
return -EAGAIN;
}
- if (ordered)
- btrfs_put_ordered_extent(ordered);
+ } else {
+ btrfs_lock_extent(&inode->io_tree, start_pos, last_pos,
+ cached_state);
+ }
- *lockstart = start_pos;
- *lockend = last_pos;
- ret = 1;
+ ordered = btrfs_lookup_ordered_range(inode, start_pos,
+ last_pos - start_pos + 1);
+ if (ordered &&
+ ordered->file_offset + ordered->num_bytes > start_pos &&
+ ordered->file_offset <= last_pos) {
+ btrfs_unlock_extent(&inode->io_tree, start_pos, last_pos,
+ cached_state);
+ folio_unlock(folio);
+ folio_put(folio);
+ btrfs_start_ordered_extent(ordered);
+ btrfs_put_ordered_extent(ordered);
+ return -EAGAIN;
}
+ if (ordered)
+ btrfs_put_ordered_extent(ordered);
+
+ *lockstart = start_pos;
+ *lockend = last_pos;
/*
* We should be called after prepare_one_folio() which should have locked
@@ -938,7 +932,7 @@ lock_and_cleanup_extent_if_need(struct btrfs_inode *inode, struct folio *folio,
*/
WARN_ON(!folio_test_locked(folio));
- return ret;
+ return 0;
}
/*
@@ -1195,7 +1189,6 @@ static int copy_one_range(struct btrfs_inode *inode, struct iov_iter *iter,
const u64 reserved_start = round_down(start, fs_info->sectorsize);
u64 reserved_len;
struct folio *folio = NULL;
- int extents_locked;
u64 lockstart;
u64 lockend;
bool only_release_metadata = false;
@@ -1253,18 +1246,16 @@ again:
reserved_len = last_block - reserved_start;
}
- extents_locked = lock_and_cleanup_extent_if_need(inode, folio, start,
- write_bytes, &lockstart,
- &lockend, nowait,
- &cached_state);
- if (extents_locked < 0) {
- if (!nowait && extents_locked == -EAGAIN)
+ ret = lock_and_cleanup_extent(inode, folio, start, write_bytes,
+ &lockstart, &lockend, nowait, &cached_state);
+ if (ret < 0) {
+ if (!nowait)
goto again;
btrfs_delalloc_release_extents(inode, reserved_len);
release_space(inode, *data_reserved, reserved_start, reserved_len,
only_release_metadata);
- return extents_locked;
+ return ret;
}
copied = copy_folio_from_iter_atomic(folio, offset_in_folio(folio, start),
@@ -1288,11 +1279,8 @@ again:
/* No copied bytes, unlock, release reserved space and exit. */
if (copied == 0) {
- if (extents_locked)
- btrfs_unlock_extent(&inode->io_tree, lockstart, lockend,
- &cached_state);
- else
- btrfs_free_extent_state(cached_state);
+ btrfs_unlock_extent(&inode->io_tree, lockstart, lockend,
+ &cached_state);
btrfs_delalloc_release_extents(inode, reserved_len);
release_space(inode, *data_reserved, reserved_start, reserved_len,
only_release_metadata);
@@ -1311,17 +1299,7 @@ again:
ret = btrfs_dirty_folio(inode, folio, start, copied, &cached_state,
only_release_metadata);
- /*
- * If we have not locked the extent range, because the range's start
- * offset is >= i_size, we might still have a non-NULL cached extent
- * state, acquired while marking the extent range as delalloc through
- * btrfs_dirty_page(). Therefore free any possible cached extent state
- * to avoid a memory leak.
- */
- if (extents_locked)
- btrfs_unlock_extent(&inode->io_tree, lockstart, lockend, &cached_state);
- else
- btrfs_free_extent_state(cached_state);
+ btrfs_unlock_extent(&inode->io_tree, lockstart, lockend, &cached_state);
btrfs_delalloc_release_extents(inode, reserved_len);
if (ret) {
@@ -2683,8 +2661,8 @@ static int btrfs_punch_hole(struct file *file, loff_t offset, loff_t len)
lockstart = round_up(offset, fs_info->sectorsize);
lockend = round_down(offset + len, fs_info->sectorsize) - 1;
- same_block = (BTRFS_BYTES_TO_BLKS(fs_info, offset))
- == (BTRFS_BYTES_TO_BLKS(fs_info, offset + len - 1));
+ same_block = (offset >> fs_info->sectorsize_bits) ==
+ ((offset + len - 1) >> fs_info->sectorsize_bits);
/*
* Only do this if we are in the same block and we aren't doing the
* entire block.
@@ -2888,7 +2866,7 @@ enum {
static int btrfs_zero_range_check_range_boundary(struct btrfs_inode *inode,
u64 offset)
{
- const u64 sectorsize = inode->root->fs_info->sectorsize;
+ const u32 sectorsize = inode->root->fs_info->sectorsize;
struct extent_map *em;
int ret;
@@ -2918,7 +2896,7 @@ static int btrfs_zero_range(struct inode *inode,
struct extent_changeset *data_reserved = NULL;
int ret;
u64 alloc_hint = 0;
- const u64 sectorsize = fs_info->sectorsize;
+ const u32 sectorsize = fs_info->sectorsize;
const u64 orig_start = offset;
const u64 orig_end = offset + len - 1;
u64 alloc_start = round_down(offset, sectorsize);
@@ -2967,8 +2945,8 @@ static int btrfs_zero_range(struct inode *inode,
}
btrfs_free_extent_map(em);
- if (BTRFS_BYTES_TO_BLKS(fs_info, offset) ==
- BTRFS_BYTES_TO_BLKS(fs_info, offset + len - 1)) {
+ if ((offset >> fs_info->sectorsize_bits) ==
+ ((offset + len - 1) >> fs_info->sectorsize_bits)) {
em = btrfs_get_extent(BTRFS_I(inode), NULL, alloc_start, sectorsize);
if (IS_ERR(em)) {
ret = PTR_ERR(em);
diff --git a/fs/btrfs/fs.c b/fs/btrfs/fs.c
index 14d83565cdee..5015d0148841 100644
--- a/fs/btrfs/fs.c
+++ b/fs/btrfs/fs.c
@@ -127,20 +127,9 @@ void btrfs_csum_final(struct btrfs_csum_ctx *ctx, u8 *out)
}
/*
- * We support the following block sizes for all systems:
- *
- * - 4K
- * This is the most common block size. For PAGE SIZE > 4K cases the subpage
- * mode is used.
- *
- * - PAGE_SIZE
- * The straightforward block size to support.
- *
- * And extra support for the following block sizes based on the kernel config:
- *
- * - MIN_BLOCKSIZE
- * This is either 4K (regular builds) or 2K (debug builds)
- * This allows testing subpage routines on x86_64.
+ * For regular builds, any block size <= page size is supported.
+ * For experimental builds, any block size between BTRFS_MIN_BLOCKSIZE
+ * and BTRFS_MAX_BLOCKSIZE (inclusive) is supported.
*/
bool __attribute_const__ btrfs_supported_blocksize(u32 blocksize)
{
@@ -148,7 +137,7 @@ bool __attribute_const__ btrfs_supported_blocksize(u32 blocksize)
ASSERT(is_power_of_2(blocksize) && blocksize >= BTRFS_MIN_BLOCKSIZE &&
blocksize <= BTRFS_MAX_BLOCKSIZE);
- if (blocksize == PAGE_SIZE || blocksize == SZ_4K || blocksize == BTRFS_MIN_BLOCKSIZE)
+ if (blocksize <= PAGE_SIZE)
return true;
#ifdef CONFIG_BTRFS_EXPERIMENTAL
/*
diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h
index 7ee9ec2b0efb..06b5884a9bcd 100644
--- a/fs/btrfs/fs.h
+++ b/fs/btrfs/fs.h
@@ -289,7 +289,8 @@ enum {
BTRFS_MOUNT_IGNOREBADROOTS | \
BTRFS_MOUNT_IGNOREDATACSUMS | \
BTRFS_MOUNT_IGNOREMETACSUMS | \
- BTRFS_MOUNT_IGNORESUPERFLAGS)
+ BTRFS_MOUNT_IGNORESUPERFLAGS | \
+ BTRFS_MOUNT_USEBACKUPROOT)
/*
* Compat flags that we support. If any incompat flags are set other than the
@@ -888,7 +889,6 @@ struct btrfs_fs_info {
u32 sectorsize_bits;
u32 block_min_order;
u32 block_max_order;
- u32 stripesize;
u32 writeback_bio_size;
u32 csum_size;
u32 csums_per_leaf;
@@ -1058,8 +1058,6 @@ static inline u64 btrfs_calc_metadata_size(const struct btrfs_fs_info *fs_info,
#define BTRFS_MAX_EXTENT_ITEM_SIZE(r) ((BTRFS_LEAF_DATA_SIZE(r->fs_info) >> 4) - \
sizeof(struct btrfs_item))
-#define BTRFS_BYTES_TO_BLKS(fs_info, bytes) ((bytes) >> (fs_info)->sectorsize_bits)
-
static inline bool btrfs_is_zoned(const struct btrfs_fs_info *fs_info)
{
return IS_ENABLED(CONFIG_BLK_DEV_ZONED) && fs_info->zone_size > 0;
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index b446c3014b24..0fcbfc393946 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -775,19 +775,28 @@ static inline void inode_should_defrag(struct btrfs_inode *inode,
static int extent_range_clear_dirty_for_io(struct btrfs_inode *inode, u64 start, u64 end)
{
+ pgoff_t index = start >> PAGE_SHIFT;
const pgoff_t end_index = end >> PAGE_SHIFT;
struct folio *folio;
int ret = 0;
- for (pgoff_t index = start >> PAGE_SHIFT; index <= end_index; index++) {
+ while (index <= end_index) {
folio = filemap_get_folio(inode->vfs_inode.i_mapping, index);
if (IS_ERR(folio)) {
if (!ret)
ret = PTR_ERR(folio);
+ index++;
continue;
}
+ /*
+ * We are about to compress the folio, so it must not be mmap
+ * writeable or we could corrupt the data as we attempt to
+ * compress it.
+ */
+ btrfs_check_folio_write_protected(folio);
btrfs_folio_clamp_clear_dirty(inode->root->fs_info, folio, start,
end + 1 - start);
+ index = folio_next_index(folio);
folio_put(folio);
}
return ret;
@@ -860,7 +869,7 @@ static void compress_file_range(struct btrfs_work *work)
struct btrfs_inode *inode = async_chunk->inode;
struct btrfs_fs_info *fs_info = inode->root->fs_info;
struct compressed_bio *cb = NULL;
- u64 blocksize = fs_info->sectorsize;
+ const u32 blocksize = fs_info->sectorsize;
u64 start = async_chunk->start;
u64 end = async_chunk->end;
u64 actual_end;
@@ -877,11 +886,6 @@ static void compress_file_range(struct btrfs_work *work)
inode_should_defrag(inode, start, end, end - start + 1, SZ_16K);
- /*
- * We need to call clear_page_dirty_for_io on each page in the range.
- * Otherwise applications with the file mmap'd can wander in and change
- * the page contents while we are compressing them.
- */
ret = extent_range_clear_dirty_for_io(inode, start, end);
/*
@@ -2317,6 +2321,13 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f
int ret;
ASSERT(folio_pos(locked_folio) == 0);
+ /*
+ * If an mmap writer could modify the folio while we copy it into an
+ * inline extent we might see only part of their modification then
+ * wrongly mark it clean again after copying, losing that write. So the
+ * folio must be write protected here.
+ */
+ btrfs_check_folio_write_protected(locked_folio);
if (btrfs_inode_can_compress(inode) &&
inode_need_compress(inode, 0, blocksize, true)) {
@@ -2865,7 +2876,7 @@ static int insert_reserved_file_extent(struct btrfs_trans_handle *trans,
u64 qgroup_reserved)
{
struct btrfs_root *root = inode->root;
- const u64 sectorsize = root->fs_info->sectorsize;
+ const u32 sectorsize = root->fs_info->sectorsize;
BTRFS_PATH_AUTO_FREE(path);
struct extent_buffer *leaf;
struct btrfs_key ins;
@@ -6832,7 +6843,7 @@ static int btrfs_mknod(struct mnt_idmap *idmap, struct inode *dir,
}
static int btrfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode;
@@ -6936,7 +6947,7 @@ static struct dentry *btrfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
inode = new_inode(dir->i_sb);
if (!inode)
return ERR_PTR(-ENOMEM);
- inode_init_owner(idmap, inode, dir, S_IFDIR | mode);
+ inode_init_owner(idmap, inode, dir, mode);
inode->i_op = &btrfs_dir_inode_operations;
inode->i_fop = &btrfs_dir_file_operations;
return ERR_PTR(btrfs_create_common(dir, dentry, inode));
@@ -10012,6 +10023,8 @@ static void btrfs_free_swapfile_pins(struct inode *inode)
struct btrfs_fs_info *fs_info = BTRFS_I(inode)->root->fs_info;
struct btrfs_swapfile_pin *sp;
struct rb_node *node, *next;
+ u64 bg_bytes_released = 0;
+ u32 bg_nr_released = 0;
spin_lock(&fs_info->swapfile_pins_lock);
node = rb_first(&fs_info->swapfile_pins);
@@ -10021,15 +10034,24 @@ static void btrfs_free_swapfile_pins(struct inode *inode)
if (sp->inode == inode) {
rb_erase(&sp->node, &fs_info->swapfile_pins);
if (sp->is_block_group) {
- btrfs_dec_block_group_swap_extents(sp->ptr,
+ struct btrfs_block_group *bg = sp->ptr;
+
+ bg_bytes_released += bg->length;
+ bg_nr_released++;
+ btrfs_dec_block_group_swap_extents(bg,
sp->bg_extent_count);
- btrfs_put_block_group(sp->ptr);
+ btrfs_put_block_group(bg);
}
kfree(sp);
}
node = next;
}
spin_unlock(&fs_info->swapfile_pins_lock);
+ btrfs_info(fs_info,
+"swapfile deactivated on root %llu ino %llu, released %llu bytes from %u block group(s)",
+ btrfs_root_id(BTRFS_I(inode)->root),
+ btrfs_ino(BTRFS_I(inode)), bg_bytes_released,
+ bg_nr_released);
}
struct btrfs_swap_info {
@@ -10107,8 +10129,10 @@ static int btrfs_swap_activate(struct swap_info_struct *sis, struct file *file,
struct btrfs_backref_share_check_ctx *backref_ctx = NULL;
struct btrfs_path *path = NULL;
int ret = 0;
+ u32 pinned_bg_nr = 0;
u64 isize;
u64 prev_extent_end = 0;
+ u64 pinned_bg_size = 0;
/*
* Acquire the inode's mmap lock to prevent races with memory mapped
@@ -10358,6 +10382,9 @@ static int btrfs_swap_activate(struct swap_info_struct *sis, struct file *file,
ret = 0;
else
goto out;
+ } else {
+ pinned_bg_size += bg->length;
+ pinned_bg_nr++;
}
if (bsi.block_len &&
@@ -10405,6 +10432,14 @@ out_unlock_mmap:
if (ret)
return ret;
+ btrfs_info(fs_info,
+"swapfile activated on root %llu ino %llu, pinned down %llu bytes from %u block group(s)",
+ btrfs_root_id(BTRFS_I(inode)->root),
+ btrfs_ino(BTRFS_I(inode)),
+ pinned_bg_size, pinned_bg_nr);
+ btrfs_warn(fs_info,
+"block groups with swapfile extents will not be scrubbed or balanced");
+
if (device)
sis->bdev = device->bdev;
*span = bsi.highest_ppage - bsi.lowest_ppage + 1;
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
index baa645e98812..07ce84070efd 100644
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -356,14 +356,21 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags |= BTRFS_INODE_NODATACOW;
}
} else {
- /*
- * Revert back under same assumptions as above
- */
- if (S_ISREG(inode->vfs_inode.i_mode)) {
- if (inode->vfs_inode.i_size == 0)
- inode_flags &= ~(BTRFS_INODE_NODATACOW |
- BTRFS_INODE_NODATASUM);
- } else {
+ /* We can only change NODATACOW for zero-sized regular file. */
+ if (S_ISREG(inode->vfs_inode.i_mode) && (inode->vfs_inode.i_size == 0)) {
+ inode_flags &= ~BTRFS_INODE_NODATACOW;
+ /*
+ * There is currently no way to change NODATASUM flag
+ * through fileattr API. If we unconditionally keep the
+ * current NODATASUM flag, chattr +C then chattr -C will
+ * keep the NODATASUM flag, and no way to remove that
+ * flag.
+ *
+ * So respect the current mount option for NODATASUM flag.
+ */
+ if (!btrfs_test_opt(fs_info, NODATASUM))
+ inode_flags &= ~BTRFS_INODE_NODATASUM;
+ } else if (!S_ISREG(inode->vfs_inode.i_mode)) {
inode_flags &= ~BTRFS_INODE_NODATACOW;
}
}
@@ -393,9 +400,9 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
/*
* 1 for inode item
- * 2 for properties
+ * 1 for property
*/
- trans = btrfs_start_transaction(root, 3);
+ trans = btrfs_start_transaction(root, 2);
if (IS_ERR(trans))
return PTR_ERR(trans);
@@ -1136,13 +1143,13 @@ out_drop:
}
static noinline int __btrfs_ioctl_snap_create(struct file *file,
- struct mnt_idmap *idmap,
const char *name, unsigned long fd, bool subvol,
bool readonly,
struct btrfs_qgroup_inherit *inherit)
{
int ret;
struct qstr qname = QSTR(name);
+ struct mnt_idmap *idmap = file_mnt_idmap(file);
if (!S_ISDIR(file_inode(file)->i_mode))
return -ENOTDIR;
@@ -1220,8 +1227,7 @@ static noinline int btrfs_ioctl_snap_create(struct file *file,
if (ret < 0)
return ret;
- return __btrfs_ioctl_snap_create(file, file_mnt_idmap(file),
- vol_args->name, vol_args->fd, subvol,
+ return __btrfs_ioctl_snap_create(file, vol_args->name, vol_args->fd, subvol,
false, NULL);
}
@@ -1264,8 +1270,7 @@ static noinline int btrfs_ioctl_snap_create_v2(struct file *file,
return ret;
}
- return __btrfs_ioctl_snap_create(file, file_mnt_idmap(file),
- vol_args->name, vol_args->fd, subvol,
+ return __btrfs_ioctl_snap_create(file, vol_args->name, vol_args->fd, subvol,
readonly, inherit);
}
@@ -1657,13 +1662,11 @@ static noinline int btrfs_ioctl_tree_search_v2(struct btrfs_root *root,
}
/*
- * Search INODE_REFs to identify path name of 'dirid' directory
- * in a 'tree_id' tree. and sets path name to 'name'.
+ * Search for an INODE_REF in a 'root' tree which identifies the path name of
+ * 'dirid'. When found, it sets 'name' with the path name.
*/
-static noinline int btrfs_search_path_in_tree(struct btrfs_fs_info *info,
- u64 tree_id, u64 dirid, char *name)
+static noinline int btrfs_search_path_in_tree(struct btrfs_root *root, u64 dirid, char *name)
{
- struct btrfs_root *root;
struct btrfs_key key;
char *ptr;
int ret = -1;
@@ -1685,13 +1688,6 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_fs_info *info,
ptr = &name[BTRFS_INO_LOOKUP_PATH_MAX - 1];
- root = btrfs_get_fs_root(info, tree_id, true);
- if (IS_ERR(root)) {
- ret = PTR_ERR(root);
- root = NULL;
- goto out;
- }
-
key.objectid = dirid;
key.type = BTRFS_INODE_REF_KEY;
key.offset = (u64)-1;
@@ -1699,11 +1695,9 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_fs_info *info,
while (1) {
ret = btrfs_search_backwards(root, &key, path);
if (ret < 0)
- goto out;
- else if (ret > 0) {
- ret = -ENOENT;
- goto out;
- }
+ return ret;
+ else if (ret > 0)
+ return -ENOENT;
l = path->nodes[0];
slot = path->slots[0];
@@ -1712,10 +1706,8 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_fs_info *info,
len = btrfs_inode_ref_name_len(l, iref);
ptr -= len + 1;
total_len += len + 1;
- if (ptr < name) {
- ret = -ENAMETOOLONG;
- goto out;
- }
+ if (ptr < name)
+ return -ENAMETOOLONG;
*(ptr + len) = '/';
read_extent_buffer(l, ptr, (unsigned long)(iref + 1), len);
@@ -1730,10 +1722,8 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_fs_info *info,
}
memmove(name, ptr, total_len);
name[total_len] = '\0';
- ret = 0;
-out:
- btrfs_put_root(root);
- return ret;
+
+ return 0;
}
static int btrfs_search_path_in_tree_user(struct mnt_idmap *idmap,
@@ -1877,6 +1867,7 @@ out_put:
static noinline int btrfs_ioctl_ino_lookup(struct btrfs_root *root,
void __user *argp)
{
+ bool new_root = false;
struct btrfs_ioctl_ino_lookup_args AUTO_KFREE(args);
int ret = 0;
@@ -1890,6 +1881,8 @@ static noinline int btrfs_ioctl_ino_lookup(struct btrfs_root *root,
*/
if (args->treeid == 0)
args->treeid = btrfs_root_id(root);
+ else
+ new_root = true;
if (args->objectid == BTRFS_FIRST_FREE_OBJECTID) {
args->name[0] = 0;
@@ -1901,9 +1894,14 @@ static noinline int btrfs_ioctl_ino_lookup(struct btrfs_root *root,
goto out;
}
- ret = btrfs_search_path_in_tree(root->fs_info,
- args->treeid, args->objectid,
- args->name);
+ if (new_root) {
+ root = btrfs_get_fs_root(root->fs_info, args->treeid, true);
+ if (IS_ERR(root))
+ return PTR_ERR(root);
+ }
+ ret = btrfs_search_path_in_tree(root, args->objectid, args->name);
+ if (new_root)
+ btrfs_put_root(root);
out:
if (ret == 0 && copy_to_user(argp, args, sizeof(*args)))
@@ -2629,7 +2627,7 @@ static long btrfs_ioctl_rm_dev_v2(struct file *file, void __user *arg)
err_drop:
mnt_drop_write_file(file);
if (bdev_file)
- bdev_fput(bdev_file);
+ btrfs_release_device_allow_freeze(bdev_file);
out:
btrfs_put_dev_args_from_path(&args);
return ret;
@@ -2679,7 +2677,7 @@ static long btrfs_ioctl_rm_dev(struct file *file, void __user *arg)
mnt_drop_write_file(file);
if (bdev_file)
- bdev_fput(bdev_file);
+ btrfs_release_device_allow_freeze(bdev_file);
out:
btrfs_put_dev_args_from_path(&args);
return ret;
@@ -3613,7 +3611,7 @@ static long btrfs_ioctl_qgroup_assign(struct file *file, void __user *arg)
{
struct inode *inode = file_inode(file);
struct btrfs_fs_info *fs_info = inode_to_fs_info(inode);
- struct btrfs_root *root = BTRFS_I(inode)->root;
+ struct btrfs_root *quota_root;
struct btrfs_ioctl_qgroup_assign_args AUTO_KFREE(sa);
struct btrfs_qgroup_list AUTO_KFREE(prealloc);
struct btrfs_trans_handle *trans;
@@ -3644,10 +3642,20 @@ static long btrfs_ioctl_qgroup_assign(struct file *file, void __user *arg)
}
}
+ mutex_lock(&fs_info->qgroup_ioctl_lock);
+ quota_root = btrfs_grab_root(fs_info->quota_root);
+ mutex_unlock(&fs_info->qgroup_ioctl_lock);
+
+ if (!quota_root) {
+ ret = -ENOTCONN;
+ goto drop_write;
+ }
+
/* 2 BTRFS_QGROUP_RELATION_KEY items. */
- trans = btrfs_start_transaction(root, 2);
+ trans = btrfs_start_transaction(quota_root, 2);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
+ btrfs_put_root(quota_root);
goto drop_write;
}
@@ -3671,6 +3679,7 @@ static long btrfs_ioctl_qgroup_assign(struct file *file, void __user *arg)
"qgroup status update failed after %s relation, marked as inconsistent",
sa->assign ? "adding" : "deleting");
err = btrfs_end_transaction(trans);
+ btrfs_put_root(quota_root);
if (err && !ret)
ret = err;
@@ -3682,7 +3691,8 @@ drop_write:
static long btrfs_ioctl_qgroup_create(struct file *file, void __user *arg)
{
struct inode *inode = file_inode(file);
- struct btrfs_root *root = BTRFS_I(inode)->root;
+ struct btrfs_fs_info *fs_info = inode_to_fs_info(inode);
+ struct btrfs_root *quota_root;
struct btrfs_ioctl_qgroup_create_args AUTO_KFREE(sa);
struct btrfs_trans_handle *trans;
int ret;
@@ -3691,7 +3701,7 @@ static long btrfs_ioctl_qgroup_create(struct file *file, void __user *arg)
if (!capable(CAP_SYS_ADMIN))
return -EPERM;
- if (!btrfs_qgroup_enabled(root->fs_info))
+ if (!btrfs_qgroup_enabled(fs_info))
return -ENOTCONN;
ret = mnt_want_write_file(file);
@@ -3714,13 +3724,23 @@ static long btrfs_ioctl_qgroup_create(struct file *file, void __user *arg)
goto drop_write;
}
+ mutex_lock(&fs_info->qgroup_ioctl_lock);
+ quota_root = btrfs_grab_root(fs_info->quota_root);
+ mutex_unlock(&fs_info->qgroup_ioctl_lock);
+
+ if (!quota_root) {
+ ret = -ENOTCONN;
+ goto drop_write;
+ }
+
/*
* 1 BTRFS_QGROUP_INFO_KEY item.
* 1 BTRFS_QGROUP_LIMIT_KEY item.
*/
- trans = btrfs_start_transaction(root, 2);
+ trans = btrfs_start_transaction(quota_root, 2);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
+ btrfs_put_root(quota_root);
goto drop_write;
}
@@ -3731,6 +3751,7 @@ static long btrfs_ioctl_qgroup_create(struct file *file, void __user *arg)
}
err = btrfs_end_transaction(trans);
+ btrfs_put_root(quota_root);
if (err && !ret)
ret = err;
@@ -3743,6 +3764,8 @@ static long btrfs_ioctl_qgroup_limit(struct file *file, void __user *arg)
{
struct inode *inode = file_inode(file);
struct btrfs_root *root = BTRFS_I(inode)->root;
+ struct btrfs_root *quota_root;
+ struct btrfs_fs_info *fs_info = root->fs_info;
struct btrfs_ioctl_qgroup_limit_args AUTO_KFREE(sa);
struct btrfs_trans_handle *trans;
int ret;
@@ -3752,7 +3775,7 @@ static long btrfs_ioctl_qgroup_limit(struct file *file, void __user *arg)
if (!capable(CAP_SYS_ADMIN))
return -EPERM;
- if (!btrfs_qgroup_enabled(root->fs_info))
+ if (!btrfs_qgroup_enabled(fs_info))
return -ENOTCONN;
ret = mnt_want_write_file(file);
@@ -3765,10 +3788,20 @@ static long btrfs_ioctl_qgroup_limit(struct file *file, void __user *arg)
goto drop_write;
}
+ mutex_lock(&fs_info->qgroup_ioctl_lock);
+ quota_root = btrfs_grab_root(fs_info->quota_root);
+ mutex_unlock(&fs_info->qgroup_ioctl_lock);
+
+ if (!quota_root) {
+ ret = -ENOTCONN;
+ goto drop_write;
+ }
+
/* 1 BTRFS_QGROUP_LIMIT_KEY item. */
- trans = btrfs_start_transaction(root, 1);
+ trans = btrfs_start_transaction(quota_root, 1);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
+ btrfs_put_root(quota_root);
goto drop_write;
}
@@ -3781,6 +3814,7 @@ static long btrfs_ioctl_qgroup_limit(struct file *file, void __user *arg)
ret = btrfs_limit_qgroup(trans, qgroupid, &sa->lim);
err = btrfs_end_transaction(trans);
+ btrfs_put_root(quota_root);
if (err && !ret)
ret = err;
diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c
index ffb654d36391..1ee52a9dcee3 100644
--- a/fs/btrfs/raid56.c
+++ b/fs/btrfs/raid56.c
@@ -2997,13 +2997,11 @@ void raid56_parity_submit_scrub_rbio(struct btrfs_raid_bio *rbio)
* This is due to the fact rbio has its own page management for its cache.
*/
void raid56_parity_cache_data_folios(struct btrfs_raid_bio *rbio,
- struct folio **data_folios, u64 data_logical)
+ void *vaddr, u64 data_logical)
{
struct btrfs_fs_info *fs_info = rbio->bioc->fs_info;
const u64 offset_in_full_stripe = data_logical -
rbio->bioc->full_stripe_logical;
- unsigned int findex = 0;
- unsigned int foffset = 0;
int ret;
/*
@@ -3026,18 +3024,10 @@ void raid56_parity_cache_data_folios(struct btrfs_raid_bio *rbio,
cur_off < offset_in_full_stripe + BTRFS_STRIPE_LEN;
cur_off += PAGE_SIZE) {
const unsigned int pindex = cur_off >> PAGE_SHIFT;
- void *kaddr;
- kaddr = kmap_local_page(rbio->stripe_pages[pindex]);
- memcpy_from_folio(kaddr, data_folios[findex], foffset, PAGE_SIZE);
- kunmap_local(kaddr);
-
- foffset += PAGE_SIZE;
- ASSERT(foffset <= folio_size(data_folios[findex]));
- if (foffset == folio_size(data_folios[findex])) {
- findex++;
- foffset = 0;
- }
+ ASSERT(cur_off - offset_in_full_stripe + PAGE_SIZE <= BTRFS_STRIPE_LEN);
+ memcpy_to_page(rbio->stripe_pages[pindex], 0,
+ vaddr + cur_off - offset_in_full_stripe, PAGE_SIZE);
}
bitmap_set(rbio->stripe_uptodate_bitmap,
offset_in_full_stripe >> fs_info->sectorsize_bits,
diff --git a/fs/btrfs/raid56.h b/fs/btrfs/raid56.h
index 1f463ecf7e41..8542648199f1 100644
--- a/fs/btrfs/raid56.h
+++ b/fs/btrfs/raid56.h
@@ -283,7 +283,7 @@ struct btrfs_raid_bio *raid56_parity_alloc_scrub_rbio(struct bio *bio,
void raid56_parity_submit_scrub_rbio(struct btrfs_raid_bio *rbio);
void raid56_parity_cache_data_folios(struct btrfs_raid_bio *rbio,
- struct folio **data_folios, u64 data_logical);
+ void *vaddr, u64 data_logical);
int btrfs_alloc_stripe_hash_table(struct btrfs_fs_info *info);
void btrfs_free_stripe_hash_table(struct btrfs_fs_info *info);
diff --git a/fs/btrfs/reflink.c b/fs/btrfs/reflink.c
index 9a49d2ecb949..28bb05a92106 100644
--- a/fs/btrfs/reflink.c
+++ b/fs/btrfs/reflink.c
@@ -691,7 +691,7 @@ static int btrfs_extent_same_range(struct btrfs_inode *src, u64 loff, u64 len,
const u64 end = dst_loff + len - 1;
struct extent_state *cached_state = NULL;
struct btrfs_fs_info *fs_info = src->root->fs_info;
- const u64 bs = fs_info->sectorsize;
+ const u32 bs = fs_info->sectorsize;
int ret;
/*
@@ -762,7 +762,7 @@ static noinline int btrfs_clone_files(struct file *file, struct file *file_src,
struct btrfs_fs_info *fs_info = inode_to_fs_info(inode);
int ret;
u64 len = olen;
- u64 bs = fs_info->sectorsize;
+ const u32 bs = fs_info->sectorsize;
u64 end;
/*
@@ -841,7 +841,7 @@ static int btrfs_remap_file_range_prep(struct file *file_in, loff_t pos_in,
{
struct btrfs_inode *inode_in = BTRFS_I(file_inode(file_in));
struct btrfs_inode *inode_out = BTRFS_I(file_inode(file_out));
- u64 bs = inode_out->root->fs_info->sectorsize;
+ const u32 bs = inode_out->root->fs_info->sectorsize;
u64 wb_len;
int ret;
diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c
index fc5c14b5adad..da54db75e7a9 100644
--- a/fs/btrfs/relocation.c
+++ b/fs/btrfs/relocation.c
@@ -339,14 +339,15 @@ static struct btrfs_backref_node *walk_down_backref(
static bool reloc_root_is_dead(const struct btrfs_root *root)
{
+ if (test_bit(BTRFS_ROOT_DEAD_RELOC_TREE, &root->state))
+ return true;
/*
- * Pair with set_bit/clear_bit in clean_dirty_subvols and
- * btrfs_update_reloc_root. We need to see the updated bit before
- * trying to access reloc_root
+ * Pairs with set_bit/clear_bit in clear_reloc_root() and
+ * btrfs_update_reloc_root(). We need to see the updated bit before
+ * trying to access root->reloc_root in our callers.
*/
smp_rmb();
- if (test_bit(BTRFS_ROOT_DEAD_RELOC_TREE, &root->state))
- return true;
+
return false;
}
@@ -1537,6 +1538,33 @@ static void clear_reloc_root(struct btrfs_root *root)
clear_bit(BTRFS_ROOT_DEAD_RELOC_TREE, &root->state);
}
+/* Drop the reloc trees of a relocation that is being deferred and retried. */
+static void abort_reloc_roots(struct reloc_control *rc, struct list_head *list)
+{
+ struct btrfs_fs_info *fs_info = rc->extent_root->fs_info;
+ struct btrfs_root *reloc_root, *tmp;
+
+ list_for_each_entry_safe(reloc_root, tmp, list, root_list) {
+ struct btrfs_root *root;
+
+ root = btrfs_get_fs_root(fs_info, reloc_root->root_key.offset, false);
+ if (!IS_ERR(root)) {
+ if (root->reloc_root == reloc_root) {
+ clear_reloc_root(root);
+ btrfs_put_root(reloc_root);
+ }
+ btrfs_put_root(root);
+ }
+
+ btrfs_set_root_refs(&reloc_root->root_item, 0);
+ memset(&reloc_root->root_item.drop_progress, 0, sizeof(struct btrfs_disk_key));
+ btrfs_set_root_drop_level(&reloc_root->root_item, 0);
+
+ list_del_init(&reloc_root->root_list);
+ list_add_tail(&reloc_root->reloc_dirty_list, &rc->dirty_subvol_roots);
+ }
+}
+
static int clean_dirty_subvols(struct reloc_control *rc)
{
struct btrfs_root *root;
@@ -1876,8 +1904,7 @@ again:
return err;
}
-static noinline_for_stack
-void merge_reloc_roots(struct reloc_control *rc)
+static noinline_for_stack int merge_reloc_roots(struct reloc_control *rc)
{
struct btrfs_fs_info *fs_info = rc->extent_root->fs_info;
struct btrfs_root *root;
@@ -1975,7 +2002,15 @@ again:
goto again;
}
out:
- if (ret) {
+ if (btrfs_is_zoned(fs_info) && ret == -EAGAIN) {
+ abort_reloc_roots(rc, &reloc_roots);
+
+ /* New reloc root may be added. */
+ mutex_lock(&fs_info->reloc_mutex);
+ list_splice_init(&rc->reloc_roots, &reloc_roots);
+ mutex_unlock(&fs_info->reloc_mutex);
+ abort_reloc_roots(rc, &reloc_roots);
+ } else if (ret) {
btrfs_handle_fs_error(fs_info, ret, NULL);
free_reloc_roots(&reloc_roots);
@@ -2001,6 +2036,7 @@ out:
*
* The remaining nodes will be cleaned up by put_reloc_control().
*/
+ return ret;
}
static void free_block_list(struct rb_root *blocks)
@@ -3730,7 +3766,9 @@ restart:
*/
err = prepare_to_merge(rc, err);
- merge_reloc_roots(rc);
+ ret = merge_reloc_roots(rc);
+ if (ret && !err)
+ err = ret;
rc->merge_reloc_tree = false;
unset_reloc_control(rc);
@@ -4114,10 +4152,10 @@ static int copy_remapped_data(struct btrfs_fs_info *fs_info, u64 old_addr,
u64 new_addr, u64 length)
{
int ret;
- u64 copy_len = min_t(u64, length, SZ_1M);
+ const u64 copy_len = min_t(u64, length, SZ_1M);
struct page **pages;
struct reloc_io_private priv;
- unsigned int nr_pages = DIV_ROUND_UP(length, PAGE_SIZE);
+ const unsigned int nr_pages = DIV_ROUND_UP(copy_len, PAGE_SIZE);
pages = kzalloc_objs(struct page *, nr_pages, GFP_NOFS);
if (!pages)
@@ -5555,6 +5593,24 @@ static noinline_for_stack int mark_garbage_root(struct btrfs_root *root)
return ret;
}
+static void release_recovered_fs_roots(struct list_head *roots, bool drop_reloc_refs)
+{
+ struct btrfs_root *root;
+ struct btrfs_root *next;
+
+ list_for_each_entry_safe(root, next, roots, reloc_dirty_list) {
+ list_del_init(&root->reloc_dirty_list);
+ if (drop_reloc_refs) {
+ struct btrfs_root *reloc_root = root->reloc_root;
+
+ ASSERT(reloc_root);
+ root->reloc_root = NULL;
+ btrfs_put_root(reloc_root);
+ }
+ btrfs_put_root(root);
+ }
+}
+
/*
* recover relocation interrupted by system crash.
*
@@ -5564,6 +5620,7 @@ static noinline_for_stack int mark_garbage_root(struct btrfs_root *root)
int btrfs_recover_relocation(struct btrfs_fs_info *fs_info)
{
LIST_HEAD(reloc_roots);
+ LIST_HEAD(recovered_roots);
struct btrfs_key key;
struct btrfs_root *fs_root;
struct btrfs_root *reloc_root;
@@ -5680,7 +5737,7 @@ int btrfs_recover_relocation(struct btrfs_fs_info *fs_info)
ret = PTR_ERR(fs_root);
list_add_tail(&reloc_root->root_list, &reloc_roots);
btrfs_end_transaction(trans);
- goto out_unset;
+ goto out_drop_reloc_refs;
}
ret = __add_reloc_root(reloc_root, rc);
@@ -5689,17 +5746,21 @@ int btrfs_recover_relocation(struct btrfs_fs_info *fs_info)
list_add_tail(&reloc_root->root_list, &reloc_roots);
btrfs_put_root(fs_root);
btrfs_end_transaction(trans);
- goto out_unset;
+ goto out_drop_reloc_refs;
}
+ ASSERT(list_empty(&fs_root->reloc_dirty_list));
fs_root->reloc_root = btrfs_grab_root(reloc_root);
- btrfs_put_root(fs_root);
+ list_add_tail(&fs_root->reloc_dirty_list, &recovered_roots);
}
ret = btrfs_commit_transaction(trans);
if (ret)
- goto out_unset;
+ goto out_drop_reloc_refs;
+ release_recovered_fs_roots(&recovered_roots, false);
- merge_reloc_roots(rc);
+ ret = merge_reloc_roots(rc);
+ if (ret)
+ goto out_unset;
unset_reloc_control(rc);
@@ -5713,6 +5774,8 @@ out_clean:
ret2 = clean_dirty_subvols(rc);
if (ret2 < 0 && !ret)
ret = ret2;
+out_drop_reloc_refs:
+ release_recovered_fs_roots(&recovered_roots, true);
out_unset:
unset_reloc_control(rc);
reloc_chunk_end(fs_info);
diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c
index d2f7ac5b6e96..f209e75f0ff5 100644
--- a/fs/btrfs/scrub.c
+++ b/fs/btrfs/scrub.c
@@ -57,12 +57,6 @@ struct scrub_ctx;
#define SCRUB_TOTAL_STRIPES (SCRUB_GROUPS_PER_SCTX * SCRUB_STRIPES_PER_GROUP)
-/*
- * The following value times PAGE_SIZE needs to be large enough to match the
- * largest node/leaf/sector size that shall be supported.
- */
-#define SCRUB_MAX_SECTORS_PER_BLOCK (BTRFS_MAX_METADATA_BLOCKSIZE / SZ_4K)
-
/* Represent one sector and its needed info to verify the content. */
struct scrub_sector_verification {
union {
@@ -129,19 +123,17 @@ enum {
scrub_bitmap_nr_last,
};
-#define SCRUB_STRIPE_MAX_FOLIOS (BTRFS_STRIPE_LEN / PAGE_SIZE)
-
/*
* Represent one contiguous range with a length of BTRFS_STRIPE_LEN.
*/
struct scrub_stripe {
struct scrub_ctx *sctx;
struct btrfs_block_group *bg;
-
- struct folio *folios[SCRUB_STRIPE_MAX_FOLIOS];
struct scrub_sector_verification *sectors;
-
struct btrfs_device *dev;
+
+ void *buffer;
+
u64 logical;
u64 physical;
@@ -227,6 +219,9 @@ struct scrub_ctx {
refcount_t refs;
};
+static_assert(BTRFS_STRIPE_LEN >= PAGE_SIZE);
+static_assert(IS_ALIGNED(BTRFS_STRIPE_LEN, PAGE_SIZE));
+
#define scrub_calc_start_bit(stripe, name, block_nr) \
({ \
unsigned int __start_bit; \
@@ -338,13 +333,10 @@ static void release_scrub_stripe(struct scrub_stripe *stripe)
if (!stripe)
return;
- for (int i = 0; i < SCRUB_STRIPE_MAX_FOLIOS; i++) {
- if (stripe->folios[i])
- folio_put(stripe->folios[i]);
- stripe->folios[i] = NULL;
- }
+ kvfree(stripe->buffer);
kfree(stripe->sectors);
kfree(stripe->csums);
+ stripe->buffer = NULL;
stripe->sectors = NULL;
stripe->csums = NULL;
stripe->sctx = NULL;
@@ -354,9 +346,6 @@ static void release_scrub_stripe(struct scrub_stripe *stripe)
static int init_scrub_stripe(struct btrfs_fs_info *fs_info,
struct scrub_stripe *stripe)
{
- const u32 min_folio_shift = PAGE_SHIFT + fs_info->block_min_order;
- int ret;
-
memset(stripe, 0, sizeof(*stripe));
stripe->nr_sectors = BTRFS_STRIPE_LEN >> fs_info->sectorsize_bits;
@@ -367,11 +356,8 @@ static int init_scrub_stripe(struct btrfs_fs_info *fs_info,
atomic_set(&stripe->pending_io, 0);
spin_lock_init(&stripe->write_error_lock);
- ASSERT(BTRFS_STRIPE_LEN >> min_folio_shift <= SCRUB_STRIPE_MAX_FOLIOS);
- ret = btrfs_alloc_folio_array(BTRFS_STRIPE_LEN >> min_folio_shift,
- fs_info->block_min_order, stripe->folios,
- GFP_NOFS);
- if (ret < 0)
+ stripe->buffer = kvmalloc(BTRFS_STRIPE_LEN, GFP_NOFS);
+ if (!stripe->buffer)
goto error;
stripe->sectors = kzalloc_objs(struct scrub_sector_verification,
@@ -682,32 +668,18 @@ static int fill_writer_pointer_gap(struct scrub_ctx *sctx, u64 physical)
return ret;
}
-static void *scrub_stripe_get_kaddr(struct scrub_stripe *stripe, int sector_nr)
-{
- struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
- const u32 min_folio_shift = PAGE_SHIFT + fs_info->block_min_order;
- u32 offset = (sector_nr << fs_info->sectorsize_bits);
- const struct folio *folio = stripe->folios[offset >> min_folio_shift];
-
- /* stripe->folios[] is allocated by us and no highmem is allowed. */
- ASSERT(folio);
- ASSERT(!folio_test_highmem(folio));
- return folio_address(folio) + offset_in_folio(folio, offset);
-}
-
-static phys_addr_t scrub_stripe_get_paddr(struct scrub_stripe *stripe, int sector_nr)
+/*
+ * Unlike the existing csum which is based on paddr, this version is fully on
+ * vaddr, so no extra per-page iteration needed.
+ */
+static void scrub_calc_vaddr_csum(struct btrfs_fs_info *fs_info,
+ void *vaddr, unsigned int len, u8 *dest)
{
- struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
- const u32 min_folio_shift = PAGE_SHIFT + fs_info->block_min_order;
- u32 offset = (sector_nr << fs_info->sectorsize_bits);
- const struct folio *folio = stripe->folios[offset >> min_folio_shift];
+ struct btrfs_csum_ctx csum;
- /* stripe->folios[] is allocated by us and no highmem is allowed. */
- ASSERT(folio);
- ASSERT(!folio_test_highmem(folio));
- /* And the range must be contained inside the folio. */
- ASSERT(offset_in_folio(folio, offset) + fs_info->sectorsize <= folio_size(folio));
- return page_to_phys(folio_page(folio, 0)) + offset_in_folio(folio, offset);
+ btrfs_csum_init(&csum, fs_info->csum_type);
+ btrfs_csum_update(&csum, vaddr, len);
+ btrfs_csum_final(&csum, dest);
}
static void scrub_verify_one_metadata(struct scrub_stripe *stripe, int sector_nr)
@@ -715,19 +687,10 @@ static void scrub_verify_one_metadata(struct scrub_stripe *stripe, int sector_nr
struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
const u32 sectors_per_tree = fs_info->nodesize >> fs_info->sectorsize_bits;
const u64 logical = stripe->logical + (sector_nr << fs_info->sectorsize_bits);
- void *first_kaddr = scrub_stripe_get_kaddr(stripe, sector_nr);
- struct btrfs_header *header = first_kaddr;
- struct btrfs_csum_ctx csum;
- u8 on_disk_csum[BTRFS_CSUM_SIZE];
+ void *first_vaddr = stripe->buffer + (sector_nr << fs_info->sectorsize_bits);
+ struct btrfs_header *header = first_vaddr;
u8 calculated_csum[BTRFS_CSUM_SIZE];
- /*
- * Here we don't have a good way to attach the pages (and subpages)
- * to a dummy extent buffer, thus we have to directly grab the members
- * from pages.
- */
- memcpy(on_disk_csum, header->csum, fs_info->csum_size);
-
if (logical != btrfs_stack_header_bytenr(header)) {
scrub_bitmap_set_meta_error(stripe, sector_nr, sectors_per_tree);
scrub_bitmap_set_error(stripe, sector_nr, sectors_per_tree);
@@ -759,23 +722,15 @@ static void scrub_verify_one_metadata(struct scrub_stripe *stripe, int sector_nr
}
/* Now check tree block csum. */
- btrfs_csum_init(&csum, fs_info->csum_type);
- btrfs_csum_update(&csum, first_kaddr + BTRFS_CSUM_SIZE,
- fs_info->sectorsize - BTRFS_CSUM_SIZE);
-
- for (int i = sector_nr + 1; i < sector_nr + sectors_per_tree; i++) {
- btrfs_csum_update(&csum, scrub_stripe_get_kaddr(stripe, i),
- fs_info->sectorsize);
- }
-
- btrfs_csum_final(&csum, calculated_csum);
- if (memcmp(calculated_csum, on_disk_csum, fs_info->csum_size) != 0) {
+ scrub_calc_vaddr_csum(fs_info, first_vaddr + BTRFS_CSUM_SIZE,
+ fs_info->nodesize - BTRFS_CSUM_SIZE, calculated_csum);
+ if (memcmp(calculated_csum, header->csum, fs_info->csum_size) != 0) {
scrub_bitmap_set_meta_error(stripe, sector_nr, sectors_per_tree);
scrub_bitmap_set_error(stripe, sector_nr, sectors_per_tree);
btrfs_warn_rl(fs_info,
"scrub: tree block %llu mirror %u has bad csum, has " BTRFS_CSUM_FMT " want " BTRFS_CSUM_FMT,
logical, stripe->mirror_num,
- BTRFS_CSUM_FMT_VALUE(fs_info->csum_size, on_disk_csum),
+ BTRFS_CSUM_FMT_VALUE(fs_info->csum_size, header->csum),
BTRFS_CSUM_FMT_VALUE(fs_info->csum_size, calculated_csum));
return;
}
@@ -801,9 +756,7 @@ static void scrub_verify_one_sector(struct scrub_stripe *stripe, int sector_nr)
struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
struct scrub_sector_verification *sector = &stripe->sectors[sector_nr];
const u32 sectors_per_tree = fs_info->nodesize >> fs_info->sectorsize_bits;
- phys_addr_t paddr = scrub_stripe_get_paddr(stripe, sector_nr);
u8 csum_buf[BTRFS_CSUM_SIZE];
- int ret;
ASSERT(sector_nr >= 0 && sector_nr < stripe->nr_sectors);
@@ -846,8 +799,10 @@ static void scrub_verify_one_sector(struct scrub_stripe *stripe, int sector_nr)
return;
}
- ret = btrfs_check_block_csum(fs_info, paddr, csum_buf, sector->csum);
- if (ret < 0) {
+ scrub_calc_vaddr_csum(fs_info,
+ stripe->buffer + (sector_nr << fs_info->sectorsize_bits),
+ fs_info->sectorsize, csum_buf);
+ if (memcmp(csum_buf, sector->csum, fs_info->csum_size)) {
scrub_bitmap_set_bit_csum_error(stripe, sector_nr);
scrub_bitmap_set_bit_error(stripe, sector_nr);
} else {
@@ -870,43 +825,65 @@ static void scrub_verify_one_stripe(struct scrub_stripe *stripe, unsigned long b
}
}
-static int calc_sector_number(struct scrub_stripe *stripe, struct bio_vec *first_bvec)
+static unsigned int calc_sector_number(const struct btrfs_bio *bbio)
{
- int i;
+ const struct scrub_stripe *stripe = bbio->private;
+ const struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
- for (i = 0; i < stripe->nr_sectors; i++) {
- if (scrub_stripe_get_kaddr(stripe, i) == bvec_virt(first_bvec))
- break;
- }
- ASSERT(i < stripe->nr_sectors);
- return i;
+ /* Scrub bbios all have their @file_offset set to the logical bytenr. */
+ ASSERT(bbio->file_offset >= stripe->logical &&
+ bbio->file_offset < stripe->logical + (stripe->nr_sectors <<
+ fs_info->sectorsize_bits),
+ "scrub bio logical=%llu stripe logical=%llu stripe len=%u",
+ bbio->file_offset, stripe->logical,
+ stripe->nr_sectors << fs_info->sectorsize_bits);
+ return (bbio->file_offset - stripe->logical) >> fs_info->sectorsize_bits;
}
/*
- * Repair read is different to the regular read:
+ * Common handling of read endio.
*
- * - Only reads the failed sectors
- * - May have extra blocksize limits
+ * The bbio will be released, so no more access to @bbio after this function.
*/
-static void scrub_repair_read_endio(struct btrfs_bio *bbio)
+static void scrub_read_endio_common(struct btrfs_bio *bbio)
{
struct scrub_stripe *stripe = bbio->private;
struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
- int sector_nr = calc_sector_number(stripe, bio_first_bvec_all(&bbio->bio));
+ unsigned int sector_nr = calc_sector_number(bbio);
const u32 bio_size = bio_get_size(&bbio->bio);
+ const u32 sectors = bio_size >> fs_info->sectorsize_bits;
- ASSERT(sector_nr < stripe->nr_sectors);
+
+ /*
+ * For vmallocated space, readers need to call invalidate_kernel_vmap_range()
+ * to manage the coherency between kernel mapping and devie space mapping.
+ */
+ if (is_vmalloc_addr(stripe->buffer))
+ invalidate_kernel_vmap_range(
+ stripe->buffer + (sector_nr << fs_info->sectorsize_bits),
+ bio_size);
if (bbio->bio.bi_status) {
- scrub_bitmap_set_io_error(stripe, sector_nr,
- bio_size >> fs_info->sectorsize_bits);
- scrub_bitmap_set_error(stripe, sector_nr,
- bio_size >> fs_info->sectorsize_bits);
+ scrub_bitmap_set_io_error(stripe, sector_nr, sectors);
+ scrub_bitmap_set_error(stripe, sector_nr, sectors);
} else {
- scrub_bitmap_clear_io_error(stripe, sector_nr,
- bio_size >> fs_info->sectorsize_bits);
+ scrub_bitmap_clear_io_error(stripe, sector_nr, sectors);
}
bio_put(&bbio->bio);
+}
+
+/*
+ * Repair read is different to the regular read:
+ *
+ * - Only reads the failed sectors
+ * - May have extra blocksize limits
+ */
+static void scrub_repair_read_endio(struct btrfs_bio *bbio)
+{
+ struct scrub_stripe *stripe = bbio->private;
+
+ scrub_read_endio_common(bbio);
+
if (atomic_dec_and_test(&stripe->pending_io))
wake_up(&stripe->io_wait);
}
@@ -921,30 +898,35 @@ static void scrub_bio_add_sector(struct btrfs_bio *bbio, struct scrub_stripe *st
int sector_nr)
{
struct btrfs_fs_info *fs_info = bbio->inode->root->fs_info;
- void *kaddr = scrub_stripe_get_kaddr(stripe, sector_nr);
+ const u32 offset = sector_nr << fs_info->sectorsize_bits;
int ret;
- ret = bio_add_page(&bbio->bio, virt_to_page(kaddr), fs_info->sectorsize,
- offset_in_page(kaddr));
- /*
- * Caller should ensure the bbio has enough size.
- * And we cannot use __bio_add_page(), which doesn't do any merge.
- *
- * Meanwhile for scrub_submit_initial_read() we fully rely on the merge
- * to create the minimal amount of bio vectors, for fs block size < page
- * size cases.
- */
+ ASSERT(offset + fs_info->sectorsize <= BTRFS_STRIPE_LEN);
+
+ if (is_vmalloc_addr(stripe->buffer)) {
+ ret = bio_add_vmalloc(&bbio->bio, stripe->buffer + offset, fs_info->sectorsize);
+ ASSERT(ret == true);
+ return;
+ }
+ ret = bio_add_page(&bbio->bio, virt_to_page(stripe->buffer + offset),
+ fs_info->sectorsize, offset_in_page(stripe->buffer + offset));
ASSERT(ret == fs_info->sectorsize);
}
static struct btrfs_bio *alloc_scrub_bbio(struct btrfs_fs_info *fs_info,
- unsigned int nr_vecs, blk_opf_t opf,
+ blk_opf_t opf,
u64 logical,
btrfs_bio_end_io_t end_io, void *private)
{
struct btrfs_bio *bbio;
- bbio = btrfs_bio_alloc(nr_vecs, opf, BTRFS_I(fs_info->btree_inode),
+ /*
+ * Stripe->buffer is allocated by kvmalloc(), which can be pages at
+ * different physical addresses, we have to ensure the bbio is large
+ * enough to contain the full stripe.
+ */
+ bbio = btrfs_bio_alloc(BTRFS_STRIPE_LEN >> PAGE_SHIFT, opf,
+ BTRFS_I(fs_info->btree_inode),
logical, end_io, private);
bbio->is_scrub = true;
bbio->bio.bi_iter.bi_sector = logical >> SECTOR_SHIFT;
@@ -976,7 +958,7 @@ static void scrub_stripe_submit_repair_read(struct scrub_stripe *stripe,
}
if (!bbio)
- bbio = alloc_scrub_bbio(fs_info, stripe->nr_sectors, REQ_OP_READ,
+ bbio = alloc_scrub_bbio(fs_info, REQ_OP_READ,
stripe->logical + (i << fs_info->sectorsize_bits),
scrub_repair_read_endio, stripe);
@@ -1245,20 +1227,9 @@ out:
static void scrub_read_endio(struct btrfs_bio *bbio)
{
struct scrub_stripe *stripe = bbio->private;
- int sector_nr = calc_sector_number(stripe, bio_first_bvec_all(&bbio->bio));
- int num_sectors;
- const u32 bio_size = bio_get_size(&bbio->bio);
- ASSERT(sector_nr < stripe->nr_sectors);
- num_sectors = bio_size >> stripe->bg->fs_info->sectorsize_bits;
+ scrub_read_endio_common(bbio);
- if (bbio->bio.bi_status) {
- scrub_bitmap_set_io_error(stripe, sector_nr, num_sectors);
- scrub_bitmap_set_error(stripe, sector_nr, num_sectors);
- } else {
- scrub_bitmap_clear_io_error(stripe, sector_nr, num_sectors);
- }
- bio_put(&bbio->bio);
if (atomic_dec_and_test(&stripe->pending_io)) {
wake_up(&stripe->io_wait);
INIT_WORK(&stripe->work, scrub_stripe_read_repair_worker);
@@ -1270,7 +1241,7 @@ static void scrub_write_endio(struct btrfs_bio *bbio)
{
struct scrub_stripe *stripe = bbio->private;
struct btrfs_fs_info *fs_info = stripe->bg->fs_info;
- int sector_nr = calc_sector_number(stripe, bio_first_bvec_all(&bbio->bio));
+ unsigned int sector_nr = calc_sector_number(bbio);
const u32 bio_size = bio_get_size(&bbio->bio);
if (bbio->bio.bi_status) {
@@ -1349,7 +1320,7 @@ static void scrub_write_sectors(struct scrub_ctx *sctx, struct scrub_stripe *str
bbio = NULL;
}
if (!bbio)
- bbio = alloc_scrub_bbio(fs_info, stripe->nr_sectors, REQ_OP_WRITE,
+ bbio = alloc_scrub_bbio(fs_info, REQ_OP_WRITE,
stripe->logical + (sector_nr << fs_info->sectorsize_bits),
scrub_write_endio, stripe);
scrub_bio_add_sector(bbio, stripe, sector_nr);
@@ -1844,7 +1815,7 @@ static void scrub_submit_extent_sector_read(struct scrub_stripe *stripe)
continue;
}
- bbio = alloc_scrub_bbio(fs_info, stripe->nr_sectors, REQ_OP_READ,
+ bbio = alloc_scrub_bbio(fs_info, REQ_OP_READ,
logical, scrub_read_endio, stripe);
}
@@ -1869,7 +1840,6 @@ static void scrub_submit_initial_read(struct scrub_ctx *sctx,
{
struct btrfs_fs_info *fs_info = sctx->fs_info;
struct btrfs_bio *bbio;
- const u32 min_folio_shift = PAGE_SHIFT + fs_info->block_min_order;
unsigned int nr_sectors = stripe_length(stripe) >> fs_info->sectorsize_bits;
int mirror = stripe->mirror_num;
@@ -1882,7 +1852,7 @@ static void scrub_submit_initial_read(struct scrub_ctx *sctx,
return;
}
- bbio = alloc_scrub_bbio(fs_info, BTRFS_STRIPE_LEN >> min_folio_shift, REQ_OP_READ,
+ bbio = alloc_scrub_bbio(fs_info, REQ_OP_READ,
stripe->logical, scrub_read_endio, stripe);
/* Read the whole range inside the chunk boundary. */
for (unsigned int cur = 0; cur < nr_sectors; cur++)
@@ -2138,7 +2108,7 @@ static int scrub_raid56_cached_parity(struct scrub_ctx *sctx,
for (int i = 0; i < data_stripes; i++) {
struct scrub_stripe *stripe = &sctx->raid56_data_stripes[i];
- raid56_parity_cache_data_folios(rbio, stripe->folios,
+ raid56_parity_cache_data_folios(rbio, stripe->buffer,
full_stripe_start + (i << BTRFS_STRIPE_LEN_SHIFT));
}
raid56_parity_submit_scrub_rbio(rbio);
@@ -3091,14 +3061,6 @@ int btrfs_scrub_dev(struct btrfs_fs_info *fs_info, u64 devid, u64 start,
/* At mount time we have ensured nodesize is in the range of [4K, 64K]. */
ASSERT(fs_info->nodesize <= BTRFS_STRIPE_LEN);
- /*
- * SCRUB_MAX_SECTORS_PER_BLOCK is calculated using the largest possible
- * value (max nodesize / min sectorsize), thus nodesize should always
- * be fine.
- */
- ASSERT(fs_info->nodesize <=
- SCRUB_MAX_SECTORS_PER_BLOCK << fs_info->sectorsize_bits);
-
/* Allocate outside of device_list_mutex */
sctx = scrub_setup_ctx(fs_info, is_dev_replace);
if (IS_ERR(sctx))
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
index 297704edf1b4..dca3570168c7 100644
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -130,10 +130,10 @@ static_assert(offsetof(struct backref_cache_entry, entry) == 0);
#define SEND_MAX_DIR_CREATED_CACHE_SIZE 64
/*
- * Max number of entries in the cache that stores directories that were already
- * created. The cache uses raw struct btrfs_lru_cache_entry entries, so it uses
- * at most 4096 bytes - sizeof(struct btrfs_lru_cache_entry) is 48 bytes, but
- * the kmalloc-64 slab is used, so we get 4096 bytes (64 bytes * 64).
+ * Maximum number of entries in the cache that stores utimes values for directories.
+ * The cache uses raw struct btrfs_lru_cache_entry entries, so it uses at most
+ * 4096 bytes - sizeof(struct btrfs_lru_cache_entry) is 48 bytes, but the
+ * kmalloc-64 slab is used, so we get 4096 bytes (64 bytes * 64).
*/
#define SEND_MAX_DIR_UTIMES_CACHE_SIZE 64
@@ -625,9 +625,8 @@ static void fs_path_unreverse(struct fs_path *p)
static inline bool is_current_inode_path(const struct send_ctx *sctx,
const struct fs_path *path)
{
- const struct fs_path *cur = &sctx->cur_inode_path;
-
- return (strncmp(path->start, cur->start, fs_path_len(cur)) == 0);
+ /* Paths are always nul terminated. */
+ return (strcmp(path->start, sctx->cur_inode_path.start) == 0);
}
static struct btrfs_path *alloc_path_for_send(void)
@@ -6033,7 +6032,7 @@ static int send_write_or_clone(struct send_ctx *sctx,
int ret = 0;
u64 offset = key->offset;
u64 end;
- u64 bs = sctx->send_root->fs_info->sectorsize;
+ const u32 bs = sctx->send_root->fs_info->sectorsize;
struct btrfs_file_extent_item *ei;
u64 disk_byte;
u64 data_offset;
diff --git a/fs/btrfs/space-info.c b/fs/btrfs/space-info.c
index e6641597b321..39a28e1bec8a 100644
--- a/fs/btrfs/space-info.c
+++ b/fs/btrfs/space-info.c
@@ -2156,7 +2156,7 @@ again:
will_reclaim = true;
reclaim = true;
}
- bg->reclaim_mark++;
+ bg->reclaim_mark = true;
spin_unlock(&bg->lock);
if (reclaim)
btrfs_mark_bg_to_reclaim(bg);
diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c
index f4e34898d581..b745794595d1 100644
--- a/fs/btrfs/super.c
+++ b/fs/btrfs/super.c
@@ -129,7 +129,6 @@ enum {
/* Rescue options */
Opt_rescue,
- Opt_usebackuproot,
/* Debugging options */
Opt_enospc_debug,
@@ -249,8 +248,6 @@ static const struct fs_parameter_spec btrfs_fs_parameters[] = {
/* Rescue options. */
fsparam_enum("rescue", Opt_rescue, btrfs_parameter_rescue),
- /* Deprecated, with alias rescue=usebackuproot */
- __fsparam(NULL, "usebackuproot", Opt_usebackuproot, fs_param_deprecated, NULL),
/* For compatibility only, alias for "rescue=nologreplay". */
fsparam_flag("norecovery", Opt_norecovery),
@@ -514,19 +511,20 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
btrfs_clear_opt(ctx->mount_opt, NODISCARD);
break;
case Opt_space_cache:
- if (result.negated) {
- btrfs_set_opt(ctx->mount_opt, NOSPACECACHE);
- btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
- btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
- } else {
- btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
- btrfs_set_opt(ctx->mount_opt, SPACE_CACHE);
- }
+ if (!result.negated)
+ btrfs_warn(NULL,
+ "v1 space cache is deprecated, falling back to no space cache");
+ btrfs_set_opt(ctx->mount_opt, NOSPACECACHE);
+ btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
+ btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
break;
case Opt_space_cache_version:
switch (result.uint_32) {
case Opt_space_cache_v1:
- btrfs_set_opt(ctx->mount_opt, SPACE_CACHE);
+ btrfs_warn(NULL,
+ "v1 space cache is deprecated, falling back to no space cache");
+ btrfs_set_opt(ctx->mount_opt, NOSPACECACHE);
+ btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE);
btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE);
break;
case Opt_space_cache_v2:
@@ -560,14 +558,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
else
btrfs_set_opt(ctx->mount_opt, AUTO_DEFRAG);
break;
- case Opt_usebackuproot:
- btrfs_warn(NULL,
- "'usebackuproot' is deprecated, use 'rescue=usebackuproot' instead");
- btrfs_set_opt(ctx->mount_opt, USEBACKUPROOT);
-
- /* If we're loading the backup roots we can't trust the space cache. */
- btrfs_set_opt(ctx->mount_opt, CLEAR_CACHE);
- break;
case Opt_skip_balance:
btrfs_set_opt(ctx->mount_opt, SKIP_BALANCE);
break;
@@ -620,6 +610,7 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
btrfs_set_opt(ctx->mount_opt, IGNORESUPERFLAGS);
btrfs_set_opt(ctx->mount_opt, IGNOREBADROOTS);
btrfs_set_opt(ctx->mount_opt, NOLOGREPLAY);
+ btrfs_set_opt(ctx->mount_opt, USEBACKUPROOT);
break;
default:
btrfs_info(NULL, "unrecognized rescue option '%s'",
@@ -668,7 +659,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param)
*/
static void btrfs_clear_oneshot_options(struct btrfs_fs_info *fs_info)
{
- btrfs_clear_opt(fs_info->mount_opt, USEBACKUPROOT);
btrfs_clear_opt(fs_info->mount_opt, CLEAR_CACHE);
btrfs_clear_opt(fs_info->mount_opt, NOSPACECACHE);
}
@@ -692,7 +682,8 @@ bool btrfs_check_options(const struct btrfs_fs_info *info,
bool ret = true;
if (!(flags & SB_RDONLY) &&
- (check_ro_option(info, *mount_opt, BTRFS_MOUNT_NOLOGREPLAY, "nologreplay") ||
+ (check_ro_option(info, *mount_opt, BTRFS_MOUNT_USEBACKUPROOT, "usebackuproot") ||
+ check_ro_option(info, *mount_opt, BTRFS_MOUNT_NOLOGREPLAY, "nologreplay") ||
check_ro_option(info, *mount_opt, BTRFS_MOUNT_IGNOREBADROOTS, "ignorebadroots") ||
check_ro_option(info, *mount_opt, BTRFS_MOUNT_IGNOREDATACSUMS, "ignoredatacsums") ||
check_ro_option(info, *mount_opt, BTRFS_MOUNT_IGNOREMETACSUMS, "ignoremetacsums") ||
diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c
index 8f9419728100..99161ea69e28 100644
--- a/fs/btrfs/transaction.c
+++ b/fs/btrfs/transaction.c
@@ -456,12 +456,12 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans,
*
* When this is zero, they can trust root->last_trans and fly
* through btrfs_record_root_in_trans without having to take the
- * lock. smp_wmb() makes sure that all the writes above are
- * done before we pop in the zero below
+ * lock. smp_wmb() makes sure readers that see the last_trans
+ * update also see IN_TRANS_SETUP set, and clear_bit_unlock()
+ * publishes the relocation setup before we clear the bit.
*/
ret = btrfs_init_reloc_root(trans, root);
- smp_mb__before_atomic();
- clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state);
+ clear_bit_unlock(BTRFS_ROOT_IN_TRANS_SETUP, &root->state);
}
return ret;
}
@@ -499,10 +499,12 @@ int btrfs_record_root_in_trans(struct btrfs_trans_handle *trans,
* see record_root_in_trans for comments about IN_TRANS_SETUP usage
* and barriers
*/
- smp_rmb();
- if (btrfs_get_root_last_trans(root) == trans->transid &&
- !test_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state))
- return 0;
+ if (btrfs_get_root_last_trans(root) == trans->transid) {
+ /* Order the last_trans load before testing IN_TRANS_SETUP. */
+ smp_rmb();
+ if (!test_bit_acquire(BTRFS_ROOT_IN_TRANS_SETUP, &root->state))
+ return 0;
+ }
mutex_lock(&fs_info->reloc_mutex);
ret = record_root_in_trans(trans, root, false);
@@ -698,8 +700,6 @@ again:
goto alloc_fail;
}
- xa_init(&h->writeback_inhibited_ebs);
-
/*
* If we are JOIN_NOLOCK we're already committing a transaction and
* waiting on this guy, so we don't need to do the sb_start_intwrite
diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h
index 5e4b1106fd90..3a57f227b5ed 100644
--- a/fs/btrfs/transaction.h
+++ b/fs/btrfs/transaction.h
@@ -7,12 +7,12 @@
#define BTRFS_TRANSACTION_H
#include <linux/atomic.h>
+#include <linux/build_bug.h>
#include <linux/refcount.h>
#include <linux/list.h>
#include <linux/time64.h>
#include <linux/mutex.h>
#include <linux/wait.h>
-#include <linux/xarray.h>
#include "btrfs_inode.h"
#include "delayed-ref.h"
@@ -23,6 +23,7 @@ struct btrfs_fs_info;
struct btrfs_root_item;
struct btrfs_root;
struct btrfs_path;
+struct extent_buffer;
/*
* Signal that a direct IO write is in progress, to avoid deadlock for sync
@@ -136,6 +137,18 @@ enum {
#define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH)
+/*
+ * Number of extent buffers a transaction handle tracks for writeback
+ * inhibition. The CLOCK reference bits pack into a u32 so this must not exceed
+ * 32, and keeping it a power of two lets the compiler reduce the CLOCK hand
+ * modulo to a mask.
+ */
+#define BTRFS_INHIBITED_EBS_SLOTS 8
+
+static_assert(BTRFS_INHIBITED_EBS_SLOTS <= 32);
+static_assert(BTRFS_INHIBITED_EBS_SLOTS != 0 &&
+ (BTRFS_INHIBITED_EBS_SLOTS & (BTRFS_INHIBITED_EBS_SLOTS - 1)) == 0);
+
struct btrfs_trans_handle {
u64 transid;
u64 bytes_reserved;
@@ -163,8 +176,14 @@ struct btrfs_trans_handle {
struct btrfs_fs_info *fs_info;
struct list_head new_bgs;
struct btrfs_block_rsv delayed_rsv;
- /* Extent buffers with writeback inhibited by this handle. */
- struct xarray writeback_inhibited_ebs;
+
+ /* Extent buffers this handle has inhibited writeback on. */
+ struct extent_buffer *inhibited_ebs[BTRFS_INHIBITED_EBS_SLOTS];
+ /* CLOCK reference bit per slot. */
+ u32 inhibited_ebs_referenced;
+ u32 nr_inhibited_ebs;
+ /* CLOCK hand. */
+ u32 inhibited_ebs_hand;
};
/*
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
index 6eab4cc73ce4..9b66eb584ece 100644
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -12,6 +12,7 @@
#include <linux/uuid.h>
#include <linux/list_sort.h>
#include <linux/namei.h>
+#include <linux/fs_struct.h>
#include "misc.h"
#include "disk-io.h"
#include "extent-tree.h"
@@ -480,7 +481,12 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
struct block_device *bdev;
int ret;
- *bdev_file = bdev_file_open_by_path(device_path, flags, holder, &fs_holder_ops);
+ if (holder)
+ *bdev_file = fs_bdev_file_open_by_path(device_path, flags,
+ holder, holder);
+ else
+ *bdev_file = bdev_file_open_by_path(device_path, flags, NULL,
+ NULL);
if (IS_ERR(*bdev_file)) {
ret = PTR_ERR(*bdev_file);
@@ -495,7 +501,7 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
if (holder) {
ret = set_blocksize(*bdev_file, BTRFS_BDEV_BLOCKSIZE);
if (ret) {
- bdev_fput(*bdev_file);
+ fs_bdev_file_release(*bdev_file, holder);
goto error;
}
}
@@ -503,7 +509,10 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
*disk_super = btrfs_read_disk_super(bdev, 0, false);
if (IS_ERR(*disk_super)) {
ret = PTR_ERR(*disk_super);
- bdev_fput(*bdev_file);
+ if (holder)
+ fs_bdev_file_release(*bdev_file, holder);
+ else
+ bdev_fput(*bdev_file);
goto error;
}
@@ -727,7 +736,7 @@ static int btrfs_open_one_device(struct btrfs_fs_devices *fs_devices,
error_free_page:
btrfs_release_disk_super(disk_super);
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, holder);
return -EINVAL;
}
@@ -740,41 +749,6 @@ const u8 *btrfs_sb_fsid_ptr(const struct btrfs_super_block *sb)
return has_metadata_uuid ? sb->metadata_uuid : sb->fsid;
}
-static bool is_same_device(struct btrfs_device *device, const char *new_path)
-{
- struct path old = { .mnt = NULL, .dentry = NULL };
- struct path new = { .mnt = NULL, .dentry = NULL };
- char AUTO_KFREE(old_path);
- bool is_same = false;
- int ret;
-
- if (!device->name)
- goto out;
-
- old_path = kzalloc(PATH_MAX, GFP_NOFS);
- if (!old_path)
- goto out;
-
- rcu_read_lock();
- ret = strscpy(old_path, rcu_dereference(device->name), PATH_MAX);
- rcu_read_unlock();
- if (ret < 0)
- goto out;
-
- ret = kern_path(old_path, LOOKUP_FOLLOW, &old);
- if (ret)
- goto out;
- ret = kern_path(new_path, LOOKUP_FOLLOW, &new);
- if (ret)
- goto out;
- if (path_equal(&old, &new))
- is_same = true;
-out:
- path_put(&old);
- path_put(&new);
- return is_same;
-}
-
/*
* Add new device to list of registered devices
*
@@ -895,7 +869,7 @@ static noinline struct btrfs_device *device_list_add(const char *path,
MAJOR(path_devt), MINOR(path_devt),
current->comm, task_pid_nr(current));
- } else if (!device->name || !is_same_device(device, path)) {
+ } else if (!device->name || device->devt != path_devt) {
const char *old_name;
/*
@@ -1087,7 +1061,7 @@ static void __btrfs_free_extra_devids(struct btrfs_fs_devices *fs_devices,
continue;
if (device->bdev_file) {
- bdev_fput(device->bdev_file);
+ fs_bdev_file_release(device->bdev_file, device->bdev_file->private_data);
device->bdev = NULL;
device->bdev_file = NULL;
fs_devices->open_devices--;
@@ -1124,7 +1098,18 @@ void btrfs_free_extra_devids(struct btrfs_fs_devices *fs_devices)
mutex_unlock(&uuid_mutex);
}
-static void btrfs_close_bdev(struct btrfs_device *device)
+/* Release a device that was made unfreezable for a membership change. */
+void btrfs_release_device_allow_freeze(struct file *bdev_file)
+{
+ struct super_block *sb = bdev_file->private_data;
+
+ /* Unregister before re-allowing (strand-safe); file still open (UAF-safe). */
+ fs_bdev_unregister(bdev_file, sb);
+ bdev_allow_freeze(file_bdev(bdev_file));
+ bdev_fput(bdev_file);
+}
+
+static void btrfs_close_bdev(struct btrfs_device *device, bool allow_freeze)
{
if (!device->bdev)
return;
@@ -1134,7 +1119,12 @@ static void btrfs_close_bdev(struct btrfs_device *device)
invalidate_bdev(device->bdev);
}
- bdev_fput(device->bdev_file);
+ /* @allow_freeze undoes a replace-time deny; unmount-close was never denied. */
+ if (allow_freeze)
+ btrfs_release_device_allow_freeze(device->bdev_file);
+ else
+ fs_bdev_file_release(device->bdev_file,
+ device->bdev_file->private_data);
}
static void btrfs_close_one_device(struct btrfs_device *device)
@@ -1155,7 +1145,7 @@ static void btrfs_close_one_device(struct btrfs_device *device)
fs_devices->missing_devices--;
}
- btrfs_close_bdev(device);
+ btrfs_close_bdev(device, false);
if (device->bdev) {
fs_devices->open_devices--;
device->bdev = NULL;
@@ -2125,8 +2115,16 @@ static int btrfs_add_dev_item(struct btrfs_trans_handle *trans,
static void update_dev_time(const char *device_path)
{
struct path path;
+ int err;
- if (!kern_path(device_path, LOOKUP_FOLLOW, &path)) {
+ if (tsk_is_kthread(current)) {
+ scoped_with_init_fs()
+ err = kern_path(device_path, LOOKUP_FOLLOW, &path);
+ } else {
+ err = kern_path(device_path, LOOKUP_FOLLOW, &path);
+ }
+
+ if (!err) {
vfs_utimes(&path, NULL);
path_put(&path);
}
@@ -2373,6 +2371,13 @@ int btrfs_rm_device(struct btrfs_fs_info *fs_info,
fs_info->fs_devices->rw_devices == 1)
return BTRFS_ERROR_DEV_ONLY_WRITABLE;
+ /* Removal and freezing are mutually exclusive; refuse if frozen now. */
+ if (device->bdev) {
+ ret = bdev_deny_freeze(device->bdev);
+ if (ret)
+ return ret;
+ }
+
if (test_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state)) {
mutex_lock(&fs_info->chunk_mutex);
list_del_init(&device->dev_alloc_list);
@@ -2399,6 +2404,8 @@ int btrfs_rm_device(struct btrfs_fs_info *fs_info,
device->devid, ret);
btrfs_abort_transaction(trans, ret);
btrfs_end_transaction(trans);
+ if (device->bdev)
+ bdev_allow_freeze(device->bdev);
return ret;
}
@@ -2490,6 +2497,8 @@ int btrfs_rm_device(struct btrfs_fs_info *fs_info,
return btrfs_commit_transaction(trans);
error_undo:
+ if (device->bdev)
+ bdev_allow_freeze(device->bdev);
if (test_bit(BTRFS_DEV_STATE_WRITEABLE, &device->dev_state)) {
mutex_lock(&fs_info->chunk_mutex);
list_add(&device->dev_alloc_list,
@@ -2534,7 +2543,8 @@ void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev)
mutex_lock(&uuid_mutex);
- btrfs_close_bdev(srcdev);
+ /* The source was made unfreezable for the replace; undo it. */
+ btrfs_close_bdev(srcdev, true);
synchronize_rcu();
btrfs_free_device(srcdev);
@@ -2555,7 +2565,8 @@ void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev)
mutex_unlock(&uuid_mutex);
}
-void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev)
+void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev,
+ bool allow_freeze)
{
struct btrfs_fs_devices *fs_devices = tgtdev->fs_info->fs_devices;
@@ -2576,7 +2587,7 @@ void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev)
btrfs_scratch_superblocks(tgtdev->fs_info, tgtdev);
- btrfs_close_bdev(tgtdev);
+ btrfs_close_bdev(tgtdev, allow_freeze);
synchronize_rcu();
btrfs_free_device(tgtdev);
}
@@ -2845,6 +2856,37 @@ next_slot:
return 0;
}
+/*
+ * Open @path for @sb with freezing denied before the holder claim is published,
+ * so a racing bdev_freeze() can never reach a claim a device add or replace may
+ * still abort. The deny is taken on a throwaway non-holder probe open, then the
+ * holder is opened by the probe's dev_t. Balanced by the caller.
+ */
+struct file *btrfs_open_device_deny_freeze(const char *path,
+ struct super_block *sb)
+{
+ struct file *probe_file, *bdev_file;
+ int ret;
+
+ /* WRITE so bdev_file_open_by_path() rejects a read-only device. */
+ probe_file = bdev_file_open_by_path(path, BLK_OPEN_WRITE, NULL, NULL);
+ if (IS_ERR(probe_file))
+ return probe_file;
+
+ ret = bdev_deny_freeze(file_bdev(probe_file));
+ if (ret) {
+ bdev_fput(probe_file);
+ return ERR_PTR(ret);
+ }
+
+ bdev_file = fs_bdev_file_open_by_dev(file_bdev(probe_file)->bd_dev,
+ BLK_OPEN_WRITE, sb, sb);
+ if (IS_ERR(bdev_file))
+ bdev_allow_freeze(file_bdev(probe_file));
+ bdev_fput(probe_file);
+ return bdev_file;
+}
+
int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path)
{
struct btrfs_root *root = fs_info->dev_root;
@@ -2863,8 +2905,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
if (sb_rdonly(sb) && !fs_devices->seeding)
return -EROFS;
- bdev_file = bdev_file_open_by_path(device_path, BLK_OPEN_WRITE,
- fs_info->sb, &fs_holder_ops);
+ /* Forbid freezing until the device is a committed member (or unwound). */
+ bdev_file = btrfs_open_device_deny_freeze(device_path, fs_info->sb);
if (IS_ERR(bdev_file))
return PTR_ERR(bdev_file);
@@ -3035,8 +3077,10 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
up_write(&sb->s_umount);
locked = false;
- if (ret) /* transaction commit */
+ if (ret) { /* transaction commit */
+ bdev_allow_freeze(file_bdev(bdev_file));
return ret;
+ }
ret = btrfs_relocate_sys_chunks(fs_info);
if (ret < 0)
@@ -3044,8 +3088,10 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
"Failed to relocate sys chunks after device initialization. This can be fixed using the \"btrfs balance\" command.");
trans = btrfs_attach_transaction(root);
if (IS_ERR(trans)) {
- if (PTR_ERR(trans) == -ENOENT)
+ if (PTR_ERR(trans) == -ENOENT) {
+ bdev_allow_freeze(file_bdev(bdev_file));
return 0;
+ }
ret = PTR_ERR(trans);
trans = NULL;
goto error_sysfs;
@@ -3065,6 +3111,7 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
/* Update ctime/mtime for blkid or udev */
update_dev_time(device_path);
+ bdev_allow_freeze(file_bdev(bdev_file));
return ret;
error_sysfs:
@@ -3094,7 +3141,7 @@ error_free_zone:
error_free_device:
btrfs_free_device(device);
error:
- bdev_fput(bdev_file);
+ btrfs_release_device_allow_freeze(bdev_file);
if (locked) {
mutex_unlock(&uuid_mutex);
up_write(&sb->s_umount);
@@ -4588,7 +4635,7 @@ again:
if (ret == -ENOSPC) {
enospc_errors++;
} else if (ret == -ETXTBSY) {
- btrfs_info(fs_info,
+ btrfs_warn(fs_info,
"skipping relocation of block group %llu due to active swapfile",
found_key.offset);
ret = 0;
@@ -6049,6 +6096,19 @@ struct btrfs_chunk_map *btrfs_alloc_chunk_map(int num_stripes, gfp_t gfp)
return map;
}
+static void set_real_chunk_type(struct btrfs_chunk_map *map)
+{
+ map->type = map->on_disk_type;
+ if (likely((map->on_disk_type & BTRFS_BLOCK_GROUP_RAID56_MASK) == 0 ||
+ nr_data_stripes(map) > 1))
+ return;
+ if (map->on_disk_type & BTRFS_BLOCK_GROUP_RAID5)
+ map->type |= BTRFS_BLOCK_GROUP_RAID1;
+ else
+ map->type |= BTRFS_BLOCK_GROUP_RAID1C3;
+ map->type &= ~BTRFS_BLOCK_GROUP_RAID56_MASK;
+}
+
static struct btrfs_block_group *create_chunk(struct btrfs_trans_handle *trans,
struct alloc_chunk_ctl *ctl,
struct btrfs_device_info *devices_info)
@@ -6067,11 +6127,10 @@ static struct btrfs_block_group *create_chunk(struct btrfs_trans_handle *trans,
map->start = start;
map->chunk_len = ctl->chunk_size;
map->stripe_size = ctl->stripe_size;
- map->type = type;
- map->io_align = BTRFS_STRIPE_LEN;
- map->io_width = BTRFS_STRIPE_LEN;
+ map->on_disk_type = type;
map->sub_stripes = ctl->sub_stripes;
map->num_stripes = ctl->num_stripes;
+ set_real_chunk_type(map);
for (int i = 0; i < ctl->ndevs; i++) {
for (int j = 0; j < ctl->dev_stripes; j++) {
@@ -6250,7 +6309,7 @@ int btrfs_chunk_alloc_add_chunk_item(struct btrfs_trans_handle *trans,
btrfs_set_stack_chunk_length(chunk, bg->length);
btrfs_set_stack_chunk_owner(chunk, BTRFS_EXTENT_TREE_OBJECTID);
btrfs_set_stack_chunk_stripe_len(chunk, BTRFS_STRIPE_LEN);
- btrfs_set_stack_chunk_type(chunk, map->type);
+ btrfs_set_stack_chunk_type(chunk, map->on_disk_type);
btrfs_set_stack_chunk_num_stripes(chunk, map->num_stripes);
btrfs_set_stack_chunk_io_align(chunk, BTRFS_STRIPE_LEN);
btrfs_set_stack_chunk_io_width(chunk, BTRFS_STRIPE_LEN);
@@ -7632,9 +7691,7 @@ static int read_one_chunk(struct btrfs_key *key, struct extent_buffer *leaf,
map->start = logical;
map->chunk_len = length;
map->num_stripes = num_stripes;
- map->io_width = btrfs_chunk_io_width(leaf, chunk);
- map->io_align = btrfs_chunk_io_align(leaf, chunk);
- map->type = type;
+ map->on_disk_type = type;
/*
* We can't use the sub_stripes value, as for profiles other than
* RAID10, they may have 0 as sub_stripes for filesystems created by
@@ -7645,6 +7702,7 @@ static int read_one_chunk(struct btrfs_key *key, struct extent_buffer *leaf,
*/
map->sub_stripes = btrfs_raid_array[index].sub_stripes;
map->verified_stripes = 0;
+ set_real_chunk_type(map);
if (num_stripes > 0)
map->stripe_size = btrfs_calc_stripe_length(map);
diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h
index 63be45c3298c..0415d74cad9b 100644
--- a/fs/btrfs/volumes.h
+++ b/fs/btrfs/volumes.h
@@ -632,9 +632,15 @@ struct btrfs_chunk_map {
u64 start;
u64 chunk_len;
u64 stripe_size;
+ /*
+ * The real type that is utilized during logical address mapping.
+ *
+ * For most profiles it matches @on_disk_type, but for single-data-RAID56,
+ * the real type will be set to RAID1/RAID1C3, to avoid unsupported
+ * operations from raid56 lib.
+ */
u64 type;
- int io_align;
- int io_width;
+ u64 on_disk_type;
int num_stripes;
int sub_stripes;
struct btrfs_io_stripe stripes[];
@@ -744,6 +750,7 @@ int btrfs_open_devices(struct btrfs_fs_devices *fs_devices,
struct btrfs_device *btrfs_scan_one_device(const char *path, bool mount_arg_dev);
int btrfs_forget_devices(dev_t devt);
void btrfs_close_devices(struct btrfs_fs_devices *fs_devices);
+void btrfs_release_device_allow_freeze(struct file *bdev_file);
void btrfs_free_extra_devids(struct btrfs_fs_devices *fs_devices);
void btrfs_assign_next_active_device(struct btrfs_device *device,
struct btrfs_device *this_dev);
@@ -768,6 +775,8 @@ struct btrfs_device *btrfs_find_device(const struct btrfs_fs_devices *fs_devices
const struct btrfs_dev_lookup_args *args);
int btrfs_shrink_device(struct btrfs_device *device, u64 new_size);
int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *path);
+struct file *btrfs_open_device_deny_freeze(const char *path,
+ struct super_block *sb);
int btrfs_balance(struct btrfs_fs_info *fs_info,
struct btrfs_balance_control *bctl,
struct btrfs_ioctl_balance_args *bargs);
@@ -788,7 +797,8 @@ int btrfs_init_writeback_bio_size(struct btrfs_fs_info *fs_info);
int btrfs_run_dev_stats(struct btrfs_trans_handle *trans);
void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev);
void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev);
-void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev);
+void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev,
+ bool allow_freeze);
unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info,
u64 logical);
u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map);
diff --git a/fs/buffer.c b/fs/buffer.c
index 9af5f061a1f8..b62e7de421b5 100644
--- a/fs/buffer.c
+++ b/fs/buffer.c
@@ -336,7 +336,7 @@ still_busy:
spin_unlock_irqrestore(&first->b_uptodate_lock, flags);
}
-struct postprocess_bh_ctx {
+struct verify_bh_ctx {
struct work_struct work;
struct buffer_head *bh;
struct fsverity_info *vi;
@@ -344,8 +344,8 @@ struct postprocess_bh_ctx {
static void verify_bh(struct work_struct *work)
{
- struct postprocess_bh_ctx *ctx =
- container_of(work, struct postprocess_bh_ctx, work);
+ struct verify_bh_ctx *ctx =
+ container_of(work, struct verify_bh_ctx, work);
struct buffer_head *bh = ctx->bh;
bool valid;
@@ -355,29 +355,6 @@ static void verify_bh(struct work_struct *work)
kfree(ctx);
}
-static void decrypt_bh(struct work_struct *work)
-{
- struct postprocess_bh_ctx *ctx =
- container_of(work, struct postprocess_bh_ctx, work);
- struct buffer_head *bh = ctx->bh;
- int err;
-
- err = fscrypt_decrypt_pagecache_blocks(bh->b_folio, bh->b_size,
- bh_offset(bh));
- if (err == 0 && ctx->vi) {
- /*
- * We use different work queues for decryption and for verity
- * because verity may require reading metadata pages that need
- * decryption, and we shouldn't recurse to the same workqueue.
- */
- INIT_WORK(&ctx->work, verify_bh);
- fsverity_enqueue_verify_work(&ctx->work);
- return;
- }
- end_buffer_async_read(bh, err == 0);
- kfree(ctx);
-}
-
/*
* I/O completion handler for block_read_full_folio() - folios
* which come unlocked at the end of I/O.
@@ -387,27 +364,21 @@ static void bh_end_async_read(struct bio *bio)
struct buffer_head *bh;
bool uptodate = bio_endio_bh(bio, &bh);
struct inode *inode = bh->b_folio->mapping->host;
- bool decrypt = fscrypt_inode_uses_fs_layer_crypto(inode);
struct fsverity_info *vi = NULL;
/* needed by ext4 */
if (bh->b_folio->index < DIV_ROUND_UP(inode->i_size, PAGE_SIZE))
vi = fsverity_get_info(inode);
- /* Decrypt (with fscrypt) and/or verify (with fsverity) if needed. */
- if (uptodate && (decrypt || vi)) {
- struct postprocess_bh_ctx *ctx = kmalloc_obj(*ctx, GFP_ATOMIC);
+ /* Verify (with fsverity) if needed. */
+ if (vi && uptodate) {
+ struct verify_bh_ctx *ctx = kmalloc_obj(*ctx, GFP_ATOMIC);
if (ctx) {
ctx->bh = bh;
ctx->vi = vi;
- if (decrypt) {
- INIT_WORK(&ctx->work, decrypt_bh);
- fscrypt_enqueue_decrypt_work(&ctx->work);
- } else {
- INIT_WORK(&ctx->work, verify_bh);
- fsverity_enqueue_verify_work(&ctx->work);
- }
+ INIT_WORK(&ctx->work, verify_bh);
+ fsverity_enqueue_verify_work(&ctx->work);
return;
}
uptodate = false;
@@ -2177,6 +2148,7 @@ void block_commit_write(struct folio *folio, size_t from, size_t to)
{
size_t block_start, block_end;
bool partial = false;
+ bool uptodate = folio_test_uptodate(folio);
unsigned blocksize;
struct buffer_head *bh, *head;
@@ -2199,6 +2171,8 @@ void block_commit_write(struct folio *folio, size_t from, size_t to)
clear_buffer_new(bh);
block_start = block_end;
+ if (uptodate && block_start >= to)
+ break;
bh = bh->b_this_page;
} while (bh != head);
diff --git a/fs/ceph/dir.c b/fs/ceph/dir.c
index ef9e92e362d3..dfa590a220ea 100644
--- a/fs/ceph/dir.c
+++ b/fs/ceph/dir.c
@@ -983,7 +983,7 @@ out:
}
static int ceph_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return ceph_mknod(idmap, dir, dentry, mode, 0);
}
@@ -1147,7 +1147,6 @@ static struct dentry *ceph_mkdir(struct mnt_idmap *idmap, struct inode *dir,
goto out;
}
- mode |= S_IFDIR;
req->r_new_inode = ceph_new_inode(dir, dentry, &mode, &as_ctx);
if (IS_ERR(req->r_new_inode)) {
ret = ERR_CAST(req->r_new_inode);
diff --git a/fs/coda/dir.c b/fs/coda/dir.c
index 835eb7fdfdad..67148edfadee 100644
--- a/fs/coda/dir.c
+++ b/fs/coda/dir.c
@@ -134,7 +134,7 @@ static inline void coda_dir_drop_nlink(struct inode *dir)
/* creation routines: create, mknod, mkdir, link, symlink */
static int coda_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *de, umode_t mode, bool excl)
+ struct dentry *de, umode_t mode)
{
int error;
const char *name=de->d_name.name;
@@ -179,7 +179,12 @@ static struct dentry *coda_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (is_root_inode(dir) && coda_iscontrol(name, len))
return ERR_PTR(-EPERM);
- attrs.va_mode = mode;
+ /*
+ * vfs_mkdir() now passes S_IFDIR in @mode, but @mode is forwarded
+ * verbatim to userspace, which has only ever been given the permission
+ * bits. Strip the type bit until venus is known to cope with it.
+ */
+ attrs.va_mode = mode & ~S_IFDIR;
error = venus_mkdir(dir->i_sb, coda_i2f(dir),
name, len, &newfid, &attrs);
if (error)
diff --git a/fs/coredump.c b/fs/coredump.c
index e68a76ff92a3..ac3cd74808c6 100644
--- a/fs/coredump.c
+++ b/fs/coredump.c
@@ -921,15 +921,10 @@ static bool coredump_file(struct core_name *cn, struct coredump_params *cprm,
* with a fully qualified path" rule is to control where
* coredumps may be placed using root privileges,
* current->fs->root must not be used. Instead, use the
- * root directory of init_task.
+ * root directory of PID 1.
*/
- struct path root;
-
- task_lock(&init_task);
- get_fs_root(init_task.fs, &root);
- task_unlock(&init_task);
- file = file_open_root(&root, cn->corename, open_flags, 0600);
- path_put(&root);
+ scoped_with_init_fs()
+ file = filp_open(cn->corename, open_flags, 0600);
} else {
file = filp_open(cn->corename, open_flags, 0600);
}
diff --git a/fs/cramfs/inode.c b/fs/cramfs/inode.c
index 4edbfccd0bbe..d4cd03f4f60d 100644
--- a/fs/cramfs/inode.c
+++ b/fs/cramfs/inode.c
@@ -504,7 +504,7 @@ static void cramfs_kill_sb(struct super_block *sb)
sb->s_mtd = NULL;
} else if (IS_ENABLED(CONFIG_CRAMFS_BLOCKDEV) && sb->s_bdev) {
sync_blockdev(sb->s_bdev);
- bdev_fput(sb->s_bdev_file);
+ fs_bdev_file_release(sb->s_bdev_file, sb);
}
kfree(sbi);
}
diff --git a/fs/crypto/Kconfig b/fs/crypto/Kconfig
index 983d8ad1f417..cd934e31dec4 100644
--- a/fs/crypto/Kconfig
+++ b/fs/crypto/Kconfig
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: GPL-2.0-only
config FS_ENCRYPTION
bool "FS Encryption (Per-file encryption)"
+ select BLK_INLINE_ENCRYPTION if BLOCK
+ select BLK_INLINE_ENCRYPTION_FALLBACK if BLOCK
select CRYPTO
select CRYPTO_SKCIPHER
select CRYPTO_LIB_AES
@@ -34,7 +36,5 @@ config FS_ENCRYPTION_ALGS
select CRYPTO_XTS
config FS_ENCRYPTION_INLINE_CRYPT
- bool "Enable fscrypt to use inline crypto"
- depends on FS_ENCRYPTION && BLK_INLINE_ENCRYPTION
- help
- Enable fscrypt to use inline encryption hardware if available.
+ bool
+ default y if FS_ENCRYPTION && BLOCK
diff --git a/fs/crypto/Makefile b/fs/crypto/Makefile
index 652c7180ec6d..b03e02f0f09d 100644
--- a/fs/crypto/Makefile
+++ b/fs/crypto/Makefile
@@ -10,5 +10,4 @@ fscrypto-y := crypto.o \
keysetup_v1.o \
policy.o
-fscrypto-$(CONFIG_BLOCK) += bio.o
-fscrypto-$(CONFIG_FS_ENCRYPTION_INLINE_CRYPT) += inline_crypt.o
+fscrypto-$(CONFIG_BLOCK) += block.o
diff --git a/fs/crypto/bio.c b/fs/crypto/bio.c
deleted file mode 100644
index d07740680602..000000000000
--- a/fs/crypto/bio.c
+++ /dev/null
@@ -1,216 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * Utility functions for file contents encryption/decryption on
- * block device-based filesystems.
- *
- * Copyright (C) 2015, Google, Inc.
- * Copyright (C) 2015, Motorola Mobility
- */
-
-#include <linux/bio.h>
-#include <linux/export.h>
-#include <linux/module.h>
-#include <linux/namei.h>
-#include <linux/pagemap.h>
-
-#include "fscrypt_private.h"
-
-/**
- * fscrypt_decrypt_bio() - decrypt the contents of a bio
- * @bio: the bio to decrypt
- *
- * Decrypt the contents of a "read" bio following successful completion of the
- * underlying disk read. The bio must be reading a whole number of blocks of an
- * encrypted file directly into the page cache. If the bio is reading the
- * ciphertext into bounce pages instead of the page cache (for example, because
- * the file is also compressed, so decompression is required after decryption),
- * then this function isn't applicable. This function may sleep, so it must be
- * called from a workqueue rather than from the bio's bi_end_io callback.
- *
- * Return: %true on success; %false on failure. On failure, bio->bi_status is
- * also set to an error status.
- */
-bool fscrypt_decrypt_bio(struct bio *bio)
-{
- struct folio_iter fi;
-
- bio_for_each_folio_all(fi, bio) {
- int err = fscrypt_decrypt_pagecache_blocks(fi.folio, fi.length,
- fi.offset);
-
- if (err) {
- bio->bi_status = errno_to_blk_status(err);
- return false;
- }
- }
- return true;
-}
-EXPORT_SYMBOL(fscrypt_decrypt_bio);
-
-struct fscrypt_zero_done {
- atomic_t pending;
- blk_status_t status;
- struct completion done;
-};
-
-static void fscrypt_zeroout_range_done(struct fscrypt_zero_done *done)
-{
- if (atomic_dec_and_test(&done->pending))
- complete(&done->done);
-}
-
-static void fscrypt_zeroout_range_end_io(struct bio *bio)
-{
- struct fscrypt_zero_done *done = bio->bi_private;
-
- if (bio->bi_status)
- cmpxchg(&done->status, 0, bio->bi_status);
- fscrypt_zeroout_range_done(done);
- bio_put(bio);
-}
-
-static int fscrypt_zeroout_range_inline_crypt(const struct inode *inode,
- loff_t pos, sector_t sector,
- u64 len)
-{
- struct fscrypt_zero_done done = {
- .pending = ATOMIC_INIT(1),
- .done = COMPLETION_INITIALIZER_ONSTACK(done.done),
- };
-
- while (len) {
- struct bio *bio;
- unsigned int n;
-
- bio = bio_alloc(inode->i_sb->s_bdev, BIO_MAX_VECS, REQ_OP_WRITE,
- GFP_NOFS);
- bio->bi_iter.bi_sector = sector;
- bio->bi_private = &done;
- bio->bi_end_io = fscrypt_zeroout_range_end_io;
- fscrypt_set_bio_crypt_ctx(bio, inode, pos, GFP_NOFS);
-
- for (n = 0; n < BIO_MAX_VECS; n++) {
- unsigned int bytes_this_page = min(len, PAGE_SIZE);
-
- __bio_add_page(bio, ZERO_PAGE(0), bytes_this_page, 0);
- len -= bytes_this_page;
- pos += bytes_this_page;
- sector += (bytes_this_page >> SECTOR_SHIFT);
- if (!len || !fscrypt_mergeable_bio(bio, inode, pos))
- break;
- }
-
- atomic_inc(&done.pending);
- blk_crypto_submit_bio(bio);
- }
-
- fscrypt_zeroout_range_done(&done);
-
- wait_for_completion(&done.done);
- return blk_status_to_errno(done.status);
-}
-
-/**
- * fscrypt_zeroout_range() - zero out a range of blocks in an encrypted file
- * @inode: the file's inode
- * @pos: the first file position (in bytes) to zero out
- * @sector: the first sector to zero out
- * @len: bytes to zero out
- *
- * Zero out filesystem blocks in an encrypted regular file on-disk, i.e. write
- * ciphertext blocks which decrypt to the all-zeroes block. The blocks must be
- * both logically and physically contiguous. It's also assumed that the
- * filesystem only uses a single block device, ->s_bdev. @len must be a
- * multiple of the file system logical block size.
- *
- * Note that since each block uses a different IV, this involves writing a
- * different ciphertext to each block; we can't simply reuse the same one.
- *
- * Return: 0 on success; -errno on failure.
- */
-int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
- sector_t sector, u64 len)
-{
- const struct fscrypt_inode_info *ci = fscrypt_get_inode_info_raw(inode);
- const unsigned int du_bits = ci->ci_data_unit_bits;
- const unsigned int du_size = 1U << du_bits;
- const unsigned int du_per_page_bits = PAGE_SHIFT - du_bits;
- const unsigned int du_per_page = 1U << du_per_page_bits;
- u64 du_index = pos >> du_bits;
- u64 du_remaining = len >> du_bits;
- struct page *pages[16]; /* write up to 16 pages at a time */
- unsigned int nr_pages;
- unsigned int i;
- unsigned int offset;
- struct bio *bio;
- int ret, err;
-
- if (len == 0)
- return 0;
-
- if (fscrypt_inode_uses_inline_crypto(inode))
- return fscrypt_zeroout_range_inline_crypt(inode, pos, sector,
- len);
-
- BUILD_BUG_ON(ARRAY_SIZE(pages) > BIO_MAX_VECS);
- nr_pages = min_t(u64, ARRAY_SIZE(pages),
- (du_remaining + du_per_page - 1) >> du_per_page_bits);
-
- /*
- * We need at least one page for ciphertext. Allocate the first one
- * from a mempool, with __GFP_DIRECT_RECLAIM set so that it can't fail.
- *
- * Any additional page allocations are allowed to fail, as they only
- * help performance, and waiting on the mempool for them could deadlock.
- */
- for (i = 0; i < nr_pages; i++) {
- pages[i] = fscrypt_alloc_bounce_page(i == 0 ? GFP_NOFS :
- GFP_NOWAIT);
- if (!pages[i])
- break;
- }
- nr_pages = i;
- if (WARN_ON_ONCE(nr_pages <= 0))
- return -EINVAL;
-
- /* This always succeeds since __GFP_DIRECT_RECLAIM is set. */
- bio = bio_alloc(inode->i_sb->s_bdev, nr_pages, REQ_OP_WRITE, GFP_NOFS);
-
- do {
- bio->bi_iter.bi_sector = sector;
-
- i = 0;
- offset = 0;
- do {
- err = fscrypt_crypt_data_unit(ci, FS_ENCRYPT, du_index,
- ZERO_PAGE(0), pages[i],
- du_size, offset);
- if (err)
- goto out;
- du_index++;
- sector += 1U << (du_bits - SECTOR_SHIFT);
- du_remaining--;
- offset += du_size;
- if (offset == PAGE_SIZE || du_remaining == 0) {
- ret = bio_add_page(bio, pages[i++], offset, 0);
- if (WARN_ON_ONCE(ret != offset)) {
- err = -EIO;
- goto out;
- }
- offset = 0;
- }
- } while (i != nr_pages && du_remaining != 0);
-
- err = submit_bio_wait(bio);
- if (err)
- goto out;
- bio_reset(bio, inode->i_sb->s_bdev, REQ_OP_WRITE);
- } while (du_remaining != 0);
- err = 0;
-out:
- bio_put(bio);
- for (i = 0; i < nr_pages; i++)
- fscrypt_free_bounce_page(pages[i]);
- return err;
-}
-EXPORT_SYMBOL(fscrypt_zeroout_range);
diff --git a/fs/crypto/inline_crypt.c b/fs/crypto/block.c
index 66b9c9150fed..5193f8ba3ee0 100644
--- a/fs/crypto/inline_crypt.c
+++ b/fs/crypto/block.c
@@ -1,20 +1,20 @@
// SPDX-License-Identifier: GPL-2.0
/*
- * Inline encryption support for fscrypt
+ * File contents en/decryption on block-based filesystems
*
* Copyright 2019 Google LLC
*/
/*
- * With "inline encryption", the block layer handles the decryption/encryption
- * as part of the bio, instead of the filesystem doing the crypto itself via
- * crypto API. See Documentation/block/inline-encryption.rst. fscrypt still
- * provides the key and IV to use.
+ * This file implements fscrypt's file contents en/decryption using blk-crypto
+ * (Documentation/block/inline-encryption.rst). fscrypt assigns a bio_crypt_ctx
+ * with a key and IV to each bio, and the block layer does the en/decryption.
+ *
+ * This file's exported functions are called only by block-based filesystems.
*/
#include <linux/blk-crypto.h>
#include <linux/blkdev.h>
-#include <linux/buffer_head.h>
#include <linux/export.h>
#include <linux/sched/mm.h>
#include <linux/slab.h>
@@ -62,84 +62,19 @@ static unsigned int fscrypt_get_dun_bytes(const struct fscrypt_inode_info *ci)
* helpful for debugging problems where the "wrong" implementation is used.
*/
static void fscrypt_log_blk_crypto_impl(struct fscrypt_mode *mode,
- struct block_device **devs,
- unsigned int num_devs,
- const struct blk_crypto_config *cfg)
+ struct block_device *dev,
+ const struct blk_crypto_key *blk_key)
{
- unsigned int i;
-
- for (i = 0; i < num_devs; i++) {
- if (!IS_ENABLED(CONFIG_BLK_INLINE_ENCRYPTION_FALLBACK) ||
- blk_crypto_config_supported_natively(devs[i], cfg)) {
- if (!xchg(&mode->logged_blk_crypto_native, 1))
- pr_info("fscrypt: %s using blk-crypto (native)\n",
- mode->friendly_name);
- } else if (!xchg(&mode->logged_blk_crypto_fallback, 1)) {
- pr_info("fscrypt: %s using blk-crypto-fallback\n",
+ if (blk_crypto_config_supported_natively(dev, &blk_key->crypto_cfg)) {
+ if (!xchg(&mode->logged_blk_crypto_native, 1))
+ pr_info("fscrypt: %s using blk-crypto (native)\n",
mode->friendly_name);
- }
+ } else if (!xchg(&mode->logged_blk_crypto_fallback, 1)) {
+ pr_info("fscrypt: %s using blk-crypto-fallback\n",
+ mode->friendly_name);
}
}
-/* Enable inline encryption for this file if supported. */
-int fscrypt_select_encryption_impl(struct fscrypt_inode_info *ci,
- bool is_hw_wrapped_key)
-{
- const struct inode *inode = ci->ci_inode;
- struct super_block *sb = inode->i_sb;
- struct blk_crypto_config crypto_cfg;
- struct block_device *devs[FSCRYPT_MAX_DEVICES];
- unsigned int num_devs;
- unsigned int i;
-
- /* The file must need contents encryption, not filenames encryption */
- if (!S_ISREG(inode->i_mode))
- return 0;
-
- /* The crypto mode must have a blk-crypto counterpart */
- if (ci->ci_mode->blk_crypto_mode == BLK_ENCRYPTION_MODE_INVALID)
- return 0;
-
- /* The filesystem must be mounted with -o inlinecrypt */
- if (!(sb->s_flags & SB_INLINECRYPT))
- return 0;
-
- /*
- * When a page contains multiple logically contiguous filesystem blocks,
- * some filesystem code only calls fscrypt_mergeable_bio() for the first
- * block in the page. This is fine for most of fscrypt's IV generation
- * strategies, where contiguous blocks imply contiguous IVs. But it
- * doesn't work with IV_INO_LBLK_32. For now, simply exclude
- * IV_INO_LBLK_32 with blocksize != PAGE_SIZE from inline encryption.
- */
- if ((fscrypt_policy_flags(&ci->ci_policy) &
- FSCRYPT_POLICY_FLAG_IV_INO_LBLK_32) &&
- sb->s_blocksize != PAGE_SIZE)
- return 0;
-
- /*
- * On all the filesystem's block devices, blk-crypto must support the
- * crypto configuration that the file would use.
- */
- crypto_cfg.crypto_mode = ci->ci_mode->blk_crypto_mode;
- crypto_cfg.data_unit_size = 1U << ci->ci_data_unit_bits;
- crypto_cfg.dun_bytes = fscrypt_get_dun_bytes(ci);
- crypto_cfg.key_type = is_hw_wrapped_key ?
- BLK_CRYPTO_KEY_TYPE_HW_WRAPPED : BLK_CRYPTO_KEY_TYPE_RAW;
-
- num_devs = fscrypt_get_devices(sb, devs);
- for (i = 0; i < num_devs; i++) {
- if (!blk_crypto_config_supported(devs[i], &crypto_cfg))
- return 0;
- }
-
- fscrypt_log_blk_crypto_impl(ci->ci_mode, devs, num_devs, &crypto_cfg);
-
- ci->ci_inlinecrypt = true;
-
- return 0;
-}
-
int fscrypt_prepare_inline_crypt_key(struct fscrypt_prepared_key *prep_key,
const u8 *key_bytes, size_t key_size,
bool is_hw_wrapped,
@@ -147,7 +82,8 @@ int fscrypt_prepare_inline_crypt_key(struct fscrypt_prepared_key *prep_key,
{
const struct inode *inode = ci->ci_inode;
struct super_block *sb = inode->i_sb;
- enum blk_crypto_mode_num crypto_mode = ci->ci_mode->blk_crypto_mode;
+ bool inlinecrypt = sb->s_flags & SB_INLINECRYPT;
+ struct fscrypt_mode *mode = ci->ci_mode;
enum blk_crypto_key_type key_type = is_hw_wrapped ?
BLK_CRYPTO_KEY_TYPE_HW_WRAPPED : BLK_CRYPTO_KEY_TYPE_RAW;
struct blk_crypto_key *blk_key;
@@ -156,15 +92,28 @@ int fscrypt_prepare_inline_crypt_key(struct fscrypt_prepared_key *prep_key,
unsigned int i;
int err;
+ if (is_hw_wrapped && !inlinecrypt) {
+ /*
+ * blk_crypto_init_key() would catch this anyway, but this
+ * provides a clearer error message.
+ */
+ fscrypt_err(
+ inode,
+ "Hardware-wrapped keys require inline encryption (-o inlinecrypt)");
+ return -EINVAL;
+ }
+
blk_key = kmalloc_obj(*blk_key);
if (!blk_key)
return -ENOMEM;
err = blk_crypto_init_key(blk_key, key_bytes, key_size, key_type,
- crypto_mode, fscrypt_get_dun_bytes(ci),
- 1U << ci->ci_data_unit_bits);
+ mode->blk_crypto_mode,
+ fscrypt_get_dun_bytes(ci),
+ 1U << ci->ci_data_unit_bits,
+ inlinecrypt ? BLK_CRYPTO_CFG_ALLOW_HW : 0);
if (err) {
- fscrypt_err(inode, "error %d initializing blk-crypto key", err);
+ fscrypt_err(inode, "Error %d initializing blk-crypto key", err);
goto fail;
}
@@ -174,9 +123,16 @@ int fscrypt_prepare_inline_crypt_key(struct fscrypt_prepared_key *prep_key,
err = blk_crypto_start_using_key(devs[i], blk_key);
if (err)
break;
+ fscrypt_log_blk_crypto_impl(mode, devs[i], blk_key);
}
if (err) {
- fscrypt_err(inode, "error %d starting to use blk-crypto", err);
+ if (err == -EOPNOTSUPP && is_hw_wrapped)
+ fscrypt_err(
+ inode,
+ "Hardware-wrapped key required, but no suitable inline encryption capabilities are available");
+ else
+ fscrypt_err(inode,
+ "Error %d starting to use blk-crypto", err);
goto fail;
}
@@ -238,12 +194,6 @@ int fscrypt_derive_sw_secret(struct super_block *sb,
return err;
}
-bool __fscrypt_inode_uses_inline_crypto(const struct inode *inode)
-{
- return fscrypt_get_inode_info_raw(inode)->ci_inlinecrypt;
-}
-EXPORT_SYMBOL_GPL(__fscrypt_inode_uses_inline_crypto);
-
static void fscrypt_generate_dun(const struct fscrypt_inode_info *ci,
loff_t pos, u64 dun[BLK_CRYPTO_DUN_ARRAY_SIZE])
{
@@ -266,8 +216,8 @@ static void fscrypt_generate_dun(const struct fscrypt_inode_info *ci,
* @gfp_mask: memory allocation flags - these must be a waiting mask so that
* bio_crypt_set_ctx can't fail.
*
- * If the contents of the file should be encrypted (or decrypted) with inline
- * encryption, then assign the appropriate encryption context to the bio.
+ * If the contents of the file should be encrypted (or decrypted), then assign
+ * the appropriate encryption context to the bio.
*
* Normally the bio should be newly allocated (i.e. no pages added yet), as
* otherwise fscrypt_mergeable_bio() won't work as intended.
@@ -280,7 +230,7 @@ void fscrypt_set_bio_crypt_ctx(struct bio *bio, const struct inode *inode,
const struct fscrypt_inode_info *ci;
u64 dun[BLK_CRYPTO_DUN_ARRAY_SIZE];
- if (!fscrypt_inode_uses_inline_crypto(inode))
+ if (!fscrypt_needs_contents_encryption(inode))
return;
ci = fscrypt_get_inode_info_raw(inode);
@@ -295,12 +245,12 @@ EXPORT_SYMBOL_GPL(fscrypt_set_bio_crypt_ctx);
* @inode: the inode for the next part of the I/O
* @pos: the next file position (in bytes) in the I/O
*
- * When building a bio which may contain data which should undergo inline
- * encryption (or decryption) via fscrypt, filesystems should call this function
- * to ensure that the resulting bio contains only contiguous data unit numbers.
- * This will return false if the next part of the I/O cannot be merged with the
- * bio because either the encryption key would be different or the encryption
- * data unit numbers would be discontiguous.
+ * When building a bio which may contain data which should undergo encryption
+ * (or decryption) via fscrypt, filesystems should call this function to ensure
+ * that the resulting bio contains only contiguous data unit numbers. This will
+ * return false if the next part of the I/O cannot be merged with the bio
+ * because either the encryption key would be different or the encryption data
+ * unit numbers would be discontiguous.
*
* fscrypt_set_bio_crypt_ctx() must have already been called on the bio.
*
@@ -317,7 +267,7 @@ bool fscrypt_mergeable_bio(struct bio *bio, const struct inode *inode,
const struct fscrypt_inode_info *ci;
u64 next_dun[BLK_CRYPTO_DUN_ARRAY_SIZE];
- if (!!bc != fscrypt_inode_uses_inline_crypto(inode))
+ if (!!bc != fscrypt_needs_contents_encryption(inode))
return false;
if (!bc)
return true;
@@ -337,49 +287,6 @@ bool fscrypt_mergeable_bio(struct bio *bio, const struct inode *inode,
EXPORT_SYMBOL_GPL(fscrypt_mergeable_bio);
/**
- * fscrypt_dio_supported() - check whether DIO (direct I/O) is supported on an
- * inode, as far as encryption is concerned
- * @inode: the inode in question
- *
- * Return: %true if there are no encryption constraints that prevent DIO from
- * being supported; %false if DIO is unsupported. (Note that in the
- * %true case, the filesystem might have other, non-encryption-related
- * constraints that prevent DIO from actually being supported. Also, on
- * encrypted files the filesystem is still responsible for only allowing
- * DIO when requests are filesystem-block-aligned.)
- */
-bool fscrypt_dio_supported(struct inode *inode)
-{
- int err;
-
- /* If the file is unencrypted, no veto from us. */
- if (!fscrypt_needs_contents_encryption(inode))
- return true;
-
- /*
- * We only support DIO with inline crypto, not fs-layer crypto.
- *
- * To determine whether the inode is using inline crypto, we have to set
- * up the key if it wasn't already done. This is because in the current
- * design of fscrypt, the decision of whether to use inline crypto or
- * not isn't made until the inode's encryption key is being set up. In
- * the DIO read/write case, the key will always be set up already, since
- * the file will be open. But in the case of statx(), the key might not
- * be set up yet, as the file might not have been opened yet.
- */
- err = fscrypt_require_key(inode);
- if (err) {
- /*
- * Key unavailable or couldn't be set up. This edge case isn't
- * worth worrying about; just report that DIO is unsupported.
- */
- return false;
- }
- return fscrypt_inode_uses_inline_crypto(inode);
-}
-EXPORT_SYMBOL_GPL(fscrypt_dio_supported);
-
-/**
* fscrypt_limit_io_blocks() - limit I/O blocks to avoid discontiguous DUNs
* @inode: the file on which I/O is being done
* @lblk: the block at which the I/O is being started from
@@ -404,7 +311,7 @@ u64 fscrypt_limit_io_blocks(const struct inode *inode, u64 lblk, u64 nr_blocks)
const struct fscrypt_inode_info *ci;
u32 dun;
- if (!fscrypt_inode_uses_inline_crypto(inode))
+ if (!fscrypt_needs_contents_encryption(inode))
return nr_blocks;
if (nr_blocks <= 1)
@@ -422,3 +329,87 @@ u64 fscrypt_limit_io_blocks(const struct inode *inode, u64 lblk, u64 nr_blocks)
return min_t(u64, nr_blocks, (u64)U32_MAX + 1 - dun);
}
EXPORT_SYMBOL_GPL(fscrypt_limit_io_blocks);
+
+struct fscrypt_zero_done {
+ atomic_t pending;
+ blk_status_t status;
+ struct completion done;
+};
+
+static void fscrypt_zeroout_range_done(struct fscrypt_zero_done *done)
+{
+ if (atomic_dec_and_test(&done->pending))
+ complete(&done->done);
+}
+
+static void fscrypt_zeroout_range_end_io(struct bio *bio)
+{
+ struct fscrypt_zero_done *done = bio->bi_private;
+
+ if (bio->bi_status)
+ cmpxchg(&done->status, 0, bio->bi_status);
+ fscrypt_zeroout_range_done(done);
+ bio_put(bio);
+}
+
+/**
+ * fscrypt_zeroout_range() - zero out a range of blocks in an encrypted file
+ * @inode: the file's inode
+ * @pos: the first file position (in bytes) to zero out
+ * @sector: the first sector to zero out
+ * @len: bytes to zero out
+ *
+ * Zero out filesystem blocks in an encrypted regular file on-disk, i.e. write
+ * ciphertext blocks which decrypt to the all-zeroes block. The blocks must be
+ * both logically and physically contiguous. It's also assumed that the
+ * filesystem only uses a single block device, ->s_bdev. @len must be a
+ * multiple of the file system logical block size.
+ *
+ * Note that since each block uses a different IV, this involves writing a
+ * different ciphertext to each block; we can't simply reuse the same one.
+ *
+ * Return: 0 on success; -errno on failure.
+ */
+int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
+ sector_t sector, u64 len)
+{
+ struct fscrypt_zero_done done = {
+ .pending = ATOMIC_INIT(1),
+ .done = COMPLETION_INITIALIZER_ONSTACK(done.done),
+ };
+
+ if (len == 0)
+ return 0;
+
+ do {
+ struct bio *bio;
+ unsigned int n;
+
+ bio = bio_alloc(inode->i_sb->s_bdev, BIO_MAX_VECS, REQ_OP_WRITE,
+ GFP_NOFS);
+ bio->bi_iter.bi_sector = sector;
+ bio->bi_private = &done;
+ bio->bi_end_io = fscrypt_zeroout_range_end_io;
+ fscrypt_set_bio_crypt_ctx(bio, inode, pos, GFP_NOFS);
+
+ for (n = 0; n < BIO_MAX_VECS; n++) {
+ unsigned int bytes_this_page = min(len, PAGE_SIZE);
+
+ __bio_add_page(bio, ZERO_PAGE(0), bytes_this_page, 0);
+ len -= bytes_this_page;
+ pos += bytes_this_page;
+ sector += (bytes_this_page >> SECTOR_SHIFT);
+ if (!len || !fscrypt_mergeable_bio(bio, inode, pos))
+ break;
+ }
+
+ atomic_inc(&done.pending);
+ blk_crypto_submit_bio(bio);
+ } while (len);
+
+ fscrypt_zeroout_range_done(&done);
+
+ wait_for_completion(&done.done);
+ return blk_status_to_errno(done.status);
+}
+EXPORT_SYMBOL(fscrypt_zeroout_range);
diff --git a/fs/crypto/crypto.c b/fs/crypto/crypto.c
index 570a2231c945..5286a124b0d9 100644
--- a/fs/crypto/crypto.c
+++ b/fs/crypto/crypto.c
@@ -38,18 +38,11 @@ MODULE_PARM_DESC(num_prealloc_crypto_pages,
static mempool_t *fscrypt_bounce_page_pool = NULL;
-static struct workqueue_struct *fscrypt_read_workqueue;
static DEFINE_MUTEX(fscrypt_init_mutex);
struct kmem_cache *fscrypt_inode_info_cachep;
-void fscrypt_enqueue_decrypt_work(struct work_struct *work)
-{
- queue_work(fscrypt_read_workqueue, work);
-}
-EXPORT_SYMBOL(fscrypt_enqueue_decrypt_work);
-
-struct page *fscrypt_alloc_bounce_page(gfp_t gfp_flags)
+static struct page *fscrypt_alloc_bounce_page(gfp_t gfp_flags)
{
if (WARN_ON_ONCE(!fscrypt_bounce_page_pool)) {
/*
@@ -65,8 +58,7 @@ struct page *fscrypt_alloc_bounce_page(gfp_t gfp_flags)
* fscrypt_free_bounce_page() - free a ciphertext bounce page
* @bounce_page: the bounce page to free, or NULL
*
- * Free a bounce page that was allocated by fscrypt_encrypt_pagecache_blocks(),
- * or by fscrypt_alloc_bounce_page() directly.
+ * Free a bounce page that was allocated by fscrypt_encrypt_pagecache_blocks().
*/
void fscrypt_free_bounce_page(struct page *bounce_page)
{
@@ -91,7 +83,7 @@ void fscrypt_generate_iv(union fscrypt_iv *iv, u64 index,
{
u8 flags = fscrypt_policy_flags(&ci->ci_policy);
- memset(iv, 0, ci->ci_mode->ivsize);
+ memset(iv, 0, sizeof(*iv));
if (flags & FSCRYPT_POLICY_FLAG_IV_INO_LBLK_64) {
WARN_ON_ONCE(index > U32_MAX);
@@ -107,17 +99,23 @@ void fscrypt_generate_iv(union fscrypt_iv *iv, u64 index,
}
/* Encrypt or decrypt a single "data unit" of file contents. */
-int fscrypt_crypt_data_unit(const struct fscrypt_inode_info *ci,
- fscrypt_direction_t rw, u64 index,
- struct page *src_page, struct page *dest_page,
- unsigned int len, unsigned int offs)
+static int fscrypt_crypt_data_unit(const struct fscrypt_inode_info *ci,
+ fscrypt_direction_t rw, u64 index,
+ struct page *src_page,
+ struct page *dest_page, unsigned int len,
+ unsigned int offs)
{
- struct crypto_sync_skcipher *tfm = ci->ci_enc_key.tfm;
- SYNC_SKCIPHER_REQUEST_ON_STACK(req, tfm);
+ struct crypto_sync_skcipher *tfm;
union fscrypt_iv iv;
struct scatterlist dst, src;
int err;
+ if (WARN_ON_ONCE(ci == NULL)) /* File hasn't been opened yet? */
+ return -ENOKEY;
+ tfm = ci->ci_enc_key.tfm;
+ if (WARN_ON_ONCE(tfm == NULL)) /* Called on block-based filesystem? */
+ return -ENOKEY;
+
if (WARN_ON_ONCE(len <= 0))
return -EINVAL;
if (WARN_ON_ONCE(len % FSCRYPT_CONTENTS_ALIGNMENT != 0))
@@ -125,18 +123,22 @@ int fscrypt_crypt_data_unit(const struct fscrypt_inode_info *ci,
fscrypt_generate_iv(&iv, index, ci);
- skcipher_request_set_callback(
- req, CRYPTO_TFM_REQ_MAY_BACKLOG | CRYPTO_TFM_REQ_MAY_SLEEP,
- NULL, NULL);
- sg_init_table(&dst, 1);
- sg_set_page(&dst, dest_page, len, offs);
- sg_init_table(&src, 1);
- sg_set_page(&src, src_page, len, offs);
- skcipher_request_set_crypt(req, &src, &dst, len, &iv);
- if (rw == FS_DECRYPT)
- err = crypto_skcipher_decrypt(req);
- else
- err = crypto_skcipher_encrypt(req);
+ {
+ SYNC_SKCIPHER_REQUEST_ON_STACK(req, tfm);
+ skcipher_request_set_callback(req,
+ CRYPTO_TFM_REQ_MAY_BACKLOG |
+ CRYPTO_TFM_REQ_MAY_SLEEP,
+ NULL, NULL);
+ sg_init_table(&dst, 1);
+ sg_set_page(&dst, dest_page, len, offs);
+ sg_init_table(&src, 1);
+ sg_set_page(&src, src_page, len, offs);
+ skcipher_request_set_crypt(req, &src, &dst, len, &iv);
+ if (rw == FS_DECRYPT)
+ err = crypto_skcipher_decrypt(req);
+ else
+ err = crypto_skcipher_encrypt(req);
+ }
if (err)
fscrypt_err(ci->ci_inode,
"%scryption failed for data unit %llu: %d",
@@ -160,7 +162,7 @@ int fscrypt_crypt_data_unit(const struct fscrypt_inode_info *ci,
* which the plaintext data was located in the source page. Any other parts of
* the bounce page will be left uninitialized.
*
- * This is for use by the filesystem's ->writepages() method.
+ * This is for use by the ->writepages() method of non-block-based filesystems.
*
* The bounce page allocation is mempool-backed, so it will always succeed when
* @gfp_flags includes __GFP_DIRECT_RECLAIM, e.g. when it's GFP_NOFS. However,
@@ -174,14 +176,20 @@ struct page *fscrypt_encrypt_pagecache_blocks(struct folio *folio,
{
const struct inode *inode = folio->mapping->host;
const struct fscrypt_inode_info *ci = fscrypt_get_inode_info_raw(inode);
- const unsigned int du_bits = ci->ci_data_unit_bits;
- const unsigned int du_size = 1U << du_bits;
+ unsigned int du_bits;
+ unsigned int du_size;
struct page *ciphertext_page;
- u64 index = ((u64)folio->index << (PAGE_SHIFT - du_bits)) +
- (offs >> du_bits);
+ u64 index;
unsigned int i;
int err;
+ if (WARN_ON_ONCE(ci == NULL)) /* File hasn't been opened yet? */
+ return ERR_PTR(-ENOKEY);
+
+ du_bits = ci->ci_data_unit_bits;
+ du_size = 1U << du_bits;
+ index = (folio_pos(folio) + offs) >> du_bits;
+
VM_BUG_ON_FOLIO(folio_test_large(folio), folio);
if (WARN_ON_ONCE(!folio_test_locked(folio)))
return ERR_PTR(-EINVAL);
@@ -222,7 +230,8 @@ EXPORT_SYMBOL(fscrypt_encrypt_pagecache_blocks);
* arbitrary page, not necessarily in the original pagecache page. The @inode
* and @lblk_num must be specified, as they can't be determined from @page.
*
- * This is not compatible with fscrypt_operations::supports_subblock_data_units.
+ * This function only supports non-block-based filesystems that don't support
+ * sub-block data units (as indicated by the fscrypt_operations fields).
*
* Return: 0 on success; -errno on failure
*/
@@ -239,50 +248,6 @@ int fscrypt_encrypt_block_inplace(const struct inode *inode, struct page *page,
EXPORT_SYMBOL(fscrypt_encrypt_block_inplace);
/**
- * fscrypt_decrypt_pagecache_blocks() - Decrypt data from a pagecache folio
- * @folio: the pagecache folio containing the data to decrypt
- * @len: size of the data to decrypt, in bytes
- * @offs: offset within @folio of the data to decrypt, in bytes
- *
- * Decrypt data that has just been read from an encrypted file. The data must
- * be located in a pagecache folio that is still locked and not yet uptodate.
- * The length and offset of the data must be aligned to the file's crypto data
- * unit size. Alignment to the filesystem block size fulfills this requirement,
- * as the filesystem block size is always a multiple of the data unit size.
- *
- * Return: 0 on success; -errno on failure
- */
-int fscrypt_decrypt_pagecache_blocks(struct folio *folio, size_t len,
- size_t offs)
-{
- const struct inode *inode = folio->mapping->host;
- const struct fscrypt_inode_info *ci = fscrypt_get_inode_info_raw(inode);
- const unsigned int du_bits = ci->ci_data_unit_bits;
- const unsigned int du_size = 1U << du_bits;
- u64 index = ((u64)folio->index << (PAGE_SHIFT - du_bits)) +
- (offs >> du_bits);
- size_t i;
- int err;
-
- if (WARN_ON_ONCE(!folio_test_locked(folio)))
- return -EINVAL;
-
- if (WARN_ON_ONCE(len <= 0 || !IS_ALIGNED(len | offs, du_size)))
- return -EINVAL;
-
- for (i = offs; i < offs + len; i += du_size, index++) {
- struct page *page = folio_page(folio, i >> PAGE_SHIFT);
-
- err = fscrypt_crypt_data_unit(ci, FS_DECRYPT, index, page,
- page, du_size, i & ~PAGE_MASK);
- if (err)
- return err;
- }
- return 0;
-}
-EXPORT_SYMBOL(fscrypt_decrypt_pagecache_blocks);
-
-/**
* fscrypt_decrypt_block_inplace() - Decrypt a filesystem block in-place
* @inode: The inode to which this block belongs
* @page: The page containing the block to decrypt
@@ -296,7 +261,8 @@ EXPORT_SYMBOL(fscrypt_decrypt_pagecache_blocks);
* arbitrary page, not necessarily in the original pagecache page. The @inode
* and @lblk_num must be specified, as they can't be determined from @page.
*
- * This is not compatible with fscrypt_operations::supports_subblock_data_units.
+ * This function only supports non-block-based filesystems that don't support
+ * sub-block data units (as indicated by the fscrypt_operations fields).
*
* Return: 0 on success; -errno on failure
*/
@@ -323,31 +289,26 @@ EXPORT_SYMBOL(fscrypt_decrypt_block_inplace);
*/
int fscrypt_initialize(struct super_block *sb)
{
- int err = 0;
mempool_t *pool;
/* pairs with smp_store_release() below */
- if (likely(smp_load_acquire(&fscrypt_bounce_page_pool)))
+ if (smp_load_acquire(&fscrypt_bounce_page_pool))
return 0;
/* No need to allocate a bounce page pool if this FS won't use it. */
if (!sb->s_cop->needs_bounce_pages)
return 0;
- mutex_lock(&fscrypt_init_mutex);
+ guard(mutex)(&fscrypt_init_mutex);
if (fscrypt_bounce_page_pool)
- goto out_unlock;
+ return 0;
- err = -ENOMEM;
pool = mempool_create_page_pool(num_prealloc_crypto_pages, 0);
if (!pool)
- goto out_unlock;
+ return -ENOMEM;
/* pairs with smp_load_acquire() above */
smp_store_release(&fscrypt_bounce_page_pool, pool);
- err = 0;
-out_unlock:
- mutex_unlock(&fscrypt_init_mutex);
- return err;
+ return 0;
}
void fscrypt_msg(const struct inode *inode, const char *level,
@@ -374,45 +335,12 @@ void fscrypt_msg(const struct inode *inode, const char *level,
va_end(args);
}
-/**
- * fscrypt_init() - Set up for fs encryption.
- *
- * Return: 0 on success; -errno on failure
- */
static int __init fscrypt_init(void)
{
- int err = -ENOMEM;
-
- /*
- * Use an unbound workqueue to allow bios to be decrypted in parallel
- * even when they happen to complete on the same CPU. This sacrifices
- * locality, but it's worthwhile since decryption is CPU-intensive.
- *
- * Also use a high-priority workqueue to prioritize decryption work,
- * which blocks reads from completing, over regular application tasks.
- */
- fscrypt_read_workqueue = alloc_workqueue("fscrypt_read_queue",
- WQ_UNBOUND | WQ_HIGHPRI,
- num_online_cpus());
- if (!fscrypt_read_workqueue)
- goto fail;
-
fscrypt_inode_info_cachep = KMEM_CACHE(fscrypt_inode_info,
- SLAB_RECLAIM_ACCOUNT);
- if (!fscrypt_inode_info_cachep)
- goto fail_free_queue;
-
- err = fscrypt_init_keyring();
- if (err)
- goto fail_free_inode_info;
-
+ SLAB_RECLAIM_ACCOUNT |
+ SLAB_PANIC);
+ fscrypt_init_keyring();
return 0;
-
-fail_free_inode_info:
- kmem_cache_destroy(fscrypt_inode_info_cachep);
-fail_free_queue:
- destroy_workqueue(fscrypt_read_workqueue);
-fail:
- return err;
}
late_initcall(fscrypt_init)
diff --git a/fs/crypto/fscrypt_private.h b/fs/crypto/fscrypt_private.h
index 0053b5c45412..74329e0953d1 100644
--- a/fs/crypto/fscrypt_private.h
+++ b/fs/crypto/fscrypt_private.h
@@ -66,9 +66,6 @@
#define FSCRYPT_CONTEXT_V1 1
#define FSCRYPT_CONTEXT_V2 2
-/* Keep this in sync with include/uapi/linux/fscrypt.h */
-#define FSCRYPT_MODE_MAX FSCRYPT_MODE_AES_256_HCTR2
-
struct fscrypt_context_v1 {
u8 version; /* FSCRYPT_CONTEXT_V1 */
u8 contents_encryption_mode;
@@ -269,14 +266,6 @@ struct fscrypt_inode_info {
/* True if ci_enc_key should be freed when this struct is freed */
u8 ci_owns_key : 1;
-#ifdef CONFIG_FS_ENCRYPTION_INLINE_CRYPT
- /*
- * True if this inode will use inline encryption (blk-crypto) instead of
- * the traditional filesystem-layer encryption.
- */
- u8 ci_inlinecrypt : 1;
-#endif
-
/* True if ci_dirhash_key is initialized */
u8 ci_dirhash_key_initialized : 1;
@@ -340,11 +329,6 @@ typedef enum {
/* crypto.c */
extern struct kmem_cache *fscrypt_inode_info_cachep;
int fscrypt_initialize(struct super_block *sb);
-int fscrypt_crypt_data_unit(const struct fscrypt_inode_info *ci,
- fscrypt_direction_t rw, u64 index,
- struct page *src_page, struct page *dest_page,
- unsigned int len, unsigned int offs);
-struct page *fscrypt_alloc_bounce_page(gfp_t gfp_flags);
void __printf(3, 4) __cold
fscrypt_msg(const struct inode *inode, const char *level, const char *fmt, ...);
@@ -411,15 +395,14 @@ void fscrypt_hkdf_expand(const struct hmac_sha512_key *hkdf, u8 context,
const u8 *info, unsigned int infolen,
u8 *okm, unsigned int okmlen);
-/* inline_crypt.c */
+/* block.c */
#ifdef CONFIG_FS_ENCRYPTION_INLINE_CRYPT
-int fscrypt_select_encryption_impl(struct fscrypt_inode_info *ci,
- bool is_hw_wrapped_key);
-
static inline bool
fscrypt_using_inline_encryption(const struct fscrypt_inode_info *ci)
{
- return ci->ci_inlinecrypt;
+ const struct inode *inode = ci->ci_inode;
+
+ return S_ISREG(inode->i_mode) && inode->i_sb->s_cop->is_block_based;
}
int fscrypt_prepare_inline_crypt_key(struct fscrypt_prepared_key *prep_key,
@@ -449,12 +432,6 @@ fscrypt_is_key_prepared(const struct fscrypt_prepared_key *prep_key,
#else /* CONFIG_FS_ENCRYPTION_INLINE_CRYPT */
-static inline int fscrypt_select_encryption_impl(struct fscrypt_inode_info *ci,
- bool is_hw_wrapped_key)
-{
- return 0;
-}
-
static inline bool
fscrypt_using_inline_encryption(const struct fscrypt_inode_info *ci)
{
@@ -718,7 +695,7 @@ int fscrypt_add_test_dummy_key(struct super_block *sb,
int fscrypt_verify_key_added(struct super_block *sb,
const u8 identifier[FSCRYPT_KEY_IDENTIFIER_SIZE]);
-int __init fscrypt_init_keyring(void);
+void __init fscrypt_init_keyring(void);
/* keysetup.c */
diff --git a/fs/crypto/keyring.c b/fs/crypto/keyring.c
index 38b73e703073..76e28d1e0064 100644
--- a/fs/crypto/keyring.c
+++ b/fs/crypto/keyring.c
@@ -497,7 +497,7 @@ static int do_add_master_key(struct super_block *sb,
struct fscrypt_master_key *mk;
int err;
- mutex_lock(&fscrypt_add_key_mutex); /* serialize find + link */
+ guard(mutex)(&fscrypt_add_key_mutex); /* serialize find + link */
mk = fscrypt_find_master_key(sb, mk_spec);
if (!mk) {
@@ -524,7 +524,6 @@ static int do_add_master_key(struct super_block *sb,
}
fscrypt_put_master_key(mk);
}
- mutex_unlock(&fscrypt_add_key_mutex);
return err;
}
@@ -1221,21 +1220,19 @@ out:
}
EXPORT_SYMBOL_GPL(fscrypt_ioctl_get_key_status);
-int __init fscrypt_init_keyring(void)
+void __init fscrypt_init_keyring(void)
{
int err;
+ /*
+ * Note that register_key_type() fails only if a key type with the same
+ * name already exists, which should never happen here.
+ */
err = register_key_type(&key_type_fscrypt_user);
if (err)
- return err;
-
+ panic("failed to register .fscrypt key type (%d)", err);
err = register_key_type(&key_type_fscrypt_provisioning);
if (err)
- goto err_unregister_fscrypt_user;
-
- return 0;
-
-err_unregister_fscrypt_user:
- unregister_key_type(&key_type_fscrypt_user);
- return err;
+ panic("failed to register fscrypt-provisioning key type (%d)",
+ err);
}
diff --git a/fs/crypto/keysetup.c b/fs/crypto/keysetup.c
index f905f9f94bdd..892044ebcaca 100644
--- a/fs/crypto/keysetup.c
+++ b/fs/crypto/keysetup.c
@@ -83,8 +83,6 @@ static struct fscrypt_mode *
select_encryption_mode(const union fscrypt_policy *policy,
const struct inode *inode)
{
- BUILD_BUG_ON(ARRAY_SIZE(fscrypt_modes) != FSCRYPT_MODE_MAX + 1);
-
if (S_ISREG(inode->i_mode))
return &fscrypt_modes[fscrypt_policy_contents_mode(policy)];
@@ -146,9 +144,9 @@ err_free_tfm:
/*
* Prepare the crypto transform object or blk-crypto key in @prep_key, given the
- * raw key, encryption mode (@ci->ci_mode), flag indicating which encryption
- * implementation (fs-layer or blk-crypto) will be used (@ci->ci_inlinecrypt),
- * and IV generation method (@ci->ci_policy.flags).
+ * raw key, encryption mode (@ci->ci_mode), predicate indicating which style of
+ * key is needed (fscrypt_using_inline_encryption(ci)), IV generation method
+ * (@ci->ci_policy.flags), and data unit size (@ci->ci_data_unit_bits).
*/
int fscrypt_prepare_key(struct fscrypt_prepared_key *prep_key,
const u8 *raw_key, const struct fscrypt_inode_info *ci)
@@ -226,26 +224,8 @@ static int setup_per_mode_enc_key(struct fscrypt_inode_info *ci,
u8 raw_mode_key[FSCRYPT_MAX_RAW_KEY_SIZE];
u8 hkdf_info[sizeof(mode_num) + sizeof(sb->s_uuid)];
unsigned int hkdf_infolen = 0;
- bool use_hw_wrapped_key = false;
int err;
- if (WARN_ON_ONCE(mode_num > FSCRYPT_MODE_MAX))
- return -EINVAL;
-
- if (mk->mk_secret.is_hw_wrapped && S_ISREG(inode->i_mode)) {
- /* Using a hardware-wrapped key for file contents encryption */
- if (!fscrypt_using_inline_encryption(ci)) {
- if (sb->s_flags & SB_INLINECRYPT)
- fscrypt_warn(ci->ci_inode,
- "Hardware-wrapped key required, but no suitable inline encryption capabilities are available");
- else
- fscrypt_warn(ci->ci_inode,
- "Hardware-wrapped keys require inline encryption (-o inlinecrypt)");
- return -EINVAL;
- }
- use_hw_wrapped_key = true;
- }
-
prep_key = fscrypt_find_mode_key(mk, hkdf_context, mode_num, ci);
if (prep_key) {
ci->ci_enc_key = *prep_key;
@@ -268,7 +248,7 @@ static int setup_per_mode_enc_key(struct fscrypt_inode_info *ci,
new_node->data_unit_bits = ci->ci_data_unit_bits;
prep_key = &new_node->key;
- if (use_hw_wrapped_key) {
+ if (mk->mk_secret.is_hw_wrapped && S_ISREG(inode->i_mode)) {
err = fscrypt_prepare_inline_crypt_key(prep_key,
mk->mk_secret.bytes,
mk->mk_secret.size, true,
@@ -287,7 +267,7 @@ static int setup_per_mode_enc_key(struct fscrypt_inode_info *ci,
hkdf_info, hkdf_infolen, raw_mode_key,
mode->keysize);
err = fscrypt_prepare_key(prep_key, raw_mode_key, ci);
- memzero_explicit(raw_mode_key, mode->keysize);
+ memzero_explicit(raw_mode_key, sizeof(raw_mode_key));
}
if (err) {
kfree(new_node);
@@ -349,18 +329,17 @@ static int fscrypt_setup_iv_ino_lblk_32_key(struct fscrypt_inode_info *ci,
/* pairs with smp_store_release() below */
if (!smp_load_acquire(&mk->mk_ino_hash_key_initialized)) {
-
- mutex_lock(&fscrypt_mode_key_setup_mutex);
-
- if (mk->mk_ino_hash_key_initialized)
- goto unlock;
-
- fscrypt_derive_siphash_key(mk, HKDF_CONTEXT_INODE_HASH_KEY,
- NULL, 0, &mk->mk_ino_hash_key);
- /* pairs with smp_load_acquire() above */
- smp_store_release(&mk->mk_ino_hash_key_initialized, true);
-unlock:
- mutex_unlock(&fscrypt_mode_key_setup_mutex);
+ guard(mutex)(&fscrypt_mode_key_setup_mutex);
+
+ if (!mk->mk_ino_hash_key_initialized) {
+ fscrypt_derive_siphash_key(mk,
+ HKDF_CONTEXT_INODE_HASH_KEY,
+ NULL, 0,
+ &mk->mk_ino_hash_key);
+ /* pairs with smp_load_acquire() above */
+ smp_store_release(&mk->mk_ino_hash_key_initialized,
+ true);
+ }
}
/*
@@ -418,7 +397,7 @@ static int fscrypt_setup_v2_file_key(struct fscrypt_inode_info *ci,
ci->ci_nonce, FSCRYPT_FILE_NONCE_SIZE,
derived_key, ci->ci_mode->keysize);
err = fscrypt_set_per_file_enc_key(ci, derived_key);
- memzero_explicit(derived_key, ci->ci_mode->keysize);
+ memzero_explicit(derived_key, sizeof(derived_key));
}
if (err)
return err;
@@ -515,10 +494,6 @@ static int setup_file_encryption_key(struct fscrypt_inode_info *ci,
if (ci->ci_policy.version != FSCRYPT_POLICY_V1)
return -ENOKEY;
- err = fscrypt_select_encryption_impl(ci, false);
- if (err)
- return err;
-
/*
* As a legacy fallback for v1 policies, search for the key in
* the current task's subscribed keyrings too. Don't move this
@@ -540,10 +515,6 @@ static int setup_file_encryption_key(struct fscrypt_inode_info *ci,
goto out_release_key;
}
- err = fscrypt_select_encryption_impl(ci, mk->mk_secret.is_hw_wrapped);
- if (err)
- goto out_release_key;
-
switch (ci->ci_policy.version) {
case FSCRYPT_POLICY_V1:
if (WARN_ON_ONCE(mk->mk_secret.is_hw_wrapped)) {
diff --git a/fs/crypto/keysetup_v1.c b/fs/crypto/keysetup_v1.c
index 7e3a58dc4b56..87fe13ccb253 100644
--- a/fs/crypto/keysetup_v1.c
+++ b/fs/crypto/keysetup_v1.c
@@ -251,7 +251,7 @@ static int setup_v1_file_key_derived(struct fscrypt_inode_info *ci,
err = fscrypt_set_per_file_enc_key(ci, derived_key);
- memzero_explicit(derived_key, derived_keysize);
+ memzero_explicit(derived_key, sizeof(derived_key));
/* No need to zeroize 'aes', as its key is not secret. */
return err;
}
diff --git a/fs/crypto/policy.c b/fs/crypto/policy.c
index 9915e39362db..a7322dba7557 100644
--- a/fs/crypto/policy.c
+++ b/fs/crypto/policy.c
@@ -177,6 +177,23 @@ static bool supported_iv_ino_lblk_policy(const struct fscrypt_policy_v2 *policy,
type, sb->s_id);
return false;
}
+
+ /*
+ * IV_INO_LBLK_32 isn't compatible with inline encryption when
+ * s_blocksize != PAGE_SIZE. In that case the DUN can wrap around in
+ * the middle of a page, but sometimes fscrypt_mergeable_bio() is called
+ * only for the first block per page. Since IV_INO_LBLK_32 exists only
+ * to support inline encryption hardware that is limited to 32-bit DUNs,
+ * just disallow IV_INO_LBLK_32 with s_blocksize != PAGE_SIZE entirely.
+ */
+ if ((policy->flags & FSCRYPT_POLICY_FLAG_IV_INO_LBLK_32) &&
+ sb->s_blocksize != PAGE_SIZE) {
+ fscrypt_warn(inode,
+ "Can't use %s policy on filesystem '%s' with block size != PAGE_SIZE",
+ type, sb->s_id);
+ return false;
+ }
+
return true;
}
@@ -507,7 +524,6 @@ int fscrypt_ioctl_set_policy(struct file *filp, const void __user *arg)
union fscrypt_policy policy;
union fscrypt_policy existing_policy;
struct inode *inode = file_inode(filp);
- u8 version;
int size;
int ret;
@@ -518,21 +534,9 @@ int fscrypt_ioctl_set_policy(struct file *filp, const void __user *arg)
if (size <= 0)
return -EINVAL;
- /*
- * We should just copy the remaining 'size - 1' bytes here, but a
- * bizarre bug in gcc 7 and earlier (fixed by gcc r255731) causes gcc to
- * think that size can be 0 here (despite the check above!) *and* that
- * it's a compile-time constant. Thus it would think copy_from_user()
- * is passed compile-time constant ULONG_MAX, causing the compile-time
- * buffer overflow check to fail, breaking the build. This only occurred
- * when building an i386 kernel with -Os and branch profiling enabled.
- *
- * Work around it by just copying the first byte again...
- */
- version = policy.version;
- if (copy_from_user(&policy, arg, size))
+ if (copy_from_user((u8 *)&policy + 1, (const u8 __user *)arg + 1,
+ size - 1))
return -EFAULT;
- policy.version = version;
if (!inode_owner_or_capable(&nop_mnt_idmap, inode))
return -EACCES;
diff --git a/fs/ecryptfs/crypto.c b/fs/ecryptfs/crypto.c
index 74b02b55e3f6..e67119b6029c 100644
--- a/fs/ecryptfs/crypto.c
+++ b/fs/ecryptfs/crypto.c
@@ -1197,7 +1197,7 @@ static int ecryptfs_read_headers_virt(char *page_virt,
} else
set_default_header_data(crypt_stat);
rc = ecryptfs_parse_packet_set(crypt_stat, (page_virt + offset),
- ecryptfs_dentry);
+ PAGE_SIZE - offset, ecryptfs_dentry);
out:
return rc;
}
diff --git a/fs/ecryptfs/ecryptfs_kernel.h b/fs/ecryptfs/ecryptfs_kernel.h
index f4f56a92bd56..58165928ed1e 100644
--- a/fs/ecryptfs/ecryptfs_kernel.h
+++ b/fs/ecryptfs/ecryptfs_kernel.h
@@ -1,5 +1,5 @@
/* SPDX-License-Identifier: GPL-2.0-or-later */
-/**
+/*
* eCryptfs: Linux filesystem encryption layer
* Kernel declarations.
*
@@ -204,7 +204,7 @@ struct ecryptfs_filename {
char dentry_name[ECRYPTFS_ENCRYPTED_DENTRY_NAME_LEN + 1];
};
-/**
+/*
* This is the primary struct associated with each encrypted file.
*
* TODO: cache align/pack?
@@ -255,7 +255,8 @@ struct ecryptfs_inode_info {
};
/**
- * ecryptfs_global_auth_tok - A key used to encrypt all new files under the mountpoint
+ * struct ecryptfs_global_auth_tok - A key used to encrypt all new files
+ * under the mountpoint
* @flags: Status flags
* @mount_crypt_stat_list: These auth_toks hang off the mount-wide
* cryptographic context. Every time a new
@@ -263,7 +264,6 @@ struct ecryptfs_inode_info {
* the auth_toks on that list to the set of
* auth_toks on the inode's crypt_stat
* @global_auth_tok_key: The key from the user's keyring for the sig
- * @global_auth_tok: The key contents
* @sig: The key identifier
*
* ecryptfs_global_auth_tok structs refer to authentication token keys
@@ -283,7 +283,7 @@ struct ecryptfs_global_auth_tok {
};
/**
- * ecryptfs_key_tfm - Persistent key tfm
+ * struct ecryptfs_key_tfm - Persistent key tfm
* @key_tfm: crypto API handle to the key
* @key_size: Key size in bytes
* @key_tfm_mutex: Mutex to ensure only one operation in eCryptfs is
@@ -306,7 +306,7 @@ struct ecryptfs_key_tfm {
extern struct mutex key_tfm_list_mutex;
-/**
+/*
* This struct is to enable a mount-wide passphrase/salt combo. This
* is more or less a stopgap to provide similar functionality to other
* crypto filesystems like EncFS or CFS until full policy support is
@@ -580,7 +580,8 @@ int ecryptfs_generate_key_packet_set(char *dest_base,
size_t *len, size_t max);
int
ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
- unsigned char *src, struct dentry *ecryptfs_dentry);
+ unsigned char *src, size_t src_size,
+ struct dentry *ecryptfs_dentry);
int ecryptfs_truncate(struct dentry *dentry, loff_t new_length);
ssize_t
ecryptfs_getxattr_lower(struct dentry *lower_dentry, struct inode *lower_inode,
diff --git a/fs/ecryptfs/inode.c b/fs/ecryptfs/inode.c
index 7aaf1913f9c6..525297c7ebd8 100644
--- a/fs/ecryptfs/inode.c
+++ b/fs/ecryptfs/inode.c
@@ -268,7 +268,7 @@ out:
static int
ecryptfs_create(struct mnt_idmap *idmap,
struct inode *directory_inode, struct dentry *ecryptfs_dentry,
- umode_t mode, bool excl)
+ umode_t mode)
{
struct inode *ecryptfs_inode;
int rc;
diff --git a/fs/ecryptfs/keystore.c b/fs/ecryptfs/keystore.c
index ebebc9551f1f..51651314b7a6 100644
--- a/fs/ecryptfs/keystore.c
+++ b/fs/ecryptfs/keystore.c
@@ -894,6 +894,12 @@ ecryptfs_parse_tag_70_packet(char **filename, size_t *filename_size,
"rc = [%d]\n", __func__, rc);
goto out;
}
+ if (s->parsed_tag_70_packet_size < (ECRYPTFS_SIG_SIZE + 2)) {
+ ecryptfs_printk(KERN_WARNING, "Invalid packet size [%zd]\n",
+ s->parsed_tag_70_packet_size);
+ rc = -EINVAL;
+ goto out;
+ }
s->block_aligned_filename_size = (s->parsed_tag_70_packet_size
- ECRYPTFS_SIG_SIZE - 1);
if ((1 + s->packet_size_len + s->parsed_tag_70_packet_size)
@@ -1384,10 +1390,20 @@ parse_tag_3_packet(struct ecryptfs_crypt_stat *crypt_stat,
}
(*new_auth_tok)->session_key.encrypted_key_size =
(body_size - (ECRYPTFS_SALT_SIZE + 5));
+ /*
+ * Although encrypted_key_size is copied into the
+ * encrypted_key[ECRYPTFS_MAX_ENCRYPTED_KEY_BYTES] buffer here,
+ * it later bounds operations on a smaller buffer:
+ * decrypt_passphrase_encrypted_session_key() sets decrypted_key_size =
+ * encrypted_key_size and decrypts into
+ * decrypted_key[ECRYPTFS_MAX_KEY_BYTES], then memcpy's into
+ * crypt_stat->key[ECRYPTFS_MAX_KEY_BYTES]. Limit to
+ * ECRYPTFS_MAX_KEY_BYTES to protect those smaller buffers.
+ */
if ((*new_auth_tok)->session_key.encrypted_key_size
- > ECRYPTFS_MAX_ENCRYPTED_KEY_BYTES) {
+ > ECRYPTFS_MAX_KEY_BYTES) {
printk(KERN_WARNING "Tag 3 packet contains key larger "
- "than ECRYPTFS_MAX_ENCRYPTED_KEY_BYTES\n");
+ "than ECRYPTFS_MAX_KEY_BYTES\n");
rc = -EINVAL;
goto out_free;
}
@@ -1537,7 +1553,7 @@ parse_tag_11_packet(unsigned char *data, unsigned char *contents,
}
(*packet_size) += length_size;
(*tag_11_contents_size) = (body_size - 14);
- if (unlikely((*packet_size) + body_size + 1 > max_packet_size)) {
+ if (unlikely((*packet_size) + body_size > max_packet_size)) {
printk(KERN_ERR "Packet size exceeds max\n");
rc = -EINVAL;
goto out;
@@ -1704,6 +1720,7 @@ out:
* ecryptfs_parse_packet_set
* @crypt_stat: The cryptographic context
* @src: Virtual address of region of memory containing the packets
+ * @src_size: Size of the packet set buffer
* @ecryptfs_dentry: The eCryptfs dentry associated with the packet set
*
* Get crypt_stat to have the file's session key if the requisite key
@@ -1714,7 +1731,7 @@ out:
* conditions.
*/
int ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
- unsigned char *src,
+ unsigned char *src, size_t src_size,
struct dentry *ecryptfs_dentry)
{
size_t i = 0;
@@ -1736,7 +1753,11 @@ int ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
* added the our &auth_tok_list */
next_packet_is_auth_tok_packet = 1;
while (next_packet_is_auth_tok_packet) {
- size_t max_packet_size = ((PAGE_SIZE - 8) - i);
+ size_t max_packet_size;
+
+ if (i >= src_size)
+ break;
+ max_packet_size = src_size - i;
switch (src[i]) {
case ECRYPTFS_TAG_3_PACKET_TYPE:
@@ -1751,12 +1772,16 @@ int ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
goto out_wipe_list;
}
i += packet_size;
+ if (i > src_size) {
+ rc = -EIO;
+ goto out_wipe_list;
+ }
rc = parse_tag_11_packet((unsigned char *)&src[i],
sig_tmp_space,
ECRYPTFS_SIG_SIZE,
&tag_11_contents_size,
&tag_11_packet_size,
- max_packet_size);
+ src_size - i);
if (rc) {
ecryptfs_printk(KERN_ERR, "No valid "
"(ecryptfs-specific) literal "
@@ -1768,6 +1793,10 @@ int ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
goto out_wipe_list;
}
i += tag_11_packet_size;
+ if (i > src_size) {
+ rc = -EIO;
+ goto out_wipe_list;
+ }
if (ECRYPTFS_SIG_SIZE != tag_11_contents_size) {
ecryptfs_printk(KERN_ERR, "Expected "
"signature of size [%d]; "
@@ -1793,6 +1822,10 @@ int ecryptfs_parse_packet_set(struct ecryptfs_crypt_stat *crypt_stat,
goto out_wipe_list;
}
i += packet_size;
+ if (i > src_size) {
+ rc = -EIO;
+ goto out_wipe_list;
+ }
crypt_stat->flags |= ECRYPTFS_ENCRYPTED;
break;
case ECRYPTFS_TAG_11_PACKET_TYPE:
diff --git a/fs/ecryptfs/messaging.c b/fs/ecryptfs/messaging.c
index 03c60f0850ca..73c7b8215e66 100644
--- a/fs/ecryptfs/messaging.c
+++ b/fs/ecryptfs/messaging.c
@@ -166,6 +166,7 @@ int ecryptfs_exorcise_daemon(struct ecryptfs_daemon *daemon)
mutex_unlock(&daemon->mux);
goto out;
}
+ mutex_lock(&ecryptfs_msg_ctx_lists_mux);
list_for_each_entry_safe(msg_ctx, msg_ctx_tmp,
&daemon->msg_ctx_out_queue, daemon_out_list) {
list_del(&msg_ctx->daemon_out_list);
@@ -174,6 +175,7 @@ int ecryptfs_exorcise_daemon(struct ecryptfs_daemon *daemon)
"the out queue of a dying daemon\n", __func__);
ecryptfs_msg_ctx_alloc_to_free(msg_ctx);
}
+ mutex_unlock(&ecryptfs_msg_ctx_lists_mux);
hlist_del(&daemon->euid_chain);
mutex_unlock(&daemon->mux);
kfree_sensitive(daemon);
@@ -284,9 +286,16 @@ ecryptfs_send_message_locked(char *data, int data_len, u8 msg_type,
mutex_unlock(&ecryptfs_msg_ctx_lists_mux);
rc = ecryptfs_send_miscdev(data, data_len, *msg_ctx, msg_type, 0,
daemon);
- if (rc)
+ if (rc) {
printk(KERN_ERR "%s: Error attempting to send message to "
"userspace daemon; rc = [%d]\n", __func__, rc);
+ mutex_lock(&ecryptfs_msg_ctx_lists_mux);
+ mutex_lock(&(*msg_ctx)->mux);
+ ecryptfs_msg_ctx_alloc_to_free(*msg_ctx);
+ mutex_unlock(&(*msg_ctx)->mux);
+ mutex_unlock(&ecryptfs_msg_ctx_lists_mux);
+ *msg_ctx = NULL;
+ }
out:
return rc;
}
diff --git a/fs/ecryptfs/miscdev.c b/fs/ecryptfs/miscdev.c
index 5a7d08149922..68804399a5df 100644
--- a/fs/ecryptfs/miscdev.c
+++ b/fs/ecryptfs/miscdev.c
@@ -360,7 +360,7 @@ ecryptfs_miscdev_write(struct file *file, const char __user *buf,
u32 seq;
size_t packet_size, packet_size_length;
char *data;
- unsigned char packet_size_peek[ECRYPTFS_MAX_PKT_LEN_SIZE];
+ unsigned char packet_size_peek[ECRYPTFS_MAX_PKT_LEN_SIZE] = { };
ssize_t rc;
if (count == 0) {
@@ -376,7 +376,8 @@ ecryptfs_miscdev_write(struct file *file, const char __user *buf,
}
if (copy_from_user(packet_size_peek, &buf[PKT_LEN_OFFSET],
- sizeof(packet_size_peek))) {
+ min_t(size_t, count - PKT_LEN_OFFSET,
+ sizeof(packet_size_peek)))) {
printk(KERN_WARNING "%s: Error while inspecting packet size\n",
__func__);
return -EFAULT;
diff --git a/fs/ecryptfs/mmap.c b/fs/ecryptfs/mmap.c
index 2c2b12fedeae..07a6c8448100 100644
--- a/fs/ecryptfs/mmap.c
+++ b/fs/ecryptfs/mmap.c
@@ -355,24 +355,17 @@ out:
*/
static int ecryptfs_write_inode_size_to_header(struct inode *ecryptfs_inode)
{
- char *file_size_virt;
+ __be64 file_size;
int rc;
- file_size_virt = kmalloc(sizeof(u64), GFP_KERNEL);
- if (!file_size_virt) {
- rc = -ENOMEM;
- goto out;
- }
- put_unaligned_be64(i_size_read(ecryptfs_inode), file_size_virt);
- rc = ecryptfs_write_lower(ecryptfs_inode, file_size_virt, 0,
- sizeof(u64));
- kfree(file_size_virt);
+ file_size = cpu_to_be64(i_size_read(ecryptfs_inode));
+ rc = ecryptfs_write_lower(ecryptfs_inode, (char *)&file_size, 0,
+ sizeof(file_size));
if (rc < 0)
printk(KERN_ERR "%s: Error writing file size to header; "
"rc = [%d]\n", __func__, rc);
else
rc = 0;
-out:
return rc;
}
@@ -510,21 +503,8 @@ static sector_t ecryptfs_bmap(struct address_space *mapping, sector_t block)
return block;
}
-#include <linux/buffer_head.h>
-
const struct address_space_operations ecryptfs_aops = {
- /*
- * XXX: This is pretty broken for multiple reasons: ecryptfs does not
- * actually use buffer_heads, and ecryptfs will crash without
- * CONFIG_BLOCK. But it matches the behavior before the default for
- * address_space_operations without the ->dirty_folio method was
- * cleaned up, so this is the best we can do without maintainer
- * feedback.
- */
-#ifdef CONFIG_BLOCK
- .dirty_folio = block_dirty_folio,
- .invalidate_folio = block_invalidate_folio,
-#endif
+ .dirty_folio = filemap_dirty_folio,
.writepages = ecryptfs_writepages,
.read_folio = ecryptfs_read_folio,
.write_begin = ecryptfs_write_begin,
diff --git a/fs/ecryptfs/super.c b/fs/ecryptfs/super.c
index 3bc21d677564..686b2b4a9cb5 100644
--- a/fs/ecryptfs/super.c
+++ b/fs/ecryptfs/super.c
@@ -150,6 +150,13 @@ static int ecryptfs_show_options(struct seq_file *m, struct dentry *root)
if (mount_crypt_stat->global_default_cipher_key_size)
seq_printf(m, ",ecryptfs_key_bytes=%zd",
mount_crypt_stat->global_default_cipher_key_size);
+ if (mount_crypt_stat->flags & ECRYPTFS_GLOBAL_ENCRYPT_FILENAMES) {
+ seq_printf(m, ",ecryptfs_fn_cipher=%s",
+ mount_crypt_stat->global_default_fn_cipher_name);
+ if (mount_crypt_stat->global_default_fn_cipher_key_bytes)
+ seq_printf(m, ",ecryptfs_fn_key_bytes=%zd",
+ mount_crypt_stat->global_default_fn_cipher_key_bytes);
+ }
if (mount_crypt_stat->flags & ECRYPTFS_PLAINTEXT_PASSTHROUGH_ENABLED)
seq_printf(m, ",ecryptfs_passthrough");
if (mount_crypt_stat->flags & ECRYPTFS_XATTR_METADATA_ENABLED)
diff --git a/fs/efivarfs/inode.c b/fs/efivarfs/inode.c
index 95dcad83da11..f0d009555fc6 100644
--- a/fs/efivarfs/inode.c
+++ b/fs/efivarfs/inode.c
@@ -75,7 +75,7 @@ static bool efivarfs_valid_name(const char *str, int len)
}
static int efivarfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode = NULL;
struct efivar_entry *var;
diff --git a/fs/efs/Kconfig b/fs/efs/Kconfig
deleted file mode 100644
index 0833e533df9d..000000000000
--- a/fs/efs/Kconfig
+++ /dev/null
@@ -1,16 +0,0 @@
-# SPDX-License-Identifier: GPL-2.0-only
-config EFS_FS
- tristate "EFS file system support (read only)"
- depends on BLOCK
- select BUFFER_HEAD
- help
- EFS is an older file system used for non-ISO9660 CD-ROMs and hard
- disk partitions by SGI's IRIX operating system (IRIX 6.0 and newer
- uses the XFS file system for hard disk partitions however).
-
- This implementation only offers read-only access. If you don't know
- what all this is about, it's safe to say N. For more information
- about EFS see its home page at <http://aeschi.ch.eu.org/efs/>.
-
- To compile the EFS file system support as a module, choose M here: the
- module will be called efs.
diff --git a/fs/efs/Makefile b/fs/efs/Makefile
deleted file mode 100644
index 85e5b88f9471..000000000000
--- a/fs/efs/Makefile
+++ /dev/null
@@ -1,8 +0,0 @@
-# SPDX-License-Identifier: GPL-2.0-only
-#
-# Makefile for the linux efs-filesystem routines.
-#
-
-obj-$(CONFIG_EFS_FS) += efs.o
-
-efs-objs := super.o inode.o namei.o dir.o file.o symlink.o
diff --git a/fs/efs/dir.c b/fs/efs/dir.c
deleted file mode 100644
index 35ad0092c115..000000000000
--- a/fs/efs/dir.c
+++ /dev/null
@@ -1,105 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * dir.c
- *
- * Copyright (c) 1999 Al Smith
- */
-
-#include <linux/buffer_head.h>
-#include <linux/filelock.h>
-#include "efs.h"
-
-static int efs_readdir(struct file *, struct dir_context *);
-
-const struct file_operations efs_dir_operations = {
- .llseek = generic_file_llseek,
- .read = generic_read_dir,
- .iterate_shared = efs_readdir,
- .setlease = generic_setlease,
-};
-
-const struct inode_operations efs_dir_inode_operations = {
- .lookup = efs_lookup,
-};
-
-static int efs_readdir(struct file *file, struct dir_context *ctx)
-{
- struct inode *inode = file_inode(file);
- efs_block_t block;
- int slot;
-
- if (inode->i_size & (EFS_DIRBSIZE-1))
- pr_warn("%s(): directory size not a multiple of EFS_DIRBSIZE\n",
- __func__);
-
- /* work out where this entry can be found */
- block = ctx->pos >> EFS_DIRBSIZE_BITS;
-
- /* each block contains at most 256 slots */
- slot = ctx->pos & 0xff;
-
- /* look at all blocks */
- while (block < inode->i_blocks) {
- struct efs_dir *dirblock;
- struct buffer_head *bh;
-
- /* read the dir block */
- bh = sb_bread(inode->i_sb, efs_bmap(inode, block));
-
- if (!bh) {
- pr_err("%s(): failed to read dir block %d\n",
- __func__, block);
- break;
- }
-
- dirblock = (struct efs_dir *) bh->b_data;
-
- if (be16_to_cpu(dirblock->magic) != EFS_DIRBLK_MAGIC) {
- pr_err("%s(): invalid directory block\n", __func__);
- brelse(bh);
- break;
- }
-
- for (; slot < dirblock->slots; slot++) {
- struct efs_dentry *dirslot;
- efs_ino_t inodenum;
- const char *nameptr;
- int namelen;
-
- if (dirblock->space[slot] == 0)
- continue;
-
- dirslot = (struct efs_dentry *) (((char *) bh->b_data) + EFS_SLOTAT(dirblock, slot));
-
- inodenum = be32_to_cpu(dirslot->inode);
- namelen = dirslot->namelen;
- nameptr = dirslot->name;
- pr_debug("%s(): block %d slot %d/%d: inode %u, name \"%s\", namelen %u\n",
- __func__, block, slot, dirblock->slots-1,
- inodenum, nameptr, namelen);
- if (!namelen)
- continue;
- /* found the next entry */
- ctx->pos = (block << EFS_DIRBSIZE_BITS) | slot;
-
- /* sanity check */
- if (nameptr - (char *) dirblock + namelen > EFS_DIRBSIZE) {
- pr_warn("directory entry %d exceeds directory block\n",
- slot);
- continue;
- }
-
- /* copy filename and data in dirslot */
- if (!dir_emit(ctx, nameptr, namelen, inodenum, DT_UNKNOWN)) {
- brelse(bh);
- return 0;
- }
- }
- brelse(bh);
-
- slot = 0;
- block++;
- }
- ctx->pos = (block << EFS_DIRBSIZE_BITS) | slot;
- return 0;
-}
diff --git a/fs/efs/efs.h b/fs/efs/efs.h
deleted file mode 100644
index 918d2b9abb76..000000000000
--- a/fs/efs/efs.h
+++ /dev/null
@@ -1,144 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 */
-/*
- * Copyright (c) 1999 Al Smith, <Al.Smith@aeschi.ch.eu.org>
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang.
- * Portions derived from IRIX header files (c) 1988 Silicon Graphics
- */
-#ifndef _EFS_EFS_H_
-#define _EFS_EFS_H_
-
-#ifdef pr_fmt
-#undef pr_fmt
-#endif
-
-#define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
-
-#include <linux/fs.h>
-#include <linux/uaccess.h>
-
-#define EFS_VERSION "1.0a"
-
-/* 1 block is 512 bytes */
-#define EFS_BLOCKSIZE_BITS 9
-#define EFS_BLOCKSIZE (1 << EFS_BLOCKSIZE_BITS)
-
-typedef int32_t efs_block_t;
-typedef uint32_t efs_ino_t;
-
-#define EFS_DIRECTEXTENTS 12
-
-/*
- * layout of an extent, in memory and on disk. 8 bytes exactly.
- */
-typedef union extent_u {
- unsigned char raw[8];
- struct extent_s {
- unsigned int ex_magic:8; /* magic # (zero) */
- unsigned int ex_bn:24; /* basic block */
- unsigned int ex_length:8; /* numblocks in this extent */
- unsigned int ex_offset:24; /* logical offset into file */
- } cooked;
-} efs_extent;
-
-typedef struct edevs {
- __be16 odev;
- __be32 ndev;
-} efs_devs;
-
-/*
- * extent based filesystem inode as it appears on disk. The efs inode
- * is exactly 128 bytes long.
- */
-struct efs_dinode {
- __be16 di_mode; /* mode and type of file */
- __be16 di_nlink; /* number of links to file */
- __be16 di_uid; /* owner's user id */
- __be16 di_gid; /* owner's group id */
- __be32 di_size; /* number of bytes in file */
- __be32 di_atime; /* time last accessed */
- __be32 di_mtime; /* time last modified */
- __be32 di_ctime; /* time created */
- __be32 di_gen; /* generation number */
- __be16 di_numextents; /* # of extents */
- u_char di_version; /* version of inode */
- u_char di_spare; /* spare - used by AFS */
- union di_addr {
- efs_extent di_extents[EFS_DIRECTEXTENTS];
- efs_devs di_dev; /* device for IFCHR/IFBLK */
- } di_u;
-};
-
-/* efs inode storage in memory */
-struct efs_inode_info {
- int numextents;
- int lastextent;
-
- efs_extent extents[EFS_DIRECTEXTENTS];
- struct inode vfs_inode;
-};
-
-#include <linux/efs_fs_sb.h>
-
-#define EFS_DIRBSIZE_BITS EFS_BLOCKSIZE_BITS
-#define EFS_DIRBSIZE (1 << EFS_DIRBSIZE_BITS)
-
-struct efs_dentry {
- __be32 inode;
- unsigned char namelen;
- char name[3];
-};
-
-#define EFS_DENTSIZE (sizeof(struct efs_dentry) - 3 + 1)
-#define EFS_MAXNAMELEN ((1 << (sizeof(char) * 8)) - 1)
-
-#define EFS_DIRBLK_HEADERSIZE 4
-#define EFS_DIRBLK_MAGIC 0xbeef /* moo */
-
-struct efs_dir {
- __be16 magic;
- unsigned char firstused;
- unsigned char slots;
-
- unsigned char space[EFS_DIRBSIZE - EFS_DIRBLK_HEADERSIZE];
-};
-
-#define EFS_MAXENTS \
- ((EFS_DIRBSIZE - EFS_DIRBLK_HEADERSIZE) / \
- (EFS_DENTSIZE + sizeof(char)))
-
-#define EFS_SLOTAT(dir, slot) EFS_REALOFF((dir)->space[slot])
-
-#define EFS_REALOFF(offset) ((offset << 1))
-
-
-static inline struct efs_inode_info *INODE_INFO(struct inode *inode)
-{
- return container_of(inode, struct efs_inode_info, vfs_inode);
-}
-
-static inline struct efs_sb_info *SUPER_INFO(struct super_block *sb)
-{
- return sb->s_fs_info;
-}
-
-struct statfs;
-struct fid;
-
-extern const struct inode_operations efs_dir_inode_operations;
-extern const struct file_operations efs_dir_operations;
-extern const struct address_space_operations efs_symlink_aops;
-
-extern struct inode *efs_iget(struct super_block *, unsigned long);
-extern efs_block_t efs_map_block(struct inode *, efs_block_t);
-extern int efs_get_block(struct inode *, sector_t, struct buffer_head *, int);
-
-extern struct dentry *efs_lookup(struct inode *, struct dentry *, unsigned int);
-extern struct dentry *efs_fh_to_dentry(struct super_block *sb, struct fid *fid,
- int fh_len, int fh_type);
-extern struct dentry *efs_fh_to_parent(struct super_block *sb, struct fid *fid,
- int fh_len, int fh_type);
-extern struct dentry *efs_get_parent(struct dentry *);
-extern int efs_bmap(struct inode *, int);
-
-#endif /* _EFS_EFS_H_ */
diff --git a/fs/efs/file.c b/fs/efs/file.c
deleted file mode 100644
index 9153dfe79bbc..000000000000
--- a/fs/efs/file.c
+++ /dev/null
@@ -1,42 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * file.c
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang.
- */
-
-#include <linux/buffer_head.h>
-#include "efs.h"
-
-int efs_get_block(struct inode *inode, sector_t iblock,
- struct buffer_head *bh_result, int create)
-{
- int error = -EROFS;
- long phys;
-
- if (create)
- return error;
- if (iblock >= inode->i_blocks)
- return 0;
-
- phys = efs_map_block(inode, iblock);
- if (phys)
- map_bh(bh_result, inode->i_sb, phys);
- return 0;
-}
-
-int efs_bmap(struct inode *inode, efs_block_t block) {
-
- if (block < 0) {
- pr_warn("%s(): block < 0\n", __func__);
- return 0;
- }
-
- /* are we about to read past the end of a file ? */
- if (!(block < inode->i_blocks))
- return 0;
-
- return efs_map_block(inode, block);
-}
diff --git a/fs/efs/inode.c b/fs/efs/inode.c
deleted file mode 100644
index 4b132729e638..000000000000
--- a/fs/efs/inode.c
+++ /dev/null
@@ -1,315 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0-only
-/*
- * inode.c
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang,
- * and from work (c) 1998 Mike Shaver.
- */
-
-#include <linux/buffer_head.h>
-#include <linux/module.h>
-#include <linux/fs.h>
-#include "efs.h"
-#include <linux/efs_fs_sb.h>
-
-static int efs_read_folio(struct file *file, struct folio *folio)
-{
- return block_read_full_folio(folio, efs_get_block);
-}
-
-static sector_t _efs_bmap(struct address_space *mapping, sector_t block)
-{
- return generic_block_bmap(mapping,block,efs_get_block);
-}
-
-static const struct address_space_operations efs_aops = {
- .read_folio = efs_read_folio,
- .bmap = _efs_bmap
-};
-
-static inline void extent_copy(efs_extent *src, efs_extent *dst) {
- /*
- * this is slightly evil. it doesn't just copy
- * efs_extent from src to dst, it also mangles
- * the bits so that dst ends up in cpu byte-order.
- */
-
- dst->cooked.ex_magic = (unsigned int) src->raw[0];
- dst->cooked.ex_bn = ((unsigned int) src->raw[1] << 16) |
- ((unsigned int) src->raw[2] << 8) |
- ((unsigned int) src->raw[3] << 0);
- dst->cooked.ex_length = (unsigned int) src->raw[4];
- dst->cooked.ex_offset = ((unsigned int) src->raw[5] << 16) |
- ((unsigned int) src->raw[6] << 8) |
- ((unsigned int) src->raw[7] << 0);
- return;
-}
-
-struct inode *efs_iget(struct super_block *super, unsigned long ino)
-{
- int i, inode_index;
- dev_t device;
- u32 rdev;
- struct buffer_head *bh;
- struct efs_sb_info *sb = SUPER_INFO(super);
- struct efs_inode_info *in;
- efs_block_t block, offset;
- struct efs_dinode *efs_inode;
- struct inode *inode;
-
- inode = iget_locked(super, ino);
- if (!inode)
- return ERR_PTR(-ENOMEM);
- if (!(inode_state_read_once(inode) & I_NEW))
- return inode;
-
- in = INODE_INFO(inode);
-
- /*
- ** EFS layout:
- **
- ** | cylinder group | cylinder group | cylinder group ..etc
- ** |inodes|data |inodes|data |inodes|data ..etc
- **
- ** work out the inode block index, (considering initially that the
- ** inodes are stored as consecutive blocks). then work out the block
- ** number of that inode given the above layout, and finally the
- ** offset of the inode within that block.
- */
-
- inode_index = inode->i_ino /
- (EFS_BLOCKSIZE / sizeof(struct efs_dinode));
-
- block = sb->fs_start + sb->first_block +
- (sb->group_size * (inode_index / sb->inode_blocks)) +
- (inode_index % sb->inode_blocks);
-
- offset = (inode->i_ino %
- (EFS_BLOCKSIZE / sizeof(struct efs_dinode))) *
- sizeof(struct efs_dinode);
-
- bh = sb_bread(inode->i_sb, block);
- if (!bh) {
- pr_warn("%s() failed at block %d\n", __func__, block);
- goto read_inode_error;
- }
-
- efs_inode = (struct efs_dinode *) (bh->b_data + offset);
-
- inode->i_mode = be16_to_cpu(efs_inode->di_mode);
- set_nlink(inode, be16_to_cpu(efs_inode->di_nlink));
- i_uid_write(inode, (uid_t)be16_to_cpu(efs_inode->di_uid));
- i_gid_write(inode, (gid_t)be16_to_cpu(efs_inode->di_gid));
- inode->i_size = be32_to_cpu(efs_inode->di_size);
- inode_set_atime(inode, be32_to_cpu(efs_inode->di_atime), 0);
- inode_set_mtime(inode, be32_to_cpu(efs_inode->di_mtime), 0);
- inode_set_ctime(inode, be32_to_cpu(efs_inode->di_ctime), 0);
-
- /* this is the number of blocks in the file */
- if (inode->i_size == 0) {
- inode->i_blocks = 0;
- } else {
- inode->i_blocks = ((inode->i_size - 1) >> EFS_BLOCKSIZE_BITS) + 1;
- }
-
- rdev = be16_to_cpu(efs_inode->di_u.di_dev.odev);
- if (rdev == 0xffff) {
- rdev = be32_to_cpu(efs_inode->di_u.di_dev.ndev);
- if (sysv_major(rdev) > 0xfff)
- device = 0;
- else
- device = MKDEV(sysv_major(rdev), sysv_minor(rdev));
- } else
- device = old_decode_dev(rdev);
-
- /* get the number of extents for this object */
- in->numextents = be16_to_cpu(efs_inode->di_numextents);
- in->lastextent = 0;
-
- /* copy the extents contained within the inode to memory */
- for(i = 0; i < EFS_DIRECTEXTENTS; i++) {
- extent_copy(&(efs_inode->di_u.di_extents[i]), &(in->extents[i]));
- if (i < in->numextents && in->extents[i].cooked.ex_magic != 0) {
- pr_warn("extent %d has bad magic number in inode %llu\n",
- i, inode->i_ino);
- brelse(bh);
- goto read_inode_error;
- }
- }
-
- brelse(bh);
- pr_debug("efs_iget(): inode %llu, extents %d, mode %o\n",
- inode->i_ino, in->numextents, inode->i_mode);
- switch (inode->i_mode & S_IFMT) {
- case S_IFDIR:
- inode->i_op = &efs_dir_inode_operations;
- inode->i_fop = &efs_dir_operations;
- break;
- case S_IFREG:
- inode->i_fop = &generic_ro_fops;
- inode->i_data.a_ops = &efs_aops;
- break;
- case S_IFLNK:
- inode->i_op = &page_symlink_inode_operations;
- inode_nohighmem(inode);
- inode->i_data.a_ops = &efs_symlink_aops;
- break;
- case S_IFCHR:
- case S_IFBLK:
- case S_IFIFO:
- init_special_inode(inode, inode->i_mode, device);
- break;
- default:
- pr_warn("unsupported inode mode %o\n", inode->i_mode);
- goto read_inode_error;
- break;
- }
-
- unlock_new_inode(inode);
- return inode;
-
-read_inode_error:
- pr_warn("failed to read inode %llu\n", inode->i_ino);
- iget_failed(inode);
- return ERR_PTR(-EIO);
-}
-
-static inline efs_block_t
-efs_extent_check(efs_extent *ptr, efs_block_t block, struct efs_sb_info *sb) {
- efs_block_t start;
- efs_block_t length;
- efs_block_t offset;
-
- /*
- * given an extent and a logical block within a file,
- * can this block be found within this extent ?
- */
- start = ptr->cooked.ex_bn;
- length = ptr->cooked.ex_length;
- offset = ptr->cooked.ex_offset;
-
- if ((block >= offset) && (block < offset+length)) {
- return(sb->fs_start + start + block - offset);
- } else {
- return 0;
- }
-}
-
-efs_block_t efs_map_block(struct inode *inode, efs_block_t block) {
- struct efs_sb_info *sb = SUPER_INFO(inode->i_sb);
- struct efs_inode_info *in = INODE_INFO(inode);
- struct buffer_head *bh = NULL;
-
- int cur, last, first = 1;
- int ibase, ioffset, dirext, direxts, indext, indexts;
- efs_block_t iblock, result = 0, lastblock = 0;
- efs_extent ext, *exts;
-
- last = in->lastextent;
-
- if (in->numextents <= EFS_DIRECTEXTENTS) {
- /* first check the last extent we returned */
- if ((result = efs_extent_check(&in->extents[last], block, sb)))
- return result;
-
- /* if we only have one extent then nothing can be found */
- if (in->numextents == 1) {
- pr_err("%s() failed to map (1 extent)\n", __func__);
- return 0;
- }
-
- direxts = in->numextents;
-
- /*
- * check the stored extents in the inode
- * start with next extent and check forwards
- */
- for(dirext = 1; dirext < direxts; dirext++) {
- cur = (last + dirext) % in->numextents;
- if ((result = efs_extent_check(&in->extents[cur], block, sb))) {
- in->lastextent = cur;
- return result;
- }
- }
-
- pr_err("%s() failed to map block %u (dir)\n", __func__, block);
- return 0;
- }
-
- pr_debug("%s(): indirect search for logical block %u\n",
- __func__, block);
- direxts = in->extents[0].cooked.ex_offset;
- indexts = in->numextents;
-
- for(indext = 0; indext < indexts; indext++) {
- cur = (last + indext) % indexts;
-
- /*
- * work out which direct extent contains `cur'.
- *
- * also compute ibase: i.e. the number of the first
- * indirect extent contained within direct extent `cur'.
- *
- */
- ibase = 0;
- for(dirext = 0; cur < ibase && dirext < direxts; dirext++) {
- ibase += in->extents[dirext].cooked.ex_length *
- (EFS_BLOCKSIZE / sizeof(efs_extent));
- }
-
- if (dirext == direxts) {
- /* should never happen */
- pr_err("couldn't find direct extent for indirect extent %d (block %u)\n",
- cur, block);
- if (bh) brelse(bh);
- return 0;
- }
-
- /* work out block number and offset of this indirect extent */
- iblock = sb->fs_start + in->extents[dirext].cooked.ex_bn +
- (cur - ibase) /
- (EFS_BLOCKSIZE / sizeof(efs_extent));
- ioffset = (cur - ibase) %
- (EFS_BLOCKSIZE / sizeof(efs_extent));
-
- if (first || lastblock != iblock) {
- if (bh) brelse(bh);
-
- bh = sb_bread(inode->i_sb, iblock);
- if (!bh) {
- pr_err("%s() failed at block %d\n",
- __func__, iblock);
- return 0;
- }
- pr_debug("%s(): read indirect extent block %d\n",
- __func__, iblock);
- first = 0;
- lastblock = iblock;
- }
-
- exts = (efs_extent *) bh->b_data;
-
- extent_copy(&(exts[ioffset]), &ext);
-
- if (ext.cooked.ex_magic != 0) {
- pr_err("extent %d has bad magic number in block %d\n",
- cur, iblock);
- if (bh) brelse(bh);
- return 0;
- }
-
- if ((result = efs_extent_check(&ext, block, sb))) {
- if (bh) brelse(bh);
- in->lastextent = cur;
- return result;
- }
- }
- if (bh) brelse(bh);
- pr_err("%s() failed to map block %u (indir)\n", __func__, block);
- return 0;
-}
-
-MODULE_DESCRIPTION("Extent File System (efs)");
-MODULE_LICENSE("GPL");
diff --git a/fs/efs/namei.c b/fs/efs/namei.c
deleted file mode 100644
index 38961ee1d1af..000000000000
--- a/fs/efs/namei.c
+++ /dev/null
@@ -1,120 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * namei.c
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang.
- */
-
-#include <linux/buffer_head.h>
-#include <linux/string.h>
-#include <linux/exportfs.h>
-#include "efs.h"
-
-
-static efs_ino_t efs_find_entry(struct inode *inode, const char *name, int len)
-{
- struct buffer_head *bh;
-
- int slot, namelen;
- char *nameptr;
- struct efs_dir *dirblock;
- struct efs_dentry *dirslot;
- efs_ino_t inodenum;
- efs_block_t block;
-
- if (inode->i_size & (EFS_DIRBSIZE-1))
- pr_warn("%s(): directory size not a multiple of EFS_DIRBSIZE\n",
- __func__);
-
- for(block = 0; block < inode->i_blocks; block++) {
-
- bh = sb_bread(inode->i_sb, efs_bmap(inode, block));
- if (!bh) {
- pr_err("%s(): failed to read dir block %d\n",
- __func__, block);
- return 0;
- }
-
- dirblock = (struct efs_dir *) bh->b_data;
-
- if (be16_to_cpu(dirblock->magic) != EFS_DIRBLK_MAGIC) {
- pr_err("%s(): invalid directory block\n", __func__);
- brelse(bh);
- return 0;
- }
-
- for (slot = 0; slot < dirblock->slots; slot++) {
- dirslot = (struct efs_dentry *) (((char *) bh->b_data) + EFS_SLOTAT(dirblock, slot));
-
- namelen = dirslot->namelen;
- nameptr = dirslot->name;
-
- if ((namelen == len) && (!memcmp(name, nameptr, len))) {
- inodenum = be32_to_cpu(dirslot->inode);
- brelse(bh);
- return inodenum;
- }
- }
- brelse(bh);
- }
- return 0;
-}
-
-struct dentry *efs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags)
-{
- efs_ino_t inodenum;
- struct inode *inode = NULL;
-
- inodenum = efs_find_entry(dir, dentry->d_name.name, dentry->d_name.len);
- if (inodenum)
- inode = efs_iget(dir->i_sb, inodenum);
-
- return d_splice_alias(inode, dentry);
-}
-
-static struct inode *efs_nfs_get_inode(struct super_block *sb, u64 ino,
- u32 generation)
-{
- struct inode *inode;
-
- if (ino == 0)
- return ERR_PTR(-ESTALE);
- inode = efs_iget(sb, ino);
- if (IS_ERR(inode))
- return ERR_CAST(inode);
-
- if (generation && inode->i_generation != generation) {
- iput(inode);
- return ERR_PTR(-ESTALE);
- }
-
- return inode;
-}
-
-struct dentry *efs_fh_to_dentry(struct super_block *sb, struct fid *fid,
- int fh_len, int fh_type)
-{
- return generic_fh_to_dentry(sb, fid, fh_len, fh_type,
- efs_nfs_get_inode);
-}
-
-struct dentry *efs_fh_to_parent(struct super_block *sb, struct fid *fid,
- int fh_len, int fh_type)
-{
- return generic_fh_to_parent(sb, fid, fh_len, fh_type,
- efs_nfs_get_inode);
-}
-
-struct dentry *efs_get_parent(struct dentry *child)
-{
- struct dentry *parent = ERR_PTR(-ENOENT);
- efs_ino_t ino;
-
- ino = efs_find_entry(d_inode(child), "..", 2);
- if (ino)
- parent = d_obtain_alias(efs_iget(child->d_sb, ino));
-
- return parent;
-}
diff --git a/fs/efs/super.c b/fs/efs/super.c
deleted file mode 100644
index 11fea3bbce7c..000000000000
--- a/fs/efs/super.c
+++ /dev/null
@@ -1,368 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * super.c
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang.
- */
-
-#include <linux/init.h>
-#include <linux/module.h>
-#include <linux/exportfs.h>
-#include <linux/slab.h>
-#include <linux/buffer_head.h>
-#include <linux/vfs.h>
-#include <linux/blkdev.h>
-#include <linux/fs_context.h>
-#include "efs.h"
-#include <linux/efs_vh.h>
-#include <linux/efs_fs_sb.h>
-
-static int efs_statfs(struct dentry *dentry, struct kstatfs *buf);
-static int efs_init_fs_context(struct fs_context *fc);
-
-static void efs_kill_sb(struct super_block *s)
-{
- struct efs_sb_info *sbi = SUPER_INFO(s);
- kill_block_super(s);
- kfree(sbi);
-}
-
-static struct pt_types sgi_pt_types[] = {
- {0x00, "SGI vh"},
- {0x01, "SGI trkrepl"},
- {0x02, "SGI secrepl"},
- {0x03, "SGI raw"},
- {0x04, "SGI bsd"},
- {SGI_SYSV, "SGI sysv"},
- {0x06, "SGI vol"},
- {SGI_EFS, "SGI efs"},
- {0x08, "SGI lv"},
- {0x09, "SGI rlv"},
- {0x0A, "SGI xfs"},
- {0x0B, "SGI xfslog"},
- {0x0C, "SGI xlv"},
- {0x82, "Linux swap"},
- {0x83, "Linux native"},
- {0, NULL}
-};
-
-/*
- * File system definition and registration.
- */
-static struct file_system_type efs_fs_type = {
- .owner = THIS_MODULE,
- .name = "efs",
- .kill_sb = efs_kill_sb,
- .fs_flags = FS_REQUIRES_DEV,
- .init_fs_context = efs_init_fs_context,
-};
-MODULE_ALIAS_FS("efs");
-
-static struct kmem_cache * efs_inode_cachep;
-
-static struct inode *efs_alloc_inode(struct super_block *sb)
-{
- struct efs_inode_info *ei;
- ei = alloc_inode_sb(sb, efs_inode_cachep, GFP_KERNEL);
- if (!ei)
- return NULL;
- return &ei->vfs_inode;
-}
-
-static void efs_free_inode(struct inode *inode)
-{
- kmem_cache_free(efs_inode_cachep, INODE_INFO(inode));
-}
-
-static void init_once(void *foo)
-{
- struct efs_inode_info *ei = (struct efs_inode_info *) foo;
-
- inode_init_once(&ei->vfs_inode);
-}
-
-static int __init init_inodecache(void)
-{
- efs_inode_cachep = kmem_cache_create("efs_inode_cache",
- sizeof(struct efs_inode_info), 0,
- SLAB_RECLAIM_ACCOUNT|SLAB_ACCOUNT,
- init_once);
- if (efs_inode_cachep == NULL)
- return -ENOMEM;
- return 0;
-}
-
-static void destroy_inodecache(void)
-{
- /*
- * Make sure all delayed rcu free inodes are flushed before we
- * destroy cache.
- */
- rcu_barrier();
- kmem_cache_destroy(efs_inode_cachep);
-}
-
-static const struct super_operations efs_superblock_operations = {
- .alloc_inode = efs_alloc_inode,
- .free_inode = efs_free_inode,
- .statfs = efs_statfs,
-};
-
-static const struct export_operations efs_export_ops = {
- .encode_fh = generic_encode_ino32_fh,
- .fh_to_dentry = efs_fh_to_dentry,
- .fh_to_parent = efs_fh_to_parent,
- .get_parent = efs_get_parent,
-};
-
-static int __init init_efs_fs(void) {
- int err;
- pr_info(EFS_VERSION" - http://aeschi.ch.eu.org/efs/\n");
- err = init_inodecache();
- if (err)
- goto out1;
- err = register_filesystem(&efs_fs_type);
- if (err)
- goto out;
- return 0;
-out:
- destroy_inodecache();
-out1:
- return err;
-}
-
-static void __exit exit_efs_fs(void) {
- unregister_filesystem(&efs_fs_type);
- destroy_inodecache();
-}
-
-module_init(init_efs_fs)
-module_exit(exit_efs_fs)
-
-static efs_block_t efs_validate_vh(struct volume_header *vh) {
- int i;
- __be32 cs, *ui;
- int csum;
- efs_block_t sblock = 0; /* shuts up gcc */
- struct pt_types *pt_entry;
- int pt_type, slice = -1;
-
- if (be32_to_cpu(vh->vh_magic) != VHMAGIC) {
- /*
- * assume that we're dealing with a partition and allow
- * read_super() to try and detect a valid superblock
- * on the next block.
- */
- return 0;
- }
-
- ui = ((__be32 *) (vh + 1)) - 1;
- for(csum = 0; ui >= ((__be32 *) vh);) {
- cs = *ui--;
- csum += be32_to_cpu(cs);
- }
- if (csum) {
- pr_warn("SGI disklabel: checksum bad, label corrupted\n");
- return 0;
- }
-
-#ifdef DEBUG
- pr_debug("bf: \"%16s\"\n", vh->vh_bootfile);
-
- for(i = 0; i < NVDIR; i++) {
- int j;
- char name[VDNAMESIZE+1];
-
- for(j = 0; j < VDNAMESIZE; j++) {
- name[j] = vh->vh_vd[i].vd_name[j];
- }
- name[j] = (char) 0;
-
- if (name[0]) {
- pr_debug("vh: %8s block: 0x%08x size: 0x%08x\n",
- name, (int) be32_to_cpu(vh->vh_vd[i].vd_lbn),
- (int) be32_to_cpu(vh->vh_vd[i].vd_nbytes));
- }
- }
-#endif
-
- for(i = 0; i < NPARTAB; i++) {
- pt_type = (int) be32_to_cpu(vh->vh_pt[i].pt_type);
- for(pt_entry = sgi_pt_types; pt_entry->pt_name; pt_entry++) {
- if (pt_type == pt_entry->pt_type) break;
- }
-#ifdef DEBUG
- if (be32_to_cpu(vh->vh_pt[i].pt_nblks)) {
- pr_debug("pt %2d: start: %08d size: %08d type: 0x%02x (%s)\n",
- i, (int)be32_to_cpu(vh->vh_pt[i].pt_firstlbn),
- (int)be32_to_cpu(vh->vh_pt[i].pt_nblks),
- pt_type, (pt_entry->pt_name) ?
- pt_entry->pt_name : "unknown");
- }
-#endif
- if (IS_EFS(pt_type)) {
- sblock = be32_to_cpu(vh->vh_pt[i].pt_firstlbn);
- slice = i;
- }
- }
-
- if (slice == -1) {
- pr_notice("partition table contained no EFS partitions\n");
-#ifdef DEBUG
- } else {
- pr_info("using slice %d (type %s, offset 0x%x)\n", slice,
- (pt_entry->pt_name) ? pt_entry->pt_name : "unknown",
- sblock);
-#endif
- }
- return sblock;
-}
-
-static int efs_validate_super(struct efs_sb_info *sb, struct efs_super *super) {
-
- if (!IS_EFS_MAGIC(be32_to_cpu(super->fs_magic)))
- return -1;
-
- sb->fs_magic = be32_to_cpu(super->fs_magic);
- sb->total_blocks = be32_to_cpu(super->fs_size);
- sb->first_block = be32_to_cpu(super->fs_firstcg);
- sb->group_size = be32_to_cpu(super->fs_cgfsize);
- sb->data_free = be32_to_cpu(super->fs_tfree);
- sb->inode_free = be32_to_cpu(super->fs_tinode);
- sb->inode_blocks = be16_to_cpu(super->fs_cgisize);
- sb->total_groups = be16_to_cpu(super->fs_ncg);
-
- return 0;
-}
-
-static int efs_fill_super(struct super_block *s, struct fs_context *fc)
-{
- struct efs_sb_info *sb;
- struct buffer_head *bh;
- struct inode *root;
-
- sb = kzalloc_obj(struct efs_sb_info);
- if (!sb)
- return -ENOMEM;
- s->s_fs_info = sb;
- s->s_time_min = 0;
- s->s_time_max = U32_MAX;
-
- s->s_magic = EFS_SUPER_MAGIC;
- if (!sb_set_blocksize(s, EFS_BLOCKSIZE)) {
- pr_err("device does not support %d byte blocks\n",
- EFS_BLOCKSIZE);
- return invalf(fc, "device does not support %d byte blocks\n",
- EFS_BLOCKSIZE);
- }
-
- /* read the vh (volume header) block */
- bh = sb_bread(s, 0);
-
- if (!bh) {
- pr_err("cannot read volume header\n");
- return -EIO;
- }
-
- /*
- * if this returns zero then we didn't find any partition table.
- * this isn't (yet) an error - just assume for the moment that
- * the device is valid and go on to search for a superblock.
- */
- sb->fs_start = efs_validate_vh((struct volume_header *) bh->b_data);
- brelse(bh);
-
- if (sb->fs_start == -1) {
- return -EINVAL;
- }
-
- bh = sb_bread(s, sb->fs_start + EFS_SUPER);
- if (!bh) {
- pr_err("cannot read superblock\n");
- return -EIO;
- }
-
- if (efs_validate_super(sb, (struct efs_super *) bh->b_data)) {
-#ifdef DEBUG
- pr_warn("invalid superblock at block %u\n",
- sb->fs_start + EFS_SUPER);
-#endif
- brelse(bh);
- return -EINVAL;
- }
- brelse(bh);
-
- if (!sb_rdonly(s)) {
-#ifdef DEBUG
- pr_info("forcing read-only mode\n");
-#endif
- s->s_flags |= SB_RDONLY;
- }
- s->s_op = &efs_superblock_operations;
- s->s_export_op = &efs_export_ops;
- root = efs_iget(s, EFS_ROOTINODE);
- if (IS_ERR(root)) {
- pr_err("get root inode failed\n");
- return PTR_ERR(root);
- }
-
- s->s_root = d_make_root(root);
- if (!(s->s_root)) {
- pr_err("get root dentry failed\n");
- return -ENOMEM;
- }
-
- return 0;
-}
-
-static int efs_get_tree(struct fs_context *fc)
-{
- return get_tree_bdev(fc, efs_fill_super);
-}
-
-static int efs_reconfigure(struct fs_context *fc)
-{
- sync_filesystem(fc->root->d_sb);
- fc->sb_flags |= SB_RDONLY;
-
- return 0;
-}
-
-static const struct fs_context_operations efs_context_opts = {
- .get_tree = efs_get_tree,
- .reconfigure = efs_reconfigure,
-};
-
-/*
- * Set up the filesystem mount context.
- */
-static int efs_init_fs_context(struct fs_context *fc)
-{
- fc->ops = &efs_context_opts;
-
- return 0;
-}
-
-static int efs_statfs(struct dentry *dentry, struct kstatfs *buf) {
- struct super_block *sb = dentry->d_sb;
- struct efs_sb_info *sbi = SUPER_INFO(sb);
- u64 id = huge_encode_dev(sb->s_bdev->bd_dev);
-
- buf->f_type = EFS_SUPER_MAGIC; /* efs magic number */
- buf->f_bsize = EFS_BLOCKSIZE; /* blocksize */
- buf->f_blocks = sbi->total_groups * /* total data blocks */
- (sbi->group_size - sbi->inode_blocks);
- buf->f_bfree = sbi->data_free; /* free data blocks */
- buf->f_bavail = sbi->data_free; /* free blocks for non-root */
- buf->f_files = sbi->total_groups * /* total inodes */
- sbi->inode_blocks *
- (EFS_BLOCKSIZE / sizeof(struct efs_dinode));
- buf->f_ffree = sbi->inode_free; /* free inodes */
- buf->f_fsid = u64_to_fsid(id);
- buf->f_namelen = EFS_MAXNAMELEN; /* max filename length */
-
- return 0;
-}
-
diff --git a/fs/efs/symlink.c b/fs/efs/symlink.c
deleted file mode 100644
index 7749feded722..000000000000
--- a/fs/efs/symlink.c
+++ /dev/null
@@ -1,50 +0,0 @@
-// SPDX-License-Identifier: GPL-2.0
-/*
- * symlink.c
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from work (c) 1995,1996 Christian Vogelgsang.
- */
-
-#include <linux/string.h>
-#include <linux/pagemap.h>
-#include <linux/buffer_head.h>
-#include "efs.h"
-
-static int efs_symlink_read_folio(struct file *file, struct folio *folio)
-{
- char *link = folio_address(folio);
- struct buffer_head *bh;
- struct inode *inode = folio->mapping->host;
- efs_block_t size = inode->i_size;
- int err;
-
- err = -ENAMETOOLONG;
- if (size > 2 * EFS_BLOCKSIZE)
- goto fail;
-
- /* read first 512 bytes of link target */
- err = -EIO;
- bh = sb_bread(inode->i_sb, efs_bmap(inode, 0));
- if (!bh)
- goto fail;
- memcpy(link, bh->b_data, (size > EFS_BLOCKSIZE) ? EFS_BLOCKSIZE : size);
- brelse(bh);
- if (size > EFS_BLOCKSIZE) {
- bh = sb_bread(inode->i_sb, efs_bmap(inode, 1));
- if (!bh)
- goto fail;
- memcpy(link + EFS_BLOCKSIZE, bh->b_data, size - EFS_BLOCKSIZE);
- brelse(bh);
- }
- link[size] = '\0';
- err = 0;
-fail:
- folio_end_read(folio, err == 0);
- return err;
-}
-
-const struct address_space_operations efs_symlink_aops = {
- .read_folio = efs_symlink_read_folio
-};
diff --git a/fs/erofs/Kconfig b/fs/erofs/Kconfig
index 4789b1077d8c..8ca1767dafb6 100644
--- a/fs/erofs/Kconfig
+++ b/fs/erofs/Kconfig
@@ -131,6 +131,20 @@ config EROFS_FS_ZIP_LZMA
Say N if you want to disable LZMA compression support.
+config EROFS_FS_ZIP_LZMA_DEFAULT_MAX_STREAMS
+ int "EROFS LZMA default maximum decompression streams"
+ depends on EROFS_FS_ZIP_LZMA
+ range 1 NR_CPUS
+ default 16
+ help
+ By default EROFS allocates one LZMA decompression stream per CPU.
+ Each stream can hold a dictionary of up to 8 MiB taken from the
+ mounted image, so on systems with many CPUs this can reserve a lot
+ of memory. This caps the default; the lzma_streams module parameter
+ still overrides it.
+
+ If unsure, keep the default of 16.
+
config EROFS_FS_ZIP_DEFLATE
bool "EROFS DEFLATE compressed data support"
depends on EROFS_FS_ZIP
@@ -189,6 +203,7 @@ config EROFS_FS_PCPU_KTHREAD_HIPRI
config EROFS_FS_PAGE_CACHE_SHARE
bool "EROFS page cache share support (experimental)"
depends on EROFS_FS && EROFS_FS_XATTR
+ select FS_STACK
help
This enables page cache sharing among inodes with identical
content fingerprints on the same machine.
diff --git a/fs/erofs/decompressor_lzma.c b/fs/erofs/decompressor_lzma.c
index f6692d0f2f04..6b0cdb446c6a 100644
--- a/fs/erofs/decompressor_lzma.c
+++ b/fs/erofs/decompressor_lzma.c
@@ -51,7 +51,8 @@ static int __init z_erofs_lzma_init(void)
/* by default, use # of possible CPUs instead */
if (!z_erofs_lzma_nstrms)
- z_erofs_lzma_nstrms = num_possible_cpus();
+ z_erofs_lzma_nstrms = min_t(unsigned int, num_possible_cpus(),
+ CONFIG_EROFS_FS_ZIP_LZMA_DEFAULT_MAX_STREAMS);
for (i = 0; i < z_erofs_lzma_nstrms; ++i) {
struct z_erofs_lzma *strm = kzalloc_obj(*strm);
diff --git a/fs/erofs/internal.h b/fs/erofs/internal.h
index 580f8d9f14e7..57bd21859c65 100644
--- a/fs/erofs/internal.h
+++ b/fs/erofs/internal.h
@@ -288,8 +288,8 @@ struct erofs_inode {
struct erofs_inode_fingerprint fingerprint;
spinlock_t ishare_lock;
};
- /* for each real inode */
- struct inode *sharedinode;
+ /* for each real filesystem inode */
+ struct dentry *sharedentry;
};
#endif
/* the corresponding vfs inode */
diff --git a/fs/erofs/ishare.c b/fs/erofs/ishare.c
index a1a20a5ba548..51b81aaf9ac3 100644
--- a/fs/erofs/ishare.c
+++ b/fs/erofs/ishare.c
@@ -2,14 +2,13 @@
/*
* Copyright (C) 2024, Alibaba Cloud
*/
+#include <linux/backing-file.h>
#include <linux/xxhash.h>
#include <linux/mount.h>
#include <linux/security.h>
#include "internal.h"
#include "xattr.h"
-#include "../internal.h"
-
static struct vfsmount *erofs_ishare_mnt;
static int erofs_ishare_iget5_eq(struct inode *inode, void *data)
@@ -33,10 +32,12 @@ static int erofs_ishare_iget5_set(struct inode *inode, void *data)
bool erofs_ishare_fill_inode(struct inode *inode)
{
+ static const struct file_operations empty_fops = {};
struct erofs_sb_info *sbi = EROFS_SB(inode->i_sb);
const struct address_space_operations *aops;
struct erofs_inode *vi = EROFS_I(inode);
struct erofs_inode_fingerprint fp;
+ struct dentry *sd;
struct inode *si;
aops = erofs_get_aops(inode);
@@ -49,7 +50,9 @@ bool erofs_ishare_fill_inode(struct inode *inode)
xxh32(fp.opaque, fp.size, 0),
erofs_ishare_iget5_eq, erofs_ishare_iget5_set, &fp);
if (si && (inode_state_read_once(si) & I_NEW)) {
+ si->i_fop = &empty_fops;
si->i_mapping->a_ops = aops;
+ si->i_mode = 0444 | S_IFREG;
si->i_size = inode->i_size;
unlock_new_inode(si);
} else {
@@ -65,7 +68,12 @@ bool erofs_ishare_fill_inode(struct inode *inode)
return false;
}
}
- vi->sharedinode = si;
+ sd = d_obtain_alias(si); /* disconnected denties for sharedinodes */
+ if (IS_ERR(sd)) {
+ iput(si);
+ return false;
+ }
+ vi->sharedentry = sd;
INIT_LIST_HEAD(&vi->ishare_list);
spin_lock(&EROFS_I(si)->ishare_lock);
list_add(&vi->ishare_list, &EROFS_I(si)->ishare_list);
@@ -75,48 +83,40 @@ bool erofs_ishare_fill_inode(struct inode *inode)
void erofs_ishare_free_inode(struct inode *inode)
{
- struct erofs_inode *vi = EROFS_I(inode);
- struct inode *sharedinode = vi->sharedinode;
+ struct erofs_inode *vi = EROFS_I(inode), *svi;
- if (!sharedinode)
+ if (!vi->sharedentry)
return;
- spin_lock(&EROFS_I(sharedinode)->ishare_lock);
+ svi = EROFS_I(d_inode(vi->sharedentry));
+ spin_lock(&svi->ishare_lock);
list_del(&vi->ishare_list);
- spin_unlock(&EROFS_I(sharedinode)->ishare_lock);
- iput(sharedinode);
- vi->sharedinode = NULL;
+ spin_unlock(&svi->ishare_lock);
+ dput(vi->sharedentry);
+ vi->sharedentry = NULL;
}
static int erofs_ishare_file_open(struct inode *inode, struct file *file)
{
- struct inode *sharedinode = EROFS_I(inode)->sharedinode;
- struct file *realfile;
+ struct path sharedpath = {
+ .mnt = erofs_ishare_mnt,
+ .dentry = EROFS_I(inode)->sharedentry,
+ };
+ struct file *rf;
if (file->f_flags & O_DIRECT)
return -EINVAL;
- realfile = alloc_empty_backing_file(O_RDONLY|O_NOATIME, current_cred(),
- file);
- if (IS_ERR(realfile))
- return PTR_ERR(realfile);
- ihold(sharedinode);
- realfile->f_op = &erofs_file_fops;
- realfile->f_inode = sharedinode;
- realfile->f_mapping = sharedinode->i_mapping;
- path_get(&file->f_path);
- backing_file_set_user_path(realfile, &file->f_path);
-
- file_ra_state_init(&realfile->f_ra, file->f_mapping);
- realfile->private_data = EROFS_I(inode);
- file->private_data = realfile;
+
+ rf = backing_file_open(file, file->f_flags | O_NOATIME | FMODE_NONOTIFY,
+ &sharedpath, current_cred());
+ if (IS_ERR(rf))
+ return PTR_ERR(rf);
+ file->private_data = rf;
return 0;
}
static int erofs_ishare_file_release(struct inode *inode, struct file *file)
{
- struct file *realfile = file->private_data;
-
- iput(realfile->f_inode);
- fput(realfile);
+ fput(file->private_data);
file->private_data = NULL;
return 0;
}
diff --git a/fs/erofs/super.c b/fs/erofs/super.c
index 9d8f862f309f..8ead1646f329 100644
--- a/fs/erofs/super.c
+++ b/fs/erofs/super.c
@@ -147,8 +147,8 @@ static int erofs_init_device(struct erofs_buf *buf, struct super_block *sb,
if (!sbi->devs->flatdev) {
file = erofs_is_fileio_mode(sbi) ?
filp_open(dif->path, O_RDONLY | O_LARGEFILE, 0) :
- bdev_file_open_by_path(dif->path,
- BLK_OPEN_READ, sb->s_type, NULL);
+ fs_bdev_file_open_by_path(dif->path,
+ BLK_OPEN_READ, sb->s_type, sb);
if (IS_ERR(file)) {
if (file == ERR_PTR(-ENOTBLK))
return -EINVAL;
@@ -386,6 +386,7 @@ static void erofs_default_options(struct erofs_sb_info *sbi)
enum {
Opt_user_xattr, Opt_acl, Opt_cache_strategy, Opt_dax, Opt_dax_enum,
Opt_device, Opt_domain_id, Opt_directio, Opt_fsoffset, Opt_inode_share,
+ Opt_source,
};
static const struct constant_table erofs_param_cache_strategy[] = {
@@ -402,17 +403,18 @@ static const struct constant_table erofs_dax_param_enums[] = {
};
static const struct fs_parameter_spec erofs_fs_parameters[] = {
- fsparam_flag_no("user_xattr", Opt_user_xattr),
- fsparam_flag_no("acl", Opt_acl),
- fsparam_enum("cache_strategy", Opt_cache_strategy,
+ fsparam_flag_no("user_xattr", Opt_user_xattr),
+ fsparam_flag_no("acl", Opt_acl),
+ fsparam_enum("cache_strategy", Opt_cache_strategy,
erofs_param_cache_strategy),
- fsparam_flag("dax", Opt_dax),
- fsparam_enum("dax", Opt_dax_enum, erofs_dax_param_enums),
- fsparam_string("device", Opt_device),
- fsparam_string("domain_id", Opt_domain_id),
- fsparam_flag_no("directio", Opt_directio),
- fsparam_u64("fsoffset", Opt_fsoffset),
- fsparam_flag("inode_share", Opt_inode_share),
+ fsparam_flag("dax", Opt_dax),
+ fsparam_enum("dax", Opt_dax_enum, erofs_dax_param_enums),
+ fsparam_string("device", Opt_device),
+ fsparam_string("domain_id", Opt_domain_id),
+ fsparam_flag_no("directio", Opt_directio),
+ fsparam_u64("fsoffset", Opt_fsoffset),
+ fsparam_flag("inode_share", Opt_inode_share),
+ fsparam_file_or_string("source", Opt_source),
{}
};
@@ -437,6 +439,40 @@ static bool erofs_fc_set_dax_mode(struct fs_context *fc, unsigned int mode)
return false;
}
+static int erofs_fc_parse_source(struct fs_context *fc,
+ struct fs_parameter *param)
+{
+ struct erofs_sb_info *sbi = fc->s_fs_info;
+
+ if (fc->source || sbi->dif0.file)
+ return invalf(fc, "Multiple sources");
+
+ switch (param->type) {
+ case fs_value_is_string:
+ fc->source = param->string;
+ param->string = NULL;
+ return 0;
+ case fs_value_is_file: {
+ char *buf __free(kfree) = kmalloc(PATH_MAX, GFP_KERNEL);
+ char *p;
+
+ if (!buf)
+ return -ENOMEM;
+ p = file_path(param->file, buf, PATH_MAX);
+ if (IS_ERR(p))
+ return PTR_ERR(p);
+ fc->source = kstrdup(p, GFP_KERNEL);
+ if (!fc->source)
+ return -ENOMEM;
+ sbi->dif0.file = no_free_ptr(param->file);
+ return 0;
+ }
+ default:
+ WARN_ON_ONCE(true);
+ return -EINVAL;
+ }
+}
+
static int erofs_fc_parse_param(struct fs_context *fc,
struct fs_parameter *param)
{
@@ -524,6 +560,8 @@ static int erofs_fc_parse_param(struct fs_context *fc,
else
set_opt(&sbi->opt, INODE_SHARE);
break;
+ case Opt_source:
+ return erofs_fc_parse_source(fc, param);
}
return 0;
}
@@ -743,13 +781,26 @@ static int erofs_fc_fill_super(struct super_block *sb, struct fs_context *fc)
static int erofs_fc_get_tree(struct fs_context *fc)
{
+ struct erofs_sb_info *sbi = fc->s_fs_info;
int ret;
+ if (sbi->dif0.file) {
+ if (!IS_ENABLED(CONFIG_EROFS_FS_BACKED_BY_FILE)) {
+ errorfc(fc, "source fd option not supported");
+ return -EINVAL;
+ }
+ if (!S_ISREG(file_inode(sbi->dif0.file)->i_mode) ||
+ !sbi->dif0.file->f_mapping->a_ops->read_folio) {
+ errorfc(fc, "source is unsupported");
+ return -EINVAL;
+ }
+ return get_tree_nodev(fc, erofs_fc_fill_super);
+ }
+
ret = get_tree_bdev_flags(fc, erofs_fc_fill_super,
IS_ENABLED(CONFIG_EROFS_FS_BACKED_BY_FILE) ?
GET_TREE_BDEV_QUIET_LOOKUP : 0);
if (IS_ENABLED(CONFIG_EROFS_FS_BACKED_BY_FILE) && ret == -ENOTBLK) {
- struct erofs_sb_info *sbi = fc->s_fs_info;
struct file *file;
if (!fc->source)
@@ -790,28 +841,34 @@ static int erofs_fc_reconfigure(struct fs_context *fc)
static int erofs_release_device_info(int id, void *ptr, void *data)
{
+ struct super_block *sb = data;
struct erofs_device_info *dif = ptr;
fs_put_dax(dif->dax_dev, NULL);
- if (dif->file)
- fput(dif->file);
+ if (dif->file) {
+ if (S_ISBLK(file_inode(dif->file)->i_mode))
+ fs_bdev_file_release(dif->file, sb);
+ else
+ fput(dif->file);
+ }
kfree(dif->path);
kfree(dif);
return 0;
}
-static void erofs_free_dev_context(struct erofs_dev_context *devs)
+static void erofs_free_dev_context(struct erofs_dev_context *devs,
+ struct super_block *sb)
{
if (!devs)
return;
- idr_for_each(&devs->tree, &erofs_release_device_info, NULL);
+ idr_for_each(&devs->tree, &erofs_release_device_info, sb);
idr_destroy(&devs->tree);
kfree(devs);
}
-static void erofs_sb_free(struct erofs_sb_info *sbi)
+static void erofs_sb_free(struct erofs_sb_info *sbi, struct super_block *sb)
{
- erofs_free_dev_context(sbi->devs);
+ erofs_free_dev_context(sbi->devs, sb);
kfree_sensitive(sbi->domain_id);
if (sbi->dif0.file)
fput(sbi->dif0.file);
@@ -823,8 +880,13 @@ static void erofs_fc_free(struct fs_context *fc)
{
struct erofs_sb_info *sbi = fc->s_fs_info;
- if (sbi) /* free here if an error occurs before transferring to sb */
- erofs_sb_free(sbi);
+ /*
+ * Freed here only if an error occurs before the sb is set up; at that
+ * point no block-backed device has been claimed (that happens in
+ * fill_super), so the NULL sb never reaches fs_bdev_file_release().
+ */
+ if (sbi)
+ erofs_sb_free(sbi, NULL);
}
static const struct fs_context_operations erofs_context_ops = {
@@ -878,7 +940,7 @@ static void erofs_kill_sb(struct super_block *sb)
kill_block_super(sb);
erofs_drop_internal_inodes(sbi);
fs_put_dax(sbi->dif0.dax_dev, NULL);
- erofs_sb_free(sbi);
+ erofs_sb_free(sbi, sb);
sb->s_fs_info = NULL;
}
@@ -890,7 +952,7 @@ static void erofs_put_super(struct super_block *sb)
erofs_shrinker_unregister(sb);
erofs_xattr_prefixes_cleanup(sb);
erofs_drop_internal_inodes(sbi);
- erofs_free_dev_context(sbi->devs);
+ erofs_free_dev_context(sbi->devs, sb);
sbi->devs = NULL;
}
diff --git a/fs/exec.c b/fs/exec.c
index c7b8f2d6366c..41e1684d999c 100644
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1418,6 +1418,8 @@ static void free_bprm(struct linux_binprm *bprm)
/* If a binfmt changed the interp, free it. */
if (bprm->interp != bprm->filename)
kfree(bprm->interp);
+ kfree(bprm->bpf_interp);
+ kfree(bprm->bpf_interp_arg);
kfree(bprm->fdpath);
kfree(bprm);
}
diff --git a/fs/exfat/exfat_fs.h b/fs/exfat/exfat_fs.h
index 9be50949ce34..1f020b041a3d 100644
--- a/fs/exfat/exfat_fs.h
+++ b/fs/exfat/exfat_fs.h
@@ -294,7 +294,7 @@ struct exfat_inode_info {
/* on-disk position of directory entry or 0 */
loff_t i_pos;
loff_t valid_size;
- /* page-aligned size that has been zeroed out for mmap */
+ /* block-aligned size zeroed in the page cache (>= valid_size) */
loff_t zeroed_size;
/* hash by i_location */
struct hlist_node i_hash_fat;
diff --git a/fs/exfat/file.c b/fs/exfat/file.c
index 5fc13378d35f..5e9b47ecc614 100644
--- a/fs/exfat/file.c
+++ b/fs/exfat/file.c
@@ -16,6 +16,7 @@
#include <linux/falloc.h>
#include <linux/fileattr.h>
#include <linux/iomap.h>
+#include <linux/pagemap.h>
#include "exfat_raw.h"
#include "exfat_fs.h"
@@ -654,6 +655,104 @@ int exfat_file_fsync(struct file *filp, loff_t start, loff_t end, int datasync)
return blkdev_issue_flush(inode->i_sb->s_bdev);
}
+/*
+ * exfat_zero_new_range - zero [start, end) without overwriting uptodate blocks
+ *
+ * Uptodate blocks may contain data written through a shared mapping beyond
+ * valid_size.
+ */
+static int exfat_zero_new_range(struct inode *inode, loff_t start, loff_t end)
+{
+ struct address_space *mapping = inode->i_mapping;
+ unsigned int blocksize = i_blocksize(inode);
+ loff_t pos = start;
+ int err;
+
+ while (pos < end) {
+ loff_t next = min_t(loff_t,
+ round_down(pos, PAGE_SIZE) + PAGE_SIZE, end);
+ struct folio *folio;
+ loff_t bpos;
+
+ folio = filemap_get_folio(mapping, pos >> PAGE_SHIFT);
+ if (IS_ERR(folio)) {
+ err = iomap_zero_range(inode, pos, next - pos, NULL,
+ &exfat_iomap_ops, NULL, NULL);
+ if (err < 0)
+ return err;
+ pos = next;
+ continue;
+ }
+
+ if (folio_test_uptodate(folio)) {
+ folio_lock(folio);
+ if (folio->mapping == mapping)
+ folio_mark_dirty(folio);
+ folio_unlock(folio);
+ folio_put(folio);
+ pos = next;
+ continue;
+ }
+
+ /*
+ * Zero not-uptodate block runs. iomap_zero_range() requires an
+ * unlocked folio, so recheck ->mapping after each call.
+ */
+ folio_lock(folio);
+ bpos = pos;
+ while (bpos < next) {
+ loff_t rstart, rend;
+
+ if (folio->mapping != mapping) {
+ folio_unlock(folio);
+ err = iomap_zero_range(inode, bpos, next - bpos,
+ NULL, &exfat_iomap_ops, NULL, NULL);
+ if (err < 0) {
+ folio_put(folio);
+ return err;
+ }
+ folio_lock(folio);
+ break;
+ }
+
+ if (iomap_is_partially_uptodate(folio,
+ offset_in_folio(folio, bpos), blocksize)) {
+ bpos += blocksize;
+ continue;
+ }
+
+ rstart = bpos;
+ rend = min_t(loff_t, bpos + blocksize, next);
+ while (rend < next &&
+ !iomap_is_partially_uptodate(folio,
+ offset_in_folio(folio, rend), blocksize))
+ rend = min_t(loff_t, rend + blocksize, next);
+
+ folio_unlock(folio);
+ err = iomap_zero_range(inode, rstart, rend - rstart,
+ NULL, &exfat_iomap_ops, NULL, NULL);
+ if (err < 0) {
+ folio_put(folio);
+ return err;
+ }
+ folio_lock(folio);
+ bpos = rend;
+ }
+
+ /*
+ * Dirty only a fully uptodate folio. Dirtying a partial folio could
+ * write uninitialised cache contents over valid on-disk blocks.
+ */
+ if (folio->mapping == mapping && folio_test_uptodate(folio))
+ folio_mark_dirty(folio);
+ folio_unlock(folio);
+ folio_put(folio);
+ pos = next;
+ }
+
+ return 0;
+}
+
static int exfat_extend_valid_size(struct inode *inode, loff_t new_valid_size)
{
struct exfat_inode_info *ei = EXFAT_I(inode);
@@ -661,18 +760,41 @@ static int exfat_extend_valid_size(struct inode *inode, loff_t new_valid_size)
int ret = 0;
if (old_valid_size < new_valid_size) {
+ /* Do not re-zero blocks already covered by zeroed_size. */
+ loff_t gap_start = max(old_valid_size, ei->zeroed_size);
+
if (i_size_read(inode) < new_valid_size) {
- i_size_write(inode, new_valid_size);
- mark_inode_dirty(inode);
+ /*
+ * Allocate clusters before increasing i_size. The gap
+ * may already be zeroed, so the subsequent zeroing
+ * can be skipped.
+ */
+ ret = exfat_cont_expand(inode, new_valid_size);
+ if (ret)
+ return ret;
}
- ret = iomap_zero_range(inode, old_valid_size,
- new_valid_size - old_valid_size, NULL,
- &exfat_write_iomap_ops, NULL, NULL);
+ /*
+ * Revoke writable PTEs while zeroing the gap. A racing mmap
+ * store re-faults through exfat_page_mkwrite() after valid_size
+ * is updated.
+ */
+ filemap_invalidate_lock(inode->i_mapping);
+ if (gap_start < new_valid_size)
+ unmap_mapping_range(inode->i_mapping, gap_start,
+ new_valid_size - gap_start, 0);
+ ret = exfat_zero_new_range(inode, gap_start, new_valid_size);
+ filemap_invalidate_unlock(inode->i_mapping);
if (ret) {
truncate_setsize(inode, old_valid_size);
exfat_truncate(inode);
+ return ret;
}
+
+ ei->valid_size = new_valid_size;
+ if (ei->zeroed_size < round_up(new_valid_size, i_blocksize(inode)))
+ ei->zeroed_size = round_up(new_valid_size, i_blocksize(inode));
+ mark_inode_dirty(inode);
}
return ret;
@@ -825,39 +947,39 @@ static vm_fault_t exfat_page_mkwrite(struct vm_fault *vmf)
struct inode *inode = file_inode(vmf->vma->vm_file);
struct exfat_inode_info *ei = EXFAT_I(inode);
vm_fault_t ret;
- loff_t new_valid_size, mmap_valid_size;
+ loff_t new_valid_size, mmap_valid_size, fault_page_start;
if (!inode_trylock(inode))
return VM_FAULT_RETRY;
mmap_valid_size = ((loff_t)vmf->pgoff + 1) << PAGE_SHIFT;
+ fault_page_start = ((loff_t)vmf->pgoff) << PAGE_SHIFT;
new_valid_size = min(mmap_valid_size, i_size_read(inode));
if (ei->valid_size < new_valid_size) {
- if (ei->zeroed_size < mmap_valid_size) {
+ if (ei->zeroed_size < fault_page_start) {
int err;
/*
- * Only zero the range that hasn't been zeroed yet for
- * this mmap write path. zeroed_size tracks the largest
- * page-aligned offset that has already been zeroed.
- *
- * This prevents unnecessarily zeroing out the entire
- * tail page on every page fault when userspace writes
- * data byte-by-byte through mmap (after a small
- * fallocate). It fixes data corruption in the tail page
- * while preserving the existing valid_size semantics.
+ * Zero only the gap below the faulting page. The read
+ * fault populated its folio and iomap_page_mkwrite()
+ * will dirty it.
*/
- err = iomap_zero_range(inode, ei->zeroed_size,
- mmap_valid_size - ei->zeroed_size, NULL,
- &exfat_iomap_ops, NULL, NULL);
+ err = exfat_zero_new_range(inode, ei->zeroed_size,
+ fault_page_start);
if (err < 0) {
inode_unlock(inode);
return vmf_fs_error(err);
}
- ei->zeroed_size = mmap_valid_size;
}
+ /*
+ * Track zeroed_size by block, not page, because writeback stops
+ * at i_size recording blocks wholly beyond it could skip a
+ * later required zeroing.
+ */
+ if (ei->zeroed_size < round_up(new_valid_size, i_blocksize(inode)))
+ ei->zeroed_size = round_up(new_valid_size, i_blocksize(inode));
ei->valid_size = new_valid_size;
mark_inode_dirty(inode);
}
@@ -866,7 +988,7 @@ static vm_fault_t exfat_page_mkwrite(struct vm_fault *vmf)
file_update_time(vmf->vma->vm_file);
filemap_invalidate_lock_shared(inode->i_mapping);
- ret = iomap_page_mkwrite(vmf, &exfat_write_iomap_ops, NULL);
+ ret = iomap_page_mkwrite(vmf, &exfat_iomap_ops, NULL);
filemap_invalidate_unlock_shared(inode->i_mapping);
sb_end_pagefault(inode->i_sb);
inode_unlock(inode);
@@ -876,7 +998,6 @@ static vm_fault_t exfat_page_mkwrite(struct vm_fault *vmf)
static const struct vm_operations_struct exfat_file_vm_ops = {
.fault = filemap_fault,
- .map_pages = filemap_map_pages,
.page_mkwrite = exfat_page_mkwrite,
};
@@ -887,21 +1008,6 @@ static int exfat_file_mmap_prepare(struct vm_area_desc *desc)
if (unlikely(exfat_forced_shutdown(file_inode(desc->file)->i_sb)))
return -EIO;
- if (vma_desc_test_all(desc, VMA_SHARED_BIT, VMA_MAYWRITE_BIT)) {
- struct inode *inode = file_inode(file);
- loff_t from, to;
- int err;
-
- from = ((loff_t)desc->pgoff << PAGE_SHIFT);
- to = min_t(loff_t, i_size_read(inode),
- from + vma_desc_size(desc));
- if (EXFAT_I(inode)->valid_size < to) {
- err = exfat_extend_valid_size(inode, to);
- if (err)
- return err;
- }
- }
-
file_accessed(file);
desc->vm_ops = &exfat_file_vm_ops;
return 0;
diff --git a/fs/exfat/iomap.c b/fs/exfat/iomap.c
index 190fc6471f84..d4d3ed933a63 100644
--- a/fs/exfat/iomap.c
+++ b/fs/exfat/iomap.c
@@ -175,11 +175,18 @@ static int exfat_write_iomap_end(struct inode *inode, loff_t pos, loff_t length,
if (ei->valid_size < end) {
ei->valid_size = end;
- if (ei->zeroed_size < end)
- ei->zeroed_size = end;
dirtied = true;
}
+ /*
+ * IOMAP_F_ZERO_TAIL zeroes the remainder of the last block. Track that
+ * block as zeroed so later valid_size extensions do not zero it again.
+ */
+ if (iomap->flags & IOMAP_F_ZERO_TAIL)
+ end = round_up(end, i_blocksize(inode));
+ if (ei->zeroed_size < end)
+ ei->zeroed_size = end;
+
if (dirtied || iomap->flags & IOMAP_F_SIZE_CHANGED)
mark_inode_dirty(inode);
diff --git a/fs/exfat/namei.c b/fs/exfat/namei.c
index b7d5e44ad38e..cd9c9eca58f8 100644
--- a/fs/exfat/namei.c
+++ b/fs/exfat/namei.c
@@ -538,7 +538,7 @@ out:
}
static int exfat_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct super_block *sb = dir->i_sb;
struct inode *inode;
diff --git a/fs/ext2/namei.c b/fs/ext2/namei.c
index 0d09d22fe708..8666233ec63b 100644
--- a/fs/ext2/namei.c
+++ b/fs/ext2/namei.c
@@ -99,7 +99,7 @@ struct dentry *ext2_get_parent(struct dentry *child)
*/
static int ext2_create (struct mnt_idmap * idmap,
struct inode * dir, struct dentry * dentry,
- umode_t mode, bool excl)
+ umode_t mode)
{
struct inode *inode;
int err;
@@ -236,7 +236,7 @@ static struct dentry *ext2_mkdir(struct mnt_idmap * idmap,
inode_inc_link_count(dir);
- inode = ext2_new_inode(dir, S_IFDIR | mode, &dentry->d_name);
+ inode = ext2_new_inode(dir, mode, &dentry->d_name);
err = PTR_ERR(inode);
if (IS_ERR(inode))
goto out_dir;
diff --git a/fs/ext4/crypto.c b/fs/ext4/crypto.c
index f41f320f4437..9265cfe62c83 100644
--- a/fs/ext4/crypto.c
+++ b/fs/ext4/crypto.c
@@ -236,7 +236,7 @@ static bool ext4_has_stable_inodes(struct super_block *sb)
const struct fscrypt_operations ext4_cryptops = {
.inode_info_offs = (int)offsetof(struct ext4_inode_info, i_crypt_info) -
(int)offsetof(struct ext4_inode_info, vfs_inode),
- .needs_bounce_pages = 1,
+ .is_block_based = 1,
.has_32bit_inodes = 1,
.supports_subblock_data_units = 1,
.legacy_key_prefix = "ext4:",
diff --git a/fs/ext4/dir.c b/fs/ext4/dir.c
index 17edd678fa87..8d7b81e6948e 100644
--- a/fs/ext4/dir.c
+++ b/fs/ext4/dir.c
@@ -138,6 +138,7 @@ static int ext4_readdir(struct file *file, struct dir_context *ctx)
struct buffer_head *bh = NULL;
struct fscrypt_str fstr = FSTR_INIT(NULL, 0);
struct dir_private_info *info = file->private_data;
+ bool has_csum = ext4_has_feature_metadata_csum(sb);
err = fscrypt_prepare_readdir(inode);
if (err)
@@ -149,7 +150,7 @@ static int ext4_readdir(struct file *file, struct dir_context *ctx)
return err;
/* Can we just clear INDEX flag to ignore htree information? */
- if (!ext4_has_feature_metadata_csum(sb)) {
+ if (!has_csum) {
/*
* We don't set the inode dirty flag since it's not
* critical that it gets flushed back to the disk.
@@ -235,7 +236,10 @@ static int ext4_readdir(struct file *file, struct dir_context *ctx)
* dirent right now. Scan from the start of the block
* to make sure. */
if (!inode_eq_iversion(inode, info->cookie)) {
- for (i = 0; i < sb->s_blocksize && i < offset; ) {
+ for (i = 0;
+ i <= sb->s_blocksize -
+ ext4_dir_rec_len(1, has_csum ? NULL : inode) &&
+ i < offset;) {
de = (struct ext4_dir_entry_2 *)
(bh->b_data + i);
/* It's too expensive to do a full
@@ -257,6 +261,17 @@ static int ext4_readdir(struct file *file, struct dir_context *ctx)
info->cookie = inode_query_iversion(inode);
}
+ if (unlikely(offset < sb->s_blocksize &&
+ offset > sb->s_blocksize -
+ ext4_dir_rec_len(1, has_csum ? NULL : inode))) {
+ EXT4_ERROR_FILE(file, bh->b_blocknr,
+ "bad entry in directory: %s - offset=%u, size=%lu",
+ "directory entry too close to block end",
+ offset, sb->s_blocksize);
+ ctx->pos = round_up(ctx->pos, sb->s_blocksize);
+ goto next_block;
+ }
+
while (ctx->pos < inode->i_size
&& offset < sb->s_blocksize) {
de = (struct ext4_dir_entry_2 *) (bh->b_data + offset);
@@ -312,6 +327,7 @@ static int ext4_readdir(struct file *file, struct dir_context *ctx)
ctx->pos += ext4_rec_len_from_disk(de->rec_len,
sb->s_blocksize);
}
+next_block:
if ((ctx->pos < inode->i_size) && !dir_relax_shared(inode))
goto done;
brelse(bh);
diff --git a/fs/ext4/ext4.h b/fs/ext4/ext4.h
index b37c136ea3ab..4e3b3165ee8f 100644
--- a/fs/ext4/ext4.h
+++ b/fs/ext4/ext4.h
@@ -334,7 +334,7 @@ struct ext4_io_submit {
#define EXT4_MAX_BLOCK_SIZE 65536
#define EXT4_MIN_BLOCK_LOG_SIZE 10
#define EXT4_MAX_BLOCK_LOG_SIZE 16
-#define EXT4_MAX_CLUSTER_LOG_SIZE 30
+#define EXT4_MAX_CLUSTER_LOG_SIZE 28
#ifdef __KERNEL__
# define EXT4_BLOCK_SIZE(s) ((s)->s_blocksize)
#else
@@ -1070,8 +1070,14 @@ struct ext4_inode_info {
* between readers of EAs and writers of regular file data, so
* instead we synchronize on xattr_sem when reading or changing
* EAs.
+ *
+ * EA inodes (EXT4_EA_INODE_FL) do not use xattr_sem; they reuse
+ * the space for deferred iput linkage.
*/
- struct rw_semaphore xattr_sem;
+ union {
+ struct rw_semaphore xattr_sem;
+ struct llist_node i_ea_iput_node;
+ };
/*
* Inodes with EXT4_STATE_ORPHAN_FILE use i_orphan_idx. Otherwise
@@ -1770,6 +1776,11 @@ struct ext4_sb_info {
struct ext4_es_stats s_es_stats;
struct mb_cache *s_ea_block_cache;
struct mb_cache *s_ea_inode_cache;
+
+ /* Deferred iput for EA inodes to avoid lock ordering issues */
+ struct llist_head s_ea_inode_to_free;
+ struct delayed_work s_ea_inode_work;
+
spinlock_t s_es_lock ____cacheline_aligned_in_smp;
/* Journal triggers for checksum computation */
@@ -3137,14 +3148,15 @@ int do_journal_get_write_access(handle_t *handle, struct inode *inode,
struct buffer_head *bh);
void ext4_set_inode_mapping_order(struct inode *inode);
#define FALL_BACK_TO_NONDELALLOC 1
-#define CONVERT_INLINE_DATA 2
+#define EXT4_WRITE_DATA_INLINE 2
typedef enum {
EXT4_IGET_NORMAL = 0,
EXT4_IGET_SPECIAL = 0x0001, /* OK to iget a system inode */
EXT4_IGET_HANDLE = 0x0002, /* Inode # is from a handle */
EXT4_IGET_BAD = 0x0004, /* Allow to iget a bad inode */
- EXT4_IGET_EA_INODE = 0x0008 /* Inode should contain an EA value */
+ EXT4_IGET_EA_INODE = 0x0008, /* Inode should contain an EA value */
+ EXT4_IGET_NOWAIT = 0x0010 /* Non-blocking lookup (skip if freeing) */
} ext4_iget_flags;
extern struct inode *__ext4_iget(struct super_block *sb, unsigned long ino,
@@ -3186,8 +3198,11 @@ extern int ext4_chunk_trans_extent(struct inode *inode, int nrblocks);
extern int ext4_meta_trans_blocks(struct inode *inode, int lblocks,
int pextents);
extern int ext4_block_zero_eof(struct inode *inode, loff_t from, loff_t end);
+
+#define EXT4_PARTIAL_ZERO_START 0x1
+#define EXT4_PARTIAL_ZERO_END 0x2
extern int ext4_zero_partial_blocks(struct inode *inode, loff_t lstart,
- loff_t length, bool *did_zero);
+ loff_t length, unsigned int *partial_zeroed);
extern vm_fault_t ext4_page_mkwrite(struct vm_fault *vmf);
extern qsize_t *ext4_get_reserved_space(struct inode *inode);
extern int ext4_get_projid(struct inode *inode, kprojid_t *projid);
@@ -3639,7 +3654,13 @@ struct ext4_group_info {
#define EXT4_MB_GRP_CLEAR_TRIMMED(grp) \
(clear_bit(EXT4_GROUP_INFO_WAS_TRIMMED_BIT, &((grp)->bb_state)))
#define EXT4_MB_GRP_TEST_AND_SET_READ(grp) \
- (test_and_set_bit(EXT4_GROUP_INFO_BBITMAP_READ_BIT, &((grp)->bb_state)))
+ (ext4_mb_grp_test_and_set_read((grp)))
+
+static inline int ext4_mb_grp_test_and_set_read(struct ext4_group_info *grp)
+{
+ return (test_bit(EXT4_GROUP_INFO_BBITMAP_READ_BIT, &grp->bb_state) ||
+ test_and_set_bit(EXT4_GROUP_INFO_BBITMAP_READ_BIT, &grp->bb_state));
+}
#define EXT4_MAX_CONTENTION 8
#define EXT4_CONTENTION_THRESHOLD 2
@@ -3748,7 +3769,7 @@ extern int ext4_generic_write_inline_data(struct address_space *mapping,
struct inode *inode,
loff_t pos, unsigned len,
struct folio **foliop,
- void **fsdata, bool da);
+ bool da);
extern int ext4_try_add_inline_entry(handle_t *handle,
struct ext4_filename *fname,
struct inode *dir, struct inode *inode);
@@ -3829,8 +3850,8 @@ static inline void ext4_set_de_type(struct super_block *sb,
/* readpages.c */
int ext4_read_folio(struct file *file, struct folio *folio);
void ext4_readahead(struct readahead_control *rac);
-extern int __init ext4_init_post_read_processing(void);
-extern void ext4_exit_post_read_processing(void);
+int __init ext4_init_verity_caches(void);
+void ext4_exit_verity_caches(void);
/* symlink.c */
extern const struct inode_operations ext4_encrypted_symlink_inode_operations;
@@ -3945,7 +3966,7 @@ extern void ext4_io_submit_init(struct ext4_io_submit *io,
struct writeback_control *wbc);
extern void ext4_end_io_rsv_work(struct work_struct *work);
extern void ext4_io_submit(struct ext4_io_submit *io);
-int ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *page,
+void ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *page,
size_t len);
extern struct ext4_io_end_vec *ext4_alloc_io_end_vec(ext4_io_end_t *io_end);
extern struct ext4_io_end_vec *ext4_last_io_end_vec(ext4_io_end_t *io_end);
diff --git a/fs/ext4/ext4_jbd2.c b/fs/ext4/ext4_jbd2.c
index 9a8c225f2753..b4dacd1a89e7 100644
--- a/fs/ext4/ext4_jbd2.c
+++ b/fs/ext4/ext4_jbd2.c
@@ -33,14 +33,22 @@ int ext4_inode_journal_mode(struct inode *inode)
static handle_t *ext4_get_nojournal(void)
{
handle_t *handle = current->journal_info;
- unsigned long ref_cnt = (unsigned long)handle;
- BUG_ON(ref_cnt >= EXT4_NOJOURNAL_MAX_REF_COUNT);
-
- ref_cnt++;
- handle = (handle_t *)ref_cnt;
-
- current->journal_info = handle;
+ BUG_ON(handle && !handle->h_invalid);
+
+ if (!handle) {
+ handle = jbd2_alloc_handle(GFP_NOFS);
+ if (!handle)
+ return ERR_PTR(-ENOMEM);
+ handle->h_invalid = 1;
+ /*
+ * This is done by start_this_handle() if journalling
+ * is enabled.
+ */
+ handle->saved_alloc_context = memalloc_nofs_save();
+ current->journal_info = handle;
+ }
+ handle->h_ref++;
return handle;
}
@@ -48,14 +56,14 @@ static handle_t *ext4_get_nojournal(void)
/* Decrement the non-pointer handle value */
static void ext4_put_nojournal(handle_t *handle)
{
- unsigned long ref_cnt = (unsigned long)handle;
+ BUG_ON(handle->h_ref == 0);
- BUG_ON(ref_cnt == 0);
-
- ref_cnt--;
- handle = (handle_t *)ref_cnt;
-
- current->journal_info = handle;
+ handle->h_ref--;
+ if (handle->h_ref == 0) {
+ memalloc_nofs_restore(handle->saved_alloc_context);
+ jbd2_free_handle(handle);
+ current->journal_info = NULL;
+ }
}
/*
diff --git a/fs/ext4/ext4_jbd2.h b/fs/ext4/ext4_jbd2.h
index 63d17c5201b5..2fbf48b3dfe2 100644
--- a/fs/ext4/ext4_jbd2.h
+++ b/fs/ext4/ext4_jbd2.h
@@ -182,15 +182,11 @@ handle_t *__ext4_journal_start_sb(struct inode *inode, struct super_block *sb,
int rsv_blocks, int revoke_creds);
int __ext4_journal_stop(const char *where, unsigned int line, handle_t *handle);
-#define EXT4_NOJOURNAL_MAX_REF_COUNT ((unsigned long) 4096)
-
/* Note: Do not use this for NULL handles. This is only to determine if
* a properly allocated handle is using a journal or not. */
static inline int ext4_handle_valid(handle_t *handle)
{
- if ((unsigned long)handle < EXT4_NOJOURNAL_MAX_REF_COUNT)
- return 0;
- return 1;
+ return (handle && !handle->h_invalid);
}
static inline void ext4_handle_sync(handle_t *handle)
diff --git a/fs/ext4/extents-test.c b/fs/ext4/extents-test.c
index bd7795a82607..c3836ecb89f9 100644
--- a/fs/ext4/extents-test.c
+++ b/fs/ext4/extents-test.c
@@ -126,11 +126,6 @@ struct kunit_ext_test_param {
struct kunit_ext_data_state exp_data_state[3];
};
-static void ext_kill_sb(struct super_block *sb)
-{
- generic_shutdown_super(sb);
-}
-
static int ext_init_fs_context(struct fs_context *fc)
{
return 0;
@@ -138,13 +133,13 @@ static int ext_init_fs_context(struct fs_context *fc)
static int ext_set(struct super_block *sb, struct fs_context *fc)
{
- return 0;
+ return set_anon_super_fc(sb, fc);
}
static struct file_system_type ext_fs_type = {
.name = "extents test",
.init_fs_context = ext_init_fs_context,
- .kill_sb = ext_kill_sb,
+ .kill_sb = kill_anon_super,
};
static void extents_kunit_exit(struct kunit *test)
diff --git a/fs/ext4/extents.c b/fs/ext4/extents.c
index 91c97af64b31..d5f87a7f6c05 100644
--- a/fs/ext4/extents.c
+++ b/fs/ext4/extents.c
@@ -4715,7 +4715,7 @@ static long ext4_zero_range(struct file *file, loff_t offset,
loff_t align_start, align_end, new_size = 0;
loff_t end = offset + len;
unsigned int blocksize = i_blocksize(inode);
- bool partial_zeroed = false;
+ unsigned int partial_zeroed = 0;
int ret, flags;
trace_ext4_zero_range(inode, offset, len, mode);
@@ -4734,10 +4734,16 @@ static long ext4_zero_range(struct file *file, loff_t offset,
}
flags = EXT4_GET_BLOCKS_CREATE_UNWRIT_EXT;
- /* Preallocate the range including the unaligned edges */
+ /*
+ * Preallocate the range including the unaligned edges, and zero
+ * out partial blocks if they already contain data.
+ */
if (!IS_ALIGNED(offset | end, blocksize)) {
ret = ext4_alloc_file_blocks(file, offset, len, new_size,
flags);
+ if (!ret)
+ ret = ext4_zero_partial_blocks(inode, offset, len,
+ &partial_zeroed);
if (ret)
return ret;
}
@@ -4754,6 +4760,21 @@ static long ext4_zero_range(struct file *file, loff_t offset,
/* Zero range excluding the unaligned edges */
align_start = round_up(offset, blocksize);
align_end = round_down(end, blocksize);
+
+ /*
+ * In WRITE_ZEROES mode, edges that were not partial-zeroed (clean
+ * unwritten or hole) must be allocated and zeroed as whole blocks.
+ * Expand the aligned range outward to cover them.
+ */
+ if (mode & FALLOC_FL_WRITE_ZEROES) {
+ if (!IS_ALIGNED(offset, blocksize) &&
+ !(partial_zeroed & EXT4_PARTIAL_ZERO_START))
+ align_start = round_down(offset, blocksize);
+ if (!IS_ALIGNED(end, blocksize) &&
+ !(partial_zeroed & EXT4_PARTIAL_ZERO_END))
+ align_end = round_up(end, blocksize);
+ }
+
if (align_end > align_start) {
if (mode & FALLOC_FL_WRITE_ZEROES)
flags = EXT4_GET_BLOCKS_CREATE_ZERO | EXT4_EX_NOCACHE;
@@ -4770,11 +4791,15 @@ static long ext4_zero_range(struct file *file, loff_t offset,
if (IS_ALIGNED(offset | end, blocksize))
return ret;
- /* Zero out partial block at the edges of the range */
- ret = ext4_zero_partial_blocks(inode, offset, len, &partial_zeroed);
- if (ret)
- return ret;
- if (((file->f_flags & O_SYNC) || IS_SYNC(inode)) && partial_zeroed) {
+ /*
+ * In FALLOC_FL_WRITE_ZEROES mode, edges that have been partially
+ * zeroed must be written back to ensure the entire zeroed range
+ * is converted to the written state. In SYNC mode, writeback is
+ * also required to persist the zeroed data to disk.
+ */
+ if (partial_zeroed &&
+ ((mode & FALLOC_FL_WRITE_ZEROES) ||
+ (file->f_flags & O_SYNC) || IS_SYNC(inode))) {
ret = filemap_write_and_wait_range(inode->i_mapping, offset,
end - 1);
if (ret)
diff --git a/fs/ext4/fast_commit.c b/fs/ext4/fast_commit.c
index 8e2259799614..062103e42cd8 100644
--- a/fs/ext4/fast_commit.c
+++ b/fs/ext4/fast_commit.c
@@ -200,25 +200,6 @@ static inline void ext4_fc_set_snap_err(int *snap_err, int err)
*snap_err = err;
}
-static void ext4_end_buffer_io_sync(struct bio *bio)
-{
- struct buffer_head *bh;
- bool uptodate = bio_endio_bh(bio, &bh);
-
- BUFFER_TRACE(bh, "");
- if (uptodate) {
- ext4_debug("%s: Block %lld up-to-date",
- __func__, bh->b_blocknr);
- set_buffer_uptodate(bh);
- } else {
- ext4_debug("%s: Block %lld not up-to-date",
- __func__, bh->b_blocknr);
- clear_buffer_uptodate(bh);
- }
-
- unlock_buffer(bh);
-}
-
static void ext4_fc_free_inode_snap(struct inode *inode);
static inline void ext4_fc_reset_inode(struct inode *inode)
@@ -691,7 +672,7 @@ static void ext4_fc_submit_bh(struct super_block *sb, bool is_tail)
lock_buffer(bh);
set_buffer_dirty(bh);
set_buffer_uptodate(bh);
- bh_submit(bh, REQ_OP_WRITE | write_flags, ext4_end_buffer_io_sync);
+ bh_submit(bh, REQ_OP_WRITE | write_flags, bh_end_write);
EXT4_SB(sb)->s_fc_bh = NULL;
}
@@ -2196,8 +2177,11 @@ static int ext4_fc_replay_add_range(struct super_block *sb, u8 *val)
if (ret == 0) {
/* Range is not mapped */
path = ext4_find_extent(inode, cur, path, 0);
- if (IS_ERR(path))
+ if (IS_ERR(path)) {
+ ret = PTR_ERR(path);
+ path = NULL;
goto out;
+ }
memset(&newex, 0, sizeof(newex));
newex.ee_block = cpu_to_le32(cur);
ext4_ext_store_pblock(
@@ -2209,8 +2193,11 @@ static int ext4_fc_replay_add_range(struct super_block *sb, u8 *val)
path = ext4_ext_insert_extent(NULL, inode,
path, &newex, 0);
up_write((&EXT4_I(inode)->i_data_sem));
- if (IS_ERR(path))
+ if (IS_ERR(path)) {
+ ret = PTR_ERR(path);
+ path = NULL;
goto out;
+ }
goto next;
}
@@ -2257,10 +2244,11 @@ next:
}
ext4_ext_replay_shrink_inode(inode, i_size_read(inode) >>
sb->s_blocksize_bits);
+ ret = 0;
out:
ext4_free_ext_path(path);
iput(inode);
- return 0;
+ return ret;
}
/* Replay DEL_RANGE tag */
@@ -2320,9 +2308,10 @@ ext4_fc_replay_del_range(struct super_block *sb, u8 *val)
ext4_ext_replay_shrink_inode(inode,
i_size_read(inode) >> sb->s_blocksize_bits);
ext4_mark_inode_dirty(NULL, inode);
+ ret = 0;
out:
iput(inode);
- return 0;
+ return ret;
}
static void ext4_fc_set_bitmaps_and_counters(struct super_block *sb)
diff --git a/fs/ext4/file.c b/fs/ext4/file.c
index eb1a323962b1..9a16071b719d 100644
--- a/fs/ext4/file.c
+++ b/fs/ext4/file.c
@@ -213,31 +213,60 @@ ext4_extending_io(struct inode *inode, loff_t offset, size_t len)
return false;
}
-/* Is IO overwriting allocated or initialized blocks? */
-static bool ext4_overwrite_io(struct inode *inode,
- loff_t pos, loff_t len, bool *unwritten)
+/*
+ * Does an unaligned DIO write require partial block zeroing?
+ *
+ * Partial block zeroing is performed only for the head and tail blocks
+ * when they are partially covered by the write and the underlying extent
+ * is a hole or unwritten. Middle blocks (fully covered by the write)
+ * are written as whole blocks without zeroing.
+ *
+ * When zeroing is required, two concurrent unaligned DIO writes to the
+ * same partial block can race and corrupt each other's data, so the
+ * caller must take the exclusive i_rwsem and drain in-flight DIO. When
+ * zeroing is not required, shared lock is safe -- block allocation and
+ * unwritten conversion for middle blocks are protected by i_data_sem
+ * and inode_dio_begin().
+ */
+static bool ext4_dio_needs_zeroing(struct inode *inode, loff_t pos, loff_t len)
{
struct ext4_map_blocks map;
unsigned int blkbits = inode->i_blkbits;
- int err, blklen;
+ unsigned long blockmask = inode->i_sb->s_blocksize - 1;
+ bool head_partial, tail_partial;
+ ext4_lblk_t head_lblk, tail_lblk;
+ int err;
if (pos + len > i_size_read(inode))
- return false;
+ return true;
- map.m_lblk = pos >> blkbits;
- map.m_len = EXT4_MAX_BLOCKS(len, pos, blkbits);
- blklen = map.m_len;
+ head_partial = (pos & blockmask) != 0;
+ tail_partial = ((pos + len) & blockmask) != 0;
+ head_lblk = pos >> blkbits;
+ tail_lblk = (pos + len - 1) >> blkbits;
+
+ /* Check the head partial block. */
+ if (head_partial) {
+ map.m_lblk = head_lblk;
+ map.m_len = tail_lblk - head_lblk + 1;
+ err = ext4_map_blocks(NULL, inode, &map, 0);
+ if (err <= 0 || !(map.m_flags & EXT4_MAP_MAPPED))
+ return true;
+ /* If this mapping already covers the tail block, we're done. */
+ if (!tail_partial || map.m_lblk + err > tail_lblk)
+ return false;
+ }
- err = ext4_map_blocks(NULL, inode, &map, 0);
- if (err != blklen)
- return false;
- /*
- * 'err==len' means that all of the blocks have been preallocated,
- * regardless of whether they have been initialized or not. We need to
- * check m_flags to distinguish the unwritten extents.
- */
- *unwritten = !(map.m_flags & EXT4_MAP_MAPPED);
- return true;
+ /* Check the tail partial block. */
+ if (tail_partial) {
+ map.m_lblk = tail_lblk;
+ map.m_len = 1;
+ err = ext4_map_blocks(NULL, inode, &map, 0);
+ if (err <= 0 || !(map.m_flags & EXT4_MAP_MAPPED))
+ return true;
+ }
+
+ return false;
}
static ssize_t ext4_generic_write_checks(struct kiocb *iocb,
@@ -278,7 +307,7 @@ static ssize_t ext4_write_checks(struct kiocb *iocb, struct iov_iter *from)
if (count <= 0)
return count;
- ret = file_modified(iocb->ki_filp);
+ ret = kiocb_modified(iocb);
if (ret)
return ret;
@@ -309,6 +338,13 @@ static ssize_t ext4_buffered_write_iter(struct kiocb *iocb,
return -EOPNOTSUPP;
inode_lock(inode);
+
+ /*
+ * Prevent concurrent direct I/O and buffered I/O to the same file
+ * range. Wait for in-flight DIO to finish before dirtying pages.
+ */
+ inode_dio_wait(inode);
+
ret = ext4_write_checks(iocb, from);
if (ret <= 0)
goto out;
@@ -428,16 +464,28 @@ static const struct iomap_dio_ops ext4_dio_write_ops = {
* condition requires an exclusive inode lock. If yes, then we restart the
* whole operation by releasing the shared lock and acquiring exclusive lock.
*
- * - For unaligned_io we never take shared lock as it may cause data corruption
- * when two unaligned IO tries to modify the same block e.g. while zeroing.
+ * The decision is layered, evaluated in this order:
*
- * - For extending writes case we don't take the shared lock, since it requires
- * updating inode i_disksize and/or orphan handling with exclusive lock.
+ * 1. If kiocb_modified() needs to update security info (!IS_NOSEC), upgrade
+ * to the exclusive lock -- the security update itself requires it,
+ * regardless of whether the write extends the file or is aligned.
*
- * - shared locking will only be true mostly with overwrites, including
- * initialized blocks and unwritten blocks.
+ * 2. If the write extends i_size or i_disksize, upgrade to the exclusive
+ * lock to safely update i_disksize and the orphan list, regardless of
+ * alignment.
*
- * - Otherwise we will switch to exclusive i_rwsem lock.
+ * 3. Otherwise, for aligned non-extending writes, shared lock is always
+ * sufficient regardless of extent state (written, unwritten, or hole).
+ * truncate/punch_hole cannot run while we hold the shared i_rwsem
+ * (they need it exclusively); after we release it, inode_dio_begin()
+ * keeps their inode_dio_wait() blocked until in-flight bios complete.
+ * i_data_sem serializes concurrent extent tree modifications.
+ *
+ * 4. Otherwise, the write is unaligned and non-extending. Shared lock is
+ * safe unless the DIO layer needs to perform partial block zeroing --
+ * i.e. the head or tail partial block sits on a hole or unwritten
+ * extent. In that case upgrade to the exclusive lock and drain
+ * in-flight DIO to avoid races with concurrent partial block zeroing.
*/
static ssize_t ext4_dio_write_checks(struct kiocb *iocb, struct iov_iter *from,
bool *ilock_shared, bool *extend,
@@ -448,7 +496,7 @@ static ssize_t ext4_dio_write_checks(struct kiocb *iocb, struct iov_iter *from,
loff_t offset;
size_t count;
ssize_t ret;
- bool overwrite, unaligned_io, unwritten;
+ bool needs_zeroing = false;
restart:
ret = ext4_generic_write_checks(iocb, from);
@@ -458,24 +506,22 @@ restart:
offset = iocb->ki_pos;
count = ret;
- unaligned_io = ext4_unaligned_io(inode, from, offset);
*extend = ext4_extending_io(inode, offset, count);
- overwrite = ext4_overwrite_io(inode, offset, count, &unwritten);
/*
- * Determine whether we need to upgrade to an exclusive lock. This is
- * required to change security info in file_modified(), for extending
- * I/O, any form of non-overwrite I/O, and unaligned I/O to unwritten
- * extents (as partial block zeroing may be required).
+ * For unaligned writes, check whether partial block zeroing will be
+ * needed. If so, exclusive lock is required to serialize against
+ * concurrent DIO that could race with the zeroing.
*
- * Note that unaligned writes are allowed under shared lock so long as
- * they are pure overwrites. Otherwise, concurrent unaligned writes risk
- * data corruption due to partial block zeroing in the dio layer, and so
- * the I/O must occur exclusively.
+ * For aligned writes we skip this check entirely since allocation
+ * under shared lock is safe.
*/
+ if (ext4_unaligned_io(inode, from, offset))
+ needs_zeroing = ext4_dio_needs_zeroing(inode, offset, count);
+
+ /* Determine whether we need to upgrade to an exclusive lock. */
if (*ilock_shared &&
- ((!IS_NOSEC(inode) || *extend || !overwrite ||
- (unaligned_io && unwritten)))) {
+ (!IS_NOSEC(inode) || *extend || needs_zeroing)) {
if (iocb->ki_flags & IOCB_NOWAIT) {
ret = -EAGAIN;
goto out;
@@ -489,21 +535,28 @@ restart:
/*
* Now that locking is settled, determine dio flags and exclusivity
* requirements. We don't use DIO_OVERWRITE_ONLY because we enforce
- * behavior already. The inode lock is already held exclusive if the
- * write is non-overwrite or extending, so drain all outstanding dio and
- * set the force wait dio flag.
+ * behavior already. When holding the exclusive lock for a write that
+ * needs partial block zeroing or is extending the file, we must wait
+ * for the I/O to complete synchronously:
+ *
+ * - needs_zeroing: drain in-flight DIO whose end_io could race with
+ * our partial block zeroing, and force synchronous completion so we
+ * don't leave in-flight zeroing bios for the next writer to drain.
+ *
+ * - extend: the caller must update i_disksize after I/O completion,
+ * which requires the data to be on disk first.
*/
- if (!*ilock_shared && (unaligned_io || *extend)) {
+ if (!*ilock_shared && (needs_zeroing || *extend)) {
if (iocb->ki_flags & IOCB_NOWAIT) {
ret = -EAGAIN;
goto out;
}
- if (unaligned_io && (!overwrite || unwritten))
+ if (needs_zeroing)
inode_dio_wait(inode);
*dio_flags = IOMAP_DIO_FORCE_WAIT;
}
- ret = file_modified(file);
+ ret = kiocb_modified(iocb);
if (ret < 0)
goto out;
@@ -672,6 +725,11 @@ ext4_dax_write_iter(struct kiocb *iocb, struct iov_iter *from)
count = iov_iter_count(from);
if (offset + count > EXT4_I(inode)->i_disksize) {
+ if (iocb->ki_flags & IOCB_NOWAIT) {
+ ret = -EAGAIN;
+ goto out;
+ }
+
handle = ext4_journal_start(inode, EXT4_HT_INODE, 2);
if (IS_ERR(handle)) {
ret = PTR_ERR(handle);
diff --git a/fs/ext4/inline.c b/fs/ext4/inline.c
index 8045e4ff270c..ceee69a66482 100644
--- a/fs/ext4/inline.c
+++ b/fs/ext4/inline.c
@@ -22,8 +22,7 @@
static int ext4_da_convert_inline_data_to_extent(struct address_space *mapping,
- struct inode *inode,
- void **fsdata);
+ struct inode *inode);
static int ext4_get_inline_size(struct inode *inode)
{
@@ -697,7 +696,7 @@ int ext4_generic_write_inline_data(struct address_space *mapping,
struct inode *inode,
loff_t pos, unsigned len,
struct folio **foliop,
- void **fsdata, bool da)
+ bool da)
{
int ret;
handle_t *handle;
@@ -728,7 +727,7 @@ retry_journal:
return ext4_convert_inline_data_to_extent(mapping, inode);
}
- ret = ext4_da_convert_inline_data_to_extent(mapping, inode, fsdata);
+ ret = ext4_da_convert_inline_data_to_extent(mapping, inode);
if (ret == -ENOSPC &&
ext4_should_retry_alloc(inode->i_sb, &retries))
goto retry_journal;
@@ -788,7 +787,7 @@ int ext4_try_to_write_inline_data(struct address_space *mapping,
if (pos + len > ext4_get_max_inline_size(inode))
return ext4_convert_inline_data_to_extent(mapping, inode);
return ext4_generic_write_inline_data(mapping, inode, pos, len,
- foliop, NULL, false);
+ foliop, false);
}
int ext4_write_inline_data_end(struct inode *inode, loff_t pos, unsigned len,
@@ -812,7 +811,19 @@ int ext4_write_inline_data_end(struct inode *inode, loff_t pos, unsigned len,
goto out;
}
ext4_write_lock_xattr(inode, &no_expand);
- BUG_ON(!ext4_has_inline_data(inode));
+ /*
+ * We could have raced with ext4_page_mkwrite() converting
+ * the inode and clearing the inline data flag, so we just
+ * release resources and retry the whole write.
+ */
+ if (unlikely(!ext4_has_inline_data(inode))) {
+ ext4_write_unlock_xattr(inode, &no_expand);
+ brelse(iloc.bh);
+ folio_unlock(folio);
+ folio_put(folio);
+ ext4_journal_stop(handle);
+ return 0;
+ }
/*
* ei->i_inline_off may have changed since
@@ -883,8 +894,7 @@ out:
* need to start the journal since the file's metadata isn't changed now.
*/
static int ext4_da_convert_inline_data_to_extent(struct address_space *mapping,
- struct inode *inode,
- void **fsdata)
+ struct inode *inode)
{
int ret = 0, inline_size;
struct folio *folio;
@@ -922,7 +932,6 @@ static int ext4_da_convert_inline_data_to_extent(struct address_space *mapping,
folio_mark_dirty(folio);
folio_mark_uptodate(folio);
ext4_clear_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA);
- *fsdata = (void *)CONVERT_INLINE_DATA;
out:
up_read(&EXT4_I(inode)->xattr_sem);
@@ -1454,6 +1463,8 @@ int ext4_read_inline_dir(struct file *file,
/* for other entry, the real offset in
* the buf has to be tuned accordingly.
*/
+ if (i + ext4_dir_rec_len(1, NULL) > extra_size)
+ break;
de = (struct ext4_dir_entry_2 *)
(dir_buf + i - extra_offset);
/* It's too expensive to do a full
@@ -1488,10 +1499,17 @@ int ext4_read_inline_dir(struct file *file,
continue;
}
+ /*
+ * de lives at dir_buf + ctx->pos - extra_offset, within the
+ * kmalloc(inline_size) buffer. Make sure its header fits before
+ * ext4_check_dir_entry() dereferences de->rec_len.
+ */
+ if (ctx->pos + ext4_dir_rec_len(1, NULL) > extra_size)
+ goto out;
de = (struct ext4_dir_entry_2 *)
(dir_buf + ctx->pos - extra_offset);
if (ext4_check_dir_entry(inode, file, de, iloc.bh, dir_buf,
- extra_size, ctx->pos))
+ inline_size, ctx->pos))
goto out;
if (le32_to_cpu(de->inode)) {
if (!dir_emit(ctx, de->name, de->name_len,
diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c
index ce99807c5f5b..853efadc944e 100644
--- a/fs/ext4/inode.c
+++ b/fs/ext4/inode.c
@@ -176,7 +176,6 @@ void ext4_evict_inode(struct inode *inode)
* (xattr block freeing), bitmap, group descriptor (inode freeing)
*/
int extra_credits = 6;
- struct ext4_xattr_inode_array *ea_inode_array = NULL;
bool freeze_protected = false;
trace_ext4_evict_inode(inode);
@@ -264,6 +263,7 @@ void ext4_evict_inode(struct inode *inode)
if (ext4_inode_is_fast_symlink(inode))
memset(EXT4_I(inode)->i_data, 0, sizeof(EXT4_I(inode)->i_data));
inode->i_size = 0;
+ ext4_set_inode_state(inode, EXT4_STATE_NO_EXPAND);
err = ext4_mark_inode_dirty(handle, inode);
if (err) {
ext4_warning(inode->i_sb,
@@ -281,8 +281,7 @@ void ext4_evict_inode(struct inode *inode)
}
/* Remove xattr references. */
- err = ext4_xattr_delete_inode(handle, inode, &ea_inode_array,
- extra_credits);
+ err = ext4_xattr_delete_inode(handle, inode, extra_credits);
if (err) {
ext4_warning(inode->i_sb, "xattr delete (err %d)", err);
stop_handle:
@@ -290,7 +289,6 @@ stop_handle:
ext4_orphan_del(NULL, inode);
if (freeze_protected)
sb_end_intwrite(inode->i_sb);
- ext4_xattr_inode_array_free(ea_inode_array);
goto no_delete;
}
@@ -320,7 +318,6 @@ stop_handle:
ext4_journal_stop(handle);
if (freeze_protected)
sb_end_intwrite(inode->i_sb);
- ext4_xattr_inode_array_free(ea_inode_array);
return;
no_delete:
/*
@@ -1182,6 +1179,7 @@ int ext4_block_write_begin(handle_t *handle, struct folio *folio,
int nr_wait = 0;
int i;
bool should_journal_data = ext4_should_journal_data(inode);
+ bool folio_uptodate = folio_test_uptodate(folio);
BUG_ON(!folio_test_locked(folio));
BUG_ON(to > folio_size(folio));
@@ -1193,13 +1191,13 @@ int ext4_block_write_begin(handle_t *handle, struct folio *folio,
head = create_empty_buffers(folio, blocksize, 0);
block = EXT4_PG_TO_LBLK(inode, folio->index);
- for (bh = head, block_start = 0; bh != head || !block_start;
+ for (bh = head, block_start = 0;
+ block_start < to || (!folio_uptodate && bh != head);
block++, block_start = block_end, bh = bh->b_this_page) {
block_end = block_start + blocksize;
if (block_end <= from || block_start >= to) {
- if (folio_test_uptodate(folio)) {
+ if (folio_uptodate)
set_buffer_uptodate(bh);
- }
continue;
}
if (WARN_ON_ONCE(buffer_new(bh)))
@@ -1220,7 +1218,7 @@ int ext4_block_write_begin(handle_t *handle, struct folio *folio,
if (should_journal_data)
do_journal_get_write_access(handle,
inode, bh);
- if (folio_test_uptodate(folio)) {
+ if (folio_uptodate) {
/*
* Unlike __block_write_begin() we leave
* dirtying of new uptodate buffers to
@@ -1237,7 +1235,7 @@ int ext4_block_write_begin(handle_t *handle, struct folio *folio,
continue;
}
}
- if (folio_test_uptodate(folio)) {
+ if (folio_uptodate) {
set_buffer_uptodate(bh);
continue;
}
@@ -1262,17 +1260,6 @@ int ext4_block_write_begin(handle_t *handle, struct folio *folio,
from, to);
else
folio_zero_new_buffers(folio, from, to);
- } else if (fscrypt_inode_uses_fs_layer_crypto(inode)) {
- for (i = 0; i < nr_wait; i++) {
- int err2;
-
- err2 = fscrypt_decrypt_pagecache_blocks(folio,
- blocksize, bh_offset(wait[i]));
- if (err2) {
- clear_buffer_uptodate(wait[i]);
- err = err2;
- }
- }
}
return err;
@@ -1302,6 +1289,8 @@ static int ext4_write_begin(const struct kiocb *iocb,
if (unlikely(ret))
return ret;
+ *fsdata = (void *)((unsigned long)*fsdata & ~EXT4_WRITE_DATA_INLINE);
+
trace_ext4_write_begin(inode, pos, len);
/*
* Reserve one block more for addition to orphan list in case
@@ -1316,8 +1305,10 @@ static int ext4_write_begin(const struct kiocb *iocb,
foliop);
if (ret < 0)
return ret;
- if (ret == 1)
+ if (ret == 1) {
+ *fsdata = (void *)((unsigned long)*fsdata | EXT4_WRITE_DATA_INLINE);
return 0;
+ }
}
/*
@@ -1450,8 +1441,7 @@ static int ext4_write_end(const struct kiocb *iocb,
trace_ext4_write_end(inode, pos, len, copied);
- if (ext4_has_inline_data(inode) &&
- ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA))
+ if ((unsigned long)fsdata & EXT4_WRITE_DATA_INLINE)
return ext4_write_inline_data_end(inode, pos, len, copied,
folio);
@@ -1560,8 +1550,7 @@ static int ext4_journalled_write_end(const struct kiocb *iocb,
BUG_ON(!ext4_handle_valid(handle));
- if (ext4_has_inline_data(inode) &&
- ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA))
+ if ((unsigned long)fsdata & EXT4_WRITE_DATA_INLINE)
return ext4_write_inline_data_end(inode, pos, len, copied,
folio);
@@ -2075,11 +2064,10 @@ static void mpage_folio_done(struct mpage_da_data *mpd, struct folio *folio)
folio_unlock(folio);
}
-static int mpage_submit_folio(struct mpage_da_data *mpd, struct folio *folio)
+static void mpage_submit_folio(struct mpage_da_data *mpd, struct folio *folio)
{
size_t len;
loff_t size;
- int err;
WARN_ON_ONCE(folio_pos(folio) != mpd->start_pos);
folio_clear_dirty_for_io(folio);
@@ -2101,9 +2089,7 @@ static int mpage_submit_folio(struct mpage_da_data *mpd, struct folio *folio)
if (folio_pos(folio) + len > size &&
!ext4_verity_in_progress(mpd->inode))
len = size & (len - 1);
- err = ext4_bio_write_folio(&mpd->io_submit, folio, len);
-
- return err;
+ ext4_bio_write_folio(&mpd->io_submit, folio, len);
}
#define BH_FLAGS (BIT(BH_Unwritten) | BIT(BH_Delay))
@@ -2180,8 +2166,7 @@ static bool mpage_add_bh_to_extent(struct mpage_da_data *mpd, ext4_lblk_t lblk,
* accumulated extent of buffers to map or add buffers in the page to the
* extent of buffers to map. The function returns 1 if the caller can continue
* by processing the next page, 0 if it should stop adding buffers to the
- * extent to map because we cannot extend it anymore. It can also return value
- * < 0 in case of error during IO submission.
+ * extent to map because we cannot extend it anymore.
*/
static int mpage_process_page_bufs(struct mpage_da_data *mpd,
struct buffer_head *head,
@@ -2189,7 +2174,6 @@ static int mpage_process_page_bufs(struct mpage_da_data *mpd,
ext4_lblk_t lblk)
{
struct inode *inode = mpd->inode;
- int err;
ext4_lblk_t blocks = (i_size_read(inode) + i_blocksize(inode) - 1)
>> inode->i_blkbits;
@@ -2212,9 +2196,7 @@ static int mpage_process_page_bufs(struct mpage_da_data *mpd,
} while (lblk++, (bh = bh->b_this_page) != head);
/* So far everything mapped? Submit the page for IO. */
if (mpd->map.m_len == 0) {
- err = mpage_submit_folio(mpd, head->b_folio);
- if (err < 0)
- return err;
+ mpage_submit_folio(mpd, head->b_folio);
mpage_folio_done(mpd, head->b_folio);
}
if (lblk >= blocks) {
@@ -2344,9 +2326,7 @@ static int mpage_map_and_submit_buffers(struct mpage_da_data *mpd)
if (err < 0 || map_bh)
goto out;
/* Page fully mapped - let IO run! */
- err = mpage_submit_folio(mpd, folio);
- if (err < 0)
- goto out;
+ mpage_submit_folio(mpd, folio);
mpage_folio_done(mpd, folio);
}
folio_batch_release(&fbatch);
@@ -2419,7 +2399,6 @@ static int mpage_submit_partial_folio(struct mpage_da_data *mpd)
struct inode *inode = mpd->inode;
struct folio *folio;
loff_t pos;
- int ret;
folio = filemap_get_folio(inode->i_mapping,
mpd->start_pos >> PAGE_SHIFT);
@@ -2434,9 +2413,7 @@ static int mpage_submit_partial_folio(struct mpage_da_data *mpd)
!folio_contains(folio, pos >> PAGE_SHIFT)))
return -EINVAL;
- ret = mpage_submit_folio(mpd, folio);
- if (ret)
- goto out;
+ mpage_submit_folio(mpd, folio);
/*
* Update start_pos to prevent this folio from being released in
* mpage_release_unused_pages(), it will be reset to the aligned folio
@@ -2445,10 +2422,9 @@ static int mpage_submit_partial_folio(struct mpage_da_data *mpd)
* entire folio has finished processing.
*/
mpd->start_pos = pos;
-out:
folio_unlock(folio);
folio_put(folio);
- return ret;
+ return 0;
}
/*
@@ -2694,13 +2670,25 @@ static int mpage_prepare_extent_to_map(struct mpage_da_data *mpd)
* page is already under writeback and we are not doing
* a data integrity writeback, skip the page
*/
- if (!folio_test_dirty(folio) ||
- (folio_test_writeback(folio) &&
- (mpd->wbc->sync_mode == WB_SYNC_NONE)) ||
+ if ((folio_test_writeback(folio) &&
+ mpd->wbc->sync_mode == WB_SYNC_NONE) ||
unlikely(folio->mapping != mapping)) {
folio_unlock(folio);
continue;
}
+ /*
+ * If the folio is clean, skip writing it back.
+ * Cycle the folio through the writeback state
+ * though, to clear stale xarray tags.
+ */
+ if (!folio_test_dirty(folio)) {
+ if (!folio_test_writeback(folio)) {
+ __folio_start_writeback(folio, false);
+ folio_end_writeback(folio);
+ }
+ folio_unlock(folio);
+ continue;
+ }
folio_wait_writeback(folio);
BUG_ON(folio_test_writeback(folio));
@@ -2735,9 +2723,8 @@ static int mpage_prepare_extent_to_map(struct mpage_da_data *mpd)
* through a pin.
*/
if (!mpd->can_map) {
- err = mpage_submit_folio(mpd, folio);
- if (err < 0)
- goto out;
+ mpage_submit_folio(mpd, folio);
+ err = 0;
/* Pending dirtying of journalled data? */
if (folio_test_checked(folio)) {
err = mpage_journal_page_buffers(handle,
@@ -3158,11 +3145,13 @@ static int ext4_da_write_begin(const struct kiocb *iocb,
if (ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA)) {
ret = ext4_generic_write_inline_data(mapping, inode, pos, len,
- foliop, fsdata, true);
+ foliop, true);
if (ret < 0)
return ret;
- if (ret == 1)
+ if (ret == 1) {
+ *fsdata = (void *)((unsigned long)*fsdata | EXT4_WRITE_DATA_INLINE);
return 0;
+ }
}
retry:
@@ -3291,17 +3280,15 @@ static int ext4_da_write_end(const struct kiocb *iocb,
struct folio *folio, void *fsdata)
{
struct inode *inode = mapping->host;
- int write_mode = (int)(unsigned long)fsdata;
+ unsigned long write_mode = (unsigned long)fsdata;
- if (write_mode == FALL_BACK_TO_NONDELALLOC)
+ if (write_mode & FALL_BACK_TO_NONDELALLOC)
return ext4_write_end(iocb, mapping, pos,
len, copied, folio, fsdata);
trace_ext4_da_write_end(inode, pos, len, copied);
- if (write_mode != CONVERT_INLINE_DATA &&
- ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA) &&
- ext4_has_inline_data(inode))
+ if (write_mode & EXT4_WRITE_DATA_INLINE)
return ext4_write_inline_data_end(inode, pos, len, copied,
folio);
@@ -3673,6 +3660,9 @@ static int ext4_iomap_alloc(struct inode *inode, struct ext4_map_blocks *map,
int ret, dio_credits, m_flags = 0, retries = 0;
bool force_commit = false;
+ if (flags & IOMAP_NOWAIT)
+ return -EAGAIN;
+
/*
* Trim the mapping request to the maximum value that we can map at
* once for direct I/O.
@@ -3829,9 +3819,9 @@ static int ext4_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
return ret;
out:
/*
- * When inline encryption is enabled, sometimes I/O to an encrypted file
- * has to be broken up to guarantee DUN contiguity. Handle this by
- * limiting the length of the mapping returned.
+ * Sometimes I/O to an encrypted file has to be broken up to guarantee
+ * DUN contiguity. Handle this by limiting the length of the mapping
+ * returned.
*/
map.m_len = fscrypt_limit_io_blocks(inode, map.m_lblk, map.m_len);
@@ -4027,6 +4017,10 @@ void ext4_set_aops(struct inode *inode)
* because it might have data in pagecache (eg, if called from ext4_zero_range,
* ext4_punch_hole, etc) which needs to be properly zeroed out. Otherwise a
* racing writeback can come later and flush the stale pagecache to disk.
+ *
+ * Return the loaded bh if it actually needs zeroing - in written, dirty
+ * unwritten, or delalloc state. Return NULL if it's clean (i.e., a hole or
+ * a clean unwritten block).
*/
static struct buffer_head *ext4_load_tail_bh(struct inode *inode, loff_t from)
{
@@ -4038,7 +4032,7 @@ static struct buffer_head *ext4_load_tail_bh(struct inode *inode, loff_t from)
int err = 0;
folio = __filemap_get_folio(mapping, from >> PAGE_SHIFT,
- FGP_LOCK | FGP_ACCESSED | FGP_CREAT,
+ FGP_WRITEBEGIN | FGP_ACCESSED,
mapping_gfp_constraint(mapping, ~__GFP_FS));
if (IS_ERR(folio))
return ERR_CAST(folio);
@@ -4065,8 +4059,15 @@ static struct buffer_head *ext4_load_tail_bh(struct inode *inode, loff_t from)
}
if (!buffer_mapped(bh)) {
BUFFER_TRACE(bh, "unmapped");
- ext4_get_block(inode, iblock, bh, 0);
- /* unmapped? It's a hole - nothing to do */
+ err = ext4_get_block(inode, iblock, bh, 0);
+ if (err < 0)
+ goto unlock;
+ /*
+ * It's a hole or a clean unwritten block - nothing to do.
+ * Note that a lookup-only get_block (without
+ * EXT4_GET_BLOCKS_CREATE) never sets BH_Mapped for clean
+ * unwritten extents.
+ */
if (!buffer_mapped(bh)) {
BUFFER_TRACE(bh, "still unmapped");
goto unlock;
@@ -4081,17 +4082,6 @@ static struct buffer_head *ext4_load_tail_bh(struct inode *inode, loff_t from)
err = ext4_read_bh_lock(bh, 0, true);
if (err)
goto unlock;
- if (fscrypt_inode_uses_fs_layer_crypto(inode)) {
- /* We expect the key to be set. */
- BUG_ON(!fscrypt_has_encryption_key(inode));
- err = fscrypt_decrypt_pagecache_blocks(folio,
- blocksize,
- bh_offset(bh));
- if (err) {
- clear_buffer_uptodate(bh);
- goto unlock;
- }
- }
}
return bh;
@@ -4219,6 +4209,14 @@ int ext4_block_zero_eof(struct inode *inode, loff_t from, loff_t end)
offset = from & (blocksize - 1);
if (!offset || from >= end)
return 0;
+ /*
+ * Inline data has no tail block to zero out. Note that a race with
+ * ext4_page_mkwrite() converting inline data to an extent without
+ * holding i_rwsem is safe, as that path zeroes the full block before
+ * copying in the inline data.
+ */
+ if (ext4_has_inline_data(inode))
+ return 0;
/* If we are processing an encrypted inode during orphan list handling */
if (IS_ENCRYPTED(inode) && !fscrypt_has_encryption_key(inode))
return 0;
@@ -4253,13 +4251,26 @@ int ext4_block_zero_eof(struct inode *inode, loff_t from, loff_t end)
return 0;
}
+/*
+ * Zero out the unaligned head and tail of the [lstart, lstart+length)
+ * range.
+ *
+ * On return, @partial_zeroed records which edges actually got
+ * partial-zeroed. Set EXT4_PARTIAL_ZERO_START/EXT4_PARTIAL_ZERO_END if
+ * the head/tail block got actually partially zeroed (in written, dirty
+ * unwritten or delalloc state). Cleared if the head/tail block is a
+ * hole or a clean unwritten block, in which case there is nothing that
+ * needs zeroing. When the head and tail land in the same block, both
+ * bits are set together on a successful zeroing.
+ */
int ext4_zero_partial_blocks(struct inode *inode, loff_t lstart, loff_t length,
- bool *did_zero)
+ unsigned int *partial_zeroed)
{
struct super_block *sb = inode->i_sb;
unsigned partial_start, partial_end;
ext4_fsblk_t start, end;
loff_t byte_end = (lstart + length - 1);
+ bool did_zero = false;
int err = 0;
partial_start = lstart & (sb->s_blocksize - 1);
@@ -4271,21 +4282,32 @@ int ext4_zero_partial_blocks(struct inode *inode, loff_t lstart, loff_t length,
/* Handle partial zero within the single block */
if (start == end &&
(partial_start || (partial_end != sb->s_blocksize - 1))) {
- err = ext4_block_zero_range(inode, lstart, length, did_zero,
+ err = ext4_block_zero_range(inode, lstart, length, &did_zero,
NULL);
+ if (did_zero)
+ *partial_zeroed |= (EXT4_PARTIAL_ZERO_START |
+ EXT4_PARTIAL_ZERO_END);
return err;
}
/* Handle partial zero out on the start of the range */
if (partial_start) {
err = ext4_block_zero_range(inode, lstart, sb->s_blocksize,
- did_zero, NULL);
+ &did_zero, NULL);
if (err)
return err;
+ if (did_zero)
+ *partial_zeroed |= EXT4_PARTIAL_ZERO_START;
}
/* Handle partial zero out on the end of the range */
- if (partial_end != sb->s_blocksize - 1)
+ if (partial_end != sb->s_blocksize - 1) {
+ did_zero = false;
err = ext4_block_zero_range(inode, byte_end - partial_end,
- partial_end + 1, did_zero, NULL);
+ partial_end + 1, &did_zero, NULL);
+ if (err)
+ return err;
+ if (did_zero)
+ *partial_zeroed |= EXT4_PARTIAL_ZERO_END;
+ }
return err;
}
@@ -4434,7 +4456,7 @@ int ext4_punch_hole(struct file *file, loff_t offset, loff_t length)
loff_t end = offset + length;
handle_t *handle;
unsigned int credits;
- bool partial_zeroed = false;
+ unsigned int partial_zeroed = 0;
int ret;
trace_ext4_punch_hole(inode, offset, length, 0);
@@ -5270,6 +5292,20 @@ void ext4_set_inode_mapping_order(struct inode *inode)
mapping_set_folio_order_range(inode->i_mapping, min_order, max_order);
}
+static int ext4_iget_match(struct inode *inode, u64 ino, void *data)
+{
+ if (inode->i_ino != ino)
+ return 0;
+ spin_lock(&inode->i_lock);
+ if (inode_state_read(inode) & (I_FREEING | I_WILL_FREE | I_CREATING)) {
+ spin_unlock(&inode->i_lock);
+ return -1;
+ }
+ __iget(inode);
+ spin_unlock(&inode->i_lock);
+ return 1;
+}
+
struct inode *__ext4_iget(struct super_block *sb, unsigned long ino,
ext4_iget_flags flags, const char *function,
unsigned int line)
@@ -5298,9 +5334,24 @@ struct inode *__ext4_iget(struct super_block *sb, unsigned long ino,
return ERR_PTR(-EFSCORRUPTED);
}
- inode = iget_locked(sb, ino);
- if (!inode)
- return ERR_PTR(-ENOMEM);
+ if (flags & EXT4_IGET_NOWAIT) {
+ inode = find_inode_nowait(sb, ino, ext4_iget_match, NULL);
+ if (!inode)
+ return ERR_PTR(-ENOENT);
+
+ if (inode_state_read_once(inode) & I_NEW)
+ wait_on_new_inode(inode);
+
+ if (unlikely(inode_unhashed(inode))) {
+ iput(inode);
+ return ERR_PTR(-ENOENT);
+ }
+ } else {
+ inode = iget_locked(sb, ino);
+ if (!inode)
+ return ERR_PTR(-ENOMEM);
+ }
+
if (!(inode_state_read_once(inode) & I_NEW)) {
ret = check_igot_inode(inode, flags, function, line);
if (ret) {
@@ -6183,11 +6234,8 @@ u32 ext4_dio_alignment(struct inode *inode)
return 0;
if (ext4_has_inline_data(inode))
return 0;
- if (IS_ENCRYPTED(inode)) {
- if (!fscrypt_dio_supported(inode))
- return 0;
+ if (IS_ENCRYPTED(inode))
return i_blocksize(inode);
- }
return 1; /* use the iomap defaults */
}
@@ -6206,11 +6254,7 @@ int ext4_getattr(struct mnt_idmap *idmap, const struct path *path,
stat->btime.tv_nsec = ei->i_crtime.tv_nsec;
}
- /*
- * Return the DIO alignment restrictions if requested. We only return
- * this information when requested, since on encrypted files it might
- * take a fair bit of work to get if the file wasn't opened recently.
- */
+ /* Return the DIO alignment restrictions if requested. */
if ((request_mask & STATX_DIOALIGN) && S_ISREG(inode->i_mode)) {
u32 dio_align = ext4_dio_alignment(inode);
@@ -6511,6 +6555,16 @@ static int ext4_try_to_expand_extra_isize(struct inode *inode,
return -EOVERFLOW;
/*
+ * Skip expansion during mount (!SB_ACTIVE). Expanding extra isize
+ * may move xattrs to external blocks and release ea_inodes via iput.
+ * When !SB_ACTIVE, iput triggers write_inode_now() which acquires
+ * s_writepages_rwsem, causing a deadlock with the caller's active
+ * jbd2 handle (lock order: s_writepages_rwsem -> jbd2_handle).
+ */
+ if (unlikely(!(inode->i_sb->s_flags & SB_ACTIVE)))
+ return -EBUSY;
+
+ /*
* In nojournal mode, we can immediately attempt to expand
* the inode. When journaled, we first need to obtain extra
* buffer credits since we may write into the EA block
diff --git a/fs/ext4/mballoc-test.c b/fs/ext4/mballoc-test.c
index 0424b8b0b4c3..d31780075c21 100644
--- a/fs/ext4/mballoc-test.c
+++ b/fs/ext4/mballoc-test.c
@@ -59,11 +59,6 @@ static const struct super_operations mbt_sops = {
.free_inode = mbt_free_inode,
};
-static void mbt_kill_sb(struct super_block *sb)
-{
- generic_shutdown_super(sb);
-}
-
static int mbt_init_fs_context(struct fs_context *fc)
{
return 0;
@@ -72,7 +67,7 @@ static int mbt_init_fs_context(struct fs_context *fc)
static struct file_system_type mbt_fs_type = {
.name = "mballoc test",
.init_fs_context = mbt_init_fs_context,
- .kill_sb = mbt_kill_sb,
+ .kill_sb = kill_anon_super,
};
static int mbt_mb_init(struct super_block *sb)
@@ -136,7 +131,7 @@ static void mbt_mb_release(struct super_block *sb)
static int mbt_set(struct super_block *sb, struct fs_context *fc)
{
- return 0;
+ return set_anon_super_fc(sb, fc);
}
static struct super_block *mbt_ext4_alloc_super_block(void)
diff --git a/fs/ext4/mballoc.c b/fs/ext4/mballoc.c
index ed1bd00e11cd..06171a11db12 100644
--- a/fs/ext4/mballoc.c
+++ b/fs/ext4/mballoc.c
@@ -2861,8 +2861,6 @@ ext4_group_t ext4_mb_prefetch(struct super_block *sb, ext4_group_t group,
blk_start_plug(&plug);
while (nr-- > 0) {
- struct ext4_group_desc *gdp = ext4_get_group_desc(sb, group,
- NULL);
struct ext4_group_info *grp = ext4_get_group_info(sb, group);
/*
@@ -2872,14 +2870,17 @@ ext4_group_t ext4_mb_prefetch(struct super_block *sb, ext4_group_t group,
* prefetch once, so we avoid getblk() call, which can
* be expensive.
*/
- if (gdp && grp && !EXT4_MB_GRP_TEST_AND_SET_READ(grp) &&
- EXT4_MB_GRP_NEED_INIT(grp) &&
- ext4_free_group_clusters(sb, gdp) > 0 ) {
- bh = ext4_read_block_bitmap_nowait(sb, group, true);
- if (!IS_ERR_OR_NULL(bh)) {
- if (!buffer_uptodate(bh) && cnt)
- (*cnt)++;
- brelse(bh);
+ if (grp && !EXT4_MB_GRP_TEST_AND_SET_READ(grp) &&
+ EXT4_MB_GRP_NEED_INIT(grp)) {
+ struct ext4_group_desc *gdp = ext4_get_group_desc(sb, group, NULL);
+
+ if (gdp && ext4_free_group_clusters(sb, gdp) > 0) {
+ bh = ext4_read_block_bitmap_nowait(sb, group, true);
+ if (!IS_ERR_OR_NULL(bh)) {
+ if (!buffer_uptodate(bh) && cnt)
+ (*cnt)++;
+ brelse(bh);
+ }
}
}
if (++group >= ngroups)
diff --git a/fs/ext4/migrate.c b/fs/ext4/migrate.c
index 477d43d7e294..5d60ef10fe11 100644
--- a/fs/ext4/migrate.c
+++ b/fs/ext4/migrate.c
@@ -464,6 +464,7 @@ int ext4_ext_migrate(struct inode *inode)
if (IS_ERR(tmp_inode)) {
retval = PTR_ERR(tmp_inode);
ext4_journal_stop(handle);
+ tmp_inode = NULL;
goto out_unlock;
}
/*
@@ -591,9 +592,9 @@ out_stop:
ext4_journal_stop(handle);
out_tmp_inode:
unlock_new_inode(tmp_inode);
- iput(tmp_inode);
out_unlock:
ext4_writepages_up_write(inode->i_sb, alloc_ctx);
+ iput(tmp_inode);
return retval;
}
diff --git a/fs/ext4/namei.c b/fs/ext4/namei.c
index cc49ae04a6f6..a6386c1d237f 100644
--- a/fs/ext4/namei.c
+++ b/fs/ext4/namei.c
@@ -1467,6 +1467,8 @@ int ext4_search_dir(struct buffer_head *bh, char *search_buf, int buf_size,
/* this code is executed quadratically often */
/* do minimal checking `by hand' */
if (de->name + de->name_len <= dlimit &&
+ (!ext4_hash_in_dirent(dir) ||
+ (char *)de + ext4_dir_rec_len(de->name_len, dir) <= dlimit) &&
ext4_match(dir, fname, de)) {
/* found a match - just to be sure, do
* a full check */
@@ -2811,7 +2813,7 @@ static int ext4_add_nondir(handle_t *handle,
* with d_instantiate().
*/
static int ext4_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
handle_t *handle;
struct inode *inode;
@@ -3009,7 +3011,7 @@ static struct dentry *ext4_mkdir(struct mnt_idmap *idmap, struct inode *dir,
credits = (EXT4_DATA_TRANS_BLOCKS(dir->i_sb) +
EXT4_INDEX_EXTRA_TRANS_BLOCKS + 3);
retry:
- inode = ext4_new_inode_start_handle(idmap, dir, S_IFDIR | mode,
+ inode = ext4_new_inode_start_handle(idmap, dir, mode,
&dentry->d_name,
0, NULL, EXT4_HT_DIR, credits);
handle = ext4_journal_current_handle();
diff --git a/fs/ext4/orphan.c b/fs/ext4/orphan.c
index 64ea47624233..646c3077b6d2 100644
--- a/fs/ext4/orphan.c
+++ b/fs/ext4/orphan.c
@@ -4,6 +4,7 @@
#include <linux/fs.h>
#include <linux/quotaops.h>
#include <linux/buffer_head.h>
+#include <linux/string_choices.h>
#include "ext4.h"
#include "ext4_jbd2.h"
@@ -486,14 +487,12 @@ void ext4_orphan_cleanup(struct super_block *sb, struct ext4_super_block *es)
}
}
-#define PLURAL(x) (x), ((x) == 1) ? "" : "s"
-
if (nr_orphans)
ext4_msg(sb, KERN_INFO, "%d orphan inode%s deleted",
- PLURAL(nr_orphans));
+ nr_orphans, str_plural(nr_orphans));
if (nr_truncates)
ext4_msg(sb, KERN_INFO, "%d truncate%s cleaned up",
- PLURAL(nr_truncates));
+ nr_truncates, str_plural(nr_truncates));
#ifdef CONFIG_QUOTA
/* Turn off quotas if they were enabled for orphan cleanup */
if (quota_update) {
@@ -572,6 +571,7 @@ int ext4_init_orphan_info(struct super_block *sb)
int i, j;
int ret;
int free;
+ int loaded = 0;
__le32 *bdata;
int inodes_per_ob = ext4_inodes_per_orphan_block(sb);
struct ext4_orphan_block_tail *ot;
@@ -613,6 +613,7 @@ int ext4_init_orphan_info(struct super_block *sb)
ret = -EIO;
goto out_free;
}
+ loaded++;
ot = ext4_orphan_block_tail(sb, oi->of_binfo[i].ob_bh);
if (le32_to_cpu(ot->ob_magic) != EXT4_ORPHAN_BLOCK_MAGIC) {
ext4_error(sb, "orphan file block %d: bad magic", i);
@@ -635,8 +636,10 @@ int ext4_init_orphan_info(struct super_block *sb)
iput(inode);
return 0;
out_free:
- for (i--; i >= 0; i--)
- brelse(oi->of_binfo[i].ob_bh);
+ while (loaded > 0) {
+ loaded--;
+ brelse(oi->of_binfo[loaded].ob_bh);
+ }
kvfree(oi->of_binfo);
out_put:
iput(inode);
diff --git a/fs/ext4/page-io.c b/fs/ext4/page-io.c
index bc674aa4a656..0236b6b9785a 100644
--- a/fs/ext4/page-io.c
+++ b/fs/ext4/page-io.c
@@ -103,18 +103,12 @@ static void ext4_finish_bio(struct bio *bio)
bio_for_each_folio_all(fi, bio) {
struct folio *folio = fi.folio;
- struct folio *io_folio = NULL;
struct buffer_head *bh, *head;
size_t bio_start = fi.offset;
size_t bio_end = bio_start + fi.length;
unsigned under_io = 0;
unsigned long flags;
- if (fscrypt_is_bounce_folio(folio)) {
- io_folio = folio;
- folio = fscrypt_pagecache_folio(folio);
- }
-
if (bio->bi_status) {
int err = blk_status_to_errno(bio->bi_status);
mapping_set_error(folio->mapping, err);
@@ -139,10 +133,8 @@ static void ext4_finish_bio(struct bio *bio)
}
} while ((bh = bh->b_this_page) != head);
spin_unlock_irqrestore(&head->b_uptodate_lock, flags);
- if (!under_io) {
- fscrypt_free_bounce_page(&io_folio->page);
+ if (!under_io)
folio_end_writeback(folio);
- }
}
}
@@ -453,7 +445,6 @@ static bool io_submit_need_new_bio(struct ext4_io_submit *io,
static void io_submit_add_bh(struct ext4_io_submit *io,
struct inode *inode,
struct folio *folio,
- struct folio *io_folio,
struct buffer_head *bh)
{
if (io->io_bio && io_submit_need_new_bio(io, inode, folio, bh)) {
@@ -462,20 +453,18 @@ submit_and_retry:
}
if (io->io_bio == NULL)
io_submit_init_bio(io, inode, folio, bh);
- if (!bio_add_folio(io->io_bio, io_folio, bh->b_size, bh_offset(bh)))
+ if (!bio_add_folio(io->io_bio, folio, bh->b_size, bh_offset(bh)))
goto submit_and_retry;
wbc_account_cgroup_owner(io->io_wbc, folio, bh->b_size);
io->io_next_block++;
}
-int ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *folio,
+void ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *folio,
size_t len)
{
- struct folio *io_folio = folio;
struct inode *inode = folio->mapping->host;
unsigned block_start;
struct buffer_head *bh, *head;
- int ret = 0;
int nr_to_submit = 0;
struct writeback_control *wbc = io->io_wbc;
bool keep_towrite = false;
@@ -544,70 +533,17 @@ int ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *folio,
*/
__folio_start_writeback(folio, keep_towrite);
folio_end_writeback(folio);
- return 0;
+ return;
}
bh = head = folio_buffers(folio);
- /*
- * If any blocks are being written to an encrypted file, encrypt them
- * into a bounce page. For simplicity, just encrypt until the last
- * block which might be needed. This may cause some unneeded blocks
- * (e.g. holes) to be unnecessarily encrypted, but this is rare and
- * can't happen in the common case of blocksize == PAGE_SIZE.
- */
- if (fscrypt_inode_uses_fs_layer_crypto(inode)) {
- gfp_t gfp_flags = GFP_NOFS;
- unsigned int enc_bytes = round_up(len, i_blocksize(inode));
- struct page *bounce_page;
-
- /*
- * Since bounce page allocation uses a mempool, we can only use
- * a waiting mask (i.e. request guaranteed allocation) on the
- * first page of the bio. Otherwise it can deadlock.
- */
- if (io->io_bio)
- gfp_flags = GFP_NOWAIT;
- retry_encrypt:
- bounce_page = fscrypt_encrypt_pagecache_blocks(folio,
- enc_bytes, 0, gfp_flags);
- if (IS_ERR(bounce_page)) {
- ret = PTR_ERR(bounce_page);
- if (ret == -ENOMEM &&
- (io->io_bio || wbc->sync_mode == WB_SYNC_ALL)) {
- gfp_t new_gfp_flags = GFP_NOFS;
- if (io->io_bio)
- ext4_io_submit(io);
- else
- new_gfp_flags |= __GFP_NOFAIL;
- memalloc_retry_wait(gfp_flags);
- gfp_flags = new_gfp_flags;
- goto retry_encrypt;
- }
-
- printk_ratelimited(KERN_ERR "%s: ret = %d\n", __func__, ret);
- folio_redirty_for_writepage(wbc, folio);
- do {
- if (buffer_async_write(bh)) {
- clear_buffer_async_write(bh);
- set_buffer_dirty(bh);
- }
- bh = bh->b_this_page;
- } while (bh != head);
-
- return ret;
- }
- io_folio = page_folio(bounce_page);
- }
-
__folio_start_writeback(folio, keep_towrite);
/* Now submit buffers to write */
do {
if (!buffer_async_write(bh))
continue;
- io_submit_add_bh(io, inode, folio, io_folio, bh);
+ io_submit_add_bh(io, inode, folio, bh);
} while ((bh = bh->b_this_page) != head);
-
- return 0;
}
diff --git a/fs/ext4/readpage.c b/fs/ext4/readpage.c
index dd3627c71732..c7b6cdb2e124 100644
--- a/fs/ext4/readpage.c
+++ b/fs/ext4/readpage.c
@@ -47,25 +47,15 @@
#include "ext4.h"
#include <trace/events/ext4.h>
-#define NUM_PREALLOC_POST_READ_CTXS 128
+#define NUM_VERITY_WORKS 128
-static struct kmem_cache *bio_post_read_ctx_cache;
-static mempool_t *bio_post_read_ctx_pool;
+static struct kmem_cache *ext4_verity_work_cache;
+static mempool_t *ext4_verity_work_pool;
-/* postprocessing steps for read bios */
-enum bio_post_read_step {
- STEP_INITIAL = 0,
- STEP_DECRYPT,
- STEP_VERITY,
- STEP_MAX,
-};
-
-struct bio_post_read_ctx {
+struct ext4_verity_work {
struct bio *bio;
struct fsverity_info *vi;
struct work_struct work;
- unsigned int cur_step;
- unsigned int enabled_steps;
};
static void __read_end_io(struct bio *bio)
@@ -75,40 +65,22 @@ static void __read_end_io(struct bio *bio)
bio_for_each_folio_all(fi, bio)
folio_end_read(fi.folio, bio->bi_status == 0);
if (bio->bi_private)
- mempool_free(bio->bi_private, bio_post_read_ctx_pool);
+ mempool_free(bio->bi_private, ext4_verity_work_pool);
bio_put(bio);
}
-static void bio_post_read_processing(struct bio_post_read_ctx *ctx);
-
-static void decrypt_work(struct work_struct *work)
-{
- struct bio_post_read_ctx *ctx =
- container_of(work, struct bio_post_read_ctx, work);
- struct bio *bio = ctx->bio;
-
- if (fscrypt_decrypt_bio(bio))
- bio_post_read_processing(ctx);
- else
- __read_end_io(bio);
-}
-
static void verity_work(struct work_struct *work)
{
- struct bio_post_read_ctx *ctx =
- container_of(work, struct bio_post_read_ctx, work);
+ struct ext4_verity_work *ctx =
+ container_of(work, struct ext4_verity_work, work);
struct bio *bio = ctx->bio;
struct fsverity_info *vi = ctx->vi;
/*
- * fsverity_verify_bio() may call readahead() again, and although verity
- * will be disabled for that, decryption may still be needed, causing
- * another bio_post_read_ctx to be allocated. So to guarantee that
- * mempool_alloc() never deadlocks we must free the current ctx first.
- * This is safe because verity is the last post-read step.
+ * Free the ext4_verity_work right away, since it's no longer needed.
+ * This relieves the pressure on the mempool as much as possible.
*/
- BUILD_BUG_ON(STEP_VERITY + 1 != STEP_MAX);
- mempool_free(ctx, bio_post_read_ctx_pool);
+ mempool_free(ctx, ext4_verity_work_pool);
bio->bi_private = NULL;
fsverity_verify_bio(vi, bio);
@@ -116,41 +88,6 @@ static void verity_work(struct work_struct *work)
__read_end_io(bio);
}
-static void bio_post_read_processing(struct bio_post_read_ctx *ctx)
-{
- /*
- * We use different work queues for decryption and for verity because
- * verity may require reading metadata pages that need decryption, and
- * we shouldn't recurse to the same workqueue.
- */
- switch (++ctx->cur_step) {
- case STEP_DECRYPT:
- if (ctx->enabled_steps & (1 << STEP_DECRYPT)) {
- INIT_WORK(&ctx->work, decrypt_work);
- fscrypt_enqueue_decrypt_work(&ctx->work);
- return;
- }
- ctx->cur_step++;
- fallthrough;
- case STEP_VERITY:
- if (IS_ENABLED(CONFIG_FS_VERITY) &&
- ctx->enabled_steps & (1 << STEP_VERITY)) {
- INIT_WORK(&ctx->work, verity_work);
- fsverity_enqueue_verify_work(&ctx->work);
- return;
- }
- ctx->cur_step++;
- fallthrough;
- default:
- __read_end_io(ctx->bio);
- }
-}
-
-static bool bio_post_read_required(struct bio *bio)
-{
- return bio->bi_private && !bio->bi_status;
-}
-
/*
* I/O completion handler for multipage BIOs.
*
@@ -165,36 +102,26 @@ static bool bio_post_read_required(struct bio *bio)
*/
static void mpage_end_io(struct bio *bio)
{
- if (bio_post_read_required(bio)) {
- struct bio_post_read_ctx *ctx = bio->bi_private;
+ if (IS_ENABLED(CONFIG_FS_VERITY) && bio->bi_private &&
+ !bio->bi_status) {
+ struct ext4_verity_work *ctx = bio->bi_private;
- ctx->cur_step = STEP_INITIAL;
- bio_post_read_processing(ctx);
+ INIT_WORK(&ctx->work, verity_work);
+ fsverity_enqueue_verify_work(&ctx->work);
return;
}
__read_end_io(bio);
}
-static void ext4_set_bio_post_read_ctx(struct bio *bio,
- const struct inode *inode,
- struct fsverity_info *vi)
+static void ext4_set_verity_work(struct bio *bio, struct fsverity_info *vi)
{
- unsigned int post_read_steps = 0;
-
- if (fscrypt_inode_uses_fs_layer_crypto(inode))
- post_read_steps |= 1 << STEP_DECRYPT;
-
- if (vi)
- post_read_steps |= 1 << STEP_VERITY;
-
- if (post_read_steps) {
+ if (vi) {
/* Due to the mempool, this never fails. */
- struct bio_post_read_ctx *ctx =
- mempool_alloc(bio_post_read_ctx_pool, GFP_NOFS);
+ struct ext4_verity_work *ctx =
+ mempool_alloc(ext4_verity_work_pool, GFP_NOFS);
ctx->bio = bio;
ctx->vi = vi;
- ctx->enabled_steps = post_read_steps;
bio->bi_private = ctx;
}
}
@@ -355,7 +282,7 @@ static int ext4_mpage_readpages(struct inode *inode, struct fsverity_info *vi,
bio = bio_alloc(bdev, bio_max_segs(nr_pages),
REQ_OP_READ, GFP_KERNEL);
fscrypt_set_bio_crypt_ctx(bio, inode, pos, GFP_KERNEL);
- ext4_set_bio_post_read_ctx(bio, inode, vi);
+ ext4_set_verity_work(bio, vi);
bio->bi_iter.bi_sector = first_block << (blkbits - 9);
bio->bi_end_io = mpage_end_io;
if (rac)
@@ -429,27 +356,31 @@ void ext4_readahead(struct readahead_control *rac)
ext4_mpage_readpages(inode, vi, rac, NULL);
}
-int __init ext4_init_post_read_processing(void)
+int __init ext4_init_verity_caches(void)
{
- bio_post_read_ctx_cache = KMEM_CACHE(bio_post_read_ctx, SLAB_RECLAIM_ACCOUNT);
+ if (!IS_ENABLED(CONFIG_FS_VERITY))
+ return 0;
+ ext4_verity_work_cache =
+ KMEM_CACHE(ext4_verity_work, SLAB_RECLAIM_ACCOUNT);
- if (!bio_post_read_ctx_cache)
+ if (!ext4_verity_work_cache)
goto fail;
- bio_post_read_ctx_pool =
- mempool_create_slab_pool(NUM_PREALLOC_POST_READ_CTXS,
- bio_post_read_ctx_cache);
- if (!bio_post_read_ctx_pool)
+ ext4_verity_work_pool = mempool_create_slab_pool(
+ NUM_VERITY_WORKS, ext4_verity_work_cache);
+ if (!ext4_verity_work_pool)
goto fail_free_cache;
return 0;
fail_free_cache:
- kmem_cache_destroy(bio_post_read_ctx_cache);
+ kmem_cache_destroy(ext4_verity_work_cache);
fail:
return -ENOMEM;
}
-void ext4_exit_post_read_processing(void)
+void ext4_exit_verity_caches(void)
{
- mempool_destroy(bio_post_read_ctx_pool);
- kmem_cache_destroy(bio_post_read_ctx_cache);
+ if (!IS_ENABLED(CONFIG_FS_VERITY))
+ return;
+ mempool_destroy(ext4_verity_work_pool);
+ kmem_cache_destroy(ext4_verity_work_cache);
}
diff --git a/fs/ext4/super.c b/fs/ext4/super.c
index 245f67d10ded..8e7c93a5e458 100644
--- a/fs/ext4/super.c
+++ b/fs/ext4/super.c
@@ -1303,6 +1303,8 @@ static void ext4_put_super(struct super_block *sb)
&sb->s_uuid);
ext4_unregister_li_request(sb);
+ /* Drain deferred EA inode iputs while quota is still active. */
+ flush_delayed_work(&sbi->s_ea_inode_work);
ext4_quotas_off(sb, EXT4_MAXQUOTAS);
destroy_workqueue(sbi->rsv_conversion_wq);
@@ -1423,6 +1425,13 @@ static struct inode *ext4_alloc_inode(struct super_block *sb)
memset(&ei->i_dquot, 0, sizeof(ei->i_dquot));
#endif
ei->jinode = NULL;
+ /*
+ * Reinitialize xattr_sem every allocation because EA inodes
+ * share this space with i_ea_iput_node (via union) which may
+ * have overwritten the semaphore when the slab object was
+ * previously used as an EA inode.
+ */
+ init_rwsem(&ei->xattr_sem);
INIT_LIST_HEAD(&ei->i_rsv_conversion_list);
spin_lock_init(&ei->i_completed_io_lock);
ei->i_sync_tid = 0;
@@ -1488,7 +1497,6 @@ static void init_once(void *foo)
struct ext4_inode_info *ei = foo;
INIT_LIST_HEAD(&ei->i_orphan);
- init_rwsem(&ei->xattr_sem);
init_rwsem(&ei->i_data_sem);
inode_init_once(&ei->vfs_inode);
ext4_fc_init_inode(&ei->vfs_inode);
@@ -4475,8 +4483,9 @@ static int ext4_handle_clustersize(struct super_block *sb)
sbi->s_cluster_bits = 0;
}
sbi->s_clusters_per_group = le32_to_cpu(es->s_clusters_per_group);
- if (sbi->s_clusters_per_group > sb->s_blocksize * 8) {
- ext4_msg(sb, KERN_ERR, "#clusters per group too big: %lu",
+ if (sbi->s_clusters_per_group > sb->s_blocksize * 8 ||
+ sbi->s_clusters_per_group & 7) {
+ ext4_msg(sb, KERN_ERR, "invalid #clusters per group: %lu",
sbi->s_clusters_per_group);
return -EINVAL;
}
@@ -5308,8 +5317,10 @@ static int ext4_block_group_meta_init(struct super_block *sb, int silent)
return -EINVAL;
}
if (sbi->s_inodes_per_group < sbi->s_inodes_per_block ||
- sbi->s_inodes_per_group > sb->s_blocksize * 8) {
- ext4_msg(sb, KERN_ERR, "invalid inodes per group: %lu\n",
+ sbi->s_inodes_per_group > sb->s_blocksize * 8 ||
+ sbi->s_inodes_per_group & 7 ||
+ sbi->s_inodes_per_group % sbi->s_inodes_per_block) {
+ ext4_msg(sb, KERN_ERR, "invalid inodes per group: %lu",
sbi->s_inodes_per_group);
return -EINVAL;
}
@@ -5497,6 +5508,8 @@ static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
ext4_has_feature_orphan_present(sb) ||
ext4_has_feature_journal_needs_recovery(sb));
+ ext4_init_ea_inode_work(sbi);
+
if (ext4_has_feature_mmp(sb) && !sb_rdonly(sb)) {
err = ext4_multi_mount_protect(sb, le64_to_cpu(es->s_mmp_block));
if (err)
@@ -5747,6 +5760,8 @@ static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
return 0;
failed_mount9:
+ /* Drain deferred EA inode iputs before quota shutdown */
+ flush_delayed_work(&sbi->s_ea_inode_work);
ext4_quotas_off(sb, EXT4_MAXQUOTAS);
failed_mount8: __maybe_unused
ext4_release_orphan_info(sb);
@@ -5767,6 +5782,8 @@ failed_mount4:
if (EXT4_SB(sb)->rsv_conversion_wq)
destroy_workqueue(EXT4_SB(sb)->rsv_conversion_wq);
failed_mount_wq:
+ /* Drain deferred EA inode iputs before freeing structures */
+ flush_delayed_work(&sbi->s_ea_inode_work);
ext4_xattr_destroy_cache(sbi->s_ea_inode_cache);
sbi->s_ea_inode_cache = NULL;
@@ -5777,6 +5794,8 @@ failed_mount_wq:
ext4_journal_destroy(sbi, sbi->s_journal);
}
failed_mount3a:
+ /* Drain deferred EA inode iputs from journal replay */
+ flush_delayed_work(&sbi->s_ea_inode_work);
ext4_es_unregister_shrinker(sbi);
failed_mount3:
/* flush s_sb_upd_work before sbi destroy */
@@ -5797,7 +5816,7 @@ failed_mount:
brelse(sbi->s_sbh);
if (sbi->s_journal_bdev_file) {
invalidate_bdev(file_bdev(sbi->s_journal_bdev_file));
- bdev_fput(sbi->s_journal_bdev_file);
+ fs_bdev_file_release(sbi->s_journal_bdev_file, sb);
}
out_fail:
invalidate_bdev(sb->s_bdev);
@@ -5981,9 +6000,9 @@ static struct file *ext4_get_journal_blkdev(struct super_block *sb,
struct ext4_super_block *es;
int errno;
- bdev_file = bdev_file_open_by_dev(j_dev,
+ bdev_file = fs_bdev_file_open_by_dev(j_dev,
BLK_OPEN_READ | BLK_OPEN_WRITE | BLK_OPEN_RESTRICT_WRITES,
- sb, &fs_holder_ops);
+ sb, sb);
if (IS_ERR(bdev_file)) {
ext4_msg(sb, KERN_ERR,
"failed to open journal device unknown-block(%u,%u) %pe",
@@ -6043,7 +6062,7 @@ static struct file *ext4_get_journal_blkdev(struct super_block *sb,
out_bh:
brelse(bh);
out_bdev:
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, sb);
return ERR_PTR(errno);
}
@@ -6082,7 +6101,7 @@ static journal_t *ext4_open_dev_journal(struct super_block *sb,
out_journal:
ext4_journal_destroy(EXT4_SB(sb), journal);
out_bdev:
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, sb);
return ERR_PTR(errno);
}
@@ -6447,6 +6466,7 @@ static int ext4_sync_fs(struct super_block *sb, int wait)
trace_ext4_sync_fs(sb, wait);
flush_workqueue(sbi->rsv_conversion_wq);
+ flush_delayed_work(&sbi->s_ea_inode_work);
/*
* Writeback quota in non-journalled quota case - journalled quota has
* no dirty dquots
@@ -7499,7 +7519,7 @@ static void ext4_kill_sb(struct super_block *sb)
kill_block_super(sb);
if (bdev_file)
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, sb);
}
static struct file_system_type ext4_fs_type = {
@@ -7531,7 +7551,7 @@ static int __init ext4_init_fs(void)
if (err)
goto out7;
- err = ext4_init_post_read_processing();
+ err = ext4_init_verity_caches();
if (err)
goto out6;
@@ -7580,7 +7600,7 @@ out3:
out4:
ext4_exit_pageio();
out5:
- ext4_exit_post_read_processing();
+ ext4_exit_verity_caches();
out6:
ext4_exit_pending();
out7:
@@ -7601,7 +7621,7 @@ static void __exit ext4_exit_fs(void)
ext4_exit_sysfs();
ext4_exit_system_zone();
ext4_exit_pageio();
- ext4_exit_post_read_processing();
+ ext4_exit_verity_caches();
ext4_exit_es();
ext4_exit_pending();
}
diff --git a/fs/ext4/xattr.c b/fs/ext4/xattr.c
index 982a1f831e22..2e724beced11 100644
--- a/fs/ext4/xattr.c
+++ b/fs/ext4/xattr.c
@@ -114,10 +114,6 @@ const struct xattr_handler * const ext4_xattr_handlers[] = {
#define EA_INODE_CACHE(inode) (((struct ext4_sb_info *) \
inode->i_sb->s_fs_info)->s_ea_inode_cache)
-static int
-ext4_expand_inode_array(struct ext4_xattr_inode_array **ea_inode_array,
- struct inode *inode);
-
#ifdef CONFIG_LOCKDEP
void ext4_xattr_inode_set_class(struct inode *ea_inode)
{
@@ -567,7 +563,7 @@ ext4_xattr_inode_get(struct inode *inode, struct ext4_xattr_entry *entry,
ea_inode->i_ino, true /* reusable */);
}
out:
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
return err;
}
@@ -1104,10 +1100,10 @@ static int ext4_xattr_inode_inc_ref_all(handle_t *handle, struct inode *parent,
err = ext4_xattr_inode_inc_ref(handle, ea_inode);
if (err) {
ext4_warning_inode(ea_inode, "inc ref error %d", err);
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
goto cleanup;
}
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
}
return 0;
@@ -1133,7 +1129,7 @@ cleanup:
if (err)
ext4_warning_inode(ea_inode, "cleanup dec ref error %d",
err);
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
}
return saved_err;
}
@@ -1160,7 +1156,6 @@ static void
ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
struct buffer_head *bh,
struct ext4_xattr_entry *first, bool block_csum,
- struct ext4_xattr_inode_array **ea_inode_array,
int extra_credits, bool skip_quota)
{
struct inode *ea_inode;
@@ -1197,14 +1192,6 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
if (err)
continue;
- err = ext4_expand_inode_array(ea_inode_array, ea_inode);
- if (err) {
- ext4_warning_inode(ea_inode,
- "Expand inode array err=%d", err);
- iput(ea_inode);
- continue;
- }
-
err = ext4_journal_ensure_credits_fn(handle, credits, credits,
ext4_free_metadata_revoke_credits(parent->i_sb, 1),
ext4_xattr_restart_fn(handle, parent, bh, block_csum,
@@ -1212,6 +1199,7 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
if (err < 0) {
ext4_warning_inode(ea_inode, "Ensure credits err=%d",
err);
+ ext4_put_ea_inode(ea_inode);
continue;
}
if (err > 0) {
@@ -1221,6 +1209,7 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
ext4_warning_inode(ea_inode,
"Re-get write access err=%d",
err);
+ ext4_put_ea_inode(ea_inode);
continue;
}
}
@@ -1229,6 +1218,7 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
if (err) {
ext4_warning_inode(ea_inode, "ea_inode dec ref err=%d",
err);
+ ext4_put_ea_inode(ea_inode);
continue;
}
@@ -1245,6 +1235,7 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
entry->e_value_inum = 0;
entry->e_value_size = 0;
+ ext4_put_ea_inode(ea_inode);
dirty = true;
}
@@ -1271,7 +1262,6 @@ ext4_xattr_inode_dec_ref_all(handle_t *handle, struct inode *parent,
static void
ext4_xattr_release_block(handle_t *handle, struct inode *inode,
struct buffer_head *bh,
- struct ext4_xattr_inode_array **ea_inode_array,
int extra_credits)
{
struct mb_cache *ea_block_cache = EA_BLOCK_CACHE(inode);
@@ -1313,7 +1303,6 @@ retry_ref:
ext4_xattr_inode_dec_ref_all(handle, inode, bh,
BFIRST(bh),
true /* block_csum */,
- ea_inode_array,
extra_credits,
true /* skip_quota */);
ext4_free_blocks(handle, inode, bh, 0, 1,
@@ -1505,7 +1494,7 @@ static struct inode *ext4_xattr_inode_create(handle_t *handle,
if (ext4_xattr_inode_dec_ref(handle, ea_inode))
ext4_warning_inode(ea_inode,
"cleanup dec ref error %d", err);
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
return ERR_PTR(err);
}
@@ -1550,7 +1539,7 @@ ext4_xattr_inode_cache_find(struct inode *inode, const void *value,
while (ce) {
ea_inode = ext4_iget(inode->i_sb, ce->e_value,
- EXT4_IGET_EA_INODE);
+ EXT4_IGET_EA_INODE | EXT4_IGET_NOWAIT);
if (IS_ERR(ea_inode))
goto next_entry;
ext4_xattr_inode_set_class(ea_inode);
@@ -1564,7 +1553,7 @@ ext4_xattr_inode_cache_find(struct inode *inode, const void *value,
kvfree(ea_data);
return ea_inode;
}
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
next_entry:
ce = mb_cache_entry_find_next(ea_inode_cache, ce);
}
@@ -1615,7 +1604,7 @@ static struct inode *ext4_xattr_inode_lookup_create(handle_t *handle,
ea_inode->i_ino, true /* reusable */);
return ea_inode;
out_err:
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
ext4_xattr_inode_free_quota(inode, NULL, value_len);
return ERR_PTR(err);
}
@@ -1848,7 +1837,7 @@ update_hash:
ret = 0;
out:
- iput(old_ea_inode);
+ ext4_put_ea_inode(old_ea_inode);
return ret;
}
@@ -2010,7 +1999,7 @@ clone_block:
old_ea_inode_quota = le32_to_cpu(
s->here->e_value_size);
}
- iput(tmp_inode);
+ ext4_put_ea_inode(tmp_inode);
s->here->e_value_inum = 0;
s->here->e_value_size = 0;
@@ -2150,7 +2139,7 @@ getblk_failed:
ext4_warning_inode(ea_inode,
"dec ref error=%d",
error);
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
ea_inode = NULL;
}
@@ -2182,12 +2171,8 @@ getblk_failed:
/* Drop the previous xattr block. */
if (bs->bh && bs->bh != new_bh) {
- struct ext4_xattr_inode_array *ea_inode_array = NULL;
-
ext4_xattr_release_block(handle, inode, bs->bh,
- &ea_inode_array,
0 /* extra_credits */);
- ext4_xattr_inode_array_free(ea_inode_array);
}
error = 0;
@@ -2203,7 +2188,7 @@ cleanup:
ext4_xattr_inode_free_quota(inode, ea_inode,
i_size_read(ea_inode));
}
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
}
if (ce)
mb_cache_entry_put(ea_block_cache, ce);
@@ -2285,7 +2270,7 @@ int ext4_xattr_ibody_set(handle_t *handle, struct inode *inode,
ext4_xattr_inode_free_quota(inode, ea_inode,
i_size_read(ea_inode));
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
}
return error;
}
@@ -2297,7 +2282,7 @@ int ext4_xattr_ibody_set(handle_t *handle, struct inode *inode,
header->h_magic = cpu_to_le32(0);
ext4_clear_inode_state(inode, EXT4_STATE_XATTR);
}
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
return 0;
}
@@ -2839,6 +2824,7 @@ retry:
s_min_extra_isize) {
tried_min_extra_isize++;
new_extra_isize = s_min_extra_isize;
+ error = 0;
goto retry;
}
goto cleanup;
@@ -2863,46 +2849,6 @@ cleanup:
return error;
}
-#define EIA_INCR 16 /* must be 2^n */
-#define EIA_MASK (EIA_INCR - 1)
-
-/* Add the large xattr @inode into @ea_inode_array for deferred iput().
- * If @ea_inode_array is new or full it will be grown and the old
- * contents copied over.
- */
-static int
-ext4_expand_inode_array(struct ext4_xattr_inode_array **ea_inode_array,
- struct inode *inode)
-{
- if (*ea_inode_array == NULL) {
- /*
- * Start with 15 inodes, so it fits into a power-of-two size.
- */
- (*ea_inode_array) = kmalloc_flex(**ea_inode_array, inodes,
- EIA_MASK, GFP_NOFS);
- if (*ea_inode_array == NULL)
- return -ENOMEM;
- (*ea_inode_array)->count = 0;
- } else if (((*ea_inode_array)->count & EIA_MASK) == EIA_MASK) {
- /* expand the array once all 15 + n * 16 slots are full */
- struct ext4_xattr_inode_array *new_array = NULL;
-
- new_array = kmalloc_flex(**ea_inode_array, inodes,
- (*ea_inode_array)->count + EIA_INCR,
- GFP_NOFS);
- if (new_array == NULL)
- return -ENOMEM;
- memcpy(new_array, *ea_inode_array,
- struct_size(*ea_inode_array, inodes,
- (*ea_inode_array)->count));
- kfree(*ea_inode_array);
- *ea_inode_array = new_array;
- }
- (*ea_inode_array)->count++;
- (*ea_inode_array)->inodes[(*ea_inode_array)->count - 1] = inode;
- return 0;
-}
-
/*
* ext4_xattr_delete_inode()
*
@@ -2913,7 +2859,6 @@ ext4_expand_inode_array(struct ext4_xattr_inode_array **ea_inode_array,
* references on xattr block and xattr inodes.
*/
int ext4_xattr_delete_inode(handle_t *handle, struct inode *inode,
- struct ext4_xattr_inode_array **ea_inode_array,
int extra_credits)
{
struct buffer_head *bh = NULL;
@@ -2952,7 +2897,6 @@ int ext4_xattr_delete_inode(handle_t *handle, struct inode *inode,
ext4_xattr_inode_dec_ref_all(handle, inode, iloc.bh,
IFIRST(header),
false /* block_csum */,
- ea_inode_array,
extra_credits,
false /* skip_quota */);
}
@@ -2986,12 +2930,12 @@ int ext4_xattr_delete_inode(handle_t *handle, struct inode *inode,
continue;
ext4_xattr_inode_free_quota(inode, ea_inode,
le32_to_cpu(entry->e_value_size));
- iput(ea_inode);
+ ext4_put_ea_inode(ea_inode);
}
}
- ext4_xattr_release_block(handle, inode, bh, ea_inode_array,
+ ext4_xattr_release_block(handle, inode, bh,
extra_credits);
/*
* Update i_file_acl value in the same transaction that releases
@@ -3013,16 +2957,63 @@ cleanup:
return error;
}
-void ext4_xattr_inode_array_free(struct ext4_xattr_inode_array *ea_inode_array)
+/*
+ * Worker function for deferred EA inode iput. Processes all inodes queued
+ * on s_ea_inode_to_free in a context free of xattr_sem/jbd2 handle locks.
+ */
+static void ext4_ea_inode_work(struct work_struct *work)
{
- int idx;
+ struct ext4_sb_info *sbi = container_of(to_delayed_work(work),
+ struct ext4_sb_info,
+ s_ea_inode_work);
+ struct llist_node *node = llist_del_all(&sbi->s_ea_inode_to_free);
- if (ea_inode_array == NULL)
+ while (node) {
+ struct ext4_inode_info *ei = container_of(node,
+ struct ext4_inode_info, i_ea_iput_node);
+ node = node->next;
+ iput(&ei->vfs_inode);
+ }
+}
+
+/*
+ * Release a VFS reference on an EA inode. Must be used instead of iput()
+ * in any context where xattr_sem or a jbd2 handle is held.
+ *
+ * If this is not the last reference, drops it immediately via
+ * iput_if_not_last() with no further action needed.
+ *
+ * If this is the last reference, the inode is linked onto a per-sb
+ * llist via i_ea_iput_node (embedded in ext4_inode_info, sharing space
+ * with the unused xattr_sem) and a delayed worker performs the final
+ * iput() in a clean context.
+ *
+ * Note: while an inode is on s_ea_inode_to_free, the unconsumed i_count
+ * reference (still 1) keeps it in the inode cache, so any concurrent
+ * iget() bumps i_count to >= 2 and iput_if_not_last() will succeed.
+ * Nobody will add the inode a second time until ext4_ea_inode_work()
+ * drops that reference via iput().
+ */
+void ext4_put_ea_inode(struct inode *inode)
+{
+ if (!inode)
+ return;
+ WARN_ON_ONCE(!(EXT4_I(inode)->i_flags & EXT4_EA_INODE_FL));
+ if (iput_if_not_last(inode))
return;
+ llist_add(&EXT4_I(inode)->i_ea_iput_node,
+ &EXT4_SB(inode->i_sb)->s_ea_inode_to_free);
+ /*
+ * Use a short delay to allow multiple EA inodes to accumulate,
+ * reducing workqueue wakeups when several are released together.
+ */
+ schedule_delayed_work(&EXT4_SB(inode->i_sb)->s_ea_inode_work, 1);
+}
- for (idx = 0; idx < ea_inode_array->count; ++idx)
- iput(ea_inode_array->inodes[idx]);
- kfree(ea_inode_array);
+void ext4_init_ea_inode_work(struct ext4_sb_info *sbi)
+{
+ init_llist_head(&sbi->s_ea_inode_to_free);
+ INIT_DELAYED_WORK(&sbi->s_ea_inode_work, ext4_ea_inode_work);
}
/*
diff --git a/fs/ext4/xattr.h b/fs/ext4/xattr.h
index 1fedf44d4fb6..821dc6a50e51 100644
--- a/fs/ext4/xattr.h
+++ b/fs/ext4/xattr.h
@@ -131,11 +131,6 @@ struct ext4_xattr_ibody_find {
struct ext4_iloc iloc;
};
-struct ext4_xattr_inode_array {
- unsigned int count;
- struct inode *inodes[] __counted_by(count);
-};
-
extern const struct xattr_handler ext4_xattr_user_handler;
extern const struct xattr_handler ext4_xattr_trusted_handler;
extern const struct xattr_handler ext4_xattr_security_handler;
@@ -187,9 +182,9 @@ extern int __ext4_xattr_set_credits(struct super_block *sb, struct inode *inode,
bool is_create);
extern int ext4_xattr_delete_inode(handle_t *handle, struct inode *inode,
- struct ext4_xattr_inode_array **array,
int extra_credits);
-extern void ext4_xattr_inode_array_free(struct ext4_xattr_inode_array *array);
+extern void ext4_init_ea_inode_work(struct ext4_sb_info *sbi);
+extern void ext4_put_ea_inode(struct inode *inode);
extern int ext4_expand_extra_isize_ea(struct inode *inode, int new_extra_isize,
struct ext4_inode *raw_inode, handle_t *handle);
diff --git a/fs/f2fs/compress.c b/fs/f2fs/compress.c
index 91855d91bbdd..ce88092d9ce2 100644
--- a/fs/f2fs/compress.c
+++ b/fs/f2fs/compress.c
@@ -1286,8 +1286,6 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc,
.compressed_page = NULL,
.io_type = io_type,
.io_wbc = wbc,
- .encrypted = fscrypt_inode_uses_fs_layer_crypto(cc->inode) ?
- 1 : 0,
};
struct folio *folio;
struct dnode_of_data dn;
@@ -1361,14 +1359,6 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc,
/* wait for GCed page writeback via META_MAPPING */
f2fs_wait_on_block_writeback(inode, fio.old_blkaddr);
-
- if (fio.encrypted) {
- fio.page = cc->rpages[i + 1];
- err = f2fs_encrypt_one_page(&fio);
- if (err)
- goto out_destroy_crypt;
- cc->cpages[i] = fio.encrypted_page;
- }
}
set_cluster_writeback(cc);
@@ -1406,21 +1396,15 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc,
f2fs_bug_on(fio.sbi, blkaddr == NULL_ADDR);
- if (fio.encrypted)
- fio.encrypted_page = cc->cpages[i - 1];
- else
- fio.compressed_page = cc->cpages[i - 1];
+ fio.compressed_page = cc->cpages[i - 1];
cc->cpages[i - 1] = NULL;
fio.submitted = 0;
f2fs_outplace_write_data(&dn, &fio);
if (unlikely(!fio.submitted)) {
cancel_cluster_writeback(cc, cic, i);
-
- /* To call fscrypt_finalize_bounce_page */
- i = cc->valid_nr_cpages;
*submitted = 0;
- goto out_destroy_crypt;
+ goto out_free_page_array;
}
(*submitted)++;
unlock_continue:
@@ -1452,17 +1436,8 @@ unlock_continue:
f2fs_destroy_compress_ctx(cc, false);
return 0;
-out_destroy_crypt:
+out_free_page_array:
page_array_free(sbi, cic->rpages, cc->cluster_size);
-
- if (!fio.encrypted)
- goto out_put_cic;
-
- for (--i; i >= 0; i--) {
- if (!cc->cpages[i])
- continue;
- fscrypt_finalize_bounce_page(&cc->cpages[i]);
- }
out_put_cic:
kmem_cache_free(cic_entry_slab, cic);
out_put_dnode:
diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c
index a765fda71536..1b2d9fe992ea 100644
--- a/fs/f2fs/data.c
+++ b/fs/f2fs/data.c
@@ -65,9 +65,6 @@ bool f2fs_is_cp_guaranteed(const struct folio *folio)
struct inode *inode;
struct f2fs_sb_info *sbi;
- if (fscrypt_is_bounce_folio(folio))
- return folio_test_f2fs_gcing(fscrypt_pagecache_folio(folio));
-
inode = mapping->host;
sbi = F2FS_I_SB(inode);
@@ -101,11 +98,6 @@ static enum count_type __read_io_type(struct folio *folio)
/* postprocessing steps for read bios */
enum bio_post_read_step {
-#ifdef CONFIG_FS_ENCRYPTION
- STEP_DECRYPT = BIT(0),
-#else
- STEP_DECRYPT = 0, /* compile out the decryption-related code */
-#endif
#ifdef CONFIG_F2FS_FS_COMPRESSION
STEP_DECOMPRESS = BIT(1),
#else
@@ -301,11 +293,6 @@ static void f2fs_post_read_work(struct work_struct *work)
container_of(work, struct bio_post_read_ctx, work);
struct bio *bio = ctx->bio;
- if ((ctx->enabled_steps & STEP_DECRYPT) && !fscrypt_decrypt_bio(bio)) {
- f2fs_finish_read_bio(bio, true);
- return;
- }
-
if (ctx->enabled_steps & STEP_DECOMPRESS)
f2fs_handle_step_decompress(ctx, true);
@@ -329,18 +316,11 @@ static void f2fs_read_end_io(struct bio *bio)
return;
}
- if (ctx) {
- unsigned int enabled_steps = ctx->enabled_steps &
- (STEP_DECRYPT | STEP_DECOMPRESS);
-
- /*
- * If we have only decompression step between decompression and
- * decrypt, we don't need post processing for this.
- */
- if (enabled_steps == STEP_DECOMPRESS &&
- !f2fs_low_mem_mode(sbi)) {
+ if (ctx && (ctx->enabled_steps & STEP_DECOMPRESS)) {
+ if (!f2fs_low_mem_mode(sbi)) {
+ /* Decompress inline. */
f2fs_handle_step_decompress(ctx, intask);
- } else if (enabled_steps) {
+ } else {
INIT_WORK(&ctx->work, f2fs_post_read_work);
queue_work(ctx->sbi->wq, &ctx->work);
return;
@@ -362,13 +342,6 @@ static void f2fs_write_end_bio(struct bio *bio)
struct folio *folio = fi.folio;
enum count_type type;
- if (fscrypt_is_bounce_folio(folio)) {
- struct folio *io_folio = folio;
-
- folio = fscrypt_pagecache_folio(io_folio);
- fscrypt_free_bounce_page(&io_folio->page);
- }
-
#ifdef CONFIG_F2FS_FS_COMPRESSION
if (f2fs_is_compressed_page(folio)) {
f2fs_compress_write_end_io(bio, folio);
@@ -614,11 +587,6 @@ static bool __has_merged_page(struct bio *bio, struct inode *inode,
bio_for_each_folio_all(fi, bio) {
struct folio *target = fi.folio;
- if (fscrypt_is_bounce_folio(target)) {
- target = fscrypt_pagecache_folio(target);
- if (IS_ERR(target))
- continue;
- }
if (f2fs_is_compressed_page(target)) {
target = f2fs_compress_control_folio(target);
if (IS_ERR(target))
@@ -1161,9 +1129,6 @@ static struct bio *f2fs_grab_read_bio(struct inode *inode,
f2fs_set_bio_crypt_ctx(bio, inode, first_idx, NULL, GFP_NOFS);
bio->bi_end_io = f2fs_read_end_io;
- if (fscrypt_inode_uses_fs_layer_crypto(inode))
- post_read_steps |= STEP_DECRYPT;
-
if (vi)
post_read_steps |= STEP_VERITY;
@@ -1323,10 +1288,11 @@ retry:
if (folio_test_large(folio)) {
pgoff_t folio_index = mapping_align_index(mapping, index);
+ unsigned long nr_pages = folio_nr_pages(folio);
f2fs_folio_put(folio, true);
invalidate_inode_pages2_range(mapping, folio_index,
- folio_index + folio_nr_pages(folio) - 1);
+ folio_index + nr_pages - 1);
f2fs_schedule_timeout(DEFAULT_SCHEDULE_TIMEOUT);
goto retry;
}
@@ -2852,35 +2818,6 @@ static void f2fs_readahead(struct readahead_control *rac)
f2fs_mpage_readpages(inode, vi, rac, NULL);
}
-int f2fs_encrypt_one_page(struct f2fs_io_info *fio)
-{
- struct inode *inode = fio_inode(fio);
- struct folio *mfolio;
- struct page *page;
-
- if (!f2fs_encrypted_file(inode))
- return 0;
-
- page = fio->compressed_page ? fio->compressed_page : fio->page;
-
- if (fscrypt_inode_uses_inline_crypto(inode))
- return 0;
-
- fio->encrypted_page = fscrypt_encrypt_pagecache_blocks(page_folio(page),
- PAGE_SIZE, 0, GFP_NOFS);
- if (IS_ERR(fio->encrypted_page))
- return PTR_ERR(fio->encrypted_page);
-
- mfolio = filemap_lock_folio(META_MAPPING(fio->sbi), fio->old_blkaddr);
- if (!IS_ERR(mfolio)) {
- if (folio_test_uptodate(mfolio))
- memcpy(folio_address(mfolio),
- page_address(fio->encrypted_page), PAGE_SIZE);
- f2fs_folio_put(mfolio, true);
- }
- return 0;
-}
-
static inline bool check_inplace_update_policy(struct inode *inode,
struct f2fs_io_info *fio)
{
@@ -3053,22 +2990,15 @@ got_it:
if (ipu_force ||
(__is_valid_data_blkaddr(fio->old_blkaddr) &&
need_inplace_update(fio))) {
- err = f2fs_encrypt_one_page(fio);
- if (err)
- goto out_writepage;
-
folio_start_writeback(folio);
f2fs_put_dnode(&dn);
if (fio->need_lock == LOCK_REQ)
f2fs_unlock_op(fio->sbi, &lc);
err = f2fs_inplace_write_data(fio);
- if (err) {
- if (fscrypt_inode_uses_fs_layer_crypto(inode))
- fscrypt_finalize_bounce_page(&fio->encrypted_page);
+ if (err)
folio_end_writeback(folio);
- } else {
+ else
set_inode_flag(inode, FI_UPDATE_WRITE);
- }
trace_f2fs_do_write_data_page(folio, IPU);
return err;
}
@@ -3087,10 +3017,6 @@ got_it:
fio->version = ni.version;
- err = f2fs_encrypt_one_page(fio);
- if (err)
- goto out_writepage;
-
folio_start_writeback(folio);
if (fio->compr_blocks && fio->old_blkaddr == COMPRESS_ADDR)
@@ -3975,7 +3901,7 @@ repeat:
* Will wait that below with our IO control.
*/
folio = f2fs_filemap_get_folio(mapping, index,
- FGP_LOCK | FGP_WRITE | FGP_CREAT | FGP_NOFS,
+ FGP_LOCK | FGP_WRITE | FGP_CREAT,
mapping_gfp_mask(mapping));
if (IS_ERR(folio)) {
err = PTR_ERR(folio);
@@ -4031,7 +3957,7 @@ repeat:
/*
* Although the block may be stored in the COW inode, the folio
* belongs to @inode and its data was encrypted (or not) using
- * @inode's context (see f2fs_encrypt_one_page()). Read with
+ * @inode's context (see f2fs_set_bio_crypt_ctx()). Read with
* @inode so the post-read decryption decision matches the
* folio's owner; otherwise an unencrypted @inode whose COW inode
* is encrypted hits a NULL ->i_crypt_info on decryption.
@@ -4602,9 +4528,9 @@ static int f2fs_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
iomap->offset = F2FS_BLK_TO_BYTES(map.m_lblk);
/*
- * When inline encryption is enabled, sometimes I/O to an encrypted file
- * has to be broken up to guarantee DUN contiguity. Handle this by
- * limiting the length of the mapping returned.
+ * Sometimes I/O to an encrypted file has to be broken up to guarantee
+ * DUN contiguity. Handle this by limiting the length of the mapping
+ * returned.
*/
map.m_len = fscrypt_limit_io_blocks(inode, map.m_lblk, map.m_len);
diff --git a/fs/f2fs/debug.c b/fs/f2fs/debug.c
index af88db8fdb71..ff379aff4472 100644
--- a/fs/f2fs/debug.c
+++ b/fs/f2fs/debug.c
@@ -352,10 +352,6 @@ static void update_mem_info(struct f2fs_sb_info *sbi)
get_cache:
si->cache_mem = 0;
- /* build gc */
- if (sbi->gc_thread)
- si->cache_mem += sizeof(struct f2fs_gc_kthread);
-
/* build merge flush thread */
if (SM_I(sbi)->fcc_info)
si->cache_mem += sizeof(struct flush_cmd_control);
diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h
index 8f3e632f315c..9c73b1895568 100644
--- a/fs/f2fs/f2fs.h
+++ b/fs/f2fs/f2fs.h
@@ -1364,7 +1364,6 @@ struct f2fs_io_info {
unsigned int submitted:1; /* indicate IO submission */
unsigned int in_list:1; /* indicate fio is in io_list */
unsigned int is_por:1; /* indicate IO is from recovery or not */
- unsigned int encrypted:1; /* indicate file is encrypted */
unsigned int meta_gc:1; /* require meta inode GC */
enum iostat_type io_type; /* io type */
struct writeback_control *io_wbc; /* writeback control */
@@ -1748,6 +1747,33 @@ struct decompress_io_ctx {
#define MAX_COMPRESS_LOG_SIZE 8
#define MAX_COMPRESS_WINDOW_SIZE(log_size) ((PAGE_SIZE) << (log_size))
+struct f2fs_gc_kthread {
+ struct task_struct *f2fs_gc_task;
+ wait_queue_head_t gc_wait_queue_head;
+
+ /* for gc sleep time */
+ unsigned int urgent_sleep_time;
+ unsigned int min_sleep_time;
+ unsigned int max_sleep_time;
+ unsigned int no_gc_sleep_time;
+
+ /* for changing gc mode */
+ bool gc_wake;
+
+ /* for GC_MERGE mount option */
+ wait_queue_head_t fggc_wq; /*
+ * caller of f2fs_balance_fs()
+ * will wait on this wait queue.
+ */
+
+ /* for gc control for zoned devices */
+ unsigned int no_zoned_gc_percent;
+ unsigned int boost_zoned_gc_percent;
+ unsigned int valid_thresh_ratio;
+ unsigned int boost_gc_multiple;
+ unsigned int boost_gc_greedy;
+};
+
struct f2fs_sb_info {
struct super_block *sb; /* pointer to VFS super block */
struct proc_dir_entry *s_proc; /* proc entry */
@@ -1883,7 +1909,7 @@ struct f2fs_sb_info {
* semaphore for GC, avoid
* race between GC and GC or CP
*/
- struct f2fs_gc_kthread *gc_thread; /* GC thread */
+ struct f2fs_gc_kthread gc_thread; /* GC thread */
struct atgc_management am; /* atgc management */
unsigned int cur_victim_sec; /* current victim section num */
unsigned int gc_mode; /* current GC state */
@@ -4199,7 +4225,6 @@ int f2fs_do_write_data_page(struct f2fs_io_info *fio);
int f2fs_map_blocks(struct inode *inode, struct f2fs_map_blocks *map, int flag);
int f2fs_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo,
u64 start, u64 len);
-int f2fs_encrypt_one_page(struct f2fs_io_info *fio);
bool f2fs_should_update_inplace(struct inode *inode, struct f2fs_io_info *fio);
bool f2fs_should_update_outplace(struct inode *inode, struct f2fs_io_info *fio);
int f2fs_write_single_data_page(struct folio *folio, int *submitted,
diff --git a/fs/f2fs/file.c b/fs/f2fs/file.c
index 4b52c56d71f0..9d315c3cd9c8 100644
--- a/fs/f2fs/file.c
+++ b/fs/f2fs/file.c
@@ -950,8 +950,6 @@ static bool f2fs_force_buffered_io(struct inode *inode, int rw)
{
struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
- if (!fscrypt_dio_supported(inode))
- return true;
if (fsverity_active(inode))
return true;
if (f2fs_compressed_file(inode))
@@ -996,9 +994,7 @@ int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path,
}
/*
- * Return the DIO alignment restrictions if requested. We only return
- * this information when requested, since on encrypted files it might
- * take a fair bit of work to get if the file wasn't opened recently.
+ * Return the DIO alignment restrictions if requested.
*
* f2fs sometimes supports DIO reads but not DIO writes. STATX_DIOALIGN
* cannot represent that, so in that case we report no DIO support.
@@ -1107,17 +1103,23 @@ int f2fs_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
!IS_ALIGNED(attr->ia_size,
F2FS_BLK_TO_BYTES(fi->i_cluster_size)))
return -EINVAL;
- /*
- * To prevent scattered pin block generation, we don't allow
- * smaller/equal size unaligned truncation for pinned file.
- * We only support overwrite IO to pinned file, so don't
- * care about larger size truncation.
- */
- if (f2fs_is_pinned_file(inode) &&
- attr->ia_size <= i_size_read(inode) &&
- !IS_ALIGNED(attr->ia_size,
- F2FS_BLK_TO_BYTES(CAP_BLKS_PER_SEC(sbi))))
- return -EINVAL;
+
+ if (f2fs_is_pinned_file(inode)) {
+ /*
+ * It may break section-aligned fallocate recovery
+ * mechanism, so do not allow larger size truncation.
+ */
+ if (attr->ia_size > i_size_read(inode))
+ return -EINVAL;
+ /*
+ * To prevent scattered pin block generation, we don't
+ * allow smaller/equal size unaligned truncation for
+ * pinned file.
+ */
+ else if (!IS_ALIGNED(attr->ia_size,
+ F2FS_BLK_TO_BYTES(CAP_BLKS_PER_SEC(sbi))))
+ return -EINVAL;
+ }
}
if (is_quota_modification(idmap, inode, attr)) {
diff --git a/fs/f2fs/gc.c b/fs/f2fs/gc.c
index ffaa7ba76a1b..d04633f872ef 100644
--- a/fs/f2fs/gc.c
+++ b/fs/f2fs/gc.c
@@ -31,9 +31,9 @@ static unsigned int count_bits(const unsigned long *addr,
static int gc_thread_func(void *data)
{
struct f2fs_sb_info *sbi = data;
- struct f2fs_gc_kthread *gc_th = sbi->gc_thread;
- wait_queue_head_t *wq = &sbi->gc_thread->gc_wait_queue_head;
- wait_queue_head_t *fggc_wq = &sbi->gc_thread->fggc_wq;
+ struct f2fs_gc_kthread *gc_th = &sbi->gc_thread;
+ wait_queue_head_t *wq = &sbi->gc_thread.gc_wait_queue_head;
+ wait_queue_head_t *fggc_wq = &sbi->gc_thread.fggc_wq;
unsigned int wait_ms;
struct f2fs_gc_control gc_control = {
.victim_segno = NULL_SEGNO,
@@ -193,13 +193,9 @@ next:
int f2fs_start_gc_thread(struct f2fs_sb_info *sbi)
{
- struct f2fs_gc_kthread *gc_th;
+ struct f2fs_gc_kthread *gc_th = &sbi->gc_thread;
dev_t dev = sbi->sb->s_bdev->bd_dev;
- gc_th = f2fs_kmalloc(sbi, sizeof(struct f2fs_gc_kthread), GFP_KERNEL);
- if (!gc_th)
- return -ENOMEM;
-
gc_th->urgent_sleep_time = DEF_GC_THREAD_URGENT_SLEEP_TIME;
gc_th->valid_thresh_ratio = DEF_GC_THREAD_VALID_THRESH_RATIO;
gc_th->boost_gc_multiple = BOOST_GC_MULTIPLE;
@@ -221,16 +217,14 @@ int f2fs_start_gc_thread(struct f2fs_sb_info *sbi)
gc_th->gc_wake = false;
- sbi->gc_thread = gc_th;
- init_waitqueue_head(&sbi->gc_thread->gc_wait_queue_head);
- init_waitqueue_head(&sbi->gc_thread->fggc_wq);
- sbi->gc_thread->f2fs_gc_task = kthread_run(gc_thread_func, sbi,
+ init_waitqueue_head(&gc_th->gc_wait_queue_head);
+ init_waitqueue_head(&gc_th->fggc_wq);
+ gc_th->f2fs_gc_task = kthread_run(gc_thread_func, sbi,
"f2fs_gc-%u:%u", MAJOR(dev), MINOR(dev));
if (IS_ERR(gc_th->f2fs_gc_task)) {
int err = PTR_ERR(gc_th->f2fs_gc_task);
- kfree(gc_th);
- sbi->gc_thread = NULL;
+ gc_th->f2fs_gc_task = NULL;
return err;
}
@@ -241,14 +235,14 @@ int f2fs_start_gc_thread(struct f2fs_sb_info *sbi)
void f2fs_stop_gc_thread(struct f2fs_sb_info *sbi)
{
- struct f2fs_gc_kthread *gc_th = sbi->gc_thread;
+ struct f2fs_gc_kthread *gc_th = &sbi->gc_thread;
- if (!gc_th)
+ if (!gc_th->f2fs_gc_task)
return;
+
kthread_stop(gc_th->f2fs_gc_task);
+ gc_th->f2fs_gc_task = NULL;
wake_up_all(&gc_th->fggc_wq);
- kfree(gc_th);
- sbi->gc_thread = NULL;
}
static int select_gc_type(struct f2fs_sb_info *sbi, int gc_type)
@@ -796,7 +790,7 @@ int f2fs_get_victim(struct f2fs_sb_info *sbi, unsigned int *result,
if (one_time) {
p.one_time_gc = one_time;
if (has_enough_free_secs(sbi, 0, NR_PERSISTENT_LOG))
- valid_thresh_ratio = sbi->gc_thread->valid_thresh_ratio;
+ valid_thresh_ratio = sbi->gc_thread.valid_thresh_ratio;
}
retry:
@@ -1807,9 +1801,9 @@ static int do_garbage_collect(struct f2fs_sb_info *sbi,
if (f2fs_sb_has_blkzoned(sbi) &&
!has_enough_free_blocks(sbi,
- sbi->gc_thread->boost_zoned_gc_percent))
+ sbi->gc_thread.boost_zoned_gc_percent))
window_granularity *=
- sbi->gc_thread->boost_gc_multiple;
+ sbi->gc_thread.boost_gc_multiple;
end_segno = start_segno + window_granularity;
}
diff --git a/fs/f2fs/gc.h b/fs/f2fs/gc.h
index 6c4d4567571e..b015742fb455 100644
--- a/fs/f2fs/gc.h
+++ b/fs/f2fs/gc.h
@@ -45,32 +45,7 @@
#define NR_GC_CHECKPOINT_SECS (3) /* data/node/dentry sections */
-struct f2fs_gc_kthread {
- struct task_struct *f2fs_gc_task;
- wait_queue_head_t gc_wait_queue_head;
-
- /* for gc sleep time */
- unsigned int urgent_sleep_time;
- unsigned int min_sleep_time;
- unsigned int max_sleep_time;
- unsigned int no_gc_sleep_time;
-
- /* for changing gc mode */
- bool gc_wake;
-
- /* for GC_MERGE mount option */
- wait_queue_head_t fggc_wq; /*
- * caller of f2fs_balance_fs()
- * will wait on this wait queue.
- */
- /* for gc control for zoned devices */
- unsigned int no_zoned_gc_percent;
- unsigned int boost_zoned_gc_percent;
- unsigned int valid_thresh_ratio;
- unsigned int boost_gc_multiple;
- unsigned int boost_gc_greedy;
-};
struct gc_inode_list {
struct list_head ilist;
@@ -197,6 +172,6 @@ static inline bool need_to_boost_gc(struct f2fs_sb_info *sbi)
{
if (f2fs_sb_has_blkzoned(sbi))
return !has_enough_free_blocks(sbi,
- sbi->gc_thread->boost_zoned_gc_percent);
+ sbi->gc_thread.boost_zoned_gc_percent);
return has_enough_invalid_blocks(sbi);
}
diff --git a/fs/f2fs/namei.c b/fs/f2fs/namei.c
index cac03b8e91a1..5ae647a352aa 100644
--- a/fs/f2fs/namei.c
+++ b/fs/f2fs/namei.c
@@ -366,7 +366,7 @@ fail_drop:
}
static int f2fs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct f2fs_sb_info *sbi = F2FS_I_SB(dir);
struct f2fs_lock_context lc;
@@ -742,7 +742,7 @@ static struct dentry *f2fs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (err)
return ERR_PTR(err);
- inode = f2fs_new_inode(idmap, dir, S_IFDIR | mode, NULL);
+ inode = f2fs_new_inode(idmap, dir, mode, NULL);
if (IS_ERR(inode))
return ERR_CAST(inode);
diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c
index d71ddb3ee918..ac322d63e3f9 100644
--- a/fs/f2fs/segment.c
+++ b/fs/f2fs/segment.c
@@ -452,15 +452,14 @@ void f2fs_balance_fs(struct f2fs_sb_info *sbi, bool need)
f2fs_submit_merged_write(sbi, DATA);
f2fs_submit_all_merged_ipu_writes(sbi);
- if (test_opt(sbi, GC_MERGE) && sbi->gc_thread &&
- sbi->gc_thread->f2fs_gc_task) {
+ if (test_opt(sbi, GC_MERGE) && sbi->gc_thread.f2fs_gc_task) {
DEFINE_WAIT(wait);
- prepare_to_wait(&sbi->gc_thread->fggc_wq, &wait,
+ prepare_to_wait(&sbi->gc_thread.fggc_wq, &wait,
TASK_UNINTERRUPTIBLE);
- wake_up(&sbi->gc_thread->gc_wait_queue_head);
+ wake_up(&sbi->gc_thread.gc_wait_queue_head);
io_schedule();
- finish_wait(&sbi->gc_thread->fggc_wq, &wait);
+ finish_wait(&sbi->gc_thread.fggc_wq, &wait);
} else {
struct f2fs_gc_control gc_control = {
.victim_segno = NULL_SEGNO,
@@ -3986,8 +3985,6 @@ static void do_write_page(struct f2fs_summary *sum, struct f2fs_io_info *fio)
"%s Failed to allocate data block, ino:%u, index:%lu, type:%d, old_blkaddr:0x%x, new_blkaddr:0x%x, err:%d",
__func__, fio->ino, folio->index, type,
fio->old_blkaddr, fio->new_blkaddr, err);
- if (fscrypt_inode_uses_fs_layer_crypto(folio->mapping->host))
- fscrypt_finalize_bounce_page(&fio->encrypted_page);
folio_end_writeback(folio);
if (f2fs_in_warm_node_list(folio))
f2fs_del_fsync_node_entry(fio->sbi, folio);
diff --git a/fs/f2fs/super.c b/fs/f2fs/super.c
index 9760e4efeffe..de99eead6f8d 100644
--- a/fs/f2fs/super.c
+++ b/fs/f2fs/super.c
@@ -1971,7 +1971,7 @@ static void destroy_device_list(struct f2fs_sb_info *sbi)
for (i = 0; i < sbi->s_ndevs; i++) {
if (i > 0)
- bdev_fput(FDEV(i).bdev_file);
+ fs_bdev_file_release(FDEV(i).bdev_file, sbi->sb);
#ifdef CONFIG_BLK_DEV_ZONED
kvfree(FDEV(i).blkz_seq);
#endif
@@ -2943,11 +2943,11 @@ static int __f2fs_remount(struct fs_context *fc, struct super_block *sb)
if ((flags & SB_RDONLY) ||
(F2FS_OPTION(sbi).bggc_mode == BGGC_MODE_OFF &&
!test_opt(sbi, GC_MERGE))) {
- if (sbi->gc_thread) {
+ if (sbi->gc_thread.f2fs_gc_task) {
f2fs_stop_gc_thread(sbi);
need_restart_gc = true;
}
- } else if (!sbi->gc_thread) {
+ } else if (!sbi->gc_thread.f2fs_gc_task) {
err = f2fs_start_gc_thread(sbi);
if (err)
goto restore_opts;
@@ -3168,7 +3168,7 @@ static ssize_t f2fs_quota_read(struct super_block *sb, int type, char *data,
repeat:
folio = mapping_read_folio_gfp(mapping, off >> PAGE_SHIFT,
- GFP_NOFS);
+ GFP_KERNEL);
if (IS_ERR(folio)) {
if (PTR_ERR(folio) == -ENOMEM) {
memalloc_retry_wait(GFP_NOFS);
@@ -3775,7 +3775,7 @@ f2fs_get_devices(struct super_block *sb,
static const struct fscrypt_operations f2fs_cryptops = {
.inode_info_offs = (int)offsetof(struct f2fs_inode_info, i_crypt_info) -
(int)offsetof(struct f2fs_inode_info, vfs_inode),
- .needs_bounce_pages = 1,
+ .is_block_based = 1,
.has_32bit_inodes = 1,
.supports_subblock_data_units = 1,
.legacy_key_prefix = "f2fs:",
@@ -4901,8 +4901,8 @@ static int f2fs_scan_devices(struct f2fs_sb_info *sbi)
FDEV(i).end_blk = FDEV(i).start_blk +
SEGS_TO_BLKS(sbi,
FDEV(i).total_segments) - 1;
- FDEV(i).bdev_file = bdev_file_open_by_path(
- FDEV(i).path, mode, sbi->sb, NULL);
+ FDEV(i).bdev_file = fs_bdev_file_open_by_path(
+ FDEV(i).path, mode, sbi->sb, sbi->sb);
}
}
if (IS_ERR(FDEV(i).bdev_file))
diff --git a/fs/f2fs/sysfs.c b/fs/f2fs/sysfs.c
index 665687244c93..be92c05a5420 100644
--- a/fs/f2fs/sysfs.c
+++ b/fs/f2fs/sysfs.c
@@ -75,7 +75,7 @@ static ssize_t f2fs_sbi_show(struct f2fs_attr *a,
static unsigned char *__struct_ptr(struct f2fs_sb_info *sbi, int struct_type)
{
if (struct_type == GC_THREAD)
- return (unsigned char *)sbi->gc_thread;
+ return (unsigned char *)&sbi->gc_thread;
else if (struct_type == SM_INFO)
return (unsigned char *)SM_I(sbi);
else if (struct_type == DCC_INFO)
@@ -664,20 +664,20 @@ out:
sbi->gc_mode = GC_NORMAL;
} else if (t == 1) {
sbi->gc_mode = GC_URGENT_HIGH;
- if (sbi->gc_thread) {
- sbi->gc_thread->gc_wake = true;
+ if (sbi->gc_thread.f2fs_gc_task) {
+ sbi->gc_thread.gc_wake = true;
wake_up_interruptible_all(
- &sbi->gc_thread->gc_wait_queue_head);
+ &sbi->gc_thread.gc_wait_queue_head);
wake_up_discard_thread(sbi, true);
}
} else if (t == 2) {
sbi->gc_mode = GC_URGENT_LOW;
} else if (t == 3) {
sbi->gc_mode = GC_URGENT_MID;
- if (sbi->gc_thread) {
- sbi->gc_thread->gc_wake = true;
+ if (sbi->gc_thread.f2fs_gc_task) {
+ sbi->gc_thread.gc_wake = true;
wake_up_interruptible_all(
- &sbi->gc_thread->gc_wait_queue_head);
+ &sbi->gc_thread.gc_wait_queue_head);
}
} else {
return -EINVAL;
@@ -934,14 +934,14 @@ out:
if (!strcmp(a->attr.name, "gc_boost_gc_multiple")) {
if (t < 1 || t > SEGS_PER_SEC(sbi))
return -EINVAL;
- sbi->gc_thread->boost_gc_multiple = (unsigned int)t;
+ sbi->gc_thread.boost_gc_multiple = (unsigned int)t;
return count;
}
if (!strcmp(a->attr.name, "gc_boost_gc_greedy")) {
if (t > GC_GREEDY)
return -EINVAL;
- sbi->gc_thread->boost_gc_greedy = (unsigned int)t;
+ sbi->gc_thread.boost_gc_greedy = (unsigned int)t;
return count;
}
@@ -989,8 +989,8 @@ out:
if (sbi->cprc_info.f2fs_issue_ckpt)
set_user_nice(sbi->cprc_info.f2fs_issue_ckpt,
PRIO_TO_NICE(sbi->critical_task_priority));
- if (sbi->gc_thread && sbi->gc_thread->f2fs_gc_task)
- set_user_nice(sbi->gc_thread->f2fs_gc_task,
+ if (sbi->gc_thread.f2fs_gc_task)
+ set_user_nice(sbi->gc_thread.f2fs_gc_task,
PRIO_TO_NICE(sbi->critical_task_priority));
return count;
}
diff --git a/fs/fat/namei_msdos.c b/fs/fat/namei_msdos.c
index 0fd2971ad4b1..ee6824c5d136 100644
--- a/fs/fat/namei_msdos.c
+++ b/fs/fat/namei_msdos.c
@@ -29,6 +29,9 @@ static int msdos_format_name(const unsigned char *name, int len,
unsigned char c;
int space;
+ if (len > NAME_MAX)
+ return -ENAMETOOLONG;
+
if (name[0] == '.') { /* dotfile because . and .. already done */
if (opts->dotsOK) {
/* Get rid of dot - test for it elsewhere */
@@ -262,7 +265,7 @@ static int msdos_add_entry(struct inode *dir, const unsigned char *name,
/***** Create a file */
static int msdos_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct super_block *sb = dir->i_sb;
struct inode *inode = NULL;
diff --git a/fs/fat/namei_vfat.c b/fs/fat/namei_vfat.c
index e909447873e3..139d3ef4bfae 100644
--- a/fs/fat/namei_vfat.c
+++ b/fs/fat/namei_vfat.c
@@ -755,7 +755,7 @@ error:
}
static int vfat_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct super_block *sb = dir->i_sb;
struct inode *inode;
diff --git a/fs/fs_struct.c b/fs/fs_struct.c
index 394875d06fd6..34699f3b6f88 100644
--- a/fs/fs_struct.c
+++ b/fs/fs_struct.c
@@ -8,6 +8,7 @@
#include <linux/fs_struct.h>
#include <linux/init_task.h>
#include "internal.h"
+#include "mount.h"
/*
* Replace the fs->{rootmnt,root} with {mnt,dentry}. Put the old values.
@@ -60,8 +61,11 @@ void chroot_fs_refs(const struct path *old_root, const struct path *new_root)
read_lock(&tasklist_lock);
for_each_process_thread(g, p) {
+ if (p->flags & (PF_KTHREAD | PF_EXITING | PF_DUMPCORE))
+ continue;
+
task_lock(p);
- fs = p->fs;
+ fs = p->real_fs;
if (fs) {
int hits = 0;
write_seqlock(&fs->seq);
@@ -89,12 +93,13 @@ void free_fs_struct(struct fs_struct *fs)
void exit_fs(struct task_struct *tsk)
{
- struct fs_struct *fs = tsk->fs;
+ struct fs_struct *fs = tsk->real_fs;
if (fs) {
int kill;
task_lock(tsk);
read_seqlock_excl(&fs->seq);
+ tsk->real_fs = NULL;
tsk->fs = NULL;
kill = !--fs->users;
read_sequnlock_excl(&fs->seq);
@@ -126,7 +131,7 @@ struct fs_struct *copy_fs_struct(struct fs_struct *old)
int unshare_fs_struct(void)
{
- struct fs_struct *fs = current->fs;
+ struct fs_struct *fs = current->real_fs;
struct fs_struct *new_fs = copy_fs_struct(fs);
int kill;
@@ -135,8 +140,10 @@ int unshare_fs_struct(void)
task_lock(current);
read_seqlock_excl(&fs->seq);
+ VFS_WARN_ON_ONCE(fs != current->fs);
kill = !--fs->users;
current->fs = new_fs;
+ current->real_fs = new_fs;
read_sequnlock_excl(&fs->seq);
task_unlock(current);
@@ -147,9 +154,99 @@ int unshare_fs_struct(void)
}
EXPORT_SYMBOL_GPL(unshare_fs_struct);
+/*
+ * PID 1 may choose to stop sharing fs_struct state with us.
+ * Either via unshare(CLONE_FS) or unshare(CLONE_NEWNS). Of
+ * course, PID 1 could have chosen to create arbitrary process
+ * trees that all share fs_struct state via CLONE_FS. This is a
+ * strong statement: We only care about PID 1 aka the thread-group
+ * leader so subthread's fs_struct state doesn't matter.
+ *
+ * PID 1 unsharing fs_struct state is a bug. PID 1 relies on
+ * various kthreads to be able to perform work based on its
+ * fs_struct state. Breaking that contract sucks for both sides.
+ * So just don't bother with extra work for this. No sane init
+ * system should ever do this.
+ *
+ * On older kernels if PID 1 unshared its filesystem state with us the
+ * kernel simply used the stale fs_struct state implicitly pinning
+ * anything that PID 1 had last used. Even if PID 1 might've moved on to
+ * some completely different fs_struct state and might've even unmounted
+ * the old root.
+ *
+ * This has hilarious consequences: Think continuing to dump coredump
+ * state into an implicitly pinned directory somewhere. Calling random
+ * binaries in the old rootfs via usermodehelpers.
+ *
+ * Be aggressive about this: We simply reject operating on stale
+ * fs_struct state by reverting to nullfs. Every kworker that does
+ * lookups after this point will fail. Every usermodehelper call will
+ * fail. Tough luck but let's be kind and emit a warning to userspace.
+ */
+static inline void validate_fs_switch(struct fs_struct *old_fs)
+{
+ might_sleep();
+
+ if (likely(current->pid != 1))
+ return;
+ /* @old_fs may be dangling but for comparison it's fine */
+ if (old_fs != userspace_init_fs)
+ return;
+ pr_warn("VFS: Pid 1 stopped sharing filesystem state\n");
+ set_fs_root(userspace_init_fs, &init_fs.root);
+ set_fs_pwd(userspace_init_fs, &init_fs.root);
+}
+
+struct fs_struct *switch_fs_struct(struct fs_struct *new_fs)
+{
+ struct fs_struct *fs;
+
+ scoped_guard(task_lock, current) {
+ fs = current->fs;
+ VFS_WARN_ON_ONCE(fs != current->real_fs);
+ read_seqlock_excl(&fs->seq);
+ current->fs = new_fs;
+ current->real_fs = new_fs;
+ if (--fs->users)
+ new_fs = NULL;
+ else
+ new_fs = fs;
+ read_sequnlock_excl(&fs->seq);
+ }
+
+ validate_fs_switch(fs);
+ return new_fs;
+}
+
/* to be mentioned only in INIT_TASK */
struct fs_struct init_fs = {
.users = 1,
.seq = __SEQLOCK_UNLOCKED(init_fs.seq),
.umask = 0022,
};
+
+struct fs_struct *userspace_init_fs __ro_after_init;
+EXPORT_SYMBOL_GPL(userspace_init_fs);
+
+void __init init_userspace_fs(void)
+{
+ struct mount *m;
+ struct path root;
+
+ /* Move PID 1 from nullfs into the initramfs. */
+ m = topmost_overmount(current->nsproxy->mnt_ns->root);
+ root.mnt = &m->mnt;
+ root.dentry = root.mnt->mnt_root;
+
+ VFS_WARN_ON_ONCE(current->pid != 1);
+
+ set_fs_root(current->fs, &root);
+ set_fs_pwd(current->fs, &root);
+
+ /* Hold a reference for the global pointer. */
+ read_seqlock_excl(&current->fs->seq);
+ current->fs->users++;
+ read_sequnlock_excl(&current->fs->seq);
+
+ userspace_init_fs = current->fs;
+}
diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c
index 0e2a1039fa43..d4e0029810c0 100644
--- a/fs/fuse/dir.c
+++ b/fs/fuse/dir.c
@@ -1084,7 +1084,7 @@ static int fuse_mknod(struct mnt_idmap *idmap, struct inode *dir,
}
static int fuse_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *entry, umode_t mode, bool excl)
+ struct dentry *entry, umode_t mode)
{
return fuse_mknod(idmap, dir, entry, mode, 0);
}
@@ -1117,6 +1117,14 @@ static struct dentry *fuse_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (!fm->fc->dont_mask)
mode &= ~current_umask();
+ /*
+ * vfs_mkdir() now passes S_IFDIR in @mode, but @mode is forwarded
+ * verbatim to the userspace server which has only ever been given the
+ * permission bits. Strip the type bit until the protocol is known to
+ * cope with it.
+ */
+ mode &= ~S_IFDIR;
+
memset(&inarg, 0, sizeof(inarg));
inarg.mode = mode;
inarg.umask = current_umask();
diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c
index 8a77794bbd4a..f361876c5583 100644
--- a/fs/gfs2/inode.c
+++ b/fs/gfs2/inode.c
@@ -963,15 +963,14 @@ fail:
* @dir: The directory in which to create the file
* @dentry: The dentry of the new file
* @mode: The mode of the new file
- * @excl: Force fail if inode exists
*
* Returns: errno
*/
static int gfs2_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
- return gfs2_create_inode(dir, dentry, NULL, S_IFREG | mode, 0, NULL, 0, excl);
+ return gfs2_create_inode(dir, dentry, NULL, S_IFREG | mode, 0, NULL, 0, 1);
}
/**
@@ -1351,7 +1350,7 @@ static struct dentry *gfs2_mkdir(struct mnt_idmap *idmap, struct inode *dir,
{
unsigned dsize = gfs2_max_stuffed_size(GFS2_I(dir));
- return ERR_PTR(gfs2_create_inode(dir, dentry, NULL, S_IFDIR | mode, 0, NULL, dsize, 0));
+ return ERR_PTR(gfs2_create_inode(dir, dentry, NULL, mode, 0, NULL, dsize, 0));
}
/**
diff --git a/fs/hfs/dir.c b/fs/hfs/dir.c
index e13450bb933e..e1f1fb351464 100644
--- a/fs/hfs/dir.c
+++ b/fs/hfs/dir.c
@@ -184,7 +184,7 @@ static int hfs_dir_release(struct inode *inode, struct file *file)
* the directory and the name (and its length) of the new file.
*/
static int hfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode;
int res;
@@ -219,7 +219,7 @@ static struct dentry *hfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct inode *inode;
int res;
- inode = hfs_new_inode(dir, &dentry->d_name, S_IFDIR | mode);
+ inode = hfs_new_inode(dir, &dentry->d_name, mode);
if (IS_ERR(inode))
return ERR_CAST(inode);
diff --git a/fs/hfs/super.c b/fs/hfs/super.c
index a466c401f6bb..ecdafc658928 100644
--- a/fs/hfs/super.c
+++ b/fs/hfs/super.c
@@ -82,7 +82,7 @@ void hfs_mark_mdb_dirty(struct super_block *sb)
spin_lock(&sbi->work_lock);
if (!sbi->work_queued) {
delay = msecs_to_jiffies(dirty_writeback_interval * 10);
- queue_delayed_work(system_long_wq, &sbi->mdb_work, delay);
+ queue_delayed_work(system_dfl_long_wq, &sbi->mdb_work, delay);
sbi->work_queued = 1;
}
spin_unlock(&sbi->work_lock);
diff --git a/fs/hfsplus/dir.c b/fs/hfsplus/dir.c
index 8bf6c7cdd9a8..51fcba2e6d40 100644
--- a/fs/hfsplus/dir.c
+++ b/fs/hfsplus/dir.c
@@ -562,7 +562,7 @@ out:
}
static int hfsplus_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0);
}
@@ -570,7 +570,7 @@ static int hfsplus_create(struct mnt_idmap *idmap, struct inode *dir,
static struct dentry *hfsplus_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode)
{
- return ERR_PTR(hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFDIR, 0));
+ return ERR_PTR(hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0));
}
static int hfsplus_rename(struct mnt_idmap *idmap,
diff --git a/fs/hfsplus/super.c b/fs/hfsplus/super.c
index 5777e31de45a..ff7d6b3336a6 100644
--- a/fs/hfsplus/super.c
+++ b/fs/hfsplus/super.c
@@ -312,7 +312,7 @@ void hfsplus_mark_mdb_dirty(struct super_block *sb)
spin_lock(&sbi->work_lock);
if (!sbi->work_queued) {
delay = msecs_to_jiffies(dirty_writeback_interval * 10);
- queue_delayed_work(system_long_wq, &sbi->sync_work, delay);
+ queue_delayed_work(system_dfl_long_wq, &sbi->sync_work, delay);
sbi->work_queued = 1;
}
spin_unlock(&sbi->work_lock);
diff --git a/fs/hostfs/hostfs_kern.c b/fs/hostfs/hostfs_kern.c
index abe86d72d9ef..7add056d47d8 100644
--- a/fs/hostfs/hostfs_kern.c
+++ b/fs/hostfs/hostfs_kern.c
@@ -593,7 +593,7 @@ static struct inode *hostfs_iget(struct super_block *sb, char *name)
}
static int hostfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode;
char *name;
diff --git a/fs/hpfs/namei.c b/fs/hpfs/namei.c
index 353e13a615f5..9446f4038874 100644
--- a/fs/hpfs/namei.c
+++ b/fs/hpfs/namei.c
@@ -105,10 +105,10 @@ static struct dentry *hpfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (!uid_eq(result->i_uid, current_fsuid()) ||
!gid_eq(result->i_gid, current_fsgid()) ||
- result->i_mode != (mode | S_IFDIR)) {
+ result->i_mode != mode) {
result->i_uid = current_fsuid();
result->i_gid = current_fsgid();
- result->i_mode = mode | S_IFDIR;
+ result->i_mode = mode;
hpfs_write_inode_nolock(result);
}
hpfs_update_directory_times(dir);
@@ -129,7 +129,7 @@ bail:
}
static int hpfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
const unsigned char *name = dentry->d_name.name;
unsigned len = dentry->d_name.len;
diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c
index a578fba8e2fc..7611a8470ea2 100644
--- a/fs/hugetlbfs/inode.c
+++ b/fs/hugetlbfs/inode.c
@@ -971,7 +971,7 @@ static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir
struct dentry *dentry, umode_t mode)
{
int retval = hugetlbfs_mknod(idmap, dir, dentry,
- mode | S_IFDIR, 0);
+ mode, 0);
if (!retval)
inc_nlink(dir);
return ERR_PTR(retval);
@@ -979,7 +979,7 @@ static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir
static int hugetlbfs_create(struct mnt_idmap *idmap,
struct inode *dir, struct dentry *dentry,
- umode_t mode, bool excl)
+ umode_t mode)
{
return hugetlbfs_mknod(idmap, dir, dentry, mode | S_IFREG, 0);
}
diff --git a/fs/inode.c b/fs/inode.c
index 31c5b9ee3a81..a31aa7cb47f6 100644
--- a/fs/inode.c
+++ b/fs/inode.c
@@ -763,21 +763,18 @@ void clear_inode(struct inode *inode)
fsverity_cleanup_inode(inode);
/*
- * We have to cycle the i_pages lock here because reclaim can be in the
- * process of removing the last page (in __filemap_remove_folio())
- * and we must not free the mapping under it.
+ * We have to cycle the i_pages lock here because reclaim
+ * can be in the process of removing the last page (in
+ * __filemap_remove_folio()) and we must not free the mapping
+ * under it. We also remove nodes which are empty; these
+ * can occur in two different ways. The first is that radix
+ * tree expansion can fail partway and the second is that THP
+ * collapse_file() can allocate some temporary nodes and not
+ * clean them up.
*/
- xa_lock_irq(&inode->i_data.i_pages);
+ xa_destroy(&inode->i_data.i_pages);
+
BUG_ON(inode->i_data.nrpages);
- /*
- * Almost always, mapping_empty(&inode->i_data) here; but there are
- * two known and long-standing ways in which nodes may get left behind
- * (when deep radix-tree node allocation failed partway; or when THP
- * collapse_file() failed). Until those two known cases are cleaned up,
- * or a cleanup function is called here, do not BUG_ON(!mapping_empty),
- * nor even WARN_ON(!mapping_empty).
- */
- xa_unlock_irq(&inode->i_data.i_pages);
BUG_ON(!(inode_state_read_once(inode) & I_FREEING));
BUG_ON(inode_state_read_once(inode) & I_CLEAR);
BUG_ON(!list_empty(&inode->i_wb_list));
diff --git a/fs/internal.h b/fs/internal.h
index 355d93f92208..174f06357555 100644
--- a/fs/internal.h
+++ b/fs/internal.h
@@ -137,6 +137,7 @@ extern int reconfigure_super(struct fs_context *);
extern bool super_trylock_shared(struct super_block *sb);
struct super_block *user_get_super(dev_t, bool excl);
void put_super(struct super_block *sb);
+void __init super_dev_init(void);
extern bool mount_capable(struct fs_context *);
/*
diff --git a/fs/iomap/buffered-io.c b/fs/iomap/buffered-io.c
index 6d9a2efd4bee..f669fcdca2f0 100644
--- a/fs/iomap/buffered-io.c
+++ b/fs/iomap/buffered-io.c
@@ -792,7 +792,7 @@ EXPORT_SYMBOL_GPL(iomap_is_partially_uptodate);
*/
struct folio *iomap_get_folio(struct iomap_iter *iter, loff_t pos, size_t len)
{
- fgf_t fgp = FGP_WRITEBEGIN | FGP_NOFS;
+ fgf_t fgp = FGP_WRITEBEGIN;
if (iter->flags & IOMAP_NOWAIT)
fgp |= FGP_NOWAIT;
@@ -1182,7 +1182,6 @@ static bool iomap_write_end(struct iomap_iter *iter, size_t len, size_t copied,
static int iomap_write_iter(struct iomap_iter *iter, struct iov_iter *i,
const struct iomap_write_ops *write_ops)
{
- ssize_t total_written = 0;
int status = 0;
struct address_space *mapping = iter->inode->i_mapping;
size_t chunk = mapping_max_folio_size(mapping);
@@ -1278,12 +1277,11 @@ retry:
goto retry;
}
} else {
- total_written += written;
iomap_iter_advance(iter, written);
}
} while (iov_iter_count(i) && iomap_length(iter));
- return total_written ? 0 : status;
+ return status;
}
ssize_t
diff --git a/fs/iomap/direct-io.c b/fs/iomap/direct-io.c
index e2cd5f92babe..64c6b7286c20 100644
--- a/fs/iomap/direct-io.c
+++ b/fs/iomap/direct-io.c
@@ -10,6 +10,7 @@
#include <linux/iomap.h>
#include <linux/task_io_accounting_ops.h>
#include <linux/fserror.h>
+#include <linux/init.h>
#include "internal.h"
#include "trace.h"
@@ -88,9 +89,9 @@ static inline enum fserror_type iomap_dio_err_type(const struct iomap_dio *dio)
return FSERR_DIRECTIO_READ;
}
-static inline bool should_report_dio_fserror(const struct iomap_dio *dio)
+static inline bool should_report_dio_fserror(int error)
{
- switch (dio->error) {
+ switch (error) {
case 0:
case -EAGAIN:
case -ENOTBLK:
@@ -110,7 +111,7 @@ ssize_t iomap_dio_complete(struct iomap_dio *dio)
if (dops && dops->end_io)
ret = dops->end_io(iocb, dio->size, ret, dio->flags);
- if (should_report_dio_fserror(dio))
+ if (should_report_dio_fserror(dio->error))
fserror_report_io(file_inode(iocb->ki_filp),
iomap_dio_err_type(dio), offset, dio->size,
dio->error, GFP_NOFS);
@@ -403,6 +404,14 @@ out_put_bio:
return ret;
}
+static inline unsigned int iomap_dio_alignment(struct inode *inode,
+ struct block_device *bdev, unsigned int dio_flags)
+{
+ if (dio_flags & IOMAP_DIO_FSBLOCK_ALIGNED)
+ return i_blocksize(inode);
+ return bdev_logical_block_size(bdev);
+}
+
static int iomap_dio_bio_iter(struct iomap_iter *iter, struct iomap_dio *dio)
{
const struct iomap *iomap = &iter->iomap;
@@ -421,10 +430,7 @@ static int iomap_dio_bio_iter(struct iomap_iter *iter, struct iomap_dio *dio)
* File systems that write out of place and always allocate new blocks
* need each bio to be block aligned as that's the unit of allocation.
*/
- if (dio->flags & IOMAP_DIO_FSBLOCK_ALIGNED)
- alignment = fs_block_size;
- else
- alignment = bdev_logical_block_size(iomap->bdev);
+ alignment = iomap_dio_alignment(inode, iomap->bdev, dio->flags);
if ((pos | length) & (alignment - 1))
return -EINVAL;
@@ -893,12 +899,277 @@ out_free_dio:
}
EXPORT_SYMBOL_GPL(__iomap_dio_rw);
+struct iomap_dio_simple {
+ struct kiocb *iocb;
+ size_t size;
+ unsigned int dio_flags;
+ struct work_struct work;
+ /*
+ * Align @bio to a cacheline boundary so that, combined with the
+ * front_pad passed to bioset_init(), the bio sits at the start of
+ * a cacheline in memory returned by the (HWCACHE-aligned) bio
+ * slab. This keeps the hot fields block layer touches on submit
+ * and completion (bi_iter, bi_status, ...) within a single line.
+ */
+ struct bio bio ____cacheline_aligned_in_smp;
+};
+
+static struct bio_set iomap_dio_simple_pool;
+
+static ssize_t iomap_dio_simple_complete(struct iomap_dio_simple *sr)
+{
+ struct bio *bio = &sr->bio;
+ struct kiocb *iocb = sr->iocb;
+ struct inode *inode = file_inode(iocb->ki_filp);
+ ssize_t ret;
+
+ if (unlikely(bio->bi_status)) {
+ ret = blk_status_to_errno(bio->bi_status);
+ if (should_report_dio_fserror(ret))
+ fserror_report_io(inode, FSERR_DIRECTIO_READ,
+ iocb->ki_pos, sr->size, ret,
+ GFP_NOFS);
+ } else {
+ ret = sr->size;
+ iocb->ki_pos += ret;
+ }
+
+ if (sr->dio_flags & IOMAP_DIO_USER_BACKED) {
+ bio_check_pages_dirty(bio);
+ } else {
+ bio_release_pages(bio, false);
+ bio_put(bio);
+ }
+ inode_dio_end(inode);
+ trace_iomap_dio_complete(iocb, ret < 0 ? ret : 0, ret);
+ return ret;
+}
+
+static void iomap_dio_simple_complete_work(struct work_struct *work)
+{
+ struct iomap_dio_simple *sr =
+ container_of(work, struct iomap_dio_simple, work);
+ struct kiocb *iocb = sr->iocb;
+
+ WRITE_ONCE(iocb->private, NULL);
+ iocb->ki_complete(iocb, iomap_dio_simple_complete(sr));
+}
+
+static void iomap_dio_simple_end_io(struct bio *bio)
+{
+ struct iomap_dio_simple *sr =
+ container_of(bio, struct iomap_dio_simple, bio);
+ struct kiocb *iocb = sr->iocb;
+
+ if (unlikely(sr->bio.bi_status)) {
+ struct inode *inode = file_inode(iocb->ki_filp);
+
+ INIT_WORK(&sr->work, iomap_dio_simple_complete_work);
+ queue_work(inode->i_sb->s_dio_done_wq, &sr->work);
+ return;
+ }
+
+ WRITE_ONCE(iocb->private, NULL);
+ iocb->ki_complete(iocb, iomap_dio_simple_complete(sr));
+}
+
+static inline bool
+iomap_dio_simple_supported(struct kiocb *iocb, struct iov_iter *iter,
+ const struct iomap_dio_ops *dops,
+ unsigned int dio_flags, size_t done_before)
+{
+ struct inode *inode = file_inode(iocb->ki_filp);
+ size_t count = iov_iter_count(iter);
+
+ if (dops || done_before)
+ return false;
+ if (iov_iter_rw(iter) != READ)
+ return false;
+ if (!count)
+ return false;
+ /*
+ * Simple dio is an optimization for small IO. Filter out large IO
+ * early as it's the most common case to fail for typical direct IO
+ * workloads.
+ */
+ if (count > inode->i_sb->s_blocksize)
+ return false;
+ if (dio_flags & (IOMAP_DIO_FORCE_WAIT | IOMAP_DIO_PARTIAL |
+ IOMAP_DIO_BOUNCE))
+ return false;
+ if (iocb->ki_pos + count > i_size_read(inode))
+ return false;
+ if (IS_ENCRYPTED(inode))
+ return false;
+
+ return true;
+}
+
+/*
+ * Fast path for small, block-aligned direct I/Os that map to a single
+ * contiguous on-disk extent.
+ *
+ * iomap_dio_simple_supported() enforces the cheap up-front constraints before
+ * entering this path.
+ *
+ * @dops must be NULL: a non-NULL @dops means the caller wants its
+ * ->end_io / ->submit_io hooks invoked, and in particular wants its bios to be
+ * allocated from the filesystem-private @dops->bio_set (whose front_pad sizes a
+ * filesystem-private wrapper around the bio). The fast path instead allocates
+ * from the shared iomap_dio_simple_pool, whose front_pad matches struct
+ * iomap_dio_simple; the two wrappers are not interchangeable, so we must fall
+ * back to __iomap_dio_rw() in that case.
+ *
+ * @done_before must be zero: a non-zero caller-accumulated residual cannot be
+ * carried through a single-bio inline completion.
+ *
+ * @iter must describe a non-empty READ no larger than the inode block size:
+ * writes, zero-length I/O, and larger requests need the generic iomap direct
+ * I/O path.
+ *
+ * @dio_flags must not request IOMAP_DIO_FORCE_WAIT, IOMAP_DIO_PARTIAL, or
+ * IOMAP_DIO_BOUNCE: this path does not support forced waiting, partial direct
+ * I/O, or bouncing. The range must also stay within i_size and encrypted
+ * inodes must use the generic iomap direct I/O path.
+ *
+ * -ENOTBLK is the private sentinel returned by iomap_dio_simple() when it
+ * decides the request does not fit the fast path. In that case we proceed to
+ * the generic __iomap_dio_rw() slow path. Any other errno is a real result and
+ * is propagated as-is, in particular -EAGAIN for IOCB_NOWAIT must reach the
+ * caller.
+ */
+static ssize_t
+iomap_dio_simple(struct kiocb *iocb, struct iov_iter *iter,
+ const struct iomap_ops *ops, void *private,
+ unsigned int dio_flags)
+{
+ struct inode *inode = file_inode(iocb->ki_filp);
+ size_t count = iov_iter_count(iter);
+ bool wait_for_completion = is_sync_kiocb(iocb);
+ struct iomap_iter iomi = {
+ .inode = inode,
+ .pos = iocb->ki_pos,
+ .len = count,
+ .flags = IOMAP_DIRECT,
+ .private = private,
+ };
+ struct iomap_dio_simple *sr;
+ unsigned int alignment;
+ struct bio *bio;
+ ssize_t ret;
+
+ if (iocb->ki_flags & IOCB_NOWAIT)
+ iomi.flags |= IOMAP_NOWAIT;
+
+ ret = kiocb_write_and_wait(iocb, count);
+ if (ret)
+ return ret;
+
+ inode_dio_begin(inode);
+
+ ret = ops->iomap_begin(inode, iomi.pos, count, iomi.flags,
+ &iomi.iomap, &iomi.srcmap);
+ if (ret) {
+ inode_dio_end(inode);
+ return ret;
+ }
+
+ if (iomi.iomap.type != IOMAP_MAPPED ||
+ iomi.iomap.offset + iomi.iomap.length < iomi.pos + count ||
+ (iomi.iomap.flags & IOMAP_F_INTEGRITY)) {
+ ret = -ENOTBLK;
+ goto out_iomap_end;
+ }
+
+ alignment = iomap_dio_alignment(inode, iomi.iomap.bdev, dio_flags);
+ if ((iomi.pos | count) & (alignment - 1)) {
+ ret = -EINVAL;
+ goto out_iomap_end;
+ }
+
+ if (!wait_for_completion && unlikely(!inode->i_sb->s_dio_done_wq)) {
+ ret = sb_init_dio_done_wq(inode->i_sb);
+ if (ret < 0)
+ goto out_iomap_end;
+ }
+
+ trace_iomap_dio_rw_begin(iocb, iter, dio_flags, 0);
+
+ if (user_backed_iter(iter))
+ dio_flags |= IOMAP_DIO_USER_BACKED;
+
+ bio = bio_alloc_bioset(iomi.iomap.bdev,
+ bio_iov_vecs_to_alloc(iter, BIO_MAX_VECS),
+ REQ_OP_READ, GFP_KERNEL, &iomap_dio_simple_pool);
+ sr = container_of(bio, struct iomap_dio_simple, bio);
+ sr->iocb = iocb;
+ sr->dio_flags = dio_flags;
+
+ bio->bi_iter.bi_sector = iomap_sector(&iomi.iomap, iomi.pos);
+ bio->bi_ioprio = iocb->ki_ioprio;
+
+ ret = bio_iov_iter_get_pages(bio, iter, alignment - 1);
+ if (unlikely(ret))
+ goto out_bio_put;
+
+ if (bio->bi_iter.bi_size != count) {
+ iov_iter_revert(iter, bio->bi_iter.bi_size);
+ ret = -ENOTBLK;
+ goto out_bio_release_pages;
+ }
+
+ sr->size = bio->bi_iter.bi_size;
+
+ if (dio_flags & IOMAP_DIO_USER_BACKED)
+ bio_set_pages_dirty(bio);
+
+ if (iocb->ki_flags & IOCB_NOWAIT)
+ bio->bi_opf |= REQ_NOWAIT;
+ if ((iocb->ki_flags & IOCB_HIPRI) && !wait_for_completion) {
+ bio->bi_opf |= REQ_POLLED;
+ WRITE_ONCE(iocb->private, bio);
+ }
+
+ if (ops->iomap_end)
+ ops->iomap_end(inode, iomi.pos, count, count, iomi.flags,
+ &iomi.iomap);
+
+ if (!wait_for_completion) {
+ bio->bi_end_io = iomap_dio_simple_end_io;
+ submit_bio(bio);
+ trace_iomap_dio_rw_queued(inode, iomi.pos, count);
+ return -EIOCBQUEUED;
+ }
+
+ submit_bio_wait(bio);
+ return iomap_dio_simple_complete(sr);
+
+out_bio_release_pages:
+ bio_release_pages(bio, false);
+out_bio_put:
+ bio_put(bio);
+out_iomap_end:
+ if (ops->iomap_end)
+ ops->iomap_end(inode, iomi.pos, count, 0, iomi.flags,
+ &iomi.iomap);
+ inode_dio_end(inode);
+ return ret;
+}
+
ssize_t
iomap_dio_rw(struct kiocb *iocb, struct iov_iter *iter,
const struct iomap_ops *ops, const struct iomap_dio_ops *dops,
unsigned int dio_flags, void *private, size_t done_before)
{
struct iomap_dio *dio;
+ ssize_t ret;
+
+ if (iomap_dio_simple_supported(iocb, iter, dops, dio_flags,
+ done_before)) {
+ ret = iomap_dio_simple(iocb, iter, ops, private, dio_flags);
+ if (ret != -ENOTBLK)
+ return ret;
+ }
dio = __iomap_dio_rw(iocb, iter, ops, dops, dio_flags, private,
done_before);
@@ -907,3 +1178,11 @@ iomap_dio_rw(struct kiocb *iocb, struct iov_iter *iter,
return iomap_dio_complete(dio);
}
EXPORT_SYMBOL_GPL(iomap_dio_rw);
+
+static int __init iomap_dio_init(void)
+{
+ return bioset_init(&iomap_dio_simple_pool, 4,
+ offsetof(struct iomap_dio_simple, bio),
+ BIOSET_NEED_BVECS | BIOSET_PERCPU_CACHE);
+}
+fs_initcall(iomap_dio_init);
diff --git a/fs/iomap/ioend.c b/fs/iomap/ioend.c
index 30468d51b5ad..fb636dce43af 100644
--- a/fs/iomap/ioend.c
+++ b/fs/iomap/ioend.c
@@ -13,6 +13,7 @@
struct bio_set iomap_ioend_bioset;
EXPORT_SYMBOL_GPL(iomap_ioend_bioset);
+static struct bio_set iomap_ioend_split_bioset;
struct iomap_ioend *iomap_init_ioend(struct inode *inode,
struct bio *bio, loff_t file_offset, u16 ioend_flags)
@@ -488,7 +489,8 @@ struct iomap_ioend *iomap_split_ioend(struct iomap_ioend *ioend,
sector_offset = ALIGN_DOWN(sector_offset << SECTOR_SHIFT,
i_blocksize(ioend->io_inode)) >> SECTOR_SHIFT;
- split = bio_split(bio, sector_offset, GFP_NOFS, &iomap_ioend_bioset);
+ split = bio_split(bio, sector_offset, GFP_NOFS,
+ &iomap_ioend_split_bioset);
if (IS_ERR(split))
return ERR_CAST(split);
split->bi_private = bio->bi_private;
@@ -511,8 +513,23 @@ EXPORT_SYMBOL_GPL(iomap_split_ioend);
static int __init iomap_ioend_init(void)
{
- return bioset_init(&iomap_ioend_bioset, 4 * (PAGE_SIZE / SECTOR_SIZE),
+ const unsigned int nr_mempool_entries = 4 * (PAGE_SIZE / SECTOR_SIZE);
+ int error;
+
+ error = bioset_init(&iomap_ioend_bioset, nr_mempool_entries,
offsetof(struct iomap_ioend, io_bio),
BIOSET_NEED_BVECS);
+ if (error)
+ return error;
+ error = bioset_init(&iomap_ioend_split_bioset, nr_mempool_entries,
+ offsetof(struct iomap_ioend, io_bio),
+ BIOSET_NEED_BVECS);
+ if (error)
+ goto out_exit_ioend_bioset;
+ return 0;
+
+out_exit_ioend_bioset:
+ bioset_exit(&iomap_ioend_bioset);
+ return error;
}
fs_initcall(iomap_ioend_init);
diff --git a/fs/isofs/compress.c b/fs/isofs/compress.c
index 397568b9c7e7..3fda92358e22 100644
--- a/fs/isofs/compress.c
+++ b/fs/isofs/compress.c
@@ -65,12 +65,14 @@ static loff_t zisofs_uncompress_block(struct inode *inode, loff_t block_start,
/* Empty block? */
if (block_size == 0) {
for ( i = 0 ; i < pcount ; i++ ) {
+ unsigned int off = i ? 0 : poffset;
+
if (!pages[i])
continue;
- memzero_page(pages[i], 0, PAGE_SIZE);
+ memzero_page(pages[i], off, PAGE_SIZE - off);
SetPageUptodate(pages[i]);
}
- return ((loff_t)pcount) << PAGE_SHIFT;
+ return (((loff_t)pcount) << PAGE_SHIFT) - poffset;
}
/* Because zlib is not thread-safe, do all the I/O at the top. */
diff --git a/fs/jbd2/checkpoint.c b/fs/jbd2/checkpoint.c
index 1508e2f54462..513273712010 100644
--- a/fs/jbd2/checkpoint.c
+++ b/fs/jbd2/checkpoint.c
@@ -358,15 +358,16 @@ int jbd2_cleanup_journal_tail(journal_t *journal)
/*
* journal_shrink_one_cp_list
*
- * Find all the written-back checkpoint buffers in the given list
- * and try to release them. If the whole transaction is released, set
- * the 'released' parameter. Return the number of released checkpointed
- * buffers.
+ * Find written-back checkpoint buffers in the given list and try to release
+ * them. If 'nr_to_scan' is set, scan at most that many buffers. If the whole
+ * transaction is released, set the 'released' parameter. Return the number of
+ * released checkpointed buffers.
*
* Called with j_list_lock held.
*/
static unsigned long journal_shrink_one_cp_list(struct journal_head *jh,
enum jbd2_shrink_type type,
+ unsigned long *nr_to_scan,
bool *released)
{
struct journal_head *last_jh;
@@ -375,13 +376,15 @@ static unsigned long journal_shrink_one_cp_list(struct journal_head *jh,
int ret;
*released = false;
- if (!jh)
+ if (!jh || (nr_to_scan && !*nr_to_scan))
return 0;
last_jh = jh->b_cpprev;
do {
jh = next_jh;
next_jh = jh->b_cpnext;
+ if (nr_to_scan)
+ (*nr_to_scan)--;
if (type == JBD2_SHRINK_DESTROY) {
ret = __jbd2_journal_remove_checkpoint(jh);
@@ -389,7 +392,7 @@ static unsigned long journal_shrink_one_cp_list(struct journal_head *jh,
ret = jbd2_journal_try_remove_checkpoint(jh);
if (ret < 0) {
if (type == JBD2_SHRINK_BUSY_SKIP)
- continue;
+ goto next;
break;
}
}
@@ -400,9 +403,10 @@ static unsigned long journal_shrink_one_cp_list(struct journal_head *jh,
break;
}
+next:
if (need_resched())
break;
- } while (jh != last_jh);
+ } while (jh != last_jh && (!nr_to_scan || *nr_to_scan));
return nr_freed;
}
@@ -424,7 +428,6 @@ unsigned long jbd2_journal_shrink_checkpoint_list(journal_t *journal,
tid_t first_tid = 0, last_tid = 0, next_tid = 0;
tid_t tid = 0;
unsigned long nr_freed = 0;
- unsigned long freed;
bool first_set = false;
again:
@@ -457,10 +460,9 @@ again:
next_transaction = transaction->t_cpnext;
tid = transaction->t_tid;
- freed = journal_shrink_one_cp_list(transaction->t_checkpoint_list,
- JBD2_SHRINK_BUSY_SKIP, &released);
- nr_freed += freed;
- (*nr_to_scan) -= min(*nr_to_scan, freed);
+ nr_freed += journal_shrink_one_cp_list(transaction->t_checkpoint_list,
+ JBD2_SHRINK_BUSY_SKIP,
+ nr_to_scan, &released);
if (*nr_to_scan == 0)
break;
if (need_resched() || spin_needbreak(&journal->j_list_lock))
@@ -516,7 +518,7 @@ void __jbd2_journal_clean_checkpoint_list(journal_t *journal,
transaction = next_transaction;
next_transaction = transaction->t_cpnext;
journal_shrink_one_cp_list(transaction->t_checkpoint_list,
- type, &released);
+ type, NULL, &released);
/*
* This function only frees up some memory if possible so we
* dont have an obligation to finish processing. Bail out if
diff --git a/fs/jbd2/journal.c b/fs/jbd2/journal.c
index 09efa337649e..00f5a98f3d4f 100644
--- a/fs/jbd2/journal.c
+++ b/fs/jbd2/journal.c
@@ -94,6 +94,7 @@ EXPORT_SYMBOL(jbd2_journal_init_jbd_inode);
EXPORT_SYMBOL(jbd2_journal_release_jbd_inode);
EXPORT_SYMBOL(jbd2_journal_begin_ordered_truncate);
EXPORT_SYMBOL(jbd2_inode_cache);
+EXPORT_SYMBOL(jbd2_handle_cache);
#ifdef CONFIG_JBD2_DEBUG
void __jbd2_debug(int level, const char *file, const char *func,
diff --git a/fs/jffs2/dir.c b/fs/jffs2/dir.c
index c4088c3b4ac0..656c920864c5 100644
--- a/fs/jffs2/dir.c
+++ b/fs/jffs2/dir.c
@@ -26,7 +26,7 @@
static int jffs2_readdir (struct file *, struct dir_context *);
static int jffs2_create (struct mnt_idmap *, struct inode *,
- struct dentry *, umode_t, bool);
+ struct dentry *, umode_t);
static struct dentry *jffs2_lookup (struct inode *,struct dentry *,
unsigned int);
static int jffs2_link (struct dentry *,struct inode *,struct dentry *);
@@ -163,7 +163,7 @@ static int jffs2_readdir(struct file *file, struct dir_context *ctx)
static int jffs2_create(struct mnt_idmap *idmap, struct inode *dir_i,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct jffs2_raw_inode *ri;
struct jffs2_inode_info *f, *dir_f;
@@ -462,8 +462,6 @@ static struct dentry *jffs2_mkdir (struct mnt_idmap *idmap, struct inode *dir_i,
uint32_t alloclen;
int ret;
- mode |= S_IFDIR;
-
ri = jffs2_alloc_raw_inode();
if (!ri)
return ERR_PTR(-ENOMEM);
diff --git a/fs/jffs2/wbuf.c b/fs/jffs2/wbuf.c
index 8ff7a0b6add2..3b7803c75d58 100644
--- a/fs/jffs2/wbuf.c
+++ b/fs/jffs2/wbuf.c
@@ -1177,7 +1177,7 @@ void jffs2_dirty_trigger(struct jffs2_sb_info *c)
return;
delay = msecs_to_jiffies(dirty_writeback_interval * 10);
- if (queue_delayed_work(system_long_wq, &c->wbuf_dwork, delay))
+ if (queue_delayed_work(system_dfl_long_wq, &c->wbuf_dwork, delay))
jffs2_dbg(1, "%s()\n", __func__);
}
diff --git a/fs/jfs/namei.c b/fs/jfs/namei.c
index 442d62679262..8a36c218f0f7 100644
--- a/fs/jfs/namei.c
+++ b/fs/jfs/namei.c
@@ -61,7 +61,7 @@ static inline void free_ea_wmap(struct inode *inode)
*
*/
static int jfs_create(struct mnt_idmap *idmap, struct inode *dip,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
int rc = 0;
tid_t tid; /* transaction id */
@@ -223,7 +223,7 @@ static struct dentry *jfs_mkdir(struct mnt_idmap *idmap, struct inode *dip,
* block there while holding dtree page, so we allocate the inode &
* begin the transaction before we search the directory.
*/
- ip = ialloc(dip, S_IFDIR | mode);
+ ip = ialloc(dip, mode);
if (IS_ERR(ip)) {
rc = PTR_ERR(ip);
goto out2;
diff --git a/fs/kernel_read_file.c b/fs/kernel_read_file.c
index de32c95d823d..9c2ba9240083 100644
--- a/fs/kernel_read_file.c
+++ b/fs/kernel_read_file.c
@@ -150,18 +150,13 @@ ssize_t kernel_read_file_from_path_initns(const char *path, loff_t offset,
enum kernel_read_file_id id)
{
struct file *file;
- struct path root;
ssize_t ret;
if (!path || !*path)
return -EINVAL;
- task_lock(&init_task);
- get_fs_root(init_task.fs, &root);
- task_unlock(&init_task);
-
- file = file_open_root(&root, path, O_RDONLY, 0);
- path_put(&root);
+ scoped_with_init_fs()
+ file = filp_open(path, O_RDONLY, 0);
if (IS_ERR(file))
return PTR_ERR(file);
diff --git a/fs/lockd/lockd.h b/fs/lockd/lockd.h
index e418a50c4180..14cc952fe81a 100644
--- a/fs/lockd/lockd.h
+++ b/fs/lockd/lockd.h
@@ -314,7 +314,7 @@ void nsm_release(struct nsm_handle *nsm);
* This is used in garbage collection and resource reclaim
* A return value != 0 means destroy the lock/block/share
*/
-typedef int (*nlm_host_match_fn_t)(void *cur, struct nlm_host *ref);
+typedef int (*nlm_host_match_fn_t)(void *owner, struct nlm_host *ref);
/*
* Server-side lock handling
diff --git a/fs/lockd/svc.c b/fs/lockd/svc.c
index 490551369ef2..ee90e743064a 100644
--- a/fs/lockd/svc.c
+++ b/fs/lockd/svc.c
@@ -47,7 +47,7 @@
static struct svc_program nlmsvc_program;
-const struct nlmsvc_binding *nlmsvc_ops;
+const struct nlmsvc_binding __rcu *nlmsvc_ops;
EXPORT_SYMBOL_GPL(nlmsvc_ops);
static DEFINE_MUTEX(nlmsvc_mutex);
@@ -142,7 +142,7 @@ lockd(void *vrqstp)
nlmsvc_retry_blocked(rqstp);
svc_recv(rqstp, 0);
}
- if (nlmsvc_ops)
+ if (rcu_access_pointer(nlmsvc_ops))
nlmsvc_invalidate_all();
nlm_shutdown_hosts();
cancel_delayed_work_sync(&ln->grace_period_end);
diff --git a/fs/lockd/svc4proc.c b/fs/lockd/svc4proc.c
index 78e675470c4b..b73004a7987e 100644
--- a/fs/lockd/svc4proc.c
+++ b/fs/lockd/svc4proc.c
@@ -128,7 +128,7 @@ nlm4svc_lookup_host(struct svc_rqst *rqstp, string caller, bool monitored)
{
struct nlm_host *host;
- if (!nlmsvc_ops)
+ if (!rcu_access_pointer(nlmsvc_ops))
return NULL;
host = nlmsvc_lookup_host(rqstp, caller.data, caller.len);
if (!host)
@@ -872,7 +872,8 @@ static __be32 nlm4svc_proc_granted_msg(struct svc_rqst *rqstp)
struct nlm4_testargs_wrapper *argp = rqstp->rq_argp;
struct nlm_host *host;
- host = nlm4svc_lookup_host(rqstp, argp->xdrgen.alock.caller_name, false);
+ host = nlmsvc_lookup_host(rqstp, argp->xdrgen.alock.caller_name.data,
+ argp->xdrgen.alock.caller_name.len);
if (!host)
return rpc_system_err;
@@ -894,7 +895,7 @@ static __be32 nlm4svc_proc_granted_res(struct svc_rqst *rqstp)
{
struct nlm4_res_wrapper *argp = rqstp->rq_argp;
- if (!nlmsvc_ops)
+ if (!rcu_access_pointer(nlmsvc_ops))
return rpc_success;
if (nlm4_netobj_to_cookie(&argp->cookie, &argp->xdrgen.cookie))
diff --git a/fs/lockd/svclock.c b/fs/lockd/svclock.c
index e48d31f14a65..e628b5d35507 100644
--- a/fs/lockd/svclock.c
+++ b/fs/lockd/svclock.c
@@ -47,40 +47,6 @@ static const struct rpc_call_ops nlmsvc_grant_ops;
static LIST_HEAD(nlm_blocked);
static DEFINE_SPINLOCK(nlm_blocked_lock);
-#if IS_ENABLED(CONFIG_SUNRPC_DEBUG)
-static const char *nlmdbg_cookie2a(const struct lockd_cookie *cookie)
-{
- /*
- * We can get away with a static buffer because this is only called
- * from lockd, which is single-threaded.
- */
- static char buf[2*NLM_MAXCOOKIELEN+1];
- unsigned int i, len = sizeof(buf);
- char *p = buf;
-
- len--; /* allow for trailing \0 */
- if (len < 3)
- return "???";
- for (i = 0 ; i < cookie->len ; i++) {
- if (len < 2) {
- strcpy(p-3, "...");
- break;
- }
- sprintf(p, "%02x", cookie->data[i]);
- p += 2;
- len -= 2;
- }
- *p = '\0';
-
- return buf;
-}
-#else
-static inline const char *nlmdbg_cookie2a(const struct lockd_cookie *cookie)
-{
- return "???";
-}
-#endif
-
/*
* Insert a blocked lock into the global list
*/
@@ -155,11 +121,12 @@ nlmsvc_lookup_block(struct nlm_file *file, struct lockd_lock *lock)
spin_lock(&nlm_blocked_lock);
list_for_each_entry(block, &nlm_blocked, b_list) {
fl = &block->b_call->a_args.lock.fl;
- dprintk("lockd: check f=%p pd=%d %Ld-%Ld ty=%d cookie=%s\n",
+ dprintk("lockd: check f=%p pd=%d %Ld-%Ld ty=%d cookie=%*phN\n",
block->b_file, fl->c.flc_pid,
(long long)fl->fl_start,
(long long)fl->fl_end, fl->c.flc_type,
- nlmdbg_cookie2a(&block->b_call->a_args.cookie));
+ block->b_call->a_args.cookie.len,
+ block->b_call->a_args.cookie.data);
if (block->b_file == file && nlm_compare_locks(fl, &lock->fl)) {
kref_get(&block->b_count);
spin_unlock(&nlm_blocked_lock);
@@ -198,7 +165,8 @@ nlmsvc_find_block(struct lockd_cookie *cookie)
return NULL;
found:
- dprintk("nlmsvc_find_block(%s): block=%p\n", nlmdbg_cookie2a(cookie), block);
+ dprintk("nlmsvc_find_block(%*phN): block=%p\n",
+ cookie->len, cookie->data, block);
kref_get(&block->b_count);
spin_unlock(&nlm_blocked_lock);
return block;
diff --git a/fs/lockd/svcproc.c b/fs/lockd/svcproc.c
index 386a881b520f..d410b8c69893 100644
--- a/fs/lockd/svcproc.c
+++ b/fs/lockd/svcproc.c
@@ -133,7 +133,7 @@ nlm3svc_lookup_host(struct svc_rqst *rqstp, string caller, bool monitored)
{
struct nlm_host *host;
- if (!nlmsvc_ops)
+ if (!rcu_access_pointer(nlmsvc_ops))
return NULL;
host = nlmsvc_lookup_host(rqstp, caller.data, caller.len);
if (!host)
@@ -924,7 +924,7 @@ static __be32 nlmsvc_proc_granted_res(struct svc_rqst *rqstp)
{
struct nlm_res_wrapper *argp = rqstp->rq_argp;
- if (!nlmsvc_ops)
+ if (!rcu_access_pointer(nlmsvc_ops))
return rpc_success;
if (nlm_netobj_to_cookie(&argp->cookie, &argp->xdrgen.cookie))
diff --git a/fs/lockd/svcsubs.c b/fs/lockd/svcsubs.c
index a0d1a6fbf61e..ffab76278bb2 100644
--- a/fs/lockd/svcsubs.c
+++ b/fs/lockd/svcsubs.c
@@ -90,22 +90,35 @@ int lock_to_openmode(struct file_lock *lock)
static __be32 nlm_do_fopen(struct svc_rqst *rqstp,
struct nlm_file *file, int mode)
{
+ const struct nlmsvc_binding *ops;
__be32 nlmerr = nlm__int__failed;
__be32 deferred = 0;
int error;
int m;
+ rcu_read_lock();
+ ops = rcu_dereference(nlmsvc_ops);
+ if (!ops || !try_module_get(ops->owner)) {
+ rcu_read_unlock();
+ return nlm__int__failed;
+ }
+ rcu_read_unlock();
+
for (m = O_RDONLY; m <= O_WRONLY; m++) {
struct file **fp = &file->f_file[m];
if (mode != O_RDWR && mode != m)
continue;
- if (*fp)
+ if (*fp) {
+ module_put(ops->owner);
return nlm_granted;
+ }
- error = nlmsvc_ops->fopen(rqstp, &file->f_handle, fp, m);
- if (!error)
+ error = ops->fopen(rqstp, &file->f_handle, fp, m);
+ if (!error) {
+ module_put(ops->owner);
return nlm_granted;
+ }
dprintk("lockd: open failed (errno %d)\n", error);
switch (error) {
@@ -122,6 +135,7 @@ static __be32 nlm_do_fopen(struct svc_rqst *rqstp,
}
}
+ module_put(ops->owner);
return deferred ? deferred : nlmerr;
}
@@ -186,6 +200,33 @@ out_free:
}
/*
+ * Release the struct file references held by a nlm_file.
+ */
+static void nlm_release_files(struct nlm_file *file)
+{
+ const struct nlmsvc_binding *ops;
+ bool have_ops;
+
+ rcu_read_lock();
+ ops = rcu_dereference(nlmsvc_ops);
+ have_ops = ops && try_module_get(ops->owner);
+ rcu_read_unlock();
+
+ if (have_ops) {
+ if (file->f_file[O_RDONLY])
+ ops->fclose(file->f_file[O_RDONLY]);
+ if (file->f_file[O_WRONLY])
+ ops->fclose(file->f_file[O_WRONLY]);
+ module_put(ops->owner);
+ } else {
+ if (file->f_file[O_RDONLY])
+ fput(file->f_file[O_RDONLY]);
+ if (file->f_file[O_WRONLY])
+ fput(file->f_file[O_WRONLY]);
+ }
+}
+
+/*
* Delete a file after having released all locks, blocks and shares
*/
static inline void
@@ -194,10 +235,7 @@ nlm_delete_file(struct nlm_file *file)
nlm_debug_print_file("closing file", file);
if (!hlist_unhashed(&file->f_list)) {
hlist_del(&file->f_list);
- if (file->f_file[O_RDONLY])
- nlmsvc_ops->fclose(file->f_file[O_RDONLY]);
- if (file->f_file[O_WRONLY])
- nlmsvc_ops->fclose(file->f_file[O_WRONLY]);
+ nlm_release_files(file);
kfree(file);
} else {
printk(KERN_WARNING "lockd: attempt to release unknown file!\n");
@@ -312,12 +350,10 @@ nlm_file_inuse(struct nlm_file *file)
return 0;
}
-static void nlm_close_files(struct nlm_file *file)
+static void nlm_file_release(struct nlm_file *file)
{
- if (file->f_file[O_RDONLY])
- nlmsvc_ops->fclose(file->f_file[O_RDONLY]);
- if (file->f_file[O_WRONLY])
- nlmsvc_ops->fclose(file->f_file[O_WRONLY]);
+ if (!nlm_file_inuse(file))
+ nlm_delete_file(file);
}
/*
@@ -327,32 +363,41 @@ static int
nlm_traverse_files(void *data, nlm_host_match_fn_t match,
int (*is_failover_file)(void *data, struct nlm_file *file))
{
- struct hlist_node *next;
- struct nlm_file *file;
+ struct nlm_file *file, *next;
int i, ret = 0;
mutex_lock(&nlm_file_mutex);
for (i = 0; i < FILE_NRHASH; i++) {
- hlist_for_each_entry_safe(file, next, &nlm_files[i], f_list) {
- if (is_failover_file && !is_failover_file(data, file))
- continue;
+ file = hlist_entry_safe(nlm_files[i].first,
+ struct nlm_file, f_list);
+ if (file)
file->f_count++;
- mutex_unlock(&nlm_file_mutex);
-
- /* Traverse locks, blocks and shares of this file
- * and update file->f_locks count */
- if (nlm_inspect_file(data, file, match))
- ret = 1;
+ while (file) {
+ /*
+ * Pin the next neighbour before we drop the mutex
+ * for nlm_inspect_file(); a concurrent
+ * nlm_release_file() under the same mutex would
+ * otherwise be free to unlink and kfree it during
+ * the unlock window, leaving us to dereference a
+ * freed slab when we walked to next afterwards.
+ */
+ next = hlist_entry_safe(file->f_list.next,
+ struct nlm_file, f_list);
+ if (next)
+ next->f_count++;
+
+ if (!is_failover_file || is_failover_file(data, file)) {
+ mutex_unlock(&nlm_file_mutex);
+
+ if (nlm_inspect_file(data, file, match))
+ ret = 1;
+
+ mutex_lock(&nlm_file_mutex);
+ }
- mutex_lock(&nlm_file_mutex);
file->f_count--;
- /* No more references to this file. Let go of it. */
- if (list_empty(&file->f_blocks) && !file->f_locks
- && !file->f_shares && !file->f_count) {
- hlist_del(&file->f_list);
- nlm_close_files(file);
- kfree(file);
- }
+ nlm_file_release(file);
+ file = next;
}
}
mutex_unlock(&nlm_file_mutex);
@@ -512,7 +557,7 @@ EXPORT_SYMBOL_GPL(nlmsvc_unlock_all_by_sb);
static int
nlmsvc_match_ip(void *datap, struct nlm_host *host)
{
- return rpc_cmp_addr(nlm_srcaddr(host), datap);
+ return rpc_cmp_addr(nlm_srcaddr(datap), (struct sockaddr *)host);
}
/**
diff --git a/fs/minix/namei.c b/fs/minix/namei.c
index 263e4ba8b1c8..5525ba367ed7 100644
--- a/fs/minix/namei.c
+++ b/fs/minix/namei.c
@@ -64,7 +64,7 @@ static int minix_tmpfile(struct mnt_idmap *idmap, struct inode *dir,
}
static int minix_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return minix_mknod(&nop_mnt_idmap, dir, dentry, mode, 0);
}
@@ -110,7 +110,7 @@ static struct dentry *minix_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct inode * inode;
int err;
- inode = minix_new_inode(dir, S_IFDIR | mode);
+ inode = minix_new_inode(dir, mode);
if (IS_ERR(inode))
return ERR_CAST(inode);
diff --git a/fs/namei.c b/fs/namei.c
index 19ce43c9a6e6..e4ef48583a3b 100644
--- a/fs/namei.c
+++ b/fs/namei.c
@@ -4140,11 +4140,6 @@ EXPORT_SYMBOL(end_renaming);
* after setgid stripping allows the same ordering for both non-POSIX ACL and
* POSIX ACL supporting filesystems.
*
- * Note that it's currently valid for @type to be 0 if a directory is created.
- * Filesystems raise that flag individually and we need to check whether each
- * filesystem can deal with receiving S_IFDIR from the vfs before we enforce a
- * non-zero type.
- *
* Returns: mode to be passed to the filesystem
*/
static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap,
@@ -4190,7 +4185,7 @@ int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode,
return error;
if (!dir->i_op->create)
- return -EACCES; /* shouldn't it be ENOSYS? */
+ return -EOPNOTSUPP;
mode = vfs_prepare_mode(idmap, dir, mode, S_IALLUGO, S_IFREG);
error = security_inode_create(dir, dentry, mode);
@@ -4199,7 +4194,7 @@ int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode,
error = try_break_deleg(dir, LEASE_BREAK_DIR_CREATE, di);
if (error)
return error;
- error = dir->i_op->create(idmap, dir, dentry, mode, true);
+ error = dir->i_op->create(idmap, dir, dentry, mode);
if (!error)
fsnotify_create(dir, dentry);
return error;
@@ -4501,12 +4496,11 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file,
file->f_mode |= FMODE_CREATED;
audit_inode_child(dir_inode, dentry, AUDIT_TYPE_CHILD_CREATE);
if (!dir_inode->i_op->create) {
- error = -EACCES;
+ error = -EOPNOTSUPP;
goto out_dput;
}
- error = dir_inode->i_op->create(idmap, dir_inode, dentry,
- mode, open_flag & O_EXCL);
+ error = dir_inode->i_op->create(idmap, dir_inode, dentry, mode);
if (error)
goto out_dput;
}
@@ -5117,7 +5111,7 @@ int vfs_mknod(struct mnt_idmap *idmap, struct inode *dir,
return -EPERM;
if (!dir->i_op->mknod)
- return -EPERM;
+ return -EOPNOTSUPP;
mode = vfs_prepare_mode(idmap, dir, mode, mode, mode);
error = devcgroup_inode_mknod(mode, dev);
@@ -5256,11 +5250,11 @@ struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (error)
goto err;
- error = -EPERM;
+ error = -EOPNOTSUPP;
if (!dir->i_op->mkdir)
goto err;
- mode = vfs_prepare_mode(idmap, dir, mode, S_IRWXUGO | S_ISVTX, 0);
+ mode = vfs_prepare_mode(idmap, dir, mode, S_IRWXUGO | S_ISVTX, S_IFDIR);
error = security_inode_mkdir(dir, dentry, mode);
if (error)
goto err;
@@ -5360,7 +5354,7 @@ int vfs_rmdir(struct mnt_idmap *idmap, struct inode *dir,
return error;
if (!dir->i_op->rmdir)
- return -EPERM;
+ return -EOPNOTSUPP;
dget(dentry);
inode_lock(dentry->d_inode);
@@ -5496,7 +5490,7 @@ int vfs_unlink(struct mnt_idmap *idmap, struct inode *dir,
return error;
if (!dir->i_op->unlink)
- return -EPERM;
+ return -EOPNOTSUPP;
inode_lock(target);
if (IS_SWAPFILE(target))
@@ -5647,7 +5641,7 @@ int vfs_symlink(struct mnt_idmap *idmap, struct inode *dir,
return error;
if (!dir->i_op->symlink)
- return -EPERM;
+ return -EOPNOTSUPP;
error = security_inode_symlink(dir, dentry, oldname);
if (error)
@@ -5769,7 +5763,7 @@ int vfs_link(struct dentry *old_dentry, struct mnt_idmap *idmap,
if (HAS_UNMAPPED_ID(idmap, inode))
return -EPERM;
if (!dir->i_op->link)
- return -EPERM;
+ return -EOPNOTSUPP;
if (S_ISDIR(inode->i_mode))
return -EPERM;
@@ -5978,7 +5972,7 @@ int vfs_rename(struct renamedata *rd)
return error;
if (!old_dir->i_op->rename)
- return -EPERM;
+ return -EOPNOTSUPP;
/*
* If we are going to change the parent - check write permissions,
diff --git a/fs/namespace.c b/fs/namespace.c
index 3d5cd5bf3b05..6b81b77f9303 100644
--- a/fs/namespace.c
+++ b/fs/namespace.c
@@ -2908,6 +2908,9 @@ static int do_change_type(const struct path *path, int ms_flags)
for (m = mnt; m; m = (recurse ? next_mnt(m, mnt) : NULL))
change_mnt_propagation(m, type);
+ guard(mount_locked_reader)();
+ touch_mnt_namespace(mnt->mnt_ns);
+
return 0;
}
@@ -3481,6 +3484,10 @@ static int do_set_group(const struct path *from_path, const struct path *to_path
list_add(&to->mnt_share, &from->mnt_share);
set_mnt_shared(to);
}
+
+ guard(mount_locked_reader)();
+ touch_mnt_namespace(to->mnt_ns);
+
return 0;
}
@@ -6184,12 +6191,14 @@ static void __init init_mount_tree(void)
struct path root;
/*
- * We create two mounts:
+ * We create three mounts:
*
* (1) nullfs with mount id 1
* (2) mutable rootfs with mount id 2
+ * (3) private nullfs for kthreads (SB_KERNMOUNT)
*
- * with (2) mounted on top of (1).
+ * with (2) mounted on top of (1). The init_task's root and pwd
+ * are pointed at (3) so all kthreads start isolated in nullfs.
*/
nullfs_mnt = vfs_kern_mount(&nullfs_fs_type, 0, "nullfs", NULL);
if (IS_ERR(nullfs_mnt))
@@ -6229,12 +6238,14 @@ static void __init init_mount_tree(void)
init_mnt_ns.nr_mounts++;
}
+ nullfs_mnt = kern_mount(&nullfs_fs_type);
+ if (IS_ERR(nullfs_mnt))
+ panic("VFS: Failed to create private nullfs instance");
+ root.mnt = nullfs_mnt;
+ root.dentry = nullfs_mnt->mnt_root;
+
init_task.nsproxy->mnt_ns = &init_mnt_ns;
get_mnt_ns(&init_mnt_ns);
-
- /* The root and pwd always point to the mutable rootfs. */
- root.mnt = mnt;
- root.dentry = mnt->mnt_root;
set_fs_pwd(current->fs, &root);
set_fs_root(current->fs, &root);
@@ -6262,6 +6273,8 @@ void __init mnt_init(void)
if (!mount_hashtable || !mountpoint_hashtable)
panic("Failed to allocate mount hash table\n");
+ super_dev_init();
+
kernfs_init();
err = sysfs_init();
diff --git a/fs/nfs/blocklayout/dev.c b/fs/nfs/blocklayout/dev.c
index bb35f88501ce..368d20daf67b 100644
--- a/fs/nfs/blocklayout/dev.c
+++ b/fs/nfs/blocklayout/dev.c
@@ -4,6 +4,7 @@
*/
#include <linux/sunrpc/svc.h>
#include <linux/blkdev.h>
+#include <linux/fs_struct.h>
#include <linux/nfs4.h>
#include <linux/nfs_fs.h>
#include <linux/nfs_xdr.h>
@@ -363,15 +364,22 @@ static struct file *
bl_open_path(struct pnfs_block_volume *v, const char *prefix)
{
struct file *bdev_file;
- const char *devname;
+ const char *devname __free(kfree) = NULL;
devname = kasprintf(GFP_KERNEL, "/dev/disk/by-id/%s%*phN",
prefix, v->scsi.designator_len, v->scsi.designator);
if (!devname)
return ERR_PTR(-ENOMEM);
- bdev_file = bdev_file_open_by_path(devname,
- BLK_OPEN_READ | BLK_OPEN_WRITE, NULL, NULL);
+ if (tsk_is_kthread(current)) {
+ scoped_with_init_fs()
+ bdev_file = bdev_file_open_by_path(devname,
+ BLK_OPEN_READ | BLK_OPEN_WRITE,
+ NULL, NULL);
+ } else {
+ bdev_file = bdev_file_open_by_path(devname,
+ BLK_OPEN_READ | BLK_OPEN_WRITE, NULL, NULL);
+ }
if (IS_ERR(bdev_file)) {
dprintk("failed to open device %s (%ld)\n",
devname, PTR_ERR(bdev_file));
@@ -380,7 +388,6 @@ bl_open_path(struct pnfs_block_volume *v, const char *prefix)
file_bdev(bdev_file)->bd_disk->disk_name);
}
- kfree(devname);
return bdev_file;
}
diff --git a/fs/nfs/callback.c b/fs/nfs/callback.c
index ff4e9fd38e83..bc282b744f34 100644
--- a/fs/nfs/callback.c
+++ b/fs/nfs/callback.c
@@ -231,8 +231,9 @@ int nfs_callback_up(u32 minorversion, struct rpc_xprt *xprt)
cb_info->users++;
err_net:
if (!cb_info->users) {
+ xprt_svc_shutdown_bc(xprt);
svc_set_num_threads(cb_info->serv, 0, 0);
- svc_destroy(&cb_info->serv);
+ xprt_svc_destroy_nullify_bc(xprt, &cb_info->serv);
}
err_create:
mutex_unlock(&nfs_callback_mutex);
@@ -254,6 +255,7 @@ void nfs_callback_down(int minorversion, struct net *net, struct rpc_xprt *xprt)
mutex_lock(&nfs_callback_mutex);
serv = cb_info->serv;
+ xprt_svc_shutdown_bc(xprt);
nfs_callback_down_net(minorversion, serv, net);
cb_info->users--;
if (cb_info->users == 0) {
diff --git a/fs/nfs/dir.c b/fs/nfs/dir.c
index c7caffb31935..36f2e8588922 100644
--- a/fs/nfs/dir.c
+++ b/fs/nfs/dir.c
@@ -2427,9 +2427,9 @@ out_err:
}
int nfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
- return nfs_do_create(dir, dentry, mode, excl ? O_EXCL : 0);
+ return nfs_do_create(dir, dentry, mode, O_EXCL);
}
EXPORT_SYMBOL_GPL(nfs_create);
@@ -2474,7 +2474,7 @@ struct dentry *nfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
dir->i_sb->s_id, dir->i_ino, dentry);
attr.ia_valid = ATTR_MODE;
- attr.ia_mode = mode | S_IFDIR;
+ attr.ia_mode = mode;
trace_nfs_mkdir_enter(dir, dentry);
ret = NFS_PROTO(dir)->mkdir(dir, dentry, &attr);
diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h
index e4533f583632..7f96a258af76 100644
--- a/fs/nfs/internal.h
+++ b/fs/nfs/internal.h
@@ -395,7 +395,7 @@ extern unsigned long nfs_access_cache_scan(struct shrinker *shrink,
struct dentry *nfs_lookup(struct inode *, struct dentry *, unsigned int);
void nfs_d_prune_case_insensitive_aliases(struct inode *inode);
int nfs_create(struct mnt_idmap *, struct inode *, struct dentry *,
- umode_t, bool);
+ umode_t);
struct dentry *nfs_mkdir(struct mnt_idmap *, struct inode *, struct dentry *,
umode_t);
int nfs_rmdir(struct inode *, struct dentry *);
diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c
index 1360409d8de9..7d98e9a98580 100644
--- a/fs/nfs/nfs4proc.c
+++ b/fs/nfs/nfs4proc.c
@@ -10364,6 +10364,7 @@ static void nfs41_free_stateid_release(void *calldata)
struct nfs_free_stateid_data *data = calldata;
struct nfs_client *clp = data->server->nfs_client;
+ nfs_sb_deactive(data->server->super);
nfs_put_client(clp);
kfree(calldata);
}
@@ -10402,17 +10403,22 @@ static int nfs41_free_stateid(struct nfs_server *server,
struct nfs_free_stateid_data *data;
struct rpc_task *task;
struct nfs_client *clp = server->nfs_client;
+ int ret = -EIO;
if (!refcount_inc_not_zero(&clp->cl_count))
- return -EIO;
+ return ret;
+ if (!nfs_sb_active(server->super))
+ goto out_put_clp;
nfs4_state_protect(clp, NFS_SP4_MACH_CRED_STATEID,
&task_setup.rpc_client, &msg);
dprintk("NFS call free_stateid %p\n", stateid);
data = kmalloc_obj(*data);
- if (!data)
- return -ENOMEM;
+ if (!data) {
+ ret = -ENOMEM;
+ goto out_put_server;
+ }
data->server = server;
nfs4_stateid_copy(&data->args.stateid, stateid);
@@ -10428,6 +10434,11 @@ static int nfs41_free_stateid(struct nfs_server *server,
rpc_put_task(task);
stateid->type = NFS4_FREED_STATEID_TYPE;
return 0;
+out_put_server:
+ nfs_sb_deactive(server->super);
+out_put_clp:
+ nfs_put_client(clp);
+ return ret;
}
static void
diff --git a/fs/nfs_common/nfslocalio.c b/fs/nfs_common/nfslocalio.c
index dd715cdb6c04..85aa03a7b020 100644
--- a/fs/nfs_common/nfslocalio.c
+++ b/fs/nfs_common/nfslocalio.c
@@ -292,8 +292,22 @@ struct nfsd_file *nfs_open_local_fh(nfs_uuid_t *uuid,
localio = nfs_to->nfsd_open_local_fh(net, uuid->dom, rpc_clnt, cred,
nfs_fh, pnf, fmode);
if (!IS_ERR(localio) && nfs_uuid_add_file(uuid, nfl) < 0) {
- /* Delete the cached file when racing with nfs_uuid_put() */
+ /*
+ * Delete the cached file when racing with nfs_uuid_put().
+ * Since nfl->nfs_uuid was never published via
+ * rcu_assign_pointer(), nfs_close_local_fh() will early-return
+ * and cannot clean up after us. Drop the slot's file ref and
+ * its paired net ref, then drop the caller-owned nfsd_file ref
+ * (+1) and the entry-time nfsd_net ref carried via nf->nf_net,
+ * and return -ENXIO so the caller never dereferences the
+ * now-cleared localio.
+ */
+ struct nfsd_file __rcu *tmp =
+ (struct nfsd_file __force __rcu *)localio;
+
nfs_to_nfsd_file_put_local(pnf);
+ nfs_to_nfsd_file_put_local(&tmp);
+ localio = ERR_PTR(-ENXIO);
}
nfs_to_nfsd_net_put(net);
diff --git a/fs/nfsd/filecache.c b/fs/nfsd/filecache.c
index 24511c3208db..b9548eb17c77 100644
--- a/fs/nfsd/filecache.c
+++ b/fs/nfsd/filecache.c
@@ -55,6 +55,17 @@
/* We only care about NFSD_MAY_READ/WRITE for this cache */
#define NFSD_FILE_MAY_MASK (NFSD_MAY_READ|NFSD_MAY_WRITE|NFSD_MAY_LOCALIO)
+/* If the shrinker runs between calls to list_lru_walk_node() in
+ * nfsd_file_gc(), the "remaining" count will be wrong. This could
+ * result in premature freeing of some files. This may not matter much
+ * but is easy to fix with this spinlock which temporarily disables
+ * the shrinker.
+ *
+ * It also serializes callers of nfsd_file_dispose_list_delayed()
+ * against per-net shutdown.
+ */
+static DEFINE_SPINLOCK(nfsd_gc_lock);
+
static DEFINE_PER_CPU(unsigned long, nfsd_file_cache_hits);
static DEFINE_PER_CPU(unsigned long, nfsd_file_acquisitions);
static DEFINE_PER_CPU(unsigned long, nfsd_file_allocations);
@@ -62,16 +73,12 @@ static DEFINE_PER_CPU(unsigned long, nfsd_file_releases);
static DEFINE_PER_CPU(unsigned long, nfsd_file_total_age);
static DEFINE_PER_CPU(unsigned long, nfsd_file_evictions);
-struct nfsd_fcache_disposal {
- spinlock_t lock;
- struct list_head freeme;
-};
-
static struct kmem_cache *nfsd_file_slab;
static struct kmem_cache *nfsd_file_mark_slab;
static struct list_lru nfsd_file_lru;
static unsigned long nfsd_file_flags;
static struct fsnotify_group *nfsd_file_fsnotify_group;
+static struct fsnotify_group *nfsd_dir_fsnotify_group;
static struct delayed_work nfsd_filecache_laundrette;
static struct rhltable nfsd_file_rhltable
____cacheline_aligned_in_smp;
@@ -147,7 +154,7 @@ static void
nfsd_file_mark_put(struct nfsd_file_mark *nfm)
{
if (refcount_dec_and_test(&nfm->nfm_ref)) {
- fsnotify_destroy_mark(&nfm->nfm_mark, nfsd_file_fsnotify_group);
+ fsnotify_destroy_mark(&nfm->nfm_mark, nfm->nfm_mark.group);
fsnotify_put_mark(&nfm->nfm_mark);
}
}
@@ -155,37 +162,40 @@ nfsd_file_mark_put(struct nfsd_file_mark *nfm)
static struct nfsd_file_mark *
nfsd_file_mark_find_or_create(struct inode *inode)
{
- int err;
- struct fsnotify_mark *mark;
struct nfsd_file_mark *nfm = NULL, *new;
+ struct fsnotify_group *group;
+ struct fsnotify_mark *mark;
+ int err;
+
+ group = S_ISDIR(inode->i_mode) ? nfsd_dir_fsnotify_group : nfsd_file_fsnotify_group;
do {
- fsnotify_group_lock(nfsd_file_fsnotify_group);
- mark = fsnotify_find_inode_mark(inode,
- nfsd_file_fsnotify_group);
+ fsnotify_group_lock(group);
+ mark = fsnotify_find_inode_mark(inode, group);
if (mark) {
nfm = nfsd_file_mark_get(container_of(mark,
struct nfsd_file_mark,
nfm_mark));
- fsnotify_group_unlock(nfsd_file_fsnotify_group);
+ fsnotify_group_unlock(group);
if (nfm) {
fsnotify_put_mark(mark);
break;
}
/* Avoid soft lockup race with nfsd_file_mark_put() */
- fsnotify_destroy_mark(mark, nfsd_file_fsnotify_group);
+ fsnotify_destroy_mark(mark, group);
fsnotify_put_mark(mark);
} else {
- fsnotify_group_unlock(nfsd_file_fsnotify_group);
+ fsnotify_group_unlock(group);
}
/* allocate a new nfm */
new = kmem_cache_alloc(nfsd_file_mark_slab, GFP_KERNEL);
if (!new)
return NULL;
- fsnotify_init_mark(&new->nfm_mark, nfsd_file_fsnotify_group);
+ fsnotify_init_mark(&new->nfm_mark, group);
new->nfm_mark.mask = FS_ATTRIB|FS_DELETE_SELF;
refcount_set(&new->nfm_ref, 1);
+ mutex_init(&new->nfm_recalc_mutex);
err = fsnotify_add_inode_mark(&new->nfm_mark, inode, 0);
@@ -327,8 +337,11 @@ static void nfsd_file_lru_add(struct nfsd_file *nf)
refcount_inc(&nf->nf_ref);
if (list_lru_add_obj(&nfsd_file_lru, &nf->nf_lru))
trace_nfsd_file_lru_add(nf);
- else
- WARN_ON(1);
+ else {
+ refcount_dec(&nf->nf_ref);
+ WARN_ON_ONCE(1);
+ return;
+ }
nfsd_file_schedule_laundrette();
}
@@ -419,25 +432,31 @@ nfsd_file_dispose_list(struct list_head *dispose)
}
/**
- * nfsd_file_dispose_list_delayed - move list of dead files to net's freeme list
+ * nfsd_file_dispose_list_delayed - queue dead files for nfsd thread disposal
* @dispose: list of nfsd_files to be disposed
*
- * Transfers each file to the "freeme" list for its nfsd_net, to eventually
- * be disposed of by the per-net garbage collector.
+ * Transfers each file to the dispose list in its nfsd_net and wakes an nfsd
+ * thread to do the actual close. This keeps the cost of fput() in the nfsd
+ * threads rather than in the shrinker or GC worker.
+ *
+ * All callers must hold nfsd_gc_lock, so that nfsd_file_cache_shutdown_net()
+ * can synchronize against them before draining the per-net dispose list.
+ * This guarantees nf_net is still live when we call net_generic().
*/
static void
nfsd_file_dispose_list_delayed(struct list_head *dispose)
{
- while(!list_empty(dispose)) {
+ lockdep_assert_held(&nfsd_gc_lock);
+
+ while (!list_empty(dispose)) {
struct nfsd_file *nf = list_first_entry(dispose,
struct nfsd_file, nf_gc);
struct nfsd_net *nn = net_generic(nf->nf_net, nfsd_net_id);
- struct nfsd_fcache_disposal *l = nn->fcache_disposal;
struct svc_serv *serv;
- spin_lock(&l->lock);
- list_move_tail(&nf->nf_gc, &l->freeme);
- spin_unlock(&l->lock);
+ spin_lock(&nn->fcache_dispose_lock);
+ list_move_tail(&nf->nf_gc, &nn->fcache_dispose_list);
+ spin_unlock(&nn->fcache_dispose_lock);
/*
* The filecache laundrette is shut down after the
@@ -461,21 +480,28 @@ nfsd_file_dispose_list_delayed(struct list_head *dispose)
*/
void nfsd_file_net_dispose(struct nfsd_net *nn)
{
- struct nfsd_fcache_disposal *l = nn->fcache_disposal;
-
- if (!list_empty(&l->freeme)) {
+ if (!list_empty(&nn->fcache_dispose_list)) {
LIST_HEAD(dispose);
int i;
- spin_lock(&l->lock);
- for (i = 0; i < 8 && !list_empty(&l->freeme); i++)
- list_move(l->freeme.next, &dispose);
- spin_unlock(&l->lock);
- if (!list_empty(&l->freeme))
- /* Wake up another thread to share the work
+ spin_lock(&nn->fcache_dispose_lock);
+ for (i = 0; i < 8 && !list_empty(&nn->fcache_dispose_list); i++)
+ list_move(nn->fcache_dispose_list.next, &dispose);
+ spin_unlock(&nn->fcache_dispose_lock);
+ if (!list_empty(&nn->fcache_dispose_list)) {
+ /*
+ * Wake up another thread to share the work
* *before* doing any actual disposing.
+ *
+ * The filecache laundrette is shut down after
+ * the nn->nfsd_serv pointer is cleared, but
+ * before the svc_serv is freed.
*/
- svc_wake_up(nn->nfsd_serv);
+ struct svc_serv *serv = nn->nfsd_serv;
+
+ if (serv)
+ svc_wake_up(serv);
+ }
nfsd_file_dispose_list(&dispose);
}
}
@@ -552,13 +578,6 @@ nfsd_file_gc_cb(struct list_head *item, struct list_lru_one *lru,
return nfsd_file_lru_cb(item, lru, arg);
}
-/* If the shrinker runs between calls to list_lru_walk_node() in
- * nfsd_file_gc(), the "remaining" count will be wrong. This could
- * result in premature freeing of some files. This may not matter much
- * but is easy to fix with this spinlock which temporarily disables
- * the shrinker.
- */
-static DEFINE_SPINLOCK(nfsd_gc_lock);
static void
nfsd_file_gc(void)
{
@@ -581,9 +600,9 @@ nfsd_file_gc(void)
remaining = 0;
}
}
+ nfsd_file_dispose_list_delayed(&dispose);
spin_unlock(&nfsd_gc_lock);
trace_nfsd_file_gc_removed(ret, list_lru_count(&nfsd_file_lru));
- nfsd_file_dispose_list_delayed(&dispose);
}
static void
@@ -611,9 +630,9 @@ nfsd_file_lru_scan(struct shrinker *s, struct shrink_control *sc)
ret = list_lru_shrink_walk(&nfsd_file_lru, sc,
nfsd_file_lru_cb, &dispose);
+ nfsd_file_dispose_list_delayed(&dispose);
spin_unlock(&nfsd_gc_lock);
trace_nfsd_file_shrinker_removed(ret, list_lru_count(&nfsd_file_lru));
- nfsd_file_dispose_list_delayed(&dispose);
return ret;
}
@@ -686,11 +705,11 @@ nfsd_file_queue_for_close(struct inode *inode, struct list_head *dispose)
}
/**
- * nfsd_file_close_inode - attempt a delayed close of a nfsd_file
+ * nfsd_file_close_inode - attempt a deferred close of a nfsd_file
* @inode: inode of the file to attempt to remove
*
* Close out any open nfsd_files that can be reaped for @inode. The
- * actual freeing is deferred to the dispose_list_delayed infrastructure.
+ * actual freeing is deferred to the nfsd service threads.
*
* This is used by the fsnotify callbacks and setlease notifier.
*/
@@ -699,8 +718,10 @@ nfsd_file_close_inode(struct inode *inode)
{
LIST_HEAD(dispose);
+ spin_lock(&nfsd_gc_lock);
nfsd_file_queue_for_close(inode, &dispose);
nfsd_file_dispose_list_delayed(&dispose);
+ spin_unlock(&nfsd_gc_lock);
}
/**
@@ -812,12 +833,36 @@ nfsd_file_fsnotify_handle_event(struct fsnotify_mark *mark, u32 mask,
return 0;
}
+#ifdef CONFIG_NFSD_V4
+static int
+nfsd_dir_fsnotify_handle_event(struct fsnotify_group *group, u32 mask,
+ const void *data, int data_type, struct inode *dir,
+ const struct qstr *name, u32 cookie,
+ struct fsnotify_iter_info *iter_info)
+{
+ return nfsd_handle_dir_event(mask, dir, data, data_type, name);
+}
+#else
+static int
+nfsd_dir_fsnotify_handle_event(struct fsnotify_group *group, u32 mask,
+ const void *data, int data_type, struct inode *dir,
+ const struct qstr *name, u32 cookie,
+ struct fsnotify_iter_info *iter_info)
+{
+ return 0;
+}
+#endif
static const struct fsnotify_ops nfsd_file_fsnotify_ops = {
.handle_inode_event = nfsd_file_fsnotify_handle_event,
.free_mark = nfsd_file_mark_free,
};
+static const struct fsnotify_ops nfsd_dir_fsnotify_ops = {
+ .handle_event = nfsd_dir_fsnotify_handle_event,
+ .free_mark = nfsd_file_mark_free,
+};
+
int
nfsd_file_cache_init(void)
{
@@ -869,8 +914,7 @@ nfsd_file_cache_init(void)
goto out_shrinker;
}
- nfsd_file_fsnotify_group = fsnotify_alloc_group(&nfsd_file_fsnotify_ops,
- 0);
+ nfsd_file_fsnotify_group = fsnotify_alloc_group(&nfsd_file_fsnotify_ops, 0);
if (IS_ERR(nfsd_file_fsnotify_group)) {
pr_err("nfsd: unable to create fsnotify group: %ld\n",
PTR_ERR(nfsd_file_fsnotify_group));
@@ -879,11 +923,23 @@ nfsd_file_cache_init(void)
goto out_notifier;
}
+ nfsd_dir_fsnotify_group = fsnotify_alloc_group(&nfsd_dir_fsnotify_ops, 0);
+ if (IS_ERR(nfsd_dir_fsnotify_group)) {
+ pr_err("nfsd: unable to create fsnotify group: %ld\n",
+ PTR_ERR(nfsd_dir_fsnotify_group));
+ ret = PTR_ERR(nfsd_dir_fsnotify_group);
+ nfsd_dir_fsnotify_group = NULL;
+ goto out_notify_group;
+ }
+
INIT_DELAYED_WORK(&nfsd_filecache_laundrette, nfsd_file_gc_worker);
out:
if (ret)
clear_bit(NFSD_FILE_CACHE_UP, &nfsd_file_flags);
return ret;
+out_notify_group:
+ fsnotify_put_group(nfsd_file_fsnotify_group);
+ nfsd_file_fsnotify_group = NULL;
out_notifier:
lease_unregister_notifier(&nfsd_file_lease_notifier);
out_shrinker:
@@ -940,42 +996,14 @@ __nfsd_file_cache_purge(struct net *net)
nfsd_file_dispose_list(&dispose);
}
-static struct nfsd_fcache_disposal *
-nfsd_alloc_fcache_disposal(void)
-{
- struct nfsd_fcache_disposal *l;
-
- l = kmalloc_obj(*l);
- if (!l)
- return NULL;
- spin_lock_init(&l->lock);
- INIT_LIST_HEAD(&l->freeme);
- return l;
-}
-
-static void
-nfsd_free_fcache_disposal(struct nfsd_fcache_disposal *l)
-{
- nfsd_file_dispose_list(&l->freeme);
- kfree(l);
-}
-
-static void
-nfsd_free_fcache_disposal_net(struct net *net)
-{
- struct nfsd_net *nn = net_generic(net, nfsd_net_id);
- struct nfsd_fcache_disposal *l = nn->fcache_disposal;
-
- nfsd_free_fcache_disposal(l);
-}
-
int
nfsd_file_cache_start_net(struct net *net)
{
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
- nn->fcache_disposal = nfsd_alloc_fcache_disposal();
- return nn->fcache_disposal ? 0 : -ENOMEM;
+ spin_lock_init(&nn->fcache_dispose_lock);
+ INIT_LIST_HEAD(&nn->fcache_dispose_list);
+ return 0;
}
/**
@@ -994,8 +1022,18 @@ nfsd_file_cache_purge(struct net *net)
void
nfsd_file_cache_shutdown_net(struct net *net)
{
+ struct nfsd_net *nn = net_generic(net, nfsd_net_id);
+
nfsd_file_cache_purge(net);
- nfsd_free_fcache_disposal_net(net);
+ /*
+ * Ensure any in-progress shrinker, GC, or fsnotify/lease callback
+ * (all of which hold nfsd_gc_lock while calling
+ * nfsd_file_dispose_list_delayed()) has fully completed before
+ * draining the per-net dispose list.
+ */
+ spin_lock(&nfsd_gc_lock);
+ spin_unlock(&nfsd_gc_lock);
+ nfsd_file_dispose_list(&nn->fcache_dispose_list);
}
void
@@ -1019,6 +1057,8 @@ nfsd_file_cache_shutdown(void)
rcu_barrier();
fsnotify_put_group(nfsd_file_fsnotify_group);
nfsd_file_fsnotify_group = NULL;
+ fsnotify_put_group(nfsd_dir_fsnotify_group);
+ nfsd_dir_fsnotify_group = NULL;
kmem_cache_destroy(nfsd_file_slab);
nfsd_file_slab = NULL;
fsnotify_wait_marks_destroyed();
@@ -1223,11 +1263,9 @@ out:
open_file:
trace_nfsd_file_alloc(nf);
- if (type == S_IFREG)
- nf->nf_mark = nfsd_file_mark_find_or_create(inode);
-
- if (type != S_IFREG || nf->nf_mark) {
- if (file) {
+ nf->nf_mark = nfsd_file_mark_find_or_create(inode);
+ if (nf->nf_mark) {
+ if (file && (file->f_mode & FMODE_OPENED)) {
get_file(file);
nf->nf_file = file;
status = nfs_ok;
@@ -1374,12 +1412,12 @@ nfsd_file_acquire_local(struct net *net, struct svc_cred *cred,
* @rqstp: the RPC transaction being executed
* @fhp: the NFS filehandle of the file just created
* @may_flags: NFSD_MAY_ settings for the file
- * @file: cached, already-open file (may be NULL)
+ * @file: cached, already-open file (may be NULL or not yet opened)
* @pnf: OUT: new or found "struct nfsd_file" object
*
* Acquire a nfsd_file object that is not GC'ed. If one doesn't already exist,
- * and @file is non-NULL, use it to instantiate a new nfsd_file instead of
- * opening a new one.
+ * and @file has FMODE_OPENED set, use it to instantiate a new nfsd_file
+ * instead of opening a new one.
*
* Return values:
* %nfs_ok - @pnf points to an nfsd_file with its reference
@@ -1474,3 +1512,54 @@ int nfsd_file_cache_stats_show(struct seq_file *m, void *v)
seq_printf(m, "mean age (ms): -\n");
return 0;
}
+
+/**
+ * nfsd_fsnotify_recalc_mask - recalculate the fsnotify mask for a nfsd_file
+ * @nf: nfsd_file to recalculate the mask on
+ *
+ * When a directory nfsd_file has a delegation added or removed, that may
+ * change the events that nfsd requires from the VFS layer. This function
+ * recalculates the fsnotify mask based on the leases present.
+ */
+void nfsd_fsnotify_recalc_mask(struct nfsd_file *nf)
+{
+ struct inode *inode = file_inode(nf->nf_file);
+ u32 lease_mask, set = 0, clear = 0;
+ struct fsnotify_mark *mark;
+
+ /* This is only needed when adding or removing dir delegs */
+ if (!S_ISDIR(inode->i_mode) || !nf->nf_mark)
+ return;
+
+ mark = &nf->nf_mark->nfm_mark;
+
+ /*
+ * The mark is shared by every nfsd_file on this inode, so concurrent
+ * delegation add/remove on the same directory can recalc it in
+ * parallel. Serialize the read of the lease state and the update of
+ * the mark so that a recalc working from a stale snapshot of the
+ * lease list can't clobber a concurrent recalc's update.
+ */
+ mutex_lock(&nf->nf_mark->nfm_recalc_mutex);
+
+ /* Set up notifications for any ignored delegation events */
+ lease_mask = inode_lease_ignore_mask(inode);
+
+ if (lease_mask & FL_IGN_DIR_CREATE)
+ set |= FS_CREATE | FS_MOVED_TO;
+ else
+ clear |= FS_CREATE | FS_MOVED_TO;
+
+ if (lease_mask & FL_IGN_DIR_DELETE)
+ set |= FS_DELETE | FS_MOVED_FROM;
+ else
+ clear |= FS_DELETE | FS_MOVED_FROM;
+
+ if (lease_mask & FL_IGN_DIR_RENAME)
+ set |= FS_RENAME;
+ else
+ clear |= FS_RENAME;
+
+ fsnotify_modify_mark_mask(mark, set, clear);
+ mutex_unlock(&nf->nf_mark->nfm_recalc_mutex);
+}
diff --git a/fs/nfsd/filecache.h b/fs/nfsd/filecache.h
index 683b6437cacc..b224902b438d 100644
--- a/fs/nfsd/filecache.h
+++ b/fs/nfsd/filecache.h
@@ -26,6 +26,8 @@
struct nfsd_file_mark {
struct fsnotify_mark nfm_mark;
refcount_t nfm_ref;
+ /* serializes nfsd_fsnotify_recalc_mask() against itself */
+ struct mutex nfm_recalc_mutex;
};
/*
@@ -86,4 +88,5 @@ __be32 nfsd_file_acquire_local(struct net *net, struct svc_cred *cred,
__be32 nfsd_file_acquire_dir(struct svc_rqst *rqstp, struct svc_fh *fhp,
struct nfsd_file **pnf);
int nfsd_file_cache_stats_show(struct seq_file *m, void *v);
+void nfsd_fsnotify_recalc_mask(struct nfsd_file *nf);
#endif /* _FS_NFSD_FILECACHE_H */
diff --git a/fs/nfsd/flexfilelayoutxdr.c b/fs/nfsd/flexfilelayoutxdr.c
index f9f7e38cba13..374e52d3064a 100644
--- a/fs/nfsd/flexfilelayoutxdr.c
+++ b/fs/nfsd/flexfilelayoutxdr.c
@@ -30,19 +30,24 @@ nfsd4_ff_encode_layoutget(struct xdr_stream *xdr,
struct ff_idmap uid;
struct ff_idmap gid;
- fh_len = 4 + fl->fh.size;
+ fh_len = 4 + xdr_align_size(fl->fh.size);
uid.len = sprintf(uid.buf, "%u", from_kuid(&init_user_ns, fl->uid));
gid.len = sprintf(gid.buf, "%u", from_kgid(&init_user_ns, fl->gid));
- /* 8 + len for recording the length, name, and padding */
- ds_len = 20 + sizeof(stateid_opaque_t) + 4 + fh_len +
- 8 + uid.len + 8 + gid.len;
+ /* data server entry: deviceid + efficiency + stateid + fh list +
+ * user + group + flags + stats_collect_hint
+ */
+ ds_len = 16 + 4 + 4 + sizeof(stateid_opaque_t) + 4 + fh_len +
+ 4 + xdr_align_size(uid.len) +
+ 4 + xdr_align_size(gid.len) +
+ 4 + 4;
+ /* mirror: ds_count + ds */
mirror_len = 4 + ds_len;
- /* The layout segment */
- len = 20 + mirror_len;
+ /* stripe_unit + mirror_count + mirror */
+ len = 12 + mirror_len;
p = xdr_reserve_space(xdr, sizeof(__be32) + len);
if (!p)
@@ -94,7 +99,8 @@ nfsd4_ff_encode_getdeviceinfo(struct xdr_stream *xdr,
}
/* len + padding for two strings */
- addr_len = 16 + da->netaddr.netid_len + da->netaddr.addr_len;
+ addr_len = 8 + xdr_align_size(da->netaddr.netid_len) +
+ xdr_align_size(da->netaddr.addr_len);
ver_len = 20;
len = 4 + ver_len + 4 + addr_len;
diff --git a/fs/nfsd/localio.c b/fs/nfsd/localio.c
index be710d809a3b..c3eb0557b3e1 100644
--- a/fs/nfsd/localio.c
+++ b/fs/nfsd/localio.c
@@ -97,11 +97,15 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom,
}
nfsd_file_get(localio);
again:
+ rcu_read_lock();
new = unrcu_pointer(cmpxchg(pnf, NULL, RCU_INITIALIZER(localio)));
if (new) {
/* Some other thread installed an nfsd_file */
- if (nfsd_file_get(new) == NULL)
+ if (nfsd_file_get(new) == NULL) {
+ rcu_read_unlock();
goto again;
+ }
+ rcu_read_unlock();
/*
* Drop the ref we were going to install (both file and
* net) and the one we were going to return (only file).
@@ -110,6 +114,8 @@ nfsd_open_local_fh(struct net *net, struct auth_domain *dom,
nfsd_net_put(net);
nfsd_file_put(localio);
localio = new;
+ } else {
+ rcu_read_unlock();
}
} else
nfsd_net_put(net);
diff --git a/fs/nfsd/lockd.c b/fs/nfsd/lockd.c
index 6fe1325815e0..72a5b499839d 100644
--- a/fs/nfsd/lockd.c
+++ b/fs/nfsd/lockd.c
@@ -92,6 +92,7 @@ nlm_fclose(struct file *filp)
}
static const struct nlmsvc_binding nfsd_nlm_ops = {
+ .owner = THIS_MODULE,
.fopen = nlm_fopen, /* open file for locking */
.fclose = nlm_fclose, /* close file */
};
@@ -100,11 +101,12 @@ void
nfsd_lockd_init(void)
{
dprintk("nfsd: initializing lockd\n");
- nlmsvc_ops = &nfsd_nlm_ops;
+ rcu_assign_pointer(nlmsvc_ops, &nfsd_nlm_ops);
}
void
nfsd_lockd_shutdown(void)
{
- nlmsvc_ops = NULL;
+ RCU_INIT_POINTER(nlmsvc_ops, NULL);
+ synchronize_rcu();
}
diff --git a/fs/nfsd/netns.h b/fs/nfsd/netns.h
index 27da1a3edacb..03724bef10a7 100644
--- a/fs/nfsd/netns.h
+++ b/fs/nfsd/netns.h
@@ -28,6 +28,16 @@ struct cld_net;
struct nfsd_net_cb;
struct nfsd4_client_tracking_ops;
+enum nfsd_net_flag {
+ NFSD_NET_GRACE_ENDED,
+ NFSD_NET_GRACE_END_FORCED,
+ NFSD_NET_IN_GRACE,
+ NFSD_NET_SOMEBODY_RECLAIMED,
+ NFSD_NET_TRACK_RECLAIM_COMPLETES,
+ NFSD_NET_UP,
+ NFSD_NET_LOCKD_UP,
+};
+
enum {
/* cache misses due only to checksum comparison failures */
NFSD_STATS_PAYLOAD_MISSES,
@@ -66,9 +76,9 @@ struct nfsd_net {
struct cache_detail *nametoid_cache;
struct lock_manager nfsd4_manager;
- bool grace_ended;
- bool grace_end_forced;
+ unsigned long flags;
time64_t boot_time;
+ time64_t boot_time_bt; /* same instant in CLOCK_BOOTTIME */
struct dentry *nfsd_client_dir;
@@ -84,6 +94,7 @@ struct nfsd_net {
*/
struct list_head *reclaim_str_hashtbl;
int reclaim_str_hashtbl_size;
+ struct rw_semaphore reclaim_str_hashtbl_lock;
struct list_head *conf_id_hashtbl;
struct rb_root conf_name_tree;
struct list_head *unconf_id_hashtbl;
@@ -96,7 +107,10 @@ struct nfsd_net {
* close_lru holds (open) stateowner queue ordered by nfs4_stateowner.so_time
* for last close replay.
*
- * All of the above fields are protected by the client_mutex.
+ * reclaim_str_hashtbl[], reclaim_str_hashtbl_size are protected by
+ * reclaim_str_hashtbl_lock.
+ *
+ * All of the remaining fields are protected by the client_lock.
*/
struct list_head client_lru;
struct list_head close_lru;
@@ -117,19 +131,13 @@ struct nfsd_net {
spinlock_t blocked_locks_lock;
struct file *rec_file;
- bool in_grace;
const struct nfsd4_client_tracking_ops *client_tracking_ops;
time64_t nfsd4_lease;
time64_t nfsd4_grace;
- bool somebody_reclaimed;
- bool track_reclaim_completes;
atomic_t nr_reclaim_complete;
- bool nfsd_net_up;
- bool lockd_up;
-
seqlock_t writeverf_lock;
unsigned char writeverf[8];
@@ -209,7 +217,8 @@ struct nfsd_net {
/* utsname taken from the process that starts the server */
char nfsd_name[UNX_MAXNODENAME+1];
- struct nfsd_fcache_disposal *fcache_disposal;
+ spinlock_t fcache_dispose_lock;
+ struct list_head fcache_dispose_list;
siphash_key_t siphash_key;
diff --git a/fs/nfsd/nfs2acl.c b/fs/nfsd/nfs2acl.c
index 76305b86c1a9..827f90194c43 100644
--- a/fs/nfsd/nfs2acl.c
+++ b/fs/nfsd/nfs2acl.c
@@ -115,14 +115,19 @@ static __be32 nfsacld_proc_setacl(struct svc_rqst *rqstp)
inode_lock(inode);
- error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry, ACL_TYPE_ACCESS,
- argp->acl_access);
- if (error)
- goto out_drop_lock;
- error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry, ACL_TYPE_DEFAULT,
- argp->acl_default);
- if (error)
- goto out_drop_lock;
+ error = 0;
+ if (argp->mask & NFS_ACL) {
+ error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry,
+ ACL_TYPE_ACCESS, argp->acl_access);
+ if (error)
+ goto out_drop_lock;
+ }
+ if (argp->mask & NFS_DFACL) {
+ error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry,
+ ACL_TYPE_DEFAULT, argp->acl_default);
+ if (error)
+ goto out_drop_lock;
+ }
inode_unlock(inode);
diff --git a/fs/nfsd/nfs3acl.c b/fs/nfsd/nfs3acl.c
index e87731380be8..a87f9d7f32be 100644
--- a/fs/nfsd/nfs3acl.c
+++ b/fs/nfsd/nfs3acl.c
@@ -105,12 +105,17 @@ static __be32 nfsd3_proc_setacl(struct svc_rqst *rqstp)
inode_lock(inode);
- error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry, ACL_TYPE_ACCESS,
- argp->acl_access);
- if (error)
- goto out_drop_lock;
- error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry, ACL_TYPE_DEFAULT,
- argp->acl_default);
+ error = 0;
+ if (argp->mask & NFS_ACL) {
+ error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry,
+ ACL_TYPE_ACCESS, argp->acl_access);
+ if (error)
+ goto out_drop_lock;
+ }
+ if (argp->mask & NFS_DFACL) {
+ error = set_posix_acl(&nop_mnt_idmap, fh->fh_dentry,
+ ACL_TYPE_DEFAULT, argp->acl_default);
+ }
out_drop_lock:
inode_unlock(inode);
diff --git a/fs/nfsd/nfs3proc.c b/fs/nfsd/nfs3proc.c
index aeda7a802bdf..bbaef884f893 100644
--- a/fs/nfsd/nfs3proc.c
+++ b/fs/nfsd/nfs3proc.c
@@ -29,6 +29,25 @@ static int nfs3_ftypes[] = {
S_IFIFO, /* NF3FIFO */
};
+/*
+ * Reject a client-supplied atime or mtime whose nanoseconds field is out
+ * of range. Such a value is well-formed on the wire but is not a valid
+ * timespec64, and storing it verbatim can corrupt on-disk timestamps.
+ * tv_nsec is a long, so it is cast to unsigned long (the same width) to
+ * catch both an over-large value and one that became negative when an
+ * out-of-range u32 wire nseconds was assigned to a 32-bit long.
+ */
+static bool nfsd3_time_in_range(const struct iattr *iap)
+{
+ if ((iap->ia_valid & ATTR_ATIME_SET) &&
+ (unsigned long)iap->ia_atime.tv_nsec >= NSEC_PER_SEC)
+ return false;
+ if ((iap->ia_valid & ATTR_MTIME_SET) &&
+ (unsigned long)iap->ia_mtime.tv_nsec >= NSEC_PER_SEC)
+ return false;
+ return true;
+}
+
static __be32 nfsd3_map_status(__be32 status)
{
switch (status) {
@@ -101,9 +120,14 @@ nfsd3_proc_setattr(struct svc_rqst *rqstp)
SVCFH_fmt(&argp->fh));
fh_copy(&resp->fh, &argp->fh);
+ if (!nfsd3_time_in_range(&argp->attrs)) {
+ resp->status = nfserr_inval;
+ goto out;
+ }
if (argp->check_guard)
guardtime = &argp->guardtime;
resp->status = nfsd_setattr(rqstp, &resp->fh, &attrs, guardtime);
+out:
resp->status = nfsd3_map_status(resp->status);
return rpc_success;
}
@@ -265,7 +289,9 @@ nfsd3_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp,
trace_nfsd_vfs_create(rqstp, fhp, S_IFREG, argp->name, argp->len);
- if (isdotent(argp->name, argp->len))
+ if (!nfsd3_time_in_range(iap))
+ return nfserr_inval;
+ if (name_is_dot_dotdot(argp->name, argp->len))
return nfserr_exist;
if (!(iap->ia_valid & ATTR_MODE))
iap->ia_mode = 0;
@@ -400,8 +426,13 @@ nfsd3_proc_mkdir(struct svc_rqst *rqstp)
argp->attrs.ia_valid &= ~ATTR_SIZE;
fh_copy(&resp->dirfh, &argp->fh);
fh_init(&resp->fh, NFS3_FHSIZE);
+ if (!nfsd3_time_in_range(&argp->attrs)) {
+ resp->status = nfserr_inval;
+ goto out;
+ }
resp->status = nfsd_create(rqstp, &resp->dirfh, argp->name, argp->len,
&attrs, S_IFDIR, 0, &resp->fh);
+out:
resp->status = nfsd3_map_status(resp->status);
return rpc_success;
}
@@ -415,6 +446,10 @@ nfsd3_proc_symlink(struct svc_rqst *rqstp)
.na_iattr = &argp->attrs,
};
+ if (!nfsd3_time_in_range(&argp->attrs)) {
+ resp->status = nfserr_inval;
+ goto out;
+ }
if (argp->tlen == 0) {
resp->status = nfserr_inval;
goto out;
@@ -471,6 +506,11 @@ nfsd3_proc_mknod(struct svc_rqst *rqstp)
goto out;
}
+ if (!nfsd3_time_in_range(&argp->attrs)) {
+ resp->status = nfserr_inval;
+ goto out;
+ }
+
type = nfs3_ftypes[argp->ftype];
resp->status = nfsd_create(rqstp, &resp->dirfh, argp->name, argp->len,
&attrs, type, rdev, &resp->fh);
diff --git a/fs/nfsd/nfs3xdr.c b/fs/nfsd/nfs3xdr.c
index 2ff9a991a8fb..e481804bb120 100644
--- a/fs/nfsd/nfs3xdr.c
+++ b/fs/nfsd/nfs3xdr.c
@@ -987,7 +987,7 @@ compose_entry_fh(struct nfsd3_readdirres *cd, struct svc_fh *fhp,
dparent = cd->fh.fh_dentry;
exp = cd->fh.fh_export;
- if (isdotent(name, namlen)) {
+ if (name_is_dot_dotdot(name, namlen)) {
if (namlen == 2) {
dchild = dget_parent(dparent);
/*
diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c
index 50827405468d..71dcb448fa0a 100644
--- a/fs/nfsd/nfs4callback.c
+++ b/fs/nfsd/nfs4callback.c
@@ -108,6 +108,8 @@ static int decode_cb_fattr4(struct xdr_stream *xdr, uint32_t *bitmap,
if (!xdrgen_decode_fattr4_time_deleg_access(xdr, &access))
return -EIO;
+ if (access.nseconds >= NSEC_PER_SEC)
+ return -EIO;
fattr->ncf_cb_atime.tv_sec = access.seconds;
fattr->ncf_cb_atime.tv_nsec = access.nseconds;
@@ -117,6 +119,8 @@ static int decode_cb_fattr4(struct xdr_stream *xdr, uint32_t *bitmap,
if (!xdrgen_decode_fattr4_time_deleg_modify(xdr, &modify))
return -EIO;
+ if (modify.nseconds >= NSEC_PER_SEC)
+ return -EIO;
fattr->ncf_cb_mtime.tv_sec = modify.seconds;
fattr->ncf_cb_mtime.tv_nsec = modify.nseconds;
@@ -456,13 +460,20 @@ static void encode_cb_sequence4args(struct xdr_stream *xdr,
const struct nfsd4_callback *cb,
struct nfs4_cb_compound_hdr *hdr)
{
- struct nfsd4_session *session = cb->cb_clp->cl_cb_session;
+ struct nfsd4_session *session;
struct nfsd4_referring_call_list *rcl;
__be32 *p;
if (hdr->minorversion == 0)
return;
+ rcu_read_lock();
+ session = rcu_dereference(cb->cb_clp->cl_cb_session);
+ if (!session) {
+ rcu_read_unlock();
+ return;
+ }
+
encode_nfs_cb_opnum4(xdr, OP_CB_SEQUENCE);
encode_sessionid4(xdr, session);
@@ -478,6 +489,7 @@ static void encode_cb_sequence4args(struct xdr_stream *xdr,
encode_referring_call_list4(xdr, rcl);
hdr->nops++;
+ rcu_read_unlock();
}
static void update_cb_slot_table(struct nfsd4_session *ses, u32 target)
@@ -529,21 +541,32 @@ static void update_cb_slot_table(struct nfsd4_session *ses, u32 target)
static int decode_cb_sequence4resok(struct xdr_stream *xdr,
struct nfsd4_callback *cb)
{
- struct nfsd4_session *session = cb->cb_clp->cl_cb_session;
+ struct nfsd4_session *session;
int status = -ESERVERFAULT;
__be32 *p;
u32 seqid, slotid, target;
+ rcu_read_lock();
+ session = rcu_dereference(cb->cb_clp->cl_cb_session);
+ if (!session) {
+ rcu_read_unlock();
+ cb->cb_seq_status = -NFS4ERR_BADSESSION;
+ return -NFS4ERR_BADSESSION;
+ }
+
/*
* If the server returns different values for sessionID, slotID or
* sequence number, the server is looney tunes.
*/
p = xdr_inline_decode(xdr, NFS4_MAX_SESSIONID_LEN + 4 + 4 + 4 + 4);
- if (unlikely(p == NULL))
+ if (unlikely(p == NULL)) {
+ rcu_read_unlock();
goto out_overflow;
+ }
if (memcmp(p, session->se_sessionid.data, NFS4_MAX_SESSIONID_LEN)) {
dprintk("NFS: %s Invalid session id\n", __func__);
+ rcu_read_unlock();
goto out;
}
p += XDR_QUADLEN(NFS4_MAX_SESSIONID_LEN);
@@ -551,12 +574,14 @@ static int decode_cb_sequence4resok(struct xdr_stream *xdr,
seqid = be32_to_cpup(p++);
if (seqid != session->se_cb_seq_nr[cb->cb_held_slot]) {
dprintk("NFS: %s Invalid sequence number\n", __func__);
+ rcu_read_unlock();
goto out;
}
slotid = be32_to_cpup(p++);
if (slotid != cb->cb_held_slot) {
dprintk("NFS: %s Invalid slotid\n", __func__);
+ rcu_read_unlock();
goto out;
}
@@ -564,6 +589,7 @@ static int decode_cb_sequence4resok(struct xdr_stream *xdr,
target = be32_to_cpup(p++);
update_cb_slot_table(session, target);
+ rcu_read_unlock();
status = 0;
out:
cb->cb_seq_status = status;
@@ -865,6 +891,84 @@ static void encode_stateowner(struct xdr_stream *xdr, struct nfs4_stateowner *so
xdr_encode_opaque(p, so->so_owner.data, so->so_owner.len);
}
+static void nfs4_xdr_enc_cb_notify(struct rpc_rqst *req,
+ struct xdr_stream *xdr,
+ const void *data)
+{
+ const struct nfsd4_callback *cb = data;
+ struct nfsd4_cb_notify *ncn = container_of(cb, struct nfsd4_cb_notify, ncn_cb);
+ struct nfs4_delegation *dp = container_of(ncn, struct nfs4_delegation, dl_cb_notify);
+ struct nfs4_cb_compound_hdr hdr = {
+ .ident = 0,
+ .minorversion = cb->cb_clp->cl_minorversion,
+ };
+ struct CB_NOTIFY4args args;
+ unsigned int start;
+
+ WARN_ON_ONCE(hdr.minorversion == 0);
+
+ encode_cb_compound4args(xdr, &hdr);
+ encode_cb_sequence4args(xdr, cb, &hdr);
+
+ /*
+ * nfsd4_cb_notify_prepare() sized the payload against a single page,
+ * but did not account for the compound, sequence, stateid, and
+ * filehandle encoded here. If the variable-length encode overflows the
+ * backchannel send buffer, roll back to before the operation so that a
+ * truncated CB_NOTIFY is never placed on the wire.
+ */
+ start = xdr_stream_pos(xdr);
+
+ if (xdr_stream_encode_u32(xdr, OP_CB_NOTIFY) < 0)
+ goto out_err;
+
+ args.cna_stateid.seqid = dp->dl_stid.sc_stateid.si_generation;
+ memcpy(&args.cna_stateid.other, &dp->dl_stid.sc_stateid.si_opaque,
+ ARRAY_SIZE(args.cna_stateid.other));
+ args.cna_fh.len = dp->dl_stid.sc_file->fi_fhandle.fh_size;
+ args.cna_fh.data = dp->dl_stid.sc_file->fi_fhandle.fh_raw;
+ args.cna_changes.count = ncn->ncn_nf_cnt;
+ args.cna_changes.element = ncn->ncn_nf;
+ if (!xdrgen_encode_CB_NOTIFY4args(xdr, &args))
+ goto out_err;
+
+ hdr.nops++;
+ encode_cb_nops(&hdr);
+ return;
+
+out_err:
+ /*
+ * Drop the CB_NOTIFY op and emit a valid CB_SEQUENCE-only compound so
+ * the client still advances its slot. Flag the failure so the done
+ * handler recalls the delegation and the missed notification is not
+ * silently lost. The flag is written here in the transmit path and read
+ * in the done handler; the two are serialized phases of the same
+ * rpc_task, so no additional barrier is needed.
+ */
+ ncn->ncn_encode_err = true;
+ xdr_truncate_encode(xdr, start);
+ encode_cb_nops(&hdr);
+}
+
+static int nfs4_xdr_dec_cb_notify(struct rpc_rqst *rqstp,
+ struct xdr_stream *xdr,
+ void *data)
+{
+ struct nfsd4_callback *cb = data;
+ struct nfs4_cb_compound_hdr hdr;
+ int status;
+
+ status = decode_cb_compound4res(xdr, &hdr);
+ if (unlikely(status))
+ return status;
+
+ status = decode_cb_sequence4res(xdr, cb);
+ if (unlikely(status || cb->cb_seq_status))
+ return status;
+
+ return decode_cb_op_status(xdr, OP_CB_NOTIFY, &cb->cb_status);
+}
+
static void nfs4_xdr_enc_cb_notify_lock(struct rpc_rqst *req,
struct xdr_stream *xdr,
const void *data)
@@ -1026,6 +1130,7 @@ static const struct rpc_procinfo nfs4_cb_procedures[] = {
#ifdef CONFIG_NFSD_PNFS
PROC(CB_LAYOUT, COMPOUND, cb_layout, cb_layout),
#endif
+ PROC(CB_NOTIFY, COMPOUND, cb_notify, cb_notify),
PROC(CB_NOTIFY_LOCK, COMPOUND, cb_notify_lock, cb_notify_lock),
PROC(CB_OFFLOAD, COMPOUND, cb_offload, cb_offload),
PROC(CB_RECALL_ANY, COMPOUND, cb_recall_any, cb_recall_any),
@@ -1150,9 +1255,8 @@ static int setup_callback_client(struct nfs4_client *clp, struct nfs4_cb_conn *c
} else {
if (!conn->cb_xprt || !ses)
return -EINVAL;
- clp->cl_cb_session = ses;
args.bc_xprt = conn->cb_xprt;
- args.prognumber = clp->cl_cb_session->se_cb_prog;
+ args.prognumber = ses->se_cb_prog;
args.protocol = conn->cb_xprt->xpt_class->xcl_ident |
XPRT_TRANSPORT_BC;
args.authflavor = ses->se_cb_sec.flavor;
@@ -1170,8 +1274,10 @@ static int setup_callback_client(struct nfs4_client *clp, struct nfs4_cb_conn *c
return -ENOMEM;
}
- if (clp->cl_minorversion != 0)
+ if (clp->cl_minorversion != 0) {
clp->cl_cb_conn.cb_xprt = conn->cb_xprt;
+ rcu_assign_pointer(clp->cl_cb_session, ses);
+ }
clp->cl_cb_client = client;
clp->cl_cb_cred = cred;
rcu_read_lock();
@@ -1278,18 +1384,33 @@ static int grab_slot(struct nfsd4_session *ses)
static bool nfsd41_cb_get_slot(struct nfsd4_callback *cb, struct rpc_task *task)
{
struct nfs4_client *clp = cb->cb_clp;
- struct nfsd4_session *ses = clp->cl_cb_session;
+ struct nfsd4_session *ses;
if (cb->cb_held_slot >= 0)
return true;
+
+ rcu_read_lock();
+ ses = rcu_dereference(clp->cl_cb_session);
+ if (!ses) {
+ rcu_read_unlock();
+ rpc_sleep_on(&clp->cl_cb_waitq, task, NULL);
+ return false;
+ }
cb->cb_held_slot = grab_slot(ses);
if (cb->cb_held_slot < 0) {
+ rcu_read_unlock();
rpc_sleep_on(&clp->cl_cb_waitq, task, NULL);
/* Race breaker */
- cb->cb_held_slot = grab_slot(ses);
+ rcu_read_lock();
+ ses = rcu_dereference(clp->cl_cb_session);
+ if (ses)
+ cb->cb_held_slot = grab_slot(ses);
+ rcu_read_unlock();
if (cb->cb_held_slot < 0)
return false;
rpc_wake_up_queued_task(&clp->cl_cb_waitq, task);
+ } else {
+ rcu_read_unlock();
}
return true;
}
@@ -1297,12 +1418,17 @@ static bool nfsd41_cb_get_slot(struct nfsd4_callback *cb, struct rpc_task *task)
static void nfsd41_cb_release_slot(struct nfsd4_callback *cb)
{
struct nfs4_client *clp = cb->cb_clp;
- struct nfsd4_session *ses = clp->cl_cb_session;
+ struct nfsd4_session *ses;
if (cb->cb_held_slot >= 0) {
- spin_lock(&ses->se_lock);
- ses->se_cb_slot_avail |= BIT(cb->cb_held_slot);
- spin_unlock(&ses->se_lock);
+ rcu_read_lock();
+ ses = rcu_dereference(clp->cl_cb_session);
+ if (ses) {
+ spin_lock(&ses->se_lock);
+ ses->se_cb_slot_avail |= BIT(cb->cb_held_slot);
+ spin_unlock(&ses->se_lock);
+ }
+ rcu_read_unlock();
cb->cb_held_slot = -1;
rpc_wake_up_next(&clp->cl_cb_waitq);
}
@@ -1319,6 +1445,16 @@ static void nfsd41_destroy_cb(struct nfsd4_callback *cb)
else
clear_bit(NFSD4_CALLBACK_RUNNING, &cb->cb_flags);
+ /*
+ * Order the clear of NFSD4_CALLBACK_RUNNING above before the ->release()
+ * callback below. A release op may re-check producer-side state to decide
+ * whether to requeue itself (see nfsd4_cb_notify_release()), and that
+ * check must not be reordered ahead of the clear. The plain clear_bit()
+ * path carries no ordering; clear_and_wake_up_bit() already issues this
+ * barrier internally, so the extra one is harmless there.
+ */
+ smp_mb__after_atomic();
+
if (cb->cb_ops && cb->cb_ops->release)
cb->cb_ops->release(cb);
nfsd41_cb_inflight_end(clp);
@@ -1434,22 +1570,35 @@ static void nfsd4_cb_prepare(struct rpc_task *task, void *calldata)
trace_nfsd_cb_rpc_prepare(clp);
cb->cb_seq_status = 1;
cb->cb_status = 0;
- if (minorversion && !nfsd41_cb_get_slot(cb, task))
- return;
+ if (minorversion) {
+ if (!rcu_access_pointer(clp->cl_cb_session)) {
+ rpc_exit(task, -EIO);
+ return;
+ }
+ if (!nfsd41_cb_get_slot(cb, task))
+ return;
+ }
rpc_call_start(task);
}
/* Returns true if CB_COMPOUND processing should continue */
static bool nfsd4_cb_sequence_done(struct rpc_task *task, struct nfsd4_callback *cb)
{
- struct nfsd4_session *session = cb->cb_clp->cl_cb_session;
+ struct nfsd4_session *session;
bool ret = false;
if (cb->cb_held_slot < 0)
goto requeue;
+ rcu_read_lock();
+ session = rcu_dereference(cb->cb_clp->cl_cb_session);
+ if (!session) {
+ rcu_read_unlock();
+ goto requeue;
+ }
+
/* This is the operation status code for CB_SEQUENCE */
- trace_nfsd_cb_seq_status(task, cb);
+ trace_nfsd_cb_seq_status(task, cb, session);
switch (cb->cb_seq_status) {
case 0:
/*
@@ -1481,12 +1630,16 @@ static bool nfsd4_cb_sequence_done(struct rpc_task *task, struct nfsd4_callback
fallthrough;
case -NFS4ERR_BADSESSION:
nfsd4_mark_cb_fault(cb->cb_clp);
+ rcu_read_unlock();
goto requeue;
case -NFS4ERR_DELAY:
cb->cb_seq_status = 1;
- if (RPC_SIGNALLED(task) || !rpc_restart_call(task))
+ if (RPC_SIGNALLED(task) || !rpc_restart_call(task)) {
+ rcu_read_unlock();
goto requeue;
+ }
rpc_delay(task, 2 * HZ);
+ rcu_read_unlock();
return false;
case -NFS4ERR_SEQ_MISORDERED:
case -NFS4ERR_BADSLOT:
@@ -1498,11 +1651,13 @@ static bool nfsd4_cb_sequence_done(struct rpc_task *task, struct nfsd4_callback
*/
nfsd4_mark_cb_fault(cb->cb_clp);
cb->cb_held_slot = -1;
+ rcu_read_unlock();
goto retry_nowait;
default:
nfsd4_mark_cb_fault(cb->cb_clp);
}
- trace_nfsd_cb_free_slot(task, cb);
+ trace_nfsd_cb_free_slot(task, cb, session);
+ rcu_read_unlock();
nfsd41_cb_release_slot(cb);
return ret;
retry_nowait:
@@ -1624,7 +1779,15 @@ static struct nfsd4_conn * __nfsd4_find_backchannel(struct nfs4_client *clp)
* Note there isn't a lot of locking in this code; instead we depend on
* the fact that it is run from clp->cl_callback_wq, which won't run two
* work items at once. So, for example, clp->cl_callback_wq handles all
- * access of cl_cb_client and all calls to rpc_create or rpc_shutdown_client.
+ * access of cl_cb_client, and all calls to rpc_create or
+ * rpc_shutdown_client.
+ *
+ * cl_cb_session is written only from cl_callback_wq (via
+ * rcu_assign_pointer) and read from rpciod under rcu_read_lock (via
+ * rcu_dereference) by encode_cb_sequence4args(), decode_cb_sequence4resok(),
+ * nfsd4_cb_sequence_done(), and the cb-slot helpers. Sessions are freed
+ * with kfree_rcu() so that rpciod readers in an RCU read-side critical
+ * section never dereference a freed session.
*/
static void nfsd4_process_cb_update(struct nfsd4_callback *cb)
{
@@ -1676,6 +1839,7 @@ static void nfsd4_process_cb_update(struct nfsd4_callback *cb)
nfsd4_mark_cb_down(clp);
if (c)
svc_xprt_put(c->cn_xprt);
+ rcu_assign_pointer(clp->cl_cb_session, ses);
return;
}
}
@@ -1715,7 +1879,10 @@ nfsd4_run_cb_work(struct work_struct *work)
if (!test_and_clear_bit(NFSD4_CALLBACK_REQUEUE, &cb->cb_flags)) {
if (cb->cb_ops && cb->cb_ops->prepare)
- cb->cb_ops->prepare(cb);
+ if (!cb->cb_ops->prepare(cb)) {
+ nfsd41_destroy_cb(cb);
+ return;
+ }
}
cb->cb_msg.rpc_cred = clp->cl_cb_cred;
diff --git a/fs/nfsd/nfs4layouts.c b/fs/nfsd/nfs4layouts.c
index f34320e4c2f4..22bcb6d09f70 100644
--- a/fs/nfsd/nfs4layouts.c
+++ b/fs/nfsd/nfs4layouts.c
@@ -247,13 +247,21 @@ nfsd4_alloc_layout_stateid(struct nfsd4_compound_state *cstate,
nfsd4_init_cb(&ls->ls_recall, clp, &nfsd4_cb_layout_ops,
NFSPROC4_CLNT_CB_LAYOUT);
- if (parent->sc_type == SC_TYPE_DELEG)
- ls->ls_file = nfsd_file_get(fp->fi_deleg_file);
- else
+ if (parent->sc_type == SC_TYPE_DELEG) {
+ rcu_read_lock();
+ ls->ls_file = nfsd_file_get(rcu_dereference(fp->fi_deleg_file));
+ rcu_read_unlock();
+ } else {
ls->ls_file = find_any_file(fp);
- BUG_ON(!ls->ls_file);
+ }
+
+ if (!ls->ls_file) {
+ nfs4_put_stid(stp);
+ return NULL;
+ }
ls->ls_fenced = false;
+ ls->ls_fence_inflight = false;
ls->ls_fence_delay = 0;
INIT_DELAYED_WORK(&ls->ls_fence_work, nfsd4_layout_fence_worker);
@@ -652,7 +660,7 @@ nfsd4_cb_layout_fail(struct nfs4_layout_stateid *ls, struct nfsd_file *file)
}
}
-static void
+static bool
nfsd4_cb_layout_prepare(struct nfsd4_callback *cb)
{
struct nfs4_layout_stateid *ls =
@@ -661,6 +669,7 @@ nfsd4_cb_layout_prepare(struct nfsd4_callback *cb)
mutex_lock(&ls->ls_mutex);
nfs4_inc_and_copy_stateid(&ls->ls_recall_sid, &ls->ls_stid);
mutex_unlock(&ls->ls_mutex);
+ return true;
}
static int
@@ -791,15 +800,6 @@ nfsd4_layout_fence_worker(struct work_struct *work)
struct nfs4_client *clp;
struct nfsd_net *nn;
- /*
- * The workqueue clears WORK_STRUCT_PENDING before invoking
- * this callback. Re-arm immediately so that
- * delayed_work_pending() returns true while the fence
- * operation is in progress, preventing
- * lm_breaker_timedout() from taking a duplicate reference.
- */
- mod_delayed_work(system_dfl_wq, &ls->ls_fence_work, 0);
-
spin_lock(&ls->ls_lock);
if (list_empty(&ls->ls_layouts)) {
spin_unlock(&ls->ls_lock);
@@ -809,6 +809,9 @@ dispose:
nfsd4_close_layout(ls);
ls->ls_fenced = true;
+ spin_lock(&ls->ls_lock);
+ ls->ls_fence_inflight = false;
+ spin_unlock(&ls->ls_lock);
nfs4_put_stid(&ls->ls_stid);
return;
}
@@ -894,18 +897,26 @@ nfsd4_layout_lm_breaker_timedout(struct file_lease *fl)
if ((!nfsd4_layout_ops[ls->ls_layout_type]->fence_client) ||
ls->ls_fenced)
return true;
- if (delayed_work_pending(&ls->ls_fence_work))
- return false;
/*
* Make sure layout has not been returned yet before
- * taking a reference count on the layout stateid.
+ * taking a reference count on the layout stateid. The
+ * ls_fence_inflight flag is set together with the sc_count
+ * increment under ls_lock so that a fence worker invocation
+ * already in progress (which has cleared WORK_STRUCT_PENDING
+ * but not yet reached dispose:) cannot be coalesced with a
+ * fresh schedule that takes an extra unmatched reference.
*/
spin_lock(&ls->ls_lock);
+ if (ls->ls_fence_inflight) {
+ spin_unlock(&ls->ls_lock);
+ return false;
+ }
if (list_empty(&ls->ls_layouts) ||
!refcount_inc_not_zero(&ls->ls_stid.sc_count)) {
spin_unlock(&ls->ls_lock);
return true;
}
+ ls->ls_fence_inflight = true;
spin_unlock(&ls->ls_lock);
mod_delayed_work(system_dfl_wq, &ls->ls_fence_work, 0);
diff --git a/fs/nfsd/nfs4proc.c b/fs/nfsd/nfs4proc.c
index 8561540ab2db..669896be08b6 100644
--- a/fs/nfsd/nfs4proc.c
+++ b/fs/nfsd/nfs4proc.c
@@ -259,7 +259,7 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp,
__be32 status;
int host_err;
- if (isdotent(open->op_fname, open->op_fnamelen))
+ if (name_is_dot_dotdot(open->op_fname, open->op_fnamelen))
return nfserr_exist;
if (!(iap->ia_valid & ATTR_MODE))
iap->ia_mode = 0;
@@ -306,10 +306,6 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp,
goto out;
}
- status = fh_compose(resfhp, fhp->fh_export, child, fhp);
- if (status != nfs_ok)
- goto out;
-
v_mtime = 0;
v_atime = 0;
if (nfsd4_create_is_exclusive(open->op_createmode)) {
@@ -335,6 +331,10 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp,
if (status != nfs_ok)
goto out;
+ status = fh_compose(resfhp, fhp->fh_export, child, fhp);
+ if (status != nfs_ok)
+ goto out;
+
switch (open->op_createmode) {
case NFS4_CREATE_UNCHECKED:
if (!d_is_reg(child))
@@ -385,6 +385,10 @@ nfsd4_create_file(struct svc_rqst *rqstp, struct svc_fh *fhp,
open->op_created = true;
fh_fill_post_attrs(fhp);
+ status = fh_compose(resfhp, fhp->fh_export, child, fhp);
+ if (status != nfs_ok)
+ goto out;
+
/* A newly created file already has a file size of zero. */
if ((iap->ia_valid & ATTR_SIZE) && (iap->ia_size == 0))
iap->ia_valid &= ~ATTR_SIZE;
@@ -663,7 +667,7 @@ nfsd4_open(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
pr_warn("nfsd4_process_open2 failed to open newly-created file: status=%u\n",
be32_to_cpu(status));
if (reclaim && !status)
- nn->somebody_reclaimed = true;
+ set_bit(NFSD_NET_SOMEBODY_RECLAIMED, &nn->flags);
out:
if (open->op_filp) {
fput(open->op_filp);
@@ -677,8 +681,6 @@ out:
nfsd4_cleanup_open_state(cstate, open);
nfsd4_bump_seqid(cstate, status);
out_err:
- posix_acl_release(open->op_dpacl);
- posix_acl_release(open->op_pacl);
return status;
}
@@ -700,6 +702,13 @@ static __be32 nfsd4_open_omfg(struct svc_rqst *rqstp, struct nfsd4_compound_stat
return nfsd4_open(rqstp, cstate, &op->u);
}
+static void
+nfsd4_open_release(union nfsd4_op_u *u)
+{
+ posix_acl_release(u->open.op_dpacl);
+ posix_acl_release(u->open.op_pacl);
+}
+
/*
* filehandle-manipulating ops.
*/
@@ -839,6 +848,20 @@ nfsd4_create(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
if (status)
goto out_aftermask;
+ /* Sanitize cr_type to avoid returning ATTRNOTSUPP. */
+ switch (create->cr_type) {
+ case NF4LNK:
+ case NF4BLK:
+ case NF4CHR:
+ case NF4SOCK:
+ case NF4FIFO:
+ case NF4DIR:
+ break;
+ default:
+ status = nfserr_badtype;
+ goto out_aftermask;
+ }
+
if (create->cr_acl) {
if (attrs.na_dpacl || attrs.na_pacl) {
status = nfserr_inval;
@@ -846,6 +869,8 @@ nfsd4_create(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
}
status = nfsd4_acl_to_attr(create->cr_type, create->cr_acl,
&attrs);
+ if (status != nfs_ok)
+ goto out_aftermask;
}
current->fs->umask = create->cr_umask;
switch (create->cr_type) {
@@ -1253,7 +1278,7 @@ nfsd4_setattr(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
if (deleg_attrs) {
status = nfserr_bad_stateid;
- if (st->sc_type & SC_TYPE_DELEG) {
+ if (st && (st->sc_type & SC_TYPE_DELEG)) {
struct nfs4_delegation *dp = delegstateid(st);
/* Only for *_ATTRS_DELEG flavors */
@@ -1562,6 +1587,10 @@ static bool nfsd4_copy_on_sb(const struct nfsd4_copy *copy,
* nfsd4_cancel_copy_by_sb - cancel async copy operations on @sb
* @net: net namespace containing the copy operations
* @sb: targeted superblock
+ *
+ * Context: Caller must hold nfsd_mutex with NFSD_NET_UP set. Outside
+ * that window nn->conf_id_hashtbl is unallocated or freed,
+ * so the walk would dereference a NULL or dangling pointer.
*/
void nfsd4_cancel_copy_by_sb(struct net *net, struct super_block *sb)
{
@@ -1571,6 +1600,7 @@ void nfsd4_cancel_copy_by_sb(struct net *net, struct super_block *sb)
unsigned int idhashval;
LIST_HEAD(to_cancel);
+ lockdep_assert_held(&nfsd_mutex);
spin_lock(&nn->client_lock);
for (idhashval = 0; idhashval < CLIENT_HASH_SIZE; idhashval++) {
struct list_head *head = &nn->conf_id_hashtbl[idhashval];
@@ -1950,6 +1980,7 @@ static ssize_t _nfsd_copy_file_range(struct nfsd4_copy *copy,
/* See RFC 7862 p.67: */
if (bytes_total == 0)
bytes_total = ULLONG_MAX;
+ since = READ_ONCE(dst->f_wb_err);
do {
/* Only async copies can be stopped here */
if (kthread_should_stop())
@@ -1965,13 +1996,14 @@ static ssize_t _nfsd_copy_file_range(struct nfsd4_copy *copy,
} while (bytes_total > 0 && nfsd4_copy_is_async(copy));
/* for a non-zero asynchronous copy do a commit of data */
if (nfsd4_copy_is_async(copy) && copy->cp_res.wr_bytes_written > 0) {
- since = READ_ONCE(dst->f_wb_err);
end = copy->cp_dst_pos + copy->cp_res.wr_bytes_written - 1;
status = vfs_fsync_range(dst, copy->cp_dst_pos, end, 0);
if (!status)
status = filemap_check_wb_err(dst->f_mapping, since);
if (!status)
set_bit(NFSD4_COPY_F_COMMITTED, &copy->cp_flags);
+ else if (status != -EAGAIN && status != -ESTALE)
+ nfsd_reset_write_verifier(copy->cp_nn);
}
return bytes_copied;
}
@@ -2153,16 +2185,14 @@ nfsd4_copy(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
}
status = nfsd4_setup_inter_ssc(rqstp, cstate, copy);
if (status) {
- trace_nfsd_copy_done(copy, status);
- return nfserr_offload_denied;
+ status = nfserr_offload_denied;
+ goto out;
}
} else {
trace_nfsd_copy_intra(copy);
status = nfsd4_setup_intra_ssc(rqstp, cstate, copy);
- if (status) {
- trace_nfsd_copy_done(copy, status);
- return status;
- }
+ if (status)
+ goto out;
}
memcpy(&copy->fh, &cstate->current_fh.fh_handle,
@@ -2521,12 +2551,19 @@ nfsd4_verify(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
return status == nfserr_same ? nfs_ok : status;
}
+#define SUPPORTED_NOTIFY_MASK (BIT(NOTIFY4_CHANGE_DIR_ATTRS) | \
+ BIT(NOTIFY4_REMOVE_ENTRY) | \
+ BIT(NOTIFY4_ADD_ENTRY) | \
+ BIT(NOTIFY4_RENAME_ENTRY) | \
+ BIT(NOTIFY4_GFLAG_EXTEND))
+
static __be32
nfsd4_get_dir_delegation(struct svc_rqst *rqstp,
struct nfsd4_compound_state *cstate,
union nfsd4_op_u *u)
{
struct nfsd4_get_dir_delegation *gdd = &u->get_dir_delegation;
+ u32 requested = gdd->gdda_notification_types[0];
struct nfs4_delegation *dd;
struct nfsd_file *nf;
__be32 status;
@@ -2536,6 +2573,21 @@ nfsd4_get_dir_delegation(struct svc_rqst *rqstp,
return status;
/*
+ * Offer no notifications to an order-aware client. RFC8881bis section
+ * 16.2.13 defines order-aware as NOTIFY4_CFLAG_ORDER being set or
+ * NOTIFY4_GFLAG_EXTEND being reset. Such a client expects cookie and
+ * previous-entry information with its notifications (e.g. 27.4.5), and
+ * nfsd does not track or emit directory offset information. Per
+ * 16.2.11.3 the alternative would be to recall the delegation, so it's
+ * simpler to just decline the notifications here.
+ */
+ if (!(requested & BIT(NOTIFY4_GFLAG_EXTEND)) ||
+ (requested & BIT(NOTIFY4_CFLAG_ORDER)))
+ requested = 0;
+
+ gdd->gddr_notification[0] = requested & SUPPORTED_NOTIFY_MASK;
+
+ /*
* RFC 8881, section 18.39.3 says:
*
* "The server may refuse to grant the delegation. In that case, the
@@ -2556,6 +2608,10 @@ nfsd4_get_dir_delegation(struct svc_rqst *rqstp,
gdd->gddrnf_status = GDD4_OK;
memcpy(&gdd->gddr_stateid, &dd->dl_stid.sc_stateid, sizeof(gdd->gddr_stateid));
+ gdd->gddr_child_attributes[0] = dd->dl_child_attrs[0];
+ gdd->gddr_child_attributes[1] = dd->dl_child_attrs[1];
+ gdd->gddr_dir_attributes[0] = dd->dl_dir_attrs[0];
+ gdd->gddr_dir_attributes[1] = dd->dl_dir_attrs[1];
nfs4_put_stid(&dd->dl_stid);
return nfs_ok;
}
@@ -3119,9 +3175,22 @@ nfsd4_proc_compound(struct svc_rqst *rqstp)
op->status = nfsd4_open_omfg(rqstp, cstate, op);
goto encode_op;
}
- if (!current_fh->fh_dentry &&
- !HAS_FH_FLAG(current_fh, NFSD4_FH_FOREIGN)) {
- if (!(op->opdesc->op_flags & ALLOWED_WITHOUT_FH)) {
+ if (!current_fh->fh_dentry) {
+ if (HAS_FH_FLAG(current_fh, NFSD4_FH_FOREIGN)) {
+ /*
+ * FOREIGN fh from inter-SSC PUTFH: only
+ * SAVEFH may proceed with a NULL fh_dentry.
+ * Per RFC 7862 S15.2.3, validation of a
+ * foreign fh is deferred to the operation
+ * that consumes it, and NFS4ERR_STALE is
+ * returned at that point.
+ */
+ if (op->opnum != OP_SAVEFH &&
+ !(op->opdesc->op_flags & ALLOWED_WITHOUT_FH)) {
+ op->status = nfserr_stale;
+ goto encode_op;
+ }
+ } else if (!(op->opdesc->op_flags & ALLOWED_WITHOUT_FH)) {
op->status = nfserr_nofilehandle;
goto encode_op;
}
@@ -3185,6 +3254,9 @@ encode_op:
status = op->status;
}
+ if (op->opdesc && op->opdesc->op_release)
+ op->opdesc->op_release(&op->u);
+
trace_nfsd_compound_status(args->client_opcnt, resp->opcnt,
status, nfsd4_op_name(op->opnum));
@@ -3506,8 +3578,8 @@ static u32 nfsd4_get_dir_delegation_rsize(const struct svc_rqst *rqstp,
op_encode_verifier_maxsz +
op_encode_stateid_maxsz +
2 /* gddr_notification */ +
- 2 /* gddr_child_attributes */ +
- 2 /* gddr_dir_attributes */);
+ 3 /* gddr_child_attributes */ +
+ 3 /* gddr_dir_attributes */) * sizeof(__be32);
}
#ifdef CONFIG_NFSD_PNFS
@@ -3684,6 +3756,7 @@ static const struct nfsd4_operation nfsd4_ops[] = {
},
[OP_OPEN] = {
.op_func = nfsd4_open,
+ .op_release = nfsd4_open_release,
.op_flags = OP_HANDLES_WRONGSEC | OP_MODIFIES_SOMETHING,
.op_name = "OP_OPEN",
.op_rsize_bop = nfsd4_open_rsize,
diff --git a/fs/nfsd/nfs4recover.c b/fs/nfsd/nfs4recover.c
index 6ea25a52d2f4..d513971fb119 100644
--- a/fs/nfsd/nfs4recover.c
+++ b/fs/nfsd/nfs4recover.c
@@ -167,7 +167,7 @@ out_end:
end_creating(dentry);
out:
if (status == 0) {
- if (nn->in_grace)
+ if (test_bit(NFSD_NET_IN_GRACE, &nn->flags))
__nfsd4_create_reclaim_record_grace(clp, dname, nn);
vfs_fsync(nn->rec_file, 0);
} else {
@@ -285,10 +285,12 @@ __nfsd4_remove_reclaim_record_grace(const char *dname, int len,
return;
}
name.len = len;
+ down_write(&nn->reclaim_str_hashtbl_lock);
crp = nfsd4_find_reclaim_client(name, nn);
- kfree(name.data);
if (crp)
nfs4_remove_reclaim_record(crp, nn);
+ up_write(&nn->reclaim_str_hashtbl_lock);
+ kfree(name.data);
}
static void
@@ -317,7 +319,7 @@ nfsd4_remove_clid_dir(struct nfs4_client *clp)
nfs4_reset_creds(original_cred);
if (status == 0) {
vfs_fsync(nn->rec_file, 0);
- if (nn->in_grace)
+ if (test_bit(NFSD_NET_IN_GRACE, &nn->flags))
__nfsd4_remove_reclaim_record_grace(dname,
HEXDIR_LEN, nn);
}
@@ -373,7 +375,7 @@ nfsd4_recdir_purge_old(struct nfsd_net *nn)
{
int status;
- nn->in_grace = false;
+ clear_bit(NFSD_NET_IN_GRACE, &nn->flags);
if (!nn->rec_file)
return;
status = mnt_want_write_file(nn->rec_file);
@@ -455,7 +457,7 @@ nfsd4_init_recdir(struct net *net)
nfs4_reset_creds(original_cred);
if (!status)
- nn->in_grace = true;
+ set_bit(NFSD_NET_IN_GRACE, &nn->flags);
return status;
}
@@ -484,6 +486,7 @@ nfs4_legacy_state_init(struct net *net)
for (i = 0; i < CLIENT_HASH_SIZE; i++)
INIT_LIST_HEAD(&nn->reclaim_str_hashtbl[i]);
nn->reclaim_str_hashtbl_size = 0;
+ init_rwsem(&nn->reclaim_str_hashtbl_lock);
return 0;
}
@@ -598,13 +601,16 @@ nfsd4_check_legacy_client(struct nfs4_client *clp)
goto out_enoent;
}
name.len = HEXDIR_LEN;
+ down_read(&nn->reclaim_str_hashtbl_lock);
crp = nfsd4_find_reclaim_client(name, nn);
- kfree(name.data);
if (crp) {
set_bit(NFSD4_CLIENT_STABLE, &clp->cl_flags);
crp->cr_clp = clp;
- return 0;
}
+ up_read(&nn->reclaim_str_hashtbl_lock);
+ kfree(name.data);
+ if (crp)
+ return 0;
out_enoent:
return -ENOENT;
@@ -1176,6 +1182,7 @@ nfsd4_cld_check(struct nfs4_client *clp)
return 0;
/* look for it in the reclaim hashtable otherwise */
+ down_read(&nn->reclaim_str_hashtbl_lock);
crp = nfsd4_find_reclaim_client(clp->cl_name, nn);
if (crp)
goto found;
@@ -1191,6 +1198,7 @@ nfsd4_cld_check(struct nfs4_client *clp)
if (!name.data) {
dprintk("%s: failed to allocate memory for name.data!\n",
__func__);
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
}
name.len = HEXDIR_LEN;
@@ -1201,9 +1209,11 @@ nfsd4_cld_check(struct nfs4_client *clp)
}
#endif
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
found:
crp->cr_clp = clp;
+ up_read(&nn->reclaim_str_hashtbl_lock);
return 0;
}
@@ -1215,6 +1225,7 @@ nfsd4_cld_check_v2(struct nfs4_client *clp)
struct cld_net *cn = nn->cld_net;
#endif
struct nfs4_client_reclaim *crp;
+ unsigned int princhashlen;
char *principal = NULL;
/* did we already find that this client is stable? */
@@ -1222,6 +1233,7 @@ nfsd4_cld_check_v2(struct nfs4_client *clp)
return 0;
/* look for it in the reclaim hashtable otherwise */
+ down_read(&nn->reclaim_str_hashtbl_lock);
crp = nfsd4_find_reclaim_client(clp->cl_name, nn);
if (crp)
goto found;
@@ -1237,6 +1249,7 @@ nfsd4_cld_check_v2(struct nfs4_client *clp)
if (!name.data) {
dprintk("%s: failed to allocate memory for name.data\n",
__func__);
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
}
name.len = HEXDIR_LEN;
@@ -1247,23 +1260,31 @@ nfsd4_cld_check_v2(struct nfs4_client *clp)
}
#endif
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
found:
- if (crp->cr_princhash.len) {
+ princhashlen = crp->cr_princhash.len;
+ if (princhashlen) {
u8 digest[SHA256_DIGEST_SIZE];
+ u8 *pdata;
if (clp->cl_cred.cr_raw_principal)
principal = clp->cl_cred.cr_raw_principal;
else if (clp->cl_cred.cr_principal)
principal = clp->cl_cred.cr_principal;
- if (principal == NULL)
+ if (principal == NULL) {
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
+ }
sha256(principal, strlen(principal), digest);
- if (memcmp(crp->cr_princhash.data, digest,
- crp->cr_princhash.len))
+ pdata = crp->cr_princhash.data;
+ if (memcmp(pdata, digest, princhashlen)) {
+ up_read(&nn->reclaim_str_hashtbl_lock);
return -ENOENT;
+ }
}
crp->cr_clp = clp;
+ up_read(&nn->reclaim_str_hashtbl_lock);
return 0;
}
@@ -1362,7 +1383,8 @@ nfs4_cld_state_init(struct net *net)
for (i = 0; i < CLIENT_HASH_SIZE; i++)
INIT_LIST_HEAD(&nn->reclaim_str_hashtbl[i]);
nn->reclaim_str_hashtbl_size = 0;
- nn->track_reclaim_completes = true;
+ init_rwsem(&nn->reclaim_str_hashtbl_lock);
+ set_bit(NFSD_NET_TRACK_RECLAIM_COMPLETES, &nn->flags);
atomic_set(&nn->nr_reclaim_complete, 0);
return 0;
@@ -1373,7 +1395,7 @@ nfs4_cld_state_shutdown(struct net *net)
{
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
- nn->track_reclaim_completes = false;
+ clear_bit(NFSD_NET_TRACK_RECLAIM_COMPLETES, &nn->flags);
kfree(nn->reclaim_str_hashtbl);
}
diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c
index a42f34842d77..a4398dc861a5 100644
--- a/fs/nfsd/nfs4state.c
+++ b/fs/nfsd/nfs4state.c
@@ -55,6 +55,7 @@
#include "netns.h"
#include "pnfs.h"
#include "filecache.h"
+#include "nfs4xdr_gen.h"
#include "trace.h"
#define NFSDDBG_FACILITY NFSDDBG_PROC
@@ -91,6 +92,8 @@ static void _free_cpntf_state_locked(struct nfsd_net *nn, struct nfs4_cpntf_stat
static void nfsd4_file_hash_remove(struct nfs4_file *fi);
static void deleg_reaper(struct nfsd_net *nn);
+static const struct lease_manager_operations nfsd_lease_mng_ops;
+
/* Locking: */
enum nfsd4_st_mutex_lock_subclass {
@@ -124,6 +127,7 @@ static void free_session(struct nfsd4_session *);
static const struct nfsd4_callback_ops nfsd4_cb_recall_ops;
static const struct nfsd4_callback_ops nfsd4_cb_notify_lock_ops;
static const struct nfsd4_callback_ops nfsd4_cb_getattr_ops;
+static const struct nfsd4_callback_ops nfsd4_cb_notify_ops;
static struct workqueue_struct *laundry_wq;
@@ -355,12 +359,13 @@ remove_blocked_locks(struct nfs4_lockowner *lo)
}
}
-static void
+static bool
nfsd4_cb_notify_lock_prepare(struct nfsd4_callback *cb)
{
struct nfsd4_blocked_lock *nbl = container_of(cb,
struct nfsd4_blocked_lock, nbl_cb);
locks_delete_block(&nbl->nbl_lock);
+ return true;
}
static int
@@ -1120,29 +1125,31 @@ static void block_delegations(struct knfsd_fh *fh)
}
static struct nfs4_delegation *
-alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp,
- struct nfs4_clnt_odstate *odstate, u32 dl_type)
+__alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp,
+ struct nfs4_clnt_odstate *odstate, u32 dl_type,
+ void (*sc_free)(struct nfs4_stid *))
{
struct nfs4_delegation *dp;
struct nfs4_stid *stid;
long n;
- dprintk("NFSD alloc_init_deleg\n");
+ if (delegation_blocked(&fp->fi_fhandle))
+ return NULL;
+
n = atomic_long_inc_return(&num_delegations);
if (n < 0 || n > max_delegations)
goto out_dec;
- if (delegation_blocked(&fp->fi_fhandle))
- goto out_dec;
- stid = nfs4_alloc_stid(clp, deleg_slab, nfs4_free_deleg);
+
+ stid = nfs4_alloc_stid(clp, deleg_slab, sc_free);
if (stid == NULL)
goto out_dec;
- dp = delegstateid(stid);
/*
* delegation seqid's are never incremented. The 4.1 special
* meaning of seqid 0 isn't meaningful, really, but let's avoid
- * 0 anyway just for consistency and use 1:
+ * 0 anyway just for consistency and use 1.
*/
+ dp = delegstateid(stid);
dp->dl_stid.sc_stateid.si_generation = 1;
INIT_LIST_HEAD(&dp->dl_perfile);
INIT_LIST_HEAD(&dp->dl_perclnt);
@@ -1152,19 +1159,79 @@ alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp,
dp->dl_type = dl_type;
dp->dl_retries = 1;
dp->dl_recalled = false;
- nfsd4_init_cb(&dp->dl_recall, dp->dl_stid.sc_client,
- &nfsd4_cb_recall_ops, NFSPROC4_CLNT_CB_RECALL);
- nfsd4_init_cb(&dp->dl_cb_fattr.ncf_getattr, dp->dl_stid.sc_client,
- &nfsd4_cb_getattr_ops, NFSPROC4_CLNT_CB_GETATTR);
- dp->dl_cb_fattr.ncf_file_modified = false;
get_nfs4_file(fp);
dp->dl_stid.sc_file = fp;
+ nfsd4_init_cb(&dp->dl_recall, dp->dl_stid.sc_client,
+ &nfsd4_cb_recall_ops, NFSPROC4_CLNT_CB_RECALL);
return dp;
out_dec:
atomic_long_dec(&num_delegations);
return NULL;
}
+static struct nfs4_delegation *
+alloc_init_deleg(struct nfs4_client *clp, struct nfs4_file *fp,
+ struct nfs4_clnt_odstate *odstate, u32 dl_type)
+{
+ struct nfs4_delegation *dp;
+
+ dp = __alloc_init_deleg(clp, fp, odstate, dl_type, nfs4_free_deleg);
+ if (!dp)
+ return NULL;
+
+ nfsd4_init_cb(&dp->dl_cb_fattr.ncf_getattr, dp->dl_stid.sc_client,
+ &nfsd4_cb_getattr_ops, NFSPROC4_CLNT_CB_GETATTR);
+ dp->dl_cb_fattr.ncf_file_modified = false;
+ return dp;
+}
+
+static void nfs4_free_dir_deleg(struct nfs4_stid *stid)
+{
+ struct nfs4_delegation *dp = delegstateid(stid);
+ struct nfsd4_cb_notify *ncn = &dp->dl_cb_notify;
+ int i;
+
+ for (i = 0; i < ncn->ncn_evt_cnt; ++i)
+ nfsd_notify_event_put(ncn->ncn_evt[i]);
+ kfree(ncn->ncn_nf);
+ for (i = 0; i < NOTIFY4_PAGE_ARRAY_SIZE; i++) {
+ if (!ncn->ncn_pages[i])
+ break;
+ put_page(ncn->ncn_pages[i]);
+ }
+ nfs4_free_deleg(stid);
+}
+
+static struct nfs4_delegation *
+alloc_init_dir_deleg(struct nfs4_client *clp, struct nfs4_file *fp)
+{
+ struct nfs4_delegation *dp;
+ struct nfsd4_cb_notify *ncn;
+ int npages;
+
+ dp = __alloc_init_deleg(clp, fp, NULL, NFS4_OPEN_DELEGATE_READ, nfs4_free_dir_deleg);
+ if (!dp)
+ return NULL;
+
+ ncn = &dp->dl_cb_notify;
+
+ npages = alloc_pages_bulk(GFP_KERNEL, NOTIFY4_PAGE_ARRAY_SIZE, ncn->ncn_pages);
+ if (npages != NOTIFY4_PAGE_ARRAY_SIZE) {
+ nfs4_put_stid(&dp->dl_stid);
+ return NULL;
+ }
+
+ ncn->ncn_nf = kcalloc(NOTIFY4_EVENT_QUEUE_SIZE, sizeof(*ncn->ncn_nf), GFP_KERNEL);
+ if (!ncn->ncn_nf) {
+ nfs4_put_stid(&dp->dl_stid);
+ return NULL;
+ }
+ spin_lock_init(&ncn->ncn_lock);
+ nfsd4_init_cb(&ncn->ncn_cb, dp->dl_stid.sc_client,
+ &nfsd4_cb_notify_ops, NFSPROC4_CLNT_CB_NOTIFY);
+ return dp;
+}
+
void
nfs4_put_stid(struct nfs4_stid *s)
{
@@ -1209,7 +1276,9 @@ static void put_deleg_file(struct nfs4_file *fp)
spin_lock(&fp->fi_lock);
if (--fp->fi_delegees == 0) {
- swap(nf, fp->fi_deleg_file);
+ nf = rcu_dereference_protected(fp->fi_deleg_file,
+ lockdep_is_held(&fp->fi_lock));
+ RCU_INIT_POINTER(fp->fi_deleg_file, NULL);
swap(rnf, fp->fi_rdeleg_file);
}
spin_unlock(&fp->fi_lock);
@@ -1247,12 +1316,13 @@ static void nfsd4_finalize_deleg_timestamps(struct nfs4_delegation *dp, struct f
static void nfs4_unlock_deleg_lease(struct nfs4_delegation *dp)
{
struct nfs4_file *fp = dp->dl_stid.sc_file;
- struct nfsd_file *nf = fp->fi_deleg_file;
+ struct nfsd_file *nf = rcu_dereference_protected(fp->fi_deleg_file, 1);
WARN_ON_ONCE(!fp->fi_delegees);
nfsd4_finalize_deleg_timestamps(dp, nf->nf_file);
kernel_setlease(nf->nf_file, F_UNLCK, NULL, (void **)&dp);
+ nfsd_fsnotify_recalc_mask(nf);
put_deleg_file(fp);
}
@@ -1873,14 +1943,20 @@ static void revoke_one_stid(struct nfsd_net *nn, struct nfs4_client *clp,
* being released. Thus nfsd will no longer prevent the filesystem from being
* unmounted.
*
- * The clients which own the states will subsequently being notified that the
+ * The clients which own the states will subsequently be notified that the
* states have been "admin-revoked".
+ *
+ * Context: Caller must hold nfsd_mutex with NFSD_NET_UP set. Outside
+ * that window nn->conf_id_hashtbl is unallocated or freed,
+ * so the walk would dereference a NULL or dangling pointer.
*/
void nfsd4_revoke_states(struct nfsd_net *nn, struct super_block *sb)
{
unsigned int idhashval;
unsigned int sc_types;
+ lockdep_assert_held(&nfsd_mutex);
+
sc_types = SC_TYPE_OPEN | SC_TYPE_LOCK | SC_TYPE_DELEG | SC_TYPE_LAYOUT;
spin_lock(&nn->client_lock);
@@ -1946,12 +2022,18 @@ static struct nfs4_stid *find_one_export_stid(struct nfs4_client *clp,
*
* Userspace (exportfs -u) sends this after removing the last client
* for a path, enabling the underlying filesystem to be unmounted.
+ *
+ * Context: Caller must hold nfsd_mutex with NFSD_NET_UP set. Outside
+ * that window nn->conf_id_hashtbl is unallocated or freed,
+ * so the walk would dereference a NULL or dangling pointer.
*/
void nfsd4_revoke_export_states(struct nfsd_net *nn, const struct path *path)
{
unsigned int idhashval;
unsigned int sc_types;
+ lockdep_assert_held(&nfsd_mutex);
+
sc_types = SC_TYPE_OPEN | SC_TYPE_LOCK | SC_TYPE_DELEG | SC_TYPE_LAYOUT;
spin_lock(&nn->client_lock);
@@ -2052,12 +2134,10 @@ gen_sessionid(struct nfsd4_session *ses)
static struct shrinker *nfsd_slot_shrinker;
static DEFINE_SPINLOCK(nfsd_session_list_lock);
static LIST_HEAD(nfsd_session_list);
-/* The sum of "target_slots-1" on every session. The shrinker can push this
- * down, though it can take a little while for the memory to actually
- * be freed. The "-1" is because we can never free slot 0 while the
- * session is active.
- */
+/* The sum of "target_slots" on every session, slot 0 included. */
static atomic_t nfsd_total_target_slots = ATOMIC_INIT(0);
+/* Session count, subtracted from the sum to exclude slot 0. */
+static atomic_t nfsd_total_sessions = ATOMIC_INIT(0);
static void
free_session_slots(struct nfsd4_session *ses, int from)
@@ -2081,26 +2161,21 @@ free_session_slots(struct nfsd4_session *ses, int from)
}
ses->se_fchannel.maxreqs = from;
if (ses->se_target_maxslots > from) {
- int new_target = from ?: 1;
- atomic_sub(ses->se_target_maxslots - new_target, &nfsd_total_target_slots);
- ses->se_target_maxslots = new_target;
+ int delta = ses->se_target_maxslots - from;
+
+ atomic_sub(delta, &nfsd_total_target_slots);
+ /* Retain one slot so the session can make forward progress. */
+ ses->se_target_maxslots = from ?: 1;
}
}
-/**
- * reduce_session_slots - reduce the target max-slots of a session if possible
- * @ses: The session to affect
- * @dec: how much to decrease the target by
- *
+/*
* This interface can be used by a shrinker to reduce the target max-slots
* for a session so that some slots can eventually be freed.
* It uses spin_trylock() as it may be called in a context where another
* spinlock is held that has a dependency on client_lock. As shrinkers are
- * best-effort, skiping a session is client_lock is already held has no
- * great coast
- *
- * Return value:
- * The number of slots that the target was reduced by.
+ * best-effort, skipping a session with the client_lock already held has no
+ * great cost.
*/
static int
reduce_session_slots(struct nfsd4_session *ses, int dec)
@@ -2179,7 +2254,7 @@ static struct nfsd4_session *alloc_session(struct nfsd4_channel_attrs *fattrs,
fattrs->maxreqs = i;
memcpy(&new->se_fchannel, fattrs, sizeof(struct nfsd4_channel_attrs));
new->se_target_maxslots = i;
- atomic_add(i - 1, &nfsd_total_target_slots);
+ atomic_add(i, &nfsd_total_target_slots);
new->se_cb_slot_avail = ~0U;
new->se_cb_highest_slot = min(battrs->maxreqs - 1,
NFSD_BC_SLOT_TABLE_SIZE - 1);
@@ -2295,7 +2370,7 @@ static void __free_session(struct nfsd4_session *ses)
{
free_session_slots(ses, 0);
xa_destroy(&ses->se_slots);
- kfree(ses);
+ kfree_rcu(ses, rcu_head);
}
static void free_session(struct nfsd4_session *ses)
@@ -2304,21 +2379,51 @@ static void free_session(struct nfsd4_session *ses)
__free_session(ses);
}
+/**
+ * nfsd_slot_shrinker_count - report reclaimable DRC slots
+ * @s: shrinker descriptor (unused)
+ * @sc: shrink control (unused)
+ *
+ * Return: a positive count of reclaimable slots, or SHRINK_EMPTY when
+ * there is nothing to reclaim.
+ */
static unsigned long
-nfsd_slot_count(struct shrinker *s, struct shrink_control *sc)
+nfsd_slot_shrinker_count(struct shrinker *s, struct shrink_control *sc)
{
- unsigned long cnt = atomic_read(&nfsd_total_target_slots);
+ int count;
+
+ /*
+ * To prevent session deadlock, one slot of each session (slot 0)
+ * is not reclaimable while the session is active. Thus the number
+ * of sessions is subtracted from the total number of target slots.
+ */
+ count = atomic_read(&nfsd_total_target_slots) -
+ atomic_read(&nfsd_total_sessions);
- return cnt ? cnt : SHRINK_EMPTY;
+ return count > 0 ? count : SHRINK_EMPTY;
}
+/**
+ * nfsd_slot_shrinker_scan - reclaim DRC slots under memory pressure
+ * @s: shrinker descriptor (unused)
+ * @sc: shrink control; @sc->nr_to_scan bounds the sessions visited,
+ * @sc->nr_scanned reports how many were visited
+ *
+ * Return: the number of session slots NFSD will release.
+ */
static unsigned long
-nfsd_slot_scan(struct shrinker *s, struct shrink_control *sc)
+nfsd_slot_shrinker_scan(struct shrinker *s, struct shrink_control *sc)
{
struct nfsd4_session *ses;
unsigned long scanned = 0;
unsigned long freed = 0;
+ /*
+ * Each visited session releases at most one slot. After
+ * nr_to_scan sessions have been visited, the list head is
+ * rotated past the last visited session so the next scan
+ * resumes from there.
+ */
spin_lock(&nfsd_session_list_lock);
list_for_each_entry(ses, &nfsd_session_list, se_all_sessions) {
freed += reduce_session_slots(ses, 1);
@@ -2360,6 +2465,7 @@ static void init_session(struct svc_rqst *rqstp, struct nfsd4_session *new, stru
spin_lock(&nfsd_session_list_lock);
list_add_tail(&new->se_all_sessions, &nfsd_session_list);
+ atomic_inc(&nfsd_total_sessions);
spin_unlock(&nfsd_session_list_lock);
{
@@ -2433,6 +2539,7 @@ unhash_session(struct nfsd4_session *ses)
spin_unlock(&ses->se_client->cl_lock);
spin_lock(&nfsd_session_list_lock);
list_del(&ses->se_all_sessions);
+ atomic_dec(&nfsd_total_sessions);
spin_unlock(&nfsd_session_list_lock);
}
@@ -2581,7 +2688,17 @@ unhash_client_locked(struct nfs4_client *clp)
spin_lock(&nfsd_session_list_lock);
list_for_each_entry(ses, &clp->cl_sessions, se_perclnt) {
list_del_init(&ses->se_hash);
- list_del_init(&ses->se_all_sessions);
+ /*
+ * unhash_client_locked() can run more than once for a
+ * client; the session stays on cl_sessions across calls.
+ * The first pass empties se_all_sessions via
+ * list_del_init(), so skip the decrement on later passes
+ * to keep nfsd_total_sessions from being double-counted.
+ */
+ if (!list_empty(&ses->se_all_sessions)) {
+ list_del_init(&ses->se_all_sessions);
+ atomic_dec(&nfsd_total_sessions);
+ }
}
spin_unlock(&nfsd_session_list_lock);
spin_unlock(&clp->cl_lock);
@@ -2675,16 +2792,23 @@ static void inc_reclaim_complete(struct nfs4_client *clp)
{
struct nfsd_net *nn = net_generic(clp->net, nfsd_net_id);
- if (!nn->track_reclaim_completes)
+ if (!test_bit(NFSD_NET_TRACK_RECLAIM_COMPLETES, &nn->flags))
return;
- if (!nfsd4_find_reclaim_client(clp->cl_name, nn))
+
+ down_read(&nn->reclaim_str_hashtbl_lock);
+ if (!nfsd4_find_reclaim_client(clp->cl_name, nn)) {
+ up_read(&nn->reclaim_str_hashtbl_lock);
return;
+ }
if (atomic_inc_return(&nn->nr_reclaim_complete) ==
nn->reclaim_str_hashtbl_size) {
+ up_read(&nn->reclaim_str_hashtbl_lock);
printk(KERN_INFO "NFSD: all clients done reclaiming, ending NFSv4 grace period (net %x)\n",
clp->net->ns.inum);
nfsd4_end_grace(nn);
+ return;
}
+ up_read(&nn->reclaim_str_hashtbl_lock);
}
static void expire_client(struct nfs4_client *clp)
@@ -3140,7 +3264,8 @@ static int nfs4_show_deleg(struct seq_file *s, struct nfs4_stid *st)
/* XXX: lease time, whether it's being recalled. */
spin_lock(&nf->fi_lock);
- file = nf->fi_deleg_file;
+ file = rcu_dereference_protected(nf->fi_deleg_file,
+ lockdep_is_held(&nf->fi_lock));
if (file) {
seq_puts(s, ", ");
nfs4_show_superblock(s, file);
@@ -3359,6 +3484,212 @@ nfsd4_cb_getattr_release(struct nfsd4_callback *cb)
nfs4_put_stid(&dp->dl_stid);
}
+static void nfsd_break_one_deleg(struct nfs4_delegation *dp)
+{
+ bool queued;
+
+ if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &dp->dl_recall.cb_flags))
+ return;
+
+ /*
+ * When called from the lease break (nfsd_break_deleg_cb()) the state
+ * code is serialized by the flc_lock and the lease has not been
+ * removed yet, so sc_count is known to be nonzero. The CB_NOTIFY
+ * callback paths reach here from a workqueue without the flc_lock,
+ * where the delegation may already be unhashed with sc_count at zero.
+ * Use refcount_inc_not_zero() so both cases are safe, and bail if the
+ * delegation is already being torn down.
+ */
+ if (!refcount_inc_not_zero(&dp->dl_stid.sc_count)) {
+ clear_bit(NFSD4_CALLBACK_RUNNING, &dp->dl_recall.cb_flags);
+ return;
+ }
+ queued = nfsd4_run_cb(&dp->dl_recall);
+ WARN_ON_ONCE(!queued);
+ if (!queued) {
+ refcount_dec(&dp->dl_stid.sc_count);
+ clear_bit(NFSD4_CALLBACK_RUNNING, &dp->dl_recall.cb_flags);
+ }
+}
+
+static bool
+nfsd4_cb_notify_prepare(struct nfsd4_callback *cb)
+{
+ struct nfsd4_cb_notify *ncn = container_of(cb, struct nfsd4_cb_notify, ncn_cb);
+ struct nfs4_delegation *dp = container_of(ncn, struct nfs4_delegation, dl_cb_notify);
+ struct nfsd_notify_event *events[NOTIFY4_EVENT_QUEUE_SIZE];
+ struct xdr_buf xdr = { .buflen = PAGE_SIZE * NOTIFY4_PAGE_ARRAY_SIZE,
+ .pages = ncn->ncn_pages };
+ int limit = NOTIFY4_EVENT_QUEUE_SIZE;
+ struct xdr_stream stream;
+ struct nfsd_file *nf;
+ bool error = false;
+ int count, i;
+
+ /* Save a slot for dir attr update if requested */
+ if (dp->dl_notify_mask & BIT(NOTIFY4_CHANGE_DIR_ATTRS))
+ --limit;
+
+ /* Clear any failure recorded by a previous transmit. */
+ ncn->ncn_encode_err = false;
+
+ xdr_init_encode_pages(&stream, &xdr);
+
+ spin_lock(&ncn->ncn_lock);
+ count = ncn->ncn_evt_cnt;
+
+ /* spurious queueing? */
+ if (count == 0) {
+ spin_unlock(&ncn->ncn_lock);
+ return false;
+ }
+
+ memcpy(events, ncn->ncn_evt, sizeof(*events) * count);
+ ncn->ncn_evt_cnt = 0;
+ spin_unlock(&ncn->ncn_lock);
+
+ /*
+ * We can't keep up! Drop the queued events and recall. The queue must
+ * be drained here: out_recall leaves ncn_evt_cnt at 0, so the release
+ * op won't see leftover events and requeue this callback forever.
+ */
+ if (count > limit) {
+ for (i = 0; i < count; ++i)
+ nfsd_notify_event_put(events[i]);
+ goto out_recall;
+ }
+
+ rcu_read_lock();
+ nf = nfsd_file_get(rcu_dereference(dp->dl_stid.sc_file->fi_deleg_file));
+ rcu_read_unlock();
+ if (!nf) {
+ for (i = 0; i < count; ++i)
+ nfsd_notify_event_put(events[i]);
+ goto out_recall;
+ }
+
+ for (i = 0; i < count; ++i) {
+ struct nfsd_notify_event *nne = events[i];
+
+ if (!error) {
+ u32 *maskp = (u32 *)xdr_reserve_space(&stream, sizeof(*maskp));
+ u8 *p;
+
+ if (!maskp) {
+ error = true;
+ goto put_event;
+ }
+
+ p = nfsd4_encode_notify_event(&stream, nne, dp, nf, maskp);
+ if (!p) {
+ pr_notice("Could not generate CB_NOTIFY from fsnotify mask 0x%x\n",
+ nne->ne_mask);
+ error = true;
+ goto put_event;
+ }
+
+ ncn->ncn_nf[i].notify_mask.count = 1;
+ ncn->ncn_nf[i].notify_mask.element = maskp;
+ ncn->ncn_nf[i].notify_vals.data = p;
+ ncn->ncn_nf[i].notify_vals.len = (u8 *)stream.p - p;
+ }
+put_event:
+ nfsd_notify_event_put(nne);
+ }
+ if (!error && (dp->dl_notify_mask & BIT(NOTIFY4_CHANGE_DIR_ATTRS))) {
+ u32 *maskp = (u32 *)xdr_reserve_space(&stream, sizeof(*maskp));
+ u8 *p;
+
+ if (maskp)
+ p = nfsd4_encode_dir_attr_change(&stream, dp, nf);
+ else
+ p = ERR_PTR(-ENOBUFS);
+
+ if (IS_ERR(p)) {
+ /*
+ * The client asked to be told about dir attr changes
+ * but the change could not be encoded. RFC 8881
+ * s10.9.4 requires the server to recall the delegation
+ * rather than drop a requested notification, so fall
+ * through to recall. A NULL return instead means there
+ * were no attributes to report, so omit the event in
+ * that case.
+ */
+ error = true;
+ } else if (p) {
+ *maskp = BIT(NOTIFY4_CHANGE_DIR_ATTRS);
+ ncn->ncn_nf[count].notify_mask.count = 1;
+ ncn->ncn_nf[count].notify_mask.element = maskp;
+ ncn->ncn_nf[count].notify_vals.data = p;
+ ncn->ncn_nf[count].notify_vals.len = (u8 *)stream.p - p;
+ ++count;
+ }
+ }
+ if (!error) {
+ ncn->ncn_nf_cnt = count;
+ nfsd_file_put(nf);
+ return true;
+ }
+ nfsd_file_put(nf);
+out_recall:
+ nfsd_break_one_deleg(dp);
+ return false;
+}
+
+static int
+nfsd4_cb_notify_done(struct nfsd4_callback *cb,
+ struct rpc_task *task)
+{
+ struct nfsd4_cb_notify *ncn = container_of(cb, struct nfsd4_cb_notify, ncn_cb);
+ struct nfs4_delegation *dp = container_of(ncn, struct nfs4_delegation, dl_cb_notify);
+
+ if (dp->dl_stid.sc_status)
+ return 1;
+
+ /*
+ * The CB_NOTIFY op overflowed the send buffer and was dropped from the
+ * compound. The notification is lost, so recall the delegation rather
+ * than leaving the client unaware of the directory change.
+ */
+ if (ncn->ncn_encode_err) {
+ nfsd_break_one_deleg(dp);
+ return 1;
+ }
+
+ switch (task->tk_status) {
+ case -NFS4ERR_DELAY:
+ rpc_delay(task, 2 * HZ);
+ return 0;
+ default:
+ /* For any other hard error, recall the deleg */
+ nfsd_break_one_deleg(dp);
+ fallthrough;
+ case 0:
+ return 1;
+ }
+}
+
+static void nfsd4_run_cb_notify(struct nfsd4_cb_notify *ncn);
+
+static void
+nfsd4_cb_notify_release(struct nfsd4_callback *cb)
+{
+ struct nfsd4_cb_notify *ncn =
+ container_of(cb, struct nfsd4_cb_notify, ncn_cb);
+ struct nfs4_delegation *dp =
+ container_of(ncn, struct nfs4_delegation, dl_cb_notify);
+
+ /*
+ * Drain events that arrived while this callback was in flight, but
+ * don't requeue against a revoked delegation: there's no point in
+ * notifying a client that no longer holds it, and doing so can pin the
+ * stid and spin the workqueue.
+ */
+ if (!dp->dl_stid.sc_status && READ_ONCE(ncn->ncn_evt_cnt) > 0)
+ nfsd4_run_cb_notify(ncn);
+ nfs4_put_stid(&dp->dl_stid);
+}
+
static const struct nfsd4_callback_ops nfsd4_cb_recall_any_ops = {
.done = nfsd4_cb_recall_any_done,
.release = nfsd4_cb_recall_any_release,
@@ -3371,6 +3702,13 @@ static const struct nfsd4_callback_ops nfsd4_cb_getattr_ops = {
.opcode = OP_CB_GETATTR,
};
+static const struct nfsd4_callback_ops nfsd4_cb_notify_ops = {
+ .prepare = nfsd4_cb_notify_prepare,
+ .done = nfsd4_cb_notify_done,
+ .release = nfsd4_cb_notify_release,
+ .opcode = OP_CB_NOTIFY,
+};
+
static void nfs4_cb_getattr(struct nfs4_cb_fattr *ncf)
{
struct nfs4_delegation *dp =
@@ -3414,7 +3752,7 @@ static struct nfs4_client *create_client(struct xdr_netobj name,
clp->cl_time = ktime_get_boottime_seconds();
copy_verf(clp, verf);
memcpy(&clp->cl_addr, sa, sizeof(struct sockaddr_storage));
- clp->cl_cb_session = NULL;
+ RCU_INIT_POINTER(clp->cl_cb_session, NULL);
clp->net = net;
clp->cl_nfsd_dentry = nfsd_client_mkdir(
nn, &clp->cl_nfsdfs,
@@ -4496,6 +4834,19 @@ static void nfsd4_construct_sequence_response(struct nfsd4_session *session,
seq->status_flags |= SEQ4_STATUS_ADMIN_STATE_REVOKED;
}
+static bool nfsd4_slots_inuse(struct nfsd4_session *ses, int from)
+{
+ int i;
+
+ for (i = from; i < ses->se_fchannel.maxreqs; i++) {
+ struct nfsd4_slot *slot = xa_load(&ses->se_slots, i);
+
+ if (slot->sl_flags & NFSD4_SLOT_INUSE)
+ return true;
+ }
+ return false;
+}
+
__be32
nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
union nfsd4_op_u *u)
@@ -4575,7 +4926,9 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
if (session->se_target_maxslots < session->se_fchannel.maxreqs &&
slot->sl_generation == session->se_slot_gen &&
- seq->maxslots <= session->se_target_maxslots)
+ seq->maxslots <= session->se_target_maxslots &&
+ seq->slotid < session->se_target_maxslots &&
+ !nfsd4_slots_inuse(session, session->se_target_maxslots))
/* Client acknowledged our reduce maxreqs */
free_session_slots(session, session->se_target_maxslots);
@@ -4608,15 +4961,26 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
* gently try to allocate another 20%. This allows
* fairly quick growth without grossly over-shooting what
* the client might use.
+ *
+ * Bound that growth by the service's thread ceiling:
+ * slots beyond the nfsd thread count cannot raise this
+ * client's throughput, only deepen its backlog. Cap each
+ * session independently, since a session cannot use
+ * another's slots; a shared budget would let idle sessions
+ * pin an active client small. Compare against the
+ * configured maximum, not the running thread count, so a
+ * client resuming from idle can grow back before the pool
+ * scales up.
*/
if (seq->slotid == session->se_fchannel.maxreqs - 1 &&
- session->se_target_maxslots >= session->se_fchannel.maxreqs &&
- session->se_fchannel.maxreqs < NFSD_MAX_SLOTS_PER_SESSION) {
+ session->se_target_maxslots >= session->se_fchannel.maxreqs) {
int s = session->se_fchannel.maxreqs;
- int cnt = DIV_ROUND_UP(s, 5);
+ int ceiling = min_t(int, NFSD_MAX_SLOTS_PER_SESSION,
+ svc_serv_maxthreads(rqstp->rq_server));
+ int cnt = min(DIV_ROUND_UP(s, 5), ceiling - s);
void *prev_slot;
- do {
+ while (cnt-- > 0) {
/*
* GFP_NOWAIT both allows allocation under a
* spinlock, and only succeeds if there is
@@ -4624,13 +4988,14 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
*/
slot = nfsd4_alloc_slot(&session->se_fchannel, s,
GFP_NOWAIT);
+ if (!slot)
+ break;
prev_slot = xa_load(&session->se_slots, s);
- if (xa_is_value(prev_slot) && slot) {
+ if (xa_is_value(prev_slot)) {
slot->sl_seqid = xa_to_value(prev_slot);
slot->sl_flags |= NFSD4_SLOT_REUSED;
}
- if (slot &&
- !xa_is_err(xa_store(&session->se_slots, s, slot,
+ if (!xa_is_err(xa_store(&session->se_slots, s, slot,
GFP_NOWAIT))) {
s += 1;
session->se_fchannel.maxreqs = s;
@@ -4639,9 +5004,9 @@ nfsd4_sequence(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
session->se_target_maxslots = s;
} else {
kfree(slot);
- slot = NULL;
+ break;
}
- } while (slot && --cnt > 0);
+ }
}
out:
@@ -4922,7 +5287,7 @@ static void nfsd4_file_init(const struct svc_fh *fh, struct nfs4_file *fp)
INIT_LIST_HEAD(&fp->fi_delegations);
INIT_LIST_HEAD(&fp->fi_clnt_odstate);
fh_copy_shallow(&fp->fi_fhandle, &fh->fh_handle);
- fp->fi_deleg_file = NULL;
+ RCU_INIT_POINTER(fp->fi_deleg_file, NULL);
fp->fi_rdeleg_file = NULL;
fp->fi_had_conflict = false;
fp->fi_share_deny = 0;
@@ -5018,8 +5383,6 @@ nfsd4_init_leases_net(struct nfsd_net *nn)
nn->nfsd4_lease = 90; /* default lease time */
nn->nfsd4_grace = 90;
- nn->somebody_reclaimed = false;
- nn->track_reclaim_completes = false;
nn->clverifier_counter = get_random_u32();
nn->clientid_base = get_random_u32();
nn->clientid_counter = nn->clientid_base + 1;
@@ -5172,6 +5535,7 @@ static void nfsd4_drop_revoked_stid(struct nfs4_stid *s)
case SC_TYPE_DELEG:
dp = delegstateid(s);
list_del_init(&dp->dl_recall_lru);
+ s->sc_status |= SC_STATUS_FREED;
spin_unlock(&cl->cl_lock);
nfs4_put_stid(s);
break;
@@ -5528,7 +5892,7 @@ bool nfsd_wait_for_delegreturn(struct svc_rqst *rqstp, struct inode *inode)
return timeo > 0;
}
-static void nfsd4_cb_recall_prepare(struct nfsd4_callback *cb)
+static bool nfsd4_cb_recall_prepare(struct nfsd4_callback *cb)
{
struct nfs4_delegation *dp = cb_to_delegation(cb);
struct nfsd_net *nn = net_generic(dp->dl_stid.sc_client->net,
@@ -5549,6 +5913,7 @@ static void nfsd4_cb_recall_prepare(struct nfsd4_callback *cb)
list_add_tail(&dp->dl_recall_lru, &nn->del_recall_lru);
}
spin_unlock(&nn->deleg_lock);
+ return true;
}
static int nfsd4_cb_recall_done(struct nfsd4_callback *cb,
@@ -5598,27 +5963,6 @@ static const struct nfsd4_callback_ops nfsd4_cb_recall_ops = {
.opcode = OP_CB_RECALL,
};
-static void nfsd_break_one_deleg(struct nfs4_delegation *dp)
-{
- bool queued;
-
- if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &dp->dl_recall.cb_flags))
- return;
-
- /*
- * We're assuming the state code never drops its reference
- * without first removing the lease. Since we're in this lease
- * callback (and since the lease code is serialized by the
- * flc_lock) we know the server hasn't removed the lease yet, and
- * we know it's safe to take a reference.
- */
- refcount_inc(&dp->dl_stid.sc_count);
- queued = nfsd4_run_cb(&dp->dl_recall);
- WARN_ON_ONCE(!queued);
- if (!queued)
- refcount_dec(&dp->dl_stid.sc_count);
-}
-
/* Called from break_lease() with flc_lock held. */
static bool
nfsd_break_deleg_cb(struct file_lease *fl)
@@ -5664,6 +6008,10 @@ static bool nfsd_breaker_owns_lease(struct file_lease *fl)
struct svc_rqst *rqst;
struct nfs4_client *clp;
+ /* Only nfsd leases */
+ if (fl->fl_lmops != &nfsd_lease_mng_ops)
+ return false;
+
rqst = nfsd_current_rqst();
if (!nfsd_v4client(rqst))
return false;
@@ -6043,7 +6391,22 @@ static bool nfsd4_cb_channel_good(struct nfs4_client *clp)
return clp->cl_minorversion && clp->cl_cb_state == NFSD4_CB_UNKNOWN;
}
-static struct file_lease *nfs4_alloc_init_lease(struct nfs4_delegation *dp)
+static unsigned int
+nfsd_notify_to_ignore(u32 notify)
+{
+ unsigned int mask = 0;
+
+ if (notify & BIT(NOTIFY4_REMOVE_ENTRY))
+ mask |= FL_IGN_DIR_DELETE;
+ if (notify & BIT(NOTIFY4_ADD_ENTRY))
+ mask |= FL_IGN_DIR_CREATE;
+ if (notify & BIT(NOTIFY4_RENAME_ENTRY))
+ mask |= FL_IGN_DIR_RENAME;
+
+ return mask;
+}
+
+static struct file_lease *nfs4_alloc_init_lease(struct nfs4_delegation *dp, u32 notify)
{
struct file_lease *fl;
@@ -6051,11 +6414,11 @@ static struct file_lease *nfs4_alloc_init_lease(struct nfs4_delegation *dp)
if (!fl)
return NULL;
fl->fl_lmops = &nfsd_lease_mng_ops;
- fl->c.flc_flags = FL_DELEG;
+ fl->c.flc_flags = FL_DELEG | nfsd_notify_to_ignore(notify);
fl->c.flc_type = deleg_is_read(dp->dl_type) ? F_RDLCK : F_WRLCK;
fl->c.flc_owner = (fl_owner_t)dp;
fl->c.flc_pid = current->tgid;
- fl->c.flc_file = dp->dl_stid.sc_file->fi_deleg_file->nf_file;
+ fl->c.flc_file = rcu_dereference_protected(dp->dl_stid.sc_file->fi_deleg_file, 1)->nf_file;
return fl;
}
@@ -6063,7 +6426,7 @@ static int nfsd4_check_conflicting_opens(struct nfs4_client *clp,
struct nfs4_file *fp)
{
struct nfs4_ol_stateid *st;
- struct file *f = fp->fi_deleg_file->nf_file;
+ struct file *f = rcu_dereference_protected(fp->fi_deleg_file, 1)->nf_file;
struct inode *ino = file_inode(f);
int writes;
@@ -6140,7 +6503,7 @@ nfsd4_verify_deleg_dentry(struct nfsd4_open *open, struct nfs4_file *fp,
exp_put(exp);
dput(child);
- if (child != file_dentry(fp->fi_deleg_file->nf_file))
+ if (child != file_dentry(rcu_dereference_protected(fp->fi_deleg_file, 1)->nf_file))
return -EAGAIN;
return 0;
@@ -6246,8 +6609,9 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp,
status = -EAGAIN;
else if (nfsd4_verify_setuid_write(open, nf))
status = -EAGAIN;
- else if (!fp->fi_deleg_file) {
- fp->fi_deleg_file = nf;
+ else if (!rcu_dereference_protected(fp->fi_deleg_file,
+ lockdep_is_held(&fp->fi_lock))) {
+ rcu_assign_pointer(fp->fi_deleg_file, nf);
/* increment early to prevent fi_deleg_file from being
* cleared */
fp->fi_delegees = 1;
@@ -6268,11 +6632,11 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp,
if (stp->st_stid.sc_export)
dp->dl_stid.sc_export = exp_get(stp->st_stid.sc_export);
- fl = nfs4_alloc_init_lease(dp);
+ fl = nfs4_alloc_init_lease(dp, 0);
if (!fl)
goto out_clnt_odstate;
- status = kernel_setlease(fp->fi_deleg_file->nf_file,
+ status = kernel_setlease(rcu_dereference_protected(fp->fi_deleg_file, 1)->nf_file,
fl->c.flc_type, &fl, NULL);
if (fl)
locks_free_lease(fl);
@@ -6293,7 +6657,7 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp,
* Now that the deleg is set, check again to ensure that nothing
* raced in and changed the mode while we weren't looking.
*/
- status = nfsd4_verify_setuid_write(open, fp->fi_deleg_file);
+ status = nfsd4_verify_setuid_write(open, rcu_dereference_protected(fp->fi_deleg_file, 1));
if (status)
goto out_unlock;
@@ -6314,7 +6678,8 @@ nfs4_set_delegation(struct nfsd4_open *open, struct nfs4_ol_stateid *stp,
return dp;
out_unlock:
- kernel_setlease(fp->fi_deleg_file->nf_file, F_UNLCK, NULL, (void **)&dp);
+ kernel_setlease(rcu_dereference_protected(fp->fi_deleg_file, 1)->nf_file,
+ F_UNLCK, NULL, (void **)&dp);
out_clnt_odstate:
put_clnt_odstate(dp->dl_clnt_odstate);
nfs4_put_stid(&dp->dl_stid);
@@ -6471,8 +6836,9 @@ nfs4_open_delegation(struct svc_rqst *rqstp, struct nfsd4_open *open,
memcpy(&open->op_delegate_stateid, &dp->dl_stid.sc_stateid, sizeof(dp->dl_stid.sc_stateid));
if (open->op_share_access & NFS4_SHARE_ACCESS_WRITE) {
- struct file *f = dp->dl_stid.sc_file->fi_deleg_file->nf_file;
+ struct file *f;
+ f = rcu_dereference_protected(dp->dl_stid.sc_file->fi_deleg_file, 1)->nf_file;
if (!nfsd4_add_rdaccess_to_wrdeleg(rqstp, open, fh, stp) ||
!nfs4_delegation_stat(dp, currentfh, &stat)) {
nfs4_put_stid(&dp->dl_stid);
@@ -6728,12 +7094,21 @@ nfsd4_renew(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
static void
nfsd4_end_grace(struct nfsd_net *nn)
{
- /* do nothing if grace period already ended */
- if (nn->grace_ended)
+ /*
+ * nfsd4_end_grace() can be entered concurrently from the
+ * laundromat workqueue and from an nfsd compound thread
+ * handling RECLAIM_COMPLETE. Without serialization, both
+ * callers can observe NFSD_NET_GRACE_ENDED clear and proceed
+ * into nfsd4_record_grace_done(). For tracking ops whose
+ * grace_done drains reclaim_str_hashtbl, that results in
+ * list corruption and a double free of every
+ * nfs4_client_reclaim entry. Use an atomic test-and-set so
+ * exactly one caller proceeds.
+ */
+ if (test_and_set_bit(NFSD_NET_GRACE_ENDED, &nn->flags))
return;
trace_nfsd_grace_complete(nn);
- nn->grace_ended = true;
/*
* If the server goes down again right now, an NFSv4
* client will still be allowed to reclaim after it comes back up,
@@ -6774,10 +7149,10 @@ bool nfsd4_force_end_grace(struct nfsd_net *nn)
{
if (!nn->client_tracking_ops)
return false;
- if (READ_ONCE(nn->grace_ended))
+ if (test_bit(NFSD_NET_GRACE_ENDED, &nn->flags))
return false;
/* laundromat_work must be initialised now, though it might be disabled */
- WRITE_ONCE(nn->grace_end_forced, true);
+ set_bit(NFSD_NET_GRACE_END_FORCED, &nn->flags);
/* mod_delayed_work() doesn't queue work after
* nfs4_state_shutdown_net() has called disable_delayed_work_sync()
*/
@@ -6791,18 +7166,22 @@ bool nfsd4_force_end_grace(struct nfsd_net *nn)
*/
static bool clients_still_reclaiming(struct nfsd_net *nn)
{
- time64_t double_grace_period_end = nn->boot_time +
+ time64_t double_grace_period_end = nn->boot_time_bt +
2 * nn->nfsd4_lease;
- if (READ_ONCE(nn->grace_end_forced))
+ if (test_bit(NFSD_NET_GRACE_END_FORCED, &nn->flags))
return false;
- if (nn->track_reclaim_completes &&
- atomic_read(&nn->nr_reclaim_complete) ==
- nn->reclaim_str_hashtbl_size)
- return false;
- if (!nn->somebody_reclaimed)
+ if (test_bit(NFSD_NET_TRACK_RECLAIM_COMPLETES, &nn->flags)) {
+ int size;
+
+ down_read(&nn->reclaim_str_hashtbl_lock);
+ size = nn->reclaim_str_hashtbl_size;
+ up_read(&nn->reclaim_str_hashtbl_lock);
+ if (atomic_read(&nn->nr_reclaim_complete) == size)
+ return false;
+ }
+ if (!test_and_clear_bit(NFSD_NET_SOMEBODY_RECLAIMED, &nn->flags))
return false;
- nn->somebody_reclaimed = false;
/*
* If we've given them *two* lease times to reclaim, and they're
* still not done, give up:
@@ -6859,30 +7238,36 @@ static void nfsd4_ssc_shutdown_umount(struct nfsd_net *nn)
static void nfsd4_ssc_expire_umount(struct nfsd_net *nn)
{
bool do_wakeup = false;
- struct nfsd4_ssc_umount_item *ni = NULL;
- struct nfsd4_ssc_umount_item *tmp;
+ struct nfsd4_ssc_umount_item *ni;
+restart:
spin_lock(&nn->nfsd_ssc_lock);
- list_for_each_entry_safe(ni, tmp, &nn->nfsd_ssc_mount_list, nsui_list) {
- if (time_after(jiffies, ni->nsui_expire)) {
- if (refcount_read(&ni->nsui_refcnt) > 1)
- continue;
+ list_for_each_entry(ni, &nn->nfsd_ssc_mount_list, nsui_list) {
+ if (!time_after(jiffies, ni->nsui_expire))
+ break;
+ if (refcount_read(&ni->nsui_refcnt) > 1)
+ continue;
- /* mark being unmount */
- ni->nsui_busy = true;
- spin_unlock(&nn->nfsd_ssc_lock);
- mntput(ni->nsui_vfsmount);
- spin_lock(&nn->nfsd_ssc_lock);
+ /* Prevent concurrent setup during unmount */
+ ni->nsui_busy = true;
+ spin_unlock(&nn->nfsd_ssc_lock);
+ mntput(ni->nsui_vfsmount);
+ spin_lock(&nn->nfsd_ssc_lock);
- /* waiters need to start from begin of list */
- list_del(&ni->nsui_list);
- kfree(ni);
+ /* Force concurrent scanners to restart */
+ list_del(&ni->nsui_list);
+ kfree(ni);
- /* wakeup ssc_connect waiters */
- do_wakeup = true;
- continue;
- }
- break;
+ /* wakeup ssc_connect waiters */
+ do_wakeup = true;
+ /*
+ * Concurrent nfsd4_ssc_cancel_dul() can free any item
+ * on the list under nfsd_ssc_lock while mntput() runs
+ * above. Restart from the head; the list is short and
+ * the expire worker is periodic, so this is cheap.
+ */
+ spin_unlock(&nn->nfsd_ssc_lock);
+ goto restart;
}
if (do_wakeup)
wake_up_all(&nn->nfsd_ssc_waitq);
@@ -7195,12 +7580,12 @@ deleg_reaper(struct nfsd_net *nn)
continue;
if (atomic_read(&clp->cl_delegs_in_recall))
continue;
- if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &clp->cl_ra->ra_cb.cb_flags))
- continue;
if (ktime_get_boottime_seconds() - clp->cl_ra_time < 5)
continue;
if (clp->cl_cb_state != NFSD4_CB_UP)
continue;
+ if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &clp->cl_ra->ra_cb.cb_flags))
+ continue;
/* release in nfsd4_cb_recall_any_release */
kref_get(&clp->cl_nfsdfs.cl_ref);
@@ -7787,7 +8172,7 @@ retry:
return status;
stp = openlockstateid(s);
if (nfsd4_cstate_assign_replay(cstate, stp->st_stateowner) == -EAGAIN) {
- nfs4_put_stateowner(stp->st_stateowner);
+ nfs4_put_stid(&stp->st_stid);
goto retry;
}
@@ -8036,6 +8421,10 @@ nfsd4_delegreturn(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
if (status)
goto put_stateid;
+ status = nfs4_check_fh(&cstate->current_fh, &dp->dl_stid);
+ if (status)
+ goto put_stateid;
+
trace_nfsd_deleg_return(stateid);
destroy_delegation(dp);
smp_mb__after_atomic();
@@ -8506,6 +8895,9 @@ nfsd4_lock(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
status = nfserr_no_grace;
if (!locks_in_grace(net) && lock->lk_reclaim)
goto out;
+ if (lock->lk_reclaim &&
+ test_bit(NFSD4_CLIENT_RECLAIM_COMPLETE, &cstate->clp->cl_flags))
+ goto out;
if (lock->lk_reclaim)
flags |= FL_RECLAIM;
@@ -8542,10 +8934,11 @@ nfsd4_lock(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
goto out;
}
- if (lock->lk_type & (NFS4_READW_LT | NFS4_WRITEW_LT) &&
- nfsd4_has_session(cstate) &&
- locks_can_async_lock(nf->nf_file->f_op))
- flags |= FL_SLEEP;
+ if ((lock->lk_type == NFS4_READW_LT ||
+ lock->lk_type == NFS4_WRITEW_LT) &&
+ nfsd4_has_session(cstate) &&
+ locks_can_async_lock(nf->nf_file->f_op))
+ flags |= FL_SLEEP;
nbl = find_or_allocate_block(lock_sop, &fp->fi_fhandle, nn);
if (!nbl) {
@@ -8587,7 +8980,7 @@ nfsd4_lock(struct svc_rqst *rqstp, struct nfsd4_compound_state *cstate,
nfs4_inc_and_copy_stateid(&lock->lk_resp_stateid, &lock_stp->st_stid);
status = 0;
if (lock->lk_reclaim)
- nn->somebody_reclaimed = true;
+ set_bit(NFSD_NET_SOMEBODY_RECLAIMED, &nn->flags);
break;
case FILE_LOCK_DEFERRED:
kref_put(&nbl->nbl_kref, free_nbl);
@@ -8963,9 +9356,13 @@ bool
nfs4_has_reclaimed_state(struct xdr_netobj name, struct nfsd_net *nn)
{
struct nfs4_client_reclaim *crp;
+ bool found;
+ down_read(&nn->reclaim_str_hashtbl_lock);
crp = nfsd4_find_reclaim_client(name, nn);
- return (crp && crp->cr_clp);
+ found = (crp && crp->cr_clp);
+ up_read(&nn->reclaim_str_hashtbl_lock);
+ return found;
}
/*
@@ -8978,10 +9375,39 @@ nfs4_client_to_reclaim(struct xdr_netobj name, struct xdr_netobj princhash,
unsigned int strhashval;
struct nfs4_client_reclaim *crp;
+ down_write(&nn->reclaim_str_hashtbl_lock);
+
+ /*
+ * A reclaim record for this client name may already exist (for
+ * example, populated at boot from the recovery directory before
+ * an in-grace RECLAIM_COMPLETE or an nfsdcld downcall delivers
+ * the same name). Dedup here so reclaim_str_hashtbl_size stays
+ * equal to the number of distinct client names; inc_reclaim_complete
+ * relies on that equality to end the grace period via the fast path.
+ */
+ crp = nfsd4_find_reclaim_client(name, nn);
+ if (crp) {
+ if (princhash.len && crp->cr_princhash.len == 0) {
+ void *pdata = kmemdup(princhash.data, princhash.len,
+ GFP_KERNEL);
+ if (pdata) {
+ crp->cr_princhash.data = pdata;
+ crp->cr_princhash.len = princhash.len;
+ } else {
+ dprintk("%s: failed to allocate memory for princhash.data!\n",
+ __func__);
+ crp = NULL;
+ }
+ }
+ up_write(&nn->reclaim_str_hashtbl_lock);
+ return crp;
+ }
+
name.data = kmemdup(name.data, name.len, GFP_KERNEL);
if (!name.data) {
dprintk("%s: failed to allocate memory for name.data!\n",
__func__);
+ up_write(&nn->reclaim_str_hashtbl_lock);
return NULL;
}
if (princhash.len) {
@@ -8990,6 +9416,7 @@ nfs4_client_to_reclaim(struct xdr_netobj name, struct xdr_netobj princhash,
dprintk("%s: failed to allocate memory for princhash.data!\n",
__func__);
kfree(name.data);
+ up_write(&nn->reclaim_str_hashtbl_lock);
return NULL;
}
} else
@@ -9009,6 +9436,7 @@ nfs4_client_to_reclaim(struct xdr_netobj name, struct xdr_netobj princhash,
kfree(name.data);
kfree(princhash.data);
}
+ up_write(&nn->reclaim_str_hashtbl_lock);
return crp;
}
@@ -9028,6 +9456,7 @@ nfs4_release_reclaim(struct nfsd_net *nn)
struct nfs4_client_reclaim *crp = NULL;
int i;
+ down_write(&nn->reclaim_str_hashtbl_lock);
for (i = 0; i < CLIENT_HASH_SIZE; i++) {
while (!list_empty(&nn->reclaim_str_hashtbl[i])) {
crp = list_entry(nn->reclaim_str_hashtbl[i].next,
@@ -9036,6 +9465,7 @@ nfs4_release_reclaim(struct nfsd_net *nn)
}
}
WARN_ON_ONCE(nn->reclaim_str_hashtbl_size);
+ up_write(&nn->reclaim_str_hashtbl_lock);
}
/*
@@ -9113,8 +9543,9 @@ static int nfs4_state_create_net(struct net *net)
nn->conf_name_tree = RB_ROOT;
nn->unconf_name_tree = RB_ROOT;
nn->boot_time = ktime_get_real_seconds();
- nn->grace_ended = false;
- nn->grace_end_forced = false;
+ nn->boot_time_bt = ktime_get_boottime_seconds();
+ clear_bit(NFSD_NET_GRACE_ENDED, &nn->flags);
+ clear_bit(NFSD_NET_GRACE_END_FORCED, &nn->flags);
nn->nfsd4_manager.block_opens = true;
INIT_LIST_HEAD(&nn->nfsd4_manager.list);
INIT_LIST_HEAD(&nn->client_lru);
@@ -9200,7 +9631,8 @@ nfs4_state_start_net(struct net *net)
nfsd4_client_tracking_init(net);
/* safe for laundromat to run now */
enable_delayed_work(&nn->laundromat_work);
- if (nn->track_reclaim_completes && nn->reclaim_str_hashtbl_size == 0)
+ if (test_bit(NFSD_NET_TRACK_RECLAIM_COMPLETES, &nn->flags) &&
+ nn->reclaim_str_hashtbl_size == 0)
goto skip_grace;
printk(KERN_INFO "NFSD: starting %lld-second grace period (net %x)\n",
nn->nfsd4_grace, net->ns.inum);
@@ -9231,8 +9663,8 @@ nfs4_state_start(void)
rhltable_destroy(&nfs4_file_rhltable);
return -ENOMEM;
}
- nfsd_slot_shrinker->count_objects = nfsd_slot_count;
- nfsd_slot_shrinker->scan_objects = nfsd_slot_scan;
+ nfsd_slot_shrinker->count_objects = nfsd_slot_shrinker_count;
+ nfsd_slot_shrinker->scan_objects = nfsd_slot_shrinker_scan;
shrinker_register(nfsd_slot_shrinker);
set_max_delegations();
@@ -9569,6 +10001,30 @@ out_status:
return status;
}
+#define GDD_WORD0_CHILD_ATTRS (FATTR4_WORD0_TYPE | \
+ FATTR4_WORD0_CHANGE | \
+ FATTR4_WORD0_SIZE | \
+ FATTR4_WORD0_FILEID | \
+ FATTR4_WORD0_FILEHANDLE)
+
+#define GDD_WORD1_CHILD_ATTRS (FATTR4_WORD1_MODE | \
+ FATTR4_WORD1_NUMLINKS | \
+ FATTR4_WORD1_RAWDEV | \
+ FATTR4_WORD1_SPACE_USED | \
+ FATTR4_WORD1_TIME_ACCESS | \
+ FATTR4_WORD1_TIME_METADATA | \
+ FATTR4_WORD1_TIME_MODIFY | \
+ FATTR4_WORD1_TIME_CREATE)
+
+#define GDD_WORD0_DIR_ATTRS (FATTR4_WORD0_CHANGE | \
+ FATTR4_WORD0_SIZE)
+
+#define GDD_WORD1_DIR_ATTRS (FATTR4_WORD1_NUMLINKS | \
+ FATTR4_WORD1_SPACE_USED | \
+ FATTR4_WORD1_TIME_ACCESS | \
+ FATTR4_WORD1_TIME_METADATA | \
+ FATTR4_WORD1_TIME_MODIFY)
+
/**
* nfsd_get_dir_deleg - attempt to get a directory delegation
* @cstate: compound state
@@ -9576,8 +10032,7 @@ out_status:
* @nf: nfsd_file opened on the directory
*
* Given a GET_DIR_DELEGATION request @gdd, attempt to acquire a delegation
- * on the directory to which @nf refers. Note that this does not set up any
- * sort of async notifications for the delegation.
+ * on the directory to which @nf refers.
*/
struct nfs4_delegation *
nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate,
@@ -9614,8 +10069,9 @@ nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate,
/* existing delegation? */
if (nfs4_delegation_exists(clp, fp)) {
status = -EAGAIN;
- } else if (!fp->fi_deleg_file) {
- fp->fi_deleg_file = nfsd_file_get(nf);
+ } else if (!rcu_dereference_protected(fp->fi_deleg_file,
+ lockdep_is_held(&fp->fi_lock))) {
+ rcu_assign_pointer(fp->fi_deleg_file, nfsd_file_get(nf));
fp->fi_delegees = 1;
} else {
++fp->fi_delegees;
@@ -9630,14 +10086,24 @@ nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate,
/* Try to set up the lease */
status = -ENOMEM;
- dp = alloc_init_deleg(clp, fp, NULL, NFS4_OPEN_DELEGATE_READ);
+ dp = alloc_init_dir_deleg(clp, fp);
if (!dp)
goto out_delegees;
if (cstate->current_fh.fh_export)
dp->dl_stid.sc_export =
exp_get(cstate->current_fh.fh_export);
- fl = nfs4_alloc_init_lease(dp);
+ /*
+ * NB: gddr_notification[0] represents the notifications that
+ * will be granted to the client
+ */
+ dp->dl_notify_mask = gdd->gddr_notification[0];
+ dp->dl_child_attrs[0] = gdd->gdda_child_attributes[0] & GDD_WORD0_CHILD_ATTRS;
+ dp->dl_child_attrs[1] = gdd->gdda_child_attributes[1] & GDD_WORD1_CHILD_ATTRS;
+ dp->dl_dir_attrs[0] = gdd->gdda_dir_attributes[0] & GDD_WORD0_DIR_ATTRS;
+ dp->dl_dir_attrs[1] = gdd->gdda_dir_attributes[1] & GDD_WORD1_DIR_ATTRS;
+
+ fl = nfs4_alloc_init_lease(dp, dp->dl_notify_mask);
if (!fl)
goto out_put_stid;
@@ -9663,11 +10129,22 @@ nfsd_get_dir_deleg(struct nfsd4_compound_state *cstate,
if (!status) {
put_nfs4_file(fp);
+ nfsd_fsnotify_recalc_mask(nf);
return dp;
}
- /* Something failed. Drop the lease and clean up the stid */
- kernel_setlease(fp->fi_deleg_file->nf_file, F_UNLCK, NULL, (void **)&dp);
+ /*
+ * Something failed after the lease was set. Drop the lease and clean
+ * up the stid. The lease's flc_file is the fi_deleg_file (see
+ * nfs4_alloc_init_lease()), which is not necessarily this client's
+ * @nf when an earlier client already holds a delegation on @fp.
+ * generic_delete_lease() matches on flc_file, so unlock against
+ * fi_deleg_file or the lease will be leaked (and later freed with the
+ * stid, leading to a use-after-free when it's eventually broken).
+ */
+ kernel_setlease(rcu_dereference_protected(fp->fi_deleg_file, 1)->nf_file,
+ F_UNLCK, NULL, (void **)&dp);
+ nfsd_fsnotify_recalc_mask(nf);
out_put_stid:
nfs4_put_stid(&dp->dl_stid);
out_delegees:
@@ -9703,3 +10180,170 @@ void nfsd_update_cmtime_attr(struct file *f, unsigned int flags)
MINOR(inode->i_sb->s_dev),
inode->i_ino, ret);
}
+
+static void
+nfsd4_run_cb_notify(struct nfsd4_cb_notify *ncn)
+{
+ struct nfs4_delegation *dp = container_of(ncn, struct nfs4_delegation, dl_cb_notify);
+
+ if (test_and_set_bit(NFSD4_CALLBACK_RUNNING, &ncn->ncn_cb.cb_flags))
+ return;
+
+ if (!refcount_inc_not_zero(&dp->dl_stid.sc_count))
+ clear_bit(NFSD4_CALLBACK_RUNNING, &ncn->ncn_cb.cb_flags);
+ else
+ nfsd4_run_cb(&ncn->ncn_cb);
+}
+
+static struct nfsd_notify_event *
+alloc_nfsd_notify_event(u32 mask, const struct qstr *q, struct dentry *dentry,
+ struct inode *target)
+{
+ struct nfsd_notify_event *ne;
+ struct name_snapshot newname;
+ u32 newnamelen = 0;
+
+ /*
+ * For a rename, @q is the old name and the live dentry carries the new
+ * name. Snapshot the new name now, while it is guaranteed to describe
+ * this event: the dentry can be renamed again before the CB_NOTIFY work
+ * runs, which would corrupt a late read in nfsd4_encode_notify_event().
+ */
+ if (mask & FS_RENAME) {
+ take_dentry_name_snapshot(&newname, dentry);
+ newnamelen = newname.name.len;
+ }
+
+ ne = kmalloc(struct_size(ne, ne_name, q->len + 1 +
+ (newnamelen ? newnamelen + 1 : 0)), GFP_NOFS);
+ if (!ne)
+ goto out;
+
+ memcpy(ne->ne_name, q->name, q->len);
+ ne->ne_name[q->len] = '\0';
+ ne->ne_namelen = q->len;
+
+ ne->ne_newnamelen = newnamelen;
+ if (newnamelen) {
+ char *p = nfsd_notify_event_newname(ne);
+
+ memcpy(p, newname.name.name, newnamelen);
+ p[newnamelen] = '\0';
+ }
+
+ refcount_set(&ne->ne_ref, 1);
+ ne->ne_mask = mask;
+ ne->ne_dentry = dget(dentry);
+ ne->ne_target = target;
+ if (ne->ne_target)
+ ihold(ne->ne_target);
+out:
+ if (mask & FS_RENAME)
+ release_dentry_name_snapshot(&newname);
+ return ne;
+}
+
+static bool
+should_notify_deleg(u32 mask, struct file_lease *fl)
+{
+ /* Don't notify the client generating the event */
+ if (nfsd_breaker_owns_lease(fl))
+ return false;
+
+ /* Skip if this event wasn't ignored by the lease */
+ if ((mask & FS_DELETE) && !(fl->c.flc_flags & FL_IGN_DIR_DELETE))
+ return false;
+ if ((mask & FS_CREATE) && !(fl->c.flc_flags & FL_IGN_DIR_CREATE))
+ return false;
+ if ((mask & FS_RENAME) && !(fl->c.flc_flags & FL_IGN_DIR_RENAME))
+ return false;
+
+ return true;
+}
+
+static void
+nfsd_recall_all_dir_delegs(const struct inode *dir)
+{
+ struct file_lock_context *ctx = locks_inode_context(dir);
+ struct file_lock_core *flc;
+
+ spin_lock(&ctx->flc_lock);
+ list_for_each_entry(flc, &ctx->flc_lease, flc_list) {
+ struct file_lease *fl = container_of(flc, struct file_lease, c);
+
+ if (fl->fl_lmops == &nfsd_lease_mng_ops)
+ nfsd_break_deleg_cb(fl);
+ }
+ spin_unlock(&ctx->flc_lock);
+}
+
+int
+nfsd_handle_dir_event(u32 mask, const struct inode *dir, const void *data,
+ int data_type, const struct qstr *name)
+{
+ struct dentry *dentry = fsnotify_data_dentry(data, data_type);
+ struct inode *target = fsnotify_data_rename_target(data, data_type);
+ struct file_lock_context *ctx;
+ struct file_lock_core *flc;
+ struct nfsd_notify_event *evt;
+
+ trace_nfsd_handle_dir_event(mask, dir, name);
+
+ /* Normalize cross-dir rename events to create/delete */
+ if (mask & FS_MOVED_FROM) {
+ mask &= ~FS_MOVED_FROM;
+ mask |= FS_DELETE;
+ }
+ if (mask & FS_MOVED_TO) {
+ mask &= ~FS_MOVED_TO;
+ mask |= FS_CREATE;
+ }
+
+ /*
+ * FS_RENAME fires on the source directory even for a cross-dir
+ * rename, where the moved entry now lives under a different parent.
+ * NOTIFY4_RENAME_ENTRY describes an in-place rename, so reporting it
+ * here would advertise a name absent from this directory.
+ */
+ if ((mask & FS_RENAME) && dentry && d_inode(dentry->d_parent) != dir)
+ mask &= ~FS_RENAME;
+
+ /* Don't do anything if this is not an expected event */
+ if (!(mask & (FS_CREATE|FS_DELETE|FS_RENAME)))
+ return 0;
+
+ ctx = locks_inode_context(dir);
+ if (!ctx || list_empty(&ctx->flc_lease))
+ return 0;
+
+ evt = alloc_nfsd_notify_event(mask, name, dentry, target);
+ if (!evt) {
+ nfsd_recall_all_dir_delegs(dir);
+ return 0;
+ }
+
+ spin_lock(&ctx->flc_lock);
+ list_for_each_entry(flc, &ctx->flc_lease, flc_list) {
+ struct file_lease *fl = container_of(flc, struct file_lease, c);
+ struct nfs4_delegation *dp = flc->flc_owner;
+ struct nfsd4_cb_notify *ncn = &dp->dl_cb_notify;
+
+ if (!should_notify_deleg(mask, fl))
+ continue;
+
+ spin_lock(&ncn->ncn_lock);
+ if (ncn->ncn_evt_cnt >= NOTIFY4_EVENT_QUEUE_SIZE) {
+ /* We're generating notifications too fast. Recall. */
+ spin_unlock(&ncn->ncn_lock);
+ nfsd_break_deleg_cb(fl);
+ continue;
+ }
+ ncn->ncn_evt[ncn->ncn_evt_cnt++] = nfsd_notify_event_get(evt);
+ spin_unlock(&ncn->ncn_lock);
+
+ nfsd4_run_cb_notify(ncn);
+ }
+ spin_unlock(&ctx->flc_lock);
+ nfsd_notify_event_put(evt);
+ return 0;
+}
diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c
index e17488a911f7..aef48fb0fac2 100644
--- a/fs/nfsd/nfs4xdr.c
+++ b/fs/nfsd/nfs4xdr.c
@@ -98,7 +98,7 @@ check_filename(char *str, int len)
return nfserr_inval;
if (len > NFS4_MAXNAMLEN)
return nfserr_nametoolong;
- if (isdotent(str, len))
+ if (name_is_dot_dotdot(str, len))
return nfserr_badname;
for (i = 0; i < len; i++)
if (str[i] == '/')
@@ -244,7 +244,7 @@ nfsd4_decode_nfstime4(struct nfsd4_compoundargs *argp, struct timespec64 *tv)
return nfserr_bad_xdr;
p = xdr_decode_hyper(p, &tv->tv_sec);
tv->tv_nsec = be32_to_cpup(p++);
- if (tv->tv_nsec >= (u32)1000000000)
+ if ((unsigned long)tv->tv_nsec >= NSEC_PER_SEC)
return nfserr_inval;
return nfs_ok;
}
@@ -449,9 +449,18 @@ nfsd4_decode_posixacl(struct nfsd4_compoundargs *argp, struct posix_acl **acl)
if (xdr_stream_decode_u32(argp->xdr, &count) < 0)
return nfserr_bad_xdr;
+ /*
+ * The NFSv4 POSIX ACL draft doesn't define a max number of ACE's, but
+ * the NFSACL spec does. For NFSv4, cap the number of entries to the v3
+ * limit, as we want to ensure that ACLs set via NFSv4 POSIX ACL
+ * extensions are retrievable via NFSACL.
+ */
+ if (count > NFS_ACL_MAX_ENTRIES)
+ return nfserr_inval;
+
*acl = posix_acl_alloc(count, GFP_KERNEL);
if (*acl == NULL)
- return nfserr_resource;
+ return nfserr_jukebox;
(*acl)->a_count = count;
for (ace = (*acl)->a_entries; ace < (*acl)->a_entries + count; ace++) {
@@ -628,6 +637,8 @@ nfsd4_decode_fattr4(struct nfsd4_compoundargs *argp, u32 *bmval, u32 bmlen,
if (!xdrgen_decode_fattr4_time_deleg_access(argp->xdr, &access))
return nfserr_bad_xdr;
+ if (access.nseconds >= NSEC_PER_SEC)
+ return nfserr_inval;
iattr->ia_atime.tv_sec = access.seconds;
iattr->ia_atime.tv_nsec = access.nseconds;
iattr->ia_valid |= ATTR_ATIME | ATTR_ATIME_SET | ATTR_DELEG;
@@ -637,6 +648,8 @@ nfsd4_decode_fattr4(struct nfsd4_compoundargs *argp, u32 *bmval, u32 bmlen,
if (!xdrgen_decode_fattr4_time_deleg_modify(argp->xdr, &modify))
return nfserr_bad_xdr;
+ if (modify.nseconds >= NSEC_PER_SEC)
+ return nfserr_inval;
iattr->ia_mtime.tv_sec = modify.seconds;
iattr->ia_mtime.tv_nsec = modify.nseconds;
iattr->ia_ctime.tv_sec = modify.seconds;
@@ -955,6 +968,10 @@ nfsd4_decode_create(struct nfsd4_compoundargs *argp, union nfsd4_op_u *u)
case NF4LNK:
if (xdr_stream_decode_u32(argp->xdr, &create->cr_datalen) < 0)
return nfserr_bad_xdr;
+ if (create->cr_datalen == 0)
+ return nfserr_inval;
+ if (create->cr_datalen > NFS4_MAXPATHLEN)
+ return nfserr_nametoolong;
p = xdr_inline_decode(argp->xdr, create->cr_datalen);
if (!p)
return nfserr_bad_xdr;
@@ -2702,7 +2719,7 @@ nfsd4_decode_compound(struct nfsd4_compoundargs *argp)
}
static __be32 nfsd4_encode_nfs_fh4(struct xdr_stream *xdr,
- struct knfsd_fh *fh_handle)
+ const struct knfsd_fh *fh_handle)
{
return nfsd4_encode_opaque(xdr, fh_handle->fh_raw, fh_handle->fh_size);
}
@@ -3142,9 +3159,9 @@ out_resource:
struct nfsd4_fattr_args {
struct svc_rqst *rqstp;
- struct svc_fh *fhp;
struct svc_export *exp;
struct dentry *dentry;
+ struct knfsd_fh fhandle;
struct kstat stat;
struct kstatfs statfs;
struct nfs4_acl *acl;
@@ -3260,7 +3277,7 @@ static __be32 nfsd4_encode_fattr4_change(struct xdr_stream *xdr,
{
const struct svc_export *exp = args->exp;
- if (unlikely(exp->ex_flags & NFSEXP_V4ROOT)) {
+ if (exp && unlikely(exp->ex_flags & NFSEXP_V4ROOT)) {
u32 flush_time = convert_to_wallclock(exp->cd->flush_time);
if (xdr_stream_encode_u32(xdr, flush_time) != XDR_UNIT)
@@ -3292,7 +3309,7 @@ static __be32 nfsd4_encode_fattr4_fsid(struct xdr_stream *xdr,
xdr_encode_hyper(p, NFS4_REFERRAL_FSID_MINOR);
return nfs_ok;
}
- switch (fsid_source(args->fhp)) {
+ switch (fsid_source_fh(&args->fhandle, args->exp)) {
case FSIDSOURCE_FSID:
p = xdr_encode_hyper(p, (u64)args->exp->ex_fsid);
xdr_encode_hyper(p, (u64)0);
@@ -3389,7 +3406,7 @@ static __be32 nfsd4_encode_fattr4_homogeneous(struct xdr_stream *xdr,
static __be32 nfsd4_encode_fattr4_filehandle(struct xdr_stream *xdr,
const struct nfsd4_fattr_args *args)
{
- return nfsd4_encode_nfs_fh4(xdr, &args->fhp->fh_handle);
+ return nfsd4_encode_nfs_fh4(xdr, &args->fhandle);
}
static __be32 nfsd4_encode_fattr4_fileid(struct xdr_stream *xdr,
@@ -3882,6 +3899,22 @@ static const nfsd4_enc_attr nfsd4_enc_fattr4_encode_ops[] = {
#endif
};
+static __be32
+nfsd4_encode_attr_vals(struct xdr_stream *xdr, u32 *attrmask, struct nfsd4_fattr_args *args)
+{
+ DECLARE_BITMAP(attr_bitmap, ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops));
+ unsigned long bit;
+ __be32 status;
+
+ bitmap_from_arr32(attr_bitmap, attrmask, ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops));
+ for_each_set_bit(bit, attr_bitmap, ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops)) {
+ status = nfsd4_enc_fattr4_encode_ops[bit](xdr, args);
+ if (status != nfs_ok)
+ return status;
+ }
+ return nfs_ok;
+}
+
/*
* Note: @fhp can be NULL; in this case, we might have to compose the filehandle
* ourselves. @case_cache is NULL for callers that encode a single dentry
@@ -3895,7 +3928,6 @@ nfsd4_encode_fattr4(struct svc_rqst *rqstp, struct xdr_stream *xdr,
int ignore_crossmnt,
struct nfsd_case_attrs_cache *case_cache)
{
- DECLARE_BITMAP(attr_bitmap, ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops));
struct nfs4_delegation *dp = NULL;
struct nfsd4_fattr_args args;
struct svc_fh *tempfh = NULL;
@@ -3910,7 +3942,6 @@ nfsd4_encode_fattr4(struct svc_rqst *rqstp, struct xdr_stream *xdr,
.mnt = exp->ex_path.mnt,
.dentry = dentry,
};
- unsigned long bit;
WARN_ON_ONCE(bmval[1] & NFSD_WRITEONLY_ATTRS_WORD1);
WARN_ON_ONCE(!nfsd_attrs_supported(minorversion, bmval));
@@ -3988,19 +4019,22 @@ nfsd4_encode_fattr4(struct svc_rqst *rqstp, struct xdr_stream *xdr,
if (err)
goto out_nfserr;
}
- if ((attrmask[0] & (FATTR4_WORD0_FILEHANDLE | FATTR4_WORD0_FSID)) &&
- !fhp) {
- tempfh = kmalloc_obj(struct svc_fh);
- status = nfserr_jukebox;
- if (!tempfh)
- goto out;
- fh_init(tempfh, NFS4_FHSIZE);
- status = fh_compose(tempfh, exp, dentry, NULL);
- if (status)
- goto out;
- args.fhp = tempfh;
- } else
- args.fhp = fhp;
+
+ if ((attrmask[0] & (FATTR4_WORD0_FILEHANDLE | FATTR4_WORD0_FSID))) {
+ if (!fhp) {
+ tempfh = kmalloc_obj(struct svc_fh);
+ status = nfserr_jukebox;
+ if (!tempfh)
+ goto out;
+ fh_init(tempfh, NFS4_FHSIZE);
+ status = fh_compose(tempfh, exp, dentry, NULL);
+ if (status)
+ goto out;
+ fhp = tempfh;
+ }
+ fh_copy_shallow(&args.fhandle, &fhp->fh_handle);
+ }
+
if (attrmask[0] & (FATTR4_WORD0_CASE_INSENSITIVE |
FATTR4_WORD0_CASE_PRESERVING)) {
/*
@@ -4124,27 +4158,22 @@ nfsd4_encode_fattr4(struct svc_rqst *rqstp, struct xdr_stream *xdr,
#endif /* CONFIG_NFSD_V4_POSIX_ACLS */
/* attrmask */
- status = nfsd4_encode_bitmap4(xdr, attrmask[0], attrmask[1],
- attrmask[2]);
+ status = nfsd4_encode_bitmap4(xdr, attrmask[0], attrmask[1], attrmask[2]);
if (status)
goto out;
/* attr_vals */
attrlen_offset = xdr->buf->len;
- if (unlikely(!xdr_reserve_space(xdr, XDR_UNIT)))
- goto out_resource;
- bitmap_from_arr32(attr_bitmap, attrmask,
- ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops));
- for_each_set_bit(bit, attr_bitmap,
- ARRAY_SIZE(nfsd4_enc_fattr4_encode_ops)) {
- status = nfsd4_enc_fattr4_encode_ops[bit](xdr, &args);
- if (status != nfs_ok)
- goto out;
+ if (unlikely(!xdr_reserve_space(xdr, XDR_UNIT))) {
+ status = nfserr_resource;
+ goto out;
}
- attrlen = cpu_to_be32(xdr->buf->len - attrlen_offset - XDR_UNIT);
- write_bytes_to_xdr_buf(xdr->buf, attrlen_offset, &attrlen, XDR_UNIT);
- status = nfs_ok;
+ status = nfsd4_encode_attr_vals(xdr, attrmask, &args);
+ if (status == nfs_ok) {
+ attrlen = cpu_to_be32(xdr->buf->len - attrlen_offset - XDR_UNIT);
+ write_bytes_to_xdr_buf(xdr->buf, attrlen_offset, &attrlen, XDR_UNIT);
+ }
out:
#ifdef CONFIG_NFSD_V4_POSIX_ACLS
if (args.dpacl)
@@ -4167,9 +4196,286 @@ out:
out_nfserr:
status = nfserrno(err);
goto out;
-out_resource:
- status = nfserr_resource;
- goto out;
+}
+
+static bool
+setup_notify_fhandle(struct dentry *dentry, struct nfs4_delegation *dp,
+ struct nfsd_file *nf, struct nfsd4_fattr_args *args)
+{
+ struct nfs4_file *fi = dp->dl_stid.sc_file;
+ struct nfs4_client *clp = dp->dl_stid.sc_client;
+ int fileid_type, fsid_len, maxsize, flags = 0;
+ struct knfsd_fh *fhp = &args->fhandle;
+ struct inode *inode = d_inode(dentry);
+ struct inode *parent = NULL;
+ struct svc_export *exp;
+ struct fid *fid;
+ bool ret = false;
+
+ /*
+ * drop_stid_export() can clear sc_export under cl_lock and drop its
+ * reference when the delegation is admin-revoked, concurrently with
+ * this callback. Grab our own reference under cl_lock so the export
+ * can be neither NULL-raced nor freed while we encode.
+ */
+ spin_lock(&clp->cl_lock);
+ exp = dp->dl_stid.sc_export;
+ if (exp)
+ exp_get(exp);
+ spin_unlock(&clp->cl_lock);
+
+ fsid_len = key_len(fi->fi_fhandle.fh_fsid_type);
+ fhp->fh_size = 4 + fsid_len;
+
+ /* Copy first 4 bytes + fsid */
+ memcpy(&fhp->fh_raw, &fi->fi_fhandle.fh_raw, fhp->fh_size);
+
+ fid = (struct fid *)(fh_fsid(fhp) + fsid_len/4);
+ maxsize = (NFS4_FHSIZE - fhp->fh_size)/4;
+
+ /*
+ * Subtree-checking exports need a connectable filehandle so the
+ * parent can be resolved at decode time. Derive this from the
+ * delegation's export rather than the shared nfs4_file, which may
+ * have been initialized under a different export.
+ */
+ if (exp && !(exp->ex_flags & NFSEXP_NOSUBTREECHECK) &&
+ !S_ISDIR(inode->i_mode)) {
+ parent = d_inode(nf->nf_file->f_path.dentry);
+ flags = EXPORT_FH_CONNECTABLE;
+ }
+
+ fileid_type = exportfs_encode_inode_fh(inode, fid, &maxsize, parent, flags);
+ if (fileid_type < 0 || fileid_type == FILEID_INVALID)
+ goto out;
+
+ fhp->fh_fileid_type = fileid_type;
+ fhp->fh_size += maxsize * 4;
+
+ if (exp && (exp->ex_flags & NFSEXP_SIGN_FH))
+ if (!fh_append_mac(fhp, NFS4_FHSIZE, exp->cd->net))
+ goto out;
+
+ ret = true;
+out:
+ if (exp)
+ exp_put(exp);
+ return ret;
+}
+
+#define CB_NOTIFY_STATX_REQUEST_MASK (STATX_BASIC_STATS | \
+ STATX_BTIME | \
+ STATX_CHANGE_COOKIE)
+
+static bool
+nfsd4_setup_notify_entry4(struct notify_entry4 *ne, struct xdr_stream *xdr,
+ struct dentry *dentry, struct nfs4_delegation *dp,
+ struct nfsd_file *nf, char *name, u32 namelen)
+{
+ struct path path = nf->nf_file->f_path;
+ struct nfsd4_fattr_args args = { };
+ const u32 *reqmask;
+ uint32_t *attrmask;
+ __be32 status;
+ bool parent;
+ int ret;
+
+ /* Reserve space for attrmask */
+ attrmask = xdr_reserve_space(xdr, 3 * sizeof(uint32_t));
+ if (!attrmask)
+ return false;
+
+ ne->ne_file.data = name;
+ ne->ne_file.len = namelen;
+ ne->ne_attrs.attrmask.element = attrmask;
+
+ parent = (dentry == path.dentry);
+ path.dentry = dentry;
+ reqmask = parent ? dp->dl_dir_attrs : dp->dl_child_attrs;
+
+ /*
+ * A NULL or negative dentry has no attributes to report (expected,
+ * e.g. for the old entry of a rename or an entry already removed).
+ * The client may also have been granted the notification while
+ * requesting no attributes for this entry. Both cases encode an
+ * empty attribute set rather than failing: the vfs_getattr() and
+ * nfsd4_encode_attr_vals() failures below recall the delegation, so
+ * a case with nothing to fetch must short-circuit ahead of them.
+ */
+ if (!path.dentry || !d_inode(path.dentry) ||
+ (!reqmask[0] && !reqmask[1])) {
+ attrmask[0] = 0;
+ attrmask[1] = 0;
+ attrmask[2] = 0;
+ ne->ne_attrs.attr_vals.data = NULL;
+ ne->ne_attrs.attr_vals.len = 0;
+ ne->ne_attrs.attrmask.count = 1;
+ return true;
+ }
+
+ /*
+ * It is possible that the client was granted a delegation when a file
+ * was created. Note that we don't issue a CB_GETATTR here since stale
+ * attributes are presumably ok.
+ */
+ ret = vfs_getattr(&path, &args.stat, CB_NOTIFY_STATX_REQUEST_MASK, AT_STATX_SYNC_AS_STAT);
+ if (ret)
+ return false;
+
+ args.change_attr = nfsd4_change_attribute(&args.stat);
+
+ if (parent) {
+ attrmask[0] = dp->dl_dir_attrs[0];
+ attrmask[1] = dp->dl_dir_attrs[1];
+ } else {
+ attrmask[0] = dp->dl_child_attrs[0];
+ attrmask[1] = dp->dl_child_attrs[1];
+
+ if (!setup_notify_fhandle(dentry, dp, nf, &args))
+ attrmask[0] &= ~FATTR4_WORD0_FILEHANDLE;
+
+ if (!(args.stat.result_mask & STATX_BTIME))
+ attrmask[1] &= ~FATTR4_WORD1_TIME_CREATE;
+ }
+ attrmask[2] = 0;
+
+ ne->ne_attrs.attrmask.count = 2;
+ ne->ne_attrs.attr_vals.data = (u8 *)xdr->p;
+
+ status = nfsd4_encode_attr_vals(xdr, attrmask, &args);
+ if (status != nfs_ok)
+ return false;
+
+ ne->ne_attrs.attr_vals.len = (u8 *)xdr->p - ne->ne_attrs.attr_vals.data;
+ return true;
+}
+
+/**
+ * nfsd4_encode_notify_event - encode a notify
+ * @xdr: stream to which to encode the fattr4
+ * @nne: nfsd_notify_event to encode
+ * @dp: delegation where the event occurred
+ * @nf: nfsd_file on which event occurred
+ * @notify_mask: pointer to word where notification mask should be set
+ *
+ * Encode @nne into @xdr. The matching bit in @notify_mask is set on
+ * success.
+ *
+ * Return: pointer to the start of the encoded event, or NULL if the
+ * event could not be encoded.
+ */
+u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event *nne,
+ struct nfs4_delegation *dp, struct nfsd_file *nf,
+ u32 *notify_mask)
+{
+ u8 *p = NULL;
+
+ *notify_mask = 0;
+
+ if (nne->ne_mask & FS_DELETE) {
+ struct notify_remove4 nr = { };
+
+ if (!nfsd4_setup_notify_entry4(&nr.nrm_old_entry, xdr, nne->ne_dentry, dp,
+ nf, nne->ne_name, nne->ne_namelen))
+ goto out_err;
+ p = (u8 *)xdr->p;
+ if (!xdrgen_encode_notify_remove4(xdr, &nr))
+ goto out_err;
+ *notify_mask |= BIT(NOTIFY4_REMOVE_ENTRY);
+ } else if (nne->ne_mask & FS_CREATE) {
+ struct notify_add4 na = { };
+ struct notify_remove4 old = { };
+
+ if (!nfsd4_setup_notify_entry4(&na.nad_new_entry, xdr, nne->ne_dentry, dp,
+ nf, nne->ne_name, nne->ne_namelen))
+ goto out_err;
+
+ /* If a file was overwritten, report it in nad_old_entry */
+ if (nne->ne_target) {
+ if (!nfsd4_setup_notify_entry4(&old.nrm_old_entry, xdr,
+ NULL, dp, nf,
+ nne->ne_name, nne->ne_namelen))
+ goto out_err;
+ na.nad_old_entry.count = 1;
+ na.nad_old_entry.element = &old;
+ }
+
+ p = (u8 *)xdr->p;
+ if (!xdrgen_encode_notify_add4(xdr, &na))
+ goto out_err;
+
+ *notify_mask |= BIT(NOTIFY4_ADD_ENTRY);
+ } else if (nne->ne_mask & FS_RENAME) {
+ struct notify_rename4 nr = { };
+ struct notify_remove4 old = { };
+ char *newname = nfsd_notify_event_newname(nne);
+
+ /* Don't send any attributes in the old_entry since they're the same in new */
+ if (!nfsd4_setup_notify_entry4(&nr.nrn_old_entry.nrm_old_entry, xdr,
+ NULL, dp, nf, nne->ne_name,
+ nne->ne_namelen))
+ goto out_err;
+
+ if (!nfsd4_setup_notify_entry4(&nr.nrn_new_entry.nad_new_entry, xdr,
+ nne->ne_dentry, dp, nf, newname,
+ nne->ne_newnamelen))
+ goto out_err;
+
+ /* If a file was overwritten, report it in nad_old_entry */
+ if (nne->ne_target) {
+ if (!nfsd4_setup_notify_entry4(&old.nrm_old_entry, xdr,
+ NULL, dp, nf, newname,
+ nne->ne_newnamelen))
+ goto out_err;
+ nr.nrn_new_entry.nad_old_entry.count = 1;
+ nr.nrn_new_entry.nad_old_entry.element = &old;
+ }
+
+ p = (u8 *)xdr->p;
+ if (!xdrgen_encode_notify_rename4(xdr, &nr))
+ goto out_err;
+ *notify_mask |= BIT(NOTIFY4_RENAME_ENTRY);
+ }
+ return p;
+out_err:
+ pr_warn("nfsd: unable to marshal notify event to xdr stream\n");
+ return NULL;
+}
+
+/**
+ * nfsd4_encode_dir_attr_change
+ * @xdr: stream to which to encode the fattr4
+ * @dp: delegation where the event occurred
+ * @nf: nfsd_file opened on the directory
+ *
+ * Encode a dir attr change event.
+ *
+ * Return: a pointer to the start of the encoded event on success; NULL
+ * if there were no requested attributes to report, in which case the
+ * caller should omit the event; or an ERR_PTR if the event was requested
+ * but could not be marshalled into @xdr, in which case the caller should
+ * recall the delegation.
+ */
+u8 *nfsd4_encode_dir_attr_change(struct xdr_stream *xdr, struct nfs4_delegation *dp,
+ struct nfsd_file *nf)
+{
+ struct dentry *dentry = nf->nf_file->f_path.dentry;
+ struct notify_attr4 na = { };
+ u8 *p;
+
+ /* RFC 8881 s10.4.3: ne_file must be a zero-length string for dir attrs */
+ if (!nfsd4_setup_notify_entry4(&na.na_changed_entry, xdr,
+ dentry, dp, nf, "", 0))
+ return ERR_PTR(-ENOBUFS);
+
+ /* No requested attributes to report; omit the event */
+ if (!na.na_changed_entry.ne_attrs.attr_vals.len)
+ return NULL;
+
+ p = (u8 *)xdr->p;
+ if (!xdrgen_encode_notify_attr4(xdr, &na))
+ return ERR_PTR(-ENOBUFS);
+ return p;
}
static void svcxdr_init_encode_from_buffer(struct xdr_stream *xdr,
@@ -4323,7 +4629,7 @@ nfsd4_encode_entry4(void *ccdv, const char *name, int namlen,
__be32 nfserr = nfserr_toosmall;
/* In nfsv4, "." and ".." never make it onto the wire.. */
- if (name && isdotent(name, namlen)) {
+ if (name && name_is_dot_dotdot(name, namlen)) {
cd->common.err = nfs_ok;
return 0;
}
@@ -6390,9 +6696,6 @@ status:
write_bytes_to_xdr_buf(xdr->buf, op_status_offset,
&op->status, XDR_UNIT);
release:
- if (opdesc && opdesc->op_release)
- opdesc->op_release(&op->u);
-
/*
* Account for pages consumed while encoding this operation.
* The xdr_stream primitives don't manage rq_next_page.
@@ -6424,9 +6727,12 @@ void nfsd4_release_compoundargs(struct svc_rqst *rqstp)
{
struct nfsd4_compoundargs *args = rqstp->rq_argp;
+ args->opcnt = 0;
if (args->ops != args->iops) {
- vfree(args->ops);
+ void *old_ops = args->ops;
+
args->ops = args->iops;
+ kvfree_rcu_mightsleep(old_ops);
}
while (args->to_free) {
struct svcxdr_tmpbuf *tb = args->to_free;
diff --git a/fs/nfsd/nfs4xdr_gen.c b/fs/nfsd/nfs4xdr_gen.c
index 824497051b87..a6725c773768 100644
--- a/fs/nfsd/nfs4xdr_gen.c
+++ b/fs/nfsd/nfs4xdr_gen.c
@@ -1,16 +1,16 @@
// SPDX-License-Identifier: GPL-2.0
// Generated by xdrgen. Manual edits will be lost.
// XDR specification file: ../../Documentation/sunrpc/xdr/nfs4_1.x
-// XDR specification modification time: Thu Jan 8 23:12:07 2026
+// XDR specification modification time: Wed Mar 25 11:40:02 2026
#include <linux/sunrpc/svc.h>
#include "nfs4xdr_gen.h"
static bool __maybe_unused
-xdrgen_decode_int64_t(struct xdr_stream *xdr, int64_t *ptr)
+xdrgen_decode_int32_t(struct xdr_stream *xdr, int32_t *ptr)
{
- return xdrgen_decode_hyper(xdr, ptr);
+ return xdrgen_decode_int(xdr, ptr);
}
static bool __maybe_unused
@@ -20,6 +20,154 @@ xdrgen_decode_uint32_t(struct xdr_stream *xdr, uint32_t *ptr)
}
static bool __maybe_unused
+xdrgen_decode_int64_t(struct xdr_stream *xdr, int64_t *ptr)
+{
+ return xdrgen_decode_hyper(xdr, ptr);
+}
+
+static bool __maybe_unused
+xdrgen_decode_uint64_t(struct xdr_stream *xdr, uint64_t *ptr)
+{
+ return xdrgen_decode_unsigned_hyper(xdr, ptr);
+}
+
+static bool __maybe_unused
+xdrgen_decode_nfsstat4(struct xdr_stream *xdr, nfsstat4 *ptr)
+{
+ u32 val;
+
+ if (xdr_stream_decode_u32(xdr, &val) < 0)
+ return false;
+ /* Compiler may optimize to a range check for dense enums */
+ switch (val) {
+ case NFS4_OK:
+ case NFS4ERR_PERM:
+ case NFS4ERR_NOENT:
+ case NFS4ERR_IO:
+ case NFS4ERR_NXIO:
+ case NFS4ERR_ACCESS:
+ case NFS4ERR_EXIST:
+ case NFS4ERR_XDEV:
+ case NFS4ERR_NOTDIR:
+ case NFS4ERR_ISDIR:
+ case NFS4ERR_INVAL:
+ case NFS4ERR_FBIG:
+ case NFS4ERR_NOSPC:
+ case NFS4ERR_ROFS:
+ case NFS4ERR_MLINK:
+ case NFS4ERR_NAMETOOLONG:
+ case NFS4ERR_NOTEMPTY:
+ case NFS4ERR_DQUOT:
+ case NFS4ERR_STALE:
+ case NFS4ERR_BADHANDLE:
+ case NFS4ERR_BAD_COOKIE:
+ case NFS4ERR_NOTSUPP:
+ case NFS4ERR_TOOSMALL:
+ case NFS4ERR_SERVERFAULT:
+ case NFS4ERR_BADTYPE:
+ case NFS4ERR_DELAY:
+ case NFS4ERR_SAME:
+ case NFS4ERR_DENIED:
+ case NFS4ERR_EXPIRED:
+ case NFS4ERR_LOCKED:
+ case NFS4ERR_GRACE:
+ case NFS4ERR_FHEXPIRED:
+ case NFS4ERR_SHARE_DENIED:
+ case NFS4ERR_WRONGSEC:
+ case NFS4ERR_CLID_INUSE:
+ case NFS4ERR_RESOURCE:
+ case NFS4ERR_MOVED:
+ case NFS4ERR_NOFILEHANDLE:
+ case NFS4ERR_MINOR_VERS_MISMATCH:
+ case NFS4ERR_STALE_CLIENTID:
+ case NFS4ERR_STALE_STATEID:
+ case NFS4ERR_OLD_STATEID:
+ case NFS4ERR_BAD_STATEID:
+ case NFS4ERR_BAD_SEQID:
+ case NFS4ERR_NOT_SAME:
+ case NFS4ERR_LOCK_RANGE:
+ case NFS4ERR_SYMLINK:
+ case NFS4ERR_RESTOREFH:
+ case NFS4ERR_LEASE_MOVED:
+ case NFS4ERR_ATTRNOTSUPP:
+ case NFS4ERR_NO_GRACE:
+ case NFS4ERR_RECLAIM_BAD:
+ case NFS4ERR_RECLAIM_CONFLICT:
+ case NFS4ERR_BADXDR:
+ case NFS4ERR_LOCKS_HELD:
+ case NFS4ERR_OPENMODE:
+ case NFS4ERR_BADOWNER:
+ case NFS4ERR_BADCHAR:
+ case NFS4ERR_BADNAME:
+ case NFS4ERR_BAD_RANGE:
+ case NFS4ERR_LOCK_NOTSUPP:
+ case NFS4ERR_OP_ILLEGAL:
+ case NFS4ERR_DEADLOCK:
+ case NFS4ERR_FILE_OPEN:
+ case NFS4ERR_ADMIN_REVOKED:
+ case NFS4ERR_CB_PATH_DOWN:
+ case NFS4ERR_BADIOMODE:
+ case NFS4ERR_BADLAYOUT:
+ case NFS4ERR_BAD_SESSION_DIGEST:
+ case NFS4ERR_BADSESSION:
+ case NFS4ERR_BADSLOT:
+ case NFS4ERR_COMPLETE_ALREADY:
+ case NFS4ERR_CONN_NOT_BOUND_TO_SESSION:
+ case NFS4ERR_DELEG_ALREADY_WANTED:
+ case NFS4ERR_BACK_CHAN_BUSY:
+ case NFS4ERR_LAYOUTTRYLATER:
+ case NFS4ERR_LAYOUTUNAVAILABLE:
+ case NFS4ERR_NOMATCHING_LAYOUT:
+ case NFS4ERR_RECALLCONFLICT:
+ case NFS4ERR_UNKNOWN_LAYOUTTYPE:
+ case NFS4ERR_SEQ_MISORDERED:
+ case NFS4ERR_SEQUENCE_POS:
+ case NFS4ERR_REQ_TOO_BIG:
+ case NFS4ERR_REP_TOO_BIG:
+ case NFS4ERR_REP_TOO_BIG_TO_CACHE:
+ case NFS4ERR_RETRY_UNCACHED_REP:
+ case NFS4ERR_UNSAFE_COMPOUND:
+ case NFS4ERR_TOO_MANY_OPS:
+ case NFS4ERR_OP_NOT_IN_SESSION:
+ case NFS4ERR_HASH_ALG_UNSUPP:
+ case NFS4ERR_CLIENTID_BUSY:
+ case NFS4ERR_PNFS_IO_HOLE:
+ case NFS4ERR_SEQ_FALSE_RETRY:
+ case NFS4ERR_BAD_HIGH_SLOT:
+ case NFS4ERR_DEADSESSION:
+ case NFS4ERR_ENCR_ALG_UNSUPP:
+ case NFS4ERR_PNFS_NO_LAYOUT:
+ case NFS4ERR_NOT_ONLY_OP:
+ case NFS4ERR_WRONG_CRED:
+ case NFS4ERR_WRONG_TYPE:
+ case NFS4ERR_DIRDELEG_UNAVAIL:
+ case NFS4ERR_REJECT_DELEG:
+ case NFS4ERR_RETURNCONFLICT:
+ case NFS4ERR_DELEG_REVOKED:
+ case NFS4ERR_PARTNER_NOTSUPP:
+ case NFS4ERR_PARTNER_NO_AUTH:
+ case NFS4ERR_UNION_NOTSUPP:
+ case NFS4ERR_OFFLOAD_DENIED:
+ case NFS4ERR_WRONG_LFS:
+ case NFS4ERR_BADLABEL:
+ case NFS4ERR_OFFLOAD_NO_REQS:
+ case NFS4ERR_NOXATTR:
+ case NFS4ERR_XATTR2BIG:
+ break;
+ default:
+ return false;
+ }
+ *ptr = val;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_attrlist4(struct xdr_stream *xdr, attrlist4 *ptr)
+{
+ return xdrgen_decode_opaque(xdr, ptr, 0);
+}
+
+static bool __maybe_unused
xdrgen_decode_bitmap4(struct xdr_stream *xdr, bitmap4 *ptr)
{
if (xdr_stream_decode_u32(xdr, &ptr->count) < 0)
@@ -31,6 +179,24 @@ xdrgen_decode_bitmap4(struct xdr_stream *xdr, bitmap4 *ptr)
}
static bool __maybe_unused
+xdrgen_decode_verifier4(struct xdr_stream *xdr, verifier4 *ptr)
+{
+ return xdr_stream_decode_opaque_fixed(xdr, ptr, NFS4_VERIFIER_SIZE) == 0;
+}
+
+static bool __maybe_unused
+xdrgen_decode_nfs_cookie4(struct xdr_stream *xdr, nfs_cookie4 *ptr)
+{
+ return xdrgen_decode_uint64_t(xdr, ptr);
+}
+
+static bool __maybe_unused
+xdrgen_decode_nfs_fh4(struct xdr_stream *xdr, nfs_fh4 *ptr)
+{
+ return xdrgen_decode_opaque(xdr, ptr, NFS4_FHSIZE);
+}
+
+static bool __maybe_unused
xdrgen_decode_utf8string(struct xdr_stream *xdr, utf8string *ptr)
{
return xdrgen_decode_opaque(xdr, ptr, 0);
@@ -55,6 +221,29 @@ xdrgen_decode_utf8str_mixed(struct xdr_stream *xdr, utf8str_mixed *ptr)
}
static bool __maybe_unused
+xdrgen_decode_component4(struct xdr_stream *xdr, component4 *ptr)
+{
+ return xdrgen_decode_utf8str_cs(xdr, ptr);
+}
+
+static bool __maybe_unused
+xdrgen_decode_linktext4(struct xdr_stream *xdr, linktext4 *ptr)
+{
+ return xdrgen_decode_utf8str_cs(xdr, ptr);
+}
+
+static bool __maybe_unused
+xdrgen_decode_pathname4(struct xdr_stream *xdr, pathname4 *ptr)
+{
+ if (xdr_stream_decode_u32(xdr, &ptr->count) < 0)
+ return false;
+ for (u32 i = 0; i < ptr->count; i++)
+ if (!xdrgen_decode_component4(xdr, &ptr->element[i]))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
xdrgen_decode_nfstime4(struct xdr_stream *xdr, struct nfstime4 *ptr)
{
if (!xdrgen_decode_int64_t(xdr, &ptr->seconds))
@@ -65,6 +254,26 @@ xdrgen_decode_nfstime4(struct xdr_stream *xdr, struct nfstime4 *ptr)
}
static bool __maybe_unused
+xdrgen_decode_fattr4(struct xdr_stream *xdr, struct fattr4 *ptr)
+{
+ if (!xdrgen_decode_bitmap4(xdr, &ptr->attrmask))
+ return false;
+ if (!xdrgen_decode_attrlist4(xdr, &ptr->attr_vals))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_stateid4(struct xdr_stream *xdr, struct stateid4 *ptr)
+{
+ if (!xdrgen_decode_uint32_t(xdr, &ptr->seqid))
+ return false;
+ if (xdr_stream_decode_opaque_fixed(xdr, ptr->other, 12) < 0)
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
xdrgen_decode_fattr4_offline(struct xdr_stream *xdr, fattr4_offline *ptr)
{
return xdrgen_decode_bool(xdr, ptr);
@@ -366,9 +575,171 @@ xdrgen_decode_fattr4_posix_access_acl(struct xdr_stream *xdr, fattr4_posix_acces
*/
static bool __maybe_unused
-xdrgen_encode_int64_t(struct xdr_stream *xdr, const int64_t value)
+xdrgen_decode_notify_type4(struct xdr_stream *xdr, notify_type4 *ptr)
{
- return xdrgen_encode_hyper(xdr, value);
+ u32 val;
+
+ if (xdr_stream_decode_u32(xdr, &val) < 0)
+ return false;
+ /* Compiler may optimize to a range check for dense enums */
+ switch (val) {
+ case NOTIFY4_CHANGE_CHILD_ATTRS:
+ case NOTIFY4_CHANGE_DIR_ATTRS:
+ case NOTIFY4_REMOVE_ENTRY:
+ case NOTIFY4_ADD_ENTRY:
+ case NOTIFY4_RENAME_ENTRY:
+ case NOTIFY4_CHANGE_COOKIE_VERIFIER:
+ case NOTIFY4_GFLAG_EXTEND:
+ case NOTIFY4_AUFLAG_VALID:
+ case NOTIFY4_AUFLAG_USER:
+ case NOTIFY4_AUFLAG_GROUP:
+ case NOTIFY4_AUFLAG_OTHER:
+ case NOTIFY4_CHANGE_AUTH:
+ case NOTIFY4_CFLAG_ORDER:
+ case NOTIFY4_AUFLAG_GANOW:
+ case NOTIFY4_AUFLAG_GALATER:
+ case NOTIFY4_CHANGE_GA:
+ case NOTIFY4_CHANGE_AMASK:
+ break;
+ default:
+ return false;
+ }
+ *ptr = val;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_notify_entry4(struct xdr_stream *xdr, struct notify_entry4 *ptr)
+{
+ if (!xdrgen_decode_component4(xdr, &ptr->ne_file))
+ return false;
+ if (!xdrgen_decode_fattr4(xdr, &ptr->ne_attrs))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_prev_entry4(struct xdr_stream *xdr, struct prev_entry4 *ptr)
+{
+ if (!xdrgen_decode_notify_entry4(xdr, &ptr->pe_prev_entry))
+ return false;
+ if (!xdrgen_decode_nfs_cookie4(xdr, &ptr->pe_prev_entry_cookie))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_decode_notify_remove4(struct xdr_stream *xdr, struct notify_remove4 *ptr)
+{
+ if (!xdrgen_decode_notify_entry4(xdr, &ptr->nrm_old_entry))
+ return false;
+ if (!xdrgen_decode_nfs_cookie4(xdr, &ptr->nrm_old_entry_cookie))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_decode_notify_add4(struct xdr_stream *xdr, struct notify_add4 *ptr)
+{
+ if (xdr_stream_decode_u32(xdr, &ptr->nad_old_entry.count) < 0)
+ return false;
+ if (ptr->nad_old_entry.count > 1)
+ return false;
+ for (u32 i = 0; i < ptr->nad_old_entry.count; i++)
+ if (!xdrgen_decode_notify_remove4(xdr, &ptr->nad_old_entry.element[i]))
+ return false;
+ if (!xdrgen_decode_notify_entry4(xdr, &ptr->nad_new_entry))
+ return false;
+ if (xdr_stream_decode_u32(xdr, &ptr->nad_new_entry_cookie.count) < 0)
+ return false;
+ if (ptr->nad_new_entry_cookie.count > 1)
+ return false;
+ for (u32 i = 0; i < ptr->nad_new_entry_cookie.count; i++)
+ if (!xdrgen_decode_nfs_cookie4(xdr, &ptr->nad_new_entry_cookie.element[i]))
+ return false;
+ if (xdr_stream_decode_u32(xdr, &ptr->nad_prev_entry.count) < 0)
+ return false;
+ if (ptr->nad_prev_entry.count > 1)
+ return false;
+ for (u32 i = 0; i < ptr->nad_prev_entry.count; i++)
+ if (!xdrgen_decode_prev_entry4(xdr, &ptr->nad_prev_entry.element[i]))
+ return false;
+ if (!xdrgen_decode_bool(xdr, &ptr->nad_last_entry))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_decode_notify_attr4(struct xdr_stream *xdr, struct notify_attr4 *ptr)
+{
+ if (!xdrgen_decode_notify_entry4(xdr, &ptr->na_changed_entry))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_decode_notify_rename4(struct xdr_stream *xdr, struct notify_rename4 *ptr)
+{
+ if (!xdrgen_decode_notify_remove4(xdr, &ptr->nrn_old_entry))
+ return false;
+ if (!xdrgen_decode_notify_add4(xdr, &ptr->nrn_new_entry))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_notify_verifier4(struct xdr_stream *xdr, struct notify_verifier4 *ptr)
+{
+ if (!xdrgen_decode_verifier4(xdr, &ptr->nv_old_cookieverf))
+ return false;
+ if (!xdrgen_decode_verifier4(xdr, &ptr->nv_new_cookieverf))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_notifylist4(struct xdr_stream *xdr, notifylist4 *ptr)
+{
+ return xdrgen_decode_opaque(xdr, ptr, 0);
+}
+
+static bool __maybe_unused
+xdrgen_decode_notify4(struct xdr_stream *xdr, struct notify4 *ptr)
+{
+ if (!xdrgen_decode_bitmap4(xdr, &ptr->notify_mask))
+ return false;
+ if (!xdrgen_decode_notifylist4(xdr, &ptr->notify_vals))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_decode_CB_NOTIFY4args(struct xdr_stream *xdr, struct CB_NOTIFY4args *ptr)
+{
+ if (!xdrgen_decode_stateid4(xdr, &ptr->cna_stateid))
+ return false;
+ if (!xdrgen_decode_nfs_fh4(xdr, &ptr->cna_fh))
+ return false;
+ if (xdr_stream_decode_u32(xdr, &ptr->cna_changes.count) < 0)
+ return false;
+ for (u32 i = 0; i < ptr->cna_changes.count; i++)
+ if (!xdrgen_decode_notify4(xdr, &ptr->cna_changes.element[i]))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_decode_CB_NOTIFY4res(struct xdr_stream *xdr, struct CB_NOTIFY4res *ptr)
+{
+ if (!xdrgen_decode_nfsstat4(xdr, &ptr->cnr_status))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_int32_t(struct xdr_stream *xdr, const int32_t value)
+{
+ return xdrgen_encode_int(xdr, value);
}
static bool __maybe_unused
@@ -378,6 +749,30 @@ xdrgen_encode_uint32_t(struct xdr_stream *xdr, const uint32_t value)
}
static bool __maybe_unused
+xdrgen_encode_int64_t(struct xdr_stream *xdr, const int64_t value)
+{
+ return xdrgen_encode_hyper(xdr, value);
+}
+
+static bool __maybe_unused
+xdrgen_encode_uint64_t(struct xdr_stream *xdr, const uint64_t value)
+{
+ return xdrgen_encode_unsigned_hyper(xdr, value);
+}
+
+static bool __maybe_unused
+xdrgen_encode_nfsstat4(struct xdr_stream *xdr, nfsstat4 value)
+{
+ return xdr_stream_encode_u32(xdr, value) == XDR_UNIT;
+}
+
+static bool __maybe_unused
+xdrgen_encode_attrlist4(struct xdr_stream *xdr, const attrlist4 value)
+{
+ return xdr_stream_encode_opaque(xdr, value.data, value.len) >= 0;
+}
+
+static bool __maybe_unused
xdrgen_encode_bitmap4(struct xdr_stream *xdr, const bitmap4 value)
{
if (xdr_stream_encode_u32(xdr, value.count) != XDR_UNIT)
@@ -389,6 +784,24 @@ xdrgen_encode_bitmap4(struct xdr_stream *xdr, const bitmap4 value)
}
static bool __maybe_unused
+xdrgen_encode_verifier4(struct xdr_stream *xdr, const verifier4 value)
+{
+ return xdr_stream_encode_opaque_fixed(xdr, value, NFS4_VERIFIER_SIZE) >= 0;
+}
+
+static bool __maybe_unused
+xdrgen_encode_nfs_cookie4(struct xdr_stream *xdr, const nfs_cookie4 value)
+{
+ return xdrgen_encode_uint64_t(xdr, value);
+}
+
+static bool __maybe_unused
+xdrgen_encode_nfs_fh4(struct xdr_stream *xdr, const nfs_fh4 value)
+{
+ return xdr_stream_encode_opaque(xdr, value.data, value.len) >= 0;
+}
+
+static bool __maybe_unused
xdrgen_encode_utf8string(struct xdr_stream *xdr, const utf8string value)
{
return xdr_stream_encode_opaque(xdr, value.data, value.len) >= 0;
@@ -413,6 +826,29 @@ xdrgen_encode_utf8str_mixed(struct xdr_stream *xdr, const utf8str_mixed value)
}
static bool __maybe_unused
+xdrgen_encode_component4(struct xdr_stream *xdr, const component4 value)
+{
+ return xdrgen_encode_utf8str_cs(xdr, value);
+}
+
+static bool __maybe_unused
+xdrgen_encode_linktext4(struct xdr_stream *xdr, const linktext4 value)
+{
+ return xdrgen_encode_utf8str_cs(xdr, value);
+}
+
+static bool __maybe_unused
+xdrgen_encode_pathname4(struct xdr_stream *xdr, const pathname4 value)
+{
+ if (xdr_stream_encode_u32(xdr, value.count) != XDR_UNIT)
+ return false;
+ for (u32 i = 0; i < value.count; i++)
+ if (!xdrgen_encode_component4(xdr, value.element[i]))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
xdrgen_encode_nfstime4(struct xdr_stream *xdr, const struct nfstime4 *value)
{
if (!xdrgen_encode_int64_t(xdr, value->seconds))
@@ -423,6 +859,26 @@ xdrgen_encode_nfstime4(struct xdr_stream *xdr, const struct nfstime4 *value)
}
static bool __maybe_unused
+xdrgen_encode_fattr4(struct xdr_stream *xdr, const struct fattr4 *value)
+{
+ if (!xdrgen_encode_bitmap4(xdr, value->attrmask))
+ return false;
+ if (!xdrgen_encode_attrlist4(xdr, value->attr_vals))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_stateid4(struct xdr_stream *xdr, const struct stateid4 *value)
+{
+ if (!xdrgen_encode_uint32_t(xdr, value->seqid))
+ return false;
+ if (xdr_stream_encode_opaque_fixed(xdr, value->other, 12) < 0)
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
xdrgen_encode_fattr4_offline(struct xdr_stream *xdr, const fattr4_offline value)
{
return xdrgen_encode_bool(xdr, value);
@@ -567,3 +1023,137 @@ xdrgen_encode_fattr4_posix_access_acl(struct xdr_stream *xdr, const fattr4_posix
return false;
return true;
}
+
+static bool __maybe_unused
+xdrgen_encode_notify_type4(struct xdr_stream *xdr, notify_type4 value)
+{
+ return xdr_stream_encode_u32(xdr, value) == XDR_UNIT;
+}
+
+static bool __maybe_unused
+xdrgen_encode_notify_entry4(struct xdr_stream *xdr, const struct notify_entry4 *value)
+{
+ if (!xdrgen_encode_component4(xdr, value->ne_file))
+ return false;
+ if (!xdrgen_encode_fattr4(xdr, &value->ne_attrs))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_prev_entry4(struct xdr_stream *xdr, const struct prev_entry4 *value)
+{
+ if (!xdrgen_encode_notify_entry4(xdr, &value->pe_prev_entry))
+ return false;
+ if (!xdrgen_encode_nfs_cookie4(xdr, value->pe_prev_entry_cookie))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_encode_notify_remove4(struct xdr_stream *xdr, const struct notify_remove4 *value)
+{
+ if (!xdrgen_encode_notify_entry4(xdr, &value->nrm_old_entry))
+ return false;
+ if (!xdrgen_encode_nfs_cookie4(xdr, value->nrm_old_entry_cookie))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_encode_notify_add4(struct xdr_stream *xdr, const struct notify_add4 *value)
+{
+ if (value->nad_old_entry.count > 1)
+ return false;
+ if (xdr_stream_encode_u32(xdr, value->nad_old_entry.count) != XDR_UNIT)
+ return false;
+ for (u32 i = 0; i < value->nad_old_entry.count; i++)
+ if (!xdrgen_encode_notify_remove4(xdr, &value->nad_old_entry.element[i]))
+ return false;
+ if (!xdrgen_encode_notify_entry4(xdr, &value->nad_new_entry))
+ return false;
+ if (value->nad_new_entry_cookie.count > 1)
+ return false;
+ if (xdr_stream_encode_u32(xdr, value->nad_new_entry_cookie.count) != XDR_UNIT)
+ return false;
+ for (u32 i = 0; i < value->nad_new_entry_cookie.count; i++)
+ if (!xdrgen_encode_nfs_cookie4(xdr, value->nad_new_entry_cookie.element[i]))
+ return false;
+ if (value->nad_prev_entry.count > 1)
+ return false;
+ if (xdr_stream_encode_u32(xdr, value->nad_prev_entry.count) != XDR_UNIT)
+ return false;
+ for (u32 i = 0; i < value->nad_prev_entry.count; i++)
+ if (!xdrgen_encode_prev_entry4(xdr, &value->nad_prev_entry.element[i]))
+ return false;
+ if (!xdrgen_encode_bool(xdr, value->nad_last_entry))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_encode_notify_attr4(struct xdr_stream *xdr, const struct notify_attr4 *value)
+{
+ if (!xdrgen_encode_notify_entry4(xdr, &value->na_changed_entry))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_encode_notify_rename4(struct xdr_stream *xdr, const struct notify_rename4 *value)
+{
+ if (!xdrgen_encode_notify_remove4(xdr, &value->nrn_old_entry))
+ return false;
+ if (!xdrgen_encode_notify_add4(xdr, &value->nrn_new_entry))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_notify_verifier4(struct xdr_stream *xdr, const struct notify_verifier4 *value)
+{
+ if (!xdrgen_encode_verifier4(xdr, value->nv_old_cookieverf))
+ return false;
+ if (!xdrgen_encode_verifier4(xdr, value->nv_new_cookieverf))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_notifylist4(struct xdr_stream *xdr, const notifylist4 value)
+{
+ return xdr_stream_encode_opaque(xdr, value.data, value.len) >= 0;
+}
+
+static bool __maybe_unused
+xdrgen_encode_notify4(struct xdr_stream *xdr, const struct notify4 *value)
+{
+ if (!xdrgen_encode_bitmap4(xdr, value->notify_mask))
+ return false;
+ if (!xdrgen_encode_notifylist4(xdr, value->notify_vals))
+ return false;
+ return true;
+}
+
+bool
+xdrgen_encode_CB_NOTIFY4args(struct xdr_stream *xdr, const struct CB_NOTIFY4args *value)
+{
+ if (!xdrgen_encode_stateid4(xdr, &value->cna_stateid))
+ return false;
+ if (!xdrgen_encode_nfs_fh4(xdr, value->cna_fh))
+ return false;
+ if (xdr_stream_encode_u32(xdr, value->cna_changes.count) != XDR_UNIT)
+ return false;
+ for (u32 i = 0; i < value->cna_changes.count; i++)
+ if (!xdrgen_encode_notify4(xdr, &value->cna_changes.element[i]))
+ return false;
+ return true;
+}
+
+static bool __maybe_unused
+xdrgen_encode_CB_NOTIFY4res(struct xdr_stream *xdr, const struct CB_NOTIFY4res *value)
+{
+ if (!xdrgen_encode_nfsstat4(xdr, value->cnr_status))
+ return false;
+ return true;
+}
diff --git a/fs/nfsd/nfs4xdr_gen.h b/fs/nfsd/nfs4xdr_gen.h
index 1c487f1a11ab..f6a458a07406 100644
--- a/fs/nfsd/nfs4xdr_gen.h
+++ b/fs/nfsd/nfs4xdr_gen.h
@@ -1,7 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
/* Generated by xdrgen. Manual edits will be lost. */
/* XDR specification file: ../../Documentation/sunrpc/xdr/nfs4_1.x */
-/* XDR specification modification time: Thu Jan 8 23:12:07 2026 */
+/* XDR specification modification time: Wed Mar 25 11:40:02 2026 */
#ifndef _LINUX_XDRGEN_NFS4_1_DECL_H
#define _LINUX_XDRGEN_NFS4_1_DECL_H
@@ -32,4 +32,19 @@ bool xdrgen_decode_posixaceperm4(struct xdr_stream *xdr, posixaceperm4 *ptr);
bool xdrgen_encode_posixaceperm4(struct xdr_stream *xdr, const posixaceperm4 value);
+bool xdrgen_decode_notify_remove4(struct xdr_stream *xdr, struct notify_remove4 *ptr);
+bool xdrgen_encode_notify_remove4(struct xdr_stream *xdr, const struct notify_remove4 *value);
+
+bool xdrgen_decode_notify_add4(struct xdr_stream *xdr, struct notify_add4 *ptr);
+bool xdrgen_encode_notify_add4(struct xdr_stream *xdr, const struct notify_add4 *value);
+
+bool xdrgen_decode_notify_attr4(struct xdr_stream *xdr, struct notify_attr4 *ptr);
+bool xdrgen_encode_notify_attr4(struct xdr_stream *xdr, const struct notify_attr4 *value);
+
+bool xdrgen_decode_notify_rename4(struct xdr_stream *xdr, struct notify_rename4 *ptr);
+bool xdrgen_encode_notify_rename4(struct xdr_stream *xdr, const struct notify_rename4 *value);
+
+bool xdrgen_decode_CB_NOTIFY4args(struct xdr_stream *xdr, struct CB_NOTIFY4args *ptr);
+bool xdrgen_encode_CB_NOTIFY4args(struct xdr_stream *xdr, const struct CB_NOTIFY4args *value);
+
#endif /* _LINUX_XDRGEN_NFS4_1_DECL_H */
diff --git a/fs/nfsd/nfscache.c b/fs/nfsd/nfscache.c
index 154468ceccdc..18f8556d33dd 100644
--- a/fs/nfsd/nfscache.c
+++ b/fs/nfsd/nfscache.c
@@ -200,14 +200,14 @@ int nfsd_reply_cache_init(struct nfsd_net *nn)
nn->nfsd_reply_cache_shrinker->seeks = 1;
nn->nfsd_reply_cache_shrinker->private_data = nn;
- shrinker_register(nn->nfsd_reply_cache_shrinker);
-
for (i = 0; i < hashsize; i++) {
INIT_LIST_HEAD(&nn->drc_hashtbl[i].lru_head);
spin_lock_init(&nn->drc_hashtbl[i].cache_lock);
}
nn->drc_hashsize = hashsize;
+ shrinker_register(nn->nfsd_reply_cache_shrinker);
+
return 0;
out_shrinker:
kvfree(nn->drc_hashtbl);
diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c
index fa92e31d19d6..bc16fc7ca24f 100644
--- a/fs/nfsd/nfsctl.c
+++ b/fs/nfsd/nfsctl.c
@@ -296,14 +296,15 @@ static ssize_t write_unlock_fs(struct file *file, char *buf, size_t size)
* 2. Is that directory a mount point, or
* 3. Is that directory the root of an exported file system?
*/
- nfsd4_cancel_copy_by_sb(netns(file), path.dentry->d_sb);
error = nlmsvc_unlock_all_by_sb(path.dentry->d_sb);
mutex_lock(&nfsd_mutex);
nn = net_generic(netns(file), nfsd_net_id);
- if (nn->nfsd_serv)
+ if (test_bit(NFSD_NET_UP, &nn->flags)) {
+ nfsd4_cancel_copy_by_sb(netns(file), path.dentry->d_sb);
nfsd4_revoke_states(nn, path.dentry->d_sb);
- else
+ } else {
error = -EINVAL;
+ }
mutex_unlock(&nfsd_mutex);
path_put(&path);
@@ -420,6 +421,7 @@ static ssize_t write_threads(struct file *file, char *buf, size_t size)
char *mesg = buf;
int rv;
struct net *net = netns(file);
+ struct nfsd_net *nn = net_generic(net, nfsd_net_id);
if (size > 0) {
int newthreads;
@@ -430,7 +432,10 @@ static ssize_t write_threads(struct file *file, char *buf, size_t size)
return -EINVAL;
trace_nfsd_ctl_threads(net, newthreads);
mutex_lock(&nfsd_mutex);
- rv = nfsd_svc(1, &newthreads, net, file->f_cred, NULL);
+ if (newthreads > 0 || nn->nfsd_serv != NULL)
+ rv = nfsd_svc(1, &newthreads, net, file->f_cred, NULL);
+ else
+ rv = 0;
mutex_unlock(&nfsd_mutex);
if (rv < 0)
return rv;
@@ -1111,7 +1116,7 @@ static ssize_t write_v4_end_grace(struct file *file, char *buf, size_t size)
}
return scnprintf(buf, SIMPLE_TRANSACTION_LIMIT, "%c\n",
- nn->grace_ended ? 'Y' : 'N');
+ test_bit(NFSD_NET_GRACE_ENDED, &nn->flags) ? 'Y' : 'N');
}
#endif
@@ -1414,8 +1419,8 @@ static int create_proc_exports_entry(void)
unsigned int nfsd_net_id;
struct nfsd_genl_rqstp {
- struct sockaddr rq_daddr;
- struct sockaddr rq_saddr;
+ struct sockaddr_storage rq_daddr;
+ struct sockaddr_storage rq_saddr;
unsigned long rq_flags;
ktime_t rq_stime;
__be32 rq_xid;
@@ -1448,9 +1453,9 @@ static int nfsd_genl_rpc_status_compose_msg(struct sk_buff *skb,
nla_put_s64(skb, NFSD_A_RPC_STATUS_SERVICE_TIME,
ktime_to_us(genl_rqstp->rq_stime),
NFSD_A_RPC_STATUS_PAD))
- return -ENOBUFS;
+ goto out_cancel;
- switch (genl_rqstp->rq_saddr.sa_family) {
+ switch (genl_rqstp->rq_saddr.ss_family) {
case AF_INET: {
const struct sockaddr_in *s_in, *d_in;
@@ -1464,7 +1469,7 @@ static int nfsd_genl_rpc_status_compose_msg(struct sk_buff *skb,
s_in->sin_port) ||
nla_put_be16(skb, NFSD_A_RPC_STATUS_DPORT,
d_in->sin_port))
- return -ENOBUFS;
+ goto out_cancel;
break;
}
case AF_INET6: {
@@ -1480,7 +1485,7 @@ static int nfsd_genl_rpc_status_compose_msg(struct sk_buff *skb,
s_in->sin6_port) ||
nla_put_be16(skb, NFSD_A_RPC_STATUS_DPORT,
d_in->sin6_port))
- return -ENOBUFS;
+ goto out_cancel;
break;
}
}
@@ -1488,10 +1493,14 @@ static int nfsd_genl_rpc_status_compose_msg(struct sk_buff *skb,
for (i = 0; i < genl_rqstp->rq_opcnt; i++)
if (nla_put_u32(skb, NFSD_A_RPC_STATUS_COMPOUND_OPS,
genl_rqstp->rq_opnum[i]))
- return -ENOBUFS;
+ goto out_cancel;
genlmsg_end(skb, hdr);
return 0;
+
+out_cancel:
+ genlmsg_cancel(skb, hdr);
+ return -ENOBUFS;
}
/**
@@ -1519,18 +1528,28 @@ int nfsd_nl_rpc_status_get_dumpit(struct sk_buff *skb,
for (i = 0; i < nn->nfsd_serv->sv_nrpools; i++) {
struct svc_rqst *rqstp;
+ long thread_skip = 0;
if (i < cb->args[0]) /* already consumed */
continue;
+ /*
+ * The saved thread index only applies to the pool the dump
+ * was resumed in. Subsequent pools must start from thread 0,
+ * otherwise their first cb->args[1] threads are silently
+ * skipped.
+ */
+ if (i == cb->args[0])
+ thread_skip = cb->args[1];
+
rqstp_index = 0;
list_for_each_entry_rcu(rqstp,
&nn->nfsd_serv->sv_pools[i].sp_all_threads,
rq_all) {
- struct nfsd_genl_rqstp genl_rqstp;
+ struct nfsd_genl_rqstp genl_rqstp = {};
unsigned int status_counter;
- if (rqstp_index++ < cb->args[1]) /* already consumed */
+ if (rqstp_index++ < thread_skip) /* already consumed */
continue;
/*
* Acquire rq_status_counter before parsing the rqst
@@ -1551,9 +1570,9 @@ int nfsd_nl_rpc_status_get_dumpit(struct sk_buff *skb,
genl_rqstp.rq_stime = rqstp->rq_stime;
genl_rqstp.rq_opcnt = 0;
memcpy(&genl_rqstp.rq_daddr, svc_daddr(rqstp),
- sizeof(struct sockaddr));
+ sizeof(struct sockaddr_storage));
memcpy(&genl_rqstp.rq_saddr, svc_addr(rqstp),
- sizeof(struct sockaddr));
+ sizeof(struct sockaddr_storage));
#ifdef CONFIG_NFSD_V4
if (rqstp->rq_vers == NFS4_VERSION &&
@@ -1572,17 +1591,26 @@ int nfsd_nl_rpc_status_get_dumpit(struct sk_buff *skb,
#endif /* CONFIG_NFSD_V4 */
/*
- * Acquire rq_status_counter before reporting the rqst
- * fields to the user.
+ * Read-side load-load fence: order the field reads
+ * above before the counter re-read below, mirroring
+ * the smp_rmb() in the standard seqcount retry. The
+ * begin-side smp_load_acquire() above pairs with the
+ * smp_store_release() in nfsd_dispatch().
*/
- if (smp_load_acquire(&rqstp->rq_status_counter) !=
- status_counter)
+ smp_rmb();
+ if (READ_ONCE(rqstp->rq_status_counter) != status_counter)
continue;
ret = nfsd_genl_rpc_status_compose_msg(skb, cb,
&genl_rqstp);
- if (ret)
+ if (ret) {
+ if (skb->len) {
+ cb->args[0] = i;
+ cb->args[1] = rqstp_index - 1;
+ ret = skb->len;
+ }
goto out;
+ }
}
}
@@ -1944,6 +1972,60 @@ err_free_msg:
}
/**
+ * nfsd_nl_validate_listeners - sanity-check the listener list from userland
+ * @info: netlink metadata and command arguments
+ *
+ * Walk every NFSD_A_SERVER_SOCK_ADDR attribute and confirm that each entry
+ * is well-formed: it parses against the policy, carries both an address and
+ * a transport name, and the address is long enough for its family. Doing
+ * this up front lets the callers below assume every entry is valid and
+ * guarantees we make no changes when the request is malformed.
+ *
+ * Return: 0 if every entry is valid, or a negative errno otherwise.
+ */
+static int nfsd_nl_validate_listeners(struct genl_info *info)
+{
+ const struct nlattr *attr;
+ int rem;
+
+ nlmsg_for_each_attr_type(attr, NFSD_A_SERVER_SOCK_ADDR, info->nlhdr,
+ GENL_HDRLEN, rem) {
+ struct nlattr *tb[NFSD_A_SOCK_MAX + 1];
+ struct sockaddr *sa;
+ int err;
+
+ err = nla_parse_nested(tb, NFSD_A_SOCK_MAX, attr,
+ nfsd_sock_nl_policy, info->extack);
+ if (err < 0)
+ return err;
+
+ if (!tb[NFSD_A_SOCK_ADDR] || !tb[NFSD_A_SOCK_TRANSPORT_NAME])
+ return -EINVAL;
+
+ sa = nla_data(tb[NFSD_A_SOCK_ADDR]);
+ if (nla_len(tb[NFSD_A_SOCK_ADDR]) < sizeof(sa->sa_family))
+ return -EINVAL;
+
+ switch (sa->sa_family) {
+ case AF_INET:
+ if (nla_len(tb[NFSD_A_SOCK_ADDR]) <
+ sizeof(struct sockaddr_in))
+ return -EINVAL;
+ break;
+ case AF_INET6:
+ if (nla_len(tb[NFSD_A_SOCK_ADDR]) <
+ sizeof(struct sockaddr_in6))
+ return -EINVAL;
+ break;
+ default:
+ return -EAFNOSUPPORT;
+ }
+ }
+
+ return 0;
+}
+
+/**
* nfsd_nl_listener_set_doit - set the nfs running sockets
* @skb: reply buffer
* @info: netlink metadata and command arguments
@@ -1961,6 +2043,15 @@ int nfsd_nl_listener_set_doit(struct sk_buff *skb, struct genl_info *info)
bool delete = false;
int err, rem;
+ /*
+ * Validate the entire listener list before making any changes, so a
+ * malformed request fails cleanly without creating a serv or touching
+ * the existing listeners.
+ */
+ err = nfsd_nl_validate_listeners(info);
+ if (err)
+ return err;
+
mutex_lock(&nfsd_mutex);
err = nfsd_create_serv(net);
@@ -1987,16 +2078,11 @@ int nfsd_nl_listener_set_doit(struct sk_buff *skb, struct genl_info *info)
const char *xcl_name;
struct sockaddr *sa;
+ /* validated up front in nfsd_nl_validate_listeners() */
if (nla_parse_nested(tb, NFSD_A_SOCK_MAX, attr,
nfsd_sock_nl_policy, info->extack) < 0)
continue;
- if (!tb[NFSD_A_SOCK_ADDR] || !tb[NFSD_A_SOCK_TRANSPORT_NAME])
- continue;
-
- if (nla_len(tb[NFSD_A_SOCK_ADDR]) < sizeof(*sa))
- continue;
-
xcl_name = nla_data(tb[NFSD_A_SOCK_TRANSPORT_NAME]);
sa = nla_data(tb[NFSD_A_SOCK_ADDR]);
@@ -2048,16 +2134,11 @@ int nfsd_nl_listener_set_doit(struct sk_buff *skb, struct genl_info *info)
struct sockaddr *sa;
int ret;
+ /* validated up front in nfsd_nl_validate_listeners() */
if (nla_parse_nested(tb, NFSD_A_SOCK_MAX, attr,
nfsd_sock_nl_policy, info->extack) < 0)
continue;
- if (!tb[NFSD_A_SOCK_ADDR] || !tb[NFSD_A_SOCK_TRANSPORT_NAME])
- continue;
-
- if (nla_len(tb[NFSD_A_SOCK_ADDR]) < sizeof(*sa))
- continue;
-
xcl_name = nla_data(tb[NFSD_A_SOCK_TRANSPORT_NAME]);
sa = nla_data(tb[NFSD_A_SOCK_ADDR]);
@@ -2343,7 +2424,7 @@ int nfsd_nl_unlock_filesystem_doit(struct sk_buff *skb,
error = nlmsvc_unlock_all_by_sb(path.dentry->d_sb);
mutex_lock(&nfsd_mutex);
- if (nn->nfsd_serv) {
+ if (test_bit(NFSD_NET_UP, &nn->flags)) {
nfsd4_cancel_copy_by_sb(net, path.dentry->d_sb);
nfsd4_revoke_states(nn, path.dentry->d_sb);
} else {
@@ -2390,7 +2471,7 @@ int nfsd_nl_unlock_export_doit(struct sk_buff *skb, struct genl_info *info)
return error;
mutex_lock(&nfsd_mutex);
- if (nn->nfsd_serv) {
+ if (test_bit(NFSD_NET_UP, &nn->flags)) {
nfsd_file_close_export(net, &path);
nfsd4_revoke_export_states(nn, &path);
} else
@@ -2512,11 +2593,12 @@ static int __init init_nfsd(void)
{
int retval;
- nfsd_debugfs_init();
-
retval = nfsd4_init_slabs();
if (retval)
return retval;
+
+ nfsd_debugfs_init();
+
retval = nfsd4_init_pnfs();
if (retval)
goto out_free_slabs;
@@ -2561,8 +2643,8 @@ out_free_lockd:
out_free_pnfs:
nfsd4_exit_pnfs();
out_free_slabs:
- nfsd4_free_slabs();
nfsd_debugfs_exit();
+ nfsd4_free_slabs();
return retval;
}
@@ -2577,9 +2659,9 @@ static void __exit exit_nfsd(void)
unregister_pernet_subsys(&nfsd_net_ops);
nfsd_drc_slab_free();
nfsd_lockd_shutdown();
- nfsd4_free_slabs();
nfsd4_exit_pnfs();
nfsd_debugfs_exit();
+ nfsd4_free_slabs();
}
MODULE_AUTHOR("Olaf Kirch <okir@monad.swb.de>");
diff --git a/fs/nfsd/nfsd.h b/fs/nfsd/nfsd.h
index 11bce03b9031..4c5d915225df 100644
--- a/fs/nfsd/nfsd.h
+++ b/fs/nfsd/nfsd.h
@@ -332,13 +332,14 @@ void nfsd_lockd_shutdown(void);
#define nfserr_noxattr cpu_to_be32(NFS4ERR_NOXATTR)
/*
- * Error codes for internal use. We use enum to choose numbers that are
- * not already assigned, then covert to be32 resulting in a number that
- * cannot conflict with any existing be32 nfserr value.
+ * Error codes for internal use. These are based at an impossible
+ * nfsstat4 value so that, once converted to be32, they cannot conflict
+ * with any value defined by the protocol (compare the nlm__int__* codes
+ * in fs/lockd/lockd.h).
*/
enum {
/* end-of-file indicator in readdir */
- NFSERR_EOF = NFS4ERR_FIRST_FREE,
+ NFSERR_EOF = 30000,
#define nfserr_eof cpu_to_be32(NFSERR_EOF)
/* replay detected */
@@ -356,9 +357,6 @@ enum {
#define nfserr_symlink_not_dir cpu_to_be32(NFSERR_SYMLINK_NOT_DIR)
};
-/* Check for dir entries '.' and '..' */
-#define isdotent(n, l) (l < 3 && n[0] == '.' && (l == 1 || n[1] == '.'))
-
#ifdef CONFIG_NFSD_V4
/* before processing a COMPOUND operation, we have to check that there
diff --git a/fs/nfsd/nfsfh.c b/fs/nfsd/nfsfh.c
index 429ca5c6ec08..8b1a95e1d058 100644
--- a/fs/nfsd/nfsfh.c
+++ b/fs/nfsd/nfsfh.c
@@ -70,10 +70,8 @@ nfsd_mode_check(struct dentry *dentry, umode_t requested)
if (requested == 0) /* the caller doesn't care */
return nfs_ok;
if (mode == requested) {
- if (mode == S_IFDIR && !d_can_lookup(dentry)) {
- WARN_ON_ONCE(1);
+ if (mode == S_IFDIR && !d_can_lookup(dentry))
return nfserr_notdir;
- }
return nfs_ok;
}
if (mode == S_IFLNK) {
@@ -144,16 +142,15 @@ static inline __be32 check_pseudo_root(struct dentry *dentry,
/* Size of a file handle MAC, in 4-octet words */
#define FH_MAC_WORDS (sizeof(__le64) / 4)
-static bool fh_append_mac(struct svc_fh *fhp, struct net *net)
+bool fh_append_mac(struct knfsd_fh *fh, int fh_maxsize, struct net *net)
{
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
- struct knfsd_fh *fh = &fhp->fh_handle;
siphash_key_t *fh_key = nn->fh_key;
__le64 hash;
if (!fh_key)
goto out_no_key;
- if (fh->fh_size + sizeof(hash) > fhp->fh_maxsize)
+ if (fh->fh_size + sizeof(hash) > fh_maxsize)
goto out_no_space;
hash = cpu_to_le64(siphash(&fh->fh_raw, fh->fh_size, fh_key));
@@ -167,7 +164,7 @@ out_no_key:
out_no_space:
pr_warn_ratelimited("NFSD: unable to sign filehandles, fh_size %zu would be greater than fh_maxsize %d.\n",
- fh->fh_size + sizeof(hash), fhp->fh_maxsize);
+ fh->fh_size + sizeof(hash), fh_maxsize);
return false;
}
@@ -344,15 +341,19 @@ static __be32 nfsd_set_fh_dentry(struct svc_rqst *rqstp, struct net *net,
if (dentry->d_sb->s_export_op->flags & EXPORT_OP_NOWCC)
fhp->fh_no_wcc = true;
fhp->fh_64bit_cookies = true;
- if (exp->ex_flags & NFSEXP_V4ROOT)
+ if (exp->ex_flags & NFSEXP_V4ROOT) {
+ dput(dentry);
goto out;
+ }
break;
case NFS_FHSIZE:
fhp->fh_no_wcc = true;
if (EX_WGATHER(exp))
fhp->fh_use_wgather = true;
- if (exp->ex_flags & NFSEXP_V4ROOT)
+ if (exp->ex_flags & NFSEXP_V4ROOT) {
+ dput(dentry);
goto out;
+ }
}
fhp->fh_dentry = dentry;
@@ -562,7 +563,8 @@ static void _fh_update(struct svc_fh *fhp, struct svc_export *exp,
fhp->fh_handle.fh_size += maxsize * 4;
if (exp->ex_flags & NFSEXP_SIGN_FH)
- if (!fh_append_mac(fhp, exp->cd->net))
+ if (!fh_append_mac(&fhp->fh_handle, fhp->fh_maxsize,
+ exp->cd->net))
fhp->fh_handle.fh_fileid_type = FILEID_INVALID;
} else {
fhp->fh_handle.fh_fileid_type = FILEID_ROOT;
@@ -892,19 +894,20 @@ char * SVCFH_fmt(struct svc_fh *fhp)
return buf;
}
-enum fsid_source fsid_source(const struct svc_fh *fhp)
+enum fsid_source fsid_source_fh(const struct knfsd_fh *fh,
+ struct svc_export *exp)
{
- if (fhp->fh_handle.fh_version != 1)
+ if (fh->fh_version != 1)
return FSIDSOURCE_DEV;
- switch(fhp->fh_handle.fh_fsid_type) {
+ switch (fh->fh_fsid_type) {
case FSID_DEV:
case FSID_ENCODE_DEV:
case FSID_MAJOR_MINOR:
- if (exp_sb(fhp->fh_export)->s_type->fs_flags & FS_REQUIRES_DEV)
+ if (exp_sb(exp)->s_type->fs_flags & FS_REQUIRES_DEV)
return FSIDSOURCE_DEV;
break;
case FSID_NUM:
- if (fhp->fh_export->ex_flags & NFSEXP_FSID)
+ if (exp->ex_flags & NFSEXP_FSID)
return FSIDSOURCE_FSID;
break;
default:
@@ -913,13 +916,18 @@ enum fsid_source fsid_source(const struct svc_fh *fhp)
/* either a UUID type filehandle, or the filehandle doesn't
* match the export.
*/
- if (fhp->fh_export->ex_flags & NFSEXP_FSID)
+ if (exp->ex_flags & NFSEXP_FSID)
return FSIDSOURCE_FSID;
- if (fhp->fh_export->ex_uuid)
+ if (exp->ex_uuid)
return FSIDSOURCE_UUID;
return FSIDSOURCE_DEV;
}
+enum fsid_source fsid_source(const struct svc_fh *fhp)
+{
+ return fsid_source_fh(&fhp->fh_handle, fhp->fh_export);
+}
+
/**
* nfsd4_change_attribute - Generate an NFSv4 change_attribute value
* @stat: inode attributes
diff --git a/fs/nfsd/nfsfh.h b/fs/nfsd/nfsfh.h
index 5ef7191f8ad8..cdeb5eea65a8 100644
--- a/fs/nfsd/nfsfh.h
+++ b/fs/nfsd/nfsfh.h
@@ -131,6 +131,8 @@ enum fsid_source {
FSIDSOURCE_FSID,
FSIDSOURCE_UUID,
};
+extern enum fsid_source fsid_source_fh(const struct knfsd_fh *fh,
+ struct svc_export *exp);
extern enum fsid_source fsid_source(const struct svc_fh *fhp);
@@ -226,6 +228,7 @@ __be32 fh_getattr(const struct svc_fh *fhp, struct kstat *stat);
__be32 fh_compose(struct svc_fh *, struct svc_export *, struct dentry *, struct svc_fh *);
__be32 fh_update(struct svc_fh *);
void fh_put(struct svc_fh *);
+bool fh_append_mac(struct knfsd_fh *fh, int fh_maxsize, struct net *net);
static __inline__ struct svc_fh *
fh_copy(struct svc_fh *dst, const struct svc_fh *src)
diff --git a/fs/nfsd/nfsproc.c b/fs/nfsd/nfsproc.c
index 8873033d1e82..f60043632575 100644
--- a/fs/nfsd/nfsproc.c
+++ b/fs/nfsd/nfsproc.c
@@ -82,6 +82,7 @@ nfsd_proc_setattr(struct svc_rqst *rqstp)
.na_iattr = iap,
};
struct svc_fh *fhp;
+ int hosterr;
dprintk("nfsd: SETATTR %s, valid=%x, size=%ld\n",
SVCFH_fmt(&argp->fh),
@@ -117,6 +118,12 @@ nfsd_proc_setattr(struct svc_rqst *rqstp)
if (resp->status != nfs_ok)
goto out;
+ hosterr = fh_want_write(fhp);
+ if (hosterr) {
+ resp->status = nfserrno(hosterr);
+ goto out;
+ }
+
if (delta < 0)
delta = -delta;
if (delta < MAX_TOUCH_TIME_ERROR &&
@@ -298,7 +305,7 @@ nfsd_proc_create(struct svc_rqst *rqstp)
/* Check for NFSD_MAY_WRITE in nfsd_create if necessary */
resp->status = nfserr_exist;
- if (isdotent(argp->name, argp->len))
+ if (name_is_dot_dotdot(argp->name, argp->len))
goto done;
hosterr = fh_want_write(dirfhp);
if (hosterr) {
diff --git a/fs/nfsd/nfssvc.c b/fs/nfsd/nfssvc.c
index 4f1ab3222a4d..a8ea4dbfa56b 100644
--- a/fs/nfsd/nfssvc.c
+++ b/fs/nfsd/nfssvc.c
@@ -237,15 +237,21 @@ static void nfsd_net_free(struct percpu_ref *ref)
*/
#define NFSD_MAXSERVS 8192
+/**
+ * nfsd_nrthreads - report a namespace's configured nfsd thread count
+ * @net: network namespace to query
+ *
+ * Return: the configured thread ceiling, or 0 when no service runs.
+ */
int nfsd_nrthreads(struct net *net)
{
- int i, rv = 0;
+ int rv = 0;
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
+ /* nfsd_mutex keeps nn->nfsd_serv valid across the read. */
mutex_lock(&nfsd_mutex);
if (nn->nfsd_serv)
- for (i = 0; i < nn->nfsd_serv->sv_nrpools; ++i)
- rv += nn->nfsd_serv->sv_pools[i].sp_nrthrmax;
+ rv = svc_serv_maxthreads(nn->nfsd_serv);
mutex_unlock(&nfsd_mutex);
return rv;
}
@@ -351,7 +357,7 @@ static int nfsd_startup_net(struct net *net, const struct cred *cred)
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
int ret;
- if (nn->nfsd_net_up)
+ if (test_bit(NFSD_NET_UP, &nn->flags))
return 0;
ret = nfsd_startup_generic();
@@ -364,11 +370,11 @@ static int nfsd_startup_net(struct net *net, const struct cred *cred)
goto out_socks;
}
- if (nfsd_needs_lockd(nn) && !nn->lockd_up) {
+ if (nfsd_needs_lockd(nn) && !test_bit(NFSD_NET_LOCKD_UP, &nn->flags)) {
ret = lockd_up(net, cred);
if (ret)
goto out_socks;
- nn->lockd_up = true;
+ set_bit(NFSD_NET_LOCKD_UP, &nn->flags);
}
ret = nfsd_file_cache_start_net(net);
@@ -386,7 +392,7 @@ static int nfsd_startup_net(struct net *net, const struct cred *cred)
if (ret)
goto out_reply_cache;
- nn->nfsd_net_up = true;
+ set_bit(NFSD_NET_UP, &nn->flags);
return 0;
out_reply_cache:
@@ -394,9 +400,9 @@ out_reply_cache:
out_filecache:
nfsd_file_cache_shutdown_net(net);
out_lockd:
- if (nn->lockd_up) {
+ if (test_bit(NFSD_NET_LOCKD_UP, &nn->flags)) {
lockd_down(net);
- nn->lockd_up = false;
+ clear_bit(NFSD_NET_LOCKD_UP, &nn->flags);
}
out_socks:
nfsd_shutdown_generic();
@@ -407,7 +413,7 @@ static void nfsd_shutdown_net(struct net *net)
{
struct nfsd_net *nn = net_generic(net, nfsd_net_id);
- if (nn->nfsd_net_up) {
+ if (test_bit(NFSD_NET_UP, &nn->flags)) {
percpu_ref_kill_and_confirm(&nn->nfsd_net_ref, nfsd_net_done);
wait_for_completion(&nn->nfsd_net_confirm_done);
@@ -415,18 +421,18 @@ static void nfsd_shutdown_net(struct net *net)
nfs4_state_shutdown_net(net);
nfsd_reply_cache_shutdown(nn);
nfsd_file_cache_shutdown_net(net);
- if (nn->lockd_up) {
+ if (test_bit(NFSD_NET_LOCKD_UP, &nn->flags)) {
lockd_down(net);
- nn->lockd_up = false;
+ clear_bit(NFSD_NET_LOCKD_UP, &nn->flags);
}
wait_for_completion(&nn->nfsd_net_free_done);
}
percpu_ref_exit(&nn->nfsd_net_ref);
- if (nn->nfsd_net_up)
+ if (test_bit(NFSD_NET_UP, &nn->flags))
nfsd_shutdown_generic();
- nn->nfsd_net_up = false;
+ clear_bit(NFSD_NET_UP, &nn->flags);
}
static DEFINE_SPINLOCK(nfsd_notifier_lock);
@@ -815,7 +821,7 @@ nfsd_acl_init_request(struct svc_rqst *rqstp,
ret->mismatch.lovers = NFSD_ACL_NRVERS;
for (i = NFSD_ACL_MINVERS; i < NFSD_ACL_NRVERS; i++) {
- if (nfsd_support_acl_version(rqstp->rq_vers) &&
+ if (nfsd_support_acl_version(i) &&
nfsd_vers(nn, i, NFSD_TEST)) {
ret->mismatch.lovers = i;
break;
@@ -825,7 +831,7 @@ nfsd_acl_init_request(struct svc_rqst *rqstp,
return rpc_prog_unavail;
ret->mismatch.hivers = NFSD_ACL_MINVERS;
for (i = NFSD_ACL_NRVERS - 1; i >= NFSD_ACL_MINVERS; i--) {
- if (nfsd_support_acl_version(rqstp->rq_vers) &&
+ if (nfsd_support_acl_version(i) &&
nfsd_vers(nn, i, NFSD_TEST)) {
ret->mismatch.hivers = i;
break;
@@ -960,6 +966,20 @@ nfsd(void *vrqstp)
return 0;
}
+/*
+ * Set rq_status_counter back to an even value, indicating that the rqstp
+ * fields are no longer meaningful to a lockless reader. This pairs with the
+ * odd-valued store made once the request has been decoded, and must run on
+ * every return path that follows it so that the seq-lock like protocol used
+ * by nfsd_nl_rpc_status_get_dumpit() is not left permanently odd. The store
+ * also advances the counter so a concurrent reader detects the transition.
+ */
+static void nfsd_status_counter_set_idle(struct svc_rqst *rqstp)
+{
+ smp_store_release(&rqstp->rq_status_counter,
+ (rqstp->rq_status_counter | 1) + 1);
+}
+
/**
* nfsd_dispatch - Process an NFS or NFSACL or LOCALIO Request
* @rqstp: incoming request
@@ -1022,14 +1042,9 @@ int nfsd_dispatch(struct svc_rqst *rqstp)
if (!proc->pc_encode(rqstp, &rqstp->rq_res_stream))
goto out_encode_err;
- /*
- * Release rq_status_counter setting it to an even value after the rpc
- * request has been properly processed.
- */
- smp_store_release(&rqstp->rq_status_counter, rqstp->rq_status_counter + 1);
-
nfsd_cache_update(rqstp, rp, ntli->ntli_cachetype, nfs_reply);
out_cached_reply:
+ nfsd_status_counter_set_idle(rqstp);
return 1;
out_decode_err:
@@ -1040,12 +1055,14 @@ out_decode_err:
out_update_drop:
nfsd_cache_update(rqstp, rp, RC_NOCACHE, NULL);
out_dropit:
+ nfsd_status_counter_set_idle(rqstp);
return 0;
out_encode_err:
trace_nfsd_cant_encode_err(rqstp);
nfsd_cache_update(rqstp, rp, RC_NOCACHE, NULL);
*statp = rpc_system_err;
+ nfsd_status_counter_set_idle(rqstp);
return 1;
}
diff --git a/fs/nfsd/nfsxdr.c b/fs/nfsd/nfsxdr.c
index ae71e0621317..86c9dd4b334b 100644
--- a/fs/nfsd/nfsxdr.c
+++ b/fs/nfsd/nfsxdr.c
@@ -10,6 +10,16 @@
#include "auth.h"
/*
+ * Sun convention: a sattr time-useconds field of one full second (an
+ * otherwise out-of-range value) means "set this time to the current
+ * server time." It's needed to make permissions checks for the "touch"
+ * program across NFSv2 mounts work correctly. See description of
+ * sattr in section 6.1 of "NFS Illustrated" by Brent Callaghan,
+ * Addison-Wesley, ISBN 0-201-32750-5
+ */
+#define NFS2_SATTR_SET_TO_SERVER_TIME (1000000)
+
+/*
* Mapping of S_IF* types to NFS file types
*/
static const u32 nfs_ftypes[] = {
@@ -172,27 +182,29 @@ svcxdr_decode_sattr(struct svc_rqst *rqstp, struct xdr_stream *xdr,
tmp1 = be32_to_cpup(p++);
tmp2 = be32_to_cpup(p++);
if (tmp1 != (u32)-1 && tmp2 != (u32)-1) {
+ /*
+ * Range test here to prevent the multiplication from
+ * wrapping to a valid (but incorrect) value on 32-bit
+ * platforms.
+ */
+ if (tmp2 > NFS2_SATTR_SET_TO_SERVER_TIME)
+ return false;
iap->ia_valid |= ATTR_ATIME | ATTR_ATIME_SET;
iap->ia_atime.tv_sec = tmp1;
iap->ia_atime.tv_nsec = tmp2 * NSEC_PER_USEC;
+ if (tmp2 == NFS2_SATTR_SET_TO_SERVER_TIME)
+ iap->ia_valid &= ~ATTR_ATIME_SET;
}
tmp1 = be32_to_cpup(p++);
tmp2 = be32_to_cpup(p++);
if (tmp1 != (u32)-1 && tmp2 != (u32)-1) {
+ if (tmp2 > NFS2_SATTR_SET_TO_SERVER_TIME)
+ return false;
iap->ia_valid |= ATTR_MTIME | ATTR_MTIME_SET;
iap->ia_mtime.tv_sec = tmp1;
iap->ia_mtime.tv_nsec = tmp2 * NSEC_PER_USEC;
- /*
- * Passing the invalid value useconds=1000000 for mtime
- * is a Sun convention for "set both mtime and atime to
- * current server time". It's needed to make permissions
- * checks for the "touch" program across v2 mounts to
- * Solaris and Irix boxes work correctly. See description of
- * sattr in section 6.1 of "NFS Illustrated" by
- * Brent Callaghan, Addison-Wesley, ISBN 0-201-32750-5
- */
- if (tmp2 == 1000000)
+ if (tmp2 == NFS2_SATTR_SET_TO_SERVER_TIME)
iap->ia_valid &= ~(ATTR_ATIME_SET|ATTR_MTIME_SET);
}
diff --git a/fs/nfsd/state.h b/fs/nfsd/state.h
index dec83e92650d..bc0181ef1cb6 100644
--- a/fs/nfsd/state.h
+++ b/fs/nfsd/state.h
@@ -98,9 +98,9 @@ struct nfsd4_callback {
};
struct nfsd4_callback_ops {
- void (*prepare)(struct nfsd4_callback *);
- int (*done)(struct nfsd4_callback *, struct rpc_task *);
- void (*release)(struct nfsd4_callback *);
+ bool (*prepare)(struct nfsd4_callback *cb);
+ int (*done)(struct nfsd4_callback *cb, struct rpc_task *task);
+ void (*release)(struct nfsd4_callback *cb);
uint32_t opcode;
};
@@ -191,6 +191,66 @@ struct nfs4_cb_fattr {
};
/*
+ * FIXME: the current backchannel encoder can't handle a send buffer longer
+ * than a single page (see bc_malloc/bc_free).
+ */
+#define NOTIFY4_EVENT_QUEUE_SIZE 3
+#define NOTIFY4_PAGE_ARRAY_SIZE 1
+
+struct nfsd_notify_event {
+ refcount_t ne_ref; // refcount
+ u32 ne_mask; // FS_* mask from fsnotify callback
+ struct dentry *ne_dentry; // dentry reference to target
+ struct inode *ne_target; // inode overwritten by rename, or NULL
+ u32 ne_namelen; // length of ne_name (old name for a rename)
+ u32 ne_newnamelen; // length of new name (rename only), else 0
+ char ne_name[]; // entry name, then new name (rename only)
+};
+
+/*
+ * For a rename, the new name is snapshotted at event-alloc time and stored
+ * immediately after the (NUL-terminated) old name in ne_name[]. ne_dentry can
+ * be renamed again before the CB_NOTIFY work runs, so the new name must not be
+ * read from the live dentry at encode time.
+ */
+static inline char *nfsd_notify_event_newname(struct nfsd_notify_event *ne)
+{
+ return ne->ne_name + ne->ne_namelen + 1;
+}
+
+static inline struct nfsd_notify_event *nfsd_notify_event_get(struct nfsd_notify_event *ne)
+{
+ refcount_inc(&ne->ne_ref);
+ return ne;
+}
+
+static inline void nfsd_notify_event_put(struct nfsd_notify_event *ne)
+{
+ if (refcount_dec_and_test(&ne->ne_ref)) {
+ iput(ne->ne_target);
+ dput(ne->ne_dentry);
+ kfree(ne);
+ }
+}
+
+/*
+ * Represents a directory delegation. The callback is for handling CB_NOTIFYs.
+ * As notifications from fsnotify come in, allocate a new event, take the ncn_lock,
+ * and add it to the ncn_evt queue. The CB_NOTIFY prepare handler will take the
+ * lock, clean out the list and process it.
+ */
+struct nfsd4_cb_notify {
+ spinlock_t ncn_lock; // protects the evt queue and count
+ int ncn_evt_cnt; // count of events in ncn_evt
+ int ncn_nf_cnt; // count of valid entries in ncn_nf
+ struct nfsd_notify_event *ncn_evt[NOTIFY4_EVENT_QUEUE_SIZE]; // list of events
+ struct page *ncn_pages[NOTIFY4_PAGE_ARRAY_SIZE]; // for encoding
+ struct notify4 *ncn_nf; // array of notify4's to be sent
+ bool ncn_encode_err; // did encoding fail?
+ struct nfsd4_callback ncn_cb; // notify4 callback
+};
+
+/*
* Represents a delegation stateid. The nfs4_client holds references to these
* and they are put when it is being destroyed or when the delegation is
* returned by the client:
@@ -226,13 +286,22 @@ struct nfs4_delegation {
bool dl_written;
bool dl_setattr;
- /* for CB_GETATTR */
- struct nfs4_cb_fattr dl_cb_fattr;
+ union {
+ /* for CB_GETATTR */
+ struct nfs4_cb_fattr dl_cb_fattr;
+ /* for CB_NOTIFY */
+ struct nfsd4_cb_notify dl_cb_notify;
+ };
/* For delegated timestamps */
struct timespec64 dl_atime;
struct timespec64 dl_mtime;
struct timespec64 dl_ctime;
+
+ /* For dir delegations */
+ u32 dl_notify_mask;
+ u32 dl_child_attrs[2];
+ u32 dl_dir_attrs[2];
};
static inline bool deleg_is_read(u32 dl_type)
@@ -384,6 +453,7 @@ struct nfsd4_session {
u16 se_slot_gen;
bool se_dead;
u32 se_target_maxslots;
+ struct rcu_head rcu_head;
};
/* formatted contents of nfs4_sessionid */
@@ -496,7 +566,7 @@ struct nfs4_client {
#define NFSD4_CB_FAULT 3
int cl_cb_state;
struct nfsd4_callback cl_cb_null;
- struct nfsd4_session *cl_cb_session;
+ struct nfsd4_session __rcu *cl_cb_session;
/* for all client information that callback code might need: */
spinlock_t cl_lock;
@@ -691,7 +761,7 @@ struct nfs4_file {
*/
atomic_t fi_access[2];
u32 fi_share_deny;
- struct nfsd_file *fi_deleg_file;
+ struct nfsd_file __rcu *fi_deleg_file;
struct nfsd_file *fi_rdeleg_file;
int fi_delegees;
struct knfsd_fh fi_fhandle;
@@ -754,6 +824,7 @@ struct nfs4_layout_stateid {
struct delayed_work ls_fence_work;
unsigned int ls_fence_delay;
bool ls_fenced;
+ bool ls_fence_inflight;
};
static inline struct nfs4_layout_stateid *layoutstateid(struct nfs4_stid *s)
@@ -774,6 +845,7 @@ enum nfsd4_cb_op {
NFSPROC4_CLNT_CB_NOTIFY_LOCK,
NFSPROC4_CLNT_CB_RECALL_ANY,
NFSPROC4_CLNT_CB_GETATTR,
+ NFSPROC4_CLNT_CB_NOTIFY,
};
/* Returns true iff a is later than b: */
@@ -848,6 +920,8 @@ void nfsd_update_cmtime_attr(struct file *f, unsigned int flags);
extern struct nfs4_client_reclaim *nfs4_client_to_reclaim(struct xdr_netobj name,
struct xdr_netobj princhash, struct nfsd_net *nn);
extern bool nfs4_has_reclaimed_state(struct xdr_netobj name, struct nfsd_net *nn);
+int nfsd_handle_dir_event(u32 mask, const struct inode *dir, const void *data,
+ int data_type, const struct qstr *name);
void put_nfs4_file(struct nfs4_file *fi);
extern void nfs4_put_cpntf_state(struct nfsd_net *nn,
diff --git a/fs/nfsd/trace.h b/fs/nfsd/trace.h
index 1c5a1e50f946..db0a0dc70660 100644
--- a/fs/nfsd/trace.h
+++ b/fs/nfsd/trace.h
@@ -12,6 +12,7 @@
#include <linux/sunrpc/clnt.h>
#include <linux/sunrpc/xprt.h>
#include <trace/misc/fs.h>
+#include <trace/misc/fsnotify.h>
#include <trace/misc/nfs.h>
#include <trace/misc/sunrpc.h>
@@ -271,7 +272,7 @@ TRACE_EVENT_CONDITION(nfsd_fh_verify,
TP_CONDITION(rqstp != NULL),
TP_STRUCT__entry(
__field(unsigned int, netns_ino)
- __sockaddr(server, rqstp->rq_xprt->xpt_remotelen)
+ __sockaddr(server, rqstp->rq_xprt->xpt_locallen)
__sockaddr(client, rqstp->rq_xprt->xpt_remotelen)
__field(u32, xid)
__field(u32, fh_hash)
@@ -310,7 +311,7 @@ TRACE_EVENT_CONDITION(nfsd_fh_verify_err,
TP_CONDITION(rqstp != NULL && error),
TP_STRUCT__entry(
__field(unsigned int, netns_ino)
- __sockaddr(server, rqstp->rq_xprt->xpt_remotelen)
+ __sockaddr(server, rqstp->rq_xprt->xpt_locallen)
__sockaddr(client, rqstp->rq_xprt->xpt_remotelen)
__field(u32, xid)
__field(u32, fh_hash)
@@ -1377,6 +1378,28 @@ TRACE_EVENT(nfsd_file_fsnotify_handle_event,
__entry->nlink, __entry->mode, __entry->mask)
);
+TRACE_EVENT(nfsd_handle_dir_event,
+ TP_PROTO(u32 mask, const struct inode *dir, const struct qstr *name),
+ TP_ARGS(mask, dir, name),
+ TP_STRUCT__entry(
+ __field(u32, mask)
+ __field(dev_t, s_dev)
+ __field(u64, i_ino)
+ __string_len(name, name ? name->name : NULL,
+ name ? name->len : 0)
+ ),
+ TP_fast_assign(
+ __entry->mask = mask;
+ __entry->s_dev = dir ? dir->i_sb->s_dev : 0;
+ __entry->i_ino = dir ? dir->i_ino : 0;
+ __assign_str(name);
+ ),
+ TP_printk("inode=0x%x:0x%x:0x%llx mask=%s name=%s",
+ MAJOR(__entry->s_dev), MINOR(__entry->s_dev),
+ __entry->i_ino, show_fsnotify_mask(__entry->mask),
+ __get_str(name))
+);
+
DECLARE_EVENT_CLASS(nfsd_file_gc_class,
TP_PROTO(
const struct nfsd_file *nf
@@ -1677,6 +1700,7 @@ TRACE_EVENT(nfsd_cb_setup_err,
{ OP_CB_RECALL, "CB_RECALL" }, \
{ OP_CB_LAYOUTRECALL, "CB_LAYOUTRECALL" }, \
{ OP_CB_RECALL_ANY, "CB_RECALL_ANY" }, \
+ { OP_CB_NOTIFY, "CB_NOTIFY" }, \
{ OP_CB_NOTIFY_LOCK, "CB_NOTIFY_LOCK" }, \
{ OP_CB_OFFLOAD, "CB_OFFLOAD" })
@@ -1727,9 +1751,10 @@ DEFINE_NFSD_CB_LIFETIME_EVENT(bc_shutdown);
TRACE_EVENT(nfsd_cb_seq_status,
TP_PROTO(
const struct rpc_task *task,
- const struct nfsd4_callback *cb
+ const struct nfsd4_callback *cb,
+ const struct nfsd4_session *session
),
- TP_ARGS(task, cb),
+ TP_ARGS(task, cb, session),
TP_STRUCT__entry(
__field(unsigned int, task_id)
__field(unsigned int, client_id)
@@ -1741,8 +1766,6 @@ TRACE_EVENT(nfsd_cb_seq_status,
__field(int, seq_status)
),
TP_fast_assign(
- const struct nfs4_client *clp = cb->cb_clp;
- const struct nfsd4_session *session = clp->cl_cb_session;
const struct nfsd4_sessionid *sid =
(struct nfsd4_sessionid *)&session->se_sessionid;
@@ -1768,9 +1791,10 @@ TRACE_EVENT(nfsd_cb_seq_status,
TRACE_EVENT(nfsd_cb_free_slot,
TP_PROTO(
const struct rpc_task *task,
- const struct nfsd4_callback *cb
+ const struct nfsd4_callback *cb,
+ const struct nfsd4_session *session
),
- TP_ARGS(task, cb),
+ TP_ARGS(task, cb, session),
TP_STRUCT__entry(
__field(unsigned int, task_id)
__field(unsigned int, client_id)
@@ -1781,8 +1805,6 @@ TRACE_EVENT(nfsd_cb_free_slot,
__field(u32, slot_seqno)
),
TP_fast_assign(
- const struct nfs4_client *clp = cb->cb_clp;
- const struct nfsd4_session *session = clp->cl_cb_session;
const struct nfsd4_sessionid *sid =
(struct nfsd4_sessionid *)&session->se_sessionid;
diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c
index 1e89c7ff9493..9e05c3949cc1 100644
--- a/fs/nfsd/vfs.c
+++ b/fs/nfsd/vfs.c
@@ -139,16 +139,17 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp,
err = follow_down(&path, follow_flags);
if (err < 0)
goto out;
+
if (path.mnt == exp->ex_path.mnt && path.dentry == dentry &&
nfsd_mountpoint(dentry, exp) == 2) {
/* This is only a mountpoint in some other namespace */
- path_put(&path);
goto out;
}
exp2 = rqst_exp_get_by_name(rqstp, &path);
if (IS_ERR(exp2)) {
err = PTR_ERR(exp2);
+ exp2 = NULL;
/*
* We normally allow NFS clients to continue
* "underneath" a mountpoint that is not exported.
@@ -158,10 +159,7 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp,
*/
if (err == -ENOENT && !(exp->ex_flags & NFSEXP_V4ROOT))
err = 0;
- path_put(&path);
- goto out;
- }
- if (nfsd_v4client(rqstp) ||
+ } else if (nfsd_v4client(rqstp) ||
(exp->ex_flags & NFSEXP_CROSSMOUNT) || EX_NOHIDE(exp2)) {
/* successfully crossed mount point */
/*
@@ -175,9 +173,10 @@ nfsd_cross_mnt(struct svc_rqst *rqstp, struct dentry **dpp,
*expp = exp2;
exp2 = exp;
}
- path_put(&path);
- exp_put(exp2);
out:
+ path_put(&path);
+ if (exp2)
+ exp_put(exp2);
return err;
}
@@ -256,7 +255,7 @@ nfsd_lookup_dentry(struct svc_rqst *rqstp, struct svc_fh *fhp,
exp = exp_get(fhp->fh_export);
/* Lookup the name, but don't follow links */
- if (isdotent(name, len)) {
+ if (name_is_dot_dotdot(name, len)) {
if (len==1)
dentry = dget(dparent);
else if (dparent != exp->ex_path.dentry)
@@ -419,21 +418,22 @@ nfsd_sanitize_attrs(struct inode *inode, struct iattr *iap)
}
static __be32
-nfsd_get_write_access(struct svc_rqst *rqstp, struct svc_fh *fhp,
- struct iattr *iap)
+nfsd_may_truncate(struct svc_rqst *rqstp, struct svc_fh *fhp,
+ struct iattr *iap)
{
struct inode *inode = d_inode(fhp->fh_dentry);
- if (iap->ia_size < inode->i_size) {
- __be32 err;
+ if (iap->ia_size >= i_size_read(inode))
+ return nfs_ok;
- err = nfsd_permission(&rqstp->rq_cred,
- fhp->fh_export, fhp->fh_dentry,
- NFSD_MAY_TRUNC | NFSD_MAY_OWNER_OVERRIDE);
- if (err)
- return err;
- }
- return nfserrno(get_write_access(inode));
+ return nfsd_permission(&rqstp->rq_cred, fhp->fh_export, fhp->fh_dentry,
+ NFSD_MAY_TRUNC | NFSD_MAY_OWNER_OVERRIDE);
+}
+
+static __be32
+nfsd_get_write_access(struct svc_fh *fhp)
+{
+ return nfserrno(get_write_access(d_inode(fhp->fh_dentry)));
}
static int __nfsd_setattr(struct dentry *dentry, struct iattr *iap)
@@ -560,12 +560,17 @@ nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp,
* setattr call.
*/
if (size_change) {
- err = nfsd_get_write_access(rqstp, fhp, iap);
+ err = nfsd_get_write_access(fhp);
if (err)
return err;
}
inode_lock(inode);
+ if (size_change) {
+ err = nfsd_may_truncate(rqstp, fhp, iap);
+ if (err)
+ goto out_unlock;
+ }
err = fh_fill_pre_attrs(fhp);
if (err)
goto out_unlock;
@@ -1374,6 +1379,7 @@ nfsd_direct_write(struct svc_rqst *rqstp, struct svc_fh *fhp,
struct file *file = nf->nf_file;
unsigned int nsegs, i;
ssize_t host_err;
+ size_t expected;
nsegs = nfsd_write_dio_iters_init(nf, rqstp->rq_bvec, nvecs,
kiocb, *cnt, segments);
@@ -1395,11 +1401,13 @@ nfsd_direct_write(struct svc_rqst *rqstp, struct svc_fh *fhp,
kiocb->ki_flags |= IOCB_DONTCACHE;
}
+ expected = iov_iter_count(&segments[i].iter);
+
host_err = vfs_iocb_iter_write(file, kiocb, &segments[i].iter);
if (host_err < 0)
return host_err;
*cnt += host_err;
- if (host_err < segments[i].iter.count)
+ if (host_err < (ssize_t)expected)
break; /* partial write */
}
@@ -1876,7 +1884,7 @@ nfsd_create(struct svc_rqst *rqstp, struct svc_fh *fhp,
trace_nfsd_vfs_create(rqstp, fhp, type, fname, flen);
- if (isdotent(fname, flen))
+ if (name_is_dot_dotdot(fname, flen))
return nfserr_exist;
err = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_NOP);
@@ -1978,7 +1986,7 @@ nfsd_symlink(struct svc_rqst *rqstp, struct svc_fh *fhp,
if (!flen || path[0] == '\0')
goto out;
err = nfserr_exist;
- if (isdotent(fname, flen))
+ if (name_is_dot_dotdot(fname, flen))
goto out;
err = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_CREATE);
@@ -2055,7 +2063,7 @@ nfsd_link(struct svc_rqst *rqstp, struct svc_fh *ffhp,
if (!len)
goto out;
err = nfserr_exist;
- if (isdotent(name, len))
+ if (name_is_dot_dotdot(name, len))
goto out;
err = nfs_ok;
@@ -2166,7 +2174,8 @@ nfsd_rename(struct svc_rqst *rqstp, struct svc_fh *ffhp, char *fname, int flen,
tdentry = tfhp->fh_dentry;
err = nfserr_perm;
- if (!flen || isdotent(fname, flen) || !tlen || isdotent(tname, tlen))
+ if (!flen || name_is_dot_dotdot(fname, flen) ||
+ !tlen || name_is_dot_dotdot(tname, tlen))
goto out;
err = nfserr_xdev;
@@ -2288,7 +2297,7 @@ nfsd_unlink(struct svc_rqst *rqstp, struct svc_fh *fhp, int type,
trace_nfsd_vfs_unlink(rqstp, fhp, fname, flen);
err = nfserr_acces;
- if (!flen || isdotent(fname, flen))
+ if (!flen || name_is_dot_dotdot(fname, flen))
goto out;
err = fh_verify(rqstp, fhp, S_IFDIR, NFSD_MAY_REMOVE);
if (err)
diff --git a/fs/nfsd/xdr4.h b/fs/nfsd/xdr4.h
index 85574b2a139a..805c7122eb93 100644
--- a/fs/nfsd/xdr4.h
+++ b/fs/nfsd/xdr4.h
@@ -970,6 +970,11 @@ __be32 nfsd4_encode_fattr_to_buf(__be32 **p, int words,
struct svc_fh *fhp, struct svc_export *exp,
struct dentry *dentry,
u32 *bmval, struct svc_rqst *, int ignore_crossmnt);
+u8 *nfsd4_encode_notify_event(struct xdr_stream *xdr, struct nfsd_notify_event *nne,
+ struct nfs4_delegation *dd, struct nfsd_file *nf,
+ u32 *notify_mask);
+u8 *nfsd4_encode_dir_attr_change(struct xdr_stream *xdr, struct nfs4_delegation *dp,
+ struct nfsd_file *nf);
extern __be32 nfsd4_setclientid(struct svc_rqst *rqstp,
struct nfsd4_compound_state *, union nfsd4_op_u *u);
extern __be32 nfsd4_setclientid_confirm(struct svc_rqst *rqstp,
diff --git a/fs/nfsd/xdr4cb.h b/fs/nfsd/xdr4cb.h
index f4e29c0c701c..b06d0170d7c4 100644
--- a/fs/nfsd/xdr4cb.h
+++ b/fs/nfsd/xdr4cb.h
@@ -33,6 +33,18 @@
cb_sequence_dec_sz + \
op_dec_sz)
+#define NFS4_enc_cb_notify_sz (cb_compound_enc_hdr_sz + \
+ cb_sequence_enc_sz + \
+ 1 + enc_stateid_sz + \
+ enc_nfs4_fh_sz + \
+ 1 + \
+ NOTIFY4_EVENT_QUEUE_SIZE * \
+ (2 + (NFS4_OPAQUE_LIMIT >> 2)))
+
+#define NFS4_dec_cb_notify_sz (cb_compound_dec_hdr_sz + \
+ cb_sequence_dec_sz + \
+ op_dec_sz)
+
#define NFS4_enc_cb_notify_lock_sz (cb_compound_enc_hdr_sz + \
cb_sequence_enc_sz + \
2 + 1 + \
diff --git a/fs/nilfs2/namei.c b/fs/nilfs2/namei.c
index e2fe95de3d71..77c5f7f74fbf 100644
--- a/fs/nilfs2/namei.c
+++ b/fs/nilfs2/namei.c
@@ -86,7 +86,7 @@ nilfs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags)
* with d_instantiate().
*/
static int nilfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode;
struct nilfs_transaction_info ti;
@@ -231,7 +231,7 @@ static struct dentry *nilfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
inc_nlink(dir);
- inode = nilfs_new_inode(dir, S_IFDIR | mode);
+ inode = nilfs_new_inode(dir, mode);
err = PTR_ERR(inode);
if (IS_ERR(inode))
goto out_dir;
diff --git a/fs/notify/fanotify/fanotify.c b/fs/notify/fanotify/fanotify.c
index a3555bebad63..b59f0aa43c4b 100644
--- a/fs/notify/fanotify/fanotify.c
+++ b/fs/notify/fanotify/fanotify.c
@@ -599,6 +599,7 @@ static struct fanotify_event *fanotify_alloc_perm_event(const void *data,
pevent->hdr.pad = 0;
pevent->hdr.len = 0;
pevent->state = FAN_EVENT_INIT;
+ pevent->watchdog_cnt = 0;
pevent->path = *path;
/* NULL ppos means no range info */
pevent->ppos = range ? &range->pos : NULL;
diff --git a/fs/ntfs/attrib.c b/fs/ntfs/attrib.c
index 239b7bcbaedf..d354c3b0fae1 100644
--- a/fs/ntfs/attrib.c
+++ b/fs/ntfs/attrib.c
@@ -697,6 +697,11 @@ static bool ntfs_non_resident_attr_value_is_valid(const struct attr_record *a)
attr_len = le32_to_cpu(a->length);
min_len = offsetof(struct attr_record, data.non_resident.initialized_size) +
sizeof(a->data.non_resident.initialized_size);
+
+ /* Sparse and compressed attributes have the extra compressed_size field */
+ if (a->flags & (ATTR_IS_SPARSE | ATTR_COMPRESSION_MASK))
+ min_len += sizeof(a->data.non_resident.compressed_size);
+
if (attr_len < min_len)
return false;
@@ -4293,6 +4298,16 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
ni->initialized_size = newsize;
ctx->attr->data.non_resident.initialized_size = cpu_to_le64(newsize);
}
+
+ /*
+ * Drop any page-cache folios that now lie beyond the shrunk
+ * attribute. The clusters backing them have just been freed and the
+ * runlist truncated, so leaving stale dirty folios around makes a
+ * later writeback map a vcn past the new allocation, which fails with
+ * -ENOENT and loses the write.
+ */
+ truncate_inode_pages(VFS_I(ni)->i_mapping, newsize);
+
/* Update data size in the index. */
if (ni->type == AT_DATA && ni->name == AT_UNNAMED)
NInoSetFileNameDirty(ni);
diff --git a/fs/ntfs/compress.c b/fs/ntfs/compress.c
index 76bd806b41ed..ea29fade9b9b 100644
--- a/fs/ntfs/compress.c
+++ b/fs/ntfs/compress.c
@@ -97,26 +97,6 @@ void free_compression_buffers(void)
}
/*
- * zero_partial_compressed_page - zero out of bounds compressed page region
- * @page: page to zero
- * @initialized_size: initialized size of the attribute
- */
-static void zero_partial_compressed_page(struct page *page,
- const s64 initialized_size)
-{
- u8 *kp = page_address(page);
- unsigned int kp_ofs;
-
- ntfs_debug("Zeroing page region outside initialized size.");
- if (((s64)page->__folio_index << PAGE_SHIFT) >= initialized_size) {
- clear_page(kp);
- return;
- }
- kp_ofs = initialized_size & ~PAGE_MASK;
- memset(kp + kp_ofs, 0, PAGE_SIZE - kp_ofs);
-}
-
-/*
* handle_bounds_compressed_page - test for&handle out of bounds compressed page
* @page: page to check and handle
* @i_size: file size
@@ -125,9 +105,21 @@ static void zero_partial_compressed_page(struct page *page,
static inline void handle_bounds_compressed_page(struct page *page,
const loff_t i_size, const s64 initialized_size)
{
- if ((page->__folio_index >= (initialized_size >> PAGE_SHIFT)) &&
- (initialized_size < i_size))
- zero_partial_compressed_page(page, initialized_size);
+ loff_t pos = page_offset(page);
+
+ if ((pos + PAGE_SIZE > initialized_size) &&
+ (initialized_size < i_size)) {
+ size_t offset;
+
+ ntfs_debug("Zeroing page region outside initialized size.");
+ if (pos >= initialized_size)
+ offset = 0;
+ else
+ offset = offset_in_page(initialized_size);
+ zero_user_segment(page, offset, PAGE_SIZE);
+ } else {
+ flush_dcache_page(page);
+ }
}
/*
@@ -185,6 +177,7 @@ static int ntfs_decompress(struct page *dest_pages[], int completed_pages[],
/* Variables for uncompressed data / destination. */
struct page *dp; /* Current destination page being worked on. */
+ u8 *dp_kaddr; /* Local kmap for the current destination page. */
u8 *dp_addr; /* Current pointer into dp. */
u8 *dp_sb_start; /* Start of current sub-block in dp. */
u8 *dp_sb_end; /* End of current sb in dp (dp_sb_start + NTFS_SB_SIZE). */
@@ -199,6 +192,7 @@ static int ntfs_decompress(struct page *dest_pages[], int completed_pages[],
/* Default error code. */
int err = -EOVERFLOW;
+ dp_kaddr = NULL;
ntfs_debug("Entering, cb_size = 0x%x.", cb_size);
do_next_sb:
ntfs_debug("Beginning sub-block at offset = 0x%zx in the cb.",
@@ -231,8 +225,6 @@ return_error:
*/
handle_bounds_compressed_page(dp, i_size,
initialized_size);
- flush_dcache_page(dp);
- kunmap_local(page_address(dp));
SetPageUptodate(dp);
unlock_page(dp);
if (di == xpage)
@@ -278,7 +270,8 @@ return_error:
}
/* We have a valid destination page. Setup the destination pointers. */
- dp_addr = (u8 *)page_address(dp) + do_sb_start;
+ dp_kaddr = kmap_local_page(dp);
+ dp_addr = dp_kaddr + do_sb_start;
/* Now, we are ready to process the current sub-block (sb). */
if (!(le16_to_cpup((__le16 *)cb) & NTFS_SB_IS_COMPRESSED)) {
@@ -299,6 +292,8 @@ return_error:
/* Advance destination position to next sub-block. */
*dest_ofs += NTFS_SB_SIZE;
*dest_ofs &= ~PAGE_MASK;
+ kunmap_local(dp_kaddr);
+ dp_kaddr = NULL;
if (!(*dest_ofs)) {
finalize_page:
/*
@@ -333,6 +328,8 @@ do_next_tag:
}
/* We have finished the current sub-block. */
*dest_ofs &= ~PAGE_MASK;
+ kunmap_local(dp_kaddr);
+ dp_kaddr = NULL;
if (!(*dest_ofs))
goto finalize_page;
goto do_next_sb;
@@ -438,6 +435,8 @@ do_next_tag:
goto do_next_tag;
return_overflow:
+ if (dp_kaddr)
+ kunmap_local(dp_kaddr);
ntfs_error(NULL, "Failed. Returning -EOVERFLOW.");
goto return_error;
}
@@ -465,14 +464,14 @@ int ntfs_read_compressed_block(struct folio *folio)
struct page *page = &folio->page;
loff_t i_size;
s64 initialized_size;
- struct address_space *mapping = page->mapping;
+ struct address_space *mapping = folio->mapping;
struct ntfs_inode *ni = NTFS_I(mapping->host);
struct ntfs_volume *vol = ni->vol;
struct super_block *sb = vol->sb;
struct runlist_element *rl;
unsigned long flags;
u8 *cb, *cb_pos, *cb_end;
- unsigned long offset, index = page->__folio_index;
+ unsigned long offset, index = folio->index;
u32 cb_size = ni->itype.compressed.block_size;
u64 cb_size_mask = cb_size - 1UL;
s64 vcn;
@@ -566,7 +565,6 @@ int ntfs_read_compressed_block(struct folio *folio)
* least wasting our time.
*/
if (!PageDirty(page) && (!PageUptodate(page))) {
- kmap_local_page(page);
continue;
}
unlock_page(page);
@@ -652,8 +650,7 @@ lock_retry_remap:
}
lock_page(lpage);
- memcpy(cb_pos, page_address(lpage) + page_ofs,
- vol->cluster_size);
+ memcpy_from_page(cb_pos, lpage, page_ofs, vol->cluster_size);
unlock_page(lpage);
put_page(lpage);
cb_pos += vol->cluster_size;
@@ -692,14 +689,7 @@ lock_retry_remap:
for (; cur_page < cb_max_page; cur_page++) {
page = pages[cur_page];
if (page) {
- if (likely(!cur_ofs))
- clear_page(page_address(page));
- else
- memset(page_address(page) + cur_ofs, 0,
- PAGE_SIZE -
- cur_ofs);
- flush_dcache_page(page);
- kunmap_local(page_address(page));
+ memzero_page(page, cur_ofs, PAGE_SIZE - cur_ofs);
SetPageUptodate(page);
unlock_page(page);
if (cur_page == xpage)
@@ -717,8 +707,7 @@ lock_retry_remap:
if (cb_max_ofs && cb_pos < cb_end) {
page = pages[cur_page];
if (page)
- memset(page_address(page) + cur_ofs, 0,
- cb_max_ofs - cur_ofs);
+ memzero_page(page, cur_ofs, cb_max_ofs - cur_ofs);
/*
* No need to update cb_pos at this stage:
* cb_pos += cb_max_ofs - cur_ofs;
@@ -739,7 +728,7 @@ lock_retry_remap:
for (; cur_page < cb_max_page; cur_page++) {
page = pages[cur_page];
if (page)
- memcpy(page_address(page) + cur_ofs, cb_pos,
+ memcpy_to_page(page, cur_ofs, cb_pos,
PAGE_SIZE - cur_ofs);
cb_pos += PAGE_SIZE - cur_ofs;
cur_ofs = 0;
@@ -750,7 +739,7 @@ lock_retry_remap:
if (cb_max_ofs && cb_pos < cb_end) {
page = pages[cur_page];
if (page)
- memcpy(page_address(page) + cur_ofs, cb_pos,
+ memcpy_to_page(page, cur_ofs, cb_pos,
cb_max_ofs - cur_ofs);
cb_pos += cb_max_ofs - cur_ofs;
cur_ofs = cb_max_ofs;
@@ -767,8 +756,6 @@ lock_retry_remap:
*/
handle_bounds_compressed_page(page, i_size,
initialized_size);
- flush_dcache_page(page);
- kunmap_local(page_address(page));
SetPageUptodate(page);
unlock_page(page);
if (cur2_page == xpage)
@@ -804,7 +791,6 @@ lock_retry_remap:
page = pages[prev_cur_page];
if (page) {
flush_dcache_page(page);
- kunmap_local(page_address(page));
unlock_page(page);
if (prev_cur_page != xpage)
put_page(page);
@@ -822,14 +808,15 @@ lock_retry_remap:
for (cur_page = 0; cur_page < max_page; cur_page++) {
page = pages[cur_page];
if (page) {
+ folio = page_folio(page);
+
ntfs_error(vol->sb,
"Still have pages left! Terminating them with extreme prejudice. Inode 0x%llx, page index 0x%lx.",
- ni->mft_no, page->__folio_index);
- flush_dcache_page(page);
- kunmap_local(page_address(page));
- unlock_page(page);
+ ni->mft_no, folio->index);
+ flush_dcache_folio(folio);
+ folio_unlock(folio);
if (cur_page != xpage)
- put_page(page);
+ folio_put(folio);
pages[cur_page] = NULL;
}
}
@@ -864,7 +851,6 @@ err_out:
page = pages[i];
if (page) {
flush_dcache_page(page);
- kunmap_local(page_address(page));
unlock_page(page);
if (i != xpage)
put_page(page);
@@ -908,6 +894,12 @@ struct compress_context {
s16 prev[NTFS_SB_SIZE];
};
+struct ntfs_compress_workspace {
+ struct page **pages;
+ char *outbuf;
+ unsigned int nr_pages;
+};
+
/*
* Hash the next 3-byte sequence in the input buffer
*/
@@ -1084,12 +1076,11 @@ static void ntfs_skip_position(struct compress_context *pctx, const int i)
*
* Returns the size of the compressed block, including the
* header (minimal size is 2, maximum size is 4098)
- * 0 if an error has been met.
+ * A negative error code if an error has been met.
*/
-static unsigned int ntfs_compress_block(const char *inbuf, const int bufsize,
- char *outbuf)
+static int ntfs_compress_block(struct compress_context *pctx,
+ const char *inbuf, const int bufsize, char *outbuf)
{
- struct compress_context *pctx;
int i; /* current position */
int j; /* end of best match from current position */
int k; /* end of best match from next position */
@@ -1104,10 +1095,6 @@ static unsigned int ntfs_compress_block(const char *inbuf, const int bufsize,
int tag; /* current value of tag */
int ntag; /* count of bits still undefined in tag */
- pctx = kvzalloc(sizeof(struct compress_context), GFP_NOFS);
- if (!pctx)
- return -ENOMEM;
-
/*
* All hash chains start as empty. The special value '-1' indicates the
* end of each hash chain.
@@ -1263,22 +1250,76 @@ static unsigned int ntfs_compress_block(const char *inbuf, const int bufsize,
xout = NTFS_SB_SIZE + 2;
}
- /*
- * Free the compression context and return the total number of bytes
- * written to 'outbuf'.
- */
- kvfree(pctx);
return xout;
}
+static int ntfs_compress_workspace_init(struct ntfs_inode *ni,
+ struct ntfs_compress_workspace *ws)
+{
+ unsigned int size, i;
+
+ size = ni->itype.compressed.block_size + 2 *
+ (ni->itype.compressed.block_size / NTFS_SB_SIZE) + 2;
+ ws->nr_pages = DIV_ROUND_UP(size, PAGE_SIZE);
+ ws->pages = kcalloc(ws->nr_pages, sizeof(*ws->pages), GFP_NOFS);
+ if (!ws->pages)
+ return -ENOMEM;
+
+ for (i = 0; i < ws->nr_pages; i++) {
+ ws->pages[i] = alloc_page(GFP_NOFS);
+ if (!ws->pages[i])
+ goto free_pages;
+ }
+
+ ws->outbuf = vmap(ws->pages, ws->nr_pages, VM_MAP, PAGE_KERNEL);
+ if (!ws->outbuf)
+ goto free_pages;
+ return 0;
+
+free_pages:
+ while (i)
+ put_page(ws->pages[--i]);
+ kfree(ws->pages);
+ return -ENOMEM;
+}
+
+static void ntfs_compress_workspace_free(struct ntfs_compress_workspace *ws)
+{
+ unsigned int i;
+
+ vunmap(ws->outbuf);
+ for (i = 0; i < ws->nr_pages; i++)
+ put_page(ws->pages[i]);
+ kfree(ws->pages);
+}
+
+static void ntfs_copy_cb(struct page **pages, int pages_per_cb,
+ unsigned int page_offset,
+ struct ntfs_compress_workspace *ws, unsigned int bytes)
+{
+ unsigned int copied = 0, i;
+
+ for (i = 0; i < pages_per_cb && copied < bytes; i++) {
+ unsigned int offset = i ? 0 : page_offset;
+ unsigned int len = min(bytes - copied, PAGE_SIZE - offset);
+ void *addr = kmap_local_page(pages[i]);
+
+ memcpy(ws->outbuf + copied, addr + offset, len);
+ kunmap_local(addr);
+ copied += len;
+ }
+}
+
static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
- int pages_per_cb)
+ int pages_per_cb, unsigned int page_offset,
+ struct compress_context *ctx, struct ntfs_compress_workspace *ws)
{
struct ntfs_volume *vol = ni->vol;
- char *outbuf = NULL, *pbuf, *inbuf;
- u32 compsz, p, insz = pages_per_cb << PAGE_SHIFT;
+ char *outbuf = ws->outbuf, *pbuf;
+ u32 compsz, p, insz = ni->itype.compressed.block_size;
s32 rounded, bio_size;
- unsigned int sz, bsz;
+ int sz;
+ unsigned int bsz;
bool fail = false, allzeroes;
/* a single compressed zero */
static char onezero[] = {0x01, 0xb0, 0x00, 0x00};
@@ -1286,54 +1327,36 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
static char twozeroes[] = {0x02, 0xb0, 0x00, 0x00, 0x00};
/* more compressed zeroes, to be followed by some count */
static char morezeroes[] = {0x03, 0xb0, 0x02, 0x00};
- struct page **pages_disk = NULL, *pg;
- s64 bio_lcn;
+ s64 bio_lcn, bio_pos;
struct runlist_element *rlc, *rl;
int i, err;
- int pages_count = (round_up(ni->itype.compressed.block_size + 2 *
- (ni->itype.compressed.block_size / NTFS_SB_SIZE) + 2, PAGE_SIZE)) / PAGE_SIZE;
+ u32 cb_clusters = ni->itype.compressed.block_clusters;
size_t new_rl_count;
struct bio *bio = NULL;
- loff_t new_length;
+ loff_t cb_pos, new_length;
s64 new_vcn;
- inbuf = vmap(pages, pages_per_cb, VM_MAP, PAGE_KERNEL_RO);
- if (!inbuf)
- return -ENOMEM;
-
- /* may need 2 extra bytes per block and 2 more bytes */
- pages_disk = kcalloc(pages_count, sizeof(struct page *), GFP_NOFS);
- if (!pages_disk) {
- vunmap(inbuf);
- return -ENOMEM;
- }
-
- for (i = 0; i < pages_count; i++) {
- pg = alloc_page(GFP_KERNEL);
- if (!pg) {
- err = -ENOMEM;
- goto out;
- }
- pages_disk[i] = pg;
- lock_page(pg);
- kmap_local_page(pg);
- }
-
- outbuf = vmap(pages_disk, pages_count, VM_MAP, PAGE_KERNEL);
- if (!outbuf) {
- err = -ENOMEM;
- goto out;
- }
-
compsz = 0;
allzeroes = true;
for (p = 0; (p < insz) && !fail; p += NTFS_SB_SIZE) {
+ unsigned int input_offset = page_offset + p;
+ unsigned int page_idx = input_offset >> PAGE_SHIFT;
+ const char *input;
+ void *addr;
+
if ((p + NTFS_SB_SIZE) < insz)
bsz = NTFS_SB_SIZE;
else
bsz = insz - p;
pbuf = &outbuf[compsz];
- sz = ntfs_compress_block(&inbuf[p], bsz, pbuf);
+ addr = kmap_local_page(pages[page_idx]);
+ input = addr + offset_in_page(input_offset);
+ sz = ntfs_compress_block(ctx, input, bsz, pbuf);
+ kunmap_local(addr);
+ if (sz < 0) {
+ err = sz;
+ goto out;
+ }
/* fail if all the clusters (or more) are needed */
if (!sz || ((compsz + sz + vol->cluster_size + 2) >
ni->itype.compressed.block_size))
@@ -1360,28 +1383,25 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
}
}
+ cb_pos = pos & ~((loff_t)ni->itype.compressed.block_size - 1);
+ new_vcn = ntfs_bytes_to_cluster(vol, cb_pos);
+
if (!fail && !allzeroes) {
outbuf[compsz++] = 0;
outbuf[compsz++] = 0;
rounded = ((compsz - 1) | (vol->cluster_size - 1)) + 1;
memset(&outbuf[compsz], 0, rounded - compsz);
bio_size = rounded;
- pages = pages_disk;
} else if (allzeroes) {
- err = 0;
+ err = ntfs_non_resident_attr_punch_hole(ni, new_vcn, cb_clusters);
goto out;
} else {
+ ntfs_copy_cb(pages, pages_per_cb, page_offset, ws, insz);
bio_size = insz;
}
- new_vcn = ntfs_bytes_to_cluster(vol,
- pos & ~((loff_t)ni->itype.compressed.block_size - 1));
new_length = ntfs_bytes_to_cluster(vol, round_up(bio_size, vol->cluster_size));
- err = ntfs_non_resident_attr_punch_hole(ni, new_vcn, ni->itype.compressed.block_clusters);
- if (err < 0)
- goto out;
-
rlc = ntfs_cluster_alloc(vol, new_vcn, new_length, -1, DATA_ZONE,
false, true, true);
if (IS_ERR(rlc)) {
@@ -1390,74 +1410,56 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
}
bio_lcn = rlc->lcn;
+ bio_pos = ntfs_cluster_to_bytes(vol, bio_lcn);
+ bio = bio_alloc(vol->sb->s_bdev, DIV_ROUND_UP(bio_size, PAGE_SIZE),
+ REQ_OP_WRITE, GFP_NOIO);
+ bio->bi_iter.bi_sector = ntfs_bytes_to_sector(vol, bio_pos);
+
+ for (i = 0; bio_size; i++) {
+ unsigned int len = min_t(unsigned int, bio_size, PAGE_SIZE);
+
+ if (bio_add_page(bio, ws->pages[i], len, 0) != len) {
+ err = -EIO;
+ bio_put(bio);
+ goto free_rlc;
+ }
+ bio_size -= len;
+ }
+
+ err = submit_bio_wait(bio);
+ bio_put(bio);
+ if (err)
+ goto free_rlc;
+
+ /* Do not discard the old compression block until the new one is safe. */
+ err = ntfs_non_resident_attr_punch_hole(ni, new_vcn, cb_clusters);
+ if (err)
+ goto free_rlc;
+
down_write(&ni->runlist.lock);
rl = ntfs_runlists_merge(&ni->runlist, rlc, 0, &new_rl_count);
if (IS_ERR(rl)) {
up_write(&ni->runlist.lock);
ntfs_error(vol->sb, "Failed to merge runlists");
err = PTR_ERR(rl);
- if (ntfs_cluster_free_from_rl(vol, rlc))
- ntfs_error(vol->sb, "Failed to free hot clusters.");
- kvfree(rlc);
- goto out;
+ goto free_rlc;
}
ni->runlist.count = new_rl_count;
ni->runlist.rl = rl;
+ rlc = NULL;
err = ntfs_attr_update_mapping_pairs(ni, 0);
up_write(&ni->runlist.lock);
- if (err) {
+ if (err)
err = -EIO;
- goto out;
- }
+ goto out;
- i = 0;
- while (bio_size > 0) {
- int page_size;
-
- if (bio_size >= PAGE_SIZE) {
- page_size = PAGE_SIZE;
- bio_size -= PAGE_SIZE;
- } else {
- page_size = bio_size;
- bio_size = 0;
- }
-
-setup_bio:
- if (!bio) {
- bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE,
- GFP_NOIO);
- bio->bi_iter.bi_sector =
- ntfs_bytes_to_sector(vol,
- ntfs_cluster_to_bytes(vol, bio_lcn + i));
- }
-
- if (!bio_add_page(bio, pages[i], page_size, 0)) {
- err = submit_bio_wait(bio);
- bio_put(bio);
- if (err)
- goto out;
- bio = NULL;
- goto setup_bio;
- }
- i++;
- }
-
- err = submit_bio_wait(bio);
- bio_put(bio);
+free_rlc:
+ if (ntfs_cluster_free_from_rl(vol, rlc))
+ ntfs_error(vol->sb, "Failed to free hot clusters.");
+ kvfree(rlc);
out:
- vunmap(outbuf);
- for (i = 0; i < pages_count; i++) {
- pg = pages_disk[i];
- if (pg) {
- kunmap_local(page_address(pg));
- unlock_page(pg);
- put_page(pg);
- }
- }
- kfree(pages_disk);
- vunmap(inbuf);
NInoSetFileNameDirty(ni);
mark_mft_record_dirty(ni);
@@ -1467,31 +1469,39 @@ out:
int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
struct iov_iter *from)
{
+ struct ntfs_compress_workspace ws = {};
+ struct compress_context *ctx;
struct folio *folio;
struct page **pages = NULL, *page;
- int pages_per_cb = ni->itype.compressed.block_size >> PAGE_SHIFT;
+ int pages_per_cb;
int cb_size = ni->itype.compressed.block_size, cb_off, err = 0;
int i, ip;
size_t written = 0;
struct address_space *mapping = VFS_I(ni)->i_mapping;
- if (NInoCompressed(ni) && pos + count > ni->allocated_size) {
- int err;
- loff_t end = pos + count;
-
- err = ntfs_attr_expand(ni, end,
- round_up(end, ni->itype.compressed.block_size));
- if (err)
- return err;
- }
+ pages_per_cb = DIV_ROUND_UP(offset_in_page(pos & ~(cb_size - 1)) +
+ cb_size, PAGE_SIZE);
pages = kmalloc_array(pages_per_cb, sizeof(struct page *), GFP_NOFS);
if (!pages)
return -ENOMEM;
+ ctx = kvzalloc_obj(*ctx, GFP_NOFS);
+ if (!ctx) {
+ kfree(pages);
+ return -ENOMEM;
+ }
+ err = ntfs_compress_workspace_init(ni, &ws);
+ if (err) {
+ kvfree(ctx);
+ kfree(pages);
+ return err;
+ }
while (count) {
pgoff_t index;
size_t copied, bytes;
+ unsigned int page_offset;
+ bool full_cb;
int off;
off = pos & (cb_size - 1);
@@ -1500,7 +1510,11 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
bytes = count;
cb_off = pos & ~(cb_size - 1);
+ page_offset = offset_in_page(cb_off);
+ pages_per_cb = DIV_ROUND_UP(page_offset + cb_size, PAGE_SIZE);
index = cb_off >> PAGE_SHIFT;
+ full_cb = !off && bytes == cb_size && !page_offset &&
+ !(cb_size & (PAGE_SIZE - 1));
if (unlikely(fault_in_iov_iter_readable(from, bytes))) {
err = -EFAULT;
@@ -1508,7 +1522,10 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
}
for (i = 0; i < pages_per_cb; i++) {
- folio = read_mapping_folio(mapping, index + i, NULL);
+ if (full_cb)
+ folio = filemap_grab_folio(mapping, index + i);
+ else
+ folio = read_mapping_folio(mapping, index + i, NULL);
if (IS_ERR(folio)) {
for (ip = 0; ip < i; ip++) {
folio_unlock(page_folio(pages[ip]));
@@ -1518,7 +1535,8 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
goto out;
}
- folio_lock(folio);
+ if (!full_cb)
+ folio_lock(folio);
pages[i] = folio_page(folio, 0);
}
@@ -1548,13 +1566,26 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
}
}
- err = ntfs_write_cb(ni, pos, pages, pages_per_cb);
+ if (!copied) {
+ err = -EFAULT;
+ goto release_pages;
+ }
+ err = ntfs_write_cb(ni, pos, pages, pages_per_cb, page_offset, ctx, &ws);
+ if (!err && pos + copied > ni->initialized_size) {
+ mutex_lock(&ni->mrec_lock);
+ err = ntfs_attr_set_initialized_size(ni, pos + copied);
+ mutex_unlock(&ni->mrec_lock);
+ }
+
+release_pages:
for (i = 0; i < pages_per_cb; i++) {
folio = page_folio(pages[i]);
- if (i < ip) {
+ if (!err) {
folio_clear_dirty(folio);
folio_mark_uptodate(folio);
+ } else {
+ folio_clear_uptodate(folio);
}
folio_unlock(folio);
folio_put(folio);
@@ -1570,6 +1601,8 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
}
out:
+ ntfs_compress_workspace_free(&ws);
+ kvfree(ctx);
kfree(pages);
if (err < 0)
written = err;
diff --git a/fs/ntfs/dir.c b/fs/ntfs/dir.c
index 6fa9ae3377cb..2d594cbb4ebe 100644
--- a/fs/ntfs/dir.c
+++ b/fs/ntfs/dir.c
@@ -966,13 +966,14 @@ filldir:
*/
private = file->private_data;
kfree(private->key);
- private->key = kmalloc(le16_to_cpu(next->key_length), GFP_KERNEL);
+ private->key = kmemdup(&next->key.file_name,
+ le16_to_cpu(next->key_length),
+ GFP_KERNEL);
if (!private->key) {
err = -ENOMEM;
goto out;
}
- memcpy(private->key, &next->key.file_name, le16_to_cpu(next->key_length));
private->key_length = next->key_length;
break;
}
diff --git a/fs/ntfs/ea.c b/fs/ntfs/ea.c
index 0cd192752b7c..4fb10d51211c 100644
--- a/fs/ntfs/ea.c
+++ b/fs/ntfs/ea.c
@@ -196,6 +196,9 @@ static int ntfs_set_ea(struct inode *inode, const char *name, size_t name_len,
struct ea_attr *p_ea;
u32 ea_info_qsize = 0;
char *ea_buf = NULL;
+ char *new_ea_buf;
+ char *old_ea_buf = NULL;
+ struct ea_information old_ea_info;
size_t new_ea_size = ALIGN(struct_size(p_ea, ea_name, 1 + name_len + val_size), 4);
s64 ea_off, ea_info_size, all_ea_size, ea_size;
@@ -249,6 +252,22 @@ create_ea_info:
err = -EEXIST;
goto out;
}
+ if ((flags & XATTR_REPLACE) && !val_size) {
+ old_ea_info = *p_ea_info;
+ old_ea_buf = kvmemdup(ea_buf, all_ea_size, GFP_NOFS);
+ if (!old_ea_buf) {
+ err = -ENOMEM;
+ goto out;
+ }
+ }
+
+ /* Check the final $EA size before removing the old entry. */
+ if (val_size &&
+ ntfs_attr_size_bounds_check(ni->vol, AT_EA,
+ ea_info_qsize - ea_size + new_ea_size)) {
+ err = -EFBIG;
+ goto out;
+ }
p_ea = (struct ea_attr *)(ea_buf + ea_off);
@@ -267,17 +286,39 @@ create_ea_info:
ea_info_qsize -= ea_size;
p_ea_info->ea_query_length = cpu_to_le32(ea_info_qsize);
- err = ntfs_write_ea(ni, AT_EA_INFORMATION, (char *)p_ea_info, 0,
- sizeof(struct ea_information), false);
- if (err)
- goto out;
+ if ((flags & XATTR_REPLACE) && !val_size && !ea_info_qsize) {
+ err = ntfs_attr_remove(ni, AT_EA, AT_UNNAMED, 0);
+ if (err)
+ goto out;
- err = ntfs_write_ea(ni, AT_EA, ea_buf, 0, ea_info_qsize, true);
- if (err)
+ err = ntfs_attr_remove(ni, AT_EA_INFORMATION, AT_UNNAMED, 0);
+ if (err) {
+ /* Restore the original $EA if $EA_INFORMATION removal failed. */
+ ntfs_attr_add(ni, AT_EA, AT_UNNAMED, 0, old_ea_buf,
+ all_ea_size);
+ ea_info_qsize = le32_to_cpu(old_ea_info.ea_query_length);
+ }
goto out;
+ }
if ((flags & XATTR_REPLACE) && !val_size) {
- /* Remove xattr. */
+ err = ntfs_write_ea(ni, AT_EA, ea_buf, 0, ea_info_qsize,
+ true);
+ if (err) {
+ ntfs_write_ea(ni, AT_EA, old_ea_buf, 0,
+ all_ea_size, false);
+ goto out;
+ }
+
+ err = ntfs_write_ea(ni, AT_EA_INFORMATION, (char *)p_ea_info,
+ 0, sizeof(struct ea_information), false);
+ if (err) {
+ ntfs_write_ea(ni, AT_EA, old_ea_buf, 0,
+ all_ea_size, false);
+ ntfs_write_ea(ni, AT_EA_INFORMATION,
+ (char *)&old_ea_info, 0,
+ sizeof(old_ea_info), false);
+ }
goto out;
}
} else {
@@ -285,22 +326,30 @@ create_ea_info:
err = -ENODATA;
goto out;
}
- }
- kvfree(ea_buf);
+ if (ntfs_attr_size_bounds_check(ni->vol, AT_EA,
+ ea_info_qsize + new_ea_size)) {
+ err = -EFBIG;
+ goto out;
+ }
+ }
alloc_new_ea:
- ea_buf = kzalloc(new_ea_size, GFP_NOFS);
- if (!ea_buf) {
+ new_ea_buf = kvzalloc(ea_info_qsize + new_ea_size, GFP_NOFS);
+ if (!new_ea_buf) {
err = -ENOMEM;
goto out;
}
+ if (ea_info_qsize)
+ memcpy(new_ea_buf, ea_buf, ea_info_qsize);
+ kvfree(ea_buf);
+ ea_buf = new_ea_buf;
+ p_ea = (struct ea_attr *)(ea_buf + ea_info_qsize);
/*
* EA and REPARSE_POINT compatibility not checked any more,
* required by Windows 10, but having both may lead to
* problems with earlier versions.
*/
- p_ea = (struct ea_attr *)ea_buf;
memcpy(p_ea->ea_name, name, name_len);
p_ea->ea_name_length = name_len;
p_ea->ea_name[name_len] = 0;
@@ -312,8 +361,7 @@ alloc_new_ea:
p_ea_info->ea_length = cpu_to_le16(ea_packed);
p_ea_info->ea_query_length = cpu_to_le32(ea_info_qsize + new_ea_size);
- if (ea_packed > 0xffff ||
- ntfs_attr_size_bounds_check(ni->vol, AT_EA, new_ea_size)) {
+ if (ea_packed > 0xffff) {
err = -EFBIG;
goto out;
}
@@ -322,13 +370,13 @@ alloc_new_ea:
* no EA or EA_INFORMATION : add them
*/
if (!ntfs_attr_exist(ni, AT_EA, AT_UNNAMED, 0)) {
- err = ntfs_attr_add(ni, AT_EA, AT_UNNAMED, 0, (char *)p_ea,
- new_ea_size);
+ err = ntfs_attr_add(ni, AT_EA, AT_UNNAMED, 0, ea_buf,
+ ea_info_qsize + new_ea_size);
if (err)
goto out;
} else {
- err = ntfs_write_ea(ni, AT_EA, (char *)p_ea, ea_info_qsize,
- new_ea_size, false);
+ err = ntfs_write_ea(ni, AT_EA, ea_buf, 0,
+ ea_info_qsize + new_ea_size, true);
if (err)
goto out;
}
@@ -348,6 +396,7 @@ out:
NInoClearHasEA(ni);
kvfree(ea_buf);
+ kvfree(old_ea_buf);
kvfree(p_ea_info);
return err;
@@ -581,7 +630,8 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
struct mft_record *m;
struct attr_record *a;
__le16 new_aflags;
- int mp_size, mp_ofs, name_ofs, arec_size, err;
+ u16 old_name_ofs, old_mp_ofs;
+ int mp_size, mp_ofs, name_ofs, old_arec_size, arec_size, err;
m = map_mft_record(ni);
if (IS_ERR(m))
@@ -613,8 +663,10 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
else
new_aflags &= ~ATTR_IS_COMPRESSED;
- if (new_aflags == a->flags)
- return 0;
+ if (new_aflags == a->flags) {
+ err = 0;
+ goto err_out;
+ }
if ((new_aflags & (ATTR_IS_SPARSE | ATTR_IS_COMPRESSED)) ==
(ATTR_IS_SPARSE | ATTR_IS_COMPRESSED)) {
@@ -623,15 +675,40 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
goto err_out;
}
- if (!a->non_resident)
- goto out;
+ if (!a->non_resident) {
+ if (!(new_aflags & (ATTR_IS_SPARSE | ATTR_IS_COMPRESSED)))
+ return 0;
- if (a->data.non_resident.data_size) {
- pr_err("Can't change sparsed/compressed for non-empty file\n");
- err = -EOPNOTSUPP;
- goto err_out;
+ if (le32_to_cpu(a->data.resident.value_length)) {
+ pr_err("Can't change sparse/compressed for non-empty file");
+ err = -EOPNOTSUPP;
+ goto err_out;
+ }
+
+ err = ntfs_attr_make_non_resident(ni, 0);
+ if (err)
+ goto err_out;
+
+ ntfs_attr_reinit_search_ctx(ctx);
+ err = ntfs_attr_lookup(ni->type, ni->name,
+ ni->name_len, CASE_SENSITIVE,
+ 0, NULL, 0, ctx);
+ if (err) {
+ err = -EINVAL;
+ goto err_out;
+ }
+ a = ctx->attr;
+ } else {
+ if (a->data.non_resident.data_size) {
+ pr_err("Can't change sparsed/compressed for non-empty file");
+ err = -EOPNOTSUPP;
+ goto err_out;
+ }
}
+ old_name_ofs = le16_to_cpu(a->name_offset);
+ old_mp_ofs = le16_to_cpu(a->data.non_resident.mapping_pairs_offset);
+
if (new_aflags & (ATTR_IS_SPARSE | ATTR_IS_COMPRESSED))
name_ofs = (offsetof(struct attr_record,
data.non_resident.compressed_size) +
@@ -649,11 +726,37 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
mp_ofs = (name_ofs + a->name_length * sizeof(__le16) + 7) & ~7;
arec_size = (mp_ofs + mp_size + 7) & ~7;
+ old_arec_size = le32_to_cpu(a->length);
+
+ /*
+ * Move payloads before shrinking the record. Otherwise resizing moves
+ * the following attribute over the old payload before it can be copied.
+ */
+ if (arec_size < old_arec_size) {
+ if (a->name_length && name_ofs != old_name_ofs)
+ memmove((u8 *)a + name_ofs, (u8 *)a + old_name_ofs,
+ a->name_length * sizeof(__le16));
+ if (mp_ofs != old_mp_ofs)
+ memmove((u8 *)a + mp_ofs, (u8 *)a + old_mp_ofs, mp_size);
+ }
err = ntfs_attr_record_resize(m, a, arec_size);
if (unlikely(err))
goto err_out;
+ /*
+ * When compressed/sparse state changes, the non-resident header grows or
+ * shrinks by the compressed_size field. Update the in-record payload layout
+ * to match the new offsets before exposing the new mapping_pairs_offset.
+ */
+ if (arec_size > old_arec_size) {
+ if (mp_ofs != old_mp_ofs)
+ memmove((u8 *)a + mp_ofs, (u8 *)a + old_mp_ofs, mp_size);
+ if (a->name_length)
+ memmove((u8 *)a + name_ofs, (u8 *)a + old_name_ofs,
+ a->name_length * sizeof(__le16));
+ }
+
if (new_aflags & (ATTR_IS_SPARSE | ATTR_IS_COMPRESSED)) {
a->data.non_resident.compression_unit = 0;
if (new_aflags & ATTR_IS_COMPRESSED || ni->vol->major_ver < 3)
@@ -674,28 +777,31 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
ni->itype.compressed.block_size_bits = 0;
ni->itype.compressed.block_clusters = 0;
}
-
- if (new_aflags & ATTR_IS_SPARSE) {
- NInoSetSparse(ni);
- ni->flags |= FILE_ATTR_SPARSE_FILE;
- }
-
- if (new_aflags & ATTR_IS_COMPRESSED) {
- NInoSetCompressed(ni);
- ni->flags |= FILE_ATTR_COMPRESSED;
- }
} else {
- ni->flags &= ~(FILE_ATTR_SPARSE_FILE | FILE_ATTR_COMPRESSED);
a->data.non_resident.compression_unit = 0;
- NInoClearSparse(ni);
- NInoClearCompressed(ni);
}
a->name_offset = cpu_to_le16(name_ofs);
a->data.non_resident.mapping_pairs_offset = cpu_to_le16(mp_ofs);
-out:
a->flags = new_aflags;
+
+ if (new_aflags & ATTR_IS_SPARSE) {
+ NInoSetSparse(ni);
+ ni->flags |= FILE_ATTR_SPARSE_FILE;
+ } else {
+ NInoClearSparse(ni);
+ ni->flags &= ~FILE_ATTR_SPARSE_FILE;
+ }
+
+ if (new_aflags & ATTR_IS_COMPRESSED) {
+ NInoSetCompressed(ni);
+ ni->flags |= FILE_ATTR_COMPRESSED;
+ } else {
+ NInoClearCompressed(ni);
+ ni->flags &= ~FILE_ATTR_COMPRESSED;
+ }
+
mark_mft_record_dirty(ctx->ntfs_ino);
err_out:
if (ctx)
diff --git a/fs/ntfs/file.c b/fs/ntfs/file.c
index 6a7b638e523d..a6855d62e1fc 100644
--- a/fs/ntfs/file.c
+++ b/fs/ntfs/file.c
@@ -268,23 +268,19 @@ static int ntfs_setattr_size(struct inode *vi, struct iattr *attr)
return err;
inode_dio_wait(vi);
- truncate_setsize(vi, attr->ia_size);
+ if (attr->ia_size > old_size) {
+ truncate_pagecache(vi, old_size);
+ i_size_write(vi, attr->ia_size);
+ pagecache_isize_extended(vi, old_size, attr->ia_size);
+ } else
+ truncate_setsize(vi, attr->ia_size);
+
err = ntfs_truncate_vfs(vi, attr->ia_size, old_size);
if (err) {
i_size_write(vi, old_size);
return err;
}
- if (NInoNonResident(ni) && attr->ia_size > old_size &&
- old_size % PAGE_SIZE != 0) {
- loff_t len = min_t(loff_t,
- round_up(old_size, PAGE_SIZE) - old_size,
- attr->ia_size - old_size);
- err = iomap_zero_range(vi, old_size, len,
- NULL, &ntfs_seek_iomap_ops,
- &ntfs_iomap_folio_ops, NULL);
- }
-
return err;
}
@@ -346,14 +342,17 @@ int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
if (ia_valid & ATTR_MODE)
flags |= NTFS_EA_MODE;
+ mutex_lock(&ni->mrec_lock);
+ err = ntfs_ea_set_wsl_inode(vi, 0, NULL, flags);
+ mutex_unlock(&ni->mrec_lock);
+ if (err)
+ goto out;
+
+ /* Apply mount masks after saving the unmasked POSIX mode. */
if (S_ISDIR(vi->i_mode))
vi->i_mode &= ~vol->dmask;
else
vi->i_mode &= ~vol->fmask;
-
- mutex_lock(&ni->mrec_lock);
- ntfs_ea_set_wsl_inode(vi, 0, NULL, flags);
- mutex_unlock(&ni->mrec_lock);
}
mark_inode_dirty(vi);
@@ -535,6 +534,31 @@ out:
return ret;
}
+static int ntfs_expand_for_write(struct ntfs_inode *ni, loff_t end)
+{
+ struct ntfs_volume *vol = ni->vol;
+ loff_t prealloc_size = 0;
+ int err;
+
+ if (end <= ni->data_size)
+ return 0;
+
+ if (NInoCompressed(ni)) {
+ if (end > ni->allocated_size)
+ prealloc_size = round_up(end,
+ ni->itype.compressed.block_size);
+ } else if (end > ni->allocated_size &&
+ end < ni->allocated_size + vol->preallocated_size) {
+ prealloc_size = ni->allocated_size + vol->preallocated_size;
+ }
+
+ mutex_lock(&ni->mrec_lock);
+ err = ntfs_attr_expand(ni, end, prealloc_size);
+ mutex_unlock(&ni->mrec_lock);
+
+ return err;
+}
+
static ssize_t ntfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from)
{
struct file *file = iocb->ki_filp;
@@ -543,7 +567,7 @@ static ssize_t ntfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from)
struct ntfs_volume *vol = ni->vol;
ssize_t ret;
ssize_t count;
- loff_t pos;
+ loff_t pos, end;
int err;
loff_t old_data_size, old_init_size;
@@ -580,10 +604,24 @@ static ssize_t ntfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from)
pos = iocb->ki_pos;
count = ret;
+ end = pos + count;
old_data_size = ni->data_size;
old_init_size = ni->initialized_size;
+ if (end > old_data_size) {
+ ret = ntfs_expand_for_write(ni, end);
+ if (ret < 0)
+ goto out;
+ }
+
+ if (NInoNonResident(ni) && !NInoCompressed(ni) &&
+ end > old_init_size) {
+ ret = ntfs_extend_initialized_size(vi, pos, end);
+ if (ret < 0)
+ goto out;
+ }
+
if (NInoNonResident(ni) && NInoCompressed(ni)) {
ret = ntfs_compress_write(ni, pos, count, from);
if (ret > 0)
@@ -655,7 +693,7 @@ static int ntfs_file_mmap_prepare(struct vm_area_desc *desc)
from + desc->end - desc->start);
if (NTFS_I(inode)->initialized_size < to) {
- err = ntfs_extend_initialized_size(inode, to, to, false);
+ err = ntfs_extend_initialized_size(inode, to, to);
if (err)
return err;
}
@@ -1126,13 +1164,9 @@ out:
filemap_invalidate_unlock(vi->i_mapping);
if (!err) {
if (mode == 0 && NInoNonResident(ni) &&
- offset > old_size && old_size % PAGE_SIZE != 0) {
- loff_t len = min_t(loff_t,
- round_up(old_size, PAGE_SIZE) - old_size,
- offset - old_size);
- err = iomap_zero_range(vi, old_size, len, NULL,
- &ntfs_seek_iomap_ops,
- &ntfs_iomap_folio_ops, NULL);
+ offset > old_size) {
+ truncate_pagecache(vi, old_size);
+ pagecache_isize_extended(vi, old_size, offset);
}
NInoSetFileNameDirty(ni);
inode_set_mtime_to_ts(vi, inode_set_ctime_current(vi));
diff --git a/fs/ntfs/index.c b/fs/ntfs/index.c
index faa7ee920a3a..409759eab55d 100644
--- a/fs/ntfs/index.c
+++ b/fs/ntfs/index.c
@@ -1112,6 +1112,7 @@ static struct index_block *ntfs_ir_to_ib(struct index_root *ir, s64 ib_vcn)
struct index_entry *ie_last;
char *ies_start, *ies_end;
int i;
+ u32 ib_cap;
ntfs_debug("Entering\n");
@@ -1127,6 +1128,16 @@ static struct index_block *ntfs_ir_to_ib(struct index_root *ir, s64 ib_vcn)
* as well, which can never have any data.
*/
i = (char *)ie_last - ies_start + le16_to_cpu(ie_last->length);
+
+ /* Entries must fit in the allocated index block */
+ ib_cap = le32_to_cpu(ib->index.allocated_size) -
+ le32_to_cpu(ib->index.entries_offset);
+ if ((u32)i > ib_cap) {
+ ntfs_error(NULL, "Entries (%d B) exceed IB capacity", i);
+ kvfree(ib);
+ return NULL;
+ }
+
memcpy(ntfs_ie_get_first(&ib->index), ies_start, i);
ib->index.flags = ir->index.flags;
diff --git a/fs/ntfs/inode.c b/fs/ntfs/inode.c
index 7381a18cfadd..50e244aa372e 100644
--- a/fs/ntfs/inode.c
+++ b/fs/ntfs/inode.c
@@ -2401,7 +2401,7 @@ int ntfs_show_options(struct seq_file *sf, struct dentry *root)
}
int ntfs_extend_initialized_size(struct inode *vi, const loff_t offset,
- const loff_t new_size, bool bsync)
+ const loff_t new_size)
{
struct ntfs_inode *ni = NTFS_I(vi);
loff_t old_init_size;
@@ -2428,10 +2428,6 @@ int ntfs_extend_initialized_size(struct inode *vi, const loff_t offset,
&ntfs_iomap_folio_ops, NULL);
if (err)
return err;
- if (bsync)
- err = filemap_write_and_wait_range(vi->i_mapping,
- old_init_size,
- offset - 1);
}
diff --git a/fs/ntfs/inode.h b/fs/ntfs/inode.h
index 9aacd5787ffe..c6d065aaecd5 100644
--- a/fs/ntfs/inode.h
+++ b/fs/ntfs/inode.h
@@ -352,7 +352,7 @@ static inline void ntfs_commit_inode(struct inode *vi)
int ntfs_inode_sync_filename(struct ntfs_inode *ni);
int ntfs_extend_initialized_size(struct inode *vi, const loff_t offset,
- const loff_t new_size, bool bsync);
+ const loff_t new_size);
void ntfs_set_vfs_operations(struct inode *inode, mode_t mode, dev_t dev);
struct folio *ntfs_get_locked_folio(struct address_space *mapping,
pgoff_t index, pgoff_t end_index, struct file_ra_state *ra);
diff --git a/fs/ntfs/iomap.c b/fs/ntfs/iomap.c
index 52eecf5cb256..26a1831a2c18 100644
--- a/fs/ntfs/iomap.c
+++ b/fs/ntfs/iomap.c
@@ -675,21 +675,7 @@ static int ntfs_write_iomap_begin_non_resident(struct inode *inode, loff_t offse
loff_t length, unsigned int flags,
struct iomap *iomap, int ntfs_iomap_flags)
{
- struct ntfs_inode *ni = NTFS_I(inode);
-
- if (ntfs_iomap_flags & (NTFS_IOMAP_FLAGS_BEGIN | NTFS_IOMAP_FLAGS_DIO) &&
- offset + length > ni->initialized_size) {
- int ret;
-
- ret = ntfs_extend_initialized_size(inode, offset,
- offset + length,
- ntfs_iomap_flags &
- NTFS_IOMAP_FLAGS_DIO);
- if (ret < 0)
- return ret;
- }
-
- mutex_lock(&ni->mrec_lock);
+ mutex_lock(&NTFS_I(inode)->mrec_lock);
if (ntfs_iomap_flags & NTFS_IOMAP_FLAGS_BEGIN)
return ntfs_write_simple_iomap_begin_non_resident(inode, offset,
length, iomap);
@@ -705,28 +691,10 @@ static int __ntfs_write_iomap_begin(struct inode *inode, loff_t offset,
struct iomap *iomap, int ntfs_iomap_flags)
{
struct ntfs_inode *ni = NTFS_I(inode);
- loff_t end = offset + length;
if (NVolShutdown(ni->vol))
return -EIO;
- if (ntfs_iomap_flags & (NTFS_IOMAP_FLAGS_BEGIN | NTFS_IOMAP_FLAGS_DIO) &&
- end > ni->data_size) {
- struct ntfs_volume *vol = ni->vol;
- int ret;
-
- mutex_lock(&ni->mrec_lock);
- if (end > ni->allocated_size &&
- end < ni->allocated_size + vol->preallocated_size)
- ret = ntfs_attr_expand(ni, end,
- ni->allocated_size + vol->preallocated_size);
- else
- ret = ntfs_attr_expand(ni, end, 0);
- mutex_unlock(&ni->mrec_lock);
- if (ret)
- return ret;
- }
-
if (!NInoNonResident(ni)) {
mutex_lock(&ni->mrec_lock);
return ntfs_write_iomap_begin_resident(inode, offset, iomap);
diff --git a/fs/ntfs/mft.c b/fs/ntfs/mft.c
index fd20d7abd6f5..f95e433885a0 100644
--- a/fs/ntfs/mft.c
+++ b/fs/ntfs/mft.c
@@ -2420,7 +2420,7 @@ mft_rec_already_initialized:
* record.
*/
- (*ni)->mrec = kmalloc(vol->mft_record_size, GFP_NOFS);
+ (*ni)->mrec = kmemdup(m, vol->mft_record_size, GFP_NOFS);
if (!(*ni)->mrec) {
folio_unlock(folio);
kunmap_local(m);
@@ -2429,7 +2429,6 @@ mft_rec_already_initialized:
goto undo_mftbmp_alloc;
}
- memcpy((*ni)->mrec, m, vol->mft_record_size);
post_read_mst_fixup((struct ntfs_record *)(*ni)->mrec, vol->mft_record_size);
ntfs_mft_mark_dirty(folio);
folio_unlock(folio);
diff --git a/fs/ntfs/namei.c b/fs/ntfs/namei.c
index 5ff25e9aaa32..0de5b63ef407 100644
--- a/fs/ntfs/namei.c
+++ b/fs/ntfs/namei.c
@@ -61,12 +61,12 @@ static int ntfs_check_bad_windows_name(struct ntfs_volume *vol,
const __le16 *wc,
unsigned int wc_len)
{
- if (ntfs_check_bad_char(wc, wc_len))
- return -EINVAL;
-
if (!NVolCheckWindowsNames(vol))
return 0;
+ if (ntfs_check_bad_char(wc, wc_len))
+ return -EINVAL;
+
/* Check for trailing space or dot. */
if (wc_len > 0 &&
(wc[wc_len - 1] == cpu_to_le16(' ') ||
@@ -685,7 +685,8 @@ static struct ntfs_inode *__ntfs_create(struct mnt_idmap *idmap, struct inode *d
mutex_unlock(&dir_ni->mrec_lock);
mutex_unlock(&ni->mrec_lock);
- ni->flags = fn->file_attributes;
+ ni->flags = fn->file_attributes |
+ (ni->flags & FILE_ATTRIBUTE_RECALL_ON_OPEN);
/* Set the sequence number. */
vi->i_generation = ni->seq_no;
set_nlink(vi, 1);
@@ -736,7 +737,7 @@ err_out:
}
static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct ntfs_volume *vol = NTFS_SB(dir->i_sb);
struct ntfs_inode *ni;
@@ -1082,7 +1083,7 @@ static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
if (!(vol->vol_flags & VOLUME_IS_DIRTY))
ntfs_set_volume_flags(vol, VOLUME_IS_DIRTY);
- ni = __ntfs_create(idmap, dir, uname, uname_len, S_IFDIR | mode, 0, NULL, 0);
+ ni = __ntfs_create(idmap, dir, uname, uname_len, mode, 0, NULL, 0);
kmem_cache_free(ntfs_name_cache, uname);
if (IS_ERR(ni)) {
err = PTR_ERR(ni);
diff --git a/fs/ntfs/reparse.c b/fs/ntfs/reparse.c
index fa523dc3691e..ced4c675d91f 100644
--- a/fs/ntfs/reparse.c
+++ b/fs/ntfs/reparse.c
@@ -317,8 +317,7 @@ unsigned int ntfs_make_symlink(struct ntfs_inode *ni)
} else
ni->flags &= ~FILE_ATTR_REPARSE_POINT;
- if (reparse_attr)
- kvfree(reparse_attr);
+ kvfree(reparse_attr);
return mode;
}
@@ -358,8 +357,7 @@ unsigned int ntfs_reparse_tag_dt_types(struct ntfs_volume *vol, unsigned long mr
}
}
- if (reparse_attr)
- kvfree(reparse_attr);
+ kvfree(reparse_attr);
iput(vi);
return dt_type;
diff --git a/fs/ntfs/runlist.c b/fs/ntfs/runlist.c
index cbb6576cf725..8e0fd400e7f7 100644
--- a/fs/ntfs/runlist.c
+++ b/fs/ntfs/runlist.c
@@ -71,29 +71,46 @@ static inline void ntfs_rl_mc(struct runlist_element *dstbase, int dst,
* On success, return a pointer to the newly allocated, or recycled, memory.
* On error, return -errno.
*/
-struct runlist_element *ntfs_rl_realloc(struct runlist_element *rl,
- int old_size, int new_size)
+static inline struct runlist_element *ntfs_rl_realloc_gfp(struct runlist_element *rl,
+ int old_size, int new_size, gfp_t gfp)
{
struct runlist_element *new_rl;
+ size_t new_bytes;
+
+ if (old_size < 0 || new_size < 0)
+ return ERR_PTR(-EINVAL);
- old_size = old_size * sizeof(*rl);
- new_size = new_size * sizeof(*rl);
if (old_size == new_size)
return rl;
- new_rl = kvzalloc(new_size, GFP_NOFS);
+ if (check_mul_overflow(new_size, sizeof(*rl), &new_bytes))
+ return ERR_PTR(-EINVAL);
+
+ new_rl = kvzalloc(new_bytes, gfp);
if (unlikely(!new_rl))
return ERR_PTR(-ENOMEM);
if (likely(rl != NULL)) {
- if (unlikely(old_size > new_size))
- old_size = new_size;
- memcpy(new_rl, rl, old_size);
+ size_t old_bytes;
+
+ if (check_mul_overflow(old_size, sizeof(*rl), &old_bytes)) {
+ kvfree(new_rl);
+ return ERR_PTR(-EINVAL);
+ }
+ if (unlikely(old_bytes > new_bytes))
+ old_bytes = new_bytes;
+ memcpy(new_rl, rl, old_bytes);
kvfree(rl);
}
return new_rl;
}
+struct runlist_element *ntfs_rl_realloc(struct runlist_element *rl,
+ int old_size, int new_size)
+{
+ return ntfs_rl_realloc_gfp(rl, old_size, new_size, GFP_NOFS);
+}
+
/*
* ntfs_rl_realloc_nofail - Reallocate memory for runlists
* @rl: original runlist
@@ -118,21 +135,8 @@ struct runlist_element *ntfs_rl_realloc(struct runlist_element *rl,
static inline struct runlist_element *ntfs_rl_realloc_nofail(struct runlist_element *rl,
int old_size, int new_size)
{
- struct runlist_element *new_rl;
-
- old_size = old_size * sizeof(*rl);
- new_size = new_size * sizeof(*rl);
- if (old_size == new_size)
- return rl;
-
- new_rl = kvmalloc(new_size, GFP_NOFS | __GFP_NOFAIL);
- if (likely(rl != NULL)) {
- if (unlikely(old_size > new_size))
- old_size = new_size;
- memcpy(new_rl, rl, old_size);
- kvfree(rl);
- }
- return new_rl;
+ return ntfs_rl_realloc_gfp(rl, old_size, new_size,
+ GFP_NOFS | __GFP_NOFAIL);
}
/*
diff --git a/fs/ntfs3/attrib.c b/fs/ntfs3/attrib.c
index c621a4c582f9..3628a737d7cc 100644
--- a/fs/ntfs3/attrib.c
+++ b/fs/ntfs3/attrib.c
@@ -1515,7 +1515,7 @@ int attr_wof_frame_info(struct ntfs_inode *ni, struct ATTRIB *attr,
u8 bytes_per_off;
char *addr;
struct folio *folio;
- int i, err;
+ int i, err = 0;
__le32 *off32;
__le64 *off64;
diff --git a/fs/ntfs3/dir.c b/fs/ntfs3/dir.c
index 873d52233003..6f8ad8428f49 100644
--- a/fs/ntfs3/dir.c
+++ b/fs/ntfs3/dir.c
@@ -25,6 +25,11 @@ int ntfs_utf16_to_nls(struct ntfs_sb_info *sbi, const __le16 *name, u32 len,
static_assert(sizeof(wchar_t) == sizeof(__le16));
+ if (buf_len <= 0)
+ return -EINVAL;
+
+ buf_len -= 1;
+
if (!nls) {
/* UTF-16 -> UTF-8 */
ret = utf16s_to_utf8s((wchar_t *)name, len, UTF16_LITTLE_ENDIAN,
@@ -273,6 +278,12 @@ out:
return err == -ENOENT ? NULL : err ? ERR_PTR(err) : inode;
}
+static inline bool de_fname_fits(const struct NTFS_DE *e, u32 e_size,
+ const struct ATTR_FILE_NAME *fname)
+{
+ return sizeof(struct NTFS_DE) + fname_full_size(fname) <= e_size;
+}
+
/*
* returns false if 'ctx' if full
*/
@@ -281,7 +292,7 @@ static inline bool ntfs_dir_emit(struct ntfs_sb_info *sbi,
u8 *name, struct dir_context *ctx)
{
const struct ATTR_FILE_NAME *fname;
- unsigned long ino;
+ u64 ino;
int name_len;
u32 dt_type;
@@ -305,15 +316,13 @@ static inline bool ntfs_dir_emit(struct ntfs_sb_info *sbi,
if (sbi->options->nohidden && (fname->dup.fa & FILE_ATTRIBUTE_HIDDEN))
return true;
- if (sizeof(struct NTFS_DE) +
- offsetof(struct ATTR_FILE_NAME, name) +
- fname->name_len * sizeof(short) > le16_to_cpu(e->size))
+ if (!de_fname_fits(e, le16_to_cpu(e->size), fname))
return true;
name_len = ntfs_utf16_to_nls(sbi, fname->name, fname->name_len, name,
PATH_MAX);
if (name_len <= 0) {
- ntfs_warn(sbi->sb, "failed to convert name for inode %lx.",
+ ntfs_warn(sbi->sb, "failed to convert name for inode %llx.",
ino);
return true;
}
@@ -576,6 +585,23 @@ out:
return err;
}
+/*
+ * Return fname when @e passes the same checks as ntfs_dir_emit() before
+ * exposing an entry (valid key, non-DOS, fname fits in e->size).
+ */
+static inline const struct ATTR_FILE_NAME *
+de_countable_fname(const struct NTFS_DE *e, u32 e_size)
+{
+ const struct ATTR_FILE_NAME *fname;
+
+ fname = de_get_fname(e);
+ if (!fname || fname->type == FILE_NAME_DOS ||
+ !de_fname_fits(e, e_size, fname))
+ return NULL;
+
+ return fname;
+}
+
static int ntfs_dir_count(struct inode *dir, bool *is_empty, size_t *dirs,
size_t *files)
{
@@ -615,13 +641,10 @@ static int ntfs_dir_count(struct inode *dir, bool *is_empty, size_t *dirs,
if (de_is_last(e))
break;
- fname = de_get_fname(e);
+ fname = de_countable_fname(e, e_size);
if (!fname)
continue;
- if (fname->type == FILE_NAME_DOS)
- continue;
-
if (is_empty) {
*is_empty = false;
if (!dirs && !files)
diff --git a/fs/ntfs3/file.c b/fs/ntfs3/file.c
index d601f088618c..fa8e2e56ff3f 100644
--- a/fs/ntfs3/file.c
+++ b/fs/ntfs3/file.c
@@ -894,7 +894,7 @@ static ssize_t ntfs_file_read_iter(struct kiocb *iocb, struct iov_iter *iter)
out:
inode_unlock_shared(inode);
- file_accessed(iocb->ki_filp);
+ file_accessed(file);
return err;
}
diff --git a/fs/ntfs3/frecord.c b/fs/ntfs3/frecord.c
index 2b49bc077558..3ad309d38fe8 100644
--- a/fs/ntfs3/frecord.c
+++ b/fs/ntfs3/frecord.c
@@ -169,7 +169,7 @@ out:
int ni_load_mi(struct ntfs_inode *ni, const struct ATTR_LIST_ENTRY *le,
struct mft_inode **mi)
{
- CLST rno;
+ u64 rno;
if (!le) {
*mi = &ni->mi;
@@ -768,10 +768,23 @@ int ni_create_attr_list(struct ntfs_inode *ni)
rs = sbi->record_size;
/*
- * Skip estimating exact memory requirement.
- * Looks like one record_size is always enough.
+ * Compute the exact size of the attribute list. Each attribute in the
+ * record yields one ATTR_LIST_ENTRY of le_size(name_len) bytes. The
+ * minimum on-disk attribute is SIZEOF_RESIDENT (0x18) bytes, but an
+ * unnamed one expands to le_size(0) (0x20) here, so a record crafted
+ * with many such attributes needs more than a single record_size; the
+ * previous fixed kzalloc(record_size) could therefore be overflowed by
+ * an attacker-controlled record.
*/
- le = kzalloc(al_aligned(rs), GFP_NOFS);
+ lsize = 0;
+ attr = NULL;
+ while ((attr = mi_enum_attr(ni, &ni->mi, attr)))
+ lsize += le_size(attr->name_len);
+
+ if (!lsize)
+ return -EINVAL;
+
+ le = kzalloc(al_aligned(lsize), GFP_NOFS);
if (!le)
return -ENOMEM;
@@ -781,7 +794,6 @@ int ni_create_attr_list(struct ntfs_inode *ni)
attr = NULL;
nb = 0;
free_b = 0;
- attr = NULL;
for (; (attr = mi_enum_attr(ni, &ni->mi, attr)); le = Add2Ptr(le, sz)) {
sz = le_size(attr->name_len);
@@ -2919,7 +2931,6 @@ loff_t ni_seek_data_or_hole(struct ntfs_inode *ni, loff_t offset, bool data)
break;
}
}
-
}
vbo = (u64)vcn << cluster_bits;
@@ -2961,8 +2972,8 @@ int ni_write_parents(struct ntfs_inode *ni, int sync)
if (IS_ERR(dir)) {
ntfs_inode_warn(
&ni->vfs_inode,
- "failed to open parent directory r=%lx to write",
- (long)ino_get(&fname->home));
+ "failed to open parent directory r=%llx to write",
+ (u64)ino_get(&fname->home));
continue;
}
@@ -3081,8 +3092,8 @@ static bool ni_update_parent(struct ntfs_inode *ni, struct NTFS_DUP_INFO *dup,
if (IS_ERR(dir)) {
ntfs_inode_warn(
&ni->vfs_inode,
- "failed to open parent directory r=%lx to update",
- (long)ino_get(&fname->home));
+ "failed to open parent directory r=%llx to update",
+ (u64)ino_get(&fname->home));
continue;
}
diff --git a/fs/ntfs3/fslog.c b/fs/ntfs3/fslog.c
index f038c799e7ac..cc24c23e4db9 100644
--- a/fs/ntfs3/fslog.c
+++ b/fs/ntfs3/fslog.c
@@ -2613,7 +2613,6 @@ bool check_index_header(const struct INDEX_HDR *hdr, size_t bytes)
const bool has_subnode = hdr_has_subnode(hdr);
__le16 mask;
u32 min_de, de_off, used, total;
- const struct NTFS_DE *e;
if (has_subnode) {
min_de = sizeof(struct NTFS_DE) + sizeof(u64);
@@ -2632,8 +2631,8 @@ bool check_index_header(const struct INDEX_HDR *hdr, size_t bytes)
return false;
}
- e = (const struct NTFS_DE *)((const u8 *)hdr + de_off);
for (;;) {
+ const struct NTFS_DE *e = Add2Ptr(hdr, de_off);
u16 esize = le16_to_cpu(e->size);
u16 key_size = le16_to_cpu(e->key_size);
u16 data_size;
@@ -2649,7 +2648,6 @@ bool check_index_header(const struct INDEX_HDR *hdr, size_t bytes)
if (de_is_last(e)) {
if (key_size)
return false;
-
break;
}
@@ -2658,7 +2656,6 @@ bool check_index_header(const struct INDEX_HDR *hdr, size_t bytes)
return false;
de_off += esize;
- e = (const struct NTFS_DE *)((const u8 *)hdr + de_off);
}
return true;
@@ -3544,8 +3541,7 @@ move_data:
* bound here so the memmove cannot reach past the entry.
*/
if (le16_to_cpu(e->view.data_off) > le16_to_cpu(e->size) ||
- le16_to_cpu(e->view.data_off) + dlen >
- le16_to_cpu(e->size))
+ le16_to_cpu(e->view.data_off) + dlen > le16_to_cpu(e->size))
goto dirty_vol;
memmove(Add2Ptr(e, le16_to_cpu(e->view.data_off)), data, dlen);
@@ -3756,8 +3752,7 @@ move_data:
/* See UpdateRecordDataRoot for the rationale. */
if (le16_to_cpu(e->view.data_off) > le16_to_cpu(e->size) ||
- le16_to_cpu(e->view.data_off) + dlen >
- le16_to_cpu(e->size))
+ le16_to_cpu(e->view.data_off) + dlen > le16_to_cpu(e->size))
goto dirty_vol;
memmove(Add2Ptr(e, le16_to_cpu(e->view.data_off)), data, dlen);
@@ -4653,11 +4648,11 @@ copy_lcns:
}
/*
- * find_dp() only validates that target_vcn is the first
- * cluster covered by dp. The walk through lrh->lcns_follow
- * further entries must stay within the allocated
- * dp->page_lcns[] array, which is sized by dp->lcns_follow.
- */
+ * find_dp() only validates that target_vcn is the first
+ * cluster covered by dp. The walk through lrh->lcns_follow
+ * further entries must stay within the allocated
+ * dp->page_lcns[] array, which is sized by dp->lcns_follow.
+ */
if (le64_to_cpu(lrh->target_vcn) - le64_to_cpu(dp->vcn) + t16 >
le32_to_cpu(dp->lcns_follow)) {
err = -EINVAL;
diff --git a/fs/ntfs3/fsntfs.c b/fs/ntfs3/fsntfs.c
index bc7469d0a34d..7c4db816c43d 100644
--- a/fs/ntfs3/fsntfs.c
+++ b/fs/ntfs3/fsntfs.c
@@ -2302,8 +2302,8 @@ int ntfs_reparse_init(struct ntfs_sb_info *sbi)
goto out;
}
- root_r = resident_data(attr);
- if (root_r->type != ATTR_ZERO ||
+ root_r = resident_data_ex(attr, sizeof(struct INDEX_ROOT));
+ if (!root_r || root_r->type != ATTR_ZERO ||
root_r->rule != NTFS_COLLATION_TYPE_UINTS) {
err = -EINVAL;
goto out;
@@ -2340,8 +2340,8 @@ int ntfs_objid_init(struct ntfs_sb_info *sbi)
goto out;
}
- root = resident_data(attr);
- if (root->type != ATTR_ZERO ||
+ root = resident_data_ex(attr, sizeof(struct INDEX_ROOT));
+ if (!root || root->type != ATTR_ZERO ||
root->rule != NTFS_COLLATION_TYPE_UINTS) {
err = -EINVAL;
goto out;
diff --git a/fs/ntfs3/index.c b/fs/ntfs3/index.c
index 2b439ac04356..4afacfc6ff89 100644
--- a/fs/ntfs3/index.c
+++ b/fs/ntfs3/index.c
@@ -612,8 +612,8 @@ static const struct NTFS_DE *hdr_insert_head(struct INDEX_HDR *hdr,
static bool index_hdr_check(const struct INDEX_HDR *hdr, u32 bytes)
{
const bool has_subnode = hdr_has_subnode(hdr);
- const u16 min_size = sizeof(struct NTFS_DE) +
- (has_subnode ? sizeof(u64) : 0);
+ const u16 min_size =
+ sizeof(struct NTFS_DE) + (has_subnode ? sizeof(u64) : 0);
u32 end = le32_to_cpu(hdr->used);
u32 tot = le32_to_cpu(hdr->total);
u32 off = le32_to_cpu(hdr->de_off);
@@ -2131,8 +2131,7 @@ static struct indx_node *indx_find_buffer(struct ntfs_index *indx,
if (err)
return ERR_PTR(err);
- r = indx_find_buffer(indx, ni, root, vbn, n,
- depth + 1);
+ r = indx_find_buffer(indx, ni, root, vbn, n, depth + 1);
if (r)
return r;
}
diff --git a/fs/ntfs3/inode.c b/fs/ntfs3/inode.c
index 0c9bd669117d..4ecd7ebaa655 100644
--- a/fs/ntfs3/inode.c
+++ b/fs/ntfs3/inode.c
@@ -36,7 +36,7 @@ static struct inode *ntfs_read_mft(struct inode *inode,
bool is_match = false;
bool is_root = false;
bool is_dir;
- unsigned long ino = inode->i_ino;
+ u64 ino = inode->i_ino;
u32 rp_fa = 0, asize, t32;
u16 roff, rsize, names = 0, links = 0;
const struct ATTR_FILE_NAME *fname = NULL;
@@ -79,7 +79,7 @@ static struct inode *ntfs_read_mft(struct inode *inode,
;
} else if (ref->seq != rec->seq) {
err = -EINVAL;
- ntfs_err(sb, "MFT: r=%lx, expect seq=%x instead of %x!", ino,
+ ntfs_err(sb, "MFT: r=%llx, expect seq=%x instead of %x!", ino,
le16_to_cpu(ref->seq), le16_to_cpu(rec->seq));
goto out;
} else if (!is_rec_inuse(rec)) {
@@ -127,10 +127,16 @@ next_attr:
if (le && le->vcn) {
/* This is non primary attribute segment. Ignore if not MFT. */
- if (ino != MFT_REC_MFT || attr->type != ATTR_DATA)
+ if (ino != MFT_REC_MFT)
+ goto next_attr;
+
+ if (attr->type == ATTR_DATA)
+ run = &ni->file.run;
+ else if (attr->type == ATTR_BITMAP)
+ run = &sbi->mft.bitmap.run;
+ else
goto next_attr;
- run = &ni->file.run;
asize = le32_to_cpu(attr->size);
goto attr_unpack_run;
}
@@ -606,15 +612,17 @@ static void ntfs_iomap_read_end_io(struct bio *bio)
}
static void ntfs_iomap_bio_submit_read(const struct iomap_iter *iter,
- struct iomap_read_folio_ctx *ctx)
+ struct iomap_read_folio_ctx *ctx)
{
iomap_bio_submit_read_endio(iter, ctx, ntfs_iomap_read_end_io);
}
+// clang-format off
static const struct iomap_read_ops ntfs_iomap_bio_read_ops = {
.read_folio_range = iomap_bio_read_folio_range,
.submit_read = ntfs_iomap_bio_submit_read,
};
+// clang-format on
static int ntfs_read_folio(struct file *file, struct folio *folio)
{
diff --git a/fs/ntfs3/lznt.c b/fs/ntfs3/lznt.c
index f818d9785004..5dcb7674790c 100644
--- a/fs/ntfs3/lznt.c
+++ b/fs/ntfs3/lznt.c
@@ -240,8 +240,10 @@ static inline ssize_t decompress_chunk(u8 *unc, u8 *unc_end, const u8 *cmpr,
if (up - unc > LZNT_CHUNK_SIZE)
return -EINVAL;
/* Correct index */
- while (index < ARRAY_SIZE(s_max_off) - 1 && unc + s_max_off[index] < up)
+ while (index < ARRAY_SIZE(s_max_off) - 1 &&
+ unc + s_max_off[index] < up) {
index += 1;
+ }
/* Check the current flag for zero. */
if (!(ch & (1 << bit))) {
diff --git a/fs/ntfs3/namei.c b/fs/ntfs3/namei.c
index c59de5f2fa97..7bf889a4ca35 100644
--- a/fs/ntfs3/namei.c
+++ b/fs/ntfs3/namei.c
@@ -22,7 +22,7 @@ int fill_name_de(struct ntfs_sb_info *sbi, void *buf, const struct qstr *name,
{
int err;
struct NTFS_DE *e = buf;
- u16 data_size;
+ u16 data_size, real_size, aligned_size;
struct ATTR_FILE_NAME *fname = (struct ATTR_FILE_NAME *)(e + 1);
#ifndef CONFIG_NTFS3_64BIT_CLUSTER
@@ -53,7 +53,12 @@ int fill_name_de(struct ntfs_sb_info *sbi, void *buf, const struct qstr *name,
fname->type = FILE_NAME_POSIX;
data_size = fname_full_size(fname);
- e->size = cpu_to_le16(ALIGN(data_size, 8) + sizeof(struct NTFS_DE));
+ real_size = data_size + sizeof(struct NTFS_DE);
+ aligned_size = ALIGN(data_size, 8) + sizeof(struct NTFS_DE);
+ if (aligned_size > real_size)
+ memset((char *)buf + real_size, 0, aligned_size - real_size);
+
+ e->size = cpu_to_le16(aligned_size);
e->key_size = cpu_to_le16(data_size);
e->flags = 0;
e->res = 0;
@@ -105,7 +110,7 @@ static struct dentry *ntfs_lookup(struct inode *dir, struct dentry *dentry,
* ntfs_create - inode_operations::create
*/
static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return ntfs_create_inode(idmap, dir, dentry, NULL, S_IFREG | mode, 0,
NULL, 0, NULL);
@@ -213,7 +218,7 @@ static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode)
{
return ERR_PTR(ntfs_create_inode(idmap, dir, dentry, NULL,
- S_IFDIR | mode, 0, NULL, 0, NULL));
+ mode, 0, NULL, 0, NULL));
}
/*
diff --git a/fs/ntfs3/ntfs_fs.h b/fs/ntfs3/ntfs_fs.h
index d98d7e474476..6bd1a439e92f 100644
--- a/fs/ntfs3/ntfs_fs.h
+++ b/fs/ntfs3/ntfs_fs.h
@@ -401,7 +401,7 @@ struct ntfs_inode {
struct rw_semaphore run_lock;
/* Unpacked runs from just one record. */
struct runs_tree run;
- /*
+ /*
* Pairs [vcn, len] for all delay allocated clusters.
* Normal file always contains delayed clusters in one fragment.
* TODO: use 2 CLST per pair instead of 3.
@@ -886,8 +886,8 @@ int run_unpack_ex(struct runs_tree *run, struct ntfs_sb_info *sbi, CLST ino,
#else
#define run_unpack_ex run_unpack
#endif
-int run_get_highest_vcn(CLST vcn, const u8 *run_buf, size_t run_buf_size,
- u64 *highest_vcn);
+int run_get_highest_vcn(CLST vcn, const u8 *run_buf, size_t run_buf_size,
+ u64 *highest_vcn);
int run_clone(const struct runs_tree *run, struct runs_tree *new_run);
bool run_remove_range(struct runs_tree *run, CLST vcn, CLST len, CLST *done);
CLST run_len(const struct runs_tree *run);
diff --git a/fs/ntfs3/run.c b/fs/ntfs3/run.c
index 3ebf0154eda3..6e3ef89fc666 100644
--- a/fs/ntfs3/run.c
+++ b/fs/ntfs3/run.c
@@ -1265,8 +1265,8 @@ int run_unpack_ex(struct runs_tree *run, struct ntfs_sb_info *sbi, CLST ino,
* Return the highest vcn from a mapping pairs array
* it used while replaying log file.
*/
-int run_get_highest_vcn(CLST vcn, const u8 *run_buf, size_t run_buf_size,
- u64 *highest_vcn)
+int run_get_highest_vcn(CLST vcn, const u8 *run_buf, size_t run_buf_size,
+ u64 *highest_vcn)
{
const u8 *run_last = run_buf + run_buf_size;
u64 vcn64 = vcn;
@@ -1279,7 +1279,7 @@ int run_get_highest_vcn(CLST vcn, const u8 *run_buf, size_t run_buf_size,
if (size_size > 8 || offset_size > 8)
return -EINVAL;
- if (run_buf + size_size + offset_size > run_last)
+ if (run_buf + size_size + offset_size > run_last)
return -EINVAL;
len = run_unpack_s64(run_buf, size_size, 0);
@@ -1357,7 +1357,8 @@ bool run_remove_range(struct runs_tree *run, CLST vcn, CLST len, CLST *done)
if (r_end > end) {
/* Remove a middle part, split. */
CLST tail_lcn = r->lcn == SPARSE_LCN ?
- SPARSE_LCN : (r->lcn + (end - r->vcn));
+ SPARSE_LCN :
+ (r->lcn + (end - r->vcn));
*done += len;
r->len = d;
diff --git a/fs/ntfs3/super.c b/fs/ntfs3/super.c
index 3305fe406cb2..94cb57007893 100644
--- a/fs/ntfs3/super.c
+++ b/fs/ntfs3/super.c
@@ -1189,7 +1189,7 @@ read_boot:
#ifdef CONFIG_NTFS3_64BIT_CLUSTER
if (clusters >= (1ull << (64 - cluster_bits)))
sbi->maxbytes = -1;
- sbi->maxbytes_sparse = -1;
+ sbi->maxbytes_sparse = MAX_LFS_FILESIZE;
sb->s_maxbytes = MAX_LFS_FILESIZE;
#else
/* Maximum size for sparse file. */
@@ -1458,7 +1458,10 @@ static int ntfs_fill_super(struct super_block *sb, struct fs_context *fc)
Add2Ptr(a, roff),
le32_to_cpu(a->size) - roff);
if (err < 0) {
- ntfs_err(sb, "Failed to unpack $MFT bitmap extent (%d).", err);
+ ntfs_err(
+ sb,
+ "Failed to unpack $MFT bitmap extent (%d).",
+ err);
goto put_inode_out;
}
err = 0;
@@ -1866,8 +1869,7 @@ static int ntfs_init_fs_context(struct fs_context *fc)
/* Default options. */
opts->fs_uid = current_uid();
opts->fs_gid = current_gid();
- opts->fs_fmask_inv = ~current_umask();
- opts->fs_dmask_inv = ~current_umask();
+ opts->fs_fmask_inv = opts->fs_dmask_inv = ~current_umask();
opts->prealloc = 1;
#ifdef CONFIG_NTFS3_FS_POSIX_ACL
@@ -1928,7 +1930,6 @@ static struct file_system_type ntfs_fs_type = {
.kill_sb = ntfs3_kill_sb,
.fs_flags = FS_REQUIRES_DEV | FS_ALLOW_IDMAP,
};
-
// clang-format on
static int __init init_ntfs_fs(void)
diff --git a/fs/ntfs3/xattr.c b/fs/ntfs3/xattr.c
index 04814dd29375..7a81369a1173 100644
--- a/fs/ntfs3/xattr.c
+++ b/fs/ntfs3/xattr.c
@@ -660,7 +660,6 @@ static noinline int ntfs_set_acl_ex(struct mnt_idmap *idmap,
inode->i_mode = old_mode;
goto out;
}
- inode->i_mode = mode;
}
set_cached_acl(inode, type, acl);
inode_set_ctime_current(inode);
diff --git a/fs/nullfs.c b/fs/nullfs.c
index fdbd3e5d3d71..e06352c7b2cc 100644
--- a/fs/nullfs.c
+++ b/fs/nullfs.c
@@ -4,6 +4,8 @@
#include <linux/fs_context.h>
#include <linux/magic.h>
+#include "mount.h"
+
static const struct super_operations nullfs_super_operations = {
.statfs = simple_statfs,
};
@@ -40,14 +42,9 @@ static int nullfs_fs_fill_super(struct super_block *s, struct fs_context *fc)
return 0;
}
-/*
- * For now this is a single global instance. If needed we can make it
- * mountable by userspace at which point we will need to make it
- * multi-instance.
- */
static int nullfs_fs_get_tree(struct fs_context *fc)
{
- return get_tree_single(fc, nullfs_fs_fill_super);
+ return get_tree_nodev(fc, nullfs_fs_fill_super);
}
static const struct fs_context_operations nullfs_fs_context_ops = {
@@ -57,9 +54,8 @@ static const struct fs_context_operations nullfs_fs_context_ops = {
static int nullfs_init_fs_context(struct fs_context *fc)
{
fc->ops = &nullfs_fs_context_ops;
- fc->global = true;
- fc->sb_flags = SB_NOUSER;
- fc->s_iflags = SB_I_NOEXEC | SB_I_NODEV;
+ fc->sb_flags |= SB_NOUSER;
+ fc->s_iflags |= SB_I_NOEXEC | SB_I_NODEV;
return 0;
}
diff --git a/fs/ocfs2/dlmfs/dlmfs.c b/fs/ocfs2/dlmfs/dlmfs.c
index 5821e33df78f..53df5dd10ad0 100644
--- a/fs/ocfs2/dlmfs/dlmfs.c
+++ b/fs/ocfs2/dlmfs/dlmfs.c
@@ -422,7 +422,7 @@ static struct dentry *dlmfs_mkdir(struct mnt_idmap * idmap,
goto bail;
}
- inode = dlmfs_get_inode(dir, dentry, mode | S_IFDIR);
+ inode = dlmfs_get_inode(dir, dentry, mode);
if (!inode) {
status = -ENOMEM;
mlog_errno(status);
@@ -453,8 +453,7 @@ bail:
static int dlmfs_create(struct mnt_idmap *idmap,
struct inode *dir,
struct dentry *dentry,
- umode_t mode,
- bool excl)
+ umode_t mode)
{
int status = 0;
struct inode *inode;
diff --git a/fs/ocfs2/namei.c b/fs/ocfs2/namei.c
index 8368a7f3d4a2..10ce03e4557e 100644
--- a/fs/ocfs2/namei.c
+++ b/fs/ocfs2/namei.c
@@ -657,7 +657,7 @@ static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap,
trace_ocfs2_mkdir(dir, dentry, dentry->d_name.len, dentry->d_name.name,
OCFS2_I(dir)->ip_blkno, mode);
- ret = ocfs2_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFDIR, 0);
+ ret = ocfs2_mknod(&nop_mnt_idmap, dir, dentry, mode, 0);
if (ret)
mlog_errno(ret);
@@ -667,8 +667,7 @@ static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap,
static int ocfs2_create(struct mnt_idmap *idmap,
struct inode *dir,
struct dentry *dentry,
- umode_t mode,
- bool excl)
+ umode_t mode)
{
int ret;
diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c
index bcac614ff02a..c62e389d4dd6 100644
--- a/fs/ocfs2/super.c
+++ b/fs/ocfs2/super.c
@@ -1882,7 +1882,6 @@ static void ocfs2_dismount_volume(struct super_block *sb, int mnt_err)
ocfs2_delete_osb(osb);
kfree(osb);
- sb->s_dev = 0;
sb->s_fs_info = NULL;
}
diff --git a/fs/omfs/dir.c b/fs/omfs/dir.c
index 2ed541fccf33..692297cf84e7 100644
--- a/fs/omfs/dir.c
+++ b/fs/omfs/dir.c
@@ -282,11 +282,11 @@ out_free_inode:
static struct dentry *omfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode)
{
- return ERR_PTR(omfs_add_node(dir, dentry, mode | S_IFDIR));
+ return ERR_PTR(omfs_add_node(dir, dentry, mode));
}
static int omfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return omfs_add_node(dir, dentry, mode | S_IFREG);
}
diff --git a/fs/orangefs/namei.c b/fs/orangefs/namei.c
index 75e65e72c2d6..8ebc34e112d5 100644
--- a/fs/orangefs/namei.c
+++ b/fs/orangefs/namei.c
@@ -18,8 +18,7 @@
static int orangefs_create(struct mnt_idmap *idmap,
struct inode *dir,
struct dentry *dentry,
- umode_t mode,
- bool exclusive)
+ umode_t mode)
{
struct orangefs_inode_s *parent = ORANGEFS_I(dir);
struct orangefs_kernel_op_s *new_op;
@@ -333,7 +332,7 @@ static struct dentry *orangefs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
ref = new_op->downcall.resp.mkdir.refn;
- inode = orangefs_new_inode(dir->i_sb, dir, S_IFDIR | mode, 0, &ref);
+ inode = orangefs_new_inode(dir->i_sb, dir, mode, 0, &ref);
if (IS_ERR(inode)) {
gossip_err("*** Failed to allocate orangefs dir inode\n");
ret = PTR_ERR(inode);
diff --git a/fs/overlayfs/dir.c b/fs/overlayfs/dir.c
index a033743dbf51..610c0116252c 100644
--- a/fs/overlayfs/dir.c
+++ b/fs/overlayfs/dir.c
@@ -689,8 +689,8 @@ static int ovl_create_or_link(struct dentry *dentry, struct inode *inode,
return err;
}
-static int ovl_create_object(struct dentry *dentry, int mode, dev_t rdev,
- const char *link)
+static int ovl_create_object(struct mnt_idmap *idmap, struct dentry *dentry,
+ int mode, dev_t rdev, const char *link)
{
int err;
struct inode *inode;
@@ -717,7 +717,7 @@ static int ovl_create_object(struct dentry *dentry, int mode, dev_t rdev,
inode_state_set(inode, I_CREATING);
spin_unlock(&inode->i_lock);
- inode_init_owner(&nop_mnt_idmap, inode, dentry->d_parent->d_inode, mode);
+ inode_init_owner(idmap, inode, dentry->d_parent->d_inode, mode);
attr.mode = inode->i_mode;
err = ovl_create_or_link(dentry, inode, &attr, false);
@@ -732,15 +732,15 @@ out:
}
static int ovl_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
- return ovl_create_object(dentry, (mode & 07777) | S_IFREG, 0, NULL);
+ return ovl_create_object(idmap, dentry, (mode & 07777) | S_IFREG, 0, NULL);
}
static struct dentry *ovl_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode)
{
- return ERR_PTR(ovl_create_object(dentry, (mode & 07777) | S_IFDIR, 0, NULL));
+ return ERR_PTR(ovl_create_object(idmap, dentry, (mode & 07777) | S_IFDIR, 0, NULL));
}
static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir,
@@ -750,13 +750,13 @@ static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir,
if (S_ISCHR(mode) && rdev == WHITEOUT_DEV)
return -EPERM;
- return ovl_create_object(dentry, mode, rdev, NULL);
+ return ovl_create_object(idmap, dentry, mode, rdev, NULL);
}
static int ovl_symlink(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, const char *link)
{
- return ovl_create_object(dentry, S_IFLNK, 0, link);
+ return ovl_create_object(idmap, dentry, S_IFLNK, 0, link);
}
static int ovl_set_link_redirect(struct dentry *dentry)
@@ -1444,7 +1444,7 @@ static int ovl_tmpfile(struct mnt_idmap *idmap, struct inode *dir,
if (!inode)
goto drop_write;
- inode_init_owner(&nop_mnt_idmap, inode, dir, mode);
+ inode_init_owner(idmap, inode, dir, mode);
err = ovl_create_tmpfile(file, dentry, inode, inode->i_mode);
if (err)
goto put_inode;
diff --git a/fs/overlayfs/inode.c b/fs/overlayfs/inode.c
index bc71231cad53..401cb8c75520 100644
--- a/fs/overlayfs/inode.c
+++ b/fs/overlayfs/inode.c
@@ -26,10 +26,18 @@ int ovl_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
bool full_copy_up = false;
struct dentry *upperdentry;
- err = setattr_prepare(&nop_mnt_idmap, dentry, attr);
+ err = setattr_prepare(idmap, dentry, attr);
if (err)
return err;
+ /* Rebase ownership from the mount idmap into overlay id space. */
+ if (attr->ia_valid & ATTR_UID)
+ attr->ia_vfsuid = VFSUIDT_INIT(from_vfsuid(idmap,
+ i_user_ns(d_inode(dentry)), attr->ia_vfsuid));
+ if (attr->ia_valid & ATTR_GID)
+ attr->ia_vfsgid = VFSGIDT_INIT(from_vfsgid(idmap,
+ i_user_ns(d_inode(dentry)), attr->ia_vfsgid));
+
if (attr->ia_valid & ATTR_SIZE) {
/* Truncate should trigger data copy up as well */
full_copy_up = true;
@@ -172,6 +180,8 @@ int ovl_getattr(struct mnt_idmap *idmap, const struct path *path,
int fsid = 0;
int err;
bool metacopy_blocks = false;
+ vfsuid_t vfsuid;
+ vfsgid_t vfsgid;
metacopy_blocks = ovl_is_metacopy_dentry(dentry);
@@ -284,6 +294,12 @@ int ovl_getattr(struct mnt_idmap *idmap, const struct path *path,
if (!is_dir && ovl_test_flag(OVL_INDEX, d_inode(dentry)))
stat->nlink = dentry->d_inode->i_nlink;
+ /* Map ownership of the real inode through the overlay mount idmap. */
+ vfsuid = make_vfsuid(idmap, i_user_ns(inode), stat->uid);
+ vfsgid = make_vfsgid(idmap, i_user_ns(inode), stat->gid);
+ stat->uid = vfsuid_into_kuid(vfsuid);
+ stat->gid = vfsgid_into_kgid(vfsgid);
+
return err;
}
@@ -306,7 +322,7 @@ int ovl_permission(struct mnt_idmap *idmap,
* Check overlay inode with the creds of task and underlying inode
* with creds of mounter
*/
- err = generic_permission(&nop_mnt_idmap, inode, mask);
+ err = generic_permission(idmap, inode, mask);
if (err)
return err;
@@ -534,7 +550,7 @@ int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry,
return -EOPNOTSUPP;
if (type == ACL_TYPE_DEFAULT && !S_ISDIR(inode->i_mode))
return acl ? -EACCES : 0;
- if (!inode_owner_or_capable(&nop_mnt_idmap, inode))
+ if (!inode_owner_or_capable(idmap, inode))
return -EPERM;
/*
@@ -542,8 +558,8 @@ int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry,
* be done with mounter's capabilities and so that won't do it for us).
*/
if (unlikely(inode->i_mode & S_ISGID) && type == ACL_TYPE_ACCESS &&
- !in_group_p(inode->i_gid) &&
- !capable_wrt_inode_uidgid(&nop_mnt_idmap, inode, CAP_FSETID)) {
+ !in_group_or_capable(idmap, inode,
+ i_gid_into_vfsgid(idmap, inode))) {
struct iattr iattr = { .ia_valid = ATTR_KILL_SGID };
err = ovl_setattr(&nop_mnt_idmap, dentry, &iattr);
diff --git a/fs/overlayfs/overlayfs.h b/fs/overlayfs/overlayfs.h
index b75df37f70ac..e0d8c6152e9f 100644
--- a/fs/overlayfs/overlayfs.h
+++ b/fs/overlayfs/overlayfs.h
@@ -320,6 +320,7 @@ static inline int ovl_do_setxattr(struct ovl_fs *ofs, struct dentry *dentry,
const char *name, const void *value,
size_t size, int flags)
{
+ /* Use vfs_setxattr(), not __vfs_setxattr(): it idmaps the security.capability rootid. */
int err = vfs_setxattr(ovl_upper_mnt_idmap(ofs), dentry, name,
value, size, flags);
diff --git a/fs/overlayfs/super.c b/fs/overlayfs/super.c
index 60f0b7ceef0a..f2889cf9bc07 100644
--- a/fs/overlayfs/super.c
+++ b/fs/overlayfs/super.c
@@ -1573,7 +1573,7 @@ struct file_system_type ovl_fs_type = {
.name = "overlay",
.init_fs_context = ovl_init_fs_context,
.parameters = ovl_parameter_spec,
- .fs_flags = FS_USERNS_MOUNT,
+ .fs_flags = FS_USERNS_MOUNT | FS_ALLOW_IDMAP,
.kill_sb = kill_anon_super,
};
MODULE_ALIAS_FS("overlay");
diff --git a/fs/overlayfs/xattrs.c b/fs/overlayfs/xattrs.c
index 859e80ae6f40..5ae44b9c8790 100644
--- a/fs/overlayfs/xattrs.c
+++ b/fs/overlayfs/xattrs.c
@@ -84,6 +84,7 @@ static int ovl_xattr_get(struct dentry *dentry, struct inode *inode, const char
struct path realpath;
ovl_i_path_real(inode, &realpath);
+ /* Use vfs_getxattr(), not __vfs_getxattr(): it idmaps the security.capability rootid. */
with_ovl_creds(dentry->d_sb)
return vfs_getxattr(mnt_idmap(realpath.mnt), realpath.dentry, name, value, size);
}
diff --git a/fs/posix_acl.c b/fs/posix_acl.c
index 3dc62c1c2708..18b302f94174 100644
--- a/fs/posix_acl.c
+++ b/fs/posix_acl.c
@@ -747,8 +747,6 @@ static int posix_acl_fix_xattr_common(const void *value, size_t size)
count = posix_acl_xattr_count(size);
if (count < 0)
return -EINVAL;
- if (count == 0)
- return 0;
return count;
}
diff --git a/fs/proc/array.c b/fs/proc/array.c
index 479ea8cb4ef4..f6f75d206762 100644
--- a/fs/proc/array.c
+++ b/fs/proc/array.c
@@ -168,8 +168,8 @@ static inline void task_state(struct seq_file *m, struct pid_namespace *ns,
cred = get_task_cred(p);
task_lock(p);
- if (p->fs)
- umask = p->fs->umask;
+ if (p->real_fs)
+ umask = p->real_fs->umask;
if (p->files)
max_fds = files_fdtable(p->files)->max_fds;
task_unlock(p);
diff --git a/fs/proc/base.c b/fs/proc/base.c
index 780f81259052..6a39de424f62 100644
--- a/fs/proc/base.c
+++ b/fs/proc/base.c
@@ -211,8 +211,8 @@ static int get_task_root(struct task_struct *task, struct path *root)
int result = -ENOENT;
task_lock(task);
- if (task->fs) {
- get_fs_root(task->fs, root);
+ if (task->real_fs) {
+ get_fs_root(task->real_fs, root);
result = 0;
}
task_unlock(task);
@@ -225,8 +225,8 @@ static int proc_cwd_link(struct dentry *dentry, struct path *path,
int result = -ENOENT;
task_lock(task);
- if (task->fs) {
- get_fs_pwd(task->fs, path);
+ if (task->real_fs) {
+ get_fs_pwd(task->real_fs, path);
result = 0;
}
task_unlock(task);
diff --git a/fs/proc_namespace.c b/fs/proc_namespace.c
index 5c555db68aa2..036356c0a55b 100644
--- a/fs/proc_namespace.c
+++ b/fs/proc_namespace.c
@@ -254,13 +254,13 @@ static int mounts_open_common(struct inode *inode, struct file *file,
}
ns = nsp->mnt_ns;
get_mnt_ns(ns);
- if (!task->fs) {
+ if (!task->real_fs) {
task_unlock(task);
put_task_struct(task);
ret = -ENOENT;
goto err_put_ns;
}
- get_fs_root(task->fs, &root);
+ get_fs_root(task->real_fs, &root);
task_unlock(task);
put_task_struct(task);
diff --git a/fs/ramfs/inode.c b/fs/ramfs/inode.c
index 3987639ed132..0a88ede48e0a 100644
--- a/fs/ramfs/inode.c
+++ b/fs/ramfs/inode.c
@@ -121,14 +121,14 @@ out:
static struct dentry *ramfs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct dentry *dentry, umode_t mode)
{
- int retval = ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFDIR, 0);
+ int retval = ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode, 0);
if (!retval)
inc_nlink(dir);
return ERR_PTR(retval);
}
static int ramfs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFREG, 0);
}
diff --git a/fs/romfs/super.c b/fs/romfs/super.c
index ac55193bf398..7f783d6173e8 100644
--- a/fs/romfs/super.c
+++ b/fs/romfs/super.c
@@ -240,6 +240,8 @@ static struct dentry *romfs_lookup(struct inode *dir, struct dentry *dentry,
if ((be32_to_cpu(ri.next) & ROMFH_TYPE) == ROMFH_HRD)
offset = be32_to_cpu(ri.spec) & ROMFH_MASK;
inode = romfs_iget(dir->i_sb, offset);
+ if (IS_ERR(inode))
+ return ERR_CAST(inode);
break;
}
@@ -262,6 +264,8 @@ static const struct inode_operations romfs_dir_inode_operations = {
.lookup = romfs_lookup,
};
+#define ROMFS_MAX_HARDLINK_DEPTH 64
+
/*
* get a romfs inode based on its position in the image (which doubles as the
* inode number)
@@ -273,6 +277,7 @@ static struct inode *romfs_iget(struct super_block *sb, unsigned long pos)
struct inode *i;
unsigned long nlen;
unsigned nextfh;
+ unsigned int depth = 0;
int ret;
umode_t mode;
@@ -289,6 +294,9 @@ static struct inode *romfs_iget(struct super_block *sb, unsigned long pos)
if ((nextfh & ROMFH_TYPE) != ROMFH_HRD)
break;
+ if (++depth > ROMFS_MAX_HARDLINK_DEPTH)
+ return ERR_PTR(-ELOOP);
+
pos = be32_to_cpu(ri.spec) & ROMFH_MASK;
}
@@ -587,7 +595,7 @@ static void romfs_kill_sb(struct super_block *sb)
#ifdef CONFIG_ROMFS_ON_BLOCK
if (sb->s_bdev) {
sync_blockdev(sb->s_bdev);
- bdev_fput(sb->s_bdev_file);
+ fs_bdev_file_release(sb->s_bdev_file, sb);
}
#endif
}
diff --git a/fs/smb/client/cifs_swn.c b/fs/smb/client/cifs_swn.c
index 9951817d0d7f..fe10719e627e 100644
--- a/fs/smb/client/cifs_swn.c
+++ b/fs/smb/client/cifs_swn.c
@@ -425,7 +425,7 @@ static struct cifs_swn_reg *cifs_find_swn_reg(struct cifs_tcon *tcon)
/*
* Get a registration for the tcon's server and share name, allocating a new one if it does not
- * exists
+ * exist.
*/
static struct cifs_swn_reg *cifs_get_swn_reg(struct cifs_tcon *tcon)
{
@@ -443,7 +443,7 @@ static struct cifs_swn_reg *cifs_get_swn_reg(struct cifs_tcon *tcon)
goto unlock;
}
- reg = kmalloc_obj(struct cifs_swn_reg, GFP_ATOMIC);
+ reg = kmalloc_obj(struct cifs_swn_reg, GFP_KERNEL);
if (reg == NULL) {
ret = -ENOMEM;
goto fail_unlock;
@@ -451,7 +451,7 @@ static struct cifs_swn_reg *cifs_get_swn_reg(struct cifs_tcon *tcon)
kref_init(&reg->ref_count);
- reg->id = idr_alloc(&cifs_swnreg_idr, reg, 1, 0, GFP_ATOMIC);
+ reg->id = idr_alloc(&cifs_swnreg_idr, reg, 1, 0, GFP_KERNEL);
if (reg->id < 0) {
cifs_dbg(FYI, "%s: failed to allocate registration id\n", __func__);
ret = reg->id;
diff --git a/fs/smb/client/cifsacl.c b/fs/smb/client/cifsacl.c
index 9424281a7674..12005f46307d 100644
--- a/fs/smb/client/cifsacl.c
+++ b/fs/smb/client/cifsacl.c
@@ -68,6 +68,9 @@ cifs_idmap_key_instantiate(struct key *key, struct key_preparsed_payload *prep)
{
char *payload;
+ if (prep->datalen > U16_MAX)
+ return -EINVAL;
+
/*
* If the payload is less than or equal to the size of a pointer, then
* an allocation here is wasteful. Just copy the data directly to the
diff --git a/fs/smb/client/cifsfs.h b/fs/smb/client/cifsfs.h
index 854e672a4e37..287c632392c4 100644
--- a/fs/smb/client/cifsfs.h
+++ b/fs/smb/client/cifsfs.h
@@ -54,7 +54,7 @@ void cifs_sb_deactive(struct super_block *sb);
extern const struct inode_operations cifs_dir_inode_ops;
struct inode *cifs_root_iget(struct super_block *sb);
int cifs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *direntry, umode_t mode, bool excl);
+ struct dentry *direntry, umode_t mode);
int cifs_atomic_open(struct inode *dir, struct dentry *direntry,
struct file *file, unsigned int oflags, umode_t mode);
int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir,
diff --git a/fs/smb/client/cifssmb.c b/fs/smb/client/cifssmb.c
index 40162d5554ea..1f77512252e7 100644
--- a/fs/smb/client/cifssmb.c
+++ b/fs/smb/client/cifssmb.c
@@ -1681,8 +1681,10 @@ CIFSSMBRead(const unsigned int xid, struct cifs_io_parms *io_parms,
pSMB->hdr.PidHigh = cpu_to_le16((__u16)(pid >> 16));
/* tcon and ses pointer are checked in smb_init */
- if (tcon->ses->server == NULL)
+ if (!tcon->ses->server) {
+ cifs_small_buf_release(pSMB);
return -ECONNABORTED;
+ }
pSMB->AndXCommand = 0xFF; /* none */
pSMB->Fid = netfid;
@@ -1796,8 +1798,10 @@ CIFSSMBWrite(const unsigned int xid, struct cifs_io_parms *io_parms,
pSMB->hdr.PidHigh = cpu_to_le16((__u16)(pid >> 16));
/* tcon and ses pointer are checked in smb_init */
- if (tcon->ses->server == NULL)
+ if (!tcon->ses->server) {
+ cifs_buf_release(pSMB);
return -ECONNABORTED;
+ }
pSMB->AndXCommand = 0xFF; /* none */
pSMB->Fid = netfid;
@@ -2077,8 +2081,10 @@ CIFSSMBWrite2(const unsigned int xid, struct cifs_io_parms *io_parms,
pSMB->hdr.PidHigh = cpu_to_le16((__u16)(pid >> 16));
/* tcon and ses pointer are checked in smb_init */
- if (tcon->ses->server == NULL)
+ if (!tcon->ses->server) {
+ cifs_small_buf_release(pSMB);
return -ECONNABORTED;
+ }
pSMB->AndXCommand = 0xFF; /* none */
pSMB->Fid = netfid;
diff --git a/fs/smb/client/dir.c b/fs/smb/client/dir.c
index 88a4a1787ff0..7803bd5bd01f 100644
--- a/fs/smb/client/dir.c
+++ b/fs/smb/client/dir.c
@@ -645,7 +645,7 @@ out_free_xid:
* hashed-positive by calling d_instantiate().
*/
int cifs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *direntry, umode_t mode, bool excl)
+ struct dentry *direntry, umode_t mode)
{
struct cifs_sb_info *cifs_sb = CIFS_SB(dir);
int rc;
diff --git a/fs/smb/client/inode.c b/fs/smb/client/inode.c
index b2806371bfde..dc64474d16f6 100644
--- a/fs/smb/client/inode.c
+++ b/fs/smb/client/inode.c
@@ -2287,6 +2287,13 @@ struct dentry *cifs_mkdir(struct mnt_idmap *idmap, struct inode *inode,
const char *full_path;
void *page;
+ /*
+ * vfs_mkdir() now passes S_IFDIR in @mode, but @mode is forwarded
+ * verbatim to the server and in the past only contained permission
+ * bits. Strip the type bit until SMB is verified to deal with it.
+ */
+ mode &= ~S_IFDIR;
+
cifs_dbg(FYI, "In cifs_mkdir, mode = %04ho inode = 0x%p\n",
mode, inode);
diff --git a/fs/smb/client/smb1maperror.c b/fs/smb/client/smb1maperror.c
index ab3d09613c91..395299f9121b 100644
--- a/fs/smb/client/smb1maperror.c
+++ b/fs/smb/client/smb1maperror.c
@@ -234,11 +234,7 @@ int __init smb1_init_maperror(void)
if (rc)
return rc;
- rc = mapping_table_ERRSRV_is_sorted();
- if (rc)
- return rc;
-
- return rc;
+ return mapping_table_ERRSRV_is_sorted();
}
#if IS_ENABLED(CONFIG_SMB1_KUNIT_TESTS)
diff --git a/fs/smb/server/ksmbd_netlink.h b/fs/smb/server/ksmbd_netlink.h
index 8ccd57fd904b..c9e1b0b689d7 100644
--- a/fs/smb/server/ksmbd_netlink.h
+++ b/fs/smb/server/ksmbd_netlink.h
@@ -377,6 +377,7 @@ enum KSMBD_TREE_CONN_STATUS {
#define KSMBD_SHARE_FLAG_UPDATE BIT(14)
#define KSMBD_SHARE_FLAG_CROSSMNT BIT(15)
#define KSMBD_SHARE_FLAG_CONTINUOUS_AVAILABILITY BIT(16)
+#define KSMBD_SHARE_FLAG_HIDE_UNREADABLE BIT(17)
/*
* Tree connect request flags.
diff --git a/fs/smb/server/mgmt/share_config.c b/fs/smb/server/mgmt/share_config.c
index 6f97f8d39657..e00aee155935 100644
--- a/fs/smb/server/mgmt/share_config.c
+++ b/fs/smb/server/mgmt/share_config.c
@@ -9,6 +9,7 @@
#include <linux/rwsem.h>
#include <linux/parser.h>
#include <linux/namei.h>
+#include <linux/fs_struct.h>
#include <linux/sched.h>
#include <linux/mm.h>
@@ -193,7 +194,8 @@ static struct ksmbd_share_config *share_config_request(struct ksmbd_work *work,
goto out;
}
- ret = kern_path(share->path, 0, &share->vfs_path);
+ scoped_with_init_fs()
+ ret = kern_path(share->path, 0, &share->vfs_path);
ksmbd_revert_fsids(work);
if (ret) {
ksmbd_debug(SMB, "failed to access '%s'\n",
diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c
index c1ba5e01aa7f..0caf1598aae7 100644
--- a/fs/smb/server/smb2pdu.c
+++ b/fs/smb/server/smb2pdu.c
@@ -9,6 +9,7 @@
#include <net/addrconf.h>
#include <linux/syscalls.h>
#include <linux/namei.h>
+#include <linux/fs_struct.h>
#include <linux/statfs.h>
#include <linux/ethtool.h>
#include <linux/falloc.h>
@@ -64,6 +65,9 @@ static void __wbuf(struct ksmbd_work *work, void **req, void **rsp)
/* Windows reports automatic write-time updates at roughly 15 ms resolution. */
#define KSMBD_WRITE_TIME_RESOLUTION (15ULL * 10000)
+/* MAXFILESIZE in [MS-FSA] 2.1.5.3 Server Requests a Write. */
+#define SMB2_MAX_FILE_SIZE 0xfffffff0000ULL
+
/**
* check_session_id() - check for valid session id in smb header
* @conn: connection instance
@@ -1970,6 +1974,21 @@ int smb2_sess_setup(struct ksmbd_work *work)
goto out_err;
}
+ if (conn->dialect == SMB311_PROT_ID) {
+ struct channel *chann;
+ unsigned long index;
+
+ down_read(&sess->chann_lock);
+ xa_for_each(&sess->ksmbd_chann_list, index, chann) {
+ if (conn->cipher_type != chann->conn->cipher_type)
+ rc = -EINVAL;
+ break;
+ }
+ up_read(&sess->chann_lock);
+ if (rc)
+ goto out_err;
+ }
+
if (!(req->hdr.Flags & SMB2_FLAGS_SIGNED)) {
rc = -EINVAL;
goto out_err;
@@ -2334,6 +2353,10 @@ out_err1:
if (conn->dialect == SMB311_PROT_ID &&
conn->compress_algorithm != SMB3_COMPRESS_NONE)
rsp->ShareFlags |= cpu_to_le32(SMB2_SHAREFLAG_COMPRESS_DATA);
+ if (share && test_share_config_flag(share,
+ KSMBD_SHARE_FLAG_HIDE_UNREADABLE))
+ rsp->ShareFlags |=
+ cpu_to_le32(SMB2_SHAREFLAG_ACCESS_BASED_DIRECTORY_ENUM);
rc = ksmbd_iov_pin_rsp(work, rsp, sizeof(struct smb2_tree_connect_rsp));
if (rc)
@@ -2642,6 +2665,22 @@ out:
return err;
}
+static bool smb2_is_private_ea(const char *name, size_t name_len)
+{
+ if (name_len == SD_PREFIX_LEN &&
+ !strncasecmp(name, SD_PREFIX, SD_PREFIX_LEN))
+ return true;
+ if (name_len == DOS_ATTRIBUTE_PREFIX_LEN &&
+ !strncasecmp(name, DOS_ATTRIBUTE_PREFIX,
+ DOS_ATTRIBUTE_PREFIX_LEN))
+ return true;
+ if (name_len >= STREAM_PREFIX_LEN &&
+ !strncasecmp(name, STREAM_PREFIX, STREAM_PREFIX_LEN))
+ return true;
+
+ return false;
+}
+
/**
* smb2_set_ea() - handler for setting extended attributes using set
* info command
@@ -2683,6 +2722,10 @@ static int smb2_set_ea(struct smb2_ea_info *eabuf, unsigned int buf_len,
rc = -EINVAL;
break;
}
+ if (smb2_is_private_ea(eabuf->name, eabuf->EaNameLength)) {
+ rc = -EACCES;
+ break;
+ }
memcpy(attr_name, XATTR_USER_PREFIX, XATTR_USER_PREFIX_LEN);
memcpy(&attr_name[XATTR_USER_PREFIX_LEN], eabuf->name,
@@ -3518,6 +3561,8 @@ int smb2_open(struct ksmbd_work *work)
file_present = true;
if (req->CreateOptions & FILE_DELETE_ON_CLOSE_LE) {
+ struct xattr_dos_attrib da;
+
/*
* If file exists with under flags, return access
* denied error.
@@ -3531,6 +3576,16 @@ int smb2_open(struct ksmbd_work *work)
if (!test_tree_conn_flag(tcon, KSMBD_TREE_CONN_FLAG_WRITABLE)) {
ksmbd_debug(SMB,
"User does not have write permission\n");
+ rc = -EACCES;
+ goto err_out;
+ }
+
+ if (test_share_config_flag(tcon->share_conf,
+ KSMBD_SHARE_FLAG_STORE_DOS_ATTRS) &&
+ ksmbd_vfs_get_dos_attrib_xattr(mnt_idmap(path.mnt),
+ path.dentry, &da) > 0 &&
+ da.attr & FILE_ATTRIBUTE_READONLY) {
+ rsp->hdr.Status = STATUS_CANNOT_DELETE;
rc = -EACCES;
goto err_out;
}
@@ -3548,6 +3603,13 @@ int smb2_open(struct ksmbd_work *work)
rc = 0;
}
+ if (!file_present && req->CreateOptions & FILE_DELETE_ON_CLOSE_LE &&
+ req->FileAttributes & FILE_ATTRIBUTE_READONLY_LE) {
+ rsp->hdr.Status = STATUS_CANNOT_DELETE;
+ rc = -EACCES;
+ goto err_out;
+ }
+
/*
* An explicit ::$DATA suffix names the unnamed data stream and is
* canonicalized to a NULL stream name (base file), but the request
@@ -3607,9 +3669,18 @@ int smb2_open(struct ksmbd_work *work)
if (file_present && !(req->CreateOptions & FILE_DELETE_ON_CLOSE_LE)) {
rc = smb_check_perm_dacl(conn, &path, &daccess,
- sess->user->uid);
+ req->DesiredAccess,
+ sess->user->uid, false);
if (rc)
goto err_out;
+
+ if (maximal_access_ctxt) {
+ maximal_access = FILE_MAXIMAL_ACCESS_LE;
+ rc = smb_check_perm_dacl(conn, &path, &maximal_access,
+ 0, sess->user->uid, false);
+ if (rc)
+ goto err_out;
+ }
}
if (daccess & FILE_MAXIMAL_ACCESS_LE) {
@@ -4179,8 +4250,13 @@ err_out2:
rsp->hdr.Status = STATUS_INVALID_PARAMETER;
else if (rc == -EOPNOTSUPP)
rsp->hdr.Status = STATUS_NOT_SUPPORTED;
- else if (rc == -EACCES || rc == -ESTALE || rc == -EXDEV)
- rsp->hdr.Status = STATUS_ACCESS_DENIED;
+ else if ((rc == -EACCES || rc == -ESTALE || rc == -EXDEV) &&
+ !rsp->hdr.Status) {
+ if (req->DesiredAccess & FILE_ACCESS_SYSTEM_SECURITY_LE)
+ rsp->hdr.Status = STATUS_PRIVILEGE_NOT_HELD;
+ else
+ rsp->hdr.Status = STATUS_ACCESS_DENIED;
+ }
else if (rc == -ENOENT)
rsp->hdr.Status = STATUS_OBJECT_NAME_INVALID;
else if (rc == -EPERM)
@@ -4578,6 +4654,7 @@ static int process_query_dir_entries(struct smb2_query_dir_private *priv)
for (i = 0; i < priv->d_info->num_entry; i++) {
struct dentry *dent;
+ struct path path;
if (dentry_name(priv->d_info, priv->info_level))
return -EINVAL;
@@ -4600,6 +4677,23 @@ static int process_query_dir_entries(struct smb2_query_dir_private *priv)
continue;
}
+ if (test_share_config_flag(priv->work->tcon->share_conf,
+ KSMBD_SHARE_FLAG_HIDE_UNREADABLE)) {
+ __le32 daccess = FILE_READ_DATA_LE | FILE_READ_EA_LE |
+ FILE_READ_ATTRIBUTES_LE;
+
+ path.mnt = priv->dir_fp->filp->f_path.mnt;
+ path.dentry = dent;
+ rc = smb_check_perm_dacl(priv->work->conn, &path,
+ &daccess, daccess,
+ priv->work->sess->user->uid,
+ true);
+ if (rc) {
+ dput(dent);
+ continue;
+ }
+ }
+
ksmbd_kstat.kstat = &kstat;
if (priv->info_level != FILE_NAMES_INFORMATION) {
rc = ksmbd_vfs_fill_dentry_attrs(priv->work,
@@ -5028,21 +5122,30 @@ err_out2:
/**
* buffer_check_err() - helper function to check buffer errors
* @reqOutputBufferLength: max buffer length expected in command response
+ * @fixed_len: minimum fixed response length
* @rsp: query info response buffer contains output buffer length
* @rsp_org: base response buffer pointer in case of chained response
*
* Return: 0 on success, otherwise error
*/
static int buffer_check_err(int reqOutputBufferLength,
+ unsigned int fixed_len,
struct smb2_query_info_rsp *rsp,
void *rsp_org)
{
- if (reqOutputBufferLength < le32_to_cpu(rsp->OutputBufferLength)) {
+ unsigned int output_len = le32_to_cpu(rsp->OutputBufferLength);
+
+ if (reqOutputBufferLength < fixed_len) {
pr_err("Invalid Buffer Size Requested\n");
rsp->hdr.Status = STATUS_INFO_LENGTH_MISMATCH;
*(__be32 *)rsp_org = cpu_to_be32(sizeof(struct smb2_hdr));
return -EINVAL;
}
+
+ if (reqOutputBufferLength < output_len) {
+ rsp->hdr.Status = STATUS_BUFFER_OVERFLOW;
+ rsp->OutputBufferLength = cpu_to_le32(reqOutputBufferLength);
+ }
return 0;
}
@@ -5105,11 +5208,13 @@ static int smb2_get_info_file_pipe(struct ksmbd_session *sess,
case FILE_STANDARD_INFORMATION:
get_standard_info_pipe(rsp, rsp_org);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
+ le32_to_cpu(rsp->OutputBufferLength),
rsp, rsp_org);
break;
case FILE_INTERNAL_INFORMATION:
get_internal_info_pipe(rsp, id, rsp_org);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
+ le32_to_cpu(rsp->OutputBufferLength),
rsp, rsp_org);
break;
default:
@@ -5207,17 +5312,13 @@ static int smb2_get_ea(struct ksmbd_work *work, struct ksmbd_file *fp,
if (strncmp(name, XATTR_USER_PREFIX, XATTR_USER_PREFIX_LEN))
continue;
- if (!strncmp(&name[XATTR_USER_PREFIX_LEN], STREAM_PREFIX,
- STREAM_PREFIX_LEN))
- continue;
-
if (req->InputBufferLength &&
strncmp(&name[XATTR_USER_PREFIX_LEN], ea_req->name,
ea_req->EaNameLength))
continue;
- if (!strncmp(&name[XATTR_USER_PREFIX_LEN],
- DOS_ATTRIBUTE_PREFIX, DOS_ATTRIBUTE_PREFIX_LEN))
+ if (smb2_is_private_ea(&name[XATTR_USER_PREFIX_LEN],
+ name_len - XATTR_USER_PREFIX_LEN))
continue;
if (!strncmp(name, XATTR_USER_PREFIX, XATTR_USER_PREFIX_LEN))
@@ -5930,6 +6031,7 @@ static int smb2_get_info_file(struct ksmbd_work *work,
}
if (!rc)
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
+ le32_to_cpu(rsp->OutputBufferLength),
rsp, work->response_buf);
ksmbd_fd_put(work, fp);
@@ -5951,11 +6053,13 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
struct kstatfs stfs;
struct path path;
int rc = 0, len;
+ unsigned int fixed_len = 0;
if (!share->path)
return -EIO;
- rc = kern_path(share->path, LOOKUP_NO_SYMLINKS, &path);
+ scoped_with_init_fs()
+ rc = kern_path(share->path, LOOKUP_NO_SYMLINKS, &path);
if (rc) {
pr_err("cannot create vfs path\n");
return -EIO;
@@ -5985,6 +6089,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->DeviceCharacteristics |=
cpu_to_le32(FILE_READ_ONLY_DEVICE);
rsp->OutputBufferLength = cpu_to_le32(8);
+ fixed_len = 8;
break;
}
case FS_ATTRIBUTE_INFORMATION:
@@ -6037,6 +6142,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->FileSystemNameLen = cpu_to_le32(len);
sz = sizeof(FILE_SYSTEM_ATTRIBUTE_INFO) + len;
rsp->OutputBufferLength = cpu_to_le32(sz);
+ fixed_len = 16;
break;
}
case FS_VOLUME_INFORMATION:
@@ -6064,6 +6170,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->SupportsObjects = 0;
sz = sizeof(struct filesystem_vol_info) + len;
rsp->OutputBufferLength = cpu_to_le32(sz);
+ fixed_len = 24;
break;
}
case FS_SIZE_INFORMATION:
@@ -6076,6 +6183,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->SectorsPerAllocationUnit = cpu_to_le32(1);
info->BytesPerSector = cpu_to_le32(stfs.f_bsize);
rsp->OutputBufferLength = cpu_to_le32(24);
+ fixed_len = 24;
break;
}
case FS_FULL_SIZE_INFORMATION:
@@ -6091,6 +6199,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->SectorsPerAllocationUnit = cpu_to_le32(1);
info->BytesPerSector = cpu_to_le32(stfs.f_bsize);
rsp->OutputBufferLength = cpu_to_le32(32);
+ fixed_len = 32;
break;
}
case FS_OBJECT_ID_INFORMATION:
@@ -6111,6 +6220,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->extended_info.rel_date = 0;
memcpy(info->extended_info.version_string, "1.1.0", strlen("1.1.0"));
rsp->OutputBufferLength = cpu_to_le32(64);
+ fixed_len = 64;
break;
}
case FS_SECTOR_SIZE_INFORMATION:
@@ -6132,6 +6242,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->ByteOffsetForSectorAlignment = 0;
info->ByteOffsetForPartitionAlignment = 0;
rsp->OutputBufferLength = cpu_to_le32(28);
+ fixed_len = 28;
break;
}
case FS_CONTROL_INFORMATION:
@@ -6152,6 +6263,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->DefaultQuotaLimit = cpu_to_le64(SMB2_NO_FID);
info->Padding = 0;
rsp->OutputBufferLength = cpu_to_le32(48);
+ fixed_len = 48;
break;
}
case FS_POSIX_INFORMATION:
@@ -6172,6 +6284,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->TotalFileNodes = cpu_to_le64(stfs.f_files);
info->FreeFileNodes = cpu_to_le64(stfs.f_ffree);
rsp->OutputBufferLength = cpu_to_le32(56);
+ fixed_len = 56;
}
break;
}
@@ -6180,6 +6293,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
return -EOPNOTSUPP;
}
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
+ fixed_len,
rsp, work->response_buf);
path_put(&path);
@@ -6281,6 +6395,7 @@ release_acl:
rsp->OutputBufferLength = cpu_to_le32(secdesclen);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
+ le32_to_cpu(rsp->OutputBufferLength),
rsp, work->response_buf);
if (rc)
goto err_out;
@@ -6906,6 +7021,9 @@ static int set_file_disposition_info(struct ksmbd_work *work,
return -EACCES;
}
+ if (fp->f_ci->m_fattr & FILE_ATTRIBUTE_READONLY_LE)
+ return -EACCES;
+
inode = file_inode(fp->filp);
if (file_info->DeletePending) {
if (ksmbd_has_stream_without_delete_share(fp))
@@ -7176,8 +7294,14 @@ int smb2_set_info(struct ksmbd_work *work)
return 0;
err_out:
- if (rc == -EACCES || rc == -EPERM || rc == -EXDEV)
- rsp->hdr.Status = STATUS_ACCESS_DENIED;
+ if (rc == -EACCES || rc == -EPERM || rc == -EXDEV) {
+ if (fp && req->InfoType == SMB2_O_INFO_FILE &&
+ req->FileInfoClass == FILE_DISPOSITION_INFORMATION &&
+ fp->f_ci->m_fattr & FILE_ATTRIBUTE_READONLY_LE)
+ rsp->hdr.Status = STATUS_CANNOT_DELETE;
+ else
+ rsp->hdr.Status = STATUS_ACCESS_DENIED;
+ }
else if (rc == -EINVAL)
rsp->hdr.Status = STATUS_INVALID_PARAMETER;
else if (rc == -EMSGSIZE)
@@ -7676,8 +7800,10 @@ int smb2_write(struct ksmbd_work *work)
}
offset = le64_to_cpu(req->Offset);
- if (offset < 0)
- return -EINVAL;
+ if (offset < 0) {
+ err = -EINVAL;
+ goto out;
+ }
length = le32_to_cpu(req->Length);
if (req->Channel == SMB2_CHANNEL_RDMA_V1 ||
@@ -7691,6 +7817,19 @@ int smb2_write(struct ksmbd_work *work)
length = le32_to_cpu(req->RemainingBytes);
}
+ if (length) {
+ u64 end = (u64)offset + length;
+
+ if (end > SMB2_MAX_FILE_SIZE) {
+ err = -EINVAL;
+ goto out;
+ }
+ if (end == SMB2_MAX_FILE_SIZE) {
+ err = -EFBIG;
+ goto out;
+ }
+ }
+
if (is_rdma_channel == true) {
unsigned int ch_offset = le16_to_cpu(req->WriteChannelInfoOffset);
diff --git a/fs/smb/server/smbacl.c b/fs/smb/server/smbacl.c
index c13f07a09ab8..b5db6dcfbaa4 100644
--- a/fs/smb/server/smbacl.c
+++ b/fs/smb/server/smbacl.c
@@ -27,6 +27,9 @@ static const struct smb_sid creator_owner = {
/* security id for everyone/world system group */
static const struct smb_sid creator_group = {
1, 1, {0, 0, 0, 0, 0, 3}, {cpu_to_le32(1)} };
+/* security id for owner rights */
+static const struct smb_sid sid_owner_rights = {
+ 1, 1, {0, 0, 0, 0, 0, 3}, {cpu_to_le32(4)} };
/* security id for everyone/world system group */
static const struct smb_sid sid_everyone = {
@@ -1432,7 +1435,8 @@ bool smb_inherit_flags(int flags, bool is_dir)
}
int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
- __le32 *pdaccess, int uid)
+ __le32 *pdaccess, __le32 raw_daccess, int uid,
+ bool strict)
{
struct mnt_idmap *idmap = mnt_idmap(path->mnt);
struct smb_ntsd *pntsd = NULL;
@@ -1442,14 +1446,17 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
unsigned int dacl_offset;
size_t dacl_struct_end;
struct smb_sid sid;
- int granted = le32_to_cpu(*pdaccess & ~FILE_MAXIMAL_ACCESS_LE);
+ int requested = le32_to_cpu(*pdaccess & ~FILE_MAXIMAL_ACCESS_LE);
+ int granted = requested;
struct smb_ace *ace;
int i, found = 0;
- unsigned int access_bits = 0;
+ unsigned int access_bits = 0, denied = 0;
struct smb_ace *others_ace = NULL;
struct posix_acl_entry *pa_entry;
unsigned int sid_type = SIDOWNER;
unsigned short ace_size;
+ bool is_owner, owner_rights = false;
+ vfsuid_t vfsuid;
ksmbd_debug(SMB, "check permission using windows acl\n");
pntsd_size = ksmbd_vfs_get_sd_xattr(conn, idmap,
@@ -1479,10 +1486,13 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
goto err_out;
}
- if (*pdaccess & FILE_MAXIMAL_ACCESS_LE) {
- granted = READ_CONTROL | WRITE_DAC | FILE_READ_ATTRIBUTES |
- DELETE;
+ if (!uid)
+ sid_type = SIDUNIX_USER;
+ id_to_sid(uid, sid_type, &sid);
+ vfsuid = i_uid_into_vfsuid(idmap, d_inode(path->dentry));
+ is_owner = uid == from_kuid(&init_user_ns, vfsuid_into_kuid(vfsuid));
+ if (*pdaccess & FILE_MAXIMAL_ACCESS_LE) {
ace = (struct smb_ace *)((char *)pdacl + sizeof(struct smb_acl));
aces_size = acl_size - sizeof(struct smb_acl);
for (i = 0; i < le16_to_cpu(pdacl->num_aces); i++) {
@@ -1495,15 +1505,51 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
CIFS_SID_BASE_SIZE)
break;
aces_size -= ace_size;
- granted |= le32_to_cpu(ace->access_req);
+
+ if (ace->sid.num_subauth > SID_MAX_SUB_AUTHORITIES ||
+ ace_size < offsetof(struct smb_ace, sid) +
+ CIFS_SID_BASE_SIZE +
+ sizeof(__le32) * ace->sid.num_subauth)
+ break;
+
+ if (!compare_sids(&sid_owner_rights, &ace->sid)) {
+ owner_rights = true;
+ if (!is_owner)
+ goto next_ace;
+ }
+
+ if (ace->flags & INHERIT_ONLY_ACE ||
+ (compare_sids(&sid, &ace->sid) &&
+ compare_sids(&sid_unix_NFS_mode, &ace->sid) &&
+ compare_sids(&sid_everyone, &ace->sid) &&
+ compare_sids(&sid_authusers, &ace->sid) &&
+ compare_sids(&sid_owner_rights, &ace->sid)))
+ goto next_ace;
+
+ switch (ace->type) {
+ case ACCESS_ALLOWED_ACE_TYPE:
+ access_bits |= le32_to_cpu(ace->access_req);
+ break;
+ case ACCESS_DENIED_ACE_TYPE:
+ case ACCESS_DENIED_CALLBACK_ACE_TYPE:
+ denied |= ~access_bits &
+ le32_to_cpu(ace->access_req);
+ break;
+ }
+next_ace:
ace = (struct smb_ace *)((char *)ace + le16_to_cpu(ace->size));
}
+ if (is_owner && !owner_rights)
+ access_bits |= READ_CONTROL | WRITE_DAC |
+ FILE_READ_ATTRIBUTES | DELETE;
+ access_bits &= ~denied;
+ if ((raw_daccess & FILE_GENERIC_EXECUTE_LE) &&
+ S_ISREG(d_inode(path->dentry)->i_mode) &&
+ (access_bits & GENERIC_READ_FLAGS) == GENERIC_READ_FLAGS)
+ access_bits |= FILE_EXECUTE;
+ granted = requested | access_bits;
}
- if (!uid)
- sid_type = SIDUNIX_USER;
- id_to_sid(uid, sid_type, &sid);
-
ace = (struct smb_ace *)((char *)pdacl + sizeof(struct smb_acl));
aces_size = acl_size - sizeof(struct smb_acl);
for (i = 0; i < le16_to_cpu(pdacl->num_aces); i++) {
@@ -1527,25 +1573,16 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
found = 1;
break;
}
- if (!compare_sids(&sid_everyone, &ace->sid))
+ if (!compare_sids(&sid_everyone, &ace->sid) ||
+ !compare_sids(&sid_authusers, &ace->sid))
others_ace = ace;
ace = (struct smb_ace *)((char *)ace + le16_to_cpu(ace->size));
}
- if (*pdaccess & FILE_MAXIMAL_ACCESS_LE && found) {
- granted = READ_CONTROL | WRITE_DAC | FILE_READ_ATTRIBUTES |
- DELETE;
-
- granted |= le32_to_cpu(ace->access_req);
-
- if (!pdacl->num_aces)
- granted = GENERIC_ALL_FLAGS;
- }
-
if (IS_ENABLED(CONFIG_FS_POSIX_ACL)) {
posix_acls = get_inode_acl(d_inode(path->dentry), ACL_TYPE_ACCESS);
- if (!IS_ERR_OR_NULL(posix_acls) && !found) {
+ if (!IS_ERR_OR_NULL(posix_acls) && !found && !others_ace) {
unsigned int id = -1;
pa_entry = posix_acls->a_entries;
@@ -1583,19 +1620,27 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
}
}
- switch (ace->type) {
- case ACCESS_ALLOWED_ACE_TYPE:
- access_bits = le32_to_cpu(ace->access_req);
- break;
- case ACCESS_DENIED_ACE_TYPE:
- case ACCESS_DENIED_CALLBACK_ACE_TYPE:
- access_bits = le32_to_cpu(~ace->access_req);
- break;
+ if (!(*pdaccess & FILE_MAXIMAL_ACCESS_LE)) {
+ switch (ace->type) {
+ case ACCESS_ALLOWED_ACE_TYPE:
+ access_bits = le32_to_cpu(ace->access_req);
+ break;
+ case ACCESS_DENIED_ACE_TYPE:
+ case ACCESS_DENIED_CALLBACK_ACE_TYPE:
+ access_bits = le32_to_cpu(~ace->access_req);
+ break;
+ }
}
check_access_bits:
- if (granted &
- ~(access_bits | FILE_READ_ATTRIBUTES | READ_CONTROL | WRITE_DAC | DELETE)) {
+ if (strict) {
+ access_bits &= granted;
+ } else {
+ access_bits |= FILE_READ_ATTRIBUTES | READ_CONTROL |
+ WRITE_DAC | DELETE;
+ }
+
+ if (granted & ~access_bits) {
ksmbd_debug(SMB, "Access denied with winACL, granted : %x, access_req : %x\n",
granted, le32_to_cpu(ace->access_req));
rc = -EACCES;
diff --git a/fs/smb/server/smbacl.h b/fs/smb/server/smbacl.h
index ab21ba2cd4df..01810c16cc04 100644
--- a/fs/smb/server/smbacl.h
+++ b/fs/smb/server/smbacl.h
@@ -95,7 +95,8 @@ bool smb_inherit_flags(int flags, bool is_dir);
int smb_inherit_dacl(struct ksmbd_conn *conn, const struct path *path,
unsigned int uid, unsigned int gid);
int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path,
- __le32 *pdaccess, int uid);
+ __le32 *pdaccess, __le32 raw_daccess, int uid,
+ bool strict);
int set_info_sec(struct ksmbd_conn *conn, struct ksmbd_tree_connect *tcon,
const struct path *path, struct smb_ntsd *pntsd, int ntsd_len,
bool type_check, bool get_write);
diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c
index d324585c0566..2f111b1923d7 100644
--- a/fs/smb/server/vfs.c
+++ b/fs/smb/server/vfs.c
@@ -7,6 +7,7 @@
#include <crypto/sha2.h>
#include <linux/kernel.h>
#include <linux/fs.h>
+#include <linux/fs_struct.h>
#include <linux/filelock.h>
#include <linux/uaccess.h>
#include <linux/backing-dev.h>
@@ -67,8 +68,9 @@ static int ksmbd_vfs_path_lookup(struct ksmbd_share_config *share_conf,
}
CLASS(filename_kernel, filename)(pathname);
- err = vfs_path_parent_lookup(filename, flags, path, &last,
- root_share_path);
+ scoped_with_init_fs()
+ err = vfs_path_parent_lookup(filename, flags, path, &last,
+ root_share_path);
if (err)
return err;
@@ -297,9 +299,6 @@ static int check_lock_range(struct file *filp, loff_t start, loff_t end,
struct file_lock_context *ctx = locks_inode_context(file_inode(filp));
int error = 0;
- if (start == end)
- return 0;
-
if (!ctx || list_empty_careful(&ctx->flc_posix))
return 0;
@@ -345,7 +344,7 @@ int ksmbd_vfs_read(struct ksmbd_work *work, struct ksmbd_file *fp, size_t count,
ssize_t nbytes = 0;
struct inode *inode = file_inode(filp);
- if (S_ISDIR(inode->i_mode))
+ if (S_ISDIR(inode->i_mode) && !ksmbd_stream_fd(fp))
return -EISDIR;
if (unlikely(count == 0))
@@ -474,7 +473,8 @@ int ksmbd_vfs_write(struct ksmbd_work *work, struct ksmbd_file *fp,
if (work->conn->connection_type) {
if (!(fp->daccess & (FILE_WRITE_DATA_LE | FILE_APPEND_DATA_LE)) ||
- S_ISDIR(file_inode(fp->filp)->i_mode)) {
+ (S_ISDIR(file_inode(fp->filp)->i_mode) &&
+ !ksmbd_stream_fd(fp))) {
pr_err("no right to write(%pD)\n", fp->filp);
err = -EACCES;
goto out;
@@ -623,7 +623,8 @@ int ksmbd_vfs_link(struct ksmbd_work *work, const char *oldname,
if (ksmbd_override_fsids(work))
return -ENOMEM;
- err = kern_path(oldname, LOOKUP_NO_SYMLINKS, &oldpath);
+ scoped_with_init_fs()
+ err = kern_path(oldname, LOOKUP_NO_SYMLINKS, &oldpath);
if (err) {
pr_err("cannot get linux path for %s, err = %d\n",
oldname, err);
diff --git a/fs/super.c b/fs/super.c
index ffdcc6a2e0de..5feecf5d9038 100644
--- a/fs/super.c
+++ b/fs/super.c
@@ -24,6 +24,8 @@
#include <linux/export.h>
#include <linux/slab.h>
#include <linux/blkdev.h>
+#include <linux/memcontrol.h>
+#include <linux/rhashtable.h>
#include <linux/mount.h>
#include <linux/security.h>
#include <linux/writeback.h> /* for the emergency remount stuff */
@@ -102,7 +104,7 @@ static bool super_flags(const struct super_block *sb, unsigned int flags)
* creation will succeed and SB_BORN is set by vfs_get_tree() or we're
* woken and we'll see SB_DYING.
*
- * The caller must have acquired a temporary reference on @sb->s_count.
+ * The caller must have acquired a temporary reference on @sb->s_passive.
*
* Return: The function returns true if SB_BORN was set and with
* s_umount held. The function returns false if SB_DYING was
@@ -170,6 +172,19 @@ static void super_wake(struct super_block *sb, unsigned int flag)
}
/*
+ * The s_op->nr_cached_objects hooks (used for example by btrfs and xfs)
+ * operate on filesystem-global state and ignore sc->memcg. Driving them
+ * from per-memcg shrink_slab_memcg() invocations only burns CPU walking
+ * per-cpu counters and queueing duplicate work: the actual reclaim happens on
+ * the global path (kswapd or root direct reclaim) regardless. Restrict them
+ * to that path.
+ */
+static inline bool super_fs_objects_eligible(struct shrink_control *sc)
+{
+ return !sc->memcg || mem_cgroup_is_root(sc->memcg);
+}
+
+/*
* One thing we have to be careful of with a per-sb shrinker is that we don't
* drop the last active reference to the superblock from within the shrinker.
* If that happens we could trigger unregistering the shrinker from within the
@@ -198,7 +213,7 @@ static unsigned long super_cache_scan(struct shrinker *shrink,
if (!super_trylock_shared(sb))
return SHRINK_STOP;
- if (sb->s_op->nr_cached_objects)
+ if (sb->s_op->nr_cached_objects && super_fs_objects_eligible(sc))
fs_objects = sb->s_op->nr_cached_objects(sb, sc);
inodes = list_lru_shrink_count(&sb->s_inode_lru, sc);
@@ -259,7 +274,8 @@ static unsigned long super_cache_count(struct shrinker *shrink,
return 0;
smp_rmb();
- if (sb->s_op && sb->s_op->nr_cached_objects)
+ if (sb->s_op && sb->s_op->nr_cached_objects &&
+ super_fs_objects_eligible(sc))
total_objects = sb->s_op->nr_cached_objects(sb, sc);
total_objects += list_lru_shrink_count(&sb->s_dentry_lru, sc);
@@ -272,6 +288,8 @@ static unsigned long super_cache_count(struct shrinker *shrink,
return total_objects;
}
+static struct super_dev *super_dev_alloc(dev_t dev, struct super_block *sb);
+
static void destroy_super_work(struct work_struct *work)
{
struct super_block *s = container_of(work, struct super_block,
@@ -279,6 +297,8 @@ static void destroy_super_work(struct work_struct *work)
fsnotify_sb_free(s);
security_sb_free(s);
put_user_ns(s->s_user_ns);
+ /* Only an unregistered entry is still owned by the superblock. */
+ kfree(s->s_super_dev);
kfree(s->s_subtype);
for (int i = 0; i < SB_FREEZE_LEVELS; i++)
percpu_free_rwsem(&s->s_writers.rw_sem[i]);
@@ -367,7 +387,7 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags,
spin_lock_init(&s->s_inode_wblist_lock);
fserror_mount(s);
- s->s_count = 1;
+ refcount_set(&s->s_passive, 1);
atomic_set(&s->s_active, 1);
mutex_init(&s->s_vfs_rename_mutex);
lockdep_set_class(&s->s_vfs_rename_mutex, &type->s_vfs_rename_key);
@@ -392,6 +412,10 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags,
goto fail;
if (list_lru_init_memcg(&s->s_inode_lru, s->s_shrink))
goto fail;
+ s->s_super_dev = super_dev_alloc(0, s);
+ if (!s->s_super_dev)
+ goto fail;
+
s->s_min_writeback_pages = MIN_WRITEBACK_PAGES;
return s;
@@ -403,12 +427,17 @@ fail:
/* Superblock refcounting */
/*
- * Drop a superblock's refcount. The caller must hold sb_lock.
+ * Drop a superblock's passive reference. Must be called WITHOUT sb_lock held;
+ * put_super() acquires sb_lock itself when the final reference is dropped.
*/
-static void __put_super(struct super_block *s)
+void put_super(struct super_block *s)
{
- if (!--s->s_count) {
+ if (refcount_dec_and_test(&s->s_passive)) {
+
+ spin_lock(&sb_lock);
list_del_init(&s->s_list);
+ spin_unlock(&sb_lock);
+
WARN_ON(s->s_dentry_lru.node);
WARN_ON(s->s_inode_lru.node);
WARN_ON(s->s_mounts);
@@ -416,18 +445,109 @@ static void __put_super(struct super_block *s)
}
}
-/**
- * put_super - drop a temporary reference to superblock
- * @sb: superblock in question
- *
- * Drops a temporary reference, frees superblock if there's no
- * references left.
- */
-void put_super(struct super_block *sb)
+struct super_dev {
+ dev_t sd_dev;
+ struct super_block *sd_sb;
+ refcount_t sd_ref;
+ struct rhlist_head sd_node;
+ struct rcu_head sd_rcu;
+};
+
+static struct rhltable super_dev_table;
+static const struct rhashtable_params super_dev_params = {
+ .key_len = sizeof(dev_t),
+ .key_offset = offsetof(struct super_dev, sd_dev),
+ .head_offset = offsetof(struct super_dev, sd_node),
+};
+
+static struct super_dev *super_dev_alloc(dev_t dev, struct super_block *sb)
{
- spin_lock(&sb_lock);
- __put_super(sb);
- spin_unlock(&sb_lock);
+ struct super_dev *fsd;
+
+ fsd = kzalloc_obj(*fsd);
+ if (!fsd)
+ return NULL;
+ fsd->sd_dev = dev;
+ fsd->sd_sb = sb;
+ refcount_set(&fsd->sd_ref, 1);
+ return fsd;
+}
+
+static void super_dev_put(struct super_dev *fsd)
+{
+ /* Unlink only once unpinned, so a cursor never resumes from a removed node. */
+ if (fsd && refcount_dec_and_test(&fsd->sd_ref)) {
+ rhltable_remove(&super_dev_table, &fsd->sd_node, super_dev_params);
+ put_super(fsd->sd_sb);
+ kfree_rcu(fsd, sd_rcu);
+ }
+}
+
+void __init super_dev_init(void)
+{
+ if (rhltable_init(&super_dev_table, &super_dev_params))
+ panic("VFS: Cannot initialise super_dev_table\n");
+}
+
+static int super_dev_insert(struct super_dev *fsd)
+{
+ int err;
+
+ err = rhltable_insert(&super_dev_table, &fsd->sd_node, super_dev_params);
+ if (!err)
+ refcount_inc(&fsd->sd_sb->s_passive);
+ return err;
+}
+
+/* Register @sb under @sb->s_dev as the final fallible act of a set callback. */
+static int super_dev_register(struct super_block *sb)
+{
+ struct super_dev *fsd = sb->s_super_dev;
+ int err;
+
+ lockdep_assert_held(&sb_lock);
+ VFS_WARN_ON_ONCE(!sb->s_dev);
+ VFS_WARN_ON_ONCE(!fsd || fsd->sd_dev);
+
+ fsd->sd_dev = sb->s_dev;
+ err = super_dev_insert(fsd);
+ if (err)
+ fsd->sd_dev = 0;
+ return err;
+}
+
+static struct super_dev *super_dev_get(struct rhlist_head *pos)
+{
+ struct super_dev *sb_dev;
+
+ for (; pos; pos = rcu_dereference_all(pos->next)) {
+ sb_dev = container_of(pos, struct super_dev, sd_node);
+ if (refcount_inc_not_zero(&sb_dev->sd_ref))
+ return sb_dev;
+ }
+ return NULL;
+}
+
+static struct super_dev *super_dev_first(dev_t dev)
+{
+ struct super_dev *sb_dev;
+
+ rcu_read_lock();
+ sb_dev = super_dev_get(rhltable_lookup(&super_dev_table, &dev, super_dev_params));
+ rcu_read_unlock();
+ return sb_dev;
+}
+
+static struct super_dev *super_dev_next(struct super_dev *prev)
+{
+ struct super_dev *sb_dev;
+
+ rcu_read_lock();
+ sb_dev = super_dev_get(rcu_dereference_all(prev->sd_node.next));
+ rcu_read_unlock();
+
+ super_dev_put(prev);
+ return sb_dev;
}
static void kill_super_notify(struct super_block *sb)
@@ -449,6 +569,12 @@ static void kill_super_notify(struct super_block *sb)
hlist_del_init(&sb->s_instances);
spin_unlock(&sb_lock);
+ /* Drop sget_fc()'s claim; a never-registered entry stays with the sb. */
+ if (sb->s_super_dev->sd_dev) {
+ super_dev_put(sb->s_super_dev);
+ sb->s_super_dev = NULL;
+ }
+
/*
* Let concurrent mounts know that this thing is really dead.
* We don't need @sb->s_umount here as every concurrent caller
@@ -478,11 +604,7 @@ void deactivate_locked_super(struct super_block *s)
kill_super_notify(s);
- /*
- * Since list_lru_destroy() may sleep, we cannot call it from
- * put_super(), where we hold the sb_lock. Therefore we destroy
- * the lru lists right now.
- */
+ /* list_lru_destroy() may sleep; put_super() callers may not. */
list_lru_destroy(&s->s_dentry_lru);
list_lru_destroy(&s->s_inode_lru);
@@ -529,7 +651,7 @@ static bool grab_super(struct super_block *sb)
{
bool locked;
- sb->s_count++;
+ refcount_inc(&sb->s_passive);
spin_unlock(&sb_lock);
locked = super_lock_excl(sb);
if (locked) {
@@ -556,7 +678,7 @@ static bool grab_super(struct super_block *sb)
* lock held in read mode in case of success. On successful return,
* the caller must drop the s_umount lock when done.
*
- * Note that unlike get_super() et.al. this one does *not* bump ->s_count.
+ * Note that unlike get_super() et.al. this one does *not* bump ->s_passive.
* The reason why it's safe is that we are OK with doing trylock instead
* of down_read(). There's a couple of places that are OK with that, but
* it's very much not a general-purpose interface.
@@ -763,6 +885,7 @@ retry:
}
if (!s) {
spin_unlock(&sb_lock);
+
s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
if (!s)
return ERR_PTR(-ENOMEM);
@@ -772,11 +895,13 @@ retry:
s->s_fs_info = fc->s_fs_info;
err = set(s, fc);
if (err) {
+ VFS_WARN_ON_ONCE(s->s_super_dev->sd_dev);
s->s_fs_info = NULL;
spin_unlock(&sb_lock);
destroy_unused_super(s);
return ERR_PTR(err);
}
+ VFS_WARN_ON_ONCE(!s->s_super_dev->sd_dev);
fc->s_fs_info = NULL;
s->s_type = fc->fs_type;
s->s_iflags |= fc->s_iflags;
@@ -851,14 +976,17 @@ static void __iterate_supers(void (*f)(struct super_block *, void *), void *arg,
struct super_block *sb, *p = NULL;
bool excl = flags & SUPER_ITER_EXCL;
- guard(spinlock)(&sb_lock);
+ spin_lock(&sb_lock);
for (sb = first_super(flags);
!list_entry_is_head(sb, &super_blocks, s_list);
sb = next_super(sb, flags)) {
if (super_flags(sb, SB_DYING))
continue;
- sb->s_count++;
+
+ if (!refcount_inc_not_zero(&sb->s_passive))
+ continue;
+
spin_unlock(&sb_lock);
if (flags & SUPER_ITER_UNLOCKED) {
@@ -868,13 +996,14 @@ static void __iterate_supers(void (*f)(struct super_block *, void *), void *arg,
super_unlock(sb, excl);
}
- spin_lock(&sb_lock);
if (p)
- __put_super(p);
+ put_super(p);
p = sb;
+ spin_lock(&sb_lock);
}
+ spin_unlock(&sb_lock);
if (p)
- __put_super(p);
+ put_super(p);
}
void iterate_supers(void (*f)(struct super_block *, void *), void *arg)
@@ -903,7 +1032,9 @@ void iterate_supers_type(struct file_system_type *type,
if (super_flags(sb, SB_DYING))
continue;
- sb->s_count++;
+ if (!refcount_inc_not_zero(&sb->s_passive))
+ continue;
+
spin_unlock(&sb_lock);
locked = super_lock_shared(sb);
@@ -912,41 +1043,33 @@ void iterate_supers_type(struct file_system_type *type,
super_unlock_shared(sb);
}
- spin_lock(&sb_lock);
if (p)
- __put_super(p);
+ put_super(p);
p = sb;
+ spin_lock(&sb_lock);
}
- if (p)
- __put_super(p);
spin_unlock(&sb_lock);
+ if (p)
+ put_super(p);
}
EXPORT_SYMBOL(iterate_supers_type);
struct super_block *user_get_super(dev_t dev, bool excl)
{
- struct super_block *sb;
+ struct super_dev *sb_dev;
- spin_lock(&sb_lock);
- list_for_each_entry(sb, &super_blocks, s_list) {
- bool locked;
+ for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
+ struct super_block *sb = sb_dev->sd_sb;
- if (sb->s_dev != dev)
+ if (!super_lock(sb, excl))
continue;
- sb->s_count++;
- spin_unlock(&sb_lock);
-
- locked = super_lock(sb, excl);
- if (locked)
- return sb;
-
- spin_lock(&sb_lock);
- __put_super(sb);
- break;
+ /* The pinned entry holds a passive reference, take our own. */
+ refcount_inc(&sb->s_passive);
+ super_dev_put(sb_dev);
+ return sb;
}
- spin_unlock(&sb_lock);
return NULL;
}
@@ -1228,7 +1351,16 @@ EXPORT_SYMBOL(free_anon_bdev);
int set_anon_super(struct super_block *s, void *data)
{
- return get_anon_bdev(&s->s_dev);
+ int error;
+
+ error = get_anon_bdev(&s->s_dev);
+ if (error)
+ return error;
+
+ error = super_dev_register(s);
+ if (error)
+ free_anon_bdev(s->s_dev);
+ return error;
}
EXPORT_SYMBOL(set_anon_super);
@@ -1314,7 +1446,7 @@ EXPORT_SYMBOL(get_tree_keyed);
static int set_bdev_super(struct super_block *s, void *data)
{
s->s_dev = *(dev_t *)data;
- return 0;
+ return super_dev_register(s);
}
static int super_s_dev_set(struct super_block *s, struct fs_context *fc)
@@ -1356,197 +1488,313 @@ struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
EXPORT_SYMBOL(sget_dev);
#ifdef CONFIG_BLOCK
-/*
- * Lock the superblock that is holder of the bdev. Returns the superblock
- * pointer if we successfully locked the superblock and it is alive. Otherwise
- * we return NULL and just unlock bdev->bd_holder_lock.
- *
- * The function must be called with bdev->bd_holder_lock and releases it.
- */
-static struct super_block *bdev_super_lock(struct block_device *bdev, bool excl)
- __releases(&bdev->bd_holder_lock)
+static int fs_super_freeze(struct super_block *sb)
{
- struct super_block *sb = bdev->bd_holder;
- bool locked;
-
- lockdep_assert_held(&bdev->bd_holder_lock);
- lockdep_assert_not_held(&sb->s_umount);
- lockdep_assert_not_held(&bdev->bd_disk->open_mutex);
-
- /* Make sure sb doesn't go away from under us */
- spin_lock(&sb_lock);
- sb->s_count++;
- spin_unlock(&sb_lock);
-
- mutex_unlock(&bdev->bd_holder_lock);
-
- locked = super_lock(sb, excl);
-
- /*
- * If the superblock wasn't already SB_DYING then we hold
- * s_umount and can safely drop our temporary reference.
- */
- put_super(sb);
-
- if (!locked)
- return NULL;
-
- if (!sb->s_root || !(sb->s_flags & SB_ACTIVE)) {
- super_unlock(sb, excl);
- return NULL;
- }
+ if (sb->s_op->freeze_super)
+ return sb->s_op->freeze_super(sb,
+ FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
+ return freeze_super(sb, FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
+}
- return sb;
+static int fs_super_thaw(struct super_block *sb)
+{
+ if (sb->s_op->thaw_super)
+ return sb->s_op->thaw_super(sb,
+ FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
+ return thaw_super(sb, FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
}
static void fs_bdev_mark_dead(struct block_device *bdev, bool surprise)
{
- struct super_block *sb;
+ struct super_dev *sb_dev;
+ dev_t dev = bdev->bd_dev;
- sb = bdev_super_lock(bdev, false);
- if (!sb)
- return;
+ mutex_unlock(&bdev->bd_holder_lock);
- if (sb->s_op->remove_bdev) {
- int ret;
+ for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
+ struct super_block *sb = sb_dev->sd_sb;
- ret = sb->s_op->remove_bdev(sb, bdev);
- if (!ret) {
- super_unlock_shared(sb);
- return;
+ if (!super_lock_shared(sb))
+ continue;
+ if (sb->s_root && (sb->s_flags & SB_ACTIVE)) {
+ if (!sb->s_op->remove_bdev ||
+ sb->s_op->remove_bdev(sb, bdev)) {
+ if (!surprise)
+ sync_filesystem(sb);
+ shrink_dcache_sb(sb);
+ evict_inodes(sb);
+ if (sb->s_op->shutdown)
+ sb->s_op->shutdown(sb);
+ }
}
- /* Fallback to shutdown. */
+ super_unlock_shared(sb);
}
-
- if (!surprise)
- sync_filesystem(sb);
- shrink_dcache_sb(sb);
- evict_inodes(sb);
- if (sb->s_op->shutdown)
- sb->s_op->shutdown(sb);
-
- super_unlock_shared(sb);
}
static void fs_bdev_sync(struct block_device *bdev)
{
- struct super_block *sb;
-
- sb = bdev_super_lock(bdev, false);
- if (!sb)
- return;
+ struct super_dev *sb_dev;
+ dev_t dev = bdev->bd_dev;
- sync_filesystem(sb);
- super_unlock_shared(sb);
-}
+ mutex_unlock(&bdev->bd_holder_lock);
-static struct super_block *get_bdev_super(struct block_device *bdev)
-{
- bool active = false;
- struct super_block *sb;
+ for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
+ struct super_block *sb = sb_dev->sd_sb;
- sb = bdev_super_lock(bdev, true);
- if (sb) {
- active = atomic_inc_not_zero(&sb->s_active);
- super_unlock_excl(sb);
+ if (!super_lock_shared(sb))
+ continue;
+ if (sb->s_root && (sb->s_flags & SB_ACTIVE))
+ sync_filesystem(sb);
+ super_unlock_shared(sb);
}
- if (!active)
- return NULL;
- return sb;
}
/**
- * fs_bdev_freeze - freeze owning filesystem of block device
+ * fs_bdev_freeze - freeze every superblock using a block device
* @bdev: block device
*
- * Freeze the filesystem that owns this block device if it is still
- * active.
- *
- * A filesystem that owns multiple block devices may be frozen from each
- * block device and won't be unfrozen until all block devices are
- * unfrozen. Each block device can only freeze the filesystem once as we
- * nest freezes for block devices in the block layer.
+ * Freeze each live superblock using @bdev. A superblock owning several block
+ * devices is frozen once per device and stays frozen until all are thawed; the
+ * block layer nests these freezes so the count stays balanced.
*
- * Return: If the freeze was successful zero is returned. If the freeze
- * failed a negative error code is returned.
+ * Return: 0, or the error from the one superblock on a single-fs device. When
+ * several superblocks share @bdev a per-superblock failure is swallowed
+ * (see below), but a sync_blockdev() failure is always reported.
*/
static int fs_bdev_freeze(struct block_device *bdev)
{
- struct super_block *sb;
- int error = 0;
+ dev_t dev = bdev->bd_dev;
+ struct super_dev *sb_dev;
+ unsigned int count = 0;
+ int error = 0, err;
lockdep_assert_held(&bdev->bd_fsfreeze_mutex);
- sb = get_bdev_super(bdev);
- if (!sb)
- return -EINVAL;
+ mutex_unlock(&bdev->bd_holder_lock);
- if (sb->s_op->freeze_super)
- error = sb->s_op->freeze_super(sb,
- FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
- else
- error = freeze_super(sb,
- FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
+ for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
+ if (!get_active_super(sb_dev->sd_sb))
+ continue;
+ err = fs_super_freeze(sb_dev->sd_sb);
+ if (err && !error)
+ error = err;
+ deactivate_super(sb_dev->sd_sb);
+ count++;
+ }
+
+ /*
+ * When several superblocks share the device, keep it frozen even if some
+ * of them failed to freeze and swallow the error: rolling the rest back
+ * via thaw_super() can fail too, so neither is a clear win. A single
+ * filesystem (count == 1) still reports its error.
+ */
+ if (error && count > 1)
+ error = 0;
if (!error)
error = sync_blockdev(bdev);
- deactivate_super(sb);
return error;
}
/**
- * fs_bdev_thaw - thaw owning filesystem of block device
+ * fs_bdev_thaw - thaw every superblock using a block device
* @bdev: block device
*
- * Thaw the filesystem that owns this block device.
- *
- * A filesystem that owns multiple block devices may be frozen from each
- * block device and won't be unfrozen until all block devices are
- * unfrozen. Each block device can only freeze the filesystem once as we
- * nest freezes for block devices in the block layer.
+ * The counterpart to fs_bdev_freeze(): thaw each live superblock using @bdev.
+ * A zero return does not imply a superblock is fully unfrozen; it may have been
+ * frozen more than once (by the kernel or via another device).
*
- * Return: If the thaw was successful zero is returned. If the thaw
- * failed a negative error code is returned. If this function
- * returns zero it doesn't mean that the filesystem is unfrozen
- * as it may have been frozen multiple times (kernel may hold a
- * freeze or might be frozen from other block devices).
+ * Return: 0, or the first error on a single-fs device; a shared device swallows
+ * per-superblock errors, as fs_bdev_freeze() does.
*/
static int fs_bdev_thaw(struct block_device *bdev)
{
- struct super_block *sb;
- int error;
+ dev_t dev = bdev->bd_dev;
+ struct super_dev *sb_dev;
+ unsigned int count = 0;
+ int error = 0, err;
lockdep_assert_held(&bdev->bd_fsfreeze_mutex);
- /*
- * The block device may have been frozen before it was claimed by a
- * filesystem. Concurrently another process might try to mount that
- * frozen block device and has temporarily claimed the block device for
- * that purpose causing a concurrent fs_bdev_thaw() to end up here. The
- * mounter is already about to abort mounting because they still saw an
- * elevanted bdev->bd_fsfreeze_count so get_bdev_super() will return
- * NULL in that case.
- */
- sb = get_bdev_super(bdev);
- if (!sb)
- return -EINVAL;
+ mutex_unlock(&bdev->bd_holder_lock);
- if (sb->s_op->thaw_super)
- error = sb->s_op->thaw_super(sb,
- FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
- else
- error = thaw_super(sb,
- FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
- deactivate_super(sb);
+ for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
+ if (!get_active_super(sb_dev->sd_sb))
+ continue;
+ err = fs_super_thaw(sb_dev->sd_sb);
+ if (err && !error)
+ error = err;
+ deactivate_super(sb_dev->sd_sb);
+ count++;
+ }
+
+ /* Shared device: swallow per-superblock errors, like fs_bdev_freeze(). */
+ if (error && count > 1)
+ error = 0;
return error;
}
-const struct blk_holder_ops fs_holder_ops = {
+static const struct blk_holder_ops fs_holder_ops = {
.mark_dead = fs_bdev_mark_dead,
.sync = fs_bdev_sync,
.freeze = fs_bdev_freeze,
.thaw = fs_bdev_thaw,
};
-EXPORT_SYMBOL_GPL(fs_holder_ops);
+
+static struct super_dev *super_dev_lookup(dev_t dev, struct super_block *sb)
+{
+ struct super_dev *it;
+ struct rhlist_head *list, *pos;
+
+ RCU_LOCKDEP_WARN(!rcu_read_lock_held(), "suspicious super_dev_lookup() usage");
+ VFS_WARN_ON_ONCE(!dev);
+ VFS_WARN_ON_ONCE(!sb);
+
+ list = rhltable_lookup(&super_dev_table, &dev, super_dev_params);
+ rhl_for_each_entry_rcu(it, pos, list, sd_node) {
+ if (it->sd_sb == sb)
+ return it;
+ }
+
+ return NULL;
+}
+
+static int fs_bdev_register(struct file *bdev_file, struct super_block *sb)
+{
+ struct super_dev *sb_dev __free(kfree) = NULL;
+ dev_t dev = file_bdev(bdev_file)->bd_dev;
+ int err;
+
+ scoped_guard(rcu) {
+ sb_dev = super_dev_lookup(dev, sb);
+ if (sb_dev && refcount_inc_not_zero(&sb_dev->sd_ref)) {
+ retain_and_null_ptr(sb_dev);
+ return 0;
+ }
+ }
+
+ sb_dev = super_dev_alloc(dev, sb);
+ if (!sb_dev)
+ return -ENOMEM;
+
+ err = super_dev_insert(sb_dev);
+ if (err)
+ return err;
+
+ /* Publish the entry before reading the count; pairs with bdev_freeze(). */
+ smp_mb();
+ if (atomic_read(&file_bdev(bdev_file)->bd_fsfreeze_count) > 0) {
+ err = -EBUSY;
+ super_dev_put(sb_dev);
+ }
+
+ retain_and_null_ptr(sb_dev);
+ return err;
+}
+
+/**
+ * fs_bdev_file_open_by_dev - claim a block device on behalf of a superblock
+ * @dev: block device number
+ * @mode: open mode
+ * @holder: block-layer exclusivity token (a superblock, or the file_system_type
+ * when the device may be shared by several superblocks of that type)
+ * @sb: superblock to drive fs_holder_ops events for
+ *
+ * Open @dev with &fs_holder_ops and register that @sb uses it, so device
+ * removal/sync/freeze/thaw are propagated to @sb (and any other superblock
+ * sharing @dev). Must be paired with fs_bdev_file_release().
+ *
+ * Return: an opened block-device file or an ERR_PTR().
+ */
+struct file *fs_bdev_file_open_by_dev(dev_t dev, blk_mode_t mode, void *holder,
+ struct super_block *sb)
+{
+ struct file *bdev_file;
+ int err;
+
+ bdev_file = bdev_file_open_by_dev(dev, mode, holder, &fs_holder_ops);
+ if (IS_ERR(bdev_file))
+ return bdev_file;
+
+ err = fs_bdev_register(bdev_file, sb);
+ if (err) {
+ bdev_fput(bdev_file);
+ return ERR_PTR(err);
+ }
+ return bdev_file;
+}
+EXPORT_SYMBOL_GPL(fs_bdev_file_open_by_dev);
+
+/**
+ * fs_bdev_file_open_by_path - claim a block device on behalf of a superblock
+ * @path: path to the block device
+ * @mode: open mode
+ * @holder: block-layer exclusivity token (a superblock, or the file_system_type
+ * when the device may be shared by several superblocks of that type)
+ * @sb: superblock to drive fs_holder_ops events for
+ *
+ * Open the block device at @path with &fs_holder_ops and register that @sb
+ * uses it, so device removal/sync/freeze/thaw are propagated to @sb (and any
+ * other superblock sharing the device). Must be paired with
+ * fs_bdev_file_release().
+ *
+ * Return: an opened block-device file or an ERR_PTR().
+ */
+struct file *fs_bdev_file_open_by_path(const char *path, blk_mode_t mode,
+ void *holder, struct super_block *sb)
+{
+ struct file *bdev_file;
+ int err;
+
+ bdev_file = bdev_file_open_by_path(path, mode, holder, &fs_holder_ops);
+ if (IS_ERR(bdev_file))
+ return bdev_file;
+
+ err = fs_bdev_register(bdev_file, sb);
+ if (err) {
+ bdev_fput(bdev_file);
+ return ERR_PTR(err);
+ }
+ return bdev_file;
+}
+EXPORT_SYMBOL_GPL(fs_bdev_file_open_by_path);
+
+/**
+ * fs_bdev_unregister - drop a superblock's claim on a block device
+ * @bdev_file: file returned by fs_bdev_file_open_by_{dev,path}()
+ * @sb: superblock the device was claimed for
+ *
+ * The inverse of fs_bdev_register(): drop one claim on the {dev, @sb} entry
+ * (the last claim unregisters it; a pinning cursor defers the actual unlink)
+ * without closing the device. A caller that must act on the still-open device
+ * between unregistering and closing - e.g. re-allow freezing one denied for a
+ * membership change - pairs this with bdev_fput(). fs_bdev_file_release() is
+ * the common unregister-and-close.
+ */
+void fs_bdev_unregister(struct file *bdev_file, struct super_block *sb)
+{
+ dev_t dev = file_bdev(bdev_file)->bd_dev;
+ struct super_dev *sb_dev;
+
+ rcu_read_lock();
+ sb_dev = super_dev_lookup(dev, sb);
+ rcu_read_unlock();
+ super_dev_put(sb_dev);
+}
+EXPORT_SYMBOL_GPL(fs_bdev_unregister);
+
+/**
+ * fs_bdev_file_release - release a block device claimed for a superblock
+ * @bdev_file: file returned by fs_bdev_file_open_by_{dev,path}()
+ * @sb: superblock the device was claimed for
+ *
+ * Unregister the {dev, @sb} entry, then close the block device.
+ */
+void fs_bdev_file_release(struct file *bdev_file, struct super_block *sb)
+{
+ fs_bdev_unregister(bdev_file, sb);
+ bdev_fput(bdev_file);
+}
+EXPORT_SYMBOL_GPL(fs_bdev_file_release);
int setup_bdev_super(struct super_block *sb, int sb_flags,
struct fs_context *fc)
@@ -1555,7 +1803,7 @@ int setup_bdev_super(struct super_block *sb, int sb_flags,
struct file *bdev_file;
struct block_device *bdev;
- bdev_file = bdev_file_open_by_dev(sb->s_dev, mode, sb, &fs_holder_ops);
+ bdev_file = fs_bdev_file_open_by_dev(sb->s_dev, mode, sb, sb);
if (IS_ERR(bdev_file)) {
if (fc)
errorf(fc, "%s: Can't open blockdev", fc->source);
@@ -1569,20 +1817,19 @@ int setup_bdev_super(struct super_block *sb, int sb_flags,
* writable from userspace even for a read-only block device.
*/
if ((mode & BLK_OPEN_WRITE) && bdev_read_only(bdev)) {
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, sb);
return -EACCES;
}
- /*
- * It is enough to check bdev was not frozen before we set
- * s_bdev as freezing will wait until SB_BORN is set.
- */
+ /* The sget_fc() entry is already published; pairs with bdev_freeze(). */
+ smp_mb();
if (atomic_read(&bdev->bd_fsfreeze_count) > 0) {
if (fc)
warnf(fc, "%pg: Can't mount, blockdev is frozen", bdev);
- bdev_fput(bdev_file);
+ fs_bdev_file_release(bdev_file, sb);
return -EBUSY;
}
+
spin_lock(&sb_lock);
sb->s_bdev_file = bdev_file;
sb->s_bdev = bdev;
@@ -1671,7 +1918,7 @@ void kill_block_super(struct super_block *sb)
generic_shutdown_super(sb);
if (bdev) {
sync_blockdev(bdev);
- bdev_fput(sb->s_bdev_file);
+ fs_bdev_file_release(sb->s_bdev_file, sb);
}
}
diff --git a/fs/ubifs/dir.c b/fs/ubifs/dir.c
index 86d41e077e4d..23ec924162d6 100644
--- a/fs/ubifs/dir.c
+++ b/fs/ubifs/dir.c
@@ -303,7 +303,7 @@ static int ubifs_prepare_create(struct inode *dir, struct dentry *dentry,
}
static int ubifs_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode;
struct ubifs_info *c = dir->i_sb->s_fs_info;
@@ -1031,7 +1031,7 @@ static struct dentry *ubifs_mkdir(struct mnt_idmap *idmap, struct inode *dir,
sz_change = CALC_DENT_SIZE(fname_len(&nm));
- inode = ubifs_new_inode(c, dir, S_IFDIR | mode, false);
+ inode = ubifs_new_inode(c, dir, mode, false);
if (IS_ERR(inode)) {
err = PTR_ERR(inode);
goto out_fname;
diff --git a/fs/udf/balloc.c b/fs/udf/balloc.c
index cc6dc6e1d84d..30cec5600149 100644
--- a/fs/udf/balloc.c
+++ b/fs/udf/balloc.c
@@ -502,6 +502,8 @@ static int udf_table_prealloc_blocks(struct super_block *sb,
int8_t etype = -1;
struct udf_inode_info *iinfo;
int ret = 0;
+ /* AED block freed by udf_delete_aext(), released after unlock */
+ struct kernel_lb_addr freed = { .partitionReferenceNum = 0xFFFF };
if (first_block >= sbi->s_partmaps[partition].s_partition_len)
return 0;
@@ -541,7 +543,7 @@ static int udf_table_prealloc_blocks(struct super_block *sb,
udf_write_aext(table, &epos, &eloc,
(etype << 30) | elen, 1);
} else
- udf_delete_aext(table, epos);
+ udf_delete_aext(table, epos, &freed);
} else {
alloc_count = 0;
}
@@ -552,6 +554,8 @@ err_out:
if (alloc_count)
udf_add_free_space(sb, partition, -alloc_count);
mutex_unlock(&sbi->s_alloc_mutex);
+ if (freed.partitionReferenceNum != 0xFFFF)
+ udf_free_blocks(sb, table, &freed, 0, 1);
return alloc_count;
}
@@ -560,6 +564,8 @@ static udf_pblk_t udf_table_new_block(struct super_block *sb,
uint32_t goal, int *err)
{
struct udf_sb_info *sbi = UDF_SB(sb);
+ /* AED block freed by udf_delete_aext(), released after unlock */
+ struct kernel_lb_addr freed = { .partitionReferenceNum = 0xFFFF };
uint32_t spread = 0xFFFFFFFF, nspread = 0xFFFFFFFF;
udf_pblk_t newblock = 0;
uint32_t adsize;
@@ -643,12 +649,14 @@ static udf_pblk_t udf_table_new_block(struct super_block *sb,
if (goal_elen)
udf_write_aext(table, &goal_epos, &goal_eloc, goal_elen, 1);
else
- udf_delete_aext(table, goal_epos);
+ udf_delete_aext(table, goal_epos, &freed);
brelse(goal_epos.bh);
udf_add_free_space(sb, partition, -1);
mutex_unlock(&sbi->s_alloc_mutex);
+ if (freed.partitionReferenceNum != 0xFFFF)
+ udf_free_blocks(sb, table, &freed, 0, 1);
*err = 0;
return newblock;
}
diff --git a/fs/udf/inode.c b/fs/udf/inode.c
index 67bcf83758c8..c97914aa8d8b 100644
--- a/fs/udf/inode.c
+++ b/fs/udf/inode.c
@@ -1204,7 +1204,7 @@ static int udf_update_extents(struct inode *inode, struct kernel_long_ad *laarr,
if (startnum > endnum) {
for (i = 0; i < (startnum - endnum); i++)
- udf_delete_aext(inode, *epos);
+ udf_delete_aext(inode, *epos, NULL);
} else if (startnum < endnum) {
for (i = 0; i < (endnum - startnum); i++) {
err = udf_insert_aext(inode, *epos,
@@ -2299,6 +2299,13 @@ int udf_current_aext(struct inode *inode, struct extent_position *epos,
return -EINVAL;
}
+ if (eloc->partitionReferenceNum >= UDF_SB(inode->i_sb)->s_partitions) {
+ udf_debug("invalid partition reference %u (partitions %u)\n",
+ eloc->partitionReferenceNum,
+ UDF_SB(inode->i_sb)->s_partitions);
+ return -EFSCORRUPTED;
+ }
+
return 1;
}
@@ -2328,7 +2335,8 @@ static int udf_insert_aext(struct inode *inode, struct extent_position epos,
return ret;
}
-int8_t udf_delete_aext(struct inode *inode, struct extent_position epos)
+int8_t udf_delete_aext(struct inode *inode, struct extent_position epos,
+ struct kernel_lb_addr *freed)
{
struct extent_position oepos;
int adsize;
@@ -2378,7 +2386,19 @@ int8_t udf_delete_aext(struct inode *inode, struct extent_position epos)
elen = 0;
if (epos.bh != oepos.bh) {
- udf_free_blocks(inode->i_sb, inode, &epos.block, 0, 1);
+ /*
+ * The block that held the now-empty allocation extent must be
+ * returned to free space. When the caller already holds
+ * s_alloc_mutex (the space-table allocator in balloc.c),
+ * freeing it inline would recurse through udf_free_blocks()
+ * into udf_table_free_blocks() and deadlock re-acquiring
+ * s_alloc_mutex. In that case report the block to the caller,
+ * which frees it after dropping the lock.
+ */
+ if (freed)
+ *freed = epos.block;
+ else
+ udf_free_blocks(inode->i_sb, inode, &epos.block, 0, 1);
udf_write_aext(inode, &oepos, &eloc, elen, 1);
udf_write_aext(inode, &oepos, &eloc, elen, 1);
if (!oepos.bh) {
diff --git a/fs/udf/namei.c b/fs/udf/namei.c
index 9a3b7cef3606..b90841ac0a40 100644
--- a/fs/udf/namei.c
+++ b/fs/udf/namei.c
@@ -371,7 +371,7 @@ static int udf_add_nondir(struct dentry *dentry, struct inode *inode)
}
static int udf_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
struct inode *inode = udf_new_inode(dir, mode);
@@ -428,7 +428,7 @@ static struct dentry *udf_mkdir(struct mnt_idmap *idmap, struct inode *dir,
struct udf_inode_info *dinfo = UDF_I(dir);
struct udf_inode_info *iinfo;
- inode = udf_new_inode(dir, S_IFDIR | mode);
+ inode = udf_new_inode(dir, mode);
if (IS_ERR(inode))
return ERR_CAST(inode);
diff --git a/fs/udf/partition.c b/fs/udf/partition.c
index 2b85c9501bed..ad8dcedca263 100644
--- a/fs/udf/partition.c
+++ b/fs/udf/partition.c
@@ -55,7 +55,7 @@ uint32_t udf_get_pblock_virt15(struct super_block *sb, uint32_t block,
map = &sbi->s_partmaps[partition];
vdata = &map->s_type_specific.s_virtual;
- if (block > vdata->s_num_entries) {
+ if (block >= vdata->s_num_entries) {
udf_debug("Trying to access block beyond end of VAT (%u max %u)\n",
block, vdata->s_num_entries);
return 0xFFFFFFFF;
diff --git a/fs/udf/super.c b/fs/udf/super.c
index 7b85f5a2b79f..9686078bba64 100644
--- a/fs/udf/super.c
+++ b/fs/udf/super.c
@@ -2054,6 +2054,17 @@ static int udf_load_vrs(struct super_block *sb, struct udf_options *uopt,
return 0;
}
+static void udf_mark_buffer_dirty(struct buffer_head *bh)
+{
+ /*
+ * We set buffer uptodate unconditionally here to avoid spurious
+ * warnings from mark_buffer_dirty() when previous EIO has marked
+ * the buffer as !uptodate
+ */
+ set_buffer_uptodate(bh);
+ mark_buffer_dirty(bh);
+}
+
static void udf_finalize_lvid(struct logicalVolIntegrityDesc *lvid)
{
struct timespec64 ts;
@@ -2089,7 +2100,7 @@ static void udf_open_lvid(struct super_block *sb)
UDF_SET_FLAG(sb, UDF_FLAG_INCONSISTENT);
udf_finalize_lvid(lvid);
- mark_buffer_dirty(bh);
+ udf_mark_buffer_dirty(bh);
sbi->s_lvid_dirty = 0;
mutex_unlock(&sbi->s_alloc_mutex);
/* Make opening of filesystem visible on the media immediately */
@@ -2122,14 +2133,8 @@ static void udf_close_lvid(struct super_block *sb)
if (!UDF_QUERY_FLAG(sb, UDF_FLAG_INCONSISTENT))
lvid->integrityType = cpu_to_le32(LVID_INTEGRITY_TYPE_CLOSE);
- /*
- * We set buffer uptodate unconditionally here to avoid spurious
- * warnings from mark_buffer_dirty() when previous EIO has marked
- * the buffer as !uptodate
- */
- set_buffer_uptodate(bh);
udf_finalize_lvid(lvid);
- mark_buffer_dirty(bh);
+ udf_mark_buffer_dirty(bh);
sbi->s_lvid_dirty = 0;
mutex_unlock(&sbi->s_alloc_mutex);
/* Make closing of filesystem visible on the media immediately */
@@ -2411,7 +2416,7 @@ static int udf_sync_fs(struct super_block *sb, int wait)
* Blockdevice will be synced later so we don't have to submit
* the buffer for IO
*/
- mark_buffer_dirty(bh);
+ udf_mark_buffer_dirty(bh);
sbi->s_lvid_dirty = 0;
}
mutex_unlock(&sbi->s_alloc_mutex);
diff --git a/fs/udf/symlink.c b/fs/udf/symlink.c
index fe03745d09b1..a05d1888a2ba 100644
--- a/fs/udf/symlink.c
+++ b/fs/udf/symlink.c
@@ -36,6 +36,8 @@ static int udf_pc_to_char(struct super_block *sb, unsigned char *from,
/* Reserve one byte for terminating \0 */
tolen--;
while (elen < fromlen) {
+ if (fromlen - elen < sizeof(struct pathComponent))
+ return -EIO;
pc = (struct pathComponent *)(from + elen);
elen += sizeof(struct pathComponent);
switch (pc->componentType) {
diff --git a/fs/udf/truncate.c b/fs/udf/truncate.c
index 41b2bfd30449..0990f94b8551 100644
--- a/fs/udf/truncate.c
+++ b/fs/udf/truncate.c
@@ -159,7 +159,7 @@ void udf_discard_prealloc(struct inode *inode)
if (etype == (EXT_NOT_RECORDED_ALLOCATED >> 30)) {
lbcount -= elen;
- udf_delete_aext(inode, prev_epos);
+ udf_delete_aext(inode, prev_epos, NULL);
udf_free_blocks(inode->i_sb, inode, &eloc, 0,
DIV_ROUND_UP(elen, bsize));
}
diff --git a/fs/udf/udfdecl.h b/fs/udf/udfdecl.h
index 6d951e05c004..21de6925fc68 100644
--- a/fs/udf/udfdecl.h
+++ b/fs/udf/udfdecl.h
@@ -170,7 +170,8 @@ extern int udf_add_aext(struct inode *, struct extent_position *,
struct kernel_lb_addr *, uint32_t, int);
extern void udf_write_aext(struct inode *, struct extent_position *,
struct kernel_lb_addr *, uint32_t, int);
-extern int8_t udf_delete_aext(struct inode *, struct extent_position);
+extern int8_t udf_delete_aext(struct inode *, struct extent_position,
+ struct kernel_lb_addr *);
extern int udf_next_aext(struct inode *inode, struct extent_position *epos,
struct kernel_lb_addr *eloc, uint32_t *elen,
int8_t *etype, int inc);
diff --git a/fs/ufs/namei.c b/fs/ufs/namei.c
index 5b3c85c93242..6703f3bcf76f 100644
--- a/fs/ufs/namei.c
+++ b/fs/ufs/namei.c
@@ -70,8 +70,7 @@ static struct dentry *ufs_lookup(struct inode * dir, struct dentry *dentry, unsi
* with d_instantiate().
*/
static int ufs_create (struct mnt_idmap * idmap,
- struct inode * dir, struct dentry * dentry, umode_t mode,
- bool excl)
+ struct inode * dir, struct dentry * dentry, umode_t mode)
{
struct inode *inode;
@@ -174,7 +173,7 @@ static struct dentry *ufs_mkdir(struct mnt_idmap * idmap, struct inode * dir,
inode_inc_link_count(dir);
- inode = ufs_new_inode(dir, S_IFDIR|mode);
+ inode = ufs_new_inode(dir, mode);
err = PTR_ERR(inode);
if (IS_ERR(inode))
goto out_dir;
diff --git a/fs/ufs/super.c b/fs/ufs/super.c
index c4831a8b9b3f..6dcf6d048cce 100644
--- a/fs/ufs/super.c
+++ b/fs/ufs/super.c
@@ -672,7 +672,7 @@ void ufs_mark_sb_dirty(struct super_block *sb)
spin_lock(&sbi->work_lock);
if (!sbi->work_queued) {
delay = msecs_to_jiffies(dirty_writeback_interval * 10);
- queue_delayed_work(system_long_wq, &sbi->sync_work, delay);
+ queue_delayed_work(system_dfl_long_wq, &sbi->sync_work, delay);
sbi->work_queued = 1;
}
spin_unlock(&sbi->work_lock);
diff --git a/fs/vboxsf/dir.c b/fs/vboxsf/dir.c
index c5bd3271aa96..0b9eab157432 100644
--- a/fs/vboxsf/dir.c
+++ b/fs/vboxsf/dir.c
@@ -298,9 +298,9 @@ out:
static int vboxsf_dir_mkfile(struct mnt_idmap *idmap,
struct inode *parent, struct dentry *dentry,
- umode_t mode, bool excl)
+ umode_t mode)
{
- return vboxsf_dir_create(parent, dentry, mode, false, excl, NULL);
+ return vboxsf_dir_create(parent, dentry, mode, false, true, NULL);
}
static struct dentry *vboxsf_dir_mkdir(struct mnt_idmap *idmap,
diff --git a/fs/xfs/libxfs/xfs_da_btree.c b/fs/xfs/libxfs/xfs_da_btree.c
index 9debb95d86fa..f190c088591b 100644
--- a/fs/xfs/libxfs/xfs_da_btree.c
+++ b/fs/xfs/libxfs/xfs_da_btree.c
@@ -2354,8 +2354,7 @@ xfs_da_grow_inode_int(
* If we didn't get it and the block might work if fragmented,
* try without the CONTIG flag. Loop until we get it all.
*/
- mapp = kmalloc(sizeof(*mapp) * count,
- GFP_KERNEL | __GFP_NOFAIL);
+ mapp = kmalloc_objs(*mapp, count, GFP_KERNEL | __GFP_NOFAIL);
for (b = *bno, mapi = 0; b < *bno + count; ) {
c = (int)(*bno + count - b);
nmap = min(XFS_BMAP_MAX_NMAP, c);
diff --git a/fs/xfs/libxfs/xfs_dquot_buf.c b/fs/xfs/libxfs/xfs_dquot_buf.c
index bbada0d3cc08..f960474bed3d 100644
--- a/fs/xfs/libxfs/xfs_dquot_buf.c
+++ b/fs/xfs/libxfs/xfs_dquot_buf.c
@@ -416,6 +416,15 @@ xfs_dqinode_load(
return 0;
}
+static int
+xfs_dqinode_init(
+ struct xfs_metadir_update *upd,
+ void *priv)
+{
+ xfs_trans_log_inode(upd->tp, upd->ip, XFS_ILOG_CORE);
+ return 0;
+}
+
/* Create a metadata directory quota inode. */
int
xfs_dqinode_metadir_create(
@@ -428,35 +437,9 @@ xfs_dqinode_metadir_create(
.metafile_type = xfs_dqinode_metafile_type(type),
.path = xfs_dqinode_path(type),
};
- int error;
- error = xfs_metadir_start_create(&upd);
- if (error)
- return error;
-
- error = xfs_metadir_create(&upd, S_IFREG);
- if (error)
- goto out_cancel;
-
- xfs_trans_log_inode(upd.tp, upd.ip, XFS_ILOG_CORE);
-
- error = xfs_metadir_commit(&upd);
- if (error)
- goto out_irele;
-
- xfs_finish_inode_setup(upd.ip);
- *ipp = upd.ip;
- return 0;
-
-out_cancel:
- xfs_metadir_cancel(&upd, error);
-out_irele:
- /* Have to finish setting up the inode to ensure it's deleted. */
- if (upd.ip) {
- xfs_finish_inode_setup(upd.ip);
- xfs_irele(upd.ip);
- }
- return error;
+ return xfs_metadir_create_file(&upd, S_IFREG, xfs_dqinode_init, NULL,
+ ipp);
}
#ifndef __KERNEL__
diff --git a/fs/xfs/libxfs/xfs_metadir.c b/fs/xfs/libxfs/xfs_metadir.c
index 74c4596ee4cf..7c6b086b73db 100644
--- a/fs/xfs/libxfs/xfs_metadir.c
+++ b/fs/xfs/libxfs/xfs_metadir.c
@@ -182,7 +182,7 @@ xfs_metadir_teardown(
* Begin the process of creating a metadata file by allocating transactions
* and taking whatever resources we're going to need.
*/
-int
+static int
xfs_metadir_start_create(
struct xfs_metadir_update *upd)
{
@@ -236,7 +236,7 @@ out_teardown:
* a negative error code. If an inode is passed back, the caller must finish
* setting up the inode before releasing it.
*/
-int
+static int
xfs_metadir_create(
struct xfs_metadir_update *upd,
umode_t mode)
@@ -425,7 +425,7 @@ xfs_metadir_commit(
}
/* Cancel a metadir update and unlock/drop all resources. */
-void
+static void
xfs_metadir_cancel(
struct xfs_metadir_update *upd,
int error)
@@ -438,48 +438,64 @@ xfs_metadir_cancel(
xfs_metadir_teardown(upd, error);
}
-/* Create a metadata for the last component of the path. */
int
-xfs_metadir_mkdir(
- struct xfs_inode *dp,
- const char *path,
+xfs_metadir_create_file(
+ struct xfs_metadir_update *upd,
+ umode_t mode,
+ xfs_metadir_createfn create,
+ void *priv,
struct xfs_inode **ipp)
{
- struct xfs_metadir_update upd = {
- .dp = dp,
- .path = path,
- .metafile_type = XFS_METAFILE_DIR,
- };
int error;
- if (xfs_is_shutdown(dp->i_mount))
+ if (xfs_is_shutdown(upd->dp->i_mount))
return -EIO;
- /* Allocate a transaction to create the last directory. */
- error = xfs_metadir_start_create(&upd);
+ error = xfs_metadir_start_create(upd);
if (error)
return error;
- /* Create the subdirectory and take our reference. */
- error = xfs_metadir_create(&upd, S_IFDIR);
+ error = xfs_metadir_create(upd, mode);
if (error)
goto out_cancel;
- error = xfs_metadir_commit(&upd);
+ if (create) {
+ error = create(upd, priv);
+ if (error)
+ goto out_cancel;
+ }
+
+ error = xfs_metadir_commit(upd);
if (error)
goto out_irele;
- xfs_finish_inode_setup(upd.ip);
- *ipp = upd.ip;
+ xfs_finish_inode_setup(upd->ip);
+ *ipp = upd->ip;
return 0;
out_cancel:
- xfs_metadir_cancel(&upd, error);
+ xfs_metadir_cancel(upd, error);
out_irele:
/* Have to finish setting up the inode to ensure it's deleted. */
- if (upd.ip) {
- xfs_finish_inode_setup(upd.ip);
- xfs_irele(upd.ip);
+ if (upd->ip) {
+ xfs_finish_inode_setup(upd->ip);
+ xfs_irele(upd->ip);
}
return error;
}
+
+/* Create a metadata for the last component of the path. */
+int
+xfs_metadir_mkdir(
+ struct xfs_inode *dp,
+ const char *path,
+ struct xfs_inode **ipp)
+{
+ struct xfs_metadir_update upd = {
+ .dp = dp,
+ .path = path,
+ .metafile_type = XFS_METAFILE_DIR,
+ };
+
+ return xfs_metadir_create_file(&upd, S_IFDIR, NULL, NULL, ipp);
+}
diff --git a/fs/xfs/libxfs/xfs_metadir.h b/fs/xfs/libxfs/xfs_metadir.h
index bfecac7d3d14..e434b9d1c932 100644
--- a/fs/xfs/libxfs/xfs_metadir.h
+++ b/fs/xfs/libxfs/xfs_metadir.h
@@ -32,14 +32,16 @@ int xfs_metadir_load(struct xfs_trans *tp, struct xfs_inode *dp,
const char *path, enum xfs_metafile_type metafile_type,
struct xfs_inode **ipp);
-int xfs_metadir_start_create(struct xfs_metadir_update *upd);
-int xfs_metadir_create(struct xfs_metadir_update *upd, umode_t mode);
+typedef int (*xfs_metadir_createfn)(struct xfs_metadir_update *upd, void *priv);
+
+int xfs_metadir_create_file(struct xfs_metadir_update *upd, umode_t mode,
+ xfs_metadir_createfn create, void *priv,
+ struct xfs_inode **ipp);
int xfs_metadir_start_link(struct xfs_metadir_update *upd);
int xfs_metadir_link(struct xfs_metadir_update *upd);
int xfs_metadir_commit(struct xfs_metadir_update *upd);
-void xfs_metadir_cancel(struct xfs_metadir_update *upd, int error);
int xfs_metadir_mkdir(struct xfs_inode *dp, const char *path,
struct xfs_inode **ipp);
diff --git a/fs/xfs/libxfs/xfs_rtgroup.c b/fs/xfs/libxfs/xfs_rtgroup.c
index c85d50953218..fe7222bbe449 100644
--- a/fs/xfs/libxfs/xfs_rtgroup.c
+++ b/fs/xfs/libxfs/xfs_rtgroup.c
@@ -517,6 +517,25 @@ xfs_rtginode_irele(
*ipp = NULL;
}
+struct xfs_rtginode_create {
+ struct xfs_rtgroup *rtg;
+ enum xfs_rtg_inodes type;
+ bool init;
+};
+
+static int
+xfs_rtginode_init(
+ struct xfs_metadir_update *upd,
+ void *priv)
+{
+ struct xfs_rtginode_create *rc = priv;
+ const struct xfs_rtginode_ops *ops = &xfs_rtginode_ops[rc->type];
+
+ xfs_rtginode_lockdep_setup(upd->ip, rtg_rgno(rc->rtg), rc->type);
+ upd->ip->i_projid = rtg_rgno(rc->rtg);
+ return ops->create(rc->rtg, upd->ip, upd->tp, rc->init);
+}
+
/* Add a metadata inode for a realtime rmap btree. */
int
xfs_rtginode_create(
@@ -526,6 +545,11 @@ xfs_rtginode_create(
{
const struct xfs_rtginode_ops *ops = &xfs_rtginode_ops[type];
struct xfs_mount *mp = rtg_mount(rtg);
+ struct xfs_rtginode_create rc = {
+ .rtg = rtg,
+ .type = type,
+ .init = init,
+ };
struct xfs_metadir_update upd = {
.dp = mp->m_rtdirip,
.metafile_type = ops->metafile_type,
@@ -544,38 +568,8 @@ xfs_rtginode_create(
if (!upd.path)
return -ENOMEM;
- error = xfs_metadir_start_create(&upd);
- if (error)
- goto out_path;
-
- error = xfs_metadir_create(&upd, S_IFREG);
- if (error)
- goto out_cancel;
-
- xfs_rtginode_lockdep_setup(upd.ip, rtg_rgno(rtg), type);
-
- upd.ip->i_projid = rtg_rgno(rtg);
- error = ops->create(rtg, upd.ip, upd.tp, init);
- if (error)
- goto out_cancel;
-
- error = xfs_metadir_commit(&upd);
- if (error)
- goto out_path;
-
- kfree(upd.path);
- xfs_finish_inode_setup(upd.ip);
- rtg->rtg_inodes[type] = upd.ip;
- return 0;
-
-out_cancel:
- xfs_metadir_cancel(&upd, error);
- /* Have to finish setting up the inode to ensure it's deleted. */
- if (upd.ip) {
- xfs_finish_inode_setup(upd.ip);
- xfs_irele(upd.ip);
- }
-out_path:
+ error = xfs_metadir_create_file(&upd, S_IFREG, xfs_rtginode_init, &rc,
+ &rtg->rtg_inodes[type]);
kfree(upd.path);
return error;
}
diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c
index f27b80a199ba..22acc1411aac 100644
--- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c
+++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c
@@ -201,7 +201,7 @@ xfs_rtrefcountbt_verify(
if (fa)
return fa;
level = be16_to_cpu(block->bb_level);
- if (level > mp->m_rtrefc_maxlevels)
+ if (level >= mp->m_rtrefc_maxlevels)
return __this_address;
return xfs_btree_fsblock_verify(bp, mp->m_rtrefc_mxr[level != 0]);
@@ -651,7 +651,7 @@ xfs_iformat_rtrefcount(
numrecs = be16_to_cpu(dfp->bb_numrecs);
level = be16_to_cpu(dfp->bb_level);
- if (level > mp->m_rtrefc_maxlevels ||
+ if (level >= mp->m_rtrefc_maxlevels ||
xfs_rtrefcount_droot_space_calc(level, numrecs) > dsize) {
xfs_inode_mark_sick(ip, XFS_SICK_INO_CORE);
return -EFSCORRUPTED;
diff --git a/fs/xfs/scrub/agheader.c b/fs/xfs/scrub/agheader.c
index 9ed053b5f061..62ed5eaf08fb 100644
--- a/fs/xfs/scrub/agheader.c
+++ b/fs/xfs/scrub/agheader.c
@@ -266,7 +266,7 @@ xchk_superblock(
xchk_block_set_corrupt(sc, bp);
if (sb->sb_gquotino != cpu_to_be64(0))
- xchk_block_set_preen(sc, bp);
+ xchk_block_set_corrupt(sc, bp);
} else {
if (sb->sb_uquotino != cpu_to_be64(mp->m_sb.sb_uquotino))
xchk_block_set_preen(sc, bp);
diff --git a/fs/xfs/scrub/bmap.c b/fs/xfs/scrub/bmap.c
index 70028da1aacc..401c278725d2 100644
--- a/fs/xfs/scrub/bmap.c
+++ b/fs/xfs/scrub/bmap.c
@@ -1170,6 +1170,11 @@ xchk_bmap_attr(
}
error = xchk_bmap(sc, XFS_ATTR_FORK);
+ /* A repaired, empty attr fork no longer has mappings to check. */
+ if (error == -ENOENT && (sc->flags & XREP_ALREADY_FIXED)) {
+ xchk_mark_healthy_if_clean(sc, XFS_SICK_INO_BMBTA_ZAPPED);
+ return 0;
+ }
if (error)
return error;
diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c
index 3ec41c198351..b88427a4460c 100644
--- a/fs/xfs/scrub/inode_repair.c
+++ b/fs/xfs/scrub/inode_repair.c
@@ -1960,7 +1960,7 @@ xrep_inode_cowextsize(
/* Fix misaligned CoW extent size hints on a directory. */
if ((sc->ip->i_diflags & XFS_DIFLAG_RTINHERIT) &&
(sc->ip->i_diflags2 & XFS_DIFLAG2_COWEXTSIZE) &&
- sc->ip->i_extsize % sc->mp->m_sb.sb_rextsize > 0) {
+ xfs_extlen_to_rtxmod(sc->mp, sc->ip->i_cowextsize) > 0) {
sc->ip->i_cowextsize = 0;
sc->ip->i_diflags2 &= ~XFS_DIFLAG2_COWEXTSIZE;
}
diff --git a/fs/xfs/scrub/nlinks_repair.c b/fs/xfs/scrub/nlinks_repair.c
index fbc2ff809fc0..09e097e16689 100644
--- a/fs/xfs/scrub/nlinks_repair.c
+++ b/fs/xfs/scrub/nlinks_repair.c
@@ -232,9 +232,14 @@ xrep_nlinks_repair_inode(
* unlinked list, put it on the unlinked list.
*/
if (total_links == 0 && !xfs_inode_on_unlinked_list(ip)) {
+ if (actual_nlink)
+ clear_nlink(VFS_I(ip));
error = xfs_iunlink(sc->tp, ip);
- if (error)
+ if (error) {
+ if (actual_nlink)
+ set_nlink(VFS_I(ip), actual_nlink);
goto out_trans;
+ }
dirty = true;
}
diff --git a/fs/xfs/scrub/rtbitmap_repair.c b/fs/xfs/scrub/rtbitmap_repair.c
index dc64902d6c25..442a17bf9720 100644
--- a/fs/xfs/scrub/rtbitmap_repair.c
+++ b/fs/xfs/scrub/rtbitmap_repair.c
@@ -36,6 +36,24 @@
/* rt bitmap content repairs */
+/*
+ * Reserve enough blocks to write out a completely new bitmap file, plus twice
+ * as many blocks as we would need if we can only allocate one block per data
+ * fork mapping. This should cover the preallocation of the temporary file and
+ * exchanging the extent mappings.
+ *
+ * We cannot use xfs_exchmaps_estimate because we have not yet constructed the
+ * replacement bitmap and therefore do not know how many extents it will use.
+ * By the time we do, we will have a dirty transaction (which we cannot drop
+ * because we cannot drop the rtbitmap ILOCK) and cannot ask for more
+ * reservation.
+ */
+static inline unsigned long long
+xrep_rtbitmap_calc_blocks(struct xfs_mount *mp, unsigned long long blocks)
+{
+ return blocks + (xfs_bmbt_calc_size(mp, blocks) * 2);
+}
+
/* Set up to repair the realtime bitmap for this group. */
int
xrep_setup_rtbitmap(
@@ -56,20 +74,7 @@ xrep_setup_rtbitmap(
if (error)
return error;
- /*
- * Reserve enough blocks to write out a completely new bitmap file,
- * plus twice as many blocks as we would need if we can only allocate
- * one block per data fork mapping. This should cover the
- * preallocation of the temporary file and exchanging the extent
- * mappings.
- *
- * We cannot use xfs_exchmaps_estimate because we have not yet
- * constructed the replacement bitmap and therefore do not know how
- * many extents it will use. By the time we do, we will have a dirty
- * transaction (which we cannot drop because we cannot drop the
- * rtbitmap ILOCK) and cannot ask for more reservation.
- */
- blocks += xfs_bmbt_calc_size(mp, blocks) * 2;
+ blocks = xrep_rtbitmap_calc_blocks(mp, mp->m_sb.sb_rbmblocks);
if (blocks > UINT_MAX)
return -EOPNOTSUPP;
@@ -512,7 +517,7 @@ xrep_rtbitmap(
struct xchk_rtbitmap *rtb = sc->buf;
struct xfs_mount *mp = sc->mp;
struct xfs_group *xg = rtg_group(sc->sr.rtg);
- unsigned long long blocks = 0;
+ unsigned long long blocks;
unsigned int busy_gen;
int error;
@@ -532,15 +537,20 @@ xrep_rtbitmap(
* figure out if we need to adjust the block reservation in the
* transaction.
*/
- blocks = xfs_bmbt_calc_size(mp, rtb->rbmblocks);
+ blocks = xrep_rtbitmap_calc_blocks(mp, rtb->rbmblocks);
if (blocks > UINT_MAX)
return -EOPNOTSUPP;
if (blocks > rtb->resblks) {
- error = xfs_trans_reserve_more(sc->tp, blocks, 0);
+ uint64_t delta = blocks - rtb->resblks;
+
+ if (delta > UINT_MAX)
+ return -EOPNOTSUPP;
+
+ error = xfs_trans_reserve_more(sc->tp, delta, 0);
if (error)
return error;
- rtb->resblks += blocks;
+ rtb->resblks += delta;
}
/* Fix inode core and forks. */
diff --git a/fs/xfs/scrub/rtsummary.c b/fs/xfs/scrub/rtsummary.c
index 78f72a046887..546b335ade13 100644
--- a/fs/xfs/scrub/rtsummary.c
+++ b/fs/xfs/scrub/rtsummary.c
@@ -358,7 +358,7 @@ xchk_rtsummary(
* EFSCORRUPTED means the rtbitmap is corrupt, which is an xref
* error since we're checking the summary file.
*/
- xchk_ip_set_corrupt(sc, rbmip);
+ xchk_ip_xref_set_corrupt(sc, rbmip);
return 0;
}
if (error)
diff --git a/fs/xfs/xfs_bmap_util.c b/fs/xfs/xfs_bmap_util.c
index c88b9ade7389..268d159339d0 100644
--- a/fs/xfs/xfs_bmap_util.c
+++ b/fs/xfs/xfs_bmap_util.c
@@ -642,11 +642,26 @@ out_unlock:
return error;
}
+/*
+ * Allocate space or convert extents for a file according to @mode:
+ *
+ * XFS_ALLOC_FILE_SPACE_PREALLOC:
+ * Preallocate unwritten extents over holes across the range and mark the inode
+ * as preallocated.
+ *
+ * XFS_ALLOC_FILE_SPACE_WRITE_ZEROES:
+ * Allocate written extents over holes and convert unwritten extents in the
+ * range to written extents, initialising both to contain zeroes.
+ *
+ * This function does not update the file size; callers that extend the file
+ * are responsible for updating it once the extents are allocated.
+ */
int
xfs_alloc_file_space(
struct xfs_inode *ip,
xfs_off_t offset,
- xfs_off_t len)
+ xfs_off_t len,
+ enum xfs_alloc_file_space_mode mode)
{
xfs_mount_t *mp = ip->i_mount;
xfs_off_t count;
@@ -657,6 +672,7 @@ xfs_alloc_file_space(
int rt;
xfs_trans_t *tp;
xfs_bmbt_irec_t imaps[1], *imapp;
+ uint32_t bmapi_flags, nr_exts;
int error;
if (xfs_is_always_cow_inode(ip))
@@ -674,6 +690,19 @@ xfs_alloc_file_space(
if (len <= 0)
return -EINVAL;
+ switch (mode) {
+ case XFS_ALLOC_FILE_SPACE_PREALLOC:
+ bmapi_flags = XFS_BMAPI_PREALLOC;
+ nr_exts = XFS_IEXT_ADD_NOSPLIT_CNT;
+ break;
+ case XFS_ALLOC_FILE_SPACE_WRITE_ZEROES:
+ bmapi_flags = XFS_BMAPI_CONVERT | XFS_BMAPI_ZERO;
+ nr_exts = XFS_IEXT_WRITE_UNWRITTEN_CNT;
+ break;
+ default:
+ return -EINVAL;
+ }
+
rt = XFS_IS_REALTIME_INODE(ip);
extsz = xfs_get_extsz_hint(ip);
@@ -733,8 +762,7 @@ xfs_alloc_file_space(
if (error)
break;
- error = xfs_iext_count_extend(tp, ip, XFS_DATA_FORK,
- XFS_IEXT_ADD_NOSPLIT_CNT);
+ error = xfs_iext_count_extend(tp, ip, XFS_DATA_FORK, nr_exts);
if (error)
goto error;
@@ -748,7 +776,7 @@ xfs_alloc_file_space(
* will eventually reach the requested range.
*/
error = xfs_bmapi_write(tp, ip, startoffset_fsb,
- allocatesize_fsb, XFS_BMAPI_PREALLOC, 0, imapp,
+ allocatesize_fsb, bmapi_flags, 0, imapp,
&nimaps);
if (error) {
if (error != -ENOSR)
@@ -759,8 +787,10 @@ xfs_alloc_file_space(
allocatesize_fsb -= imapp->br_blockcount;
}
- ip->i_diflags |= XFS_DIFLAG_PREALLOC;
- xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
+ if (mode == XFS_ALLOC_FILE_SPACE_PREALLOC) {
+ ip->i_diflags |= XFS_DIFLAG_PREALLOC;
+ xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
+ }
error = xfs_trans_commit(tp);
xfs_iunlock(ip, XFS_ILOCK_EXCL);
diff --git a/fs/xfs/xfs_bmap_util.h b/fs/xfs/xfs_bmap_util.h
index eaaf094154b9..c7b48b2602f2 100644
--- a/fs/xfs/xfs_bmap_util.h
+++ b/fs/xfs/xfs_bmap_util.h
@@ -55,8 +55,13 @@ int xfs_bmap_last_extent(struct xfs_trans *tp, struct xfs_inode *ip,
int *is_empty);
/* preallocation and hole punch interface */
+enum xfs_alloc_file_space_mode {
+ XFS_ALLOC_FILE_SPACE_PREALLOC,
+ XFS_ALLOC_FILE_SPACE_WRITE_ZEROES,
+};
+
int xfs_alloc_file_space(struct xfs_inode *ip, xfs_off_t offset,
- xfs_off_t len);
+ xfs_off_t len, enum xfs_alloc_file_space_mode mode);
int xfs_free_file_space(struct xfs_inode *ip, xfs_off_t offset,
xfs_off_t len, struct xfs_zone_alloc_ctx *ac);
int xfs_collapse_file_space(struct xfs_inode *, xfs_off_t offset,
diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c
index e1465e950acc..21d67af781da 100644
--- a/fs/xfs/xfs_buf.c
+++ b/fs/xfs/xfs_buf.c
@@ -114,7 +114,7 @@ xfs_buf_free(
vfree(bp->b_addr);
else if (bp->b_flags & _XBF_KMEM)
kfree(bp->b_addr);
- else
+ else if (bp->b_addr)
folio_put(virt_to_folio(bp->b_addr));
call_rcu(&bp->b_rcu, xfs_buf_free_callback);
@@ -1631,7 +1631,7 @@ xfs_free_buftarg(
fs_put_dax(btp->bt_daxdev, btp->bt_mount);
/* the main block device is closed by kill_block_super */
if (btp->bt_bdev != btp->bt_mount->m_super->s_bdev)
- bdev_fput(btp->bt_file);
+ fs_bdev_file_release(btp->bt_file, btp->bt_mount->m_super);
kfree(btp);
}
diff --git a/fs/xfs/xfs_buf_item_recover.c b/fs/xfs/xfs_buf_item_recover.c
index 02b95b89d1b5..240deb3f7827 100644
--- a/fs/xfs/xfs_buf_item_recover.c
+++ b/fs/xfs/xfs_buf_item_recover.c
@@ -461,7 +461,7 @@ xlog_recover_validate_buf_type(
* given buffer. The bitmap in the buf log format structure indicates
* where to place the logged data.
*/
-STATIC void
+STATIC int
xlog_recover_do_reg_buffer(
struct xfs_mount *mp,
struct xlog_recover_item *item,
@@ -489,8 +489,24 @@ xlog_recover_do_reg_buffer(
ASSERT(nbits > 0);
ASSERT(item->ri_buf[i].iov_base != NULL);
ASSERT(item->ri_buf[i].iov_len % XFS_BLF_CHUNK == 0);
- ASSERT(BBTOB(bp->b_length) >=
- ((uint)bit << XFS_BLF_SHIFT) + (nbits << XFS_BLF_SHIFT));
+ /*
+ * The bitmap is only trustworthy to the extent that it
+ * describes a region that actually fits inside the buffer we
+ * read in based on the (attacker-controlled) blf_len. Do not
+ * rely on an ASSERT() for this -- it compiles away entirely on
+ * non-DEBUG kernels, which is exactly where this matters, so
+ * validate it for real and abort recovery of this buffer rather
+ * than copying past the end of it.
+ */
+ if (XFS_IS_CORRUPT(mp, BBTOB(bp->b_length) <
+ ((uint)bit << XFS_BLF_SHIFT) +
+ (nbits << XFS_BLF_SHIFT))) {
+ xfs_alert(mp,
+ "Bad buffer log item dirty bitmap (bit %d, nbits %d) for %d-byte buffer at daddr 0x%llx.",
+ bit, nbits, BBTOB(bp->b_length),
+ xfs_buf_daddr(bp));
+ return -EFSCORRUPTED;
+ }
/*
* The dirty regions logged in the buffer, even though
@@ -544,6 +560,7 @@ xlog_recover_do_reg_buffer(
ASSERT(i == item->ri_total);
xlog_recover_validate_buf_type(mp, bp, buf_f, current_lsn);
+ return 0;
}
/*
@@ -552,10 +569,10 @@ xlog_recover_do_reg_buffer(
* (ie. USR or GRP), then just toss this buffer away; don't recover it.
* Else, treat it as a regular buffer and do recovery.
*
- * Return false if the buffer was tossed and true if we recovered the buffer to
- * indicate to the caller if the buffer needs writing.
+ * Return 0 if the buffer was not recovered (tossed), 1 if it was recovered and
+ * needs writing, or a negative errno if recovery of the buffer failed.
*/
-STATIC bool
+STATIC int
xlog_recover_do_dquot_buffer(
struct xfs_mount *mp,
struct xlog *log,
@@ -564,6 +581,7 @@ xlog_recover_do_dquot_buffer(
struct xfs_buf_log_format *buf_f)
{
uint type;
+ int error;
trace_xfs_log_recover_buf_dquot_buf(log, buf_f);
@@ -571,7 +589,7 @@ xlog_recover_do_dquot_buffer(
* Filesystems are required to send in quota flags at mount time.
*/
if (!mp->m_qflags)
- return false;
+ return 0;
type = 0;
if (buf_f->blf_flags & XFS_BLF_UDQUOT_BUF)
@@ -584,10 +602,12 @@ xlog_recover_do_dquot_buffer(
* This type of quotas was turned off, so ignore this buffer
*/
if (log->l_quotaoffs_flag & type)
- return false;
+ return 0;
- xlog_recover_do_reg_buffer(mp, item, bp, buf_f, NULLCOMMITLSN);
- return true;
+ error = xlog_recover_do_reg_buffer(mp, item, bp, buf_f, NULLCOMMITLSN);
+ if (error)
+ return error;
+ return 1;
}
/*
@@ -724,7 +744,9 @@ xlog_recover_do_primary_sb_buffer(
xfs_rgnumber_t orig_rgcount = mp->m_sb.sb_rgcount;
int error;
- xlog_recover_do_reg_buffer(mp, item, bp, buf_f, current_lsn);
+ error = xlog_recover_do_reg_buffer(mp, item, bp, buf_f, current_lsn);
+ if (error)
+ return error;
if (orig_agcount == 0) {
xfs_alert(mp, "Trying to grow file system without AGs");
@@ -1081,11 +1103,11 @@ xlog_recover_buf_commit_pass2(
goto out_release;
} else if (buf_f->blf_flags &
(XFS_BLF_UDQUOT_BUF|XFS_BLF_PDQUOT_BUF|XFS_BLF_GDQUOT_BUF)) {
- bool dirty;
-
- dirty = xlog_recover_do_dquot_buffer(mp, log, item, bp, buf_f);
- if (!dirty)
+ error = xlog_recover_do_dquot_buffer(mp, log, item, bp, buf_f);
+ if (error <= 0)
goto out_release;
+ /* write dirty buffer */
+ error = 0;
} else if ((xfs_blft_from_flags(buf_f) & XFS_BLFT_SB_BUF) &&
xfs_buf_daddr(bp) == 0) {
error = xlog_recover_do_primary_sb_buffer(mp, item, bp, buf_f,
@@ -1105,7 +1127,10 @@ xlog_recover_buf_commit_pass2(
xfs_buf_relse(rtsb_bp);
}
} else {
- xlog_recover_do_reg_buffer(mp, item, bp, buf_f, current_lsn);
+ error = xlog_recover_do_reg_buffer(mp, item, bp, buf_f,
+ current_lsn);
+ if (error)
+ goto out_release;
}
/*
diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
index 845a97c9b063..0ade13b31335 100644
--- a/fs/xfs/xfs_file.c
+++ b/fs/xfs/xfs_file.c
@@ -1368,6 +1368,84 @@ xfs_falloc_force_zero(
return XFS_TEST_ERROR(ip->i_mount, XFS_ERRTAG_FORCE_ZERO_RANGE);
}
+static int
+xfs_falloc_write_zeroes(
+ struct file *file,
+ int mode,
+ loff_t offset,
+ loff_t len,
+ struct xfs_zone_alloc_ctx *ac)
+{
+ struct inode *inode = file_inode(file);
+ struct xfs_inode *ip = XFS_I(inode);
+ loff_t new_size = 0;
+ int error;
+
+ /*
+ * XXX: There is an issue with bigrtalloc inodes where there can be blocks
+ * that are written after the EOF block. This breaks the promise of no
+ * written blocks past EOF. Return EOPNOTSUPP until it is fixed.
+ */
+ if (xfs_is_always_cow_inode(ip) || xfs_inode_has_bigrtalloc(ip) ||
+ !bdev_write_zeroes_unmap_sectors(xfs_inode_buftarg(ip)->bt_bdev))
+ return -EOPNOTSUPP;
+
+ error = xfs_falloc_newsize(file, mode, offset, len, &new_size);
+ if (error)
+ return error;
+
+ /*
+ *
+ * |----------|----------|----------|----------|----------|
+ * ^ ^ ^ ^ ^ ^
+ * | | | | | |
+ * | offset | | end |
+ * | | | |
+ * offset_rd offset_ru end_rd end_ru
+ *
+ * xfs_free_file_space() punches the aligned interior offset_ru -> end_rd
+ * to holes and byte-zeroes the in-range parts of the partial edge blocks,
+ * offset -> offset_ru and end_rd -> end. xfs_zero_range() only touches
+ * already-written blocks here; it skips holes and unwritten extents, so
+ * unallocated/unwritten edge blocks are left for the allocation below.
+ */
+ error = xfs_free_file_space(ip, offset, len, ac);
+ if (error)
+ return error;
+
+ /*
+ * Publish the new size while the punched range is still a hole, then
+ * fill it with written zeroes. Like the other fallocate modes we use
+ * xfs_falloc_setsize(), but it must run *before* we convert the range
+ * to written extents: xfs_setattr_size() zeroes [old EOF, new size) via
+ * xfs_zero_range(), which skips holes, so there is nothing to re-zero.
+ * It will also writeback partial EOF block before the on-disk size is
+ * logged.
+ * Note: extending the size before allocating means a failure below
+ * leaves the file larger with unallocated holes in the new range.
+ * That is safe as holes within i_size read back as zeroes and expose
+ * no stale data while the error is propagated to the caller.
+ */
+ error = xfs_falloc_setsize(file, new_size);
+ if (error)
+ return error;
+
+ /*
+ * Allocate written, zeroed extents across the range. xfs_alloc_file_space()
+ * rounds outward to block granularity:
+ * - holes (the punched interior and any unallocated edge block) are
+ * allocated and zeroed;
+ * - unwritten extents (including unwritten edge blocks) are converted to
+ * written and zeroed;
+ * - Already written edge blocks are skipped. The out-of-range bytes of
+ * a written edge block keep their data (offset_rd -> offset and
+ * end -> end_rd); their in-range bytes (offset -> offset_ru and
+ * end_ru -> end were already zeroed by xfs_free_file_space().
+ */
+ return xfs_alloc_file_space(ip, offset, len,
+ XFS_ALLOC_FILE_SPACE_WRITE_ZEROES);
+}
+
/*
* Punch a hole and prealloc the range. We use a hole punch rather than
* unwritten extent conversion for two reasons:
@@ -1406,7 +1484,8 @@ xfs_falloc_zero_range(
len = round_up(offset + len, blksize) -
round_down(offset, blksize);
offset = round_down(offset, blksize);
- error = xfs_alloc_file_space(ip, offset, len);
+ error = xfs_alloc_file_space(ip, offset, len,
+ XFS_ALLOC_FILE_SPACE_PREALLOC);
}
if (error)
return error;
@@ -1432,7 +1511,8 @@ xfs_falloc_unshare_range(
if (error)
return error;
- error = xfs_alloc_file_space(XFS_I(inode), offset, len);
+ error = xfs_alloc_file_space(XFS_I(inode), offset, len,
+ XFS_ALLOC_FILE_SPACE_PREALLOC);
if (error)
return error;
return xfs_falloc_setsize(file, new_size);
@@ -1460,7 +1540,8 @@ xfs_falloc_allocate_range(
if (error)
return error;
- error = xfs_alloc_file_space(XFS_I(inode), offset, len);
+ error = xfs_alloc_file_space(XFS_I(inode), offset, len,
+ XFS_ALLOC_FILE_SPACE_PREALLOC);
if (error)
return error;
return xfs_falloc_setsize(file, new_size);
@@ -1470,7 +1551,7 @@ xfs_falloc_allocate_range(
(FALLOC_FL_ALLOCATE_RANGE | FALLOC_FL_KEEP_SIZE | \
FALLOC_FL_PUNCH_HOLE | FALLOC_FL_COLLAPSE_RANGE | \
FALLOC_FL_ZERO_RANGE | FALLOC_FL_INSERT_RANGE | \
- FALLOC_FL_UNSHARE_RANGE)
+ FALLOC_FL_UNSHARE_RANGE | FALLOC_FL_WRITE_ZEROES)
STATIC long
__xfs_file_fallocate(
@@ -1522,6 +1603,9 @@ __xfs_file_fallocate(
case FALLOC_FL_ALLOCATE_RANGE:
error = xfs_falloc_allocate_range(file, mode, offset, len);
break;
+ case FALLOC_FL_WRITE_ZEROES:
+ error = xfs_falloc_write_zeroes(file, mode, offset, len, ac);
+ break;
default:
error = -EOPNOTSUPP;
break;
diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c
index 6339f4956ecb..4a3299abf774 100644
--- a/fs/xfs/xfs_iops.c
+++ b/fs/xfs/xfs_iops.c
@@ -293,8 +293,7 @@ xfs_vn_create(
struct mnt_idmap *idmap,
struct inode *dir,
struct dentry *dentry,
- umode_t mode,
- bool flags)
+ umode_t mode)
{
return xfs_generic_create(idmap, dir, dentry, mode, 0, NULL);
}
@@ -306,7 +305,7 @@ xfs_vn_mkdir(
struct dentry *dentry,
umode_t mode)
{
- return ERR_PTR(xfs_generic_create(idmap, dir, dentry, mode | S_IFDIR, 0, NULL));
+ return ERR_PTR(xfs_generic_create(idmap, dir, dentry, mode, 0, NULL));
}
STATIC struct dentry *
@@ -338,7 +337,7 @@ STATIC struct dentry *
xfs_vn_ci_lookup(
struct inode *dir,
struct dentry *dentry,
- unsigned int flags)
+ unsigned int flags)
{
struct xfs_inode *ip;
struct xfs_name xname;
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 66a02d1b9ad7..216a38a354e7 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -349,6 +349,13 @@ typedef struct xfs_mount {
/* Index of uuid record in the uuid xarray. */
unsigned int m_uuid_table_index;
+
+ /*
+ * Old io_pages/ra_pages valued in the main bdev BDI, and our initial
+ * calculated values.
+ */
+ unsigned long m_old_io_pages, m_initial_io_pages;
+ unsigned long m_old_ra_pages, m_initial_ra_pages;
} xfs_mount_t;
#define M_IGEO(mp) (&(mp)->m_ino_geo)
diff --git a/fs/xfs/xfs_rtalloc.c b/fs/xfs/xfs_rtalloc.c
index 7a3f97686989..84efe5a8fb11 100644
--- a/fs/xfs/xfs_rtalloc.c
+++ b/fs/xfs/xfs_rtalloc.c
@@ -737,7 +737,7 @@ xfs_rtginode_ensure(
xfs_trans_cancel(tp);
if (error != -ENOENT)
- return 0;
+ return error;
return xfs_rtginode_create(rtg, type, true);
}
diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
index 8531d526fc44..4b2eeb7783f7 100644
--- a/fs/xfs/xfs_super.c
+++ b/fs/xfs/xfs_super.c
@@ -400,8 +400,8 @@ xfs_blkdev_get(
blk_mode_t mode;
mode = sb_open_mode(mp->m_super->s_flags);
- *bdev_filep = bdev_file_open_by_path(name, mode,
- mp->m_super, &fs_holder_ops);
+ *bdev_filep = fs_bdev_file_open_by_path(name, mode,
+ mp->m_super, mp->m_super);
if (IS_ERR(*bdev_filep)) {
error = PTR_ERR(*bdev_filep);
*bdev_filep = NULL;
@@ -526,7 +526,7 @@ xfs_open_devices(
mp->m_logdev_targp = mp->m_ddev_targp;
/* Handle won't be used, drop it */
if (logdev_file)
- bdev_fput(logdev_file);
+ fs_bdev_file_release(logdev_file, mp->m_super);
}
return 0;
@@ -541,14 +541,60 @@ xfs_open_devices(
mp->m_ddev_targp = NULL;
out_close_rtdev:
if (rtdev_file)
- bdev_fput(rtdev_file);
+ fs_bdev_file_release(rtdev_file, mp->m_super);
out_close_logdev:
if (logdev_file)
- bdev_fput(logdev_file);
+ fs_bdev_file_release(logdev_file, mp->m_super);
return error;
}
/*
+ * When using a RT device some or all data I/O is using the RT device, but
+ * the BDI is inherited from the main data device. When the underlying block
+ * device for the RT device has larger I/O sizes, the BDI settings might be
+ * incorrect, which is especially bad if the main device is a SSD and the
+ * RT device is a HDD, as the io_opt fixup in blk_apply_bdi_limits is missing
+ * for this case.
+ *
+ * Update the BDI values to the max of the data and RT device to cover our
+ * bases.
+ */
+static void
+xfs_update_bdi_rahead(
+ struct xfs_mount *mp)
+{
+ struct backing_dev_info *rt_bdi =
+ mp->m_rtdev_targp->bt_bdev->bd_disk->bdi;
+ struct backing_dev_info *sb_bdi = mp->m_super->s_bdi;
+
+ mp->m_old_io_pages = sb_bdi->io_pages;
+ mp->m_old_ra_pages = sb_bdi->ra_pages;
+
+ sb_bdi->io_pages = mp->m_initial_io_pages =
+ max(sb_bdi->io_pages, rt_bdi->io_pages);
+ sb_bdi->ra_pages = mp->m_initial_ra_pages =
+ max(sb_bdi->ra_pages, rt_bdi->ra_pages);
+}
+
+static void
+xfs_restore_bdi_rahead(
+ struct xfs_mount *mp)
+{
+ struct backing_dev_info *sb_bdi = mp->m_super->s_bdi;
+
+ if (sb_bdi->io_pages == mp->m_initial_io_pages)
+ sb_bdi->io_pages = mp->m_old_io_pages;
+ else
+ xfs_info(mp, "io_pages changed from %lu to %lu, not restoring.",
+ mp->m_initial_io_pages, sb_bdi->io_pages);
+ if (sb_bdi->ra_pages == mp->m_initial_ra_pages)
+ sb_bdi->ra_pages = mp->m_old_ra_pages;
+ else
+ xfs_info(mp, "ra_pages changed from %lu to %lu, not restoring.",
+ mp->m_initial_ra_pages, sb_bdi->ra_pages);
+}
+
+/*
* Setup xfs_mount buffer target pointers based on superblock
*/
STATIC int
@@ -585,6 +631,7 @@ xfs_setup_devices(
mp->m_sb.sb_sectsize, mp->m_sb.sb_rblocks);
if (error)
return error;
+ xfs_update_bdi_rahead(mp);
}
return 0;
@@ -2283,8 +2330,12 @@ static void
xfs_kill_sb(
struct super_block *sb)
{
+ struct xfs_mount *mp = XFS_M(sb);
+
+ if (mp->m_rtdev_targp && mp->m_rtdev_targp != mp->m_ddev_targp)
+ xfs_restore_bdi_rahead(mp);
kill_block_super(sb);
- xfs_mount_free(XFS_M(sb));
+ xfs_mount_free(mp);
}
static struct file_system_type xfs_fs_type = {
diff --git a/include/linux/binfmt_misc.h b/include/linux/binfmt_misc.h
new file mode 100644
index 000000000000..d3112a00cc19
--- /dev/null
+++ b/include/linux/binfmt_misc.h
@@ -0,0 +1,71 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef _LINUX_BINFMT_MISC_H
+#define _LINUX_BINFMT_MISC_H
+
+#include <linux/types.h>
+
+struct bpf_prog;
+struct linux_binprm;
+struct user_namespace;
+
+#define BINFMT_MISC_OPS_NAME_MAX 16
+
+/**
+ * enum bpf_binprm_flags - per-exec invocation flags a load program can request
+ * @BPF_BINPRM_PRESERVE_ARGV0: keep the caller's argv[0] (like the 'P' flag)
+ * @BPF_BINPRM_CREDENTIALS: compute credentials from the binary; implies execfd
+ * (like the 'C' flag)
+ * @BPF_BINPRM_EXECFD: pass the binary via AT_EXECFD (like the 'O' flag)
+ *
+ * Set from a load program with bpf_binprm_set_flags(). Unlike a static entry,
+ * a bpf handler chooses these per exec rather than once at registration.
+ */
+enum bpf_binprm_flags {
+ BPF_BINPRM_PRESERVE_ARGV0 = (1ULL << 0),
+ BPF_BINPRM_CREDENTIALS = (1ULL << 1),
+ BPF_BINPRM_EXECFD = (1ULL << 2),
+};
+
+/**
+ * struct binfmt_misc_ops - bpf-backed binary type handler
+ * @match: decide whether the handler applies to @bprm; consulted from the
+ * entry lookup walk like static magic and extension matching, in
+ * registration order with first-match-wins semantics; sleepable,
+ * so it can read the binary to decide, but the verifier rejects
+ * the interpreter selection kfuncs in it
+ * @load: select an interpreter for the matched @bprm via
+ * bpf_binprm_set_interp() and return zero; a match is committed, so
+ * a failure fails the exec instead of falling through to later
+ * entries; -ENOEXEC does not fail the exec but moves on to the
+ * remaining binary formats
+ * @name: name that 'B' entries reference the handler by
+ */
+struct binfmt_misc_ops {
+ bool (*match)(struct linux_binprm *bprm);
+ int (*load)(struct linux_binprm *bprm);
+ char name[BINFMT_MISC_OPS_NAME_MAX];
+};
+
+#ifdef CONFIG_BINFMT_MISC_BPF
+const struct binfmt_misc_ops *binfmt_misc_get_ops(struct user_namespace *user_ns,
+ const char *name);
+void binfmt_misc_put_ops(const struct binfmt_misc_ops *ops);
+bool bpf_prog_is_binfmt_misc_ops(const struct bpf_prog *prog);
+#else
+static inline const struct binfmt_misc_ops *
+binfmt_misc_get_ops(struct user_namespace *user_ns, const char *name)
+{
+ return NULL;
+}
+
+static inline void binfmt_misc_put_ops(const struct binfmt_misc_ops *ops)
+{
+}
+
+static inline bool bpf_prog_is_binfmt_misc_ops(const struct bpf_prog *prog)
+{
+ return false;
+}
+#endif /* CONFIG_BINFMT_MISC_BPF */
+
+#endif /* _LINUX_BINFMT_MISC_H */
diff --git a/include/linux/binfmts.h b/include/linux/binfmts.h
index 2c77e383e737..03e1794b5cbb 100644
--- a/include/linux/binfmts.h
+++ b/include/linux/binfmts.h
@@ -12,6 +12,13 @@ struct coredump_params;
#define CORENAME_MAX_SIZE 128
+/* Interpreter selection staged by a bpf binfmt_misc handler. */
+struct binfmt_misc_bpf {
+ const char *bpf_interp; /* interpreter selected by a bpf handler */
+ const char *bpf_interp_arg; /* interpreter argument from a bpf handler */
+ u64 bpf_flags; /* enum bpf_binprm_flags from a bpf handler */
+};
+
/*
* This structure is used to hold the arguments that are used when loading binaries.
*/
@@ -65,6 +72,7 @@ struct linux_binprm {
of the time same as filename, but could be
different for binfmt_{misc,script} */
const char *fdpath; /* generated filename for execveat */
+ struct binfmt_misc_bpf; /* bpf handler interpreter selection */
unsigned interp_flags;
int execfd; /* File descriptor of the executable */
unsigned long exec;
@@ -101,8 +109,8 @@ struct linux_binfmt {
#if IS_ENABLED(CONFIG_BINFMT_MISC)
struct binfmt_misc {
- struct list_head entries;
- rwlock_t entries_lock;
+ struct hlist_head entries;
+ spinlock_t entries_lock;
bool enabled;
} __randomize_layout;
diff --git a/include/linux/blk-crypto.h b/include/linux/blk-crypto.h
index f7c3cb4a342f..938ff536838c 100644
--- a/include/linux/blk-crypto.h
+++ b/include/linux/blk-crypto.h
@@ -68,6 +68,15 @@ enum blk_crypto_key_type {
*/
#define BLK_CRYPTO_SW_SECRET_SIZE 32
+/* Flags for blk_crypto_config::flags: */
+
+/*
+ * If set, inline encryption hardware will be used if available.
+ * If unset, CPU-based encryption will always be used (requires
+ * CONFIG_BLK_INLINE_ENCRYPTION_FALLBACK)
+ */
+#define BLK_CRYPTO_CFG_ALLOW_HW (1 << 0)
+
/**
* struct blk_crypto_config - an inline encryption key's crypto configuration
* @crypto_mode: encryption algorithm this key is for
@@ -77,12 +86,14 @@ enum blk_crypto_key_type {
* filesystem block size or the disk sector size.
* @dun_bytes: the maximum number of bytes of DUN used when using this key
* @key_type: the type of this key -- either raw or hardware-wrapped
+ * @flags: BLK_CRYPTO_CFG_* flags
*/
struct blk_crypto_config {
enum blk_crypto_mode_num crypto_mode;
unsigned int data_unit_size;
unsigned int dun_bytes;
enum blk_crypto_key_type key_type;
+ int flags;
};
/**
@@ -150,7 +161,7 @@ int blk_crypto_init_key(struct blk_crypto_key *blk_key,
enum blk_crypto_key_type key_type,
enum blk_crypto_mode_num crypto_mode,
unsigned int dun_bytes,
- unsigned int data_unit_size);
+ unsigned int data_unit_size, int flags);
int blk_crypto_start_using_key(struct block_device *bdev,
const struct blk_crypto_key *key);
@@ -160,8 +171,6 @@ void blk_crypto_evict_key(struct block_device *bdev,
bool blk_crypto_config_supported_natively(struct block_device *bdev,
const struct blk_crypto_config *cfg);
-bool blk_crypto_config_supported(struct block_device *bdev,
- const struct blk_crypto_config *cfg);
int blk_crypto_derive_sw_secret(struct block_device *bdev,
const u8 *eph_key, size_t eph_key_size,
diff --git a/include/linux/blk_types.h b/include/linux/blk_types.h
index 8808ee76e73c..5a725a0cd35f 100644
--- a/include/linux/blk_types.h
+++ b/include/linux/blk_types.h
@@ -66,7 +66,7 @@ struct block_device {
int bd_holders;
struct kobject *bd_holder_dir;
- atomic_t bd_fsfreeze_count; /* number of freeze requests */
+ atomic_t bd_fsfreeze_count; /* >0 freeze requests, <0 freeze deniers */
struct mutex bd_fsfreeze_mutex; /* serialize freeze/thaw */
struct partition_meta_info *bd_meta_info;
diff --git a/include/linux/blkdev.h b/include/linux/blkdev.h
index 9213a5716f95..dbb549cdfb77 100644
--- a/include/linux/blkdev.h
+++ b/include/linux/blkdev.h
@@ -126,8 +126,6 @@ struct blk_integrity {
unsigned char pi_tuple_size;
};
-typedef unsigned int __bitwise blk_mode_t;
-
/* open for reading */
#define BLK_OPEN_READ ((__force blk_mode_t)(1 << 0))
/* open for writing */
@@ -1771,13 +1769,6 @@ struct blk_holder_ops {
};
/*
- * For filesystems using @fs_holder_ops, the @holder argument passed to
- * helpers used to open and claim block devices via
- * bd_prepare_to_claim() must point to a superblock.
- */
-extern const struct blk_holder_ops fs_holder_ops;
-
-/*
* Return the correct open flags for blkdev_get_by_* for super block flags
* as stored in sb->s_flags.
*/
@@ -1837,7 +1828,10 @@ static inline int early_lookup_bdev(const char *pathname, dev_t *dev)
int bdev_freeze(struct block_device *bdev);
int bdev_thaw(struct block_device *bdev);
+int bdev_deny_freeze(struct block_device *bdev);
+void bdev_allow_freeze(struct block_device *bdev);
void bdev_fput(struct file *bdev_file);
+void bdev_yield_claim(struct file *bdev_file);
struct io_comp_batch {
struct rq_list req_list;
diff --git a/include/linux/efs_vh.h b/include/linux/efs_vh.h
deleted file mode 100644
index 206c5270f7b8..000000000000
--- a/include/linux/efs_vh.h
+++ /dev/null
@@ -1,54 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 */
-/*
- * efs_vh.h
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from IRIX header files (c) 1985 MIPS Computer Systems, Inc.
- */
-
-#ifndef __EFS_VH_H__
-#define __EFS_VH_H__
-
-#define VHMAGIC 0xbe5a941 /* volume header magic number */
-#define NPARTAB 16 /* 16 unix partitions */
-#define NVDIR 15 /* max of 15 directory entries */
-#define BFNAMESIZE 16 /* max 16 chars in boot file name */
-#define VDNAMESIZE 8
-
-struct volume_directory {
- char vd_name[VDNAMESIZE]; /* name */
- __be32 vd_lbn; /* logical block number */
- __be32 vd_nbytes; /* file length in bytes */
-};
-
-struct partition_table { /* one per logical partition */
- __be32 pt_nblks; /* # of logical blks in partition */
- __be32 pt_firstlbn; /* first lbn of partition */
- __be32 pt_type; /* use of partition */
-};
-
-struct volume_header {
- __be32 vh_magic; /* identifies volume header */
- __be16 vh_rootpt; /* root partition number */
- __be16 vh_swappt; /* swap partition number */
- char vh_bootfile[BFNAMESIZE]; /* name of file to boot */
- char pad[48]; /* device param space */
- struct volume_directory vh_vd[NVDIR]; /* other vol hdr contents */
- struct partition_table vh_pt[NPARTAB]; /* device partition layout */
- __be32 vh_csum; /* volume header checksum */
- __be32 vh_fill; /* fill out to 512 bytes */
-};
-
-/* partition type sysv is used for EFS format CD-ROM partitions */
-#define SGI_SYSV 0x05
-#define SGI_EFS 0x07
-#define IS_EFS(x) (((x) == SGI_EFS) || ((x) == SGI_SYSV))
-
-struct pt_types {
- int pt_type;
- char *pt_name;
-};
-
-#endif /* __EFS_VH_H__ */
-
diff --git a/include/linux/fs.h b/include/linux/fs.h
index 50ce731a2b78..f6942dc4fd4c 100644
--- a/include/linux/fs.h
+++ b/include/linux/fs.h
@@ -1916,8 +1916,6 @@ struct dir_context {
struct io_uring_cmd;
struct offset_ctx;
-typedef unsigned int __bitwise fop_flags_t;
-
struct file_operations {
struct module *owner;
fop_flags_t fop_flags;
@@ -2002,7 +2000,7 @@ struct inode_operations {
int (*readlink) (struct dentry *, char __user *,int);
int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,
- umode_t, bool);
+ umode_t);
int (*link) (struct dentry *,struct inode *,struct dentry *);
int (*unlink) (struct inode *,struct dentry *);
int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,
@@ -2413,6 +2411,21 @@ static inline void super_set_sysfs_name_generic(struct super_block *sb, const ch
extern void ihold(struct inode * inode);
extern void iput(struct inode *);
void iput_not_last(struct inode *);
+
+/**
+ * iput_if_not_last - drop an inode reference only if it is not the last one
+ * @inode: inode to put
+ *
+ * Returns true if the reference was dropped, false if this was the last
+ * reference and the caller must arrange for final iput() in a safe context.
+ */
+static inline bool __must_check iput_if_not_last(struct inode *inode)
+{
+ VFS_BUG_ON_INODE(inode_state_read_once(inode) & (I_FREEING | I_CLEAR), inode);
+ VFS_BUG_ON_INODE(icount_read_once(inode) < 1, inode);
+ return atomic_add_unless(&inode->i_count, -1, 1);
+}
+
int inode_update_time(struct inode *inode, enum fs_update_time type,
unsigned int flags);
int generic_update_time(struct inode *inode, enum fs_update_time type,
diff --git a/include/linux/fs/super.h b/include/linux/fs/super.h
index 405612678115..733d439f01ed 100644
--- a/include/linux/fs/super.h
+++ b/include/linux/fs/super.h
@@ -237,4 +237,12 @@ int thaw_super(struct super_block *super, enum freeze_holder who,
int sb_init_dio_done_wq(struct super_block *sb);
+struct file;
+struct file *fs_bdev_file_open_by_dev(dev_t dev, blk_mode_t mode, void *holder,
+ struct super_block *sb);
+struct file *fs_bdev_file_open_by_path(const char *path, blk_mode_t mode,
+ void *holder, struct super_block *sb);
+void fs_bdev_unregister(struct file *bdev_file, struct super_block *sb);
+void fs_bdev_file_release(struct file *bdev_file, struct super_block *sb);
+
#endif /* _LINUX_FS_SUPER_H */
diff --git a/include/linux/fs/super_types.h b/include/linux/fs/super_types.h
index ef7941e9dc79..a8cbafdeb36c 100644
--- a/include/linux/fs/super_types.h
+++ b/include/linux/fs/super_types.h
@@ -30,6 +30,7 @@ struct mount;
struct mtd_info;
struct quotactl_ops;
struct shrinker;
+struct super_dev;
struct unicode_map;
struct user_namespace;
struct workqueue_struct;
@@ -132,6 +133,7 @@ struct super_operations {
struct super_block {
struct list_head s_list; /* Keep this first */
dev_t s_dev; /* search index; _not_ kdev_t */
+ struct super_dev *s_super_dev; /* sget_fc()'s device table claim */
unsigned char s_blocksize_bits;
unsigned long s_blocksize;
loff_t s_maxbytes; /* Max file size */
@@ -145,7 +147,7 @@ struct super_block {
unsigned long s_magic;
struct dentry *s_root;
struct rw_semaphore s_umount;
- int s_count;
+ refcount_t s_passive;
atomic_t s_active;
#ifdef CONFIG_SECURITY
void *s_security;
@@ -300,7 +302,7 @@ struct super_block {
#define SB_NODIRATIME BIT(11) /* Do not update directory access times */
#define SB_SILENT BIT(15)
#define SB_POSIXACL BIT(16) /* Supports POSIX ACLs */
-#define SB_INLINECRYPT BIT(17) /* Use blk-crypto for encrypted files */
+#define SB_INLINECRYPT BIT(17) /* Use inline crypto hardware if available */
#define SB_KERNMOUNT BIT(22) /* this is a kern_mount call */
#define SB_I_VERSION BIT(23) /* Update inode I_version field */
#define SB_LAZYTIME BIT(25) /* Update the on-disk [acm]times lazily */
diff --git a/include/linux/fs_struct.h b/include/linux/fs_struct.h
index 0070764b790a..97eef8d3863d 100644
--- a/include/linux/fs_struct.h
+++ b/include/linux/fs_struct.h
@@ -6,6 +6,7 @@
#include <linux/path.h>
#include <linux/spinlock.h>
#include <linux/seqlock.h>
+#include <linux/vfsdebug.h>
struct fs_struct {
int users;
@@ -16,6 +17,7 @@ struct fs_struct {
} __randomize_layout;
extern struct kmem_cache *fs_cachep;
+extern struct fs_struct *userspace_init_fs;
extern void exit_fs(struct task_struct *);
extern void set_fs_root(struct fs_struct *, const struct path *);
@@ -40,6 +42,8 @@ static inline void get_fs_pwd(struct fs_struct *fs, struct path *pwd)
read_sequnlock_excl(&fs->seq);
}
+struct fs_struct *switch_fs_struct(struct fs_struct *new_fs);
+
extern bool current_chrooted(void);
static inline int current_umask(void)
@@ -47,4 +51,34 @@ static inline int current_umask(void)
return current->fs->umask;
}
+/*
+ * Temporarily use userspace_init_fs for path resolution in kthreads.
+ * Callers should use scoped_with_init_fs() which automatically
+ * restores the original fs_struct at scope exit.
+ */
+static inline struct fs_struct *__override_init_fs(void)
+{
+ struct fs_struct *old_fs;
+
+ old_fs = current->fs;
+ WRITE_ONCE(current->fs, userspace_init_fs);
+ return old_fs;
+}
+
+static inline void __revert_init_fs(struct fs_struct *old_fs)
+{
+ VFS_WARN_ON_ONCE(current->fs != userspace_init_fs);
+ WRITE_ONCE(current->fs, old_fs);
+}
+
+DEFINE_CLASS(__override_init_fs,
+ struct fs_struct *,
+ __revert_init_fs(_T),
+ __override_init_fs(), void)
+
+#define scoped_with_init_fs() \
+ scoped_class(__override_init_fs, __UNIQUE_ID(label))
+
+void __init init_userspace_fs(void);
+
#endif /* _LINUX_FS_STRUCT_H */
diff --git a/include/linux/fscrypt.h b/include/linux/fscrypt.h
index f6b235cd72b4..28f2108e29f3 100644
--- a/include/linux/fscrypt.h
+++ b/include/linux/fscrypt.h
@@ -72,14 +72,15 @@ struct fscrypt_operations {
ptrdiff_t inode_info_offs;
/*
- * If set, then fs/crypto/ will allocate a global bounce page pool the
- * first time an encryption key is set up for a file. The bounce page
- * pool is required by the following functions:
- *
- * - fscrypt_encrypt_pagecache_blocks()
- * - fscrypt_zeroout_range() for files not using inline crypto
- *
- * If the filesystem doesn't use those, it doesn't need to set this.
+ * Set to 1 if the filesystem is block-based. This causes fs/crypto/ to
+ * set up the key for regular files as a blk_crypto_key. The filesystem
+ * then uses fscrypt_set_bio_crypt_ctx() and similar functions.
+ */
+ unsigned int is_block_based : 1;
+
+ /*
+ * Set to 1 if the filesystem uses fscrypt_encrypt_pagecache_blocks().
+ * This enables the allocation of the bounce page pool it requires.
*/
unsigned int needs_bounce_pages : 1;
@@ -344,7 +345,6 @@ static inline void fscrypt_prepare_dentry(struct dentry *dentry,
}
/* crypto.c */
-void fscrypt_enqueue_decrypt_work(struct work_struct *);
struct page *fscrypt_encrypt_pagecache_blocks(struct folio *folio,
size_t len, size_t offs, gfp_t gfp_flags);
@@ -352,8 +352,6 @@ int fscrypt_encrypt_block_inplace(const struct inode *inode, struct page *page,
unsigned int len, unsigned int offs,
u64 lblk_num);
-int fscrypt_decrypt_pagecache_blocks(struct folio *folio, size_t len,
- size_t offs);
int fscrypt_decrypt_block_inplace(const struct inode *inode, struct page *page,
unsigned int len, unsigned int offs,
u64 lblk_num);
@@ -450,11 +448,6 @@ bool fscrypt_match_name(const struct fscrypt_name *fname,
const u8 *de_name, u32 de_name_len);
u64 fscrypt_fname_siphash(const struct inode *dir, const struct qstr *name);
-/* bio.c */
-bool fscrypt_decrypt_bio(struct bio *bio);
-int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
- sector_t sector, u64 len);
-
/* hooks.c */
int fscrypt_file_open(struct inode *inode, struct file *filp);
int __fscrypt_prepare_link(struct inode *inode, struct inode *dir,
@@ -511,9 +504,6 @@ static inline void fscrypt_prepare_dentry(struct dentry *dentry,
}
/* crypto.c */
-static inline void fscrypt_enqueue_decrypt_work(struct work_struct *work)
-{
-}
static inline struct page *fscrypt_encrypt_pagecache_blocks(struct folio *folio,
size_t len, size_t offs, gfp_t gfp_flags)
@@ -529,12 +519,6 @@ static inline int fscrypt_encrypt_block_inplace(const struct inode *inode,
return -EOPNOTSUPP;
}
-static inline int fscrypt_decrypt_pagecache_blocks(struct folio *folio,
- size_t len, size_t offs)
-{
- return -EOPNOTSUPP;
-}
-
static inline int fscrypt_decrypt_block_inplace(const struct inode *inode,
struct page *page,
unsigned int len,
@@ -751,18 +735,6 @@ static inline int fscrypt_d_revalidate(struct inode *dir, const struct qstr *nam
return 1;
}
-/* bio.c */
-static inline bool fscrypt_decrypt_bio(struct bio *bio)
-{
- return true;
-}
-
-static inline int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
- sector_t sector, u64 len)
-{
- return -EOPNOTSUPP;
-}
-
/* hooks.c */
static inline int fscrypt_file_open(struct inode *inode, struct file *filp)
@@ -862,28 +834,21 @@ static inline void fscrypt_set_ops(struct super_block *sb,
#endif /* !CONFIG_FS_ENCRYPTION */
-/* inline_crypt.c */
+/* block.c */
#ifdef CONFIG_FS_ENCRYPTION_INLINE_CRYPT
-bool __fscrypt_inode_uses_inline_crypto(const struct inode *inode);
-
void fscrypt_set_bio_crypt_ctx(struct bio *bio, const struct inode *inode,
loff_t pos, gfp_t gfp_mask);
bool fscrypt_mergeable_bio(struct bio *bio, const struct inode *inode,
loff_t pos);
-bool fscrypt_dio_supported(struct inode *inode);
-
u64 fscrypt_limit_io_blocks(const struct inode *inode, u64 lblk, u64 nr_blocks);
+int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
+ sector_t sector, u64 len);
#else /* CONFIG_FS_ENCRYPTION_INLINE_CRYPT */
-static inline bool __fscrypt_inode_uses_inline_crypto(const struct inode *inode)
-{
- return false;
-}
-
static inline void fscrypt_set_bio_crypt_ctx(struct bio *bio,
const struct inode *inode,
loff_t pos, gfp_t gfp_mask) { }
@@ -895,47 +860,18 @@ static inline bool fscrypt_mergeable_bio(struct bio *bio,
return true;
}
-static inline bool fscrypt_dio_supported(struct inode *inode)
-{
- return !fscrypt_needs_contents_encryption(inode);
-}
-
static inline u64 fscrypt_limit_io_blocks(const struct inode *inode, u64 lblk,
u64 nr_blocks)
{
return nr_blocks;
}
-#endif /* !CONFIG_FS_ENCRYPTION_INLINE_CRYPT */
-/**
- * fscrypt_inode_uses_inline_crypto() - test whether an inode uses inline
- * encryption
- * @inode: an inode. If encrypted, its key must be set up.
- *
- * Return: true if the inode requires file contents encryption and if the
- * encryption should be done in the block layer via blk-crypto rather
- * than in the filesystem layer.
- */
-static inline bool fscrypt_inode_uses_inline_crypto(const struct inode *inode)
-{
- return fscrypt_needs_contents_encryption(inode) &&
- __fscrypt_inode_uses_inline_crypto(inode);
-}
-
-/**
- * fscrypt_inode_uses_fs_layer_crypto() - test whether an inode uses fs-layer
- * encryption
- * @inode: an inode. If encrypted, its key must be set up.
- *
- * Return: true if the inode requires file contents encryption and if the
- * encryption should be done in the filesystem layer rather than in the
- * block layer via blk-crypto.
- */
-static inline bool fscrypt_inode_uses_fs_layer_crypto(const struct inode *inode)
+static inline int fscrypt_zeroout_range(const struct inode *inode, loff_t pos,
+ sector_t sector, u64 len)
{
- return fscrypt_needs_contents_encryption(inode) &&
- !__fscrypt_inode_uses_inline_crypto(inode);
+ return -EOPNOTSUPP;
}
+#endif /* !CONFIG_FS_ENCRYPTION_INLINE_CRYPT */
/**
* fscrypt_has_encryption_key() - check whether an inode has had its key set up
@@ -1123,15 +1059,4 @@ static inline int fscrypt_encrypt_symlink(struct inode *inode,
return 0;
}
-/* If *pagep is a bounce page, free it and set *pagep to the pagecache page */
-static inline void fscrypt_finalize_bounce_page(struct page **pagep)
-{
- struct page *page = *pagep;
-
- if (fscrypt_is_bounce_page(page)) {
- *pagep = fscrypt_pagecache_page(page);
- fscrypt_free_bounce_page(page);
- }
-}
-
#endif /* _LINUX_FSCRYPT_H */
diff --git a/include/linux/init_task.h b/include/linux/init_task.h
index a6cb241ea00c..61536be773f5 100644
--- a/include/linux/init_task.h
+++ b/include/linux/init_task.h
@@ -24,6 +24,7 @@
extern struct files_struct init_files;
extern struct fs_struct init_fs;
+extern struct fs_struct *userspace_init_fs;
extern struct nsproxy init_nsproxy;
#ifndef CONFIG_VIRT_CPU_ACCOUNTING_NATIVE
diff --git a/include/linux/jbd2.h b/include/linux/jbd2.h
index b68561187e90..1b42fe47c26b 100644
--- a/include/linux/jbd2.h
+++ b/include/linux/jbd2.h
@@ -510,11 +510,12 @@ struct jbd2_journal_handle
int h_err;
/* Flags [no locking] */
- unsigned int h_sync: 1;
- unsigned int h_reserved: 1;
- unsigned int h_aborted: 1;
- unsigned int h_type: 8;
- unsigned int h_line_no: 16;
+ unsigned char h_sync: 1;
+ unsigned char h_reserved: 1;
+ unsigned char h_aborted: 1;
+ unsigned char h_invalid: 1;
+ unsigned char h_type;
+ unsigned short h_line_no;
unsigned long h_start_jiffies;
unsigned int h_requested_credits;
diff --git a/include/linux/lockd/bind.h b/include/linux/lockd/bind.h
index b614e0deea72..db8207d4059f 100644
--- a/include/linux/lockd/bind.h
+++ b/include/linux/lockd/bind.h
@@ -16,17 +16,23 @@ struct svc_rqst;
struct rpc_task;
struct rpc_clnt;
struct super_block;
+struct module;
-/*
- * This is the set of functions for lockd->nfsd communication
+/**
+ * struct nlmsvc_binding - lockd -> nfsd callback table
+ * @owner: module that provides this binding.
+ * @fopen: open a file by NFS file handle on behalf of an NLM request.
+ * @fclose: close a file that was previously opened via @fopen.
+ * Implementations MUST be semantically equivalent to fput().
*/
struct nlmsvc_binding {
+ struct module *owner;
int (*fopen)(struct svc_rqst *rqstp, struct nfs_fh *f,
struct file **filp, int flags);
void (*fclose)(struct file *filp);
};
-extern const struct nlmsvc_binding *nlmsvc_ops;
+extern const struct nlmsvc_binding __rcu *nlmsvc_ops;
/*
* Similar to nfs_client_initdata, but without the NFS-specific
diff --git a/include/linux/net.h b/include/linux/net.h
index f268f395ce47..fdcf9956805c 100644
--- a/include/linux/net.h
+++ b/include/linux/net.h
@@ -285,6 +285,7 @@ int sock_recvmsg(struct socket *sock, struct msghdr *msg, int flags);
struct file *sock_alloc_file(struct socket *sock, int flags, const char *dname);
struct socket *sockfd_lookup(int fd, int *err);
struct socket *sock_from_file(struct file *file);
+int sock_read_xattr(struct socket *sock, const char *name, void *value, size_t size);
#define sockfd_put(sock) fput(sock->file)
int net_ratelimit(void);
diff --git a/include/linux/nfs4.h b/include/linux/nfs4.h
index d87be1f25273..44e5e9fa12e1 100644
--- a/include/linux/nfs4.h
+++ b/include/linux/nfs4.h
@@ -171,133 +171,6 @@ Needs to be updated if more operations are defined in future.*/
#define LAST_NFS42_OP OP_REMOVEXATTR
#define LAST_NFS4_OP LAST_NFS42_OP
-enum nfsstat4 {
- NFS4_OK = 0,
- NFS4ERR_PERM = 1,
- NFS4ERR_NOENT = 2,
- NFS4ERR_IO = 5,
- NFS4ERR_NXIO = 6,
- NFS4ERR_ACCESS = 13,
- NFS4ERR_EXIST = 17,
- NFS4ERR_XDEV = 18,
- /* Unused/reserved 19 */
- NFS4ERR_NOTDIR = 20,
- NFS4ERR_ISDIR = 21,
- NFS4ERR_INVAL = 22,
- NFS4ERR_FBIG = 27,
- NFS4ERR_NOSPC = 28,
- NFS4ERR_ROFS = 30,
- NFS4ERR_MLINK = 31,
- NFS4ERR_NAMETOOLONG = 63,
- NFS4ERR_NOTEMPTY = 66,
- NFS4ERR_DQUOT = 69,
- NFS4ERR_STALE = 70,
- NFS4ERR_BADHANDLE = 10001,
- NFS4ERR_BAD_COOKIE = 10003,
- NFS4ERR_NOTSUPP = 10004,
- NFS4ERR_TOOSMALL = 10005,
- NFS4ERR_SERVERFAULT = 10006,
- NFS4ERR_BADTYPE = 10007,
- NFS4ERR_DELAY = 10008,
- NFS4ERR_SAME = 10009,
- NFS4ERR_DENIED = 10010,
- NFS4ERR_EXPIRED = 10011,
- NFS4ERR_LOCKED = 10012,
- NFS4ERR_GRACE = 10013,
- NFS4ERR_FHEXPIRED = 10014,
- NFS4ERR_SHARE_DENIED = 10015,
- NFS4ERR_WRONGSEC = 10016,
- NFS4ERR_CLID_INUSE = 10017,
- NFS4ERR_RESOURCE = 10018,
- NFS4ERR_MOVED = 10019,
- NFS4ERR_NOFILEHANDLE = 10020,
- NFS4ERR_MINOR_VERS_MISMATCH = 10021,
- NFS4ERR_STALE_CLIENTID = 10022,
- NFS4ERR_STALE_STATEID = 10023,
- NFS4ERR_OLD_STATEID = 10024,
- NFS4ERR_BAD_STATEID = 10025,
- NFS4ERR_BAD_SEQID = 10026,
- NFS4ERR_NOT_SAME = 10027,
- NFS4ERR_LOCK_RANGE = 10028,
- NFS4ERR_SYMLINK = 10029,
- NFS4ERR_RESTOREFH = 10030,
- NFS4ERR_LEASE_MOVED = 10031,
- NFS4ERR_ATTRNOTSUPP = 10032,
- NFS4ERR_NO_GRACE = 10033,
- NFS4ERR_RECLAIM_BAD = 10034,
- NFS4ERR_RECLAIM_CONFLICT = 10035,
- NFS4ERR_BADXDR = 10036,
- NFS4ERR_LOCKS_HELD = 10037,
- NFS4ERR_OPENMODE = 10038,
- NFS4ERR_BADOWNER = 10039,
- NFS4ERR_BADCHAR = 10040,
- NFS4ERR_BADNAME = 10041,
- NFS4ERR_BAD_RANGE = 10042,
- NFS4ERR_LOCK_NOTSUPP = 10043,
- NFS4ERR_OP_ILLEGAL = 10044,
- NFS4ERR_DEADLOCK = 10045,
- NFS4ERR_FILE_OPEN = 10046,
- NFS4ERR_ADMIN_REVOKED = 10047,
- NFS4ERR_CB_PATH_DOWN = 10048,
-
- /* nfs41 */
- NFS4ERR_BADIOMODE = 10049,
- NFS4ERR_BADLAYOUT = 10050,
- NFS4ERR_BAD_SESSION_DIGEST = 10051,
- NFS4ERR_BADSESSION = 10052,
- NFS4ERR_BADSLOT = 10053,
- NFS4ERR_COMPLETE_ALREADY = 10054,
- NFS4ERR_CONN_NOT_BOUND_TO_SESSION = 10055,
- NFS4ERR_DELEG_ALREADY_WANTED = 10056,
- NFS4ERR_BACK_CHAN_BUSY = 10057, /* backchan reqs outstanding */
- NFS4ERR_LAYOUTTRYLATER = 10058,
- NFS4ERR_LAYOUTUNAVAILABLE = 10059,
- NFS4ERR_NOMATCHING_LAYOUT = 10060,
- NFS4ERR_RECALLCONFLICT = 10061,
- NFS4ERR_UNKNOWN_LAYOUTTYPE = 10062,
- NFS4ERR_SEQ_MISORDERED = 10063, /* unexpected seq.id in req */
- NFS4ERR_SEQUENCE_POS = 10064, /* [CB_]SEQ. op not 1st op */
- NFS4ERR_REQ_TOO_BIG = 10065, /* request too big */
- NFS4ERR_REP_TOO_BIG = 10066, /* reply too big */
- NFS4ERR_REP_TOO_BIG_TO_CACHE = 10067, /* rep. not all cached */
- NFS4ERR_RETRY_UNCACHED_REP = 10068, /* retry & rep. uncached */
- NFS4ERR_UNSAFE_COMPOUND = 10069, /* retry/recovery too hard */
- NFS4ERR_TOO_MANY_OPS = 10070, /* too many ops in [CB_]COMP */
- NFS4ERR_OP_NOT_IN_SESSION = 10071, /* op needs [CB_]SEQ. op */
- NFS4ERR_HASH_ALG_UNSUPP = 10072, /* hash alg. not supp. */
- /* Error 10073 is unused. */
- NFS4ERR_CLIENTID_BUSY = 10074, /* clientid has state */
- NFS4ERR_PNFS_IO_HOLE = 10075, /* IO to _SPARSE file hole */
- NFS4ERR_SEQ_FALSE_RETRY = 10076, /* retry not original */
- NFS4ERR_BAD_HIGH_SLOT = 10077, /* sequence arg bad */
- NFS4ERR_DEADSESSION = 10078, /* persistent session dead */
- NFS4ERR_ENCR_ALG_UNSUPP = 10079, /* SSV alg mismatch */
- NFS4ERR_PNFS_NO_LAYOUT = 10080, /* direct I/O with no layout */
- NFS4ERR_NOT_ONLY_OP = 10081, /* bad compound */
- NFS4ERR_WRONG_CRED = 10082, /* permissions:state change */
- NFS4ERR_WRONG_TYPE = 10083, /* current operation mismatch */
- NFS4ERR_DIRDELEG_UNAVAIL = 10084, /* no directory delegation */
- NFS4ERR_REJECT_DELEG = 10085, /* on callback */
- NFS4ERR_RETURNCONFLICT = 10086, /* outstanding layoutreturn */
- NFS4ERR_DELEG_REVOKED = 10087, /* deleg./layout revoked */
-
- /* nfs42 */
- NFS4ERR_PARTNER_NOTSUPP = 10088,
- NFS4ERR_PARTNER_NO_AUTH = 10089,
- NFS4ERR_UNION_NOTSUPP = 10090,
- NFS4ERR_OFFLOAD_DENIED = 10091,
- NFS4ERR_WRONG_LFS = 10092,
- NFS4ERR_BADLABEL = 10093,
- NFS4ERR_OFFLOAD_NO_REQS = 10094,
-
- /* xattr (RFC8276) */
- NFS4ERR_NOXATTR = 10095,
- NFS4ERR_XATTR2BIG = 10096,
-
- /* can be used for internal errors */
- NFS4ERR_FIRST_FREE
-};
-
/* error codes for internal client use */
#define NFS4ERR_RESET_TO_MDS 12001
#define NFS4ERR_RESET_TO_PNFS 12002
diff --git a/include/linux/sched.h b/include/linux/sched.h
index 662e69a2f406..384961c2d206 100644
--- a/include/linux/sched.h
+++ b/include/linux/sched.h
@@ -1191,6 +1191,7 @@ struct task_struct {
unsigned long last_switch_time;
#endif
/* Filesystem information: */
+ struct fs_struct *real_fs;
struct fs_struct *fs;
/* Open file information: */
diff --git a/include/linux/sched/task.h b/include/linux/sched/task.h
index 41ed884cffc9..e0c1ca8c6a18 100644
--- a/include/linux/sched/task.h
+++ b/include/linux/sched/task.h
@@ -31,6 +31,7 @@ struct kernel_clone_args {
u32 io_thread:1;
u32 user_worker:1;
u32 no_files:1;
+ u32 umh:1;
unsigned long stack;
unsigned long stack_size;
unsigned long tls;
diff --git a/include/linux/sunrpc/bc_xprt.h b/include/linux/sunrpc/bc_xprt.h
index 98939cb664cf..59d0cc889beb 100644
--- a/include/linux/sunrpc/bc_xprt.h
+++ b/include/linux/sunrpc/bc_xprt.h
@@ -32,6 +32,7 @@ int xprt_setup_bc(struct rpc_xprt *xprt, unsigned int min_reqs);
void xprt_destroy_bc(struct rpc_xprt *xprt, unsigned int max_reqs);
void xprt_free_bc_rqst(struct rpc_rqst *req);
unsigned int xprt_bc_max_slots(struct rpc_xprt *xprt);
+void xprt_svc_shutdown_bc(struct rpc_xprt *xprt);
void xprt_svc_destroy_nullify_bc(struct rpc_xprt *xprt, struct svc_serv **serv);
/*
@@ -71,6 +72,10 @@ static inline void xprt_free_bc_request(struct rpc_rqst *req)
{
}
+static inline void xprt_svc_shutdown_bc(struct rpc_xprt *xprt)
+{
+}
+
static inline void xprt_svc_destroy_nullify_bc(struct rpc_xprt *xprt, struct svc_serv **serv)
{
svc_destroy(serv);
diff --git a/include/linux/sunrpc/svc.h b/include/linux/sunrpc/svc.h
index 4be6204f6630..3a0152d926fb 100644
--- a/include/linux/sunrpc/svc.h
+++ b/include/linux/sunrpc/svc.h
@@ -469,6 +469,7 @@ int svc_set_pool_threads(struct svc_serv *serv, struct svc_pool *pool,
unsigned int min_threads, unsigned int max_threads);
int svc_set_num_threads(struct svc_serv *serv, unsigned int min_threads,
unsigned int nrservs);
+unsigned int svc_serv_maxthreads(const struct svc_serv *serv);
int svc_pool_stats_open(struct svc_info *si, struct file *file);
void svc_process(struct svc_rqst *rqstp);
void svc_process_bc(struct rpc_rqst *req, struct svc_rqst *rqstp);
diff --git a/include/linux/sunrpc/svc_rdma_pcl.h b/include/linux/sunrpc/svc_rdma_pcl.h
index 7516ad0fae80..6346d8cf2587 100644
--- a/include/linux/sunrpc/svc_rdma_pcl.h
+++ b/include/linux/sunrpc/svc_rdma_pcl.h
@@ -97,7 +97,7 @@ pcl_next_chunk(const struct svc_rdma_pcl *pcl, struct svc_rdma_chunk *chunk)
*/
#define pcl_for_each_segment(pos, chunk) \
for (pos = &(chunk)->ch_segments[0]; \
- pos <= &(chunk)->ch_segments[(chunk)->ch_segcount - 1]; \
+ pos < &(chunk)->ch_segments[(chunk)->ch_segcount]; \
pos++)
/**
@@ -119,6 +119,8 @@ extern bool pcl_alloc_call(struct svc_rdma_recv_ctxt *rctxt, __be32 *p);
extern bool pcl_alloc_read(struct svc_rdma_recv_ctxt *rctxt, __be32 *p);
extern bool pcl_alloc_write(struct svc_rdma_recv_ctxt *rctxt,
struct svc_rdma_pcl *pcl, __be32 *p);
+extern bool pcl_check_read_chunk_positions(struct svc_rdma_recv_ctxt *rctxt,
+ unsigned int inline_len);
extern int pcl_process_nonpayloads(const struct svc_rdma_pcl *pcl,
const struct xdr_buf *xdr,
int (*actor)(const struct xdr_buf *,
diff --git a/include/linux/sunrpc/xdrgen/nfs4_1.h b/include/linux/sunrpc/xdrgen/nfs4_1.h
index 4ac54bdbd335..356c3da9f4e0 100644
--- a/include/linux/sunrpc/xdrgen/nfs4_1.h
+++ b/include/linux/sunrpc/xdrgen/nfs4_1.h
@@ -1,7 +1,7 @@
/* SPDX-License-Identifier: GPL-2.0 */
/* Generated by xdrgen. Manual edits will be lost. */
/* XDR specification file: ../../Documentation/sunrpc/xdr/nfs4_1.x */
-/* XDR specification modification time: Thu Jan 8 23:12:07 2026 */
+/* XDR specification modification time: Wed Mar 25 11:40:02 2026 */
#ifndef _LINUX_XDRGEN_NFS4_1_DEF_H
#define _LINUX_XDRGEN_NFS4_1_DEF_H
@@ -9,15 +9,149 @@
#include <linux/types.h>
#include <linux/sunrpc/xdrgen/_defs.h>
-typedef s64 int64_t;
+typedef s32 int32_t;
typedef u32 uint32_t;
+typedef s64 int64_t;
+
+typedef u64 uint64_t;
+
+enum { NFS4_VERIFIER_SIZE = 8 };
+
+enum { NFS4_FHSIZE = 128 };
+
+enum nfsstat4 {
+ NFS4_OK = 0,
+ NFS4ERR_PERM = 1,
+ NFS4ERR_NOENT = 2,
+ NFS4ERR_IO = 5,
+ NFS4ERR_NXIO = 6,
+ NFS4ERR_ACCESS = 13,
+ NFS4ERR_EXIST = 17,
+ NFS4ERR_XDEV = 18,
+ NFS4ERR_NOTDIR = 20,
+ NFS4ERR_ISDIR = 21,
+ NFS4ERR_INVAL = 22,
+ NFS4ERR_FBIG = 27,
+ NFS4ERR_NOSPC = 28,
+ NFS4ERR_ROFS = 30,
+ NFS4ERR_MLINK = 31,
+ NFS4ERR_NAMETOOLONG = 63,
+ NFS4ERR_NOTEMPTY = 66,
+ NFS4ERR_DQUOT = 69,
+ NFS4ERR_STALE = 70,
+ NFS4ERR_BADHANDLE = 10001,
+ NFS4ERR_BAD_COOKIE = 10003,
+ NFS4ERR_NOTSUPP = 10004,
+ NFS4ERR_TOOSMALL = 10005,
+ NFS4ERR_SERVERFAULT = 10006,
+ NFS4ERR_BADTYPE = 10007,
+ NFS4ERR_DELAY = 10008,
+ NFS4ERR_SAME = 10009,
+ NFS4ERR_DENIED = 10010,
+ NFS4ERR_EXPIRED = 10011,
+ NFS4ERR_LOCKED = 10012,
+ NFS4ERR_GRACE = 10013,
+ NFS4ERR_FHEXPIRED = 10014,
+ NFS4ERR_SHARE_DENIED = 10015,
+ NFS4ERR_WRONGSEC = 10016,
+ NFS4ERR_CLID_INUSE = 10017,
+ NFS4ERR_RESOURCE = 10018,
+ NFS4ERR_MOVED = 10019,
+ NFS4ERR_NOFILEHANDLE = 10020,
+ NFS4ERR_MINOR_VERS_MISMATCH = 10021,
+ NFS4ERR_STALE_CLIENTID = 10022,
+ NFS4ERR_STALE_STATEID = 10023,
+ NFS4ERR_OLD_STATEID = 10024,
+ NFS4ERR_BAD_STATEID = 10025,
+ NFS4ERR_BAD_SEQID = 10026,
+ NFS4ERR_NOT_SAME = 10027,
+ NFS4ERR_LOCK_RANGE = 10028,
+ NFS4ERR_SYMLINK = 10029,
+ NFS4ERR_RESTOREFH = 10030,
+ NFS4ERR_LEASE_MOVED = 10031,
+ NFS4ERR_ATTRNOTSUPP = 10032,
+ NFS4ERR_NO_GRACE = 10033,
+ NFS4ERR_RECLAIM_BAD = 10034,
+ NFS4ERR_RECLAIM_CONFLICT = 10035,
+ NFS4ERR_BADXDR = 10036,
+ NFS4ERR_LOCKS_HELD = 10037,
+ NFS4ERR_OPENMODE = 10038,
+ NFS4ERR_BADOWNER = 10039,
+ NFS4ERR_BADCHAR = 10040,
+ NFS4ERR_BADNAME = 10041,
+ NFS4ERR_BAD_RANGE = 10042,
+ NFS4ERR_LOCK_NOTSUPP = 10043,
+ NFS4ERR_OP_ILLEGAL = 10044,
+ NFS4ERR_DEADLOCK = 10045,
+ NFS4ERR_FILE_OPEN = 10046,
+ NFS4ERR_ADMIN_REVOKED = 10047,
+ NFS4ERR_CB_PATH_DOWN = 10048,
+ NFS4ERR_BADIOMODE = 10049,
+ NFS4ERR_BADLAYOUT = 10050,
+ NFS4ERR_BAD_SESSION_DIGEST = 10051,
+ NFS4ERR_BADSESSION = 10052,
+ NFS4ERR_BADSLOT = 10053,
+ NFS4ERR_COMPLETE_ALREADY = 10054,
+ NFS4ERR_CONN_NOT_BOUND_TO_SESSION = 10055,
+ NFS4ERR_DELEG_ALREADY_WANTED = 10056,
+ NFS4ERR_BACK_CHAN_BUSY = 10057,
+ NFS4ERR_LAYOUTTRYLATER = 10058,
+ NFS4ERR_LAYOUTUNAVAILABLE = 10059,
+ NFS4ERR_NOMATCHING_LAYOUT = 10060,
+ NFS4ERR_RECALLCONFLICT = 10061,
+ NFS4ERR_UNKNOWN_LAYOUTTYPE = 10062,
+ NFS4ERR_SEQ_MISORDERED = 10063,
+ NFS4ERR_SEQUENCE_POS = 10064,
+ NFS4ERR_REQ_TOO_BIG = 10065,
+ NFS4ERR_REP_TOO_BIG = 10066,
+ NFS4ERR_REP_TOO_BIG_TO_CACHE = 10067,
+ NFS4ERR_RETRY_UNCACHED_REP = 10068,
+ NFS4ERR_UNSAFE_COMPOUND = 10069,
+ NFS4ERR_TOO_MANY_OPS = 10070,
+ NFS4ERR_OP_NOT_IN_SESSION = 10071,
+ NFS4ERR_HASH_ALG_UNSUPP = 10072,
+ NFS4ERR_CLIENTID_BUSY = 10074,
+ NFS4ERR_PNFS_IO_HOLE = 10075,
+ NFS4ERR_SEQ_FALSE_RETRY = 10076,
+ NFS4ERR_BAD_HIGH_SLOT = 10077,
+ NFS4ERR_DEADSESSION = 10078,
+ NFS4ERR_ENCR_ALG_UNSUPP = 10079,
+ NFS4ERR_PNFS_NO_LAYOUT = 10080,
+ NFS4ERR_NOT_ONLY_OP = 10081,
+ NFS4ERR_WRONG_CRED = 10082,
+ NFS4ERR_WRONG_TYPE = 10083,
+ NFS4ERR_DIRDELEG_UNAVAIL = 10084,
+ NFS4ERR_REJECT_DELEG = 10085,
+ NFS4ERR_RETURNCONFLICT = 10086,
+ NFS4ERR_DELEG_REVOKED = 10087,
+ NFS4ERR_PARTNER_NOTSUPP = 10088,
+ NFS4ERR_PARTNER_NO_AUTH = 10089,
+ NFS4ERR_UNION_NOTSUPP = 10090,
+ NFS4ERR_OFFLOAD_DENIED = 10091,
+ NFS4ERR_WRONG_LFS = 10092,
+ NFS4ERR_BADLABEL = 10093,
+ NFS4ERR_OFFLOAD_NO_REQS = 10094,
+ NFS4ERR_NOXATTR = 10095,
+ NFS4ERR_XATTR2BIG = 10096,
+};
+
+typedef enum nfsstat4 nfsstat4;
+
+typedef opaque attrlist4;
+
typedef struct {
u32 count;
uint32_t *element;
} bitmap4;
+typedef u8 verifier4[NFS4_VERIFIER_SIZE];
+
+typedef uint64_t nfs_cookie4;
+
+typedef opaque nfs_fh4;
+
typedef opaque utf8string;
typedef utf8string utf8str_cis;
@@ -26,11 +160,30 @@ typedef utf8string utf8str_cs;
typedef utf8string utf8str_mixed;
+typedef utf8str_cs component4;
+
+typedef utf8str_cs linktext4;
+
+typedef struct {
+ u32 count;
+ component4 *element;
+} pathname4;
+
struct nfstime4 {
int64_t seconds;
uint32_t nseconds;
};
+struct fattr4 {
+ bitmap4 attrmask;
+ attrlist4 attr_vals;
+};
+
+struct stateid4 {
+ uint32_t seqid;
+ u8 other[12];
+};
+
typedef bool fattr4_offline;
enum { FATTR4_OFFLINE = 83 };
@@ -216,11 +369,109 @@ enum { FATTR4_POSIX_DEFAULT_ACL = 91 };
enum { FATTR4_POSIX_ACCESS_ACL = 92 };
-#define NFS4_int64_t_sz \
- (XDR_hyper)
+enum notify_type4 {
+ NOTIFY4_CHANGE_CHILD_ATTRS = 0,
+ NOTIFY4_CHANGE_DIR_ATTRS = 1,
+ NOTIFY4_REMOVE_ENTRY = 2,
+ NOTIFY4_ADD_ENTRY = 3,
+ NOTIFY4_RENAME_ENTRY = 4,
+ NOTIFY4_CHANGE_COOKIE_VERIFIER = 5,
+ NOTIFY4_GFLAG_EXTEND = 6,
+ NOTIFY4_AUFLAG_VALID = 7,
+ NOTIFY4_AUFLAG_USER = 8,
+ NOTIFY4_AUFLAG_GROUP = 9,
+ NOTIFY4_AUFLAG_OTHER = 10,
+ NOTIFY4_CHANGE_AUTH = 11,
+ NOTIFY4_CFLAG_ORDER = 12,
+ NOTIFY4_AUFLAG_GANOW = 13,
+ NOTIFY4_AUFLAG_GALATER = 14,
+ NOTIFY4_CHANGE_GA = 15,
+ NOTIFY4_CHANGE_AMASK = 16,
+};
+
+typedef enum notify_type4 notify_type4;
+
+struct notify_entry4 {
+ component4 ne_file;
+ struct fattr4 ne_attrs;
+};
+
+struct prev_entry4 {
+ struct notify_entry4 pe_prev_entry;
+ nfs_cookie4 pe_prev_entry_cookie;
+};
+
+struct notify_remove4 {
+ struct notify_entry4 nrm_old_entry;
+ nfs_cookie4 nrm_old_entry_cookie;
+};
+
+struct notify_add4 {
+ struct {
+ u32 count;
+ struct notify_remove4 *element;
+ } nad_old_entry;
+ struct notify_entry4 nad_new_entry;
+ struct {
+ u32 count;
+ nfs_cookie4 *element;
+ } nad_new_entry_cookie;
+ struct {
+ u32 count;
+ struct prev_entry4 *element;
+ } nad_prev_entry;
+ bool nad_last_entry;
+};
+
+struct notify_attr4 {
+ struct notify_entry4 na_changed_entry;
+};
+
+struct notify_rename4 {
+ struct notify_remove4 nrn_old_entry;
+ struct notify_add4 nrn_new_entry;
+};
+
+struct notify_verifier4 {
+ verifier4 nv_old_cookieverf;
+ verifier4 nv_new_cookieverf;
+};
+
+typedef opaque notifylist4;
+
+struct notify4 {
+ bitmap4 notify_mask;
+ notifylist4 notify_vals;
+};
+
+struct CB_NOTIFY4args {
+ struct stateid4 cna_stateid;
+ nfs_fh4 cna_fh;
+ struct {
+ u32 count;
+ struct notify4 *element;
+ } cna_changes;
+};
+
+struct CB_NOTIFY4res {
+ nfsstat4 cnr_status;
+};
+
+#define NFS4_int32_t_sz \
+ (XDR_int)
#define NFS4_uint32_t_sz \
(XDR_unsigned_int)
+#define NFS4_int64_t_sz \
+ (XDR_hyper)
+#define NFS4_uint64_t_sz \
+ (XDR_unsigned_hyper)
+#define NFS4_nfsstat4_sz (XDR_int)
+#define NFS4_attrlist4_sz (XDR_unsigned_int)
#define NFS4_bitmap4_sz (XDR_unsigned_int)
+#define NFS4_verifier4_sz (XDR_QUADLEN(NFS4_VERIFIER_SIZE))
+#define NFS4_nfs_cookie4_sz \
+ (NFS4_uint64_t_sz)
+#define NFS4_nfs_fh4_sz (XDR_unsigned_int + XDR_QUADLEN(NFS4_FHSIZE))
#define NFS4_utf8string_sz (XDR_unsigned_int)
#define NFS4_utf8str_cis_sz \
(NFS4_utf8string_sz)
@@ -228,8 +479,17 @@ enum { FATTR4_POSIX_ACCESS_ACL = 92 };
(NFS4_utf8string_sz)
#define NFS4_utf8str_mixed_sz \
(NFS4_utf8string_sz)
+#define NFS4_component4_sz \
+ (NFS4_utf8str_cs_sz)
+#define NFS4_linktext4_sz \
+ (NFS4_utf8str_cs_sz)
+#define NFS4_pathname4_sz (XDR_unsigned_int)
#define NFS4_nfstime4_sz \
(NFS4_int64_t_sz + NFS4_uint32_t_sz)
+#define NFS4_fattr4_sz \
+ (NFS4_bitmap4_sz + NFS4_attrlist4_sz)
+#define NFS4_stateid4_sz \
+ (NFS4_uint32_t_sz + XDR_QUADLEN(12))
#define NFS4_fattr4_offline_sz \
(XDR_bool)
#define NFS4_open_arguments4_sz \
@@ -259,5 +519,27 @@ enum { FATTR4_POSIX_ACCESS_ACL = 92 };
(NFS4_aclscope4_sz)
#define NFS4_fattr4_posix_default_acl_sz (XDR_unsigned_int)
#define NFS4_fattr4_posix_access_acl_sz (XDR_unsigned_int)
+#define NFS4_notify_type4_sz (XDR_int)
+#define NFS4_notify_entry4_sz \
+ (NFS4_component4_sz + NFS4_fattr4_sz)
+#define NFS4_prev_entry4_sz \
+ (NFS4_notify_entry4_sz + NFS4_nfs_cookie4_sz)
+#define NFS4_notify_remove4_sz \
+ (NFS4_notify_entry4_sz + NFS4_nfs_cookie4_sz)
+#define NFS4_notify_add4_sz \
+ (XDR_unsigned_int + (1 * (NFS4_notify_remove4_sz)) + NFS4_notify_entry4_sz + XDR_unsigned_int + (1 * (NFS4_nfs_cookie4_sz)) + XDR_unsigned_int + (1 * (NFS4_prev_entry4_sz)) + XDR_bool)
+#define NFS4_notify_attr4_sz \
+ (NFS4_notify_entry4_sz)
+#define NFS4_notify_rename4_sz \
+ (NFS4_notify_remove4_sz + NFS4_notify_add4_sz)
+#define NFS4_notify_verifier4_sz \
+ (NFS4_verifier4_sz + NFS4_verifier4_sz)
+#define NFS4_notifylist4_sz (XDR_unsigned_int)
+#define NFS4_notify4_sz \
+ (NFS4_bitmap4_sz + NFS4_notifylist4_sz)
+#define NFS4_CB_NOTIFY4args_sz \
+ (NFS4_stateid4_sz + NFS4_nfs_fh4_sz + XDR_unsigned_int)
+#define NFS4_CB_NOTIFY4res_sz \
+ (NFS4_nfsstat4_sz)
#endif /* _LINUX_XDRGEN_NFS4_1_DEF_H */
diff --git a/include/linux/types.h b/include/linux/types.h
index 93166b0b0617..bc5dda2a3d86 100644
--- a/include/linux/types.h
+++ b/include/linux/types.h
@@ -163,6 +163,8 @@ typedef u32 dma_addr_t;
typedef unsigned int __bitwise gfp_t;
typedef unsigned int __bitwise slab_flags_t;
typedef unsigned int __bitwise fmode_t;
+typedef unsigned int __bitwise blk_mode_t;
+typedef unsigned int __bitwise fop_flags_t;
#ifdef CONFIG_PHYS_ADDR_T_64BIT
typedef u64 phys_addr_t;
diff --git a/include/uapi/asm-generic/errno.h b/include/uapi/asm-generic/errno.h
index bd78e69e0a43..c84ebf89c8b6 100644
--- a/include/uapi/asm-generic/errno.h
+++ b/include/uapi/asm-generic/errno.h
@@ -76,7 +76,7 @@
#define ENOPROTOOPT 92 /* Protocol not available */
#define EPROTONOSUPPORT 93 /* Protocol not supported */
#define ESOCKTNOSUPPORT 94 /* Socket type not supported */
-#define EOPNOTSUPP 95 /* Operation not supported on transport endpoint */
+#define EOPNOTSUPP 95 /* Operation not supported */
#define EPFNOSUPPORT 96 /* Protocol family not supported */
#define EAFNOSUPPORT 97 /* Address family not supported by protocol */
#define EADDRINUSE 98 /* Address already in use */
diff --git a/include/uapi/linux/efs_fs_sb.h b/include/uapi/linux/efs_fs_sb.h
deleted file mode 100644
index 6bad29a10faa..000000000000
--- a/include/uapi/linux/efs_fs_sb.h
+++ /dev/null
@@ -1,63 +0,0 @@
-/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
-/*
- * efs_fs_sb.h
- *
- * Copyright (c) 1999 Al Smith
- *
- * Portions derived from IRIX header files (c) 1988 Silicon Graphics
- */
-
-#ifndef __EFS_FS_SB_H__
-#define __EFS_FS_SB_H__
-
-#include <linux/types.h>
-#include <linux/magic.h>
-
-/* EFS superblock magic numbers */
-#define EFS_MAGIC 0x072959
-#define EFS_NEWMAGIC 0x07295a
-
-#define IS_EFS_MAGIC(x) ((x == EFS_MAGIC) || (x == EFS_NEWMAGIC))
-
-#define EFS_SUPER 1
-#define EFS_ROOTINODE 2
-
-/* efs superblock on disk */
-struct efs_super {
- __be32 fs_size; /* size of filesystem, in sectors */
- __be32 fs_firstcg; /* bb offset to first cg */
- __be32 fs_cgfsize; /* size of cylinder group in bb's */
- __be16 fs_cgisize; /* bb's of inodes per cylinder group */
- __be16 fs_sectors; /* sectors per track */
- __be16 fs_heads; /* heads per cylinder */
- __be16 fs_ncg; /* # of cylinder groups in filesystem */
- __be16 fs_dirty; /* fs needs to be fsck'd */
- __be32 fs_time; /* last super-block update */
- __be32 fs_magic; /* magic number */
- char fs_fname[6]; /* file system name */
- char fs_fpack[6]; /* file system pack name */
- __be32 fs_bmsize; /* size of bitmap in bytes */
- __be32 fs_tfree; /* total free data blocks */
- __be32 fs_tinode; /* total free inodes */
- __be32 fs_bmblock; /* bitmap location. */
- __be32 fs_replsb; /* Location of replicated superblock. */
- __be32 fs_lastialloc; /* last allocated inode */
- char fs_spare[20]; /* space for expansion - MUST BE ZERO */
- __be32 fs_checksum; /* checksum of volume portion of fs */
-};
-
-/* efs superblock information in memory */
-struct efs_sb_info {
- __u32 fs_magic; /* superblock magic number */
- __u32 fs_start; /* first block of filesystem */
- __u32 first_block; /* first data block in filesystem */
- __u32 total_blocks; /* total number of blocks in filesystem */
- __u32 group_size; /* # of blocks a group consists of */
- __u32 data_free; /* # of free data blocks */
- __u32 inode_free; /* # of free inodes */
- __u16 inode_blocks; /* # of blocks used for inodes in every grp */
- __u16 total_groups; /* # of groups */
-};
-
-#endif /* __EFS_FS_SB_H__ */
-
diff --git a/include/uapi/linux/fscrypt.h b/include/uapi/linux/fscrypt.h
index 3aff99f2696a..84507280b3ea 100644
--- a/include/uapi/linux/fscrypt.h
+++ b/include/uapi/linux/fscrypt.h
@@ -30,7 +30,6 @@
#define FSCRYPT_MODE_SM4_CTS 8
#define FSCRYPT_MODE_ADIANTUM 9
#define FSCRYPT_MODE_AES_256_HCTR2 10
-/* If adding a mode number > 10, update FSCRYPT_MODE_MAX in fscrypt_private.h */
/*
* Legacy policy version; ad-hoc KDF and no key verification.
diff --git a/include/uapi/linux/nfs4.h b/include/uapi/linux/nfs4.h
index 4273e0249fcb..289205b53a08 100644
--- a/include/uapi/linux/nfs4.h
+++ b/include/uapi/linux/nfs4.h
@@ -17,11 +17,9 @@
#include <linux/types.h>
#define NFS4_BITMAP_SIZE 3
-#define NFS4_VERIFIER_SIZE 8
#define NFS4_STATEID_SEQID_SIZE 4
#define NFS4_STATEID_OTHER_SIZE 12
#define NFS4_STATEID_SIZE (NFS4_STATEID_SEQID_SIZE + NFS4_STATEID_OTHER_SIZE)
-#define NFS4_FHSIZE 128
#define NFS4_MAXPATHLEN PATH_MAX
#define NFS4_MAXNAMLEN NAME_MAX
#define NFS4_OPAQUE_LIMIT 1024
diff --git a/init/init_task.c b/init/init_task.c
index b67ef6040a65..ba5c2523f7e0 100644
--- a/init/init_task.c
+++ b/init/init_task.c
@@ -162,6 +162,7 @@ struct task_struct init_task __aligned(L1_CACHE_BYTES) = {
RCU_POINTER_INITIALIZER(cred, &init_cred),
.comm = INIT_TASK_COMM,
.thread = INIT_THREAD,
+ .real_fs = &init_fs,
.fs = &init_fs,
.files = &init_files,
#ifdef CONFIG_IO_URING
diff --git a/init/initramfs.c b/init/initramfs.c
index 20a18fcda48e..4e27b97a8844 100644
--- a/init/initramfs.c
+++ b/init/initramfs.c
@@ -6,6 +6,7 @@
#include <linux/fcntl.h>
#include <linux/file.h>
#include <linux/fs.h>
+#include <linux/fs_struct.h>
#include <linux/hex.h>
#include <linux/init.h>
#include <linux/init_syscalls.h>
@@ -716,7 +717,7 @@ static void __init populate_initrd_image(char *err)
}
#endif /* CONFIG_BLK_DEV_RAM */
-static void __init do_populate_rootfs(void *unused, async_cookie_t cookie)
+static void __init unpack_initramfs(async_cookie_t cookie)
{
/* Load the built in initramfs */
char *err = unpack_to_rootfs(__initramfs_start, __initramfs_size);
@@ -724,7 +725,7 @@ static void __init do_populate_rootfs(void *unused, async_cookie_t cookie)
panic_show_mem("%s", err); /* Failed to decompress INTERNAL initramfs */
if (!initrd_start || IS_ENABLED(CONFIG_INITRAMFS_FORCE))
- goto done;
+ return;
if (IS_ENABLED(CONFIG_BLK_DEV_RAM))
printk(KERN_INFO "Trying to unpack rootfs image as initramfs...\n");
@@ -739,9 +740,14 @@ static void __init do_populate_rootfs(void *unused, async_cookie_t cookie)
printk(KERN_EMERG "Initramfs unpacking failed: %s\n", err);
#endif
}
+}
-done:
- security_initramfs_populated();
+static void __init do_populate_rootfs(void *unused, async_cookie_t cookie)
+{
+ scoped_with_init_fs() {
+ unpack_initramfs(cookie);
+ security_initramfs_populated();
+ }
/*
* If the initrd region is overlapped with crashkernel reserved region,
diff --git a/init/initramfs_test.c b/init/initramfs_test.c
index bc55306d226d..1fc990a66563 100644
--- a/init/initramfs_test.c
+++ b/init/initramfs_test.c
@@ -3,6 +3,7 @@
#include <linux/fcntl.h>
#include <linux/file.h>
#include <linux/fs.h>
+#include <linux/fs_struct.h>
#include <linux/init.h>
#include <linux/init_syscalls.h>
#include <linux/initrd.h>
@@ -116,38 +117,41 @@ static void __init initramfs_test_extract(struct kunit *test)
GFP_KERNEL);
len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
- ktime_get_real_ts64(&ts_before);
- err = unpack_to_rootfs(cpio_srcbuf, len);
- ktime_get_real_ts64(&ts_after);
- if (err) {
- KUNIT_FAIL(test, "unpack failed %s", err);
- goto out;
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ ktime_get_real_ts64(&ts_before);
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ ktime_get_real_ts64(&ts_after);
+ if (err) {
+ KUNIT_FAIL(test, "unpack failed %s", err);
+ goto out;
+ }
+
+ KUNIT_EXPECT_EQ(test, init_stat(c[0].fname, &st, 0), 0);
+ KUNIT_EXPECT_TRUE(test, S_ISREG(st.mode));
+ KUNIT_EXPECT_TRUE(test, uid_eq(st.uid, KUIDT_INIT(c[0].uid)));
+ KUNIT_EXPECT_TRUE(test, gid_eq(st.gid, KGIDT_INIT(c[0].gid)));
+ KUNIT_EXPECT_EQ(test, st.nlink, 1);
+ if (IS_ENABLED(CONFIG_INITRAMFS_PRESERVE_MTIME)) {
+ KUNIT_EXPECT_EQ(test, st.mtime.tv_sec, c[0].mtime);
+ } else {
+ KUNIT_EXPECT_GE(test, st.mtime.tv_sec, ts_before.tv_sec);
+ KUNIT_EXPECT_LE(test, st.mtime.tv_sec, ts_after.tv_sec);
+ }
+ KUNIT_EXPECT_EQ(test, st.blocks, c[0].filesize);
+
+ KUNIT_EXPECT_EQ(test, init_stat(c[1].fname, &st, 0), 0);
+ KUNIT_EXPECT_TRUE(test, S_ISDIR(st.mode));
+ if (IS_ENABLED(CONFIG_INITRAMFS_PRESERVE_MTIME)) {
+ KUNIT_EXPECT_EQ(test, st.mtime.tv_sec, c[1].mtime);
+ } else {
+ KUNIT_EXPECT_GE(test, st.mtime.tv_sec, ts_before.tv_sec);
+ KUNIT_EXPECT_LE(test, st.mtime.tv_sec, ts_after.tv_sec);
+ }
+
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_rmdir(c[1].fname), 0);
}
-
- KUNIT_EXPECT_EQ(test, init_stat(c[0].fname, &st, 0), 0);
- KUNIT_EXPECT_TRUE(test, S_ISREG(st.mode));
- KUNIT_EXPECT_TRUE(test, uid_eq(st.uid, KUIDT_INIT(c[0].uid)));
- KUNIT_EXPECT_TRUE(test, gid_eq(st.gid, KGIDT_INIT(c[0].gid)));
- KUNIT_EXPECT_EQ(test, st.nlink, 1);
- if (IS_ENABLED(CONFIG_INITRAMFS_PRESERVE_MTIME)) {
- KUNIT_EXPECT_EQ(test, st.mtime.tv_sec, c[0].mtime);
- } else {
- KUNIT_EXPECT_GE(test, st.mtime.tv_sec, ts_before.tv_sec);
- KUNIT_EXPECT_LE(test, st.mtime.tv_sec, ts_after.tv_sec);
- }
- KUNIT_EXPECT_EQ(test, st.blocks, c[0].filesize);
-
- KUNIT_EXPECT_EQ(test, init_stat(c[1].fname, &st, 0), 0);
- KUNIT_EXPECT_TRUE(test, S_ISDIR(st.mode));
- if (IS_ENABLED(CONFIG_INITRAMFS_PRESERVE_MTIME)) {
- KUNIT_EXPECT_EQ(test, st.mtime.tv_sec, c[1].mtime);
- } else {
- KUNIT_EXPECT_GE(test, st.mtime.tv_sec, ts_before.tv_sec);
- KUNIT_EXPECT_LE(test, st.mtime.tv_sec, ts_after.tv_sec);
- }
-
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
- KUNIT_EXPECT_EQ(test, init_rmdir(c[1].fname), 0);
out:
kfree(cpio_srcbuf);
}
@@ -197,8 +201,11 @@ static void __init initramfs_test_fname_overrun(struct kunit *test)
suffix_off--;
}
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NOT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NOT_NULL(test, err);
+ }
kfree(cpio_srcbuf);
}
@@ -233,22 +240,25 @@ static void __init initramfs_test_data(struct kunit *test)
len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NULL(test, err);
- file = filp_open(c[0].fname, O_RDONLY, 0);
- if (IS_ERR(file)) {
- KUNIT_FAIL(test, "open failed");
- goto out;
- }
+ file = filp_open(c[0].fname, O_RDONLY, 0);
+ if (IS_ERR(file)) {
+ KUNIT_FAIL(test, "open failed");
+ goto out;
+ }
- /* read back file contents into @cpio_srcbuf and confirm match */
- len = kernel_read(file, cpio_srcbuf, c[0].filesize, NULL);
- KUNIT_EXPECT_EQ(test, len, c[0].filesize);
- KUNIT_EXPECT_MEMEQ(test, cpio_srcbuf, c[0].data, len);
+ /* read back file contents into @cpio_srcbuf and confirm match */
+ len = kernel_read(file, cpio_srcbuf, c[0].filesize, NULL);
+ KUNIT_EXPECT_EQ(test, len, c[0].filesize);
+ KUNIT_EXPECT_MEMEQ(test, cpio_srcbuf, c[0].data, len);
- fput(file);
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ fput(file);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ }
out:
kfree(cpio_srcbuf);
}
@@ -288,25 +298,28 @@ static void __init initramfs_test_csum(struct kunit *test)
len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NULL(test, err);
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
- KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), 0);
- /* mess up the csum and confirm that unpack fails */
- c[0].csum--;
- len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
+ /* mess up the csum and confirm that unpack fails */
+ c[0].csum--;
+ len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NOT_NULL(test, err);
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NOT_NULL(test, err);
- /*
- * file (with content) is still retained in case of bad-csum abort.
- * Perhaps we should change this.
- */
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
- KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), -ENOENT);
+ /*
+ * file (with content) is still retained in case of bad-csum abort.
+ * Perhaps we should change this.
+ */
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), -ENOENT);
+ }
kfree(cpio_srcbuf);
}
@@ -344,17 +357,20 @@ static void __init initramfs_test_hardlink(struct kunit *test)
len = fill_cpio(c, ARRAY_SIZE(c), false, cpio_srcbuf);
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NULL(test, err);
- KUNIT_EXPECT_EQ(test, init_stat(c[0].fname, &st0, 0), 0);
- KUNIT_EXPECT_EQ(test, init_stat(c[1].fname, &st1, 0), 0);
- KUNIT_EXPECT_EQ(test, st0.ino, st1.ino);
- KUNIT_EXPECT_EQ(test, st0.nlink, 2);
- KUNIT_EXPECT_EQ(test, st1.nlink, 2);
+ KUNIT_EXPECT_EQ(test, init_stat(c[0].fname, &st0, 0), 0);
+ KUNIT_EXPECT_EQ(test, init_stat(c[1].fname, &st1, 0), 0);
+ KUNIT_EXPECT_EQ(test, st0.ino, st1.ino);
+ KUNIT_EXPECT_EQ(test, st0.nlink, 2);
+ KUNIT_EXPECT_EQ(test, st1.nlink, 2);
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
- KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[1].fname), 0);
+ }
kfree(cpio_srcbuf);
}
@@ -387,12 +403,15 @@ static void __init initramfs_test_many(struct kunit *test)
}
len = p - cpio_srcbuf;
- err = unpack_to_rootfs(cpio_srcbuf, len);
- KUNIT_EXPECT_NULL(test, err);
-
- for (i = 0; i < INITRAMFS_TEST_MANY_LIMIT; i++) {
- sprintf(thispath, "initramfs_test_many-%d", i);
- KUNIT_EXPECT_EQ(test, init_unlink(thispath), 0);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(cpio_srcbuf, len);
+ KUNIT_EXPECT_NULL(test, err);
+
+ for (i = 0; i < INITRAMFS_TEST_MANY_LIMIT; i++) {
+ sprintf(thispath, "initramfs_test_many-%d", i);
+ KUNIT_EXPECT_EQ(test, init_unlink(thispath), 0);
+ }
}
kfree(cpio_srcbuf);
@@ -439,22 +458,25 @@ static void __init initramfs_test_fname_pad(struct kunit *test)
memcpy(tbufs->padded_fname, "padded_fname", sizeof("padded_fname"));
len = fill_cpio(c, ARRAY_SIZE(c), false, tbufs->cpio_srcbuf);
- err = unpack_to_rootfs(tbufs->cpio_srcbuf, len);
- KUNIT_EXPECT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(tbufs->cpio_srcbuf, len);
+ KUNIT_EXPECT_NULL(test, err);
- file = filp_open(c[0].fname, O_RDONLY, 0);
- if (IS_ERR(file)) {
- KUNIT_FAIL(test, "open failed");
- goto out;
- }
+ file = filp_open(c[0].fname, O_RDONLY, 0);
+ if (IS_ERR(file)) {
+ KUNIT_FAIL(test, "open failed");
+ goto out;
+ }
- /* read back file contents into @cpio_srcbuf and confirm match */
- len = kernel_read(file, tbufs->cpio_srcbuf, c[0].filesize, NULL);
- KUNIT_EXPECT_EQ(test, len, c[0].filesize);
- KUNIT_EXPECT_MEMEQ(test, tbufs->cpio_srcbuf, c[0].data, len);
+ /* read back file contents into @cpio_srcbuf and confirm match */
+ len = kernel_read(file, tbufs->cpio_srcbuf, c[0].filesize, NULL);
+ KUNIT_EXPECT_EQ(test, len, c[0].filesize);
+ KUNIT_EXPECT_MEMEQ(test, tbufs->cpio_srcbuf, c[0].data, len);
- fput(file);
- KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ fput(file);
+ KUNIT_EXPECT_EQ(test, init_unlink(c[0].fname), 0);
+ }
out:
kfree(tbufs);
}
@@ -495,13 +517,16 @@ static void __init initramfs_test_fname_path_max(struct kunit *test)
memcpy(tbufs->fname_ok, "fname_ok", sizeof("fname_ok") - 1);
len = fill_cpio(c, ARRAY_SIZE(c), false, tbufs->cpio_src);
- /* unpack skips over fname_oversize instead of returning an error */
- err = unpack_to_rootfs(tbufs->cpio_src, len);
- KUNIT_EXPECT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ /* unpack skips over fname_oversize instead of returning an error */
+ err = unpack_to_rootfs(tbufs->cpio_src, len);
+ KUNIT_EXPECT_NULL(test, err);
- KUNIT_EXPECT_EQ(test, init_stat("fname_oversize", &st0, 0), -ENOENT);
- KUNIT_EXPECT_EQ(test, init_stat("fname_ok", &st1, 0), 0);
- KUNIT_EXPECT_EQ(test, init_rmdir("fname_ok"), 0);
+ KUNIT_EXPECT_EQ(test, init_stat("fname_oversize", &st0, 0), -ENOENT);
+ KUNIT_EXPECT_EQ(test, init_stat("fname_ok", &st1, 0), 0);
+ KUNIT_EXPECT_EQ(test, init_rmdir("fname_ok"), 0);
+ }
kfree(tbufs);
}
@@ -539,8 +564,11 @@ static void __init initramfs_test_hdr_hex(struct kunit *test)
/* inject_ox=true to add "0x" cpio field prefixes */
len = fill_cpio(c, ARRAY_SIZE(c), true, tbufs->cpio_src);
- err = unpack_to_rootfs(tbufs->cpio_src, len);
- KUNIT_EXPECT_NOT_NULL(test, err);
+ /* Tests run in a nullfs kthread; borrow the init fs for path resolution. */
+ scoped_with_init_fs() {
+ err = unpack_to_rootfs(tbufs->cpio_src, len);
+ KUNIT_EXPECT_NOT_NULL(test, err);
+ }
kfree(tbufs);
}
diff --git a/init/main.c b/init/main.c
index e363232b428b..92d34e496a33 100644
--- a/init/main.c
+++ b/init/main.c
@@ -103,6 +103,7 @@
#include <linux/stackdepot.h>
#include <linux/randomize_kstack.h>
#include <linux/pidfs.h>
+#include <linux/fs_struct.h>
#include <linux/ptdump.h>
#include <linux/time_namespace.h>
#include <linux/unaligned.h>
@@ -670,6 +671,11 @@ static __initdata DECLARE_COMPLETION(kthreadd_done);
static noinline void __ref __noreturn rest_init(void)
{
+ struct kernel_clone_args init_args = {
+ .flags = (CLONE_VM | CLONE_UNTRACED),
+ .fn = kernel_init,
+ .fn_arg = NULL,
+ };
struct task_struct *tsk;
int pid;
@@ -679,7 +685,7 @@ static noinline void __ref __noreturn rest_init(void)
* the init task will end up wanting to create kthreads, which, if
* we schedule it before we create kthreadd, will OOPS.
*/
- pid = user_mode_thread(kernel_init, NULL, CLONE_FS);
+ pid = kernel_clone(&init_args);
/*
* Pin init on the boot CPU. Task migration is not properly working
* until sched_init_smp() has been run. It will set the allowed
@@ -1540,6 +1546,8 @@ static int __ref kernel_init(void *unused)
{
int ret;
+ init_userspace_fs();
+
/*
* Wait until kthreadd is all set-up.
*/
diff --git a/ipc/mqueue.c b/ipc/mqueue.c
index 4798b375972b..d1dd36a651b0 100644
--- a/ipc/mqueue.c
+++ b/ipc/mqueue.c
@@ -608,7 +608,7 @@ out_unlock:
}
static int mqueue_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return mqueue_create_attr(dentry, mode, NULL);
}
@@ -790,7 +790,6 @@ static void __do_notify(struct mqueue_inode_info *info)
struct kernel_siginfo sig_i;
struct task_struct *task;
- /* do_mq_notify() accepts sigev_signo == 0, why?? */
if (!info->notify.sigev_signo)
break;
@@ -1281,9 +1280,9 @@ static int do_mq_notify(mqd_t mqdes, const struct sigevent *notification)
notification->sigev_notify != SIGEV_THREAD))
return -EINVAL;
if (notification->sigev_notify == SIGEV_SIGNAL &&
- !valid_signal(notification->sigev_signo)) {
+ (!notification->sigev_signo ||
+ !valid_signal(notification->sigev_signo)))
return -EINVAL;
- }
if (notification->sigev_notify == SIGEV_THREAD) {
long timeo;
diff --git a/kernel/exit.c b/kernel/exit.c
index 2c0b1c02920f..f812df279ea7 100644
--- a/kernel/exit.c
+++ b/kernel/exit.c
@@ -1116,7 +1116,7 @@ void __noreturn make_task_dead(int signr)
SYSCALL_DEFINE1(exit, int, error_code)
{
- do_exit((error_code&0xff)<<8);
+ do_exit((error_code & 0xff) << 8);
}
/*
diff --git a/kernel/fork.c b/kernel/fork.c
index 516faec8efb8..5a5e9f4e4f54 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -1618,9 +1618,27 @@ static int copy_exec_state(u64 clone_flags, struct task_struct *tsk)
return task_exec_state_copy(tsk);
}
-static int copy_fs(u64 clone_flags, struct task_struct *tsk)
+static int copy_fs(u64 clone_flags, struct task_struct *tsk, bool umh)
{
- struct fs_struct *fs = current->fs;
+ struct fs_struct *fs;
+
+ /*
+ * Usermodehelper may copy userspace_init_fs filesystem state but
+ * they don't get to create mount namespaces, share the
+ * filesystem state, or be started from a non-initial mount
+ * namespace.
+ */
+ if (umh) {
+ if (clone_flags & (CLONE_NEWNS | CLONE_FS))
+ return -EINVAL;
+ if (current->nsproxy->mnt_ns != &init_mnt_ns)
+ return -EINVAL;
+ fs = userspace_init_fs;
+ } else {
+ fs = current->fs;
+ VFS_WARN_ON_ONCE(current->fs != current->real_fs);
+ }
+
if (clone_flags & CLONE_FS) {
/* tsk->fs is already what we want */
read_seqlock_excl(&fs->seq);
@@ -1633,7 +1651,7 @@ static int copy_fs(u64 clone_flags, struct task_struct *tsk)
read_sequnlock_excl(&fs->seq);
return 0;
}
- tsk->fs = copy_fs_struct(fs);
+ tsk->real_fs = tsk->fs = copy_fs_struct(fs);
if (!tsk->fs)
return -ENOMEM;
return 0;
@@ -2277,7 +2295,7 @@ __latent_entropy struct task_struct *copy_process(
retval = copy_files(clone_flags, p, args->no_files);
if (retval)
goto bad_fork_cleanup_semundo;
- retval = copy_fs(clone_flags, p);
+ retval = copy_fs(clone_flags, p, args->umh);
if (retval)
goto bad_fork_cleanup_files;
retval = copy_sighand(clone_flags, p);
@@ -2819,6 +2837,7 @@ pid_t user_mode_thread(int (*fn)(void *), void *arg, unsigned long flags)
.exit_signal = (flags & CSIGNAL),
.fn = fn,
.fn_arg = arg,
+ .umh = 1,
};
return kernel_clone(&args);
@@ -3216,7 +3235,7 @@ static int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp
*/
int ksys_unshare(unsigned long unshare_flags)
{
- struct fs_struct *fs, *new_fs = NULL;
+ struct fs_struct *new_fs = NULL;
struct files_struct *new_fd = NULL;
struct cred *new_cred = NULL;
struct nsproxy *new_nsproxy = NULL;
@@ -3247,6 +3266,10 @@ int ksys_unshare(unsigned long unshare_flags)
if (unshare_flags & CLONE_NEWNS)
unshare_flags |= CLONE_FS;
+ /* No unsharing with overriden fs state */
+ VFS_WARN_ON_ONCE(unshare_flags & (CLONE_NEWNS | CLONE_FS) &&
+ current->fs != current->real_fs);
+
err = check_unshare_flags(unshare_flags);
if (err)
goto bad_unshare_out;
@@ -3294,23 +3317,13 @@ int ksys_unshare(unsigned long unshare_flags)
new_nsproxy = NULL;
}
- task_lock(current);
+ if (new_fs)
+ new_fs = switch_fs_struct(new_fs);
- if (new_fs) {
- fs = current->fs;
- read_seqlock_excl(&fs->seq);
- current->fs = new_fs;
- if (--fs->users)
- new_fs = NULL;
- else
- new_fs = fs;
- read_sequnlock_excl(&fs->seq);
- }
-
- if (new_fd)
+ if (new_fd) {
+ guard(task_lock)(current);
swap(current->files, new_fd);
-
- task_unlock(current);
+ }
if (new_cred) {
/* Install the new user namespace */
diff --git a/kernel/kcmp.c b/kernel/kcmp.c
index 7c1a65bd5f8d..76476aeee067 100644
--- a/kernel/kcmp.c
+++ b/kernel/kcmp.c
@@ -186,7 +186,7 @@ SYSCALL_DEFINE5(kcmp, pid_t, pid1, pid_t, pid2, int, type,
ret = kcmp_ptr(task1->files, task2->files, KCMP_FILES);
break;
case KCMP_FS:
- ret = kcmp_ptr(task1->fs, task2->fs, KCMP_FS);
+ ret = kcmp_ptr(task1->real_fs, task2->real_fs, KCMP_FS);
break;
case KCMP_SIGHAND:
ret = kcmp_ptr(task1->sighand, task2->sighand, KCMP_SIGHAND);
diff --git a/kernel/umh.c b/kernel/umh.c
index 48117c569e1a..6e2c7bb315c6 100644
--- a/kernel/umh.c
+++ b/kernel/umh.c
@@ -71,10 +71,8 @@ static int call_usermodehelper_exec_async(void *data)
spin_unlock_irq(&current->sighand->siglock);
/*
- * Initial kernel threads share ther FS with init, in order to
- * get the init root directory. But we've now created a new
- * thread that is going to execve a user process and has its own
- * 'struct fs_struct'. Reset umask to the default.
+ * Usermodehelper threads get a copy of userspace init's
+ * fs_struct. Reset umask to the default.
*/
current->fs->umask = 0022;
diff --git a/kernel/user.c b/kernel/user.c
index 7aef4e679a6a..21bafdc11379 100644
--- a/kernel/user.c
+++ b/kernel/user.c
@@ -23,9 +23,9 @@
#if IS_ENABLED(CONFIG_BINFMT_MISC)
struct binfmt_misc init_binfmt_misc = {
- .entries = LIST_HEAD_INIT(init_binfmt_misc.entries),
+ .entries = HLIST_HEAD_INIT,
.enabled = true,
- .entries_lock = __RW_LOCK_UNLOCKED(init_binfmt_misc.entries_lock),
+ .entries_lock = __SPIN_LOCK_UNLOCKED(init_binfmt_misc.entries_lock),
};
EXPORT_SYMBOL_GPL(init_binfmt_misc);
#endif
diff --git a/mm/shmem.c b/mm/shmem.c
index 6641823bed16..20c24d92da43 100644
--- a/mm/shmem.c
+++ b/mm/shmem.c
@@ -3886,7 +3886,7 @@ static struct dentry *shmem_mkdir(struct mnt_idmap *idmap, struct inode *dir,
}
static int shmem_create(struct mnt_idmap *idmap, struct inode *dir,
- struct dentry *dentry, umode_t mode, bool excl)
+ struct dentry *dentry, umode_t mode)
{
return shmem_mknod(idmap, dir, dentry, mode | S_IFREG, 0);
}
diff --git a/net/socket.c b/net/socket.c
index 63c69a0fa74e..b0256cd222f8 100644
--- a/net/socket.c
+++ b/net/socket.c
@@ -465,6 +465,31 @@ static const struct xattr_handler sockfs_user_xattr_handler = {
.set = sockfs_user_xattr_set,
};
+/**
+ * sock_read_xattr - read a user.* xattr from a socket's sockfs inode
+ * @sock: socket whose inode holds the xattr
+ * @name: full xattr name, e.g. "user.bpf_test"
+ * @value: output buffer
+ * @size: size of @value in bytes
+ *
+ * SOCK_INODE() is valid only for sockfs sockets; sock_from_file() rejects
+ * anything else (e.g. tun, tap).
+ * Lockless: simple_xattr_get() looks up the value under RCU, no inode lock.
+ *
+ * Return: length of the value on success, a negative errno on error.
+ */
+int sock_read_xattr(struct socket *sock, const char *name, void *value, size_t size)
+{
+ struct file *file = sock->file;
+ struct sockfs_inode *si;
+
+ if (!file || sock_from_file(file) != sock)
+ return -EOPNOTSUPP;
+
+ si = SOCKFS_I(SOCK_INODE(sock));
+ return simple_xattr_get(&sockfs_xa_cache, &si->xattrs, name, value, size);
+}
+
static const struct xattr_handler * const sockfs_xattr_handlers[] = {
&sockfs_xattr_handler,
&sockfs_security_xattr_handler,
diff --git a/net/sunrpc/auth_gss/auth_gss.c b/net/sunrpc/auth_gss/auth_gss.c
index 9d3fb6848f40..8ddc65e894da 100644
--- a/net/sunrpc/auth_gss/auth_gss.c
+++ b/net/sunrpc/auth_gss/auth_gss.c
@@ -2072,7 +2072,11 @@ gss_unwrap_resp_priv(struct rpc_task *task, struct rpc_cred *cred,
goto unwrap_failed;
opaque_len = be32_to_cpup(p++);
offset = (u8 *)(p) - (u8 *)head->iov_base;
- if (offset + opaque_len > rcv_buf->len)
+ if (offset > rcv_buf->len)
+ goto unwrap_failed;
+ if (opaque_len > rcv_buf->len - offset)
+ goto unwrap_failed;
+ if (opaque_len <= GSS_KRB5_TOK_HDR_LEN)
goto unwrap_failed;
maj_stat = gss_unwrap(ctx->gc_gss_ctx, offset,
diff --git a/net/sunrpc/auth_gss/gss_krb5_unseal.c b/net/sunrpc/auth_gss/gss_krb5_unseal.c
index b5fb70419faa..4d12d49434c2 100644
--- a/net/sunrpc/auth_gss/gss_krb5_unseal.c
+++ b/net/sunrpc/auth_gss/gss_krb5_unseal.c
@@ -89,6 +89,9 @@ gss_krb5_verify_mic_v2(struct krb5_ctx *ctx, struct xdr_buf *message_buffer,
dprintk("RPC: %s\n", __func__);
+ if (read_token->len < GSS_KRB5_TOK_HDR_LEN + cksum_len)
+ return GSS_S_DEFECTIVE_TOKEN;
+
memcpy(&be16_ptr, (char *) ptr, 2);
if (be16_to_cpu(be16_ptr) != KG2_TOK_MIC)
return GSS_S_DEFECTIVE_TOKEN;
diff --git a/net/sunrpc/auth_gss/gss_krb5_wrap.c b/net/sunrpc/auth_gss/gss_krb5_wrap.c
index ac4b32df42b9..d3f61c4b5a13 100644
--- a/net/sunrpc/auth_gss/gss_krb5_wrap.c
+++ b/net/sunrpc/auth_gss/gss_krb5_wrap.c
@@ -73,6 +73,8 @@ static void _rotate_left(struct xdr_buf *buf, unsigned int shift)
int shifted = 0;
int this_shift;
+ if (!buf->len)
+ return;
shift %= buf->len;
while (shifted < shift) {
this_shift = min(shift - shifted, LOCAL_BUF_LEN);
@@ -85,6 +87,8 @@ static void rotate_left(u32 base, struct xdr_buf *buf, unsigned int shift)
{
struct xdr_buf subbuf;
+ if (buf->len <= base)
+ return;
xdr_buf_subsegment(buf, &subbuf, base, buf->len - base);
_rotate_left(&subbuf, shift);
}
@@ -154,6 +158,9 @@ gss_krb5_unwrap_v2(struct krb5_ctx *kctx, int offset, int len,
dprintk("RPC: %s\n", __func__);
+ if (len - offset <= GSS_KRB5_TOK_HDR_LEN)
+ return GSS_S_DEFECTIVE_TOKEN;
+
ptr = buf->head[0].iov_base + offset;
if (be16_to_cpu(*((__be16 *)ptr)) != KG2_TOK_WRAP)
@@ -220,14 +227,16 @@ gss_krb5_unwrap_v2(struct krb5_ctx *kctx, int offset, int len,
* head buffer space rather than that actually occupied.
*/
movelen = min_t(unsigned int, buf->head[0].iov_len, len);
+ if (movelen < offset + GSS_KRB5_TOK_HDR_LEN + headskip)
+ return GSS_S_DEFECTIVE_TOKEN;
movelen -= offset + GSS_KRB5_TOK_HDR_LEN + headskip;
- BUG_ON(offset + GSS_KRB5_TOK_HDR_LEN + headskip + movelen >
- buf->head[0].iov_len);
memmove(ptr, ptr + GSS_KRB5_TOK_HDR_LEN + headskip, movelen);
buf->head[0].iov_len -= GSS_KRB5_TOK_HDR_LEN + headskip;
buf->len = len - (GSS_KRB5_TOK_HDR_LEN + headskip);
/* Trim off the trailing "extra count" and checksum blob */
+ if (ec + GSS_KRB5_TOK_HDR_LEN + tailskip > buf->len - offset)
+ return GSS_S_DEFECTIVE_TOKEN;
xdr_buf_trim(buf, ec + GSS_KRB5_TOK_HDR_LEN + tailskip);
*align = XDR_QUADLEN(GSS_KRB5_TOK_HDR_LEN + headskip);
diff --git a/net/sunrpc/auth_gss/gss_rpc_upcall.c b/net/sunrpc/auth_gss/gss_rpc_upcall.c
index 0fa4778620d9..b7f70b1adb18 100644
--- a/net/sunrpc/auth_gss/gss_rpc_upcall.c
+++ b/net/sunrpc/auth_gss/gss_rpc_upcall.c
@@ -121,12 +121,6 @@ out:
return result;
}
-void init_gssp_clnt(struct sunrpc_net *sn)
-{
- mutex_init(&sn->gssp_lock);
- sn->gssp_clnt = NULL;
-}
-
int set_gssp_clnt(struct net *net)
{
struct sunrpc_net *sn = net_generic(net, sunrpc_net_id);
diff --git a/net/sunrpc/auth_gss/gss_rpc_upcall.h b/net/sunrpc/auth_gss/gss_rpc_upcall.h
index 31e96344167e..b3c2b2b90798 100644
--- a/net/sunrpc/auth_gss/gss_rpc_upcall.h
+++ b/net/sunrpc/auth_gss/gss_rpc_upcall.h
@@ -29,7 +29,6 @@ int gssp_accept_sec_context_upcall(struct net *net,
struct gssp_upcall_data *data);
void gssp_free_upcall_data(struct gssp_upcall_data *data);
-void init_gssp_clnt(struct sunrpc_net *);
int set_gssp_clnt(struct net *);
void clear_gssp_clnt(struct sunrpc_net *);
diff --git a/net/sunrpc/auth_gss/gss_rpc_xdr.c b/net/sunrpc/auth_gss/gss_rpc_xdr.c
index fceee648d545..0549edae1ebe 100644
--- a/net/sunrpc/auth_gss/gss_rpc_xdr.c
+++ b/net/sunrpc/auth_gss/gss_rpc_xdr.c
@@ -222,7 +222,8 @@ static int gssx_dec_linux_creds(struct xdr_stream *xdr,
return 0;
out_free_groups:
- groups_free(creds->cr_group_info);
+ put_group_info(creds->cr_group_info);
+ creds->cr_group_info = NULL;
return err;
}
@@ -230,6 +231,7 @@ static int gssx_dec_option_array(struct xdr_stream *xdr,
struct gssx_option_array *oa)
{
struct svc_cred *creds;
+ bool creds_decoded = false;
u32 count, i;
__be32 *p;
int err;
@@ -242,12 +244,12 @@ static int gssx_dec_option_array(struct xdr_stream *xdr,
return 0;
/* we recognize only 1 currently: CREDS_VALUE */
- oa->count = 1;
-
oa->data = kmalloc_obj(struct gssx_option);
if (!oa->data)
return -ENOMEM;
+ oa->count = 1;
+
creds = kzalloc_obj(struct svc_cred);
if (!creds) {
err = -ENOMEM;
@@ -280,9 +282,14 @@ static int gssx_dec_option_array(struct xdr_stream *xdr,
if (length == sizeof(CREDS_VALUE) &&
memcmp(p, CREDS_VALUE, sizeof(CREDS_VALUE)) == 0) {
/* We have creds here. parse them */
+ if (creds_decoded) {
+ err = -EINVAL;
+ goto free_creds;
+ }
err = gssx_dec_linux_creds(xdr, creds);
if (err)
goto free_creds;
+ creds_decoded = true;
oa->data[0].value.len = 1; /* presence */
} else {
/* consume uninteresting buffer */
@@ -294,8 +301,10 @@ static int gssx_dec_option_array(struct xdr_stream *xdr,
return 0;
free_creds:
+ free_svc_cred(creds);
kfree(creds);
free_oa:
+ oa->count = 0;
kfree(oa->data);
oa->data = NULL;
return err;
diff --git a/net/sunrpc/auth_gss/svcauth_gss.c b/net/sunrpc/auth_gss/svcauth_gss.c
index d14209031e18..967e9d53080d 100644
--- a/net/sunrpc/auth_gss/svcauth_gss.c
+++ b/net/sunrpc/auth_gss/svcauth_gss.c
@@ -949,6 +949,8 @@ svcauth_gss_unwrap_priv(struct svc_rqst *rqstp, u32 seq, struct gss_ctx *ctx)
}
if (len > xdr_stream_remaining(xdr))
goto unwrap_failed;
+ if (len <= GSS_KRB5_TOK_HDR_LEN)
+ goto unwrap_failed;
offset = xdr_stream_pos(xdr);
saved_len = buf->len;
@@ -1466,7 +1468,6 @@ static int create_use_gss_proxy_proc_entry(struct net *net)
&use_gss_proxy_proc_ops, net);
if (!*p)
return -ENOMEM;
- init_gssp_clnt(sn);
return 0;
}
@@ -1573,6 +1574,9 @@ svcauth_gss_decode_credbody(struct xdr_stream *xdr,
u32 body_len;
__be32 *p;
+ /* Early-return paths leave deterministic state, not stale residue. */
+ memset(gc, 0, sizeof(*gc));
+
p = xdr_inline_decode(xdr, XDR_UNIT);
if (!p)
return false;
@@ -1942,6 +1946,8 @@ svcauth_gss_release(struct svc_rqst *rqstp)
if (!gsd)
goto out;
+ if (rqstp->rq_auth_stat != rpc_auth_ok)
+ goto out;
gc = &gsd->clcred;
if (gc->gc_proc != RPC_GSS_PROC_DATA)
goto out;
diff --git a/net/sunrpc/backchannel_rqst.c b/net/sunrpc/backchannel_rqst.c
index 0ffa4d01a938..1482b06e0f38 100644
--- a/net/sunrpc/backchannel_rqst.c
+++ b/net/sunrpc/backchannel_rqst.c
@@ -25,20 +25,39 @@ unsigned int xprt_bc_max_slots(struct rpc_xprt *xprt)
}
/*
- * Helper function to nullify backchannel server pointer in transport.
- * We need to synchronize setting the pointer to NULL (done so after
- * the backchannel server is shutdown) with the usage of that pointer
- * by the backchannel request processing routines
- * xprt_complete_bc_request() and rpcrdma_bc_receive_call().
+ * Close the backchannel producer side, drain any requests still
+ * queued on sv_cb_list, then destroy the callback service.
*/
void xprt_svc_destroy_nullify_bc(struct rpc_xprt *xprt, struct svc_serv **serv)
{
- spin_lock(&xprt->bc_pa_lock);
+ struct svc_serv *bc_serv = *serv;
+ struct rpc_rqst *req;
+
+ xprt_svc_shutdown_bc(xprt);
+ while ((req = lwq_dequeue(&bc_serv->sv_cb_list, struct rpc_rqst,
+ rq_bc_list)) != NULL) {
+ atomic_dec(&req->rq_xprt->bc_slot_count);
+ xprt_free_bc_request(req);
+ }
svc_destroy(serv);
+}
+EXPORT_SYMBOL_GPL(xprt_svc_destroy_nullify_bc);
+
+/*
+ * Clear the backchannel server pointer in the transport. The NULL
+ * store is serialized under bc_pa_lock against readers of
+ * xprt->bc_serv in xprt_complete_bc_request() and
+ * rpcrdma_bc_receive_call(). Clearing it before the callback service
+ * is stopped prevents a producer from enqueueing onto a service that
+ * is being torn down.
+ */
+void xprt_svc_shutdown_bc(struct rpc_xprt *xprt)
+{
+ spin_lock(&xprt->bc_pa_lock);
xprt->bc_serv = NULL;
spin_unlock(&xprt->bc_pa_lock);
}
-EXPORT_SYMBOL_GPL(xprt_svc_destroy_nullify_bc);
+EXPORT_SYMBOL_GPL(xprt_svc_shutdown_bc);
/*
* Helper routines that track the number of preallocation elements
@@ -393,7 +412,12 @@ void xprt_enqueue_bc_request(struct rpc_rqst *req)
if (bc_serv) {
lwq_enqueue(&req->rq_bc_list, &bc_serv->sv_cb_list);
svc_pool_wake_idle_thread(&bc_serv->sv_pools[0]);
+ spin_unlock(&xprt->bc_pa_lock);
+ return;
}
spin_unlock(&xprt->bc_pa_lock);
+
+ atomic_dec(&xprt->bc_slot_count);
+ xprt_free_bc_request(req);
}
EXPORT_SYMBOL_GPL(xprt_enqueue_bc_request);
diff --git a/net/sunrpc/cache.c b/net/sunrpc/cache.c
index 391037f15292..1bc04109d213 100644
--- a/net/sunrpc/cache.c
+++ b/net/sunrpc/cache.c
@@ -430,10 +430,9 @@ void sunrpc_destroy_cache_detail(struct cache_detail *cd)
list_del_init(&cd->others);
spin_unlock(&cd->hash_lock);
spin_unlock(&cache_list_lock);
- if (list_empty(&cache_list)) {
- /* module must be being unloaded so its safe to kill the worker */
- cancel_delayed_work_sync(&cache_cleaner);
- }
+ cancel_delayed_work_sync(&cache_cleaner);
+ if (!list_empty(&cache_list))
+ queue_delayed_work(system_power_efficient_wq, &cache_cleaner, 0);
}
EXPORT_SYMBOL_GPL(sunrpc_destroy_cache_detail);
diff --git a/net/sunrpc/sunrpc_syms.c b/net/sunrpc/sunrpc_syms.c
index ab88ce46afb5..1a3884a0376a 100644
--- a/net/sunrpc/sunrpc_syms.c
+++ b/net/sunrpc/sunrpc_syms.c
@@ -57,6 +57,7 @@ static __net_init int sunrpc_init_net(struct net *net)
INIT_LIST_HEAD(&sn->all_clients);
spin_lock_init(&sn->rpc_client_lock);
spin_lock_init(&sn->rpcb_clnt_lock);
+ mutex_init(&sn->gssp_lock);
return 0;
err_pipefs:
diff --git a/net/sunrpc/svc.c b/net/sunrpc/svc.c
index ae9ec4bf34f7..dd80a2eaaa74 100644
--- a/net/sunrpc/svc.c
+++ b/net/sunrpc/svc.c
@@ -476,6 +476,35 @@ __svc_init_bc(struct svc_serv *serv)
}
#endif
+static int svc_pool_init_counters(struct svc_pool *pool)
+{
+ int err;
+
+ err = percpu_counter_init(&pool->sp_messages_arrived, 0, GFP_KERNEL);
+ if (err)
+ return err;
+ err = percpu_counter_init(&pool->sp_sockets_queued, 0, GFP_KERNEL);
+ if (err)
+ goto err_sockets;
+ err = percpu_counter_init(&pool->sp_threads_woken, 0, GFP_KERNEL);
+ if (err)
+ goto err_threads;
+ return 0;
+
+err_threads:
+ percpu_counter_destroy(&pool->sp_sockets_queued);
+err_sockets:
+ percpu_counter_destroy(&pool->sp_messages_arrived);
+ return err;
+}
+
+static void svc_pool_destroy_counters(struct svc_pool *pool)
+{
+ percpu_counter_destroy(&pool->sp_messages_arrived);
+ percpu_counter_destroy(&pool->sp_sockets_queued);
+ percpu_counter_destroy(&pool->sp_threads_woken);
+}
+
/*
* Create an RPC service
*/
@@ -540,12 +569,18 @@ __svc_create(struct svc_program *prog, int nprogs, struct svc_stat *stats,
INIT_LIST_HEAD(&pool->sp_all_threads);
init_llist_head(&pool->sp_idle_threads);
- percpu_counter_init(&pool->sp_messages_arrived, 0, GFP_KERNEL);
- percpu_counter_init(&pool->sp_sockets_queued, 0, GFP_KERNEL);
- percpu_counter_init(&pool->sp_threads_woken, 0, GFP_KERNEL);
+ if (svc_pool_init_counters(pool))
+ goto out_err;
}
return serv;
+
+out_err:
+ while (i--)
+ svc_pool_destroy_counters(&serv->sv_pools[i]);
+ kfree(serv->sv_pools);
+ kfree(serv);
+ return NULL;
}
/**
@@ -624,9 +659,7 @@ svc_destroy(struct svc_serv **servp)
for (i = 0; i < serv->sv_nrpools; i++) {
struct svc_pool *pool = &serv->sv_pools[i];
- percpu_counter_destroy(&pool->sp_messages_arrived);
- percpu_counter_destroy(&pool->sp_sockets_queued);
- percpu_counter_destroy(&pool->sp_threads_woken);
+ svc_pool_destroy_counters(pool);
}
kfree(serv->sv_pools);
kfree(serv);
@@ -683,6 +716,15 @@ svc_release_buffer(struct svc_rqst *rqstp)
}
}
+static void svc_rqst_free_rcu(struct rcu_head *head)
+{
+ struct svc_rqst *rqstp = container_of(head, struct svc_rqst, rq_rcu_head);
+
+ kfree(rqstp->rq_resp);
+ kfree(rqstp->rq_argp);
+ kfree(rqstp);
+}
+
static void
svc_rqst_free(struct svc_rqst *rqstp)
{
@@ -691,10 +733,8 @@ svc_rqst_free(struct svc_rqst *rqstp)
svc_release_buffer(rqstp);
if (rqstp->rq_scratch_folio)
folio_put(rqstp->rq_scratch_folio);
- kfree(rqstp->rq_resp);
- kfree(rqstp->rq_argp);
kfree(rqstp->rq_auth_data);
- kfree_rcu(rqstp, rq_rcu_head);
+ call_rcu(&rqstp->rq_rcu_head, svc_rqst_free_rcu);
}
static struct svc_rqst *
@@ -955,6 +995,29 @@ svc_set_num_threads(struct svc_serv *serv, unsigned int min_threads,
EXPORT_SYMBOL_GPL(svc_set_num_threads);
/**
+ * svc_serv_maxthreads - report a service's configured thread ceiling
+ * @serv: RPC service to query
+ *
+ * A pooled service sizes its threads dynamically, so the number of
+ * threads running at any moment tracks recent load rather than the
+ * service's capacity. The per-pool maximum is the stable figure a
+ * consumer should size against.
+ *
+ * The caller must keep @serv valid for the duration of the call.
+ *
+ * Return: the sum of every pool's maximum thread count.
+ */
+unsigned int svc_serv_maxthreads(const struct svc_serv *serv)
+{
+ unsigned int i, max = 0;
+
+ for (i = 0; i < serv->sv_nrpools; i++)
+ max += data_race(serv->sv_pools[i].sp_nrthrmax);
+ return max;
+}
+EXPORT_SYMBOL_GPL(svc_serv_maxthreads);
+
+/**
* svc_rqst_replace_page - Replace one page in rq_respages[]
* @rqstp: svc_rqst with pages to replace
* @page: replacement page
diff --git a/net/sunrpc/svcauth_unix.c b/net/sunrpc/svcauth_unix.c
index 64a2658faddb..aebd97e7f66c 100644
--- a/net/sunrpc/svcauth_unix.c
+++ b/net/sunrpc/svcauth_unix.c
@@ -133,7 +133,7 @@ static void ip_map_init(struct cache_head *cnew, struct cache_head *citem)
struct ip_map *new = container_of(cnew, struct ip_map, h);
struct ip_map *item = container_of(citem, struct ip_map, h);
- strcpy(new->m_class, item->m_class);
+ strscpy(new->m_class, item->m_class);
new->m_addr = item->m_addr;
}
static void update(struct cache_head *cnew, struct cache_head *citem)
@@ -296,7 +296,7 @@ static struct ip_map *__ip_map_lookup(struct cache_detail *cd, char *class,
struct ip_map ip;
struct cache_head *ch;
- strcpy(ip.m_class, class);
+ strscpy(ip.m_class, class);
ip.m_addr = *addr;
ch = sunrpc_cache_lookup_rcu(cd, &ip.h,
hash_str(class, IP_HASHBITS) ^
diff --git a/net/sunrpc/xdr.c b/net/sunrpc/xdr.c
index fa6a30b5f046..cb2ef428651f 100644
--- a/net/sunrpc/xdr.c
+++ b/net/sunrpc/xdr.c
@@ -2049,7 +2049,7 @@ void xdr_buf_trim(struct xdr_buf *buf, unsigned int len)
trim -= cur;
}
fix_len:
- buf->len -= (len - trim);
+ buf->len -= min_t(unsigned int, buf->len, len - trim);
}
EXPORT_SYMBOL_GPL(xdr_buf_trim);
diff --git a/net/sunrpc/xprtrdma/ib_client.c b/net/sunrpc/xprtrdma/ib_client.c
index de49ad02053d..188f7a13397f 100644
--- a/net/sunrpc/xprtrdma/ib_client.c
+++ b/net/sunrpc/xprtrdma/ib_client.c
@@ -51,7 +51,11 @@ static struct rpcrdma_device *rpcrdma_get_client_data(struct ib_device *device)
* to be invoked when the device is removed, unless this notification
* is unregistered first.
*
- * On failure, a negative errno is returned.
+ * On failure, a negative errno is returned. rn->rn_done is left
+ * NULL on every failure path (it is armed before xa_alloc but
+ * cleared again if xa_alloc fails), so the @rn may safely be
+ * passed to rpcrdma_rn_unregister() without a separate
+ * registered/unregistered flag in the caller.
*/
int rpcrdma_rn_register(struct ib_device *device,
struct rpcrdma_notification *rn,
@@ -62,10 +66,21 @@ int rpcrdma_rn_register(struct ib_device *device,
if (!rd || test_bit(RPCRDMA_RD_F_REMOVING, &rd->rd_flags))
return -ENETUNREACH;
- if (xa_alloc(&rd->rd_xa, &rn->rn_index, rn, xa_limit_32b, GFP_KERNEL) < 0)
+ /*
+ * Arm rn_done before xa_alloc() publishes @rn: once @rn is
+ * visible in rd_xa, a concurrent rpcrdma_remove_one() can
+ * call rn->rn_done(), so the pointer must already be set.
+ *
+ * Restore NULL if xa_alloc() fails. rn_done doubles as the
+ * registration sentinel for rpcrdma_rn_unregister(); a stale
+ * value would unregister an @rn that was never inserted.
+ */
+ rn->rn_done = done;
+ if (xa_alloc(&rd->rd_xa, &rn->rn_index, rn, xa_limit_32b, GFP_KERNEL) < 0) {
+ rn->rn_done = NULL;
return -ENOMEM;
+ }
kref_get(&rd->rd_kref);
- rn->rn_done = done;
trace_rpcrdma_client_register(device, rn);
return 0;
}
@@ -83,6 +98,10 @@ static void rpcrdma_rn_release(struct kref *kref)
* rpcrdma_rn_unregister - stop device removal notifications
* @device: monitored device
* @rn: notification object that no longer wishes to be notified
+ *
+ * It is safe to call this on an @rn whose registration never
+ * completed or failed; rn_done == NULL is treated as
+ * never-registered and the call is a no-op.
*/
void rpcrdma_rn_unregister(struct ib_device *device,
struct rpcrdma_notification *rn)
@@ -92,6 +111,21 @@ void rpcrdma_rn_unregister(struct ib_device *device,
if (!rd)
return;
+ /*
+ * rn_done is the registration sentinel: rpcrdma_rn_register
+ * leaves it NULL on every failure path, clearing it again if
+ * xa_alloc fails, so a non-NULL rn_done marks a completed
+ * registration. A NULL rn_done means this notification was
+ * never registered (or its registration failed) or has
+ * already been unregistered, and the call is a no-op.
+ * Without this guard, rn_index == 0 from a kzalloc'd
+ * parent would erase another caller's slot 0 and underflow
+ * rd_kref.
+ */
+ if (!rn->rn_done)
+ return;
+ rn->rn_done = NULL;
+
trace_rpcrdma_client_unregister(device, rn);
xa_erase(&rd->rd_xa, rn->rn_index);
kref_put(&rd->rd_kref, rpcrdma_rn_release);
diff --git a/net/sunrpc/xprtrdma/svc_rdma_pcl.c b/net/sunrpc/xprtrdma/svc_rdma_pcl.c
index 1f8f7dad8b6f..8623722790f2 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_pcl.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_pcl.c
@@ -149,9 +149,6 @@ bool pcl_alloc_call(struct svc_rdma_recv_ctxt *rctxt, __be32 *p)
* cl_count is updated to be the number of chunks (ie.
* unique position values) in the Read list.
* %false: Memory allocation failed.
- *
- * TODO:
- * - Check for chunk range overlaps
*/
bool pcl_alloc_read(struct svc_rdma_recv_ctxt *rctxt, __be32 *p)
{
@@ -213,7 +210,6 @@ bool pcl_alloc_write(struct svc_rdma_recv_ctxt *rctxt,
chunk = pcl_alloc_chunk(segcount, 0);
if (!chunk)
return false;
- list_add_tail(&chunk->ch_list, &pcl->cl_chunks);
for (j = 0; j < segcount; j++) {
segment = &chunk->ch_segments[j];
@@ -225,7 +221,66 @@ bool pcl_alloc_write(struct svc_rdma_recv_ctxt *rctxt,
chunk->ch_length += segment->rs_length;
chunk->ch_segcount++;
}
+ list_add_tail(&chunk->ch_list, &pcl->cl_chunks);
+ }
+ return true;
+}
+
+/**
+ * pcl_check_read_chunk_positions - Validate Read chunk positions
+ * @rctxt: Ingress receive context with populated chunk lists
+ * @inline_len: Length of the inline RPC body after the transport header
+ *
+ * Read chunk positions are offsets in the unreduced XDR stream
+ * (RFC 8166 Section 3.4.4), so each position includes the
+ * cumulative length of preceding Read chunks. This function
+ * subtracts those lengths to recover the inline-body offset
+ * before comparing against @inline_len or the Call chunk length.
+ *
+ * Rejects frames where a Read chunk's inline-body offset exceeds
+ * the bound, where adjacent Read chunks overlap, or where any
+ * single chunk length exceeds the page budget.
+ *
+ * Return values:
+ * %true: Read chunk positions and lengths are valid
+ * %false: Malformed chunk list detected
+ */
+bool pcl_check_read_chunk_positions(struct svc_rdma_recv_ctxt *rctxt,
+ unsigned int inline_len)
+{
+ unsigned int max_len, bound, total_read;
+ struct svc_rdma_chunk *chunk, *next;
+
+ max_len = rctxt->rc_maxpages << PAGE_SHIFT;
+
+ if (!pcl_is_empty(&rctxt->rc_call_pcl)) {
+ chunk = pcl_first_chunk(&rctxt->rc_call_pcl);
+ if (chunk->ch_length > max_len)
+ return false;
+ bound = chunk->ch_length;
+ } else {
+ bound = inline_len;
+ }
+
+ if (pcl_is_empty(&rctxt->rc_read_pcl))
+ return true;
+
+ total_read = 0;
+ pcl_for_each_chunk(chunk, &rctxt->rc_read_pcl) {
+ if (chunk->ch_position - total_read > bound)
+ return false;
+ if (chunk->ch_length > max_len)
+ return false;
+
+ next = pcl_next_chunk(&rctxt->rc_read_pcl, chunk);
+ if (!next)
+ break;
+
+ if (chunk->ch_position + chunk->ch_length > next->ch_position)
+ return false;
+ total_read += chunk->ch_length;
}
+
return true;
}
diff --git a/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c b/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c
index fe9bf0371b6e..fdfed1be97da 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_recvfrom.c
@@ -440,11 +440,14 @@ static void svc_rdma_build_arg_xdr(struct svc_rqst *rqstp,
* to the first byte past the Read list. rc_read_pcl and
* rc_call_pcl cl_count fields are set to the number of
* Read segments in the list.
- * %false: Read list is corrupt. @rctxt's xdr_stream is left in an
- * unknown state.
+ * %false: Read list is corrupt or exceeds the page budget. @rctxt's
+ * xdr_stream is left in an unknown state.
*/
static bool xdr_count_read_segments(struct svc_rdma_recv_ctxt *rctxt, __be32 *p)
{
+ unsigned int maxlen = rctxt->rc_maxpages << PAGE_SHIFT;
+ unsigned int total_len = 0;
+
rctxt->rc_call_pcl.cl_count = 0;
rctxt->rc_read_pcl.cl_count = 0;
while (xdr_item_is_present(p)) {
@@ -458,6 +461,11 @@ static bool xdr_count_read_segments(struct svc_rdma_recv_ctxt *rctxt, __be32 *p)
xdr_decode_read_segment(p, &position, &handle,
&length, &offset);
+ if (length > maxlen)
+ return false;
+ total_len += length;
+ if (PAGE_ALIGN(total_len) > maxlen)
+ return false;
if (position) {
if (position & 3)
return false;
@@ -508,10 +516,13 @@ static bool xdr_check_write_chunk(struct svc_rdma_recv_ctxt *rctxt)
return false;
/* Before trusting the segcount value enough to use it in
- * a computation, perform a simple range check. This is an
- * arbitrary but sensible limit (ie, not architectural).
+ * a computation, perform a simple range check. A zero
+ * segcount describes no remote buffer and is rejected so
+ * downstream consumers never see a degenerate ch_segcount==0
+ * chunk. The upper bound is an arbitrary but sensible limit
+ * (ie, not architectural).
*/
- if (unlikely(segcount > rctxt->rc_maxpages))
+ if (segcount == 0 || unlikely(segcount > rctxt->rc_maxpages))
return false;
p = xdr_inline_decode(&rctxt->rc_stream,
@@ -719,6 +730,9 @@ static int svc_rdma_xdr_decode_req(struct xdr_buf *rq_arg,
rq_arg->head[0].iov_base = rctxt->rc_stream.p;
hdr_len = xdr_stream_pos(&rctxt->rc_stream);
+ if (!pcl_check_read_chunk_positions(rctxt,
+ rq_arg->head[0].iov_len - hdr_len))
+ goto out_inval;
rq_arg->head[0].iov_len -= hdr_len;
rq_arg->len -= hdr_len;
trace_svcrdma_decode_rqst(rctxt, rdma_argp, hdr_len);
diff --git a/net/sunrpc/xprtrdma/svc_rdma_rw.c b/net/sunrpc/xprtrdma/svc_rdma_rw.c
index f7fd22cc4a59..9aaaade99e6e 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_rw.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_rw.c
@@ -795,7 +795,7 @@ static int svc_rdma_build_read_segment(struct svc_rqst *rqstp,
len -= seg_len;
if (len && ((head->rc_curpage + 1) > rqstp->rq_maxpages))
- goto out_overrun;
+ goto out_put;
}
ret = svc_rdma_rw_ctx_init(rdma, ctxt, segment->rs_offset,
@@ -809,7 +809,8 @@ static int svc_rdma_build_read_segment(struct svc_rqst *rqstp,
cc->cc_sqecount += ret;
return 0;
-out_overrun:
+out_put:
+ svc_rdma_put_rw_ctxt(rdma, ctxt);
trace_svcrdma_page_overrun_err(&cc->cc_cid, head->rc_curpage);
return -EINVAL;
}
@@ -847,7 +848,7 @@ static int svc_rdma_build_read_chunk(struct svc_rqst *rqstp,
* svc_rdma_copy_inline_range - Copy part of the inline content into pages
* @rqstp: RPC transaction context
* @head: context for ongoing I/O
- * @offset: offset into the Receive buffer of region to copy
+ * @offset: offset into the inline content of region to copy
* @remaining: length of region to copy
*
* Take a page at a time from rqstp->rq_pages and copy the inline
@@ -864,9 +865,13 @@ static int svc_rdma_copy_inline_range(struct svc_rqst *rqstp,
unsigned int offset,
unsigned int remaining)
{
- unsigned char *dst, *src = head->rc_recv_buf;
+ unsigned char *dst, *src = head->rc_saved_arg.head[0].iov_base;
+ unsigned int inline_len = head->rc_saved_arg.head[0].iov_len;
unsigned int page_no, numpages;
+ if (offset > inline_len || remaining > inline_len - offset)
+ return -EINVAL;
+
numpages = PAGE_ALIGN(head->rc_pageoff + remaining) >> PAGE_SHIFT;
for (page_no = 0; page_no < numpages; page_no++) {
unsigned int page_len;
@@ -917,9 +922,10 @@ svc_rdma_read_multiple_chunks(struct svc_rqst *rqstp,
{
const struct svc_rdma_pcl *pcl = &head->rc_read_pcl;
struct svc_rdma_chunk *chunk, *next;
- unsigned int start, length;
+ unsigned int inline_len, start, length;
int ret;
+ inline_len = head->rc_saved_arg.head[0].iov_len;
start = 0;
chunk = pcl_first_chunk(pcl);
length = chunk->ch_position;
@@ -937,6 +943,8 @@ svc_rdma_read_multiple_chunks(struct svc_rqst *rqstp,
break;
start += length;
+ if (head->rc_readbytes > next->ch_position)
+ return -EINVAL;
length = next->ch_position - head->rc_readbytes;
ret = svc_rdma_copy_inline_range(rqstp, head, start, length);
if (ret < 0)
@@ -944,7 +952,9 @@ svc_rdma_read_multiple_chunks(struct svc_rqst *rqstp,
}
start += length;
- length = head->rc_byte_len - start;
+ if (start > inline_len)
+ return -EINVAL;
+ length = inline_len - start;
return svc_rdma_copy_inline_range(rqstp, head, start, length);
}
@@ -969,8 +979,12 @@ svc_rdma_read_multiple_chunks(struct svc_rqst *rqstp,
static int svc_rdma_read_data_item(struct svc_rqst *rqstp,
struct svc_rdma_recv_ctxt *head)
{
- return svc_rdma_build_read_chunk(rqstp, head,
- pcl_first_chunk(&head->rc_read_pcl));
+ struct svc_rdma_chunk *chunk = pcl_first_chunk(&head->rc_read_pcl);
+
+ if (chunk->ch_position > head->rc_saved_arg.head[0].iov_len)
+ return -EINVAL;
+
+ return svc_rdma_build_read_chunk(rqstp, head, chunk);
}
/**
@@ -996,17 +1010,20 @@ static int svc_rdma_read_chunk_range(struct svc_rqst *rqstp,
const struct svc_rdma_segment *segment;
int ret;
+ if (!length)
+ return 0;
+
ret = -EINVAL;
pcl_for_each_segment(segment, chunk) {
struct svc_rdma_segment dummy;
- if (offset > segment->rs_length) {
+ if (offset >= segment->rs_length) {
offset -= segment->rs_length;
continue;
}
dummy.rs_handle = segment->rs_handle;
- dummy.rs_length = min_t(u32, length, segment->rs_length) - offset;
+ dummy.rs_length = min_t(u32, length, segment->rs_length - offset);
dummy.rs_offset = segment->rs_offset + offset;
ret = svc_rdma_build_read_segment(rqstp, head, &dummy);
@@ -1015,6 +1032,8 @@ static int svc_rdma_read_chunk_range(struct svc_rqst *rqstp,
head->rc_readbytes += dummy.rs_length;
length -= dummy.rs_length;
+ if (!length)
+ break;
offset = 0;
}
return ret;
@@ -1039,14 +1058,17 @@ static int svc_rdma_read_call_chunk(struct svc_rqst *rqstp,
pcl_first_chunk(&head->rc_call_pcl);
const struct svc_rdma_pcl *pcl = &head->rc_read_pcl;
struct svc_rdma_chunk *chunk, *next;
- unsigned int start, length;
+ unsigned int call_len, start, length;
int ret;
if (pcl_is_empty(pcl))
return svc_rdma_build_read_chunk(rqstp, head, call_chunk);
+ call_len = call_chunk->ch_length;
start = 0;
chunk = pcl_first_chunk(pcl);
+ if (chunk->ch_position > call_len)
+ return -EINVAL;
length = chunk->ch_position;
ret = svc_rdma_read_chunk_range(rqstp, head, call_chunk,
start, length);
@@ -1063,6 +1085,10 @@ static int svc_rdma_read_call_chunk(struct svc_rqst *rqstp,
break;
start += length;
+ if (next->ch_position > call_len)
+ return -EINVAL;
+ if (head->rc_readbytes > next->ch_position)
+ return -EINVAL;
length = next->ch_position - head->rc_readbytes;
ret = svc_rdma_read_chunk_range(rqstp, head, call_chunk,
start, length);
@@ -1071,7 +1097,9 @@ static int svc_rdma_read_call_chunk(struct svc_rqst *rqstp,
}
start += length;
- length = call_chunk->ch_length - start;
+ if (start > call_len)
+ return -EINVAL;
+ length = call_len - start;
return svc_rdma_read_chunk_range(rqstp, head, call_chunk,
start, length);
}
diff --git a/net/sunrpc/xprtrdma/svc_rdma_sendto.c b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
index 7f6d17bf8c1f..c09659b17351 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_sendto.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_sendto.c
@@ -825,20 +825,21 @@ static int svc_rdma_xb_count_sges(const struct xdr_buf *xdr,
}
/**
- * svc_rdma_pull_up_needed - Determine whether to use pull-up
+ * svc_rdma_check_pull_up - Determine whether to use pull-up
* @rdma: controlling transport
* @sctxt: send_ctxt for the Send WR
* @write_pcl: Write chunk list provided by client
* @xdr: xdr_buf containing RPC message to transmit
*
* Returns:
- * %true if pull-up must be used
- * %false otherwise
+ * %1 if pull-up must be used
+ * %0 if pull-up is not needed
+ * %-E2BIG if the reply is too large to be pulled up
*/
-static bool svc_rdma_pull_up_needed(const struct svcxprt_rdma *rdma,
- const struct svc_rdma_send_ctxt *sctxt,
- const struct svc_rdma_pcl *write_pcl,
- const struct xdr_buf *xdr)
+static int svc_rdma_check_pull_up(const struct svcxprt_rdma *rdma,
+ const struct svc_rdma_send_ctxt *sctxt,
+ const struct svc_rdma_pcl *write_pcl,
+ const struct xdr_buf *xdr)
{
/* Resources needed for the transport header */
struct svc_rdma_pullup_data args = {
@@ -850,11 +851,22 @@ static bool svc_rdma_pull_up_needed(const struct svcxprt_rdma *rdma,
ret = pcl_process_nonpayloads(write_pcl, xdr,
svc_rdma_xb_count_sges, &args);
if (ret < 0)
- return false;
+ return 0;
if (args.pd_length < RPCRDMA_PULLUP_THRESH)
- return true;
- return args.pd_num_sges >= rdma->sc_max_send_sges;
+ return 1;
+ if (args.pd_num_sges < rdma->sc_max_send_sges)
+ return 0;
+
+ /*
+ * The reply has too many SGEs to Send inline, so it has to be
+ * linearized into sc_xprt_buf. That buffer holds only
+ * sc_max_req_size bytes, so a larger reply cannot be pulled up.
+ * RFC 8166 Section 4.5.3 requires responding with ERR_CHUNK.
+ */
+ if (args.pd_length > rdma->sc_max_req_size)
+ return -E2BIG;
+ return 1;
}
/**
@@ -910,7 +922,7 @@ static int svc_rdma_xb_linearize(const struct xdr_buf *xdr,
* Assemble the elements of @xdr into the transport header buffer.
*
* Assumptions:
- * pull_up_needed has determined that @xdr will fit in the buffer.
+ * check_pull_up has determined that @xdr will fit in the buffer.
*
* Returns:
* %0 if pull-up was successful
@@ -945,6 +957,7 @@ static int svc_rdma_pull_up_reply_msg(const struct svcxprt_rdma *rdma,
*
* Returns:
* %0 if DMA mapping was successful.
+ * %-E2BIG if the reply is too large to be pulled up
* %-EMSGSIZE if a buffer manipulation problem occurred
* %-EIO if DMA mapping failed
*
@@ -960,6 +973,7 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
.md_rdma = rdma,
.md_ctxt = sctxt,
};
+ int ret;
/* Set up the (persistently-mapped) transport header SGE. */
sctxt->sc_send_wr.num_sge = 1;
@@ -974,7 +988,10 @@ int svc_rdma_map_reply_msg(struct svcxprt_rdma *rdma,
/* For pull-up, svc_rdma_send() will sync the transport header.
* No additional DMA mapping is necessary.
*/
- if (svc_rdma_pull_up_needed(rdma, sctxt, write_pcl, xdr))
+ ret = svc_rdma_check_pull_up(rdma, sctxt, write_pcl, xdr);
+ if (ret < 0)
+ return ret;
+ if (ret)
return svc_rdma_pull_up_reply_msg(rdma, sctxt, write_pcl, xdr);
return pcl_process_nonpayloads(write_pcl, xdr,
@@ -1162,7 +1179,7 @@ int svc_rdma_sendto(struct svc_rqst *rqstp)
&rctxt->rc_reply_pcl, sctxt,
&rqstp->rq_res);
if (ret < 0)
- goto reply_chunk;
+ goto send_err;
rc_size = ret;
}
@@ -1183,10 +1200,10 @@ int svc_rdma_sendto(struct svc_rqst *rqstp)
ret = svc_rdma_send_reply_msg(rdma, sctxt, rctxt, rqstp);
if (ret < 0)
- goto put_ctxt;
+ goto send_err;
return 0;
-reply_chunk:
+send_err:
if (ret != -E2BIG && ret != -EINVAL)
goto put_ctxt;
diff --git a/net/sunrpc/xprtrdma/svc_rdma_transport.c b/net/sunrpc/xprtrdma/svc_rdma_transport.c
index 7ca71741106b..093371f9d245 100644
--- a/net/sunrpc/xprtrdma/svc_rdma_transport.c
+++ b/net/sunrpc/xprtrdma/svc_rdma_transport.c
@@ -43,6 +43,7 @@
*/
#include <linux/interrupt.h>
+#include <linux/module.h>
#include <linux/sched.h>
#include <linux/slab.h>
#include <linux/spinlock.h>
@@ -245,12 +246,16 @@ svc_rdma_parse_connect_private(struct svcxprt_rdma *newxprt,
* structure for the listening endpoint.
*
* This function creates a new xprt for the new connection and enqueues it on
- * the accept queue for the listent xprt. When the listen thread is kicked, it
+ * the accept queue for the listen xprt. When the listen thread is kicked, it
* will call the recvfrom method on the listen xprt which will accept the new
* connection.
+ *
+ * Return values:
+ * %0: Do not destroy @new_cma_id
+ * %1: Destroy @new_cma_id (allocation failure)
*/
-static void handle_connect_req(struct rdma_cm_id *new_cma_id,
- struct rdma_conn_param *param)
+static int handle_connect_req(struct rdma_cm_id *new_cma_id,
+ struct rdma_conn_param *param)
{
struct svcxprt_rdma *listen_xprt = new_cma_id->context;
struct svcxprt_rdma *newxprt;
@@ -260,7 +265,7 @@ static void handle_connect_req(struct rdma_cm_id *new_cma_id,
listen_xprt->sc_xprt.xpt_net,
ibdev_to_node(new_cma_id->device));
if (!newxprt)
- return;
+ return 1;
newxprt->sc_cm_id = new_cma_id;
new_cma_id->context = newxprt;
svc_rdma_parse_connect_private(newxprt, param);
@@ -294,6 +299,7 @@ static void handle_connect_req(struct rdma_cm_id *new_cma_id,
set_bit(XPT_CONN, &listen_xprt->sc_xprt.xpt_flags);
svc_xprt_enqueue(&listen_xprt->sc_xprt);
+ return 0;
}
/**
@@ -317,14 +323,14 @@ static int svc_rdma_listen_handler(struct rdma_cm_id *cma_id,
switch (event->event) {
case RDMA_CM_EVENT_CONNECT_REQUEST:
- handle_connect_req(cma_id, &event->param.conn);
- break;
+ return handle_connect_req(cma_id, &event->param.conn);
case RDMA_CM_EVENT_ADDR_CHANGE:
listen_id = svc_rdma_create_listen_id(cma_rdma->xpt_net,
sap, cma_xprt);
if (IS_ERR(listen_id)) {
pr_err("Listener dead, address change failed for device %s\n",
cma_id->device->name);
+ cma_xprt->sc_cm_id = NULL;
} else
cma_xprt->sc_cm_id = listen_id;
return 1;
@@ -387,7 +393,13 @@ static struct svc_xprt *svc_rdma_create(struct svc_serv *serv,
listen_id = svc_rdma_create_listen_id(net, sa, cma_xprt);
if (IS_ERR(listen_id)) {
- kfree(cma_xprt);
+ /* _svc_xprt_create() acquired one module reference and
+ * puts it on xpo_create failure. svc_xprt_free() puts
+ * a second one when the kref drops to zero. Take a
+ * compensating reference so both puts are balanced.
+ */
+ __module_get(cma_xprt->sc_xprt.xpt_class->xcl_owner);
+ svc_xprt_put(&cma_xprt->sc_xprt);
return ERR_CAST(listen_id);
}
cma_xprt->sc_cm_id = listen_id;
@@ -598,13 +610,26 @@ static struct svc_xprt *svc_rdma_accept(struct svc_xprt *xprt)
return &newxprt->sc_xprt;
errout:
- /* Take a reference in case the DTO handler runs */
- svc_xprt_get(&newxprt->sc_xprt);
- if (newxprt->sc_qp && !IS_ERR(newxprt->sc_qp))
- ib_destroy_qp(newxprt->sc_qp);
- rdma_destroy_id(newxprt->sc_cm_id);
- rpcrdma_rn_unregister(dev, &newxprt->sc_rn);
- /* This call to put will destroy the transport */
+ /*
+ * Drop the kref_init birth reference. svc_xprt_free will
+ * dispatch xpo_free = svc_rdma_free, which tears down sc_qp,
+ * sc_sq_cq, sc_rq_cq, and sc_pd under existing IS_ERR/NULL
+ * guards, and sc_rn under the rn_done sentinel guard inside
+ * rpcrdma_rn_unregister.
+ *
+ * sc_cm_id is destroyed unconditionally by svc_rdma_free; that
+ * is safe here because sc_cm_id is non-NULL by caller invariant
+ * on every path that reaches this errout: handle_connect_req
+ * installs newxprt->sc_cm_id before queueing the new xprt for
+ * accept, and svc_rdma_accept has already dereferenced it above
+ * the first goto errout.
+ *
+ * svc_handle_xprt() drops its pre-acquired module reference when
+ * ->xpo_accept() returns NULL. Take a replacement reference before
+ * freeing @newxprt, because svc_xprt_free() drops the module
+ * reference associated with @newxprt.
+ */
+ __module_get(newxprt->sc_xprt.xpt_class->xcl_owner);
svc_xprt_put(&newxprt->sc_xprt);
return NULL;
}
@@ -614,7 +639,8 @@ static void svc_rdma_detach(struct svc_xprt *xprt)
struct svcxprt_rdma *rdma =
container_of(xprt, struct svcxprt_rdma, sc_xprt);
- rdma_disconnect(rdma->sc_cm_id);
+ if (rdma->sc_cm_id)
+ rdma_disconnect(rdma->sc_cm_id);
/*
* Most close paths go through svc_rdma_xprt_deferred_close(),
@@ -634,10 +660,15 @@ static void svc_rdma_free(struct svc_xprt *xprt)
{
struct svcxprt_rdma *rdma =
container_of(xprt, struct svcxprt_rdma, sc_xprt);
- struct ib_device *device = rdma->sc_cm_id->device;
+ struct ib_device *device;
might_sleep();
+ if (!rdma->sc_cm_id)
+ goto out_free;
+
+ device = rdma->sc_cm_id->device;
+
/* This blocks until the Completion Queues are empty */
if (rdma->sc_qp && !IS_ERR(rdma->sc_qp))
ib_drain_qp(rdma->sc_qp);
@@ -662,11 +693,13 @@ static void svc_rdma_free(struct svc_xprt *xprt)
if (rdma->sc_pd && !IS_ERR(rdma->sc_pd))
ib_dealloc_pd(rdma->sc_pd);
+ if (!test_bit(XPT_LISTENER, &rdma->sc_xprt.xpt_flags))
+ rpcrdma_rn_unregister(device, &rdma->sc_rn);
+
/* Destroy the CM ID */
rdma_destroy_id(rdma->sc_cm_id);
- if (!test_bit(XPT_LISTENER, &rdma->sc_xprt.xpt_flags))
- rpcrdma_rn_unregister(device, &rdma->sc_rn);
+out_free:
kfree(rdma);
}
diff --git a/net/unix/af_unix.c b/net/unix/af_unix.c
index 10ed9421e43a..7794740fa80f 100644
--- a/net/unix/af_unix.c
+++ b/net/unix/af_unix.c
@@ -1197,17 +1197,12 @@ static struct sock *unix_find_bsd(struct sockaddr_un *sunaddr, int addr_len,
unix_mkname_bsd(sunaddr, addr_len);
if (flags & SOCK_COREDUMP) {
- struct path root;
-
- task_lock(&init_task);
- get_fs_root(init_task.fs, &root);
- task_unlock(&init_task);
-
- scoped_with_kernel_creds()
- err = vfs_path_lookup(root.dentry, root.mnt, sunaddr->sun_path,
- LOOKUP_BENEATH | LOOKUP_NO_SYMLINKS |
- LOOKUP_NO_MAGICLINKS, &path);
- path_put(&root);
+ scoped_with_init_fs() {
+ scoped_with_kernel_creds()
+ err = kern_path(sunaddr->sun_path,
+ LOOKUP_NO_SYMLINKS |
+ LOOKUP_NO_MAGICLINKS, &path);
+ }
if (err)
goto fail;
} else {
diff --git a/tools/include/uapi/linux/fscrypt.h b/tools/include/uapi/linux/fscrypt.h
index 3aff99f2696a..84507280b3ea 100644
--- a/tools/include/uapi/linux/fscrypt.h
+++ b/tools/include/uapi/linux/fscrypt.h
@@ -30,7 +30,6 @@
#define FSCRYPT_MODE_SM4_CTS 8
#define FSCRYPT_MODE_ADIANTUM 9
#define FSCRYPT_MODE_AES_256_HCTR2 10
-/* If adding a mode number > 10, update FSCRYPT_MODE_MAX in fscrypt_private.h */
/*
* Legacy policy version; ad-hoc KDF and no key verification.
diff --git a/tools/testing/selftests/bpf/bpf_experimental.h b/tools/testing/selftests/bpf/bpf_experimental.h
index 67ff7882299e..63e4e472fe36 100644
--- a/tools/testing/selftests/bpf/bpf_experimental.h
+++ b/tools/testing/selftests/bpf/bpf_experimental.h
@@ -364,6 +364,9 @@ extern void bpf_iter_dmabuf_destroy(struct bpf_iter_dmabuf *it) __weak __ksym;
extern int bpf_cgroup_read_xattr(struct cgroup *cgroup, const char *name__str,
struct bpf_dynptr *value_p) __weak __ksym;
+extern int bpf_sock_read_xattr(struct socket *sock, const char *name__str,
+ struct bpf_dynptr *value_p) __weak __ksym;
+
#define PREEMPT_BITS 8
#define SOFTIRQ_BITS 8
#define HARDIRQ_BITS 4
diff --git a/tools/testing/selftests/bpf/prog_tests/sock_xattr.c b/tools/testing/selftests/bpf/prog_tests/sock_xattr.c
new file mode 100644
index 000000000000..b5816e90f01a
--- /dev/null
+++ b/tools/testing/selftests/bpf/prog_tests/sock_xattr.c
@@ -0,0 +1,67 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/* Copyright (c) 2026 Christian Brauner */
+
+#include <errno.h>
+#include <string.h>
+#include <unistd.h>
+#include <sys/xattr.h>
+#include <sys/socket.h>
+#include <netinet/in.h>
+#include <test_progs.h>
+
+#include "sock_read_xattr.skel.h"
+
+static const char xattr_value[] = "bpf_sock_value";
+static const char xattr_name[] = "user.bpf_test";
+
+static void test_read_sock_xattr(void)
+{
+ struct sockaddr_in addr = {};
+ struct sock_read_xattr *skel = NULL;
+ struct bpf_link *link = NULL;
+ int sock_fd = -1, err;
+
+ sock_fd = socket(AF_INET, SOCK_STREAM, 0);
+ if (!ASSERT_OK_FD(sock_fd, "socket"))
+ return;
+
+ err = fsetxattr(sock_fd, xattr_name, xattr_value, sizeof(xattr_value), 0);
+ if (!ASSERT_OK(err, "fsetxattr"))
+ goto out;
+
+ skel = sock_read_xattr__open_and_load();
+ if (!ASSERT_OK_PTR(skel, "sock_read_xattr__open_and_load"))
+ goto out;
+
+ skel->bss->monitored_pid = sys_gettid();
+
+ /* Only attach the functional program; the verifier-only programs
+ * above are not pid-gated and would clobber the shared globals.
+ */
+ link = bpf_program__attach(skel->progs.read_sock_xattr);
+ if (!ASSERT_OK_PTR(link, "attach read_sock_xattr"))
+ goto out;
+
+ addr.sin_family = AF_INET;
+ addr.sin_port = htons(1234);
+ addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK);
+ /* Only the lsm/socket_connect hook matters; the connect may fail. */
+ connect(sock_fd, (struct sockaddr *)&addr, sizeof(addr));
+
+ ASSERT_EQ(skel->data->read_ret, sizeof(xattr_value), "read_ret");
+ ASSERT_STREQ(skel->bss->value, xattr_value, "value");
+
+out:
+ bpf_link__destroy(link);
+ if (sock_fd >= 0)
+ close(sock_fd);
+ sock_read_xattr__destroy(skel);
+}
+
+void test_sock_xattr(void)
+{
+ RUN_TESTS(sock_read_xattr);
+
+ if (test__start_subtest("read_sock_xattr"))
+ test_read_sock_xattr();
+}
diff --git a/tools/testing/selftests/bpf/progs/sock_read_xattr.c b/tools/testing/selftests/bpf/progs/sock_read_xattr.c
new file mode 100644
index 000000000000..c4a8eae8cc3c
--- /dev/null
+++ b/tools/testing/selftests/bpf/progs/sock_read_xattr.c
@@ -0,0 +1,54 @@
+// SPDX-License-Identifier: GPL-2.0
+/* Copyright (c) 2026 Christian Brauner */
+
+#include <vmlinux.h>
+#include <bpf/bpf_tracing.h>
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_core_read.h>
+#include "bpf_experimental.h"
+#include "bpf_misc.h"
+
+char _license[] SEC("license") = "GPL";
+
+char value[16];
+int read_ret = -1;
+__u32 monitored_pid = 0;
+
+static __always_inline void read_xattr(struct socket *sock)
+{
+ struct bpf_dynptr value_ptr;
+
+ bpf_dynptr_from_mem(value, sizeof(value), 0, &value_ptr);
+ bpf_sock_read_xattr(sock, "user.bpf_test", &value_ptr);
+}
+
+SEC("lsm.s/socket_connect")
+__success
+int BPF_PROG(trusted_sock_ptr_sleepable, struct socket *sock)
+{
+ read_xattr(sock);
+ return 0;
+}
+
+SEC("lsm/socket_connect")
+__success
+int BPF_PROG(trusted_sock_ptr_non_sleepable, struct socket *sock)
+{
+ read_xattr(sock);
+ return 0;
+}
+
+SEC("lsm.s/socket_connect")
+__success
+int BPF_PROG(read_sock_xattr, struct socket *sock)
+{
+ struct bpf_dynptr value_ptr;
+ __u32 pid = bpf_get_current_pid_tgid() >> 32;
+
+ if (pid != monitored_pid)
+ return 0;
+
+ bpf_dynptr_from_mem(value, sizeof(value), 0, &value_ptr);
+ read_ret = bpf_sock_read_xattr(sock, "user.bpf_test", &value_ptr);
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/.gitignore b/tools/testing/selftests/exec/.gitignore
index 7f3d1ae762ec..8b93b405c424 100644
--- a/tools/testing/selftests/exec/.gitignore
+++ b/tools/testing/selftests/exec/.gitignore
@@ -19,3 +19,8 @@ null-argv
xxxxxxxx*
pipe
S_I*.test
+binfmt_misc_bpf
+binfmt_bpf_interp
+binfmt_bpf_app
+*.bpf.o
+vmlinux.h
diff --git a/tools/testing/selftests/exec/Makefile b/tools/testing/selftests/exec/Makefile
index 45a3cfc435cf..ec66c1fecfc0 100644
--- a/tools/testing/selftests/exec/Makefile
+++ b/tools/testing/selftests/exec/Makefile
@@ -21,6 +21,26 @@ TEST_GEN_PROGS += recursion-depth
TEST_GEN_PROGS += null-argv
TEST_GEN_PROGS += check-exec
+# binfmt_misc bpf-backed ('B') handler test: a libbpf harness plus its
+# struct_ops objects and the test interpreter/app it routes between. Only
+# built when clang, bpftool, the vmlinux BTF and libbpf are all present
+# (HAVE_BPF_TOOLCHAIN=y forces it) so the other exec selftests don't grow
+# a bpf toolchain dependency.
+CLANG ?= clang
+BPFTOOL ?= bpftool
+VMLINUX_BTF ?= /sys/kernel/btf/vmlinux
+HAVE_BPF_TOOLCHAIN ?= $(shell command -v $(CLANG) >/dev/null 2>&1 && \
+ command -v $(BPFTOOL) >/dev/null 2>&1 && \
+ test -r $(VMLINUX_BTF) && \
+ pkg-config --exists libbpf 2>/dev/null && echo y)
+ifeq ($(HAVE_BPF_TOOLCHAIN),y)
+TEST_GEN_PROGS += binfmt_misc_bpf
+TEST_GEN_FILES += bpf_interp.bpf.o nix_origin.bpf.o
+TEST_GEN_FILES += binfmt_bpf_interp binfmt_bpf_app
+else
+$(info exec selftests: skipping binfmt_misc_bpf, needs clang, bpftool, vmlinux BTF and libbpf)
+endif
+
EXTRA_CLEAN := $(OUTPUT)/subdir.moved $(OUTPUT)/execveat.moved $(OUTPUT)/xxxxx* \
$(OUTPUT)/S_I*.test
@@ -55,3 +75,31 @@ $(OUTPUT)/script-exec.inc: $(CHECK_EXEC_SAMPLES)/script-exec.inc
cp $< $@
$(OUTPUT)/script-noexec.inc: $(CHECK_EXEC_SAMPLES)/script-noexec.inc
cp $< $@
+
+# --- binfmt_misc bpf ('B') handler test ---------------------------------
+# The struct_ops bpf objects are compiled against the running kernel's BTF.
+# CLANG/BPFTOOL/VMLINUX_BTF are set above next to the toolchain check;
+# override LIBBPF_CFLAGS/LDLIBS to point at a libbpf install.
+BPF_CFLAGS ?= -I$(OUTPUT)
+LIBBPF_CFLAGS ?=
+LIBBPF_LDLIBS ?= -lbpf -lelf -lz
+
+$(OUTPUT)/vmlinux.h:
+ $(BPFTOOL) btf dump file $(VMLINUX_BTF) format c > $@
+ sed -i '/__ksym;$$/d' $@
+
+$(OUTPUT)/%.bpf.o: %.bpf.c $(OUTPUT)/vmlinux.h
+ $(CLANG) -g -O2 -target bpf -mcpu=v3 $(BPF_CFLAGS) $(LIBBPF_CFLAGS) -c $< -o $@
+
+$(OUTPUT)/binfmt_misc_bpf: binfmt_misc_bpf.c
+ $(CC) $(CFLAGS) $(LIBBPF_CFLAGS) $(LDFLAGS) $< $(LIBBPF_LDLIBS) -o $@
+
+$(OUTPUT)/binfmt_bpf_interp: binfmt_bpf_interp.c
+ $(CC) $(CFLAGS) $(LDFLAGS) $< -o $@
+
+# PT_INTERP is set to the literal "$ORIGIN/binfmt_bpf_interp"; the nix_origin
+# handler resolves it relative to the binary at run time.
+$(OUTPUT)/binfmt_bpf_app: binfmt_bpf_app.c
+ $(CC) $(CFLAGS) $(LDFLAGS) -Wl,--dynamic-linker,'$$ORIGIN/binfmt_bpf_interp' $< -o $@
+
+EXTRA_CLEAN += $(OUTPUT)/vmlinux.h $(OUTPUT)/bpf_interp.bpf.o $(OUTPUT)/nix_origin.bpf.o
diff --git a/tools/testing/selftests/exec/binfmt_bpf_app.c b/tools/testing/selftests/exec/binfmt_bpf_app.c
new file mode 100644
index 000000000000..472270f148bc
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_app.c
@@ -0,0 +1,12 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * A relocatable binary for the binfmt_misc_bpf $ORIGIN case. The Makefile
+ * links it with PT_INTERP set to the literal "$ORIGIN/binfmt_bpf_interp"
+ * (-Wl,--dynamic-linker), which the kernel ELF loader cannot resolve. The
+ * nix_origin bpf handler resolves it relative to this binary's directory and
+ * routes execution to the co-located interpreter.
+ */
+int main(void)
+{
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_bpf_interp.c b/tools/testing/selftests/exec/binfmt_bpf_interp.c
new file mode 100644
index 000000000000..2db205f095b2
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_bpf_interp.c
@@ -0,0 +1,15 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test interpreter for the binfmt_misc_bpf selftest. A bpf-backed 'B' handler
+ * routes a matched binary here; printing this marker proves the program's
+ * chosen interpreter actually ran.
+ */
+#include <unistd.h>
+
+int main(int argc, char **argv)
+{
+ (void)argc;
+ (void)argv;
+ write(1, "BPF_INTERP_RAN\n", 15);
+ return 0;
+}
diff --git a/tools/testing/selftests/exec/binfmt_misc_bpf.c b/tools/testing/selftests/exec/binfmt_misc_bpf.c
new file mode 100644
index 000000000000..cb89d2766fe2
--- /dev/null
+++ b/tools/testing/selftests/exec/binfmt_misc_bpf.c
@@ -0,0 +1,277 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Selftest for binfmt_misc bpf-backed ('B') handlers.
+ *
+ * A handler is a struct binfmt_misc_ops struct_ops map with a sleepable match
+ * and a sleepable load program. Attaching it publishes it by name in the
+ * caller's user namespace; a 'B' entry referencing it by name in the
+ * interpreter field activates it:
+ *
+ * echo ':name:B::::<handler>:' > /proc/sys/fs/binfmt_misc/register
+ *
+ * Two self-contained cases are exercised:
+ *
+ * 1. bpf_interp: the match program matches a synthetic aarch64 ELF header
+ * from the prefetched bprm->buf and the load program routes it to a
+ * fixed interpreter of its choosing.
+ * 2. nix_origin: the match program reads the binary's program headers to
+ * commit only to a "$ORIGIN/..."-relative PT_INTERP and the load program
+ * resolves it to an interpreter co-located with the binary (the
+ * relocatable-loader case the kernel ELF loader cannot express).
+ *
+ * Both route to a test interpreter that prints BPF_INTERP_RAN, proving the
+ * program's chosen interpreter actually ran.
+ */
+#define _GNU_SOURCE
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <unistd.h>
+#include <fcntl.h>
+#include <libgen.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+
+#include <bpf/btf.h>
+#include <bpf/libbpf.h>
+
+#define INTERP_PATH "/tmp/binfmt_bpf_interp"
+#define AARCH64_PATH "/tmp/binfmt_bpf_aarch64"
+#define RELOC_DIR "/tmp/binfmt_reloc"
+#define BINFMT_REG "/proc/sys/fs/binfmt_misc/register"
+#define EXPECT "BPF_INTERP_RAN"
+
+static char testdir[512]; /* directory holding this test's built artifacts */
+
+static int copy_file(const char *src, const char *dst)
+{
+ char buf[4096];
+ int in, out;
+ ssize_t n;
+
+ in = open(src, O_RDONLY);
+ if (in < 0)
+ return -1;
+ out = open(dst, O_WRONLY | O_CREAT | O_TRUNC, 0755);
+ if (out < 0) {
+ close(in);
+ return -1;
+ }
+ while ((n = read(in, buf, sizeof(buf))) > 0) {
+ if (write(out, buf, n) != n) {
+ close(in);
+ close(out);
+ return -1;
+ }
+ }
+ close(in);
+ close(out);
+ return n < 0 ? -1 : 0;
+}
+
+/* A minimal 64-bit little-endian aarch64 ELF header, padded to the read size. */
+static int create_fake_aarch64(const char *path)
+{
+ unsigned char hdr[256] = {0};
+ int fd;
+
+ hdr[0] = 0x7f; hdr[1] = 'E'; hdr[2] = 'L'; hdr[3] = 'F';
+ hdr[4] = 2; /* ELFCLASS64 */
+ hdr[5] = 1; /* ELFDATA2LSB */
+ hdr[6] = 1; /* EV_CURRENT */
+ hdr[16] = 2; /* e_type = ET_EXEC */
+ hdr[18] = 183 & 0xff; /* e_machine = EM_AARCH64 */
+ hdr[19] = (183 >> 8) & 0xff;
+ hdr[20] = 1; /* e_version */
+
+ fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0755);
+ if (fd < 0)
+ return -1;
+ if (write(fd, hdr, sizeof(hdr)) != (ssize_t)sizeof(hdr)) {
+ close(fd);
+ return -1;
+ }
+ close(fd);
+ return 0;
+}
+
+static int register_entry(const char *name, const char *handler)
+{
+ char rule[128];
+ int fd;
+ ssize_t n;
+
+ snprintf(rule, sizeof(rule), ":%s:B::::%s:", name, handler);
+ fd = open(BINFMT_REG, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ n = write(fd, rule, strlen(rule));
+ close(fd);
+ return n < 0 ? -1 : 0;
+}
+
+static void unregister_entry(const char *name)
+{
+ char path[128];
+ int fd;
+
+ snprintf(path, sizeof(path), "/proc/sys/fs/binfmt_misc/%s", name);
+ fd = open(path, O_WRONLY);
+ if (fd >= 0) {
+ if (write(fd, "-1", 2) < 0)
+ ; /* best effort */
+ close(fd);
+ }
+}
+
+static int check_output(const char *cmd, const char *expected)
+{
+ char buf[128];
+ FILE *fp;
+
+ fp = popen(cmd, "r");
+ if (!fp)
+ return -1;
+ if (!fgets(buf, sizeof(buf), fp)) {
+ pclose(fp);
+ return -1;
+ }
+ pclose(fp);
+ return strncmp(buf, expected, strlen(expected)) ? -1 : 0;
+}
+
+/*
+ * Load @objfile, attach its struct_ops map @handler (which publishes the
+ * handler), activate a 'B' entry named @entry that references it, run @target
+ * and check it produced @expect.
+ */
+static int run_case(const char *objfile, const char *handler,
+ const char *entry, const char *target, const char *expect)
+{
+ struct bpf_object *obj;
+ struct bpf_map *map;
+ struct bpf_link *link;
+ int ret = -1;
+
+ obj = bpf_object__open_file(objfile, NULL);
+ if (!obj || libbpf_get_error(obj)) {
+ fprintf(stderr, "open %s failed\n", objfile);
+ return -1;
+ }
+ if (bpf_object__load(obj)) {
+ fprintf(stderr, "load %s failed (check dmesg for the verifier log)\n",
+ objfile);
+ goto close;
+ }
+ map = bpf_object__find_map_by_name(obj, handler);
+ if (!map) {
+ fprintf(stderr, "no struct_ops map '%s' in %s\n", handler, objfile);
+ goto close;
+ }
+ link = bpf_map__attach_struct_ops(map);
+ if (!link || libbpf_get_error(link)) {
+ fprintf(stderr, "attach struct_ops '%s' failed\n", handler);
+ goto close;
+ }
+ if (register_entry(entry, handler)) {
+ fprintf(stderr, "register 'B' entry '%s' failed\n", entry);
+ goto detach;
+ }
+ ret = check_output(target, expect);
+ unregister_entry(entry);
+detach:
+ bpf_link__destroy(link);
+close:
+ bpf_object__close(obj);
+ return ret;
+}
+
+int main(void)
+{
+ char src[600], obj[600], appdst[600], interpdst[600];
+ char exe[512];
+ ssize_t n;
+ int fail = 0;
+ struct stat st;
+ struct btf *btf;
+
+ if (getuid() != 0) {
+ fprintf(stderr, "Skipping: test must be run as root\n");
+ return 4; /* KSFT_SKIP */
+ }
+
+ /* The kernel must know struct binfmt_misc_ops (CONFIG_BINFMT_MISC_BPF). */
+ btf = btf__load_vmlinux_btf();
+ if (!btf || btf__find_by_name_kind(btf, "binfmt_misc_ops",
+ BTF_KIND_STRUCT) < 0) {
+ fprintf(stderr,
+ "Skipping: no struct binfmt_misc_ops in the kernel BTF (CONFIG_BINFMT_MISC_BPF)\n");
+ btf__free(btf);
+ return 4; /* KSFT_SKIP */
+ }
+ btf__free(btf);
+
+ n = readlink("/proc/self/exe", exe, sizeof(exe) - 1);
+ if (n < 0) {
+ perror("readlink");
+ return 1;
+ }
+ exe[n] = '\0';
+ snprintf(testdir, sizeof(testdir), "%s", dirname(exe));
+
+ if (stat("/sys/fs/bpf", &st) < 0)
+ mkdir("/sys/fs/bpf", 0755);
+ mount("bpf", "/sys/fs/bpf", "bpf", 0, NULL);
+ if (access(BINFMT_REG, F_OK) < 0)
+ mount("binfmt_misc", "/proc/sys/fs/binfmt_misc", "binfmt_misc", 0, NULL);
+
+ /* Shared test interpreter. */
+ snprintf(src, sizeof(src), "%s/binfmt_bpf_interp", testdir);
+ if (copy_file(src, INTERP_PATH)) {
+ fprintf(stderr, "cannot install %s\n", INTERP_PATH);
+ return 1;
+ }
+
+ /* Case 1: match a synthetic aarch64 header -> fixed interpreter. */
+ printf("[*] case 1: match aarch64 header -> program-chosen interpreter\n");
+ if (create_fake_aarch64(AARCH64_PATH)) {
+ fprintf(stderr, "cannot create %s\n", AARCH64_PATH);
+ return 1;
+ }
+ snprintf(obj, sizeof(obj), "%s/bpf_interp.bpf.o", testdir);
+ if (run_case(obj, "bpf_interp", "test_bpf_interp", AARCH64_PATH, EXPECT) == 0)
+ printf("[+] case 1 passed\n");
+ else {
+ printf("[-] case 1 FAILED\n");
+ fail = 1;
+ }
+ unlink(AARCH64_PATH);
+
+ /* Case 2: $ORIGIN-relative PT_INTERP -> co-located interpreter. */
+ printf("[*] case 2: $ORIGIN interpreter resolved relative to the binary\n");
+ mkdir(RELOC_DIR, 0755);
+ snprintf(appdst, sizeof(appdst), "%s/app", RELOC_DIR);
+ snprintf(interpdst, sizeof(interpdst), "%s/binfmt_bpf_interp", RELOC_DIR);
+ snprintf(src, sizeof(src), "%s/binfmt_bpf_app", testdir);
+ if (copy_file(src, appdst) ||
+ copy_file(INTERP_PATH, interpdst)) {
+ fprintf(stderr, "cannot set up %s\n", RELOC_DIR);
+ fail = 1;
+ } else {
+ snprintf(obj, sizeof(obj), "%s/nix_origin.bpf.o", testdir);
+ if (run_case(obj, "nix_origin", "test_bpf_origin", appdst, EXPECT) == 0)
+ printf("[+] case 2 passed\n");
+ else {
+ printf("[-] case 2 FAILED\n");
+ fail = 1;
+ }
+ }
+ unlink(appdst);
+ unlink(interpdst);
+ rmdir(RELOC_DIR);
+ unlink(INTERP_PATH);
+
+ if (!fail)
+ printf("[*] all binfmt_misc bpf cases passed\n");
+ return fail;
+}
diff --git a/tools/testing/selftests/exec/bpf_interp.bpf.c b/tools/testing/selftests/exec/bpf_interp.bpf.c
new file mode 100644
index 000000000000..8df2d2d01e25
--- /dev/null
+++ b/tools/testing/selftests/exec/bpf_interp.bpf.c
@@ -0,0 +1,61 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * binfmt_misc_ops handler for the selftest's fixed-interpreter case: match a
+ * 64-bit aarch64 ELF header from the prefetched buffer and route it to a fixed
+ * interpreter chosen by the program. This is the portable, self-contained
+ * equivalent of routing a foreign binary to an emulator: it matches
+ * programmatically and computes the interpreter, but points at a test binary
+ * the harness installs rather than a system emulator.
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define EI_CLASS 4
+#define ELFCLASS64 2
+#define EM_AARCH64 183
+
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+/*
+ * A magic-style decision needs nothing beyond the prefetched bprm->buf,
+ * even though the match program could read the file.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(bpf_interp_match, struct linux_binprm *bprm)
+{
+ __u16 machine;
+
+ if (bprm->buf[0] != 0x7f || bprm->buf[1] != 'E' ||
+ bprm->buf[2] != 'L' || bprm->buf[3] != 'F' ||
+ bprm->buf[EI_CLASS] != ELFCLASS64)
+ return false;
+
+ /* e_machine is a 16-bit little-endian field at offset 18. */
+ machine = (__u8)bprm->buf[18] | ((__u16)(__u8)bprm->buf[19] << 8);
+ return machine == EM_AARCH64;
+}
+
+SEC("struct_ops.s/load")
+int BPF_PROG(bpf_interp_load, struct linux_binprm *bprm)
+{
+ /*
+ * Keep the path on the (writable) stack: bpf_binprm_set_interp() takes
+ * a sized memory arg and the verifier rejects a read-only .rodata
+ * buffer for it. The harness installs the interpreter at this path.
+ */
+ char interp[] = "/tmp/binfmt_bpf_interp";
+
+ /* @path__sz includes the terminating NUL; 0 commits the selection. */
+ return bpf_binprm_set_interp(bprm, interp, sizeof(interp));
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops bpf_interp = {
+ .match = (void *)bpf_interp_match,
+ .load = (void *)bpf_interp_load,
+ .name = "bpf_interp",
+};
diff --git a/tools/testing/selftests/exec/nix_origin.bpf.c b/tools/testing/selftests/exec/nix_origin.bpf.c
new file mode 100644
index 000000000000..378e22a4c43b
--- /dev/null
+++ b/tools/testing/selftests/exec/nix_origin.bpf.c
@@ -0,0 +1,224 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * nix_origin.bpf.c - $ORIGIN-relative PT_INTERP resolution
+ *
+ * A binfmt_misc_ops handler that makes relocatable (Nix-style) ELF
+ * binaries work: if PT_INTERP starts with "$ORIGIN/", the loader is
+ * resolved relative to the directory of the binary being executed and
+ * selected via bpf_binprm_set_interp(). The match program reads the
+ * program headers itself, so anything else never commits to this
+ * handler and passes through untouched.
+ *
+ * Activate with:
+ * bpftool struct_ops register nix_origin.bpf.o /sys/fs/bpf
+ * echo ':nix-origin:B::::nix_origin:' > /proc/sys/fs/binfmt_misc/register
+ */
+#include "vmlinux.h"
+#include <bpf/bpf_helpers.h>
+#include <bpf/bpf_tracing.h>
+
+char _license[] SEC("license") = "GPL";
+
+#define PATH_MAX 4096
+#define EI_CLASS 4
+#define ELFCLASSXX 2 /* ELFCLASS64; flip to 1 for 32-bit */
+#define PT_INTERP 3
+#define MAX_PHDRS 64
+
+#define ORIGIN "$ORIGIN"
+#define ORIGIN_LEN (sizeof(ORIGIN) - 1)
+
+#define ENOENT 2
+#define ENOEXEC 8
+#define ENAMETOOLONG 36
+
+extern int bpf_dynptr_from_file(struct file *file, __u32 flags,
+ struct bpf_dynptr *ptr__uninit) __ksym;
+extern int bpf_dynptr_file_discard(struct bpf_dynptr *dynptr) __ksym;
+extern int bpf_path_d_path(const struct path *path, char *buf,
+ size_t buf__sz) __ksym;
+extern int bpf_binprm_set_interp(struct linux_binprm *bprm, const char *path,
+ size_t path__sz) __ksym;
+
+struct scratch {
+ char interp[PATH_MAX]; /* PT_INTERP as embedded in the binary */
+ char path[PATH_MAX]; /* d_path of the binary, becomes the result */
+};
+
+/* Keyed by pid: execs run concurrently and the programs can sleep. */
+struct {
+ __uint(type, BPF_MAP_TYPE_HASH);
+ __uint(max_entries, 512);
+ __type(key, __u64);
+ __type(value, struct scratch);
+} scratch_map SEC(".maps");
+
+static const struct scratch zero_scratch;
+
+/* An ELF64 binary per the prefetched header? */
+static bool is_elf64(struct linux_binprm *bprm)
+{
+ return bprm->buf[0] == 0x7f && bprm->buf[1] == 'E' &&
+ bprm->buf[2] == 'L' && bprm->buf[3] == 'F' &&
+ bprm->buf[EI_CLASS] == ELFCLASSXX;
+}
+
+/* Locate PT_INTERP; false if the file has none or looks malformed. */
+static bool find_pt_interp(struct bpf_dynptr *dp, struct elf64_phdr *phdr)
+{
+ struct elf64_hdr ehdr;
+ bool found = false;
+ int i;
+
+ if (bpf_dynptr_read(&ehdr, sizeof(ehdr), dp, 0, 0))
+ return false;
+ if (ehdr.e_phentsize != sizeof(struct elf64_phdr))
+ return false;
+
+ bpf_for(i, 0, ehdr.e_phnum) {
+ if (i >= MAX_PHDRS)
+ break;
+ if (bpf_dynptr_read(phdr, sizeof(*phdr), dp,
+ ehdr.e_phoff + i * sizeof(*phdr), 0))
+ return false;
+ if (phdr->p_type == PT_INTERP) {
+ found = true;
+ break;
+ }
+ }
+ return found;
+}
+
+/*
+ * An ELF64 binary whose PT_INTERP starts with "$ORIGIN/" is ours. The
+ * match can sleep and read the file, so the decision is made here and
+ * regular binaries never commit to this handler: later binfmt_misc
+ * entries and binfmt_elf see them as if we did not exist.
+ */
+SEC("struct_ops.s/match")
+bool BPF_PROG(nix_origin_match, struct linux_binprm *bprm)
+{
+ char prefix[ORIGIN_LEN + 1] = {};
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ bool ours = false;
+
+ if (!is_elf64(bprm))
+ return false;
+
+ /* The dynptr must be discarded on every path once requested. */
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+ if (find_pt_interp(&dp, &phdr) &&
+ phdr.p_filesz > ORIGIN_LEN + 1 &&
+ !bpf_dynptr_read(prefix, sizeof(prefix), &dp, phdr.p_offset, 0))
+ ours = !bpf_strncmp(prefix, sizeof(prefix), ORIGIN "/");
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ours;
+}
+
+/*
+ * The match is committed and already vetted the "$ORIGIN/" prefix, so
+ * everything here reads the file again from scratch: -ENOEXEC only
+ * covers a binary that changed under us and stopped being ours.
+ */
+SEC("struct_ops.s/load")
+int BPF_PROG(nix_origin_load, struct linux_binprm *bprm)
+{
+ __u32 isz, sfx, rsz, slash;
+ struct elf64_phdr phdr;
+ struct bpf_dynptr dp;
+ struct scratch *sc;
+ __u64 id;
+ int ret = -ENOEXEC, len, i;
+
+ if (bpf_dynptr_from_file(bprm->file, 0, &dp))
+ goto out;
+
+ if (!find_pt_interp(&dp, &phdr))
+ goto out;
+
+ isz = phdr.p_filesz;
+ if (isz <= ORIGIN_LEN + 1 || isz >= sizeof(sc->interp))
+ goto out;
+ /*
+ * The range check above compiles to a test on a zero-extended copy of
+ * the u64 p_filesz, so the verifier does not carry the bound to the
+ * dynptr_read() length below ("unbounded memory access"). Mask isz to
+ * the buffer size (a power of two) and force the masked value to be
+ * materialized with a barrier so the read uses the bounded register.
+ */
+ isz &= sizeof(sc->interp) - 1;
+ barrier_var(isz);
+
+ id = bpf_get_current_pid_tgid();
+ if (bpf_map_update_elem(&scratch_map, &id, &zero_scratch, BPF_ANY))
+ goto out;
+ sc = bpf_map_lookup_elem(&scratch_map, &id);
+ if (!sc)
+ goto out_del;
+
+ if (bpf_dynptr_read(sc->interp, isz, &dp, phdr.p_offset, 0))
+ goto out_del;
+ if (sc->interp[isz - 1] != '\0')
+ goto out_del;
+
+ /* Not "$ORIGIN/..." anymore? Then it is not ours anymore either. */
+ if (sc->interp[0] != '$' || sc->interp[1] != 'O' ||
+ sc->interp[2] != 'R' || sc->interp[3] != 'I' ||
+ sc->interp[4] != 'G' || sc->interp[5] != 'I' ||
+ sc->interp[6] != 'N' || sc->interp[7] != '/')
+ goto out_del;
+
+ /*
+ * From here on resolution failures fail the exec instead of falling
+ * back to binfmt_elf, which would resolve the literal "$ORIGIN/..."
+ * relative to the caller's cwd.
+ */
+ ret = -ENOENT;
+ len = bpf_path_d_path(&bprm->file->f_path, sc->path, sizeof(sc->path));
+ if (len <= 0 || len > sizeof(sc->path))
+ goto out_del;
+ /* Unreachable or unlinked ("... (deleted)") binaries can't resolve. */
+ if (sc->path[0] != '/')
+ goto out_del;
+
+ /* $ORIGIN = dirname of the binary. */
+ slash = 0;
+ bpf_for(i, 1, len - 1) {
+ if (i >= sizeof(sc->path))
+ break;
+ if (sc->path[i] == '/')
+ slash = i;
+ }
+
+ /* Splice the suffix (leading '/' and NUL included) onto the dir. */
+ sfx = isz - ORIGIN_LEN;
+ rsz = slash + sfx;
+ if (rsz > sizeof(sc->path)) {
+ ret = -ENAMETOOLONG;
+ goto out_del;
+ }
+ bpf_for(i, 0, sfx) {
+ __u32 s = ORIGIN_LEN + i, d = slash + i;
+
+ if (s >= sizeof(sc->interp) || d >= sizeof(sc->path))
+ break;
+ sc->path[d] = sc->interp[s];
+ }
+
+ ret = bpf_binprm_set_interp(bprm, sc->path, rsz);
+out_del:
+ bpf_map_delete_elem(&scratch_map, &id);
+out:
+ bpf_dynptr_file_discard(&dp);
+ return ret;
+}
+
+SEC(".struct_ops.link")
+struct binfmt_misc_ops nix_origin = {
+ .match = (void *)nix_origin_match,
+ .load = (void *)nix_origin_load,
+ .name = "nix_origin",
+};
diff --git a/tools/testing/selftests/filesystems/.gitignore b/tools/testing/selftests/filesystems/.gitignore
index a78f894157de..9eb185fb2f9d 100644
--- a/tools/testing/selftests/filesystems/.gitignore
+++ b/tools/testing/selftests/filesystems/.gitignore
@@ -6,3 +6,4 @@ file_stressor
anon_inode_test
kernfs_test
idmapped_tmpfile
+ustat_test
diff --git a/tools/testing/selftests/filesystems/Makefile b/tools/testing/selftests/filesystems/Makefile
index a7ec2ba2dd83..03be337c1f35 100644
--- a/tools/testing/selftests/filesystems/Makefile
+++ b/tools/testing/selftests/filesystems/Makefile
@@ -1,7 +1,7 @@
# SPDX-License-Identifier: GPL-2.0
CFLAGS += $(KHDR_INCLUDES)
-TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog
+TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test
TEST_GEN_PROGS += idmapped_tmpfile
TEST_GEN_PROGS_EXTENDED := dnotify_test
diff --git a/tools/testing/selftests/filesystems/overlayfs/.gitignore b/tools/testing/selftests/filesystems/overlayfs/.gitignore
index e23a18c8b37f..077f7a128168 100644
--- a/tools/testing/selftests/filesystems/overlayfs/.gitignore
+++ b/tools/testing/selftests/filesystems/overlayfs/.gitignore
@@ -1,3 +1,4 @@
# SPDX-License-Identifier: GPL-2.0-only
dev_in_maps
set_layers_via_fds
+idmapped_mounts
diff --git a/tools/testing/selftests/filesystems/overlayfs/Makefile b/tools/testing/selftests/filesystems/overlayfs/Makefile
index d3ad4a77db9b..b3185f684add 100644
--- a/tools/testing/selftests/filesystems/overlayfs/Makefile
+++ b/tools/testing/selftests/filesystems/overlayfs/Makefile
@@ -8,7 +8,9 @@ LOCAL_HDRS += ../wrappers.h log.h
TEST_GEN_PROGS := dev_in_maps
TEST_GEN_PROGS += set_layers_via_fds
+TEST_GEN_PROGS += idmapped_mounts
include ../../lib.mk
$(OUTPUT)/set_layers_via_fds: ../utils.c
+$(OUTPUT)/idmapped_mounts: ../utils.c
diff --git a/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c b/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c
new file mode 100644
index 000000000000..44a75839f4ed
--- /dev/null
+++ b/tools/testing/selftests/filesystems/overlayfs/idmapped_mounts.c
@@ -0,0 +1,501 @@
+// SPDX-License-Identifier: GPL-2.0
+#define _GNU_SOURCE
+
+#include <fcntl.h>
+#include <limits.h>
+#include <sched.h>
+#include <stdio.h>
+#include <unistd.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+
+#include <linux/mount.h>
+#include <linux/types.h>
+
+#include "kselftest_harness.h"
+#include "../wrappers.h"
+#include "../utils.h"
+
+/*
+ * An idmapping that maps the mount-visible id range [0, ID_RANGE) onto the
+ * host/overlay-final id range [ID_HOST, ID_HOST + ID_RANGE). Through such an
+ * idmapped overlay mount, an overlay-final id of ID_HOST + n is reported as n,
+ * and an id of n requested through the mount is stored as ID_HOST + n.
+ */
+#define ID_NS 0
+#define ID_HOST 10000
+#define ID_RANGE 10000
+
+/*
+ * For the composition test the lower layer's on-disk ids live in a
+ * separate range and are mapped by an idmapped lower layer onto the
+ * overlay-final range [ID_HOST, ID_HOST + ID_RANGE).
+ */
+#define LAYER_HOST 20000
+
+#ifndef MOUNT_ATTR_IDMAP
+#define MOUNT_ATTR_IDMAP 0x00100000
+#endif
+
+#ifndef __NR_mount_setattr
+#define __NR_mount_setattr 442
+#endif
+
+static inline int sys_mount_setattr(int dfd, const char *path,
+ unsigned int flags,
+ struct mount_attr *attr, size_t size)
+{
+ return syscall(__NR_mount_setattr, dfd, path, flags, attr, size);
+}
+
+static bool ovl_supported(void)
+{
+ int fd = sys_fsopen("overlay", 0);
+
+ if (fd < 0)
+ return false;
+ close(fd);
+ return true;
+}
+
+/* base/{l,u,w} owned by ID_HOST so they map to ID_NS through the idmap. */
+static int setup_layers(const char *base)
+{
+ static const char *sub[] = { "", "/l", "/u", "/w" };
+ char path[PATH_MAX];
+
+ for (size_t i = 0; i < ARRAY_SIZE(sub); i++) {
+ snprintf(path, sizeof(path), "%s%s", base, sub[i]);
+ if (mkdir(path, 0755) && errno != EEXIST)
+ return -1;
+ if (i && chown(path, ID_HOST, ID_HOST))
+ return -1;
+ }
+ return 0;
+}
+
+static int ovl_mount(const char *base, bool nfs_export)
+{
+ char lower[PATH_MAX], upper[PATH_MAX], work[PATH_MAX];
+ int fsfd, ovl;
+
+ snprintf(lower, sizeof(lower), "%s/l", base);
+ snprintf(upper, sizeof(upper), "%s/u", base);
+ snprintf(work, sizeof(work), "%s/w", base);
+
+ fsfd = sys_fsopen("overlay", 0);
+ if (fsfd < 0)
+ return -1;
+
+ if (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", "test", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "lowerdir", lower, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "upperdir", upper, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "workdir", work, 0))
+ goto err;
+ if (nfs_export &&
+ (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "index", "on", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "nfs_export", "on", 0)))
+ goto err;
+ if (sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0))
+ goto err;
+
+ ovl = sys_fsmount(fsfd, 0, 0);
+ close(fsfd);
+ return ovl;
+err:
+ close(fsfd);
+ return -1;
+}
+
+/* Idmap the (still detached, not yet visible) overlay mount @mfd. */
+static int ovl_idmap(int mfd)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ int ret, userns_fd;
+
+ /*
+ * get_userns_fd(fs_id, mount_id, range): a file whose filesystem id
+ * is fs_id + n is shown through the idmapped mount as mount_id + n.
+ * Here the overlay-final (fs side) range is [ID_HOST, ..) and the
+ * caller-visible (mount side) range is [ID_NS, ..).
+ */
+ userns_fd = get_userns_fd(ID_HOST, ID_NS, ID_RANGE);
+ if (userns_fd < 0)
+ return -1;
+
+ attr.userns_fd = userns_fd;
+ ret = sys_mount_setattr(mfd, "", AT_EMPTY_PATH, &attr, sizeof(attr));
+ close(userns_fd);
+ return ret;
+}
+
+/* Clone @path into a detached, idmapped mount usable as an overlay layer. */
+static int idmapped_layer_fd(const char *path, int nsid, int hostid, int range)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ int fd_tree, userns_fd;
+
+ fd_tree = sys_open_tree(AT_FDCWD, path,
+ OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ if (fd_tree < 0)
+ return -1;
+ userns_fd = get_userns_fd(nsid, hostid, range);
+ if (userns_fd < 0) {
+ close(fd_tree);
+ return -1;
+ }
+ attr.userns_fd = userns_fd;
+ if (sys_mount_setattr(fd_tree, "", AT_EMPTY_PATH, &attr,
+ sizeof(attr))) {
+ close(userns_fd);
+ close(fd_tree);
+ return -1;
+ }
+ close(userns_fd);
+ return fd_tree;
+}
+
+/* Overlay with a layer passed by fd (idmapped) plus a plain upper/work. */
+static int ovl_mount_lower_fd(const char *upper, const char *work, int fd_lower)
+{
+ int fsfd, ovl;
+
+ fsfd = sys_fsopen("overlay", 0);
+ if (fsfd < 0)
+ return -1;
+
+ if (sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "source", "test", 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "upperdir", upper, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_STRING, "workdir", work, 0) ||
+ sys_fsconfig(fsfd, FSCONFIG_SET_FD, "lowerdir+", NULL, fd_lower) ||
+ sys_fsconfig(fsfd, FSCONFIG_CMD_CREATE, NULL, NULL, 0))
+ goto err;
+
+ ovl = sys_fsmount(fsfd, 0, 0);
+ close(fsfd);
+ return ovl;
+err:
+ close(fsfd);
+ return -1;
+}
+
+/*
+ * Mount an overlay inside user namespace @u1 (so the overlay sb's s_user_ns is
+ * not the initial namespace) and idmap that overlay mount with @u2. Runs in a
+ * child that joins @u1; returns 0 on success.
+ */
+static int userns_overlay_child(int u1)
+{
+ struct mount_attr attr = {
+ .attr_set = MOUNT_ATTR_IDMAP,
+ };
+ struct stat st;
+ int ovl, u2;
+
+ /* Become root in the overlay sb's user namespace u1. */
+ if (!switch_userns(u1, 0, 0, false))
+ return fprintf(stderr, "userns: switch_userns: %m\n"), -1;
+ if (unshare(CLONE_NEWNS) ||
+ sys_mount(NULL, "/", NULL, MS_SLAVE | MS_REC, NULL))
+ return fprintf(stderr, "userns: unshare/slave: %m\n"), -1;
+ if (sys_mount("tmpfs", "/tmp", "tmpfs", 0, NULL))
+ return fprintf(stderr, "userns: mount tmpfs: %m\n"), -1;
+ if (setup_layers("/tmp/ovl"))
+ return fprintf(stderr, "userns: setup_layers: %m\n"), -1;
+ if (mknod("/tmp/ovl/l/file", S_IFREG | 0644, 0) ||
+ chown("/tmp/ovl/l/file", ID_HOST + 5, ID_HOST + 5))
+ return fprintf(stderr, "userns: lower file: %m\n"), -1;
+
+ ovl = ovl_mount("/tmp/ovl", false);
+ if (ovl < 0)
+ return fprintf(stderr, "userns: ovl_mount: %m\n"), -1;
+
+ /*
+ * mount_setattr() requires CAP_SYS_ADMIN over the idmap user
+ * namespace, so it must be a child of u1. Create it now, from
+ * inside u1.
+ */
+ u2 = get_userns_fd(ID_HOST, ID_NS, ID_RANGE);
+ if (u2 < 0)
+ return fprintf(stderr, "userns: get_userns_fd: %m\n"), -1;
+ attr.userns_fd = u2;
+ if (sys_mount_setattr(ovl, "", AT_EMPTY_PATH, &attr, sizeof(attr)))
+ return fprintf(stderr, "userns: mount_setattr: %m\n"), -1;
+ close(u2);
+
+ if (fstatat(ovl, "file", &st, 0))
+ return fprintf(stderr, "userns: fstatat: %m\n"), -1;
+ if (st.st_uid != ID_NS + 5 || st.st_gid != ID_NS + 5) {
+ fprintf(stderr, "userns: got %u:%u expected %u:%u\n",
+ st.st_uid, st.st_gid, ID_NS + 5, ID_NS + 5);
+ return -1;
+ }
+ return 0;
+}
+
+FIXTURE(idmapped_overlay) {
+ char base[64];
+};
+
+FIXTURE_SETUP(idmapped_overlay)
+{
+ /* Private mount namespace so test mounts need no cleanup. */
+ ASSERT_EQ(unshare(CLONE_NEWNS), 0);
+ ASSERT_EQ(sys_mount(NULL, "/", NULL, MS_SLAVE | MS_REC, NULL), 0);
+
+ /* tmpfs for the layers so we can chown them to arbitrary ids. */
+ ASSERT_EQ(sys_mount("tmpfs", "/tmp", "tmpfs", 0, NULL), 0);
+
+ snprintf(self->base, sizeof(self->base), "/tmp/ovl");
+ ASSERT_EQ(setup_layers(self->base), 0);
+}
+
+FIXTURE_TEARDOWN(idmapped_overlay)
+{
+}
+
+/* A file owned by ID_HOST + 5 is reported as ID_NS + 5 through the idmap. */
+TEST_F(idmapped_overlay, getattr)
+{
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, ID_HOST + 5, ID_HOST + 5), 0);
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ ASSERT_EQ(fstatat(ovl, "file", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 5);
+ EXPECT_EQ(st.st_gid, ID_NS + 5);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/*
+ * Every creation path initializes the new owner through the mount idmap:
+ * created as caller id ID_NS, stored on the upper layer as overlay-final
+ * ID_HOST. Covers ovl_create() (regular file), ovl_mkdir(), ovl_mknod()
+ * and ovl_symlink() (which share ovl_create_object()), plus the separate
+ * ovl_tmpfile() path.
+ */
+TEST_F(idmapped_overlay, create)
+{
+ static const char *names[] = { "reg", "dir", "fifo", "lnk" };
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl, fd;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ /* One object per creation operation, all as caller id ID_NS. */
+ fd = openat(ovl, "reg", O_CREAT | O_WRONLY | O_EXCL, 0644);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(close(fd), 0);
+ ASSERT_EQ(mkdirat(ovl, "dir", 0755), 0);
+ ASSERT_EQ(mknodat(ovl, "fifo", S_IFIFO | 0644, 0), 0);
+ ASSERT_EQ(symlinkat("target", ovl, "lnk"), 0);
+
+ for (size_t i = 0; i < ARRAY_SIZE(names); i++) {
+ /* Reported as ID_NS through the idmapped mount ... */
+ ASSERT_EQ(fstatat(ovl, names[i], &st, AT_SYMLINK_NOFOLLOW), 0);
+ EXPECT_EQ(st.st_uid, ID_NS);
+ EXPECT_EQ(st.st_gid, ID_NS);
+ /* ... and stored as ID_HOST on the upper layer. */
+ snprintf(path, sizeof(path), "%s/u/%s", self->base, names[i]);
+ ASSERT_EQ(lstat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST);
+ EXPECT_EQ(st.st_gid, ID_HOST);
+ }
+
+ /* O_TMPFILE goes through the separate ovl_tmpfile() path. */
+ fd = openat(ovl, ".", O_TMPFILE | O_WRONLY, 0644);
+ ASSERT_GE(fd, 0);
+ /* Inside the mount: caller id ID_NS. */
+ ASSERT_EQ(fstat(fd, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_NS);
+ EXPECT_EQ(st.st_gid, ID_NS);
+ /* Link it in so the upper backing file can be inspected too. */
+ ASSERT_EQ(linkat(fd, "", ovl, "tmp", AT_EMPTY_PATH), 0);
+ EXPECT_EQ(close(fd), 0);
+ snprintf(path, sizeof(path), "%s/u/tmp", self->base);
+ ASSERT_EQ(lstat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST);
+ EXPECT_EQ(st.st_gid, ID_HOST);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/* chown through the idmapped mount round-trips: ID_NS + 5 <-> ID_HOST + 5. */
+TEST_F(idmapped_overlay, chown)
+{
+ char path[PATH_MAX];
+ struct stat st;
+ int ovl, fd;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ ovl = ovl_mount(self->base, false);
+ ASSERT_GE(ovl, 0);
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ fd = openat(ovl, "f", O_CREAT | O_WRONLY | O_EXCL, 0644);
+ ASSERT_GE(fd, 0);
+ EXPECT_EQ(close(fd), 0);
+
+ ASSERT_EQ(fchownat(ovl, "f", ID_NS + 5, ID_NS + 5, 0), 0);
+
+ ASSERT_EQ(fstatat(ovl, "f", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 5);
+ EXPECT_EQ(st.st_gid, ID_NS + 5);
+
+ snprintf(path, sizeof(path), "%s/u/f", self->base);
+ ASSERT_EQ(stat(path, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_HOST + 5);
+ EXPECT_EQ(st.st_gid, ID_HOST + 5);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/*
+ * Composition: an idmapped lower layer underneath an idmapped overlay mount.
+ * An on-disk id is mapped by the layer idmap into the overlay-final range and
+ * then by the mount idmap into the caller's range:
+ *
+ * on-disk LAYER_HOST+7 --layer--> ID_HOST+7 --mount--> ID_NS+7
+ */
+TEST_F(idmapped_overlay, composition)
+{
+ char lower[PATH_MAX], upper[PATH_MAX], work[PATH_MAX], path[PATH_MAX];
+ struct stat st;
+ int ovl, fd_lower;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(lower, sizeof(lower), "%s/l", self->base);
+ snprintf(upper, sizeof(upper), "%s/u", self->base);
+ snprintf(work, sizeof(work), "%s/w", self->base);
+
+ /* Put the lower layer's ids in the on-disk [LAYER_HOST, ..) range. */
+ ASSERT_EQ(chown(lower, LAYER_HOST, LAYER_HOST), 0);
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, LAYER_HOST + 7, LAYER_HOST + 7), 0);
+
+ /* Idmapped lower: on-disk LAYER_HOST <-> overlay-final ID_HOST. */
+ fd_lower = idmapped_layer_fd(lower, LAYER_HOST, ID_HOST, ID_RANGE);
+ ASSERT_GE(fd_lower, 0);
+
+ ovl = ovl_mount_lower_fd(upper, work, fd_lower);
+ ASSERT_GE(ovl, 0);
+ EXPECT_EQ(close(fd_lower), 0);
+
+ /* Idmap the overlay mount: overlay-final ID_HOST <-> caller ID_NS. */
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ ASSERT_EQ(fstatat(ovl, "file", &st, 0), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 7);
+ EXPECT_EQ(st.st_gid, ID_NS + 7);
+
+ EXPECT_EQ(close(ovl), 0);
+}
+
+/* An idmapped overlay mount whose sb lives inside a user namespace. */
+TEST_F(idmapped_overlay, userns)
+{
+ int u1;
+ pid_t pid;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ /* u1 backs the overlay sb: identity-mapped, but not the init ns. */
+ u1 = get_userns_fd(0, 0, 65536);
+ if (u1 < 0)
+ SKIP(return, "user namespaces not available");
+
+ pid = fork();
+ ASSERT_GE(pid, 0);
+ if (pid == 0) {
+ int ret = userns_overlay_child(u1);
+
+ _exit(ret ? EXIT_FAILURE : EXIT_SUCCESS);
+ }
+ EXPECT_EQ(wait_for_pid(pid), 0);
+
+ EXPECT_EQ(close(u1), 0);
+}
+
+/*
+ * An nfs_export overlay can be idmapped, and decodable file handles round-trip
+ * through the idmapped mount with correctly mapped ownership. Overlay file
+ * handles encode object identity, not ownership, so the mount idmap does not
+ * affect them; it only maps the owner reported once a handle is reopened.
+ */
+TEST_F(idmapped_overlay, nfs_export_handles)
+{
+ char path[PATH_MAX], mnt[128];
+ union {
+ struct file_handle fh;
+ char buf[sizeof(struct file_handle) + MAX_HANDLE_SZ];
+ } fhu;
+ struct file_handle *fh = &fhu.fh;
+ struct stat st;
+ int ovl, mfd, fd, mount_id;
+
+ if (!ovl_supported())
+ SKIP(return, "overlayfs not supported");
+
+ snprintf(path, sizeof(path), "%s/l/file", self->base);
+ ASSERT_EQ(mknod(path, S_IFREG | 0644, 0), 0);
+ ASSERT_EQ(chown(path, ID_HOST + 7, ID_HOST + 7), 0);
+
+ /* nfs_export=on gives decodable overlay file handles. */
+ ovl = ovl_mount(self->base, true);
+ if (ovl < 0)
+ SKIP(return, "overlayfs nfs_export not supported");
+ ASSERT_EQ(ovl_idmap(ovl), 0);
+
+ /* Attach the idmapped mount so handles can be resolved against it. */
+ snprintf(mnt, sizeof(mnt), "%s/mnt", self->base);
+ ASSERT_EQ(mkdir(mnt, 0755), 0);
+ ASSERT_EQ(sys_move_mount(ovl, "", AT_FDCWD, mnt,
+ MOVE_MOUNT_F_EMPTY_PATH), 0);
+
+ snprintf(path, sizeof(path), "%s/file", mnt);
+ fh->handle_bytes = MAX_HANDLE_SZ;
+ ASSERT_EQ(name_to_handle_at(AT_FDCWD, path, fh, &mount_id, 0), 0);
+
+ mfd = open(mnt, O_RDONLY | O_DIRECTORY);
+ ASSERT_GE(mfd, 0);
+ fd = open_by_handle_at(mfd, fh, O_RDONLY);
+ EXPECT_EQ(close(mfd), 0);
+ ASSERT_GE(fd, 0);
+
+ ASSERT_EQ(fstat(fd, &st), 0);
+ EXPECT_EQ(st.st_uid, ID_NS + 7);
+ EXPECT_EQ(st.st_gid, ID_NS + 7);
+
+ EXPECT_EQ(close(fd), 0);
+ EXPECT_EQ(close(ovl), 0);
+}
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c b/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
index 3c0b93183348..7a293544233d 100644
--- a/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
+++ b/tools/testing/selftests/filesystems/overlayfs/set_layers_via_fds.c
@@ -624,7 +624,7 @@ TEST_F(set_layers_via_fds, set_layers_via_detached_mount_fds)
ASSERT_EQ(sys_move_mount(fd_tmpfs, "", -EBADF, "/set_layers_via_fds_tmpfs", MOVE_MOUNT_F_EMPTY_PATH), 0);
- fd_tmp = open_tree(fd_tmpfs, "u", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ fd_tmp = sys_open_tree(fd_tmpfs, "u", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(fd_tmp, 0);
layer_fds[0] = openat(fd_tmp, "upper", O_CLOEXEC | O_DIRECTORY | O_PATH);
@@ -633,25 +633,25 @@ TEST_F(set_layers_via_fds, set_layers_via_detached_mount_fds)
layer_fds[1] = openat(fd_tmp, "work", O_CLOEXEC | O_DIRECTORY | O_PATH);
ASSERT_GE(layer_fds[1], 0);
- layer_fds[2] = open_tree(fd_tmpfs, "l1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[2] = sys_open_tree(fd_tmpfs, "l1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[2], 0);
- layer_fds[3] = open_tree(fd_tmpfs, "l2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[3] = sys_open_tree(fd_tmpfs, "l2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[3], 0);
- layer_fds[4] = open_tree(fd_tmpfs, "l3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[4] = sys_open_tree(fd_tmpfs, "l3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[4], 0);
- layer_fds[5] = open_tree(fd_tmpfs, "l4", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[5] = sys_open_tree(fd_tmpfs, "l4", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[5], 0);
- layer_fds[6] = open_tree(fd_tmpfs, "d1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[6] = sys_open_tree(fd_tmpfs, "d1", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[6], 0);
- layer_fds[7] = open_tree(fd_tmpfs, "d2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[7] = sys_open_tree(fd_tmpfs, "d2", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[7], 0);
- layer_fds[8] = open_tree(fd_tmpfs, "d3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
+ layer_fds[8] = sys_open_tree(fd_tmpfs, "d3", OPEN_TREE_CLONE | OPEN_TREE_CLOEXEC);
ASSERT_GE(layer_fds[8], 0);
ASSERT_EQ(close(fd_tmpfs), 0);
diff --git a/tools/testing/selftests/filesystems/statmount/statmount_test.c b/tools/testing/selftests/filesystems/statmount/statmount_test.c
index 8dc018d47a93..60c2c544db6a 100644
--- a/tools/testing/selftests/filesystems/statmount/statmount_test.c
+++ b/tools/testing/selftests/filesystems/statmount/statmount_test.c
@@ -82,6 +82,9 @@ static void cleanup_namespace(void)
{
int ret;
+ if (f_mountinfo)
+ fclose(f_mountinfo);
+
ret = fchdir(orig_root);
if (ret == -1)
ksft_perror("fchdir to original root");
@@ -515,7 +518,7 @@ static void test_statmount_mnt_opts(void)
return;
}
- ksft_test_result_fail("didnt't find mount entry\n");
+ ksft_test_result_fail("didn't find mount entry\n");
free(sm);
free(line);
}
diff --git a/tools/testing/selftests/filesystems/ustat_test.c b/tools/testing/selftests/filesystems/ustat_test.c
new file mode 100644
index 000000000000..d429fd18d779
--- /dev/null
+++ b/tools/testing/selftests/filesystems/ustat_test.c
@@ -0,0 +1,135 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Test ustat(2): looking up superblocks by device number.
+ *
+ * ustat() resolves a device number to a mounted superblock via
+ * user_get_super(). Check that the device number of a mounted tmpfs (an
+ * anonymous device) resolves, that it stops resolving once the filesystem
+ * is unmounted and that bogus device numbers report EINVAL.
+ */
+#define _GNU_SOURCE
+#include <errno.h>
+#include <fcntl.h>
+#include <sched.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+#include <sys/mount.h>
+#include <sys/stat.h>
+#include <sys/syscall.h>
+#include <unistd.h>
+
+#include "../kselftest_harness.h"
+
+/* struct ustat is not exported through UAPI, mirror include/linux/types.h. */
+struct ustat_buf {
+ int f_tfree;
+ unsigned long f_tinode;
+ char f_fname[6];
+ char f_fpack[6];
+ /* slack in case an architecture lays the struct out differently */
+ char pad[64];
+};
+
+#ifdef __NR_ustat
+
+/*
+ * The kernel decodes @dev with new_decode_dev(), which matches the low 32
+ * bits of the st_dev encoding stat(2) returns for any major below 4096.
+ */
+static int sys_ustat(unsigned int dev, struct ustat_buf *buf)
+{
+ return syscall(__NR_ustat, dev, buf);
+}
+
+static int write_string(const char *path, const char *string)
+{
+ ssize_t len = strlen(string);
+ int fd;
+
+ fd = open(path, O_WRONLY);
+ if (fd < 0)
+ return -1;
+ if (write(fd, string, len) != len) {
+ close(fd);
+ return -1;
+ }
+ return close(fd);
+}
+
+/* Enter namespaces in which mounting a tmpfs instance is allowed. */
+static int setup_namespaces(void)
+{
+ uid_t uid = getuid();
+ gid_t gid = getgid();
+ char map[64];
+
+ if (unshare(CLONE_NEWNS | (uid ? CLONE_NEWUSER : 0)))
+ return -1;
+
+ if (uid) {
+ if (write_string("/proc/self/setgroups", "deny"))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", uid);
+ if (write_string("/proc/self/uid_map", map))
+ return -1;
+ snprintf(map, sizeof(map), "0 %d 1", gid);
+ if (write_string("/proc/self/gid_map", map))
+ return -1;
+ }
+
+ return mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL);
+}
+
+TEST(resolves_mounted_superblock)
+{
+ char dir[] = "/tmp/ustat_test.XXXXXX";
+ struct ustat_buf ub;
+ struct stat st;
+
+ ASSERT_NE(NULL, mkdtemp(dir));
+
+ if (setup_namespaces()) {
+ rmdir(dir);
+ SKIP(return, "cannot set up namespaces: %s", strerror(errno));
+ }
+
+ ASSERT_EQ(0, mount("ustat_test", dir, "tmpfs", 0, NULL));
+ ASSERT_EQ(0, stat(dir, &st));
+
+ memset(&ub, 0xff, sizeof(ub));
+ ASSERT_EQ(0, sys_ustat(st.st_dev, &ub))
+ TH_LOG("ustat(%u): %s", (unsigned int)st.st_dev,
+ strerror(errno));
+
+ ASSERT_EQ(0, umount(dir));
+
+ /* The unmount removed the superblock, the device is gone. */
+ ASSERT_EQ(-1, sys_ustat(st.st_dev, &ub));
+ ASSERT_EQ(EINVAL, errno);
+
+ rmdir(dir);
+}
+
+TEST(bogus_device_numbers)
+{
+ struct ustat_buf ub;
+
+ ASSERT_EQ(-1, sys_ustat(0, &ub));
+ ASSERT_EQ(EINVAL, errno);
+
+ /* major 4095, minor 1048575: nothing plausible lives there */
+ ASSERT_EQ(-1, sys_ustat((0xfffu << 8) | 0xffu | (0xfff00u << 12), &ub));
+ ASSERT_EQ(EINVAL, errno);
+}
+
+#else /* !__NR_ustat */
+
+TEST(unsupported)
+{
+ SKIP(return, "ustat(2) is not available on this architecture");
+}
+
+#endif /* __NR_ustat */
+
+TEST_HARNESS_MAIN
diff --git a/tools/testing/selftests/proc/proc-pidns.c b/tools/testing/selftests/proc/proc-pidns.c
index 25b9a2933c45..6f7c10fe97b3 100644
--- a/tools/testing/selftests/proc/proc-pidns.c
+++ b/tools/testing/selftests/proc/proc-pidns.c
@@ -6,6 +6,7 @@
#include <assert.h>
#include <errno.h>
+#include <fcntl.h>
#include <sched.h>
#include <stdbool.h>
#include <stdlib.h>