diff options
Diffstat (limited to 'fs')
358 files changed, 3800 insertions, 3187 deletions
diff --git a/fs/9p/acl.c b/fs/9p/acl.c index ae7e7cf7523a..c6c7c47d32b9 100644 --- a/fs/9p/acl.c +++ b/fs/9p/acl.c @@ -140,7 +140,7 @@ struct posix_acl *v9fs_iop_get_inode_acl(struct inode *inode, int type, bool rcu } -struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, +struct posix_acl *v9fs_iop_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct v9fs_session_info *v9ses; @@ -152,7 +152,7 @@ struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, return v9fs_get_cached_acl(d_inode(dentry), type); } -int v9fs_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int v9fs_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int retval; diff --git a/fs/9p/acl.h b/fs/9p/acl.h index 333cfcc281da..2d1b24abcd3f 100644 --- a/fs/9p/acl.h +++ b/fs/9p/acl.h @@ -10,9 +10,9 @@ int v9fs_get_acl(struct inode *inode, struct p9_fid *fid); struct posix_acl *v9fs_iop_get_inode_acl(struct inode *inode, int type, bool rcu); -struct posix_acl *v9fs_iop_get_acl(struct mnt_idmap *idmap, +struct posix_acl *v9fs_iop_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int v9fs_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int v9fs_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int v9fs_acl_chmod(struct inode *inode, struct p9_fid *fid); int v9fs_set_create_acl(struct inode *inode, struct p9_fid *fid, diff --git a/fs/9p/v9fs.h b/fs/9p/v9fs.h index a462bcbfc7da..54a4a4ec5c15 100644 --- a/fs/9p/v9fs.h +++ b/fs/9p/v9fs.h @@ -188,7 +188,7 @@ extern struct dentry *v9fs_vfs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags); extern int v9fs_vfs_unlink(struct inode *i, struct dentry *d); extern int v9fs_vfs_rmdir(struct inode *i, struct dentry *d); -extern int v9fs_vfs_rename(struct mnt_idmap *idmap, +extern int v9fs_vfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); diff --git a/fs/9p/v9fs_vfs.h b/fs/9p/v9fs_vfs.h index 1856d91f8703..5e7b60042498 100644 --- a/fs/9p/v9fs_vfs.h +++ b/fs/9p/v9fs_vfs.h @@ -74,7 +74,7 @@ int v9fs_file_open(struct inode *inode, struct file *file); int v9fs_uflags2omode(int uflags, int extended); void v9fs_blank_wstat(struct p9_wstat *wstat); -int v9fs_vfs_setattr_dotl(struct mnt_idmap *idmap, +int v9fs_vfs_setattr_dotl(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); int v9fs_file_fsync_dotl(struct file *filp, loff_t start, loff_t end, int datasync); diff --git a/fs/9p/vfs_addr.c b/fs/9p/vfs_addr.c index 13cf87a5f90c..170a2b91c5f0 100644 --- a/fs/9p/vfs_addr.c +++ b/fs/9p/vfs_addr.c @@ -150,7 +150,6 @@ static int v9fs_init_request(struct netfs_io_request *rreq, struct file *file) struct p9_fid *fid; struct dentry *dentry; bool writing = (rreq->origin == NETFS_READ_FOR_WRITE || - rreq->origin == NETFS_WRITETHROUGH || rreq->origin == NETFS_UNBUFFERED_WRITE || rreq->origin == NETFS_DIO_WRITE); diff --git a/fs/9p/vfs_inode.c b/fs/9p/vfs_inode.c index 3829554ca369..c95e653344f6 100644 --- a/fs/9p/vfs_inode.c +++ b/fs/9p/vfs_inode.c @@ -652,7 +652,7 @@ error: */ static int -v9fs_vfs_create(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct v9fs_session_info *v9ses = v9fs_inode2v9ses(dir); @@ -679,7 +679,7 @@ v9fs_vfs_create(struct mnt_idmap *idmap, struct inode *dir, * */ -static struct dentry *v9fs_vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *v9fs_vfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { u32 perm; @@ -858,7 +858,7 @@ int v9fs_vfs_rmdir(struct inode *i, struct dentry *d) */ int -v9fs_vfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +v9fs_vfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -966,7 +966,7 @@ error: */ static int -v9fs_vfs_getattr(struct mnt_idmap *idmap, const struct path *path, +v9fs_vfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct dentry *dentry = path->dentry; @@ -1014,7 +1014,7 @@ v9fs_vfs_getattr(struct mnt_idmap *idmap, const struct path *path, * */ -static int v9fs_vfs_setattr(struct mnt_idmap *idmap, +static int v9fs_vfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int retval, use_dentry = 0; @@ -1249,7 +1249,7 @@ static int v9fs_vfs_mkspecial(struct inode *dir, struct dentry *dentry, */ static int -v9fs_vfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { p9_debug(P9_DEBUG_VFS, " %llu,%pd,%s\n", @@ -1304,7 +1304,7 @@ v9fs_vfs_link(struct dentry *old_dentry, struct inode *dir, */ static int -v9fs_vfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct v9fs_session_info *v9ses = v9fs_inode2v9ses(dir); diff --git a/fs/9p/vfs_inode_dotl.c b/fs/9p/vfs_inode_dotl.c index 116b29e95f21..2cd2898580a9 100644 --- a/fs/9p/vfs_inode_dotl.c +++ b/fs/9p/vfs_inode_dotl.c @@ -29,7 +29,7 @@ #include "acl.h" static int -v9fs_vfs_mknod_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode, dev_t rdev); /** @@ -216,7 +216,7 @@ int v9fs_open_to_dotl_flags(int flags) * */ static int -v9fs_vfs_create_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_create_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode) { return v9fs_vfs_mknod_dotl(idmap, dir, dentry, omode, 0); @@ -344,7 +344,7 @@ out: * */ -static struct dentry *v9fs_vfs_mkdir_dotl(struct mnt_idmap *idmap, +static struct dentry *v9fs_vfs_mkdir_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode) { @@ -414,7 +414,7 @@ error: } static int -v9fs_vfs_getattr_dotl(struct mnt_idmap *idmap, +v9fs_vfs_getattr_dotl(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -508,7 +508,7 @@ static int v9fs_mapped_iattr_valid(int iattr_valid) * */ -int v9fs_vfs_setattr_dotl(struct mnt_idmap *idmap, +int v9fs_vfs_setattr_dotl(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int retval, use_dentry = 0; @@ -682,7 +682,7 @@ v9fs_stat2inode_dotl(struct p9_stat_dotl *stat, struct inode *inode, } static int -v9fs_vfs_symlink_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_symlink_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int err; @@ -809,7 +809,7 @@ v9fs_vfs_link_dotl(struct dentry *old_dentry, struct inode *dir, * */ static int -v9fs_vfs_mknod_dotl(struct mnt_idmap *idmap, struct inode *dir, +v9fs_vfs_mknod_dotl(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t omode, dev_t rdev) { int err; diff --git a/fs/9p/xattr.c b/fs/9p/xattr.c index 8604e3377ee7..dac06587f67a 100644 --- a/fs/9p/xattr.c +++ b/fs/9p/xattr.c @@ -153,7 +153,7 @@ static int v9fs_xattr_handler_get(const struct xattr_handler *handler, } static int v9fs_xattr_handler_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/Kconfig b/fs/Kconfig index e05917adcd60..46313f65ee54 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -314,7 +314,6 @@ source "fs/ecryptfs/Kconfig" source "fs/hfs/Kconfig" source "fs/hfsplus/Kconfig" source "fs/befs/Kconfig" -source "fs/bfs/Kconfig" source "fs/jffs2/Kconfig" # UBIFS File system configuration source "fs/ubifs/Kconfig" @@ -421,4 +420,12 @@ source "fs/unicode/Kconfig" config IO_WQ bool +config FDTABLE_KUNIT_TEST + bool "KUnit test for fdtable" if !KUNIT_ALL_TESTS + depends on KUNIT=y + default KUNIT_ALL_TESTS + help + This builds the fdtable KUnit tests, which tests various aspects + of the fdtable structure and allocation. + endmenu diff --git a/fs/Makefile b/fs/Makefile index 055dfc23d82b..16f1108b64e1 100644 --- a/fs/Makefile +++ b/fs/Makefile @@ -76,7 +76,6 @@ obj-$(CONFIG_CODA_FS) += coda/ obj-$(CONFIG_MINIX_FS) += minix/ obj-$(CONFIG_FAT_FS) += fat/ obj-$(CONFIG_EXFAT_FS) += exfat/ -obj-$(CONFIG_BFS_FS) += bfs/ obj-$(CONFIG_ISO9660_FS) += isofs/ obj-$(CONFIG_HFSPLUS_FS) += hfsplus/ # Before hfs to find wrapped HFS+ obj-$(CONFIG_HFS_FS) += hfs/ diff --git a/fs/adfs/adfs.h b/fs/adfs/adfs.h index 0d32b7cd99b4..6003832277f8 100644 --- a/fs/adfs/adfs.h +++ b/fs/adfs/adfs.h @@ -144,7 +144,7 @@ struct adfs_discmap { /* Inode stuff */ struct inode *adfs_iget(struct super_block *sb, struct object_info *obj); int adfs_write_inode(struct inode *inode, struct writeback_control *wbc); -int adfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int adfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); /* map.c */ diff --git a/fs/adfs/dir.c b/fs/adfs/dir.c index 11afa9e157aa..b8cc6a697a05 100644 --- a/fs/adfs/dir.c +++ b/fs/adfs/dir.c @@ -191,7 +191,7 @@ static int adfs_dir_sync(struct adfs_dir *dir) for (i = dir->nr_buffers - 1; i >= 0; i--) { struct buffer_head *bh = dir->bhs[i]; sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; } diff --git a/fs/adfs/inode.c b/fs/adfs/inode.c index 4ac442d0a8c0..34598b499372 100644 --- a/fs/adfs/inode.c +++ b/fs/adfs/inode.c @@ -299,7 +299,7 @@ out: * later. */ int -adfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) +adfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); struct super_block *sb = inode->i_sb; diff --git a/fs/affs/affs.h b/fs/affs/affs.h index d1c506c1f310..2518de96a0c8 100644 --- a/fs/affs/affs.h +++ b/fs/affs/affs.h @@ -166,17 +166,17 @@ extern const struct export_operations affs_export_ops; extern int affs_hash_name(struct super_block *sb, const u8 *name, unsigned int len); extern struct dentry *affs_lookup(struct inode *dir, struct dentry *dentry, unsigned int); extern int affs_unlink(struct inode *dir, struct dentry *dentry); -extern int affs_create(struct mnt_idmap *idmap, struct inode *dir, +extern int affs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); -extern struct dentry *affs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +extern struct dentry *affs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); extern int affs_rmdir(struct inode *dir, struct dentry *dentry); extern int affs_link(struct dentry *olddentry, struct inode *dir, struct dentry *dentry); -extern int affs_symlink(struct mnt_idmap *idmap, +extern int affs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname); -extern int affs_rename2(struct mnt_idmap *idmap, +extern int affs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); @@ -184,7 +184,7 @@ extern int affs_rename2(struct mnt_idmap *idmap, /* inode.c */ extern struct inode *affs_new_inode(struct inode *dir); -extern int affs_setattr(struct mnt_idmap *idmap, +extern int affs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void affs_evict_inode(struct inode *inode); extern struct inode *affs_iget(struct super_block *sb, diff --git a/fs/affs/inode.c b/fs/affs/inode.c index d4a3f381c4bc..2a48d7422091 100644 --- a/fs/affs/inode.c +++ b/fs/affs/inode.c @@ -213,7 +213,7 @@ affs_write_inode(struct inode *inode, struct writeback_control *wbc) } int -affs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) +affs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); int error; diff --git a/fs/affs/namei.c b/fs/affs/namei.c index 6cb52efafe5f..2e32899a32d5 100644 --- a/fs/affs/namei.c +++ b/fs/affs/namei.c @@ -242,7 +242,7 @@ affs_unlink(struct inode *dir, struct dentry *dentry) } int -affs_create(struct mnt_idmap *idmap, struct inode *dir, +affs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -274,7 +274,7 @@ affs_create(struct mnt_idmap *idmap, struct inode *dir, } struct dentry * -affs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +affs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -313,7 +313,7 @@ affs_rmdir(struct inode *dir, struct dentry *dentry) } int -affs_symlink(struct mnt_idmap *idmap, struct inode *dir, +affs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct super_block *sb = dir->i_sb; @@ -503,7 +503,7 @@ done: return retval; } -int affs_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +int affs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/afs/dir.c b/fs/afs/dir.c index 2db534a2c7cc..75f8f70cfeff 100644 --- a/fs/afs/dir.c +++ b/fs/afs/dir.c @@ -33,17 +33,17 @@ static bool afs_lookup_one_filldir(struct dir_context *ctx, const char *name, in static bool afs_lookup_filldir(struct dir_context *ctx, const char *name, int nlen, u64 ino, u32 uniquifier); #define AFS_LOOKUP ((filldir_t)0x137UL) -static int afs_create(struct mnt_idmap *idmap, struct inode *dir, +static int afs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); -static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *afs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode); static int afs_rmdir(struct inode *dir, struct dentry *dentry); static int afs_unlink(struct inode *dir, struct dentry *dentry); static int afs_link(struct dentry *from, struct inode *dir, struct dentry *dentry); -static int afs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int afs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *content); -static int afs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int afs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags); static int afs_dir_writepages(struct address_space *mapping, @@ -1310,7 +1310,7 @@ static const struct afs_operation_ops afs_mkdir_operation = { /* * create a directory on an AFS filesystem */ -static struct dentry *afs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *afs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct afs_operation *op; @@ -1632,7 +1632,7 @@ static const struct afs_operation_ops afs_create_operation = { /* * create a regular file on an AFS filesystem */ -static int afs_create(struct mnt_idmap *idmap, struct inode *dir, +static int afs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct afs_operation *op; @@ -1779,7 +1779,7 @@ static const struct afs_operation_ops afs_symlink_operation = { /* * create a symlink in an AFS filesystem */ -static int afs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int afs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *content) { struct afs_operation *op; @@ -2067,7 +2067,7 @@ static const struct afs_operation_ops afs_rename_exchange_operation = { /* * rename a file in an AFS filesystem and/or move it between directories */ -static int afs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int afs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/afs/file.c b/fs/afs/file.c index 0467742bfeee..4c78d3441785 100644 --- a/fs/afs/file.c +++ b/fs/afs/file.c @@ -400,7 +400,6 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) } break; case NETFS_WRITEBACK: - case NETFS_WRITETHROUGH: case NETFS_UNBUFFERED_WRITE: case NETFS_DIO_WRITE: if (S_ISREG(rreq->inode->i_mode)) @@ -413,7 +412,7 @@ static int afs_init_request(struct netfs_io_request *rreq, struct file *file) return 0; } -static int afs_check_write_begin(struct file *file, loff_t pos, unsigned len, +static int afs_check_write_begin(struct file *file, uoff_t pos, unsigned len, struct folio **foliop, void **_fsdata) { struct afs_vnode *vnode = AFS_FS_I(file_inode(file)); @@ -434,7 +433,7 @@ static void afs_free_request(struct netfs_io_request *rreq) * Also, estimate the number of 512 bytes blocks used, rounded up to nearest 1K * for consistency with other AFS clients. */ -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size) { struct inode *inode = &vnode->netfs.inode; loff_t i_size; @@ -448,10 +447,9 @@ void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size) } spin_unlock(&inode->i_lock); write_sequnlock(&vnode->cb_lock); - fscache_update_cookie(afs_vnode_cache(vnode), NULL, &new_i_size); } -static void afs_update_i_size(struct inode *inode, loff_t new_i_size) +static void afs_update_i_size(struct inode *inode, uoff_t new_i_size) { afs_set_i_size(AFS_FS_I(inode), new_i_size); } diff --git a/fs/afs/inode.c b/fs/afs/inode.c index 14f39a9bea6c..8e6ca6b45c6b 100644 --- a/fs/afs/inode.c +++ b/fs/afs/inode.c @@ -596,7 +596,7 @@ error: /* * read the attributes of an inode */ -int afs_getattr(struct mnt_idmap *idmap, const struct path *path, +int afs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -759,7 +759,7 @@ static const struct afs_operation_ops afs_setattr_operation = { /* * set the attributes of an inode */ -int afs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int afs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { const unsigned int supported = diff --git a/fs/afs/internal.h b/fs/afs/internal.h index 330654ed16ec..5744a347ce2e 100644 --- a/fs/afs/internal.h +++ b/fs/afs/internal.h @@ -1171,7 +1171,7 @@ extern int afs_open(struct inode *, struct file *); extern int afs_release(struct inode *, struct file *); void afs_fetch_data_async_rx(struct work_struct *work); void afs_fetch_data_immediate_cancel(struct afs_call *call); -void afs_set_i_size(struct afs_vnode *vnode, loff_t new_i_size); +void afs_set_i_size(struct afs_vnode *vnode, uoff_t new_i_size); /* * flock.c @@ -1266,9 +1266,9 @@ extern int afs_fetch_status(struct afs_vnode *, struct key *, bool, afs_access_t extern int afs_ilookup5_test_by_fid(struct inode *, void *); extern struct inode *afs_iget(struct afs_operation *, struct afs_vnode_param *); extern struct inode *afs_root_iget(struct super_block *, struct key *); -extern int afs_getattr(struct mnt_idmap *idmap, const struct path *, +extern int afs_getattr(const struct mnt_idmap *idmap, const struct path *, struct kstat *, u32, unsigned int); -extern int afs_setattr(struct mnt_idmap *idmap, struct dentry *, struct iattr *); +extern int afs_setattr(const struct mnt_idmap *idmap, struct dentry *, struct iattr *); extern void afs_evict_inode(struct inode *); extern int afs_drop_inode(struct inode *); @@ -1538,7 +1538,7 @@ extern void afs_cache_permit(struct afs_vnode *, struct key *, unsigned int, extern struct key *afs_request_key(struct afs_cell *); extern struct key *afs_request_key_rcu(struct afs_cell *); extern int afs_check_permit(struct afs_vnode *, struct key *, afs_access_t *); -extern int afs_permission(struct mnt_idmap *, struct inode *, int); +extern int afs_permission(const struct mnt_idmap *, struct inode *, int); extern void __exit afs_clean_up_permit_cache(void); /* diff --git a/fs/afs/security.c b/fs/afs/security.c index 6d00d62a65ed..fd040f7c6766 100644 --- a/fs/afs/security.c +++ b/fs/afs/security.c @@ -428,7 +428,7 @@ int afs_check_permit(struct afs_vnode *vnode, struct key *key, * - AFS ACLs are attached to directories only, and a file is controlled by its * parent directory's ACL */ -int afs_permission(struct mnt_idmap *idmap, struct inode *inode, +int afs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct afs_vnode *vnode = AFS_FS_I(inode); diff --git a/fs/afs/xattr.c b/fs/afs/xattr.c index 3770ed236f67..bcffd7236cc9 100644 --- a/fs/afs/xattr.c +++ b/fs/afs/xattr.c @@ -97,7 +97,7 @@ static const struct afs_operation_ops afs_store_acl_operation = { * Set a file's AFS3 ACL. */ static int afs_xattr_set_acl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -228,7 +228,7 @@ static const struct afs_operation_ops yfs_store_opaque_acl2_operation = { * Set a file's YFS ACL. */ static int afs_xattr_set_yfs(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -936,7 +936,7 @@ static int kill_ioctx(struct mm_struct *mm, struct kioctx *ctx, /* * exit_aio: called when the last user of mm goes away. At this point, there is - * no way for any new requests to be submited or any of the io_* syscalls to be + * no way for any new requests to be submitted or any of the io_* syscalls to be * called on the context. * * There may be outstanding kiocbs, but free_ioctx() will explicitly wait on @@ -1280,7 +1280,7 @@ static long aio_read_events_ring(struct kioctx *ctx, * The mutex can block and wake us up and that will cause * wait_event_interruptible_hrtimeout() to schedule without sleeping * and repeat. This should be rare enough that it doesn't cause - * peformance issues. See the comment in read_events() for more detail. + * performance issues. See the comment in read_events() for more detail. */ sched_annotate_sleep(); mutex_lock(&ctx->ring_lock); @@ -1869,7 +1869,12 @@ static int aio_poll_wake(struct wait_queue_entry *wait, unsigned mode, int sync, list_del_init(&req->wait.entry); list_del(&iocb->ki_list); iocb->ki_res.res = mangle_poll(mask); - if (iocb->ki_eventfd && !eventfd_signal_allowed()) { + /* + * We hold an arbitrary provider waitqueue lock here. Signaling a + * result eventfd can feed back through epoll and try to take the same + * lock again. Defer all eventfd-backed poll completions. + */ + if (iocb->ki_eventfd) { iocb = NULL; INIT_WORK(&req->work, aio_poll_put_work); schedule_work(&req->work); diff --git a/fs/anon_inodes.c b/fs/anon_inodes.c index a7b9b948e33d..8f07c8d9bda0 100644 --- a/fs/anon_inodes.c +++ b/fs/anon_inodes.c @@ -46,7 +46,7 @@ static struct inode *anon_inode_inode __ro_after_init; * Rather than mess with our internal sane inode data, just fix it * up here in getattr() by masking off the format bits. */ -int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, +int anon_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -57,7 +57,7 @@ int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int anon_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int anon_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return -EOPNOTSUPP; diff --git a/fs/attr.c b/fs/attr.c index 71888ac903c2..9706d370f34a 100644 --- a/fs/attr.c +++ b/fs/attr.c @@ -30,7 +30,7 @@ * * Return: ATTR_KILL_SGID if setgid bit needs to be removed, 0 otherwise. */ -int setattr_should_drop_sgid(struct mnt_idmap *idmap, +int setattr_should_drop_sgid(const struct mnt_idmap *idmap, const struct inode *inode) { umode_t mode = inode->i_mode; @@ -60,7 +60,7 @@ EXPORT_SYMBOL(setattr_should_drop_sgid); * Return: A mask of ATTR_KILL_S{G,U}ID indicating which - if any - setid bits * to remove, 0 otherwise. */ -int setattr_should_drop_suidgid(struct mnt_idmap *idmap, +int setattr_should_drop_suidgid(const struct mnt_idmap *idmap, struct inode *inode) { umode_t mode = inode->i_mode; @@ -91,7 +91,7 @@ EXPORT_SYMBOL(setattr_should_drop_suidgid); * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -static bool chown_ok(struct mnt_idmap *idmap, +static bool chown_ok(const struct mnt_idmap *idmap, const struct inode *inode, vfsuid_t ia_vfsuid) { vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); @@ -118,7 +118,7 @@ static bool chown_ok(struct mnt_idmap *idmap, * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -static bool chgrp_ok(struct mnt_idmap *idmap, +static bool chgrp_ok(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t ia_vfsgid) { vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); @@ -158,7 +158,7 @@ static bool chgrp_ok(struct mnt_idmap *idmap, * Should be called as the first thing in ->setattr implementations, * possibly after taking additional locks. */ -int setattr_prepare(struct mnt_idmap *idmap, struct dentry *dentry, +int setattr_prepare(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -339,7 +339,7 @@ static void setattr_copy_mgtime(struct inode *inode, const struct iattr *attr) * that for "simple" filesystems, the struct inode is the inode storage. * The caller is free to mark the inode dirty afterwards if needed. */ -void setattr_copy(struct mnt_idmap *idmap, struct inode *inode, +void setattr_copy(const struct mnt_idmap *idmap, struct inode *inode, const struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; @@ -369,7 +369,7 @@ void setattr_copy(struct mnt_idmap *idmap, struct inode *inode, } EXPORT_SYMBOL(setattr_copy); -int may_setattr(struct mnt_idmap *idmap, struct inode *inode, +int may_setattr(const struct mnt_idmap *idmap, struct inode *inode, unsigned int ia_valid) { int error; @@ -424,7 +424,7 @@ EXPORT_SYMBOL(may_setattr); * permissions. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -int notify_change(struct mnt_idmap *idmap, struct dentry *dentry, +int notify_change(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct delegated_inode *delegated_inode) { struct inode *inode = dentry->d_inode; diff --git a/fs/autofs/root.c b/fs/autofs/root.c index b36439f4521e..28f38f5d0236 100644 --- a/fs/autofs/root.c +++ b/fs/autofs/root.c @@ -11,12 +11,12 @@ #include "autofs_i.h" -static int autofs_dir_permission(struct mnt_idmap *, struct inode *, int); -static int autofs_dir_symlink(struct mnt_idmap *, struct inode *, +static int autofs_dir_permission(const struct mnt_idmap *, struct inode *, int); +static int autofs_dir_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *); static int autofs_dir_unlink(struct inode *, struct dentry *); static int autofs_dir_rmdir(struct inode *, struct dentry *); -static struct dentry *autofs_dir_mkdir(struct mnt_idmap *, struct inode *, +static struct dentry *autofs_dir_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); static long autofs_root_ioctl(struct file *, unsigned int, unsigned long); #ifdef CONFIG_COMPAT @@ -552,7 +552,7 @@ static struct dentry *autofs_lookup(struct inode *dir, return NULL; } -static int autofs_dir_permission(struct mnt_idmap *idmap, +static int autofs_dir_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { if (mask & MAY_WRITE) { @@ -572,7 +572,7 @@ static int autofs_dir_permission(struct mnt_idmap *idmap, return generic_permission(idmap, inode, mask); } -static int autofs_dir_symlink(struct mnt_idmap *idmap, +static int autofs_dir_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { @@ -724,7 +724,7 @@ static int autofs_dir_rmdir(struct inode *dir, struct dentry *dentry) return 0; } -static struct dentry *autofs_dir_mkdir(struct mnt_idmap *idmap, +static struct dentry *autofs_dir_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { diff --git a/fs/backing-file.c b/fs/backing-file.c index cc101143f921..5614cb7801e1 100644 --- a/fs/backing-file.c +++ b/fs/backing-file.c @@ -59,7 +59,7 @@ struct file *backing_tmpfile_open(const struct file *user_file, int flags, const struct path *real_parentpath, umode_t mode, const struct cred *cred) { - struct mnt_idmap *real_idmap = mnt_idmap(real_parentpath->mnt); + const struct mnt_idmap *real_idmap = mnt_idmap(real_parentpath->mnt); const struct path *user_path = &user_file->f_path; struct file *f; int error; diff --git a/fs/bad_inode.c b/fs/bad_inode.c index 486c40f73e51..bea9f4876ee0 100644 --- a/fs/bad_inode.c +++ b/fs/bad_inode.c @@ -27,7 +27,7 @@ static const struct file_operations bad_file_ops = .open = bad_file_open, }; -static int bad_inode_create(struct mnt_idmap *idmap, +static int bad_inode_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { @@ -51,14 +51,14 @@ static int bad_inode_unlink(struct inode *dir, struct dentry *dentry) return -EIO; } -static int bad_inode_symlink(struct mnt_idmap *idmap, +static int bad_inode_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { return -EIO; } -static struct dentry *bad_inode_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *bad_inode_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(-EIO); @@ -69,13 +69,13 @@ static int bad_inode_rmdir (struct inode *dir, struct dentry *dentry) return -EIO; } -static int bad_inode_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int bad_inode_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { return -EIO; } -static int bad_inode_rename2(struct mnt_idmap *idmap, +static int bad_inode_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -89,20 +89,20 @@ static int bad_inode_readlink(struct dentry *dentry, char __user *buffer, return -EIO; } -static int bad_inode_permission(struct mnt_idmap *idmap, +static int bad_inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return -EIO; } -static int bad_inode_getattr(struct mnt_idmap *idmap, +static int bad_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { return -EIO; } -static int bad_inode_setattr(struct mnt_idmap *idmap, +static int bad_inode_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs) { return -EIO; @@ -146,14 +146,14 @@ static int bad_inode_atomic_open(struct inode *inode, struct dentry *dentry, return -EIO; } -static int bad_inode_tmpfile(struct mnt_idmap *idmap, +static int bad_inode_tmpfile(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, umode_t mode) { return -EIO; } -static int bad_inode_set_acl(struct mnt_idmap *idmap, +static int bad_inode_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { diff --git a/fs/bfs/Kconfig b/fs/bfs/Kconfig deleted file mode 100644 index 8e7ef866b62a..000000000000 --- a/fs/bfs/Kconfig +++ /dev/null @@ -1,21 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -config BFS_FS - tristate "BFS file system support" - depends on BLOCK - select BUFFER_HEAD - help - Boot File System (BFS) is a file system used under SCO UnixWare to - allow the bootloader access to the kernel image and other important - files during the boot process. It is usually mounted under /stand - and corresponds to the slice marked as "STAND" in the UnixWare - partition. You should say Y if you want to read or write the files - on your /stand slice from within Linux. You then also need to say Y - to "UnixWare slices support", below. More information about the BFS - file system is contained in the file - <file:Documentation/filesystems/bfs.rst>. - - If you don't know what this is about, say N. - - To compile this as a module, choose M here: the module will be called - bfs. Note that the file system of your root partition (the one - containing the directory /) cannot be compiled as a module. diff --git a/fs/bfs/Makefile b/fs/bfs/Makefile deleted file mode 100644 index 2b6bc5eb4de9..000000000000 --- a/fs/bfs/Makefile +++ /dev/null @@ -1,8 +0,0 @@ -# SPDX-License-Identifier: GPL-2.0-only -# -# Makefile for BFS filesystem. -# - -obj-$(CONFIG_BFS_FS) += bfs.o - -bfs-objs := inode.o file.o dir.o diff --git a/fs/bfs/bfs.h b/fs/bfs/bfs.h deleted file mode 100644 index b08afe733e63..000000000000 --- a/fs/bfs/bfs.h +++ /dev/null @@ -1,69 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -/* - * fs/bfs/bfs.h - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - */ -#ifndef _FS_BFS_BFS_H -#define _FS_BFS_BFS_H - -#include <linux/bfs_fs.h> - -/* In theory BFS supports up to 512 inodes, numbered from 2 (for /) up to 513 inclusive. - In actual fact, attempting to create the 512th inode (i.e. inode No. 513 or file No. 511) - will fail with ENOSPC in bfs_add_entry(): the root directory cannot contain so many entries, counting '..'. - So, mkfs.bfs(8) should really limit its -N option to 511 and not 512. For now, we just print a warning - if a filesystem is mounted with such "impossible to fill up" number of inodes */ -#define BFS_MAX_LASTI 513 - -/* - * BFS file system in-core superblock info - */ -struct bfs_sb_info { - unsigned long si_blocks; - unsigned long si_freeb; - unsigned long si_freei; - unsigned long si_lf_eblk; - unsigned long si_lasti; - DECLARE_BITMAP(si_imap, BFS_MAX_LASTI+1); - struct mutex bfs_lock; -}; - -/* - * BFS file system in-core inode info - */ -struct bfs_inode_info { - unsigned long i_dsk_ino; /* inode number from the disk, can be 0 */ - unsigned long i_sblock; - unsigned long i_eblock; - struct mapping_metadata_bhs i_metadata_bhs; - struct inode vfs_inode; -}; - -static inline struct bfs_sb_info *BFS_SB(struct super_block *sb) -{ - return sb->s_fs_info; -} - -static inline struct bfs_inode_info *BFS_I(struct inode *inode) -{ - return container_of(inode, struct bfs_inode_info, vfs_inode); -} - - -#define printf(format, args...) \ - printk(KERN_ERR "BFS-fs: %s(): " format, __func__, ## args) - -/* inode.c */ -extern struct inode *bfs_iget(struct super_block *sb, unsigned long ino); -extern void bfs_dump_imap(const char *, struct super_block *); - -/* file.c */ -extern const struct inode_operations bfs_file_inops; -extern const struct file_operations bfs_file_operations; -extern const struct address_space_operations bfs_aops; - -/* dir.c */ -extern const struct inode_operations bfs_dir_inops; -extern const struct file_operations bfs_dir_operations; - -#endif /* _FS_BFS_BFS_H */ diff --git a/fs/bfs/dir.c b/fs/bfs/dir.c index 91a4871fa051..b944bd62f5d0 100644 --- a/fs/bfs/dir.c +++ b/fs/bfs/dir.c @@ -75,7 +75,7 @@ const struct file_operations bfs_dir_operations = { .llseek = generic_file_llseek, }; -static int bfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int bfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int err; @@ -199,7 +199,7 @@ out_brelse: return error; } -static int bfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int bfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/bfs/file.c b/fs/bfs/file.c deleted file mode 100644 index d33d6bde992b..000000000000 --- a/fs/bfs/file.c +++ /dev/null @@ -1,203 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0 -/* - * fs/bfs/file.c - * BFS file operations. - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - * - * Make the file block allocation algorithm understand the size - * of the underlying block device. - * Copyright (C) 2007 Dmitri Vorobiev <dmitri.vorobiev@gmail.com> - * - */ - -#include <linux/fs.h> -#include <linux/mpage.h> -#include <linux/buffer_head.h> -#include "bfs.h" - -#undef DEBUG - -#ifdef DEBUG -#define dprintf(x...) printf(x) -#else -#define dprintf(x...) -#endif - -const struct file_operations bfs_file_operations = { - .llseek = generic_file_llseek, - .read_iter = generic_file_read_iter, - .write_iter = generic_file_write_iter, - .mmap_prepare = generic_file_mmap_prepare, - .splice_read = filemap_splice_read, -}; - -static int bfs_move_block(unsigned long from, unsigned long to, - struct super_block *sb) -{ - struct buffer_head *bh, *new; - - bh = sb_bread(sb, from); - if (!bh) - return -EIO; - new = sb_getblk(sb, to); - memcpy(new->b_data, bh->b_data, bh->b_size); - mark_buffer_dirty(new); - bforget(bh); - brelse(new); - return 0; -} - -static int bfs_move_blocks(struct super_block *sb, unsigned long start, - unsigned long end, unsigned long where) -{ - unsigned long i; - - dprintf("%08lx-%08lx->%08lx\n", start, end, where); - for (i = start; i <= end; i++) - if(bfs_move_block(i, where + i, sb)) { - dprintf("failed to move block %08lx -> %08lx\n", i, - where + i); - return -EIO; - } - return 0; -} - -static int bfs_get_block(struct inode *inode, sector_t block, - struct buffer_head *bh_result, int create) -{ - unsigned long phys; - int err; - struct super_block *sb = inode->i_sb; - struct bfs_sb_info *info = BFS_SB(sb); - struct bfs_inode_info *bi = BFS_I(inode); - - phys = bi->i_sblock + block; - if (!create) { - if (phys <= bi->i_eblock) { - dprintf("c=%d, b=%08lx, phys=%09lx (granted)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - } - return 0; - } - - /* - * If the file is not empty and the requested block is within the - * range of blocks allocated for this file, we can grant it. - */ - if (bi->i_sblock && (phys <= bi->i_eblock)) { - dprintf("c=%d, b=%08lx, phys=%08lx (interim block granted)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - return 0; - } - - /* The file will be extended, so let's see if there is enough space. */ - if (phys >= info->si_blocks) - return -ENOSPC; - - /* The rest has to be protected against itself. */ - mutex_lock(&info->bfs_lock); - - /* - * If the last data block for this file is the last allocated - * block, we can extend the file trivially, without moving it - * anywhere. - */ - if (bi->i_eblock == info->si_lf_eblk) { - dprintf("c=%d, b=%08lx, phys=%08lx (simple extension)\n", - create, (unsigned long)block, phys); - map_bh(bh_result, sb, phys); - info->si_freeb -= phys - bi->i_eblock; - info->si_lf_eblk = bi->i_eblock = phys; - mark_inode_dirty(inode); - err = 0; - goto out; - } - - /* Ok, we have to move this entire file to the next free block. */ - phys = info->si_lf_eblk + 1; - if (phys + block >= info->si_blocks) { - err = -ENOSPC; - goto out; - } - - if (bi->i_sblock) { - err = bfs_move_blocks(inode->i_sb, bi->i_sblock, - bi->i_eblock, phys); - if (err) { - dprintf("failed to move ino=%08lx -> fs corruption\n", - inode->i_ino); - goto out; - } - } else - err = 0; - - dprintf("c=%d, b=%08lx, phys=%08lx (moved)\n", - create, (unsigned long)block, phys); - bi->i_sblock = phys; - phys += block; - info->si_lf_eblk = bi->i_eblock = phys; - - /* - * This assumes nothing can write the inode back while we are here - * and thus update inode->i_blocks! (XXX) - */ - info->si_freeb -= bi->i_eblock - bi->i_sblock + 1 - inode->i_blocks; - mark_inode_dirty(inode); - map_bh(bh_result, sb, phys); -out: - mutex_unlock(&info->bfs_lock); - return err; -} - -static int bfs_writepages(struct address_space *mapping, - struct writeback_control *wbc) -{ - return mpage_writepages(mapping, wbc, bfs_get_block); -} - -static int bfs_read_folio(struct file *file, struct folio *folio) -{ - return block_read_full_folio(folio, bfs_get_block); -} - -static void bfs_write_failed(struct address_space *mapping, loff_t to) -{ - struct inode *inode = mapping->host; - - if (to > inode->i_size) - truncate_pagecache(inode, inode->i_size); -} - -static int bfs_write_begin(const struct kiocb *iocb, - struct address_space *mapping, - loff_t pos, unsigned len, - struct folio **foliop, void **fsdata) -{ - int ret; - - ret = block_write_begin(mapping, pos, len, foliop, bfs_get_block); - if (unlikely(ret)) - bfs_write_failed(mapping, pos + len); - - return ret; -} - -static sector_t bfs_bmap(struct address_space *mapping, sector_t block) -{ - return generic_block_bmap(mapping, block, bfs_get_block); -} - -const struct address_space_operations bfs_aops = { - .dirty_folio = block_dirty_folio, - .invalidate_folio = block_invalidate_folio, - .read_folio = bfs_read_folio, - .writepages = bfs_writepages, - .write_begin = bfs_write_begin, - .write_end = generic_write_end, - .migrate_folio = buffer_migrate_folio, - .bmap = bfs_bmap, -}; - -const struct inode_operations bfs_file_inops; diff --git a/fs/bfs/inode.c b/fs/bfs/inode.c deleted file mode 100644 index 06e3a848b4ef..000000000000 --- a/fs/bfs/inode.c +++ /dev/null @@ -1,538 +0,0 @@ -// SPDX-License-Identifier: GPL-2.0-only -/* - * fs/bfs/inode.c - * BFS superblock and inode operations. - * Copyright (C) 1999-2018 Tigran Aivazian <aivazian.tigran@gmail.com> - * From fs/minix, Copyright (C) 1991, 1992 Linus Torvalds. - * Made endianness-clean by Andrew Stribblehill <ads@wompom.org>, 2005. - */ - -#include <linux/module.h> -#include <linux/mm.h> -#include <linux/slab.h> -#include <linux/init.h> -#include <linux/fs.h> -#include <linux/buffer_head.h> -#include <linux/vfs.h> -#include <linux/writeback.h> -#include <linux/uio.h> -#include <linux/uaccess.h> -#include <linux/fs_context.h> -#include "bfs.h" - -MODULE_AUTHOR("Tigran Aivazian <aivazian.tigran@gmail.com>"); -MODULE_DESCRIPTION("SCO UnixWare BFS filesystem for Linux"); -MODULE_LICENSE("GPL"); - -#undef DEBUG - -#ifdef DEBUG -#define dprintf(x...) printf(x) -#else -#define dprintf(x...) -#endif - -struct inode *bfs_iget(struct super_block *sb, unsigned long ino) -{ - struct bfs_inode *di; - struct inode *inode; - struct buffer_head *bh; - int block, off; - - inode = iget_locked(sb, ino); - if (!inode) - return ERR_PTR(-ENOMEM); - if (!(inode_state_read_once(inode) & I_NEW)) - return inode; - - if ((ino < BFS_ROOT_INO) || (ino > BFS_SB(inode->i_sb)->si_lasti)) { - printf("Bad inode number %s:%08lx\n", inode->i_sb->s_id, ino); - goto error; - } - - block = (ino - BFS_ROOT_INO) / BFS_INODES_PER_BLOCK + 1; - bh = sb_bread(inode->i_sb, block); - if (!bh) { - printf("Unable to read inode %s:%08lx\n", inode->i_sb->s_id, - ino); - goto error; - } - - off = (ino - BFS_ROOT_INO) % BFS_INODES_PER_BLOCK; - di = (struct bfs_inode *)bh->b_data + off; - - /* - * https://martin.hinner.info/fs/bfs/bfs-structure.html explains that - * BFS in SCO UnixWare environment used only lower 9 bits of di->i_mode - * value. This means that, although bfs_write_inode() saves whole - * inode->i_mode bits (which include S_IFMT bits and S_IS{UID,GID,VTX} - * bits), middle 7 bits of di->i_mode value can be garbage when these - * bits were not saved by bfs_write_inode(). - * Since we can't tell whether middle 7 bits are garbage, use only - * lower 12 bits (i.e. tolerate S_IS{UID,GID,VTX} bits possibly being - * garbage) and reconstruct S_IFMT bits for Linux environment from - * di->i_vtype value. - */ - inode->i_mode = 0x00000FFF & le32_to_cpu(di->i_mode); - if (le32_to_cpu(di->i_vtype) == BFS_VDIR) { - inode->i_mode |= S_IFDIR; - inode->i_op = &bfs_dir_inops; - inode->i_fop = &bfs_dir_operations; - } else if (le32_to_cpu(di->i_vtype) == BFS_VREG) { - inode->i_mode |= S_IFREG; - inode->i_op = &bfs_file_inops; - inode->i_fop = &bfs_file_operations; - inode->i_mapping->a_ops = &bfs_aops; - } else { - brelse(bh); - printf("Unknown vtype=%u %s:%08lx\n", - le32_to_cpu(di->i_vtype), inode->i_sb->s_id, ino); - goto error; - } - - BFS_I(inode)->i_sblock = le32_to_cpu(di->i_sblock); - BFS_I(inode)->i_eblock = le32_to_cpu(di->i_eblock); - BFS_I(inode)->i_dsk_ino = le16_to_cpu(di->i_ino); - i_uid_write(inode, le32_to_cpu(di->i_uid)); - i_gid_write(inode, le32_to_cpu(di->i_gid)); - set_nlink(inode, le32_to_cpu(di->i_nlink)); - inode->i_size = BFS_FILESIZE(di); - inode->i_blocks = BFS_FILEBLOCKS(di); - inode_set_atime(inode, le32_to_cpu(di->i_atime), 0); - inode_set_mtime(inode, le32_to_cpu(di->i_mtime), 0); - inode_set_ctime(inode, le32_to_cpu(di->i_ctime), 0); - - brelse(bh); - unlock_new_inode(inode); - return inode; - -error: - iget_failed(inode); - return ERR_PTR(-EIO); -} - -static struct bfs_inode *find_inode(struct super_block *sb, u16 ino, struct buffer_head **p) -{ - if ((ino < BFS_ROOT_INO) || (ino > BFS_SB(sb)->si_lasti)) { - printf("Bad inode number %s:%08x\n", sb->s_id, ino); - return ERR_PTR(-EIO); - } - - ino -= BFS_ROOT_INO; - - *p = sb_bread(sb, 1 + ino / BFS_INODES_PER_BLOCK); - if (!*p) { - printf("Unable to read inode %s:%08x\n", sb->s_id, ino); - return ERR_PTR(-EIO); - } - - return (struct bfs_inode *)(*p)->b_data + ino % BFS_INODES_PER_BLOCK; -} - -static int bfs_write_inode(struct inode *inode, struct writeback_control *wbc) -{ - struct bfs_sb_info *info = BFS_SB(inode->i_sb); - unsigned int ino = (u16)inode->i_ino; - unsigned long i_sblock; - struct bfs_inode *di; - struct buffer_head *bh; - - dprintf("ino=%08x\n", ino); - - di = find_inode(inode->i_sb, ino, &bh); - if (IS_ERR(di)) - return PTR_ERR(di); - - mutex_lock(&info->bfs_lock); - - if (ino == BFS_ROOT_INO) - di->i_vtype = cpu_to_le32(BFS_VDIR); - else - di->i_vtype = cpu_to_le32(BFS_VREG); - - di->i_ino = cpu_to_le16(ino); - di->i_mode = cpu_to_le32(inode->i_mode); - di->i_uid = cpu_to_le32(i_uid_read(inode)); - di->i_gid = cpu_to_le32(i_gid_read(inode)); - di->i_nlink = cpu_to_le32(inode->i_nlink); - di->i_atime = cpu_to_le32(inode_get_atime_sec(inode)); - di->i_mtime = cpu_to_le32(inode_get_mtime_sec(inode)); - di->i_ctime = cpu_to_le32(inode_get_ctime_sec(inode)); - i_sblock = BFS_I(inode)->i_sblock; - di->i_sblock = cpu_to_le32(i_sblock); - di->i_eblock = cpu_to_le32(BFS_I(inode)->i_eblock); - di->i_eoffset = cpu_to_le32(i_sblock * BFS_BSIZE + inode->i_size - 1); - - mark_buffer_dirty(bh); - brelse(bh); - mutex_unlock(&info->bfs_lock); - set_inode_metadata_writeback(inode); - return 0; -} - -static int bfs_sync_inode_metadata(struct inode *inode, - struct writeback_control *wbc) -{ - int err = 0; - struct bfs_inode *di; - struct buffer_head *bh; - - di = find_inode(inode->i_sb, (u16)inode->i_ino, &bh); - if (IS_ERR(di)) - return PTR_ERR(di); - - sync_dirty_buffer(bh); - if (buffer_write_io_error(bh)) { - err = -EIO; - goto out; - } - err = mmb_sync(&BFS_I(inode)->i_metadata_bhs); -out: - brelse(bh); - return err; -} - -static void bfs_evict_inode(struct inode *inode) -{ - unsigned long ino = inode->i_ino; - struct bfs_inode *di; - struct buffer_head *bh; - struct super_block *s = inode->i_sb; - struct bfs_sb_info *info = BFS_SB(s); - struct bfs_inode_info *bi = BFS_I(inode); - - dprintf("ino=%08lx\n", ino); - - truncate_inode_pages_final(&inode->i_data); - if (inode->i_nlink) - mmb_sync(&BFS_I(inode)->i_metadata_bhs); - mmb_invalidate(&BFS_I(inode)->i_metadata_bhs); - clear_inode(inode); - - if (inode->i_nlink) - return; - - di = find_inode(s, inode->i_ino, &bh); - if (IS_ERR(di)) - return; - - mutex_lock(&info->bfs_lock); - /* clear on-disk inode */ - memset(di, 0, sizeof(struct bfs_inode)); - mark_buffer_dirty(bh); - brelse(bh); - - if (bi->i_dsk_ino) { - if (bi->i_sblock) - info->si_freeb += bi->i_eblock + 1 - bi->i_sblock; - info->si_freei++; - clear_bit(ino, info->si_imap); - bfs_dump_imap("evict_inode", s); - } - - /* - * If this was the last file, make the previous block - * "last block of the last file" even if there is no - * real file there, saves us 1 gap. - */ - if (info->si_lf_eblk == bi->i_eblock) - info->si_lf_eblk = bi->i_sblock - 1; - mutex_unlock(&info->bfs_lock); -} - -static void bfs_put_super(struct super_block *s) -{ - struct bfs_sb_info *info = BFS_SB(s); - - if (!info) - return; - - mutex_destroy(&info->bfs_lock); - kfree(info); - s->s_fs_info = NULL; -} - -static int bfs_statfs(struct dentry *dentry, struct kstatfs *buf) -{ - struct super_block *s = dentry->d_sb; - struct bfs_sb_info *info = BFS_SB(s); - u64 id = huge_encode_dev(s->s_bdev->bd_dev); - buf->f_type = BFS_MAGIC; - buf->f_bsize = s->s_blocksize; - buf->f_blocks = info->si_blocks; - buf->f_bfree = buf->f_bavail = info->si_freeb; - buf->f_files = info->si_lasti + 1 - BFS_ROOT_INO; - buf->f_ffree = info->si_freei; - buf->f_fsid = u64_to_fsid(id); - buf->f_namelen = BFS_NAMELEN; - return 0; -} - -static struct kmem_cache *bfs_inode_cachep; - -static struct inode *bfs_alloc_inode(struct super_block *sb) -{ - struct bfs_inode_info *bi; - bi = alloc_inode_sb(sb, bfs_inode_cachep, GFP_KERNEL); - if (!bi) - return NULL; - mmb_init(&bi->i_metadata_bhs, &bi->vfs_inode.i_data); - - return &bi->vfs_inode; -} - -static void bfs_free_inode(struct inode *inode) -{ - kmem_cache_free(bfs_inode_cachep, BFS_I(inode)); -} - -static void init_once(void *foo) -{ - struct bfs_inode_info *bi = foo; - - inode_init_once(&bi->vfs_inode); -} - -static int __init init_inodecache(void) -{ - bfs_inode_cachep = kmem_cache_create("bfs_inode_cache", - sizeof(struct bfs_inode_info), - 0, (SLAB_RECLAIM_ACCOUNT| - SLAB_ACCOUNT), - init_once); - if (bfs_inode_cachep == NULL) - return -ENOMEM; - return 0; -} - -static void destroy_inodecache(void) -{ - /* - * Make sure all delayed rcu free inodes are flushed before we - * destroy cache. - */ - rcu_barrier(); - kmem_cache_destroy(bfs_inode_cachep); -} - -static const struct super_operations bfs_sops = { - .alloc_inode = bfs_alloc_inode, - .free_inode = bfs_free_inode, - .write_inode = bfs_write_inode, - .sync_inode_metadata = bfs_sync_inode_metadata, - .evict_inode = bfs_evict_inode, - .put_super = bfs_put_super, - .statfs = bfs_statfs, -}; - -void bfs_dump_imap(const char *prefix, struct super_block *s) -{ -#ifdef DEBUG - int i; - char *tmpbuf = kzalloc(PAGE_SIZE, GFP_KERNEL); - - if (!tmpbuf) - return; - for (i = BFS_SB(s)->si_lasti; i >= 0; i--) { - if (i > PAGE_SIZE - 100) break; - if (test_bit(i, BFS_SB(s)->si_imap)) - strcat(tmpbuf, "1"); - else - strcat(tmpbuf, "0"); - } - printf("%s: lasti=%08lx <%s>\n", prefix, BFS_SB(s)->si_lasti, tmpbuf); - kfree(tmpbuf); -#endif -} - -static int bfs_fill_super(struct super_block *s, struct fs_context *fc) -{ - struct buffer_head *bh, *sbh; - struct bfs_super_block *bfs_sb; - struct inode *inode; - unsigned i; - struct bfs_sb_info *info; - int ret = -EINVAL; - unsigned long i_sblock, i_eblock, i_eoff, s_size; - int silent = fc->sb_flags & SB_SILENT; - - info = kzalloc_obj(*info); - if (!info) - return -ENOMEM; - mutex_init(&info->bfs_lock); - s->s_fs_info = info; - s->s_time_min = 0; - s->s_time_max = U32_MAX; - - if (!sb_set_blocksize(s, BFS_BSIZE)) - goto out; - - sbh = sb_bread(s, 0); - if (!sbh) - goto out; - bfs_sb = (struct bfs_super_block *)sbh->b_data; - if (le32_to_cpu(bfs_sb->s_magic) != BFS_MAGIC) { - if (!silent) - printf("No BFS filesystem on %s (magic=%08x)\n", s->s_id, le32_to_cpu(bfs_sb->s_magic)); - goto out1; - } - if (BFS_UNCLEAN(bfs_sb, s) && !silent) - printf("%s is unclean, continuing\n", s->s_id); - - s->s_magic = BFS_MAGIC; - - if (le32_to_cpu(bfs_sb->s_start) > le32_to_cpu(bfs_sb->s_end) || - le32_to_cpu(bfs_sb->s_start) < sizeof(struct bfs_super_block) + sizeof(struct bfs_dirent)) { - printf("Superblock is corrupted on %s\n", s->s_id); - goto out1; - } - - info->si_lasti = (le32_to_cpu(bfs_sb->s_start) - BFS_BSIZE) / sizeof(struct bfs_inode) + BFS_ROOT_INO - 1; - if (info->si_lasti == BFS_MAX_LASTI) - printf("NOTE: filesystem %s was created with 512 inodes, the real maximum is 511, mounting anyway\n", s->s_id); - else if (info->si_lasti > BFS_MAX_LASTI) { - printf("Impossible last inode number %lu > %d on %s\n", info->si_lasti, BFS_MAX_LASTI, s->s_id); - goto out1; - } - for (i = 0; i < BFS_ROOT_INO; i++) - set_bit(i, info->si_imap); - - s->s_op = &bfs_sops; - inode = bfs_iget(s, BFS_ROOT_INO); - if (IS_ERR(inode)) { - ret = PTR_ERR(inode); - goto out1; - } - s->s_root = d_make_root(inode); - if (!s->s_root) { - ret = -ENOMEM; - goto out1; - } - - info->si_blocks = (le32_to_cpu(bfs_sb->s_end) + 1) >> BFS_BSIZE_BITS; - info->si_freeb = (le32_to_cpu(bfs_sb->s_end) + 1 - le32_to_cpu(bfs_sb->s_start)) >> BFS_BSIZE_BITS; - info->si_freei = 0; - info->si_lf_eblk = 0; - - /* can we read the last block? */ - bh = sb_bread(s, info->si_blocks - 1); - if (!bh) { - printf("Last block not available on %s: %lu\n", s->s_id, info->si_blocks - 1); - ret = -EIO; - goto out2; - } - brelse(bh); - - bh = NULL; - for (i = BFS_ROOT_INO; i <= info->si_lasti; i++) { - struct bfs_inode *di; - int block = (i - BFS_ROOT_INO) / BFS_INODES_PER_BLOCK + 1; - int off = (i - BFS_ROOT_INO) % BFS_INODES_PER_BLOCK; - unsigned long eblock; - - if (!off) { - brelse(bh); - bh = sb_bread(s, block); - } - - if (!bh) - continue; - - di = (struct bfs_inode *)bh->b_data + off; - - /* test if filesystem is not corrupted */ - - i_eoff = le32_to_cpu(di->i_eoffset); - i_sblock = le32_to_cpu(di->i_sblock); - i_eblock = le32_to_cpu(di->i_eblock); - s_size = le32_to_cpu(bfs_sb->s_end); - - if (i_sblock > info->si_blocks || - i_eblock > info->si_blocks || - i_sblock > i_eblock || - (i_eoff != le32_to_cpu(-1) && i_eoff > s_size) || - i_sblock * BFS_BSIZE > i_eoff) { - - printf("Inode 0x%08x corrupted on %s\n", i, s->s_id); - - brelse(bh); - ret = -EIO; - goto out2; - } - - if (!di->i_ino) { - info->si_freei++; - continue; - } - set_bit(i, info->si_imap); - info->si_freeb -= BFS_FILEBLOCKS(di); - - eblock = le32_to_cpu(di->i_eblock); - if (eblock > info->si_lf_eblk) - info->si_lf_eblk = eblock; - } - brelse(bh); - brelse(sbh); - bfs_dump_imap("fill_super", s); - return 0; - -out2: - dput(s->s_root); - s->s_root = NULL; -out1: - brelse(sbh); -out: - mutex_destroy(&info->bfs_lock); - kfree(info); - s->s_fs_info = NULL; - return ret; -} - -static int bfs_get_tree(struct fs_context *fc) -{ - return get_tree_bdev(fc, bfs_fill_super); -} - -static const struct fs_context_operations bfs_context_ops = { - .get_tree = bfs_get_tree, -}; - -static int bfs_init_fs_context(struct fs_context *fc) -{ - fc->ops = &bfs_context_ops; - - return 0; -} - -static struct file_system_type bfs_fs_type = { - .owner = THIS_MODULE, - .name = "bfs", - .init_fs_context = bfs_init_fs_context, - .kill_sb = kill_block_super, - .fs_flags = FS_REQUIRES_DEV, -}; -MODULE_ALIAS_FS("bfs"); - -static int __init init_bfs_fs(void) -{ - int err = init_inodecache(); - if (err) - goto out1; - err = register_filesystem(&bfs_fs_type); - if (err) - goto out; - return 0; -out: - destroy_inodecache(); -out1: - return err; -} - -static void __exit exit_bfs_fs(void) -{ - unregister_filesystem(&bfs_fs_type); - destroy_inodecache(); -} - -module_init(init_bfs_fs) -module_exit(exit_bfs_fs) diff --git a/fs/binfmt_elf.c b/fs/binfmt_elf.c index 06d0df105382..bf7f8f47548d 100644 --- a/fs/binfmt_elf.c +++ b/fs/binfmt_elf.c @@ -74,7 +74,7 @@ static int load_elf_binary(struct linux_binprm *bprm); * don't even try. */ #ifdef CONFIG_ELF_CORE -static int elf_core_dump(struct coredump_params *cprm); +static bool elf_core_dump(struct coredump_params *cprm); #else #define elf_core_dump NULL #endif @@ -1875,7 +1875,7 @@ static int fill_note_info(struct elfhdr *elf, int phdrs, return 0; info->thread->task = dump_task; - for (ct = dump_task->signal->core_state->dumper.next; ct; ct = ct->next) { + for (ct = dump_task->signal->core_state->tasks; ct; ct = ct->next) { t = kzalloc_flex(*t, notes, info->thread_notes); if (unlikely(!t)) return 0; @@ -1987,9 +1987,9 @@ static void fill_extnum_info(struct elfhdr *elf, struct elf_shdr *shdr4extnum, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_core_dump(struct coredump_params *cprm) +static bool elf_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs, i; struct elfhdr elf; loff_t offset = 0, dataoff; @@ -2020,7 +2020,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!fill_note_info(&elf, e_phnum, &info, cprm)) goto end_coredump; - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; offset += sizeof(elf); /* ELF header */ offset += segs * sizeof(struct elf_phdr); /* Program headers */ @@ -2029,7 +2029,7 @@ static int elf_core_dump(struct coredump_params *cprm) { size_t sz = info.size; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ sz += elf_coredump_extra_notes_size(); phdr4note = kmalloc_obj(*phdr4note); @@ -2093,7 +2093,7 @@ static int elf_core_dump(struct coredump_params *cprm) if (!write_note_info(&info, cprm)) goto end_coredump; - /* For cell spufs and x86 xstate */ + /* For x86 xstate */ if (elf_coredump_extra_notes_write(cprm)) goto end_coredump; @@ -2115,11 +2115,13 @@ static int elf_core_dump(struct coredump_params *cprm) goto end_coredump; } + ret = true; + end_coredump: free_note_info(&info); kfree(shdr4extnum); kfree(phdr4note); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/binfmt_elf_fdpic.c b/fs/binfmt_elf_fdpic.c index 068c46875c74..d3872169f55e 100644 --- a/fs/binfmt_elf_fdpic.c +++ b/fs/binfmt_elf_fdpic.c @@ -75,7 +75,7 @@ static int elf_fdpic_map_file_by_direct_mmap(struct elf_fdpic_params *, struct file *, struct mm_struct *); #ifdef CONFIG_ELF_CORE -static int elf_fdpic_core_dump(struct coredump_params *cprm); +static bool elf_fdpic_core_dump(struct coredump_params *cprm); #endif static struct linux_binfmt elf_fdpic_format = { @@ -1477,9 +1477,9 @@ static bool elf_fdpic_dump_segments(struct coredump_params *cprm, * and then they are actually written out. If we run out of core limit * we just truncate. */ -static int elf_fdpic_core_dump(struct coredump_params *cprm) +static bool elf_fdpic_core_dump(struct coredump_params *cprm) { - int has_dumped = 0; + bool ret = false; int segs; int i; struct elfhdr *elf = NULL; @@ -1504,7 +1504,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) if (!psinfo) goto end_coredump; - for (ct = current->signal->core_state->dumper.next; + for (ct = current->signal->core_state->tasks; ct; ct = ct->next) { tmp = elf_dump_thread_status(cprm->siginfo->si_signo, ct->task, &thread_status_size); @@ -1536,7 +1536,7 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) /* Set up header */ fill_elf_fdpic_header(elf, e_phnum); - has_dumped = 1; + cprm->state |= COREDUMP_STATE_STARTED; /* * Set up the notes in similar form to SVR4 core dumps made * with info from their /proc. @@ -1656,6 +1656,8 @@ static int elf_fdpic_core_dump(struct coredump_params *cprm) cprm->file->f_pos, offset); } + ret = true; + end_coredump: while (thread_list) { tmp = thread_list; @@ -1666,7 +1668,7 @@ end_coredump: kfree(elf); kfree(psinfo); kfree(shdr4extnum); - return has_dumped; + return ret; } #endif /* CONFIG_ELF_CORE */ diff --git a/fs/binfmt_misc.c b/fs/binfmt_misc.c index 620da85948b4..d945b4f6e158 100644 --- a/fs/binfmt_misc.c +++ b/fs/binfmt_misc.c @@ -100,6 +100,13 @@ static const struct binfmt_misc_flag *misc_flag_by_char(const char c) return NULL; } +static bool misc_valid_delim(const char c) +{ + if (!isascii(c) || !ispunct(c)) + return false; + return c != '\\'; +} + struct binfmt_misc_entry { struct hlist_node node; unsigned long flags; /* type, status, etc. */ @@ -871,10 +878,9 @@ static struct binfmt_misc_entry *create_entry(const char __user *buffer, del = *p++; /* delimiter */ - pr_debug("register: delim: %#x {%c}\n", del, del); + pr_debug("register: delim: %#x\n", del); - /* A flag-char delimiter runs the flag scan off the buffer. */ - if (misc_flag_by_char(del)) + if (!misc_valid_delim(del)) return ERR_PTR(-EINVAL); /* Pad the buffer with the delim to simplify parsing below. */ diff --git a/fs/bpf_fs_kfuncs.c b/fs/bpf_fs_kfuncs.c index 357a379ef92a..abdfbd83dc57 100644 --- a/fs/bpf_fs_kfuncs.c +++ b/fs/bpf_fs_kfuncs.c @@ -237,7 +237,7 @@ int bpf_set_dentry_xattr_locked(struct dentry *dentry, const char *name__str, * @dentry: dentry to get xattr from * @name__str: name of the xattr * - * Rmove xattr *name__str* of *dentry*. + * Remove xattr *name__str* of *dentry*. * * For security reasons, only *name__str* with prefix "security.bpf." * is allowed. @@ -305,7 +305,7 @@ __bpf_kfunc int bpf_set_dentry_xattr(struct dentry *dentry, const char *name__st * @dentry: dentry to get xattr from * @name__str: name of the xattr * - * Rmove xattr *name__str* of *dentry*. + * Remove xattr *name__str* of *dentry*. * * For security reasons, only *name__str* with prefix "security.bpf." * is allowed. diff --git a/fs/btrfs/acl.c b/fs/btrfs/acl.c index 662cdd1cbdef..10a0d733bfd1 100644 --- a/fs/btrfs/acl.c +++ b/fs/btrfs/acl.c @@ -101,7 +101,7 @@ int __btrfs_set_acl(struct btrfs_trans_handle *trans, struct inode *inode, return 0; } -int btrfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int btrfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int ret; diff --git a/fs/btrfs/acl.h b/fs/btrfs/acl.h index 0458cd51ed48..6eae2db3654d 100644 --- a/fs/btrfs/acl.h +++ b/fs/btrfs/acl.h @@ -15,7 +15,7 @@ struct mnt_idmap; struct dentry; struct posix_acl *btrfs_get_acl(struct inode *inode, int type, bool rcu); -int btrfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int btrfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int __btrfs_set_acl(struct btrfs_trans_handle *trans, struct inode *inode, struct posix_acl *acl, int type); diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 114a5c38afd3..b673851d8d2e 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -555,7 +555,7 @@ int btrfs_new_inode_prepare(struct btrfs_new_inode_args *args, int btrfs_create_new_inode(struct btrfs_trans_handle *trans, struct btrfs_new_inode_args *args); void btrfs_new_inode_args_destroy(struct btrfs_new_inode_args *args); -struct inode *btrfs_new_subvol_inode(struct mnt_idmap *idmap, +struct inode *btrfs_new_subvol_inode(const struct mnt_idmap *idmap, struct inode *dir); void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *state, u32 bits); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index f4b68205f621..1d79f5263de3 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -5428,7 +5428,7 @@ static int btrfs_setsize(struct inode *inode, struct iattr *attr) return ret; } -static int btrfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int btrfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -6976,7 +6976,7 @@ out_inode: return ret; } -static int btrfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -6990,7 +6990,7 @@ static int btrfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return btrfs_create_common(dir, dentry, inode); } -static int btrfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -7087,7 +7087,7 @@ fail: return ret; } -static struct dentry *btrfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *btrfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -7984,7 +7984,7 @@ out: return ret; } -struct inode *btrfs_new_subvol_inode(struct mnt_idmap *idmap, +struct inode *btrfs_new_subvol_inode(const struct mnt_idmap *idmap, struct inode *dir) { struct inode *inode; @@ -8178,7 +8178,7 @@ int __init btrfs_init_cachep(void) return 0; } -static int btrfs_getattr(struct mnt_idmap *idmap, +static int btrfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -8494,7 +8494,7 @@ out_notrans: return ret; } -static struct inode *new_whiteout_inode(struct mnt_idmap *idmap, +static struct inode *new_whiteout_inode(const struct mnt_idmap *idmap, struct inode *dir) { struct inode *inode; @@ -8509,7 +8509,7 @@ static struct inode *new_whiteout_inode(struct mnt_idmap *idmap, return inode; } -static int btrfs_rename(struct mnt_idmap *idmap, +static int btrfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -8809,7 +8809,7 @@ out_fscrypt_names: return ret; } -static int btrfs_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +static int btrfs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -8997,7 +8997,7 @@ out: return ret; } -static int btrfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct btrfs_fs_info *fs_info = inode_to_fs_info(dir); @@ -9313,7 +9313,7 @@ next: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to * elide calls here. */ -static int btrfs_permission(struct mnt_idmap *idmap, +static int btrfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct btrfs_root *root = BTRFS_I(inode)->root; @@ -9329,7 +9329,7 @@ static int btrfs_permission(struct mnt_idmap *idmap, return generic_permission(idmap, inode, mask); } -static int btrfs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int btrfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct btrfs_fs_info *fs_info = inode_to_fs_info(dir); diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 52aab510aea0..f3e2afe221be 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -278,7 +278,7 @@ int btrfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int btrfs_fileattr_set(struct mnt_idmap *idmap, +int btrfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct btrfs_inode *inode = BTRFS_I(d_inode(dentry)); @@ -549,7 +549,7 @@ static unsigned int create_subvol_num_items(const struct btrfs_qgroup_inherit *i return num_items; } -static noinline int create_subvol(struct mnt_idmap *idmap, +static noinline int create_subvol(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct btrfs_qgroup_inherit *inherit) { @@ -879,7 +879,7 @@ free_pending: * inside this filesystem so it's quite a bit simpler. */ static noinline int btrfs_mksubvol(struct dentry *parent, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct qstr *qname, struct btrfs_root *snap_src, bool readonly, struct btrfs_qgroup_inherit *inherit) @@ -926,7 +926,7 @@ out_dput: } static noinline int btrfs_mksnapshot(struct dentry *parent, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct qstr *qname, struct btrfs_root *root, bool readonly, @@ -1164,7 +1164,7 @@ static noinline int __btrfs_ioctl_snap_create(struct file *file, { int ret; struct qstr qname = QSTR(name); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); if (!S_ISDIR(file_inode(file)->i_mode)) return -ENOTDIR; @@ -1741,7 +1741,7 @@ static noinline int btrfs_search_path_in_tree(struct btrfs_root *root, u64 dirid return 0; } -static int btrfs_search_path_in_tree_user(struct mnt_idmap *idmap, +static int btrfs_search_path_in_tree_user(const struct mnt_idmap *idmap, struct inode *inode, struct btrfs_ioctl_ino_lookup_user_args *args) { @@ -2241,7 +2241,7 @@ static noinline int btrfs_ioctl_snap_destroy(struct file *file, struct btrfs_root *dest = NULL; struct btrfs_ioctl_vol_args AUTO_KFREE(vol_args); struct btrfs_ioctl_vol_args_v2 AUTO_KFREE(vol_args2); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); char *subvol_name, *subvol_name_ptr = NULL; int ret = 0; bool destroy_parent = false; @@ -3901,7 +3901,7 @@ static long btrfs_ioctl_quota_rescan_wait(struct btrfs_fs_info *fs_info) } static long _btrfs_ioctl_set_received_subvol(struct file *file, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct btrfs_ioctl_received_subvol_args *sa) { struct inode *inode = file_inode(file); diff --git a/fs/btrfs/ioctl.h b/fs/btrfs/ioctl.h index ccf6bed9cc24..55f86aeb3500 100644 --- a/fs/btrfs/ioctl.h +++ b/fs/btrfs/ioctl.h @@ -17,7 +17,7 @@ struct btrfs_ioctl_balance_args; long btrfs_ioctl(struct file *file, unsigned int cmd, unsigned long arg); long btrfs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); int btrfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int btrfs_fileattr_set(struct mnt_idmap *idmap, +int btrfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int btrfs_ioctl_get_supported_features(void __user *arg); void btrfs_sync_inode_flags_to_i_flags(struct btrfs_inode *inode); diff --git a/fs/btrfs/xattr.c b/fs/btrfs/xattr.c index ab55d10bd71f..a06420b9c662 100644 --- a/fs/btrfs/xattr.c +++ b/fs/btrfs/xattr.c @@ -353,7 +353,7 @@ static int btrfs_xattr_handler_get(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) @@ -395,7 +395,7 @@ static int btrfs_xattr_handler_get_security(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set_security(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, @@ -413,7 +413,7 @@ static int btrfs_xattr_handler_set_security(const struct xattr_handler *handler, } static int btrfs_xattr_handler_set_prop(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/buffer.c b/fs/buffer.c index ed966fa73b1b..1dc933ba6925 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -203,11 +203,10 @@ void bh_end_write(struct bio *bio) bool success = bio_endio_bh(bio, &bh); if (success) { - set_buffer_uptodate(bh); + clear_buffer_write_io_error(bh); } else { buffer_io_error(bh, ", lost sync page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } unlock_buffer(bh); } @@ -265,7 +264,7 @@ __find_get_block_slow(struct block_device *bdev, sector_t block, bool atomic) bh = bh->b_this_page; } while (bh != head); - /* we might be here because some of the buffers on this page are + /* we might be here because some of the buffers on this folio are * not mapped. This is due to various races between * file io on the block device and getblk. It gets dealt with * elsewhere, don't buffer_error if we had some unmapped buffers @@ -311,7 +310,7 @@ static void end_buffer_async_read(struct buffer_head *bh, int uptodate) /* * Be _very_ careful from here on. Bad things can happen if * two buffer heads end IO at almost the same time and both - * decide that the page is now completely done. + * decide that the folio is now completely done. */ first = folio_buffers(folio); spin_lock_irqsave(&first->b_uptodate_lock, flags); @@ -408,11 +407,10 @@ void bh_end_async_write(struct bio *bio) folio = bh->b_folio; if (success) { - set_buffer_uptodate(bh); + clear_buffer_write_io_error(bh); } else { buffer_io_error(bh, ", lost async page write"); mark_buffer_write_io_error(bh); - clear_buffer_uptodate(bh); } first = folio_buffers(folio); @@ -520,8 +518,8 @@ EXPORT_SYMBOL_GPL(mmb_has_buffers); * * Do this in two main stages: first we copy dirty buffers to a * temporary inode list, queueing the writes as we go. Then we clean - * up, waiting for those writes to complete. mark_buffer_dirty_inode() - * doesn't touch b_assoc_buffers list if b_mmb is not NULL so we are sure the + * up, waiting for those writes to complete. mmb_mark_buffer_dirty() + * doesn't touch b_assoc_buffers list if b_mmb is set so we are sure the * buffer stays on our list until IO completes (at which point it can be * reaped). */ @@ -542,7 +540,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) bh = BH_ENTRY(mmb->list.next); WARN_ON_ONCE(bh->b_mmb != mmb); __remove_assoc_queue(mmb, bh); - /* Avoid race with mark_buffer_dirty_inode() which does + /* Avoid race with mmb_mark_buffer_dirty() which does * a lockless check and we rely on seeing the dirty bit */ smp_mb(); if (buffer_dirty(bh) || buffer_locked(bh)) { @@ -580,7 +578,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) bh = BH_ENTRY(tmp.prev); get_bh(bh); __remove_assoc_queue(mmb, bh); - /* Avoid race with mark_buffer_dirty_inode() which does + /* Avoid race with mmb_mark_buffer_dirty() which does * a lockless check and we rely on seeing the dirty bit */ smp_mb(); if (buffer_dirty(bh)) { @@ -589,7 +587,7 @@ int mmb_sync(struct mapping_metadata_bhs *mmb) } spin_unlock(&mmb->lock); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) err = -EIO; brelse(bh); spin_lock(&mmb->lock); @@ -618,6 +616,14 @@ void write_boundary_block(struct block_device *bdev, } } +/** + * mmb_mark_buffer_dirty - Mark a metadata buffer dirty. + * @bh: The buffer to mark dirty. + * @mmb: The list of buffers to add the buffer to. + * + * Mark the buffer dirty and add it to the list if it is not already on + * a list. + */ void mmb_mark_buffer_dirty(struct buffer_head *bh, struct mapping_metadata_bhs *mmb) { @@ -686,7 +692,7 @@ bool block_dirty_folio(struct address_space *mapping, struct folio *folio) } while (bh != head); } /* - * Lock out page's memcg migration to keep PageDirty + * Lock out folio's memcg migration to keep folio dirty flag * synchronized with per-memcg dirty page counters. */ newly_dirty = !folio_test_set_dirty(folio); @@ -952,23 +958,23 @@ __getblk_slow(struct block_device *bdev, sector_t block, } /* - * The relationship between dirty buffers and dirty pages: + * The relationship between dirty buffers and dirty folios: * - * Whenever a page has any dirty buffers, the page's dirty bit is set, and - * the page is tagged dirty in the page cache. + * Whenever a folio has any dirty buffers, the folio's dirty flag is set, and + * the folio is tagged dirty in the page cache. * * At all times, the dirtiness of the buffers represents the dirtiness of - * subsections of the page. If the page has buffers, the page dirty bit is + * subsections of the folio. If the folio has buffers, the folio dirty flag is * merely a hint about the true dirty state. * - * When a page is set dirty in its entirety, all its buffers are marked dirty - * (if the page has buffers). + * When a folio is set dirty in its entirety, all its buffers are marked dirty + * (if the folio has buffers). * - * When a buffer is marked dirty, its page is dirtied, but the page's other + * When a buffer is marked dirty, its folio is dirtied, but the folio's other * buffers are not. * * Also. When blockdev buffers are explicitly read with bread(), they - * individually become uptodate. But their backing page remains not + * individually become uptodate. But their backing folio remains not * uptodate - even if all of its buffers are uptodate. A subsequent * block_read_full_folio() against that folio will discover all the uptodate * buffers, will set the folio uptodate and will perform no I/O. @@ -979,7 +985,7 @@ __getblk_slow(struct block_device *bdev, sector_t block, * @bh: the buffer_head to mark dirty * * mark_buffer_dirty() will set the dirty bit against the buffer, then set - * its backing page dirty, then tag the page as dirty in the page cache + * its backing folio dirty, then tag the folio as dirty in the page cache * and then attach the address_space's inode to its superblock's dirty * inode list. * @@ -1062,6 +1068,7 @@ EXPORT_SYMBOL(__brelse); void __bforget(struct buffer_head *bh) { clear_buffer_dirty(bh); + clear_buffer_write_io_error(bh); remove_assoc_queue(bh); __brelse(bh); } @@ -1070,12 +1077,16 @@ EXPORT_SYMBOL(__bforget); static void buffer_set_crypto_ctx(struct bio *bio, const struct buffer_head *bh, gfp_t gfp_mask) { - const struct address_space *mapping = folio_mapping(bh->b_folio); + const struct address_space *mapping; /* * The ext4 journal (jbd2) can submit a buffer_head it directly created - * for a non-pagecache page. fscrypt doesn't care about these. + * for memory that is not in the page cache at all. fscrypt doesn't + * care about these. */ + if (!bh->b_folio) + return; + mapping = bh->b_folio->mapping; if (!mapping) return; fscrypt_set_bio_crypt_ctx(bio, mapping->host, @@ -1086,7 +1097,6 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, enum rw_hint write_hint, struct writeback_control *wbc, bio_end_io_t end_bio) { - const enum req_op op = opf & REQ_OP_MASK; struct bio *bio; BUG_ON(!buffer_locked(bh)); @@ -1094,11 +1104,7 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, BUG_ON(buffer_delay(bh)); BUG_ON(buffer_unwritten(bh)); - /* - * Only clear out a write error when rewriting - */ - if (test_set_buffer_req(bh) && (op == REQ_OP_WRITE)) - clear_buffer_write_io_error(bh); + set_buffer_req(bh); if (buffer_meta(bh)) opf |= REQ_META; @@ -1107,7 +1113,8 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, bio = bio_alloc(bh->b_bdev, 1, opf, GFP_NOIO); - if (folio_test_dropbehind(bh->b_folio) && op_is_write(opf)) + if (bh->b_folio && folio_test_dropbehind(bh->b_folio) && + op_is_write(opf)) bio_set_flag(bio, BIO_COMPLETE_IN_TASK); if (IS_ENABLED(CONFIG_FS_ENCRYPTION)) @@ -1116,7 +1123,11 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, bio->bi_iter.bi_sector = bh->b_blocknr * (bh->b_size >> 9); bio->bi_write_hint = write_hint; - bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, bh_offset(bh)); + if (bh->b_folio) + bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, + bh_offset(bh)); + else + bio_add_virt_nofail(bio, bh->b_data, bh->b_size); bio->bi_end_io = end_bio; bio->bi_private = bh; @@ -1126,7 +1137,8 @@ static void __bh_submit(struct buffer_head *bh, blk_opf_t opf, if (wbc) { wbc_init_bio(wbc, bio); - wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); + if (bh->b_folio) + wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size); } blk_crypto_submit_bio(bio); @@ -1216,7 +1228,7 @@ static void bh_lru_install(struct buffer_head *bh) /* * the refcount of buffer_head in bh_lru prevents dropping the - * attached page(i.e., try_to_free_buffers) so it could cause + * attached folio (i.e., try_to_free_buffers) so it could cause * failing page migration. * Skip putting upcoming bh into bh_lru until migration is done. */ @@ -1280,7 +1292,7 @@ lookup_bh_lru(struct block_device *bdev, sector_t block, unsigned size) * Perform a pagecache lookup for the matching buffer. If it's there, refresh * it in the LRU and mark it as accessed. If it is not present then return * NULL. Atomic context callers may also return NULL if the buffer is being - * migrated; similarly the page is not marked accessed either. + * migrated; similarly the folio is not marked accessed either. */ static struct buffer_head * find_get_block_common(struct block_device *bdev, sector_t block, @@ -1289,7 +1301,7 @@ find_get_block_common(struct block_device *bdev, sector_t block, struct buffer_head *bh = lookup_bh_lru(bdev, block, size); if (bh == NULL) { - /* __find_get_block_slow will mark the page accessed */ + /* __find_get_block_slow will mark the folio accessed */ bh = __find_get_block_slow(bdev, block, atomic); if (bh) bh_lru_install(bh); @@ -1475,15 +1487,14 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio, } EXPORT_SYMBOL(folio_set_bh); -/* - * Called when truncating a buffer on a page completely. - */ - /* Bits that are cleared during an invalidate */ #define BUFFER_FLAGS_DISCARD \ (1 << BH_Mapped | 1 << BH_New | 1 << BH_Req | \ - 1 << BH_Delay | 1 << BH_Unwritten) + 1 << BH_Delay | 1 << BH_Unwritten | 1 << BH_Write_EIO) +/* + * Called when truncating a buffer on a folio completely. + */ static void discard_buffer(struct buffer_head * bh) { unsigned long b_state; @@ -1611,9 +1622,7 @@ EXPORT_SYMBOL(create_empty_buffers); * moment when something will explicitly mark the buffer dirty (hopefully that * will not happen until we will free that block ;-) We don't even need to mark * it not-uptodate - nobody can expect anything from a newly allocated buffer - * anyway. We used to use unmap_buffer() for such invalidation, but that was - * wrong. We definitely don't want to mark the alias unmapped, for example - it - * would confuse anyone who might pick it with bread() afterwards... + * anyway. * * Also.. Note that bforget() doesn't lock the buffer. So there can be * writeout I/O going on against recently-freed buffers. We don't wait on that @@ -1649,7 +1658,7 @@ void clean_bdev_aliases(struct block_device *bdev, sector_t block, sector_t len) /* Recheck when the folio is locked which pins bhs */ head = folio_buffers(folio); if (!head) - goto unlock_page; + goto unlock_folio; bh = head; do { if (!buffer_mapped(bh) || (bh->b_blocknr < block)) @@ -1662,7 +1671,7 @@ void clean_bdev_aliases(struct block_device *bdev, sector_t block, sector_t len) next: bh = bh->b_this_page; } while (bh != head); -unlock_page: +unlock_folio: folio_unlock(folio); } folio_batch_release(&fbatch); @@ -1710,7 +1719,7 @@ static struct buffer_head *folio_create_buffers(struct folio *folio, * * If block_write_full_folio() is called for regular writeback * (wbc->sync_mode == WB_SYNC_NONE) then it will redirty a folio which - * has a locked buffer. This only can happen if someone has written + * has a locked buffer. This can only happen if someone has written * the buffer directly, with bh_submit(). At the address_space level * the folio writeback flag prevents this contention from occurring. * @@ -2213,9 +2222,9 @@ int generic_write_end(const struct kiocb *iocb, struct address_space *mapping, if (old_size < pos) pagecache_isize_extended(inode, old_size, pos); /* - * Don't mark the inode dirty under page lock. First, it unnecessarily - * makes the holding time of page lock longer. Second, it forces lock - * ordering of page lock and transaction start for journaling + * Don't mark the inode dirty under folio lock. First, it unnecessarily + * makes the holding time of folio lock longer. Second, it forces lock + * ordering of folio lock and transaction start for journaling * filesystems. */ if (i_size_changed) @@ -2341,7 +2350,7 @@ int block_read_full_folio(struct folio *folio, get_block_t *get_block) * BH_Async_Read tells end_buffer_async_read() that this * buffer is not under async I/O. * - * The folio comes unlocked when it has no locked + * The folio is unlocked when it has no locked * buffer_async buffers left. * * The folio lock prevents anyone starting new async @@ -2451,7 +2460,7 @@ static int cont_expand_zero(const struct kiocb *iocb, } } - /* page covers the boundary, find the boundary offset */ + /* folio crosses the boundary, find the boundary offset */ if (index == curidx) { zerofrom = curpos & ~PAGE_MASK; /* if we will expand the thing last block will be filled */ @@ -2509,18 +2518,18 @@ EXPORT_SYMBOL(cont_write_begin); /* * block_page_mkwrite() is not allowed to change the file size as it gets - * called from a page fault handler when a page is first dirtied. Hence we must - * be careful to check for EOF conditions here. We set the page up correctly - * for a written page which means we get ENOSPC checking when writing into + * called from a page fault handler when a folio is first dirtied. Hence we must + * be careful to check for EOF conditions here. We set the folio up correctly + * for a written folio which means we get ENOSPC checking when writing into * holes and correct delalloc and unwritten extent mapping on filesystems that * support these features. * * We are not allowed to take the i_rwsem here so we have to play games to - * protect against truncate races as the page could now be beyond EOF. Because - * truncate writes the inode size before removing pages, once we have the - * page lock we can determine safely if the page is beyond EOF. If it is not - * beyond EOF, then the page is guaranteed safe against truncation until we - * unlock the page. + * protect against truncate races as the folio could now be beyond EOF. Because + * truncate writes the inode size before removing folios, once we have the + * folio lock we can determine safely if the folio is beyond EOF. If it is not + * beyond EOF, then the folio is guaranteed safe against truncation until we + * unlock the folio. * * Direct callers of this function should protect against filesystem freezing * using sb_start_pagefault() - sb_end_pagefault() functions. @@ -2538,7 +2547,7 @@ int block_page_mkwrite(struct vm_area_struct *vma, struct vm_fault *vmf, size = i_size_read(inode); if ((folio->mapping != inode->i_mapping) || (folio_pos(folio) >= size)) { - /* We overload EFAULT to mean page got truncated */ + /* We overload EFAULT to mean folio got truncated */ ret = -EFAULT; goto out_unlock; } @@ -2710,7 +2719,7 @@ int __sync_dirty_buffer(struct buffer_head *bh, blk_opf_t op_flags) bh_submit(bh, REQ_OP_WRITE | op_flags, bh_end_write); wait_on_buffer(bh); - if (!buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) return -EIO; } else { unlock_buffer(bh); diff --git a/fs/cachefiles/Kconfig b/fs/cachefiles/Kconfig index afb25b6af5aa..c9c168c7e072 100644 --- a/fs/cachefiles/Kconfig +++ b/fs/cachefiles/Kconfig @@ -17,7 +17,7 @@ config CACHEFILES_DEBUG help This permits debugging to be dynamically enabled in the filesystem caching on files module. If this is set, the debugging output may be - enabled by setting bits in /sys/modules/cachefiles/parameter/debug or + enabled by setting bits in /sys/module/cachefiles/parameters/debug or by including a debugging specifier in /etc/cachefilesd.conf. config CACHEFILES_ERROR_INJECTION diff --git a/fs/cachefiles/interface.c b/fs/cachefiles/interface.c index 50a000310a8c..789ff6abe926 100644 --- a/fs/cachefiles/interface.c +++ b/fs/cachefiles/interface.c @@ -100,73 +100,6 @@ void cachefiles_put_object(struct cachefiles_object *object, } /* - * Adjust the size of a cache file if necessary to match the DIO size. We keep - * the EOF marker a multiple of DIO blocks so that we don't fall back to doing - * non-DIO for a partial block straddling the EOF, but we also have to be - * careful of someone expanding the file and accidentally accreting the - * padding. - */ -static int cachefiles_adjust_size(struct cachefiles_object *object) -{ - struct iattr newattrs; - struct file *file = object->file; - uint64_t ni_size; - loff_t oi_size; - int ret; - - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - _enter("{OBJ%x},[%llu]", - object->debug_id, (unsigned long long) ni_size); - - if (!file) - return -ENOBUFS; - - oi_size = i_size_read(file_inode(file)); - if (oi_size == ni_size) - return 0; - - inode_lock(file_inode(file)); - - /* if there's an extension to a partial page at the end of the backing - * file, we need to discard the partial page so that we pick up new - * data after it */ - if (oi_size & ~PAGE_MASK && ni_size > oi_size) { - _debug("discard tail %llx", oi_size); - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = oi_size & PAGE_MASK; - ret = cachefiles_inject_remove_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - if (ret < 0) - goto truncate_failed; - } - - newattrs.ia_valid = ATTR_SIZE; - newattrs.ia_size = ni_size; - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = notify_change(&nop_mnt_idmap, file->f_path.dentry, - &newattrs, NULL); - -truncate_failed: - inode_unlock(file_inode(file)); - - if (ret < 0) - trace_cachefiles_io_error(NULL, file_inode(file), ret, - cachefiles_trace_notify_change_error); - if (ret == -EIO) { - cachefiles_io_error_obj(object, "Size set failed"); - ret = -ENOBUFS; - } - - _leave(" = %d", ret); - return ret; -} - -/* * Attempt to look up the nominated node in this cache */ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) @@ -198,7 +131,6 @@ static bool cachefiles_lookup_cookie(struct fscache_cookie *cookie) spin_lock(&cache->object_list_lock); list_add(&object->cache_link, &cache->object_list); spin_unlock(&cache->object_list_lock); - cachefiles_adjust_size(object); cachefiles_end_secure(cache, saved_cred); _leave(" = t"); @@ -225,14 +157,14 @@ fail: * any unused granules. */ static bool cachefiles_shorten_object(struct cachefiles_object *object, - struct file *file, loff_t new_size) + struct file *file, uoff_t new_size) { struct cachefiles_cache *cache = object->volume->cache; struct inode *inode = file_inode(file); - loff_t i_size, dio_size; + uoff_t i_size, dio_size; int ret; - dio_size = round_up(new_size, CACHEFILES_DIO_BLOCK_SIZE); + dio_size = round_up(new_size, cache->bsize); i_size = i_size_read(inode); trace_cachefiles_trunc(object, inode, i_size, dio_size, @@ -264,6 +196,7 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, } } + object->object_size = new_size; return true; } @@ -271,29 +204,38 @@ static bool cachefiles_shorten_object(struct cachefiles_object *object, * Resize the backing object. */ static void cachefiles_resize_cookie(struct netfs_cache_resources *cres, - loff_t new_size) + uoff_t new_size) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; struct fscache_cookie *cookie = object->cookie; const struct cred *saved_cred; struct file *file = cachefiles_cres_file(cres); - loff_t old_size = cookie->object_size; + uoff_t i_size = i_size_read(file_inode(file)); - _enter("%llu->%llu", old_size, new_size); + _enter("%llu->%llu", object->object_size, new_size); - if (new_size < old_size) { + /* If the file is being shrunk, we need to downsize the backing file + * and clear the end of the final block. + */ + if (new_size < object->object_size) { + if (new_size >= i_size) + goto out; cachefiles_begin_secure(cache, &saved_cred); cachefiles_shorten_object(object, file, new_size); cachefiles_end_secure(cache, saved_cred); object->cookie->object_size = new_size; + if (new_size == 0) + object->content_info = CACHEFILES_CONTENT_NO_DATA; return; } /* The file is being expanded. We don't need to do anything - * particularly. cookie->initial_size doesn't change and so the point - * at which we have to download before doesn't change. + * particularly. The tail of the last block should have been cleared + * both when it is written and when it is shrunk. */ +out: + object->object_size = new_size; cookie->object_size = new_size; } diff --git a/fs/cachefiles/internal.h b/fs/cachefiles/internal.h index c93324e0f98c..664be64ab538 100644 --- a/fs/cachefiles/internal.h +++ b/fs/cachefiles/internal.h @@ -16,8 +16,6 @@ #include <linux/cred.h> #include <linux/security.h> -#define CACHEFILES_DIO_BLOCK_SIZE 4096 - struct cachefiles_cache; struct cachefiles_object; @@ -51,12 +49,17 @@ struct cachefiles_object { struct list_head cache_link; /* Link in cache->*_list */ struct file *file; /* The file representing this object */ char *d_name; /* Backing file name */ + unsigned long flags; +#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + uoff_t object_size; /* Size of the object stored + * (independent of cookie->object_size for + * coherency reasons) + */ + atomic64_t read_limit; /* Point beyond which uncommitted writes */ int debug_id; spinlock_t lock; refcount_t ref; - enum cachefiles_content content_info:8; /* Info about content presence */ - unsigned long flags; -#define CACHEFILES_OBJECT_USING_TMPFILE 0 /* Have an unlinked tmpfile */ + enum cachefiles_content content_info; /* Info about content presence */ }; /* @@ -203,11 +206,11 @@ extern bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state); extern int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet); extern int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv); @@ -280,6 +283,7 @@ void cachefiles_withdraw_volume(struct cachefiles_volume *volume); /* * xattr.c */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file); extern int cachefiles_set_object_xattr(struct cachefiles_object *object); extern int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file); diff --git a/fs/cachefiles/io.c b/fs/cachefiles/io.c index 9540ec25b3cb..4f547d97356e 100644 --- a/fs/cachefiles/io.c +++ b/fs/cachefiles/io.c @@ -19,7 +19,7 @@ struct cachefiles_kiocb { struct kiocb iocb; refcount_t ki_refcnt; - loff_t start; + uoff_t start; union { size_t skipped; size_t len; @@ -32,6 +32,8 @@ struct cachefiles_kiocb { u64 b_writing; }; +#define IS_ERR_VALUE_LL(x) unlikely((x) >= (unsigned long long)-MAX_ERRNO) + static inline void cachefiles_put_kiocb(struct cachefiles_kiocb *ki) { if (refcount_dec_and_test(&ki->ki_refcnt)) { @@ -73,7 +75,7 @@ static void cachefiles_read_complete(struct kiocb *iocb, long ret) * Initiate a read from the cache. */ static int cachefiles_read(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, enum netfs_read_from_hole read_hole, netfs_io_terminated_t term_func, @@ -193,60 +195,81 @@ presubmission_error: } /* - * Query the occupancy of the cache in a region, returning where the next chunk - * of data starts and how long it is. + * Query the occupancy of the cache in a region, returning the extent of the + * next two chunks of cached data and the next hole. */ static int cachefiles_query_occupancy(struct netfs_cache_resources *cres, - loff_t start, size_t len, size_t granularity, - loff_t *_data_start, size_t *_data_len) + struct fscache_occupancy *occ) { struct cachefiles_object *object; + struct inode *inode; struct file *file; - loff_t off, off2; - - *_data_start = -1; - *_data_len = 0; + uoff_t read_limit; + loff_t ret; + int i; if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) return -ENOBUFS; object = cachefiles_cres_object(cres); file = cachefiles_cres_file(cres); - granularity = max_t(size_t, object->volume->cache->bsize, granularity); + inode = file_inode(file); + occ->granularity = object->volume->cache->bsize; + /* Read read_limit before content_info. */ + read_limit = atomic64_read_acquire(&object->read_limit); + + _enter("%pD,%llu,%llx-%llx/%llx", + file, inode->i_ino, occ->query_from, occ->query_to, read_limit); + + if (read_limit == 0) + goto done; + + switch (READ_ONCE(object->content_info)) { + case CACHEFILES_CONTENT_ALL: + case CACHEFILES_CONTENT_SINGLE: + if (read_limit > occ->query_from) { + occ->cached_from[0] = 0; + occ->cached_to[0] = read_limit; + occ->cached_type[0] = FSCACHE_EXTENT_DATA; + occ->query_from = ULLONG_MAX; + } + goto done; + default: + break; + } - _enter("%pD,%llu,%llx,%zx/%llx", - file, file_inode(file)->i_ino, start, len, - i_size_read(file_inode(file))); + for (i = 0; i < ARRAY_SIZE(occ->cached_from); i++) { + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_DATA); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_type[i] = FSCACHE_EXTENT_DATA; + occ->cached_from[i] = ret; + occ->query_from = ret; + + ret = cachefiles_inject_read_error(); + if (ret == 0) + ret = vfs_llseek(file, occ->query_from, SEEK_HOLE); + if (IS_ERR_VALUE_LL(ret)) { + if (ret != -ENXIO) + return ret; + occ->query_from = ULLONG_MAX; + goto done; + } + occ->cached_to[i] = ret; + occ->query_from = ret; + if (occ->query_from >= occ->query_to) + break; + } - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off < 0 && off >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - if (round_up(off, granularity) >= start + len) - return -ENODATA; /* No data in range */ - - off2 = cachefiles_inject_read_error(); - if (off2 == 0) - off2 = vfs_llseek(file, off, SEEK_HOLE); - if (off2 == -ENXIO) - return -ENODATA; /* Beyond EOF */ - if (off2 < 0 && off2 >= (loff_t)-MAX_ERRNO) - return -ENOBUFS; /* Error. */ - - /* Round away partial blocks */ - off = round_up(off, granularity); - off2 = round_down(off2, granularity); - if (off2 <= off) - return -ENODATA; - - *_data_start = off; - if (off2 > start + len) - *_data_len = len; - else - *_data_len = off2 - off; +done: + _debug("query[0] %llx-%llx", occ->cached_from[0], occ->cached_to[0]); + _debug("query[1] %llx-%llx", occ->cached_from[1], occ->cached_to[1]); return 0; } @@ -280,7 +303,7 @@ static void cachefiles_write_complete(struct kiocb *iocb, long ret) */ int __cachefiles_write(struct cachefiles_object *object, struct file *file, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -357,7 +380,7 @@ in_progress: } static int cachefiles_write(struct netfs_cache_resources *cres, - loff_t start_pos, + uoff_t start_pos, struct iov_iter *iter, netfs_io_terminated_t term_func, void *term_func_priv) @@ -375,127 +398,12 @@ static int cachefiles_write(struct netfs_cache_resources *cres, term_func, term_func_priv); } -static inline enum netfs_io_source -cachefiles_do_prepare_read(struct netfs_cache_resources *cres, - loff_t start, size_t *_len, loff_t i_size, - unsigned long *_flags, ino_t netfs_ino) -{ - enum cachefiles_prepare_read_trace why; - struct cachefiles_object *object = NULL; - struct cachefiles_cache *cache; - struct fscache_cookie *cookie = fscache_cres_cookie(cres); - const struct cred *saved_cred; - struct file *file = cachefiles_cres_file(cres); - enum netfs_io_source ret = NETFS_DOWNLOAD_FROM_SERVER; - size_t len = *_len; - loff_t off, to; - ino_t ino = file ? file_inode(file)->i_ino : 0; - - _enter("%zx @%llx/%llx", len, start, i_size); - - if (start >= i_size) { - ret = NETFS_FILL_WITH_ZEROES; - why = cachefiles_trace_read_after_eof; - goto out_no_object; - } - - if (test_bit(FSCACHE_COOKIE_NO_DATA_TO_READ, &cookie->flags)) { - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); - why = cachefiles_trace_read_no_data; - goto out_no_object; - } - - /* The object and the file may be being created in the background. */ - if (!file) { - why = cachefiles_trace_read_no_file; - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_READ)) - goto out_no_object; - file = cachefiles_cres_file(cres); - if (!file) - goto out_no_object; - ino = file_inode(file)->i_ino; - } - - object = cachefiles_cres_object(cres); - cache = object->volume->cache; - cachefiles_begin_secure(cache, &saved_cred); - off = cachefiles_inject_read_error(); - if (off == 0) - off = vfs_llseek(file, start, SEEK_DATA); - if (off < 0 && off >= (loff_t)-MAX_ERRNO) { - if (off == (loff_t)-ENXIO) { - why = cachefiles_trace_read_seek_nxio; - goto download_and_store; - } - trace_cachefiles_io_error(object, file_inode(file), off, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (off >= start + len) { - why = cachefiles_trace_read_found_hole; - goto download_and_store; - } - - if (off > start) { - off = round_up(off, cache->bsize); - len = off - start; - *_len = len; - why = cachefiles_trace_read_found_part; - goto download_and_store; - } - - to = cachefiles_inject_read_error(); - if (to == 0) - to = vfs_llseek(file, start, SEEK_HOLE); - if (to < 0 && to >= (loff_t)-MAX_ERRNO) { - trace_cachefiles_io_error(object, file_inode(file), to, - cachefiles_trace_seek_error); - why = cachefiles_trace_read_seek_error; - goto out; - } - - if (to < start + len) { - if (start + len >= i_size) - to = round_up(to, cache->bsize); - else - to = round_down(to, cache->bsize); - len = to - start; - *_len = len; - } - - why = cachefiles_trace_read_have_data; - ret = NETFS_READ_FROM_CACHE; - goto out; - -download_and_store: - __set_bit(NETFS_SREQ_COPY_TO_CACHE, _flags); -out: - cachefiles_end_secure(cache, saved_cred); -out_no_object: - trace_cachefiles_prep_read(object, start, len, *_flags, ret, why, ino, netfs_ino); - return ret; -} - -/* - * Prepare a read operation, shortening it to a cached/uncached - * boundary as appropriate. - */ -static enum netfs_io_source cachefiles_prepare_read(struct netfs_io_subrequest *subreq, - unsigned long long i_size) -{ - return cachefiles_do_prepare_read(&subreq->rreq->cache_resources, - subreq->start, &subreq->len, i_size, - &subreq->flags, subreq->rreq->inode->i_ino); -} - /* * Prepare for a write to occur. */ int __cachefiles_prepare_write(struct cachefiles_object *object, struct file *file, - loff_t *_start, size_t *_len, size_t upper_len, + uoff_t *_start, size_t *_len, size_t upper_len, bool no_space_allocated_yet) { struct cachefiles_cache *cache = object->volume->cache; @@ -504,7 +412,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, int ret; /* Round to DIO size */ - start = round_down(*_start, PAGE_SIZE); + start = round_down(*_start, cache->bsize); if (start != *_start || *_len > upper_len) { /* Probably asked to cache a streaming write written into the * pagecache when the cookie was temporarily out of service to @@ -514,7 +422,7 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return -ENOBUFS; } - *_len = round_up(len, PAGE_SIZE); + *_len = round_up(len, cache->bsize); /* We need to work out whether there's sufficient disk space to perform * the write - but we can skip that check if we have space already @@ -540,10 +448,14 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, * space, we need to see if it's fully allocated. If it's not, we may * want to cull it. */ - if (cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_check) == 0) + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, + cachefiles_has_space_check); + if (ret == 0) return 0; /* Enough space to simply overwrite the whole block */ + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace_2); + pos = cachefiles_inject_read_error(); if (pos == 0) pos = vfs_llseek(file, start, SEEK_HOLE); @@ -572,13 +484,16 @@ int __cachefiles_prepare_write(struct cachefiles_object *object, return ret; check_space: - return cachefiles_has_space(cache, 0, *_len / PAGE_SIZE, - cachefiles_has_space_for_write); + ret = cachefiles_has_space(cache, 0, *_len / cache->bsize, + cachefiles_has_space_for_write); + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_write_nospace); + return ret; } static int cachefiles_prepare_write(struct netfs_cache_resources *cres, - loff_t *_start, size_t *_len, size_t upper_len, - loff_t i_size, bool no_space_allocated_yet) + uoff_t *_start, size_t *_len, size_t upper_len, + uoff_t i_size, bool no_space_allocated_yet) { struct cachefiles_object *object = cachefiles_cres_object(cres); struct cachefiles_cache *cache = object->volume->cache; @@ -612,10 +527,14 @@ static void cachefiles_prepare_write_subreq(struct netfs_io_subrequest *subreq) stream->sreq_max_segs = BIO_MAX_VECS; if (!cachefiles_cres_file(cres)) { - if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) + if (!fscache_wait_for_operation(cres, FSCACHE_WANT_WRITE)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_waitfail); return netfs_prepare_write_failed(subreq); - if (!cachefiles_cres_file(cres)) + } + if (!cachefiles_cres_file(cres)) { + trace_netfs_sreq(subreq, netfs_sreq_trace_cache_nofile); return netfs_prepare_write_failed(subreq); + } } } @@ -628,16 +547,16 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) struct netfs_io_stream *stream = &wreq->io_streams[subreq->stream_nr]; const struct cred *saved_cred; size_t off, pre, post, len = subreq->len; - loff_t start = subreq->start; + uoff_t start = subreq->start; int ret; _enter("W=%x[%x] %llx-%llx", wreq->debug_id, subreq->debug_index, start, start + len - 1); /* We need to start on the cache granularity boundary */ - off = start & (CACHEFILES_DIO_BLOCK_SIZE - 1); + off = start & (cache->bsize - 1); if (off) { - pre = CACHEFILES_DIO_BLOCK_SIZE - off; + pre = cache->bsize - off; if (pre >= len) { fscache_count_dio_misfit(); netfs_write_subrequest_terminated(subreq, len); @@ -651,8 +570,8 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) /* We also need to end on the cache granularity boundary */ if (start + len == wreq->i_size) { - size_t part = len % CACHEFILES_DIO_BLOCK_SIZE; - size_t need = CACHEFILES_DIO_BLOCK_SIZE - part; + size_t part = len & (cache->bsize - 1); + size_t need = cache->bsize - part; if (part && stream->submit_extendable_to >= need) { len += need; @@ -661,7 +580,7 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) } } - post = len & (CACHEFILES_DIO_BLOCK_SIZE - 1); + post = len & (cache->bsize - 1); if (post) { len -= post; if (len == 0) { @@ -689,6 +608,198 @@ static void cachefiles_issue_write(struct netfs_io_subrequest *subreq) } /* + * Collect the result of buffered writeback to the cache. This includes + * copying a read to the cache. Netfslib collates the results, which might + * occur out of order, and delivers them to the cache so that it can update its + * content record. + * + * block_type is one of: + * - NETFS_CACHE_COLLECT_WRITE_DATA for a contiguous block of data + * - NETFS_CACHE_COLLECT_WRITE_GAP if a discontiguity was skipped + * - NETFS_CACHE_COLLECT_WRITE_CANCEL for a hole due to a failed/cancelled write + * + * The writes we made are all rounded out at both sides to the nearest DIO + * block boundary, so if the final block contains the EOF in the middle of it + * (rather than at the end), padding will have been written to the file. The + * backing file's filesize will have been updated if the write extended the + * file; the filesize may still change due to outstanding subreqs. + * + * The metadata in the cache file xattr records the size of the object we have + * stored, but the cache file EOF only goes up to where we've cached data to + * and, furthermore, is rounded up to the nearest DIO block boundary. + * + * Concurrent updates should be protected against by the caller. Netfslib + * holds NETFS_ICTX_WB_LOCK as a lock on writeback requests. DIO writes + * invalidate the cookie and caching is kept disabled until all users have + * unused the cookie. + */ +static void cachefiles_collect_write(struct netfs_io_request *wreq, + uoff_t start, size_t len, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + struct cachefiles_object *object = cachefiles_cres_object(cres); + struct cachefiles_cache *cache = object->volume->cache; + struct inode *inode; + struct file *file = cachefiles_cres_file(cres); + uoff_t read_limit; + uoff_t old_size = cres->cache_i_size; + uoff_t new_size; + uoff_t data_to = object->object_size; + uoff_t end = start + len; + int ret; + + if (!file) + return; + + inode = file_inode(file); + new_size = i_size_read(inode); + + _enter("%llx,%zx,%x", start, len, cache->bsize); + + if (WARN_ON(old_size & (cache->bsize - 1)) || + WARN_ON(new_size & (cache->bsize - 1)) || + WARN_ON(start & (cache->bsize - 1)) || + WARN_ON(len & (cache->bsize - 1))) { + trace_cachefiles_io_error(object, inode, -EIO, + cachefiles_trace_alignment_error); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + /* If this is recording a gap, due to discontiguous writes or lack of + * cache space, then a hole may have been introduced into the backing + * file. Treat it as a zero-length data block. + */ + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + start = end; + len = 0; + } + + /* Zeroth case: Single monolithic files are handled specially. + */ + if (wreq->origin == NETFS_WRITEBACK_SINGLE) { + if (block_type == NETFS_CACHE_COLLECT_WRITE_GAP || + block_type == NETFS_CACHE_COLLECT_WRITE_CANCEL) { + trace_cachefiles_trunc(object, inode, data_to, 0, + cachefiles_trunc_zap); + ret = cachefiles_inject_remove_error(); + if (ret == 0) + ret = vfs_truncate(&file->f_path, 0); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_trunc_error); + cachefiles_io_error_obj(object, "truncate failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + + object->content_info = CACHEFILES_CONTENT_NO_DATA; + read_limit = 0; + } else { + object->content_info = CACHEFILES_CONTENT_SINGLE; + read_limit = len; + } + goto update_sizes_2; + } + + /* First case: The backing file was empty. */ + if (old_size == 0) { + if (start == 0) + object->content_info = CACHEFILES_CONTENT_ALL; + else + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + goto update_sizes; + } + + /* Second case: The backing file is entirely within the old object size + * and thus there can be no partial tail block to deal with in the + * cache file. + */ + if (old_size <= data_to) { + if (start > old_size) + goto discontiguous; + goto update_sizes; + } + + /* Third case: The write happened entirely within the bounds of the + * current cache file's size. + */ + if (end <= old_size) + goto update_sizes; + + /* Fourth case: The write overwrote the partial tail block and extended + * the file. We only need to update the object size because netfslib + * rounds out/pads cache writes to whole disk blocks. + */ + if (start < old_size) + goto update_sizes; + + /* Fifth case: The write started from the end of the whole tail block + * and extended the file. Just extend our notion of the filesize. + */ + if (start == old_size && old_size == data_to) + goto update_sizes; + + /* Sixth case: The write continued on from the partial tail block and + * extended the file. Need to clear the gap. + */ + if (start == old_size && old_size > data_to) + goto clear_gap; + +discontiguous: + /* Seventh case: The write was beyond the EOF on the cache file, so now + * there's a hole in the file and we can no longer say in the metadata + * that we can assume we have it all. We may also need to clear the + * end of the partial tail block. + */ + /* TODO: For the moment, we will have to use SEEK_HOLE/SEEK_DATA. */ + if (object->content_info != CACHEFILES_CONTENT_BACKFS_MAP) { + object->content_info = CACHEFILES_CONTENT_BACKFS_MAP; + trace_cachefiles_coherency(object, inode->i_ino, data_to, NULL, + CACHEFILES_CONTENT_BACKFS_MAP, + cachefiles_coherency_discontiguous); + } + +clear_gap: + /* We need to clear any partial padding that got jumped over. It + * *should* be all zeros, but shared-writable mmap exists... + */ + if (old_size > data_to) { + trace_cachefiles_trunc(object, inode, data_to, old_size, + cachefiles_trunc_clear_padding); + ret = cachefiles_inject_write_error(); + if (ret == 0) + ret = vfs_fallocate(file, FALLOC_FL_ZERO_RANGE, + data_to, old_size - data_to); + if (ret < 0) { + trace_cachefiles_io_error(object, inode, ret, + cachefiles_trace_fallocate_error); + cachefiles_io_error_obj(object, "fallocate zero pad failed %d", ret); + cachefiles_remove_object_xattr(cache, object, file->f_path.dentry); + return; + } + } + +update_sizes: + read_limit = umax(old_size, end); +update_sizes_2: + cres->cache_i_size = read_limit; + + /* We need to be careful setting the object_size: we may have written + * more to the cache than to the server (due to cache DIO rounding) and + * the i_size set on the netfs inode may include unwritten data that + * the server doesn't know about yet. + */ + object->object_size = umin(read_limit, wreq->i_size); + + /* Raise the limit at which reads can access the file. */ + /* Update read_limit after content_info */ + atomic64_set_release(&object->read_limit, read_limit); +} + +/* * Clean up an operation. */ static void cachefiles_end_operation(struct netfs_cache_resources *cres) @@ -705,10 +816,10 @@ static const struct netfs_cache_ops cachefiles_netfs_cache_ops = { .read = cachefiles_read, .write = cachefiles_write, .issue_write = cachefiles_issue_write, - .prepare_read = cachefiles_prepare_read, .prepare_write = cachefiles_prepare_write, .prepare_write_subreq = cachefiles_prepare_write_subreq, .query_occupancy = cachefiles_query_occupancy, + .collect_write = cachefiles_collect_write, }; /* @@ -718,13 +829,20 @@ bool cachefiles_begin_operation(struct netfs_cache_resources *cres, enum fscache_want_state want_state) { struct cachefiles_object *object = cachefiles_cres_object(cres); + struct file *file; + + cres->dio_size = object->volume->cache->bsize; if (!cachefiles_cres_file(cres)) { cres->ops = &cachefiles_netfs_cache_ops; + cres->object_id = object->debug_id; if (object->file) { spin_lock(&object->lock); - if (!cres->cache_priv2 && object->file) - cres->cache_priv2 = get_file(object->file); + file = object->file; + if (!cres->cache_priv2 && file) { + cres->cache_priv2 = get_file(file); + cres->cache_i_size = i_size_read(file_inode(file)); + } spin_unlock(&object->lock); } } diff --git a/fs/cachefiles/namei.c b/fs/cachefiles/namei.c index 88955249a1a6..ef656a319ede 100644 --- a/fs/cachefiles/namei.c +++ b/fs/cachefiles/namei.c @@ -117,8 +117,11 @@ retry: if (d_is_negative(subdir)) { ret = cachefiles_has_space(cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(NULL, cachefiles_trace_mkdir_nospace); goto mkdir_error; + } _debug("attempt mkdir"); @@ -414,7 +417,6 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) struct dentry *fan = volume->fanout[(u8)object->cookie->key_hash]; struct file *file; const struct path parentpath = { .mnt = cache->mnt, .dentry = fan }; - uint64_t ni_size; long ret; @@ -442,31 +444,20 @@ struct file *cachefiles_create_tmpfile(struct cachefiles_object *object) if (!cachefiles_mark_inode_in_use(object, file_inode(file))) WARN_ON(1); - ni_size = object->cookie->object_size; - ni_size = round_up(ni_size, CACHEFILES_DIO_BLOCK_SIZE); - - if (ni_size > 0) { - trace_cachefiles_trunc(object, file_inode(file), 0, ni_size, - cachefiles_trunc_expand_tmpfile); - ret = cachefiles_inject_write_error(); - if (ret == 0) - ret = vfs_truncate(&file->f_path, ni_size); - if (ret < 0) { - trace_cachefiles_vfs_error( - object, file_inode(file), ret, - cachefiles_trace_trunc_error); - goto err_unuse; - } - } - ret = -EINVAL; if (unlikely(!file->f_op->read_iter) || unlikely(!file->f_op->write_iter)) { pr_notice("Cache does not support read_iter and write_iter\n"); goto err_unuse; } + + /* Preallocate space for the xattr. */ + ret = cachefiles_preset_object_xattr(object, file); + if (ret < 0) + goto err_unuse; out: cachefiles_end_secure(cache, saved_cred); + object->content_info = CACHEFILES_CONTENT_ALL; return file; err_unuse: @@ -487,8 +478,11 @@ static bool cachefiles_create_file(struct cachefiles_object *object) ret = cachefiles_has_space(object->volume->cache, 1, 0, cachefiles_has_space_for_create); - if (ret < 0) + if (ret < 0) { + if (ret == -ENOBUFS) + trace_cachefiles_no_space(object, cachefiles_trace_create_nospace); return false; + } file = cachefiles_create_tmpfile(object); if (IS_ERR(file)) diff --git a/fs/cachefiles/xattr.c b/fs/cachefiles/xattr.c index c70bf67e52b0..8ebb713482e3 100644 --- a/fs/cachefiles/xattr.c +++ b/fs/cachefiles/xattr.c @@ -35,6 +35,57 @@ struct cachefiles_vol_xattr { } __packed; /* + * Preset the state xattr on a cache file to allocate space for it. + */ +int cachefiles_preset_object_xattr(struct cachefiles_object *object, struct file *file) +{ + struct cachefiles_xattr *buf; + struct dentry *dentry = file->f_path.dentry; + unsigned int len = object->cookie->aux_len; + int ret; + + buf = kzalloc(sizeof(struct cachefiles_xattr) + min(len, sizeof(__be64)), GFP_KERNEL); + if (!buf) + return -ENOMEM; + + buf->type = CACHEFILES_COOKIE_TYPE_DATA; + buf->content = CACHEFILES_CONTENT_DIRTY; + + ret = cachefiles_inject_write_error(); + if (ret == 0) { + ret = mnt_want_write_file(file); + if (ret == 0) { + ret = vfs_setxattr(&nop_mnt_idmap, dentry, + cachefiles_xattr_cache, buf, + sizeof(struct cachefiles_xattr) + len, 0); + mnt_drop_write_file(file); + } + } + if (ret < 0) { + trace_cachefiles_vfs_error(object, file_inode(file), ret, + cachefiles_trace_setxattr_error); + trace_cachefiles_coherency(object, file_inode(file)->i_ino, + object->object_size, + buf->data, buf->content, + cachefiles_coherency_set_fail); + switch (ret) { + case -ENOMEM: + case -ENOSPC: + break; + default: + cachefiles_io_error_obj( + object, + "Failed to set xattr with error %d", ret); + break; + } + } + + kfree(buf); + _leave(" = %d", ret); + return ret; +} + +/* * set the state xattr on a cache file */ int cachefiles_set_object_xattr(struct cachefiles_object *object) @@ -43,6 +94,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) struct dentry *dentry; struct file *file = object->file; unsigned int len = object->cookie->aux_len; + uoff_t object_size = object->cookie->object_size; int ret; if (!file) @@ -55,7 +107,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) if (!buf) return -ENOMEM; - buf->object_size = cpu_to_be64(object->cookie->object_size); + buf->object_size = cpu_to_be64(object_size); buf->zero_point = 0; buf->type = CACHEFILES_COOKIE_TYPE_DATA; buf->content = object->content_info; @@ -79,15 +131,21 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object) trace_cachefiles_vfs_error(object, file_inode(file), ret, cachefiles_trace_setxattr_error); trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_fail); - if (ret != -ENOMEM) + switch (ret) { + case -ENOMEM: + break; + case -ENOSPC: + default: cachefiles_io_error_obj( object, "Failed to set xattr with error %d", ret); + break; + } } else { trace_cachefiles_coherency(object, file_inode(file)->i_ino, - buf->data, buf->content, + object_size, buf->data, buf->content, cachefiles_coherency_set_ok); } @@ -103,10 +161,12 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file { struct cachefiles_xattr *buf; struct dentry *dentry = file->f_path.dentry; + struct inode *inode = file_inode(file); unsigned int len = object->cookie->aux_len, tlen; const void *p = fscache_get_aux(object->cookie); enum cachefiles_coherency_trace why; ssize_t xlen; + uoff_t obj_size; int ret = -ESTALE; tlen = sizeof(struct cachefiles_xattr) + len; @@ -121,34 +181,39 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file if (xlen != tlen) { if (xlen < 0) { ret = xlen; - trace_cachefiles_vfs_error(object, file_inode(file), xlen, + trace_cachefiles_vfs_error(object, inode, xlen, cachefiles_trace_getxattr_error); } if (xlen == -EIO) cachefiles_io_error_obj( object, "Failed to read aux with error %zd", xlen); + obj_size = 0; why = cachefiles_coherency_check_xattr; goto out; } + obj_size = be64_to_cpu(buf->object_size); if (buf->type != CACHEFILES_COOKIE_TYPE_DATA) { why = cachefiles_coherency_check_type; } else if (memcmp(buf->data, p, len) != 0) { why = cachefiles_coherency_check_aux; - } else if (be64_to_cpu(buf->object_size) != object->cookie->object_size) { + } else if (obj_size != object->cookie->object_size) { why = cachefiles_coherency_check_objsize; } else if (buf->content == CACHEFILES_CONTENT_DIRTY) { // TODO: Begin conflict resolution pr_warn("Dirty object in cache\n"); why = cachefiles_coherency_check_dirty; } else { + object->content_info = buf->content; + object->object_size = obj_size; + atomic64_set(&object->read_limit, i_size_read(inode)); why = cachefiles_coherency_check_ok; ret = 0; } out: - trace_cachefiles_coherency(object, file_inode(file)->i_ino, + trace_cachefiles_coherency(object, inode->i_ino, obj_size, buf->data, buf->content, why); kfree(buf); return ret; @@ -163,6 +228,9 @@ int cachefiles_remove_object_xattr(struct cachefiles_cache *cache, { int ret; + trace_cachefiles_coherency(object, d_inode(dentry)->i_ino, 0, NULL, 0, + cachefiles_coherency_remove); + ret = cachefiles_inject_remove_error(); if (ret == 0) { ret = mnt_want_write(cache->mnt); diff --git a/fs/ceph/Kconfig b/fs/ceph/Kconfig index 3d64a316ca31..aa6ccd7794d2 100644 --- a/fs/ceph/Kconfig +++ b/fs/ceph/Kconfig @@ -4,6 +4,7 @@ config CEPH_FS depends on INET select CEPH_LIB select NETFS_SUPPORT + select NETFS_PGPRIV2 select FS_ENCRYPTION_ALGS if FS_ENCRYPTION default n help diff --git a/fs/ceph/acl.c b/fs/ceph/acl.c index 85d3dd48b167..124f07ae5b2d 100644 --- a/fs/ceph/acl.c +++ b/fs/ceph/acl.c @@ -87,7 +87,7 @@ retry: return acl; } -int ceph_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ceph_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int ret = 0; diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c index e598b2d424ec..42b55ce30a32 100644 --- a/fs/ceph/addr.c +++ b/fs/ceph/addr.c @@ -65,7 +65,7 @@ (CONGESTION_ON_THRESH(congestion_kb) - \ (CONGESTION_ON_THRESH(congestion_kb) >> 2)) -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata); static inline struct ceph_snap_context *page_snap_context(struct page *page) @@ -1868,7 +1868,7 @@ ceph_find_incompatible(struct folio *folio) return NULL; } -static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned int len, +static int ceph_netfs_check_write_begin(struct file *file, uoff_t pos, unsigned int len, struct folio **foliop, void **_fsdata) { struct inode *inode = file_inode(file); diff --git a/fs/ceph/dir.c b/fs/ceph/dir.c index 2e5c0ccb1b34..d9615d67bf1c 100644 --- a/fs/ceph/dir.c +++ b/fs/ceph/dir.c @@ -921,7 +921,7 @@ int ceph_handle_notrace_create(struct inode *dir, struct dentry *dentry) return PTR_ERR(result); } -static int ceph_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -988,7 +988,7 @@ out: return err; } -static int ceph_create(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ceph_mknod(idmap, dir, dentry, mode, 0); @@ -1032,7 +1032,7 @@ static int prep_encrypted_symlink_target(struct ceph_mds_request *req, } #endif -static int ceph_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ceph_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *dest) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -1106,7 +1106,7 @@ out: return err; } -static struct dentry *ceph_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ceph_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(dir->i_sb); @@ -1478,7 +1478,7 @@ out: return err; } -static int ceph_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ceph_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/ceph/file.c b/fs/ceph/file.c index bd3e3f5c269e..2c994c08ed4b 100644 --- a/fs/ceph/file.c +++ b/fs/ceph/file.c @@ -795,7 +795,7 @@ static int ceph_finish_async_create(struct inode *dir, struct inode *inode, int ceph_atomic_open(struct inode *dir, struct dentry *dentry, struct file *file, unsigned flags, umode_t mode) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct ceph_fs_client *fsc = ceph_sb_to_fs_client(dir->i_sb); struct ceph_client *cl = fsc->client; struct ceph_mds_client *mdsc = fsc->mdsc; diff --git a/fs/ceph/inode.c b/fs/ceph/inode.c index d52e2b389e0b..a695dba82554 100644 --- a/fs/ceph/inode.c +++ b/fs/ceph/inode.c @@ -2398,7 +2398,7 @@ static const char *ceph_encrypted_get_link(struct dentry *dentry, done); } -static int ceph_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int ceph_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) @@ -2568,7 +2568,7 @@ out: return ret; } -int __ceph_setattr(struct mnt_idmap *idmap, struct inode *inode, +int __ceph_setattr(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *attr, struct ceph_iattr *cia) { struct ceph_inode_info *ci = ceph_inode(inode); @@ -2921,7 +2921,7 @@ out: /* * setattr */ -int ceph_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ceph_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -3098,7 +3098,7 @@ out: * Check inode permissions. We verify we have a valid value for * the AUTH cap, then call the generic handler. */ -int ceph_permission(struct mnt_idmap *idmap, struct inode *inode, +int ceph_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int err; @@ -3145,7 +3145,7 @@ static int statx_to_caps(u32 want, umode_t mode) * Get all the attributes. If we have sufficient caps for the requested attrs, * then we can avoid talking to the MDS at all. */ -int ceph_getattr(struct mnt_idmap *idmap, const struct path *path, +int ceph_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); diff --git a/fs/ceph/mds_client.h b/fs/ceph/mds_client.h index e7a262c9c2ab..ea48ec5383ef 100644 --- a/fs/ceph/mds_client.h +++ b/fs/ceph/mds_client.h @@ -375,7 +375,7 @@ struct ceph_mds_request { int r_fmode; /* file mode, if expecting cap */ int r_request_release_offset; const struct cred *r_cred; - struct mnt_idmap *r_mnt_idmap; + const struct mnt_idmap *r_mnt_idmap; struct timespec64 r_stamp; /* for choosing which mds to send this request to */ diff --git a/fs/ceph/super.h b/fs/ceph/super.h index 72d4e30304dc..a033331bb151 100644 --- a/fs/ceph/super.h +++ b/fs/ceph/super.h @@ -1166,18 +1166,18 @@ static inline int ceph_do_getattr(struct inode *inode, int mask, bool force) { return __ceph_do_getattr(inode, NULL, mask, force); } -extern int ceph_permission(struct mnt_idmap *idmap, +extern int ceph_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); struct ceph_iattr { struct ceph_fscrypt_auth *fscrypt_auth; }; -extern int __ceph_setattr(struct mnt_idmap *idmap, struct inode *inode, +extern int __ceph_setattr(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *attr, struct ceph_iattr *cia); -extern int ceph_setattr(struct mnt_idmap *idmap, +extern int ceph_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -extern int ceph_getattr(struct mnt_idmap *idmap, +extern int ceph_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); void ceph_inode_shutdown(struct inode *inode); @@ -1252,7 +1252,7 @@ void ceph_release_acl_sec_ctx(struct ceph_acl_sec_ctx *as_ctx); #ifdef CONFIG_CEPH_FS_POSIX_ACL struct posix_acl *ceph_get_acl(struct inode *, int, bool); -int ceph_set_acl(struct mnt_idmap *idmap, +int ceph_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int ceph_pre_init_acls(struct inode *dir, umode_t *mode, struct ceph_acl_sec_ctx *as_ctx); diff --git a/fs/ceph/xattr.c b/fs/ceph/xattr.c index cc4ffbbcb719..7d77214c76c6 100644 --- a/fs/ceph/xattr.c +++ b/fs/ceph/xattr.c @@ -1352,7 +1352,7 @@ static int ceph_get_xattr_handler(const struct xattr_handler *handler, } static int ceph_set_xattr_handler(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/char_dev.c b/fs/char_dev.c index 00229e25c10f..5ce5423f6c99 100644 --- a/fs/char_dev.c +++ b/fs/char_dev.c @@ -280,7 +280,9 @@ int __register_chrdev(unsigned int major, unsigned int baseminor, cdev->owner = fops->owner; cdev->ops = fops; - kobject_set_name(&cdev->kobj, "%s", name); + err = kobject_set_name(&cdev->kobj, "%s", name); + if (err) + goto out; err = cdev_add(cdev, MKDEV(cd->major, baseminor), count); if (err) diff --git a/fs/coda/coda_linux.h b/fs/coda/coda_linux.h index dd6277d87afb..0c0d5f81653c 100644 --- a/fs/coda/coda_linux.h +++ b/fs/coda/coda_linux.h @@ -46,12 +46,12 @@ extern const struct file_operations coda_ioctl_operations; /* operations shared over more than one file */ int coda_open(struct inode *i, struct file *f); int coda_release(struct inode *i, struct file *f); -int coda_permission(struct mnt_idmap *idmap, struct inode *inode, +int coda_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int coda_revalidate_inode(struct inode *); -int coda_getattr(struct mnt_idmap *, const struct path *, struct kstat *, +int coda_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); -int coda_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int coda_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); /* this file: helpers */ char *coda_f2s(struct CodaFid *f); diff --git a/fs/coda/dir.c b/fs/coda/dir.c index 67148edfadee..a85be5962e62 100644 --- a/fs/coda/dir.c +++ b/fs/coda/dir.c @@ -73,7 +73,7 @@ static struct dentry *coda_lookup(struct inode *dir, struct dentry *entry, unsig } -int coda_permission(struct mnt_idmap *idmap, struct inode *inode, +int coda_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int error; @@ -133,7 +133,7 @@ static inline void coda_dir_drop_nlink(struct inode *dir) } /* creation routines: create, mknod, mkdir, link, symlink */ -static int coda_create(struct mnt_idmap *idmap, struct inode *dir, +static int coda_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *de, umode_t mode) { int error; @@ -166,7 +166,7 @@ err_out: return error; } -static struct dentry *coda_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *coda_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *de, umode_t mode) { struct inode *inode; @@ -233,7 +233,7 @@ static int coda_link(struct dentry *source_de, struct inode *dir_inode, } -static int coda_symlink(struct mnt_idmap *idmap, +static int coda_symlink(const struct mnt_idmap *idmap, struct inode *dir_inode, struct dentry *de, const char *symname) { @@ -300,7 +300,7 @@ static int coda_rmdir(struct inode *dir, struct dentry *de) } /* rename */ -static int coda_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int coda_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/coda/inode.c b/fs/coda/inode.c index 40b43866e6a5..c449954e23c2 100644 --- a/fs/coda/inode.c +++ b/fs/coda/inode.c @@ -294,7 +294,7 @@ static void coda_evict_inode(struct inode *inode) coda_cache_clear_inode(inode); } -int coda_getattr(struct mnt_idmap *idmap, const struct path *path, +int coda_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { int err = coda_revalidate_inode(d_inode(path->dentry)); @@ -304,7 +304,7 @@ int coda_getattr(struct mnt_idmap *idmap, const struct path *path, return err; } -int coda_setattr(struct mnt_idmap *idmap, struct dentry *de, +int coda_setattr(const struct mnt_idmap *idmap, struct dentry *de, struct iattr *iattr) { struct inode *inode = d_inode(de); diff --git a/fs/coda/pioctl.c b/fs/coda/pioctl.c index 36e35c15561a..c457e9bab94b 100644 --- a/fs/coda/pioctl.c +++ b/fs/coda/pioctl.c @@ -24,7 +24,7 @@ #include "coda_linux.h" /* pioctl ops */ -static int coda_ioctl_permission(struct mnt_idmap *idmap, +static int coda_ioctl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); static long coda_pioctl(struct file *filp, unsigned int cmd, unsigned long user_data); @@ -41,7 +41,7 @@ const struct file_operations coda_ioctl_operations = { }; /* the coda pioctl inode ops */ -static int coda_ioctl_permission(struct mnt_idmap *idmap, +static int coda_ioctl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return (mask & MAY_EXEC) ? -EACCES : 0; diff --git a/fs/configfs/configfs_internal.h b/fs/configfs/configfs_internal.h index 4bc19cd8d666..5f627e58f135 100644 --- a/fs/configfs/configfs_internal.h +++ b/fs/configfs/configfs_internal.h @@ -76,7 +76,7 @@ extern int configfs_make_dirent(struct configfs_dirent *, struct dentry *, extern int configfs_dirent_is_ready(struct configfs_dirent *); extern const unsigned char * configfs_get_name(struct configfs_dirent *sd); -extern int configfs_setattr(struct mnt_idmap *idmap, +extern int configfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); extern struct dentry *configfs_pin_fs(void); @@ -90,7 +90,7 @@ extern const struct inode_operations configfs_root_inode_operations; extern const struct inode_operations configfs_symlink_inode_operations; extern const struct dentry_operations configfs_dentry_ops; -extern int configfs_symlink(struct mnt_idmap *idmap, +extern int configfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname); extern int configfs_unlink(struct inode *dir, struct dentry *dentry); diff --git a/fs/configfs/dir.c b/fs/configfs/dir.c index cb45e151d852..0c80feec5926 100644 --- a/fs/configfs/dir.c +++ b/fs/configfs/dir.c @@ -1295,7 +1295,7 @@ out_root_unlock: } EXPORT_SYMBOL(configfs_depend_item_unlocked); -static struct dentry *configfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *configfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int ret = 0; diff --git a/fs/configfs/inode.c b/fs/configfs/inode.c index 69f1f24e890f..c92a05251a47 100644 --- a/fs/configfs/inode.c +++ b/fs/configfs/inode.c @@ -32,7 +32,7 @@ static const struct inode_operations configfs_inode_operations ={ .setattr = configfs_setattr, }; -int configfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int configfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode * inode = d_inode(dentry); diff --git a/fs/configfs/symlink.c b/fs/configfs/symlink.c index 3b31c714400f..89178f5d371a 100644 --- a/fs/configfs/symlink.c +++ b/fs/configfs/symlink.c @@ -146,7 +146,7 @@ static int get_target(const char *symname, struct config_item **target, } -int configfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int configfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int ret; diff --git a/fs/coredump.c b/fs/coredump.c index 6114839f5178..f33ece836b10 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -39,6 +39,7 @@ #include <linux/oom.h> #include <linux/compat.h> #include <linux/fs.h> +#include <linux/wait_bit.h> #include <linux/path.h> #include <linux/timekeeping.h> #include <linux/sysctl.h> @@ -51,7 +52,6 @@ #include <net/sock.h> #include <uapi/linux/pidfd.h> #include <uapi/linux/un.h> -#include <uapi/linux/coredump.h> #include <linux/uaccess.h> #include <asm/mmu_context.h> @@ -68,6 +68,8 @@ static bool dump_vma_snapshot(struct coredump_params *cprm); static void free_vma_snapshot(struct coredump_params *cprm); +static void dump_end_record(struct coredump_params *cprm); +static bool dump_flush_skip(struct coredump_params *cprm); #define CORE_FILE_NOTE_SIZE_DEFAULT (4*1024*1024) /* Define a reasonable max cap */ @@ -83,6 +85,8 @@ static int core_uses_pid; static unsigned int core_pipe_limit; static unsigned int core_sort_vma; static char core_pattern[CORENAME_MAX_SIZE] = "core"; +/* Taken around every copy in and out of core_pattern. */ +static DEFINE_SPINLOCK(core_pattern_lock); static int core_name_size = CORENAME_MAX_SIZE; unsigned int core_file_note_size_limit = CORE_FILE_NOTE_SIZE_DEFAULT; static atomic_t core_pipe_count = ATOMIC_INIT(0); @@ -98,9 +102,7 @@ struct core_name { char *corename __counted_by_ptr(size); int used, size; unsigned int core_pipe_limit; - bool core_dumped; enum coredump_type_t core_type; - u64 mask; }; static int expand_corename(struct core_name *cn, int size) @@ -240,18 +242,22 @@ static bool coredump_parse(struct core_name *cn, struct coredump_params *cprm, size_t **argv, int *argc) { const struct cred *cred = current_cred(); - const char *pat_ptr = core_pattern; + char pattern[CORENAME_MAX_SIZE]; + const char *pat_ptr = pattern; bool was_space = false; int pid_in_pattern = 0; int err = 0; - cn->mask = COREDUMP_KERNEL; + /* The sysctl handler may be publishing a new pattern. */ + scoped_guard(spinlock, &core_pattern_lock) + strscpy(pattern, core_pattern); + + cprm->mask = COREDUMP_KERNEL; if (core_pipe_limit) - cn->mask |= COREDUMP_WAIT; + cprm->mask |= COREDUMP_WAIT; cn->used = 0; cn->corename = NULL; cn->core_pipe_limit = 0; - cn->core_dumped = false; if (*pat_ptr == '|') cn->core_type = COREDUMP_PIPE; else if (*pat_ptr == '@') @@ -508,60 +514,64 @@ static int zap_threads(struct task_struct *tsk, int nr = -EAGAIN; spin_lock_irq(&tsk->sighand->siglock); - if (!(signal->flags & SIGNAL_GROUP_EXIT) && !signal->group_exec_task) { + /* A freeze requested before the dump would be lost with TIF_SIGPENDING. */ + if (!(signal->flags & SIGNAL_GROUP_EXIT) && !signal->group_exec_task && + !freezing(tsk) && !(tsk->jobctl & JOBCTL_TRAP_FREEZE)) { /* Allow SIGKILL, see prepare_signal() */ signal->core_state = core_state; nr = zap_process(signal, exit_code); clear_tsk_thread_flag(tsk, TIF_SIGPENDING); tsk->flags |= PF_DUMPCORE; - atomic_set(&core_state->nr_threads, nr); + atomic_set(&core_state->threads_remaining, nr); } spin_unlock_irq(&tsk->sighand->siglock); return nr; } +static void coredump_wait_inactive(struct core_state *core_state) +{ + struct core_thread *ptr; + + wait_var_event_state(&core_state->threads_remaining, + !atomic_read_acquire(&core_state->threads_remaining), + TASK_UNINTERRUPTIBLE | TASK_FREEZABLE); + /* + * Wait for all the threads to become inactive, so that + * all the thread context (extended register state, like + * fpu etc) gets copied to the memory. + */ + for (ptr = core_state->tasks; ptr; ptr = ptr->next) + wait_task_inactive(ptr->task, TASK_ANY); +} + static int coredump_wait(int exit_code, struct core_state *core_state) { struct task_struct *tsk = current; int core_waiters = -EBUSY; - init_completion(&core_state->startup); - core_state->dumper.task = tsk; - core_state->dumper.next = NULL; + core_state->tasks = NULL; core_waiters = zap_threads(tsk, core_state, exit_code); - if (core_waiters > 0) { - struct core_thread *ptr; - - wait_for_completion_state(&core_state->startup, - TASK_UNINTERRUPTIBLE|TASK_FREEZABLE); - /* - * Wait for all the threads to become inactive, so that - * all the thread context (extended register state, like - * fpu etc) gets copied to the memory. - */ - ptr = core_state->dumper.next; - while (ptr != NULL) { - wait_task_inactive(ptr->task, TASK_ANY); - ptr = ptr->next; - } - } + if (core_waiters > 0) + coredump_wait_inactive(core_state); return core_waiters; } -static void coredump_finish(bool core_dumped) +static void coredump_finish(enum coredump_state state) { struct core_thread *curr, *next; struct task_struct *task; spin_lock_irq(¤t->sighand->siglock); - if (core_dumped && !__fatal_signal_pending(current)) + if ((state & COREDUMP_STATE_STARTED) && !__fatal_signal_pending(current)) current->signal->group_exit_code |= 0x80; - next = current->signal->core_state->dumper.next; + next = current->signal->core_state->tasks; current->signal->core_state = NULL; spin_unlock_irq(¤t->sighand->siglock); + /* A released thread may exit and be freed before it is woken. */ + guard(rcu)(); while ((curr = next) != NULL) { next = curr->next; task = curr->task; @@ -570,6 +580,7 @@ static void coredump_finish(bool core_dumped) * ->task == NULL before we read ->next. */ smp_mb(); + /* Any wakeup now lets the thread exit, rcu keeps it alive. */ curr->task = NULL; wake_up_process(task); } @@ -577,13 +588,8 @@ static void coredump_finish(bool core_dumped) static bool dump_interrupted(void) { - /* - * SIGKILL or freezing() interrupt the coredumping. Perhaps we - * can do try_to_freeze() and check __fatal_signal_pending(), - * but then we need to teach dump_write() to restart and clear - * TIF_SIGPENDING. - */ - return fatal_signal_pending(current) || freezing(current); + /* Only SIGKILL and the freezers set it after zap_threads(). */ + return task_sigpending(current); } static void wait_for_dump_helpers(struct file *file) @@ -664,7 +670,12 @@ static int umh_coredump_setup(struct subprocess_info *info, struct cred *new) return 0; } +static_assert(sizeof(struct coredump_record_header) == COREDUMP_RECORD_HEADER_SIZE_VER0); + #ifdef CONFIG_UNIX +/* af_unix halves the send buffer to size a single skb. */ +#define COREDUMP_SOCK_SNDBUF_MIN (3 * PAGE_SIZE) + static bool coredump_sock_connect(struct core_name *cn, struct coredump_params *cprm) { struct file *file __free(fput) = NULL; @@ -690,6 +701,10 @@ static bool coredump_sock_connect(struct core_name *cn, struct coredump_params * if (retval < 0) return false; + /* Don't let a page-sized write split into several skbs. */ + socket->sk->sk_sndbuf = max_t(int, socket->sk->sk_sndbuf, + COREDUMP_SOCK_SNDBUF_MIN); + file = sock_alloc_file(socket, 0, NULL); if (IS_ERR(file)) return false; @@ -752,8 +767,39 @@ static inline bool coredump_sock_send(struct file *file, struct coredump_req *re return ret == sizeof(*req); } +static_assert(sizeof(struct coredump_req) == COREDUMP_REQ_SIZE_VER1); +static_assert(sizeof(struct coredump_ack) == COREDUMP_ACK_SIZE_VER1); static_assert(sizeof(enum coredump_mark) == sizeof(__u32)); +/* Every memory type this kernel knows. */ +#define COREDUMP_MEMORY_ALL \ + (COREDUMP_MEMORY_ANON_PRIVATE | COREDUMP_MEMORY_ANON_SHARED | \ + COREDUMP_MEMORY_FILE_PRIVATE | COREDUMP_MEMORY_FILE_SHARED | \ + COREDUMP_MEMORY_ELF_HEADERS | \ + COREDUMP_MEMORY_HUGETLB_PRIVATE | COREDUMP_MEMORY_HUGETLB_SHARED | \ + COREDUMP_MEMORY_DAX_PRIVATE | COREDUMP_MEMORY_DAX_SHARED) + +#define COREDUMP_MEMORY_TYPE_BIT(mmf) BIT((mmf) - MMF_DUMP_FILTER_SHIFT) +static_assert(COREDUMP_MEMORY_ALL == (MMF_DUMP_FILTER_MASK >> MMF_DUMP_FILTER_SHIFT)); +static_assert(COREDUMP_MEMORY_ANON_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_PRIVATE)); +static_assert(COREDUMP_MEMORY_ANON_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ANON_SHARED)); +static_assert(COREDUMP_MEMORY_FILE_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_PRIVATE)); +static_assert(COREDUMP_MEMORY_FILE_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_MAPPED_SHARED)); +static_assert(COREDUMP_MEMORY_ELF_HEADERS == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_ELF_HEADERS)); +static_assert(COREDUMP_MEMORY_HUGETLB_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_PRIVATE)); +static_assert(COREDUMP_MEMORY_HUGETLB_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_HUGETLB_SHARED)); +static_assert(COREDUMP_MEMORY_DAX_PRIVATE == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_PRIVATE)); +static_assert(COREDUMP_MEMORY_DAX_SHARED == + COREDUMP_MEMORY_TYPE_BIT(MMF_DUMP_DAX_SHARED)); + static inline bool coredump_sock_mark(struct file *file, enum coredump_mark mark) { struct msghdr msg = { .msg_flags = MSG_NOSIGNAL }; @@ -795,10 +841,14 @@ static inline void coredump_sock_shutdown(struct file *file) static bool coredump_sock_request(struct core_name *cn, struct coredump_params *cprm) { struct coredump_req req = { - .size = sizeof(struct coredump_req), - .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | - COREDUMP_REJECT | COREDUMP_WAIT, - .size_ack = sizeof(struct coredump_ack), + .size = sizeof(struct coredump_req), + .mask = COREDUMP_KERNEL | COREDUMP_USERSPACE | + COREDUMP_REJECT | COREDUMP_WAIT | + COREDUMP_RECORDS | COREDUMP_SPARSE | + COREDUMP_MEMORY_TYPES, + .size_ack = sizeof(struct coredump_ack), + .memory_types = cprm->memory_types, + .memory_types_mask = COREDUMP_MEMORY_ALL, }; struct coredump_ack ack = {}; ssize_t usize; @@ -851,7 +901,54 @@ static bool coredump_sock_request(struct core_name *cn, struct coredump_params * return false; } - cn->mask = ack.mask; + /* Records only describe a coredump the kernel writes. */ + if ((ack.mask & COREDUMP_RECORDS) && !(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Zero records only exist inside a record stream. */ + if ((ack.mask & COREDUMP_SPARSE) && !(ack.mask & COREDUMP_RECORDS)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + if (ack.mask & COREDUMP_MEMORY_TYPES) { + /* The memory types need the whole field. */ + if (usize < COREDUMP_ACK_SIZE_VER1) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_MINSIZE); + return false; + } + + /* The memory types only select what the kernel writes. */ + if (!(ack.mask & COREDUMP_KERNEL)) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_CONFLICTING); + return false; + } + + /* Refuse unknown memory types. */ + if (ack.memory_types & ~req.memory_types_mask) { + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + } else if (ack.memory_types) { + /* Like @spare the field must be zero when it isn't used. */ + coredump_sock_mark(cprm->file, COREDUMP_MARK_UNSUPPORTED); + return false; + } + + /* Record header scratch; a bvec can't point at the stack. */ + if (ack.mask & COREDUMP_RECORDS) { + cprm->record_hdr = kmalloc_obj(*cprm->record_hdr); + if (!cprm->record_hdr) + return false; + } + + /* The server's selection replaces the task's entirely. */ + if (ack.mask & COREDUMP_MEMORY_TYPES) + cprm->memory_types = ack.memory_types; + + cprm->mask = ack.mask; return coredump_sock_mark(cprm->file, COREDUMP_MARK_REQACK); } @@ -878,7 +975,7 @@ static inline bool coredump_force_suid_safe(const struct coredump_params *cprm) static bool coredump_file(struct core_name *cn, struct coredump_params *cprm, const struct linux_binfmt *binfmt) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode; struct file *file __free(fput) = NULL; int open_flags = O_CREAT | O_WRONLY | O_NOFOLLOW | O_LARGEFILE | O_EXCL; @@ -1032,29 +1129,41 @@ static bool coredump_pipe(struct core_name *cn, struct coredump_params *cprm, return true; } -static bool coredump_write(struct core_name *cn, - struct coredump_params *cprm, - const struct linux_binfmt *binfmt) +static bool coredump_write(struct coredump_params *cprm, + const struct linux_binfmt *binfmt) { - - if (dump_interrupted()) + if (dump_interrupted()) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return true; + } - if (!dump_vma_snapshot(cprm)) + if (!dump_vma_snapshot(cprm)) { + cprm->state |= COREDUMP_STATE_TRUNCATED; return false; + } file_start_write(cprm->file); - cn->core_dumped = binfmt->core_dump(cprm); + if (!binfmt->core_dump(cprm)) + cprm->state |= COREDUMP_STATE_TRUNCATED; /* - * Ensures that file size is big enough to contain the current - * file postion. This prevents gdb from complaining about - * a truncated file if the last "write" to the file was - * dump_skip. + * A trailing hole still has to land in the coredump. Seeking over + * it doesn't grow the file, so the last byte of it is written + * instead and gdb doesn't see a truncated file. Everything else + * puts the hole on the wire as it flushes it. */ if (cprm->to_skip) { - cprm->to_skip--; - dump_emit(cprm, "", 1); + bool flushed; + + if (cprm->file->f_mode & FMODE_LSEEK) { + cprm->to_skip--; + flushed = dump_emit(cprm, "", 1); + } else { + flushed = dump_flush_skip(cprm); + } + if (!flushed) + cprm->state |= COREDUMP_STATE_TRUNCATED; } + dump_end_record(cprm); file_end_write(cprm->file); free_vma_snapshot(cprm); return true; @@ -1069,7 +1178,8 @@ static void coredump_cleanup(struct core_name *cn, struct coredump_params *cprm) atomic_dec(&core_pipe_count); } kfree(cn->corename); - coredump_finish(cn->core_dumped); + kfree(cprm->record_hdr); + coredump_finish(cprm->state); } static inline bool coredump_skip(const struct coredump_params *cprm, @@ -1115,29 +1225,24 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } /* Don't even generate the coredump. */ - if (cn->mask & COREDUMP_REJECT) - return; - - /* get us an unshared descriptor table; almost always a no-op */ - /* The cell spufs coredump code reads the file descriptor tables */ - if (unshare_files()) + if (cprm->mask & COREDUMP_REJECT) return; - if ((cn->mask & COREDUMP_KERNEL) && !coredump_write(cn, cprm, binfmt)) + if ((cprm->mask & COREDUMP_KERNEL) && !coredump_write(cprm, binfmt)) return; coredump_sock_shutdown(cprm->file); /* Let the parent know that a coredump was generated. */ - if (cn->mask & COREDUMP_USERSPACE) - cn->core_dumped = true; + if (cprm->mask & COREDUMP_USERSPACE) + cprm->state |= COREDUMP_STATE_STARTED; /* * When core_pipe_limit is set we wait for the coredump server * or usermodehelper to finish before exiting so it can e.g., * inspect /proc/<pid>. */ - if (cn->mask & COREDUMP_WAIT) { + if (cprm->mask & COREDUMP_WAIT) { switch (cn->core_type) { case COREDUMP_PIPE: wait_for_dump_helpers(cprm->file); @@ -1153,6 +1258,10 @@ static void do_coredump(struct core_name *cn, struct coredump_params *cprm, } } +#define COREDUMP_TASK_MEMORY_TYPES(mm) \ + ((__mm_flags_get_word((mm)) & MMF_DUMP_FILTER_MASK) >> \ + MMF_DUMP_FILTER_SHIFT) + void vfs_coredump(const kernel_siginfo_t *siginfo) { size_t *argv __free(kfree) = NULL; @@ -1164,8 +1273,8 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) struct coredump_params cprm = { .siginfo = siginfo, .limit = rlimit(RLIMIT_CORE), - /* Snapshot MMF_DUMP_FILTER_* (unlocked) and dumpable for the dump. */ - .mm_flags = __mm_flags_get_word(mm), + /* Snapshot the memory types (unlocked) and dumpable for the dump. */ + .memory_types = COREDUMP_TASK_MEMORY_TYPES(mm), .dumpable = task_exec_state_get_dumpable(current), .vma_meta = NULL, .cpu = raw_smp_processor_id(), @@ -1191,6 +1300,8 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) if (coredump_wait(siginfo->si_signo, &core_state) < 0) return; + /* Task work must not cut the dump short, see signal_pending(). */ + guard(no_notify_signal)(); scoped_with_creds(cred) do_coredump(&cn, &cprm, &argv, &argc, binfmt); coredump_cleanup(&cn, &cprm); @@ -1202,60 +1313,181 @@ void vfs_coredump(const kernel_siginfo_t *siginfo) * do on a core-file: use only these functions to write out all the * necessary info. */ -static int __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +static bool dump_records(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_RECORDS; +} + +static bool dump_sparse(const struct coredump_params *cprm) +{ + return cprm->mask & COREDUMP_SPARSE; +} + +/* Describe the next @len bytes of the coredump. Returns the header size. */ +static size_t dump_record_init(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, + u64 len) +{ + if (!dump_records(cprm)) + return 0; + + *cprm->record_hdr = (struct coredump_record_header) { + .size = sizeof(*cprm->record_hdr), + .type = type, + .flags = flags, + .offset = cprm->pos, + .len = len, + }; + + return sizeof(*cprm->record_hdr); +} + +/* Write @iter whole or fail. @len is what it advances the coredump by. */ +static bool dump_write_iter(struct coredump_params *cprm, struct iov_iter *iter, + size_t len) { struct file *file = cprm->file; + size_t count = iov_iter_count(iter); loff_t pos = file->f_pos; ssize_t n; - if (cprm->written + nr > cprm->limit) - return 0; - if (dump_interrupted()) - return 0; - n = __kernel_write(file, addr, nr, &pos); - if (n != nr) - return 0; + n = __kernel_write_iter(file, iter, &pos); + if (n != (ssize_t)count) + return false; file->f_pos = pos; - cprm->written += n; - cprm->pos += n; + cprm->written += count; + cprm->pos += len; + + return true; +} + +/* One record, never more than a page. See __dump_emit(). */ +static bool dump_emit_chunk(struct coredump_params *cprm, const void *addr, + int nr) +{ + struct kvec kvec[2]; + struct iov_iter iter; + unsigned int nseg = 0; + size_t hdrlen; - return 1; + if (dump_interrupted()) + return false; + + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, nr); + if (hdrlen) { + kvec[nseg].iov_base = cprm->record_hdr; + kvec[nseg].iov_len = hdrlen; + nseg++; + } + kvec[nseg].iov_base = (void *)addr; + kvec[nseg].iov_len = nr; + nseg++; + + iov_iter_kvec(&iter, ITER_SOURCE, kvec, nseg, hdrlen + nr); + + return dump_write_iter(cprm, &iter, nr); +} + +static bool __dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (cprm->written + nr > cprm->limit) + return false; + + while (nr) { + int chunk = min_t(int, nr, PAGE_SIZE); + + if (!dump_emit_chunk(cprm, addr, chunk)) + return false; + + addr += chunk; + nr -= chunk; + } + + return true; } -static int __dump_skip(struct coredump_params *cprm, size_t nr) +/* Send a record that stands on its own: a header and nothing else. */ +static bool dump_emit_record(struct coredump_params *cprm, + enum coredump_record_type type, u64 flags, u64 len) +{ + struct kvec kvec; + struct iov_iter iter; + size_t hdrlen; + + hdrlen = dump_record_init(cprm, type, flags, len); + if (!hdrlen) + return false; + + kvec.iov_base = cprm->record_hdr; + kvec.iov_len = hdrlen; + iov_iter_kvec(&iter, ITER_SOURCE, &kvec, 1, hdrlen); + + return dump_write_iter(cprm, &iter, len); +} + +/* Close the record stream. Only a whole coredump gets an end record. */ +static void dump_end_record(struct coredump_params *cprm) +{ + if (cprm->state & COREDUMP_STATE_TRUNCATED) + return; + + dump_emit_record(cprm, COREDUMP_RECORD_END, 0, 0); +} + +static bool __dump_skip(struct coredump_params *cprm, size_t nr) { static char zeroes[PAGE_SIZE]; struct file *file = cprm->file; + if (dump_sparse(cprm)) { + /* Hand the server the length of the hole instead of the hole itself. */ + if (dump_interrupted()) + return false; + return dump_emit_record(cprm, COREDUMP_RECORD_ZERO, 0, nr); + } + if (file->f_mode & FMODE_LSEEK) { if (dump_interrupted() || vfs_llseek(file, nr, SEEK_CUR) < 0) - return 0; + return false; cprm->pos += nr; - return 1; + return true; } - while (nr > PAGE_SIZE) { - if (!__dump_emit(cprm, zeroes, PAGE_SIZE)) - return 0; - nr -= PAGE_SIZE; + while (nr) { + size_t chunk = min_t(size_t, nr, PAGE_SIZE); + + if (!__dump_emit(cprm, zeroes, chunk)) + return false; + + nr -= chunk; } - return __dump_emit(cprm, zeroes, nr); + return true; } -int dump_emit(struct coredump_params *cprm, const void *addr, int nr) +/* Flush the accumulated hole before writing data. */ +static bool dump_flush_skip(struct coredump_params *cprm) { if (cprm->to_skip) { if (!__dump_skip(cprm, cprm->to_skip)) - return 0; + return false; cprm->to_skip = 0; } + return true; +} + +bool dump_emit(struct coredump_params *cprm, const void *addr, int nr) +{ + if (!dump_flush_skip(cprm)) + return false; return __dump_emit(cprm, addr, nr); } EXPORT_SYMBOL(dump_emit); void dump_skip_to(struct coredump_params *cprm, unsigned long pos) { + if (WARN_ON_ONCE(pos < cprm->pos)) + return; cprm->to_skip = pos - cprm->pos; } EXPORT_SYMBOL(dump_skip_to); @@ -1267,37 +1499,32 @@ void dump_skip(struct coredump_params *cprm, size_t nr) EXPORT_SYMBOL(dump_skip); #ifdef CONFIG_ELF_CORE -static int dump_emit_page(struct coredump_params *cprm, struct page *page) +static bool dump_emit_page(struct coredump_params *cprm, struct page *page) { - struct bio_vec bvec; + struct bio_vec bvec[2]; struct iov_iter iter; - struct file *file = cprm->file; - loff_t pos; - ssize_t n; + unsigned int nseg = 0; + size_t hdrlen; if (!page) - return 0; + return false; - if (cprm->to_skip) { - if (!__dump_skip(cprm, cprm->to_skip)) - return 0; - cprm->to_skip = 0; - } + if (!dump_flush_skip(cprm)) + return false; if (cprm->written + PAGE_SIZE > cprm->limit) - return 0; + return false; if (dump_interrupted()) - return 0; - pos = file->f_pos; - bvec_set_page(&bvec, page, PAGE_SIZE, 0); - iov_iter_bvec(&iter, ITER_SOURCE, &bvec, 1, PAGE_SIZE); - n = __kernel_write_iter(cprm->file, &iter, &pos); - if (n != PAGE_SIZE) - return 0; - file->f_pos = pos; - cprm->written += PAGE_SIZE; - cprm->pos += PAGE_SIZE; + return false; - return 1; + /* Hand the record header to the same write as the page it describes. */ + hdrlen = dump_record_init(cprm, COREDUMP_RECORD_DATA, 0, PAGE_SIZE); + if (hdrlen) + bvec_set_virt(&bvec[nseg++], cprm->record_hdr, hdrlen); + bvec_set_page(&bvec[nseg++], page, PAGE_SIZE, 0); + + iov_iter_bvec(&iter, ITER_SOURCE, bvec, nseg, hdrlen + PAGE_SIZE); + + return dump_write_iter(cprm, &iter, PAGE_SIZE); } /* @@ -1329,18 +1556,19 @@ static inline struct page *dump_page_copy(struct page *src, struct page *dst) } #endif -int dump_user_range(struct coredump_params *cprm, unsigned long start, - unsigned long len) +bool dump_user_range(struct coredump_params *cprm, unsigned long start, + unsigned long len) { unsigned long addr; struct page *dump_page; - int locked, ret; + int locked; + bool ret; dump_page = dump_page_alloc(); if (!dump_page) - return 0; + return false; - ret = 0; + ret = false; locked = 0; for (addr = start; addr < start + len; addr += PAGE_SIZE) { struct page *page; @@ -1364,7 +1592,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, mmap_read_unlock(current->mm); locked = 0; } - int stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); + bool stop = !dump_emit_page(cprm, dump_page_copy(page, dump_page)); put_page(page); if (stop) goto out; @@ -1383,7 +1611,7 @@ int dump_user_range(struct coredump_params *cprm, unsigned long start, } cond_resched(); } - ret = 1; + ret = true; out: if (locked) mmap_read_unlock(current->mm); @@ -1393,14 +1621,14 @@ out: } #endif -int dump_align(struct coredump_params *cprm, int align) +bool dump_align(struct coredump_params *cprm, int align) { unsigned mod = (cprm->pos + cprm->to_skip) & (align - 1); if (align & (align - 1)) - return 0; + return false; if (mod) cprm->to_skip += align - mod; - return 1; + return true; } EXPORT_SYMBOL(dump_align); @@ -1417,11 +1645,11 @@ void validate_coredump_safety(void) } } -static inline bool check_coredump_socket(void) +static inline bool check_coredump_socket(const char *pattern) { const char *p; - if (core_pattern[0] != '@') + if (pattern[0] != '@') return true; /* @@ -1433,16 +1661,16 @@ static inline bool check_coredump_socket(void) return false; /* Must be an absolute path... */ - if (core_pattern[1] != '/') { + if (pattern[1] != '/') { /* ... or the socket request protocol... */ - if (core_pattern[1] != '@') + if (pattern[1] != '@') return false; /* ... and if so must be an absolute path. */ - if (core_pattern[2] != '/') + if (pattern[2] != '/') return false; - p = &core_pattern[2]; + p = &pattern[2]; } else { - p = &core_pattern[1]; + p = &pattern[1]; } /* The path obviously cannot exceed UNIX_PATH_MAX. */ @@ -1450,7 +1678,7 @@ static inline bool check_coredump_socket(void) return false; /* Must not contain ".." in the path. */ - if (name_contains_dotdot(core_pattern)) + if (name_contains_dotdot(pattern)) return false; return true; @@ -1459,27 +1687,35 @@ static inline bool check_coredump_socket(void) static int proc_dostring_coredump(const struct ctl_table *table, int write, void *buffer, size_t *lenp, loff_t *ppos) { + char pattern[CORENAME_MAX_SIZE]; + const struct ctl_table tmp = { + .procname = table->procname, + .data = pattern, + .maxlen = sizeof(pattern), + }; + bool changed = false; int error; - ssize_t retval; - char old_core_pattern[CORENAME_MAX_SIZE]; - if (!write) - return proc_dostring(table, write, buffer, lenp, ppos); + /* Work on a copy, proc_dostring() appends at *ppos. */ + scoped_guard(spinlock, &core_pattern_lock) + strscpy(pattern, core_pattern); - retval = strscpy(old_core_pattern, core_pattern, CORENAME_MAX_SIZE); - - error = proc_dostring(table, write, buffer, lenp, ppos); - if (error) + error = proc_dostring(&tmp, write, buffer, lenp, ppos); + if (error || !write) return error; - if (!check_coredump_socket()) { - strscpy(core_pattern, old_core_pattern, retval + 1); + if (!check_coredump_socket(pattern)) return -EINVAL; - } - if (strncmp(old_core_pattern, core_pattern, CORENAME_MAX_SIZE)) + /* Publish the validated pattern whole. */ + scoped_guard(spinlock, &core_pattern_lock) { + changed = strncmp(pattern, core_pattern, CORENAME_MAX_SIZE); + if (changed) + strscpy(core_pattern, pattern); + } + if (changed) validate_coredump_safety(); - return error; + return 0; } static const unsigned int core_file_note_size_min = CORE_FILE_NOTE_SIZE_DEFAULT; @@ -1582,15 +1818,15 @@ static bool always_dump_vma(struct vm_area_struct *vma) } #define DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER 1 +#define COREDUMP_MEMORY_TYPE_INCLUDE(types, type) \ + ((types) & COREDUMP_MEMORY_##type) /* * Decide how much of @vma's contents should be included in a core dump. */ static unsigned long vma_dump_size(struct vm_area_struct *vma, - unsigned long mm_flags) + u64 memory_types) { -#define FILTER(type) (mm_flags & (1UL << MMF_DUMP_##type)) - /* always dump the vdso and vsyscall sections */ if (always_dump_vma(vma)) goto whole; @@ -1600,18 +1836,22 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* support for DAX */ if (vma_is_dax(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(DAX_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(DAX_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, DAX_PRIVATE)) goto whole; return 0; } /* Hugetlb memory check */ if (is_vm_hugetlb_page(vma)) { - if ((vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_SHARED)) + if ((vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_SHARED)) goto whole; - if (!(vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_PRIVATE)) + if (!(vma->vm_flags & VM_SHARED) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, HUGETLB_PRIVATE)) goto whole; return 0; } @@ -1623,25 +1863,27 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, /* By default, dump shared memory if mapped from an anonymous file. */ if (vma->vm_flags & VM_SHARED) { if (file_inode(vma->vm_file)->i_nlink == 0 ? - FILTER(ANON_SHARED) : FILTER(MAPPED_SHARED)) + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_SHARED) : + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_SHARED)) goto whole; return 0; } /* Dump segments that have been written to. */ - if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && FILTER(ANON_PRIVATE)) + if ((!IS_ENABLED(CONFIG_MMU) || vma->anon_vma) && + COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ANON_PRIVATE)) goto whole; if (vma->vm_file == NULL) return 0; - if (FILTER(MAPPED_PRIVATE)) + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, FILE_PRIVATE)) goto whole; /* * If this is the beginning of an executable file mapping, * dump the first page to aid in determining what was mapped here. */ - if (FILTER(ELF_HEADERS) && + if (COREDUMP_MEMORY_TYPE_INCLUDE(memory_types, ELF_HEADERS) && vma->vm_pgoff == 0 && (vma->vm_flags & VM_READ)) { if ((READ_ONCE(file_inode(vma->vm_file)->i_mode) & 0111) != 0) return PAGE_SIZE; @@ -1657,8 +1899,6 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, return DUMP_SIZE_MAYBE_ELFHDR_PLACEHOLDER; } -#undef FILTER - return 0; whole: @@ -1743,7 +1983,7 @@ static bool dump_vma_snapshot(struct coredump_params *cprm) m->start = vma->vm_start; m->end = vma->vm_end; m->flags = vma->vm_flags; - m->dump_size = vma_dump_size(vma, cprm->mm_flags); + m->dump_size = vma_dump_size(vma, cprm->memory_types); m->pgoff = vma->vm_pgoff; m->file = vma->vm_file; if (m->file) @@ -469,8 +469,6 @@ static void dax_folio_init(void *entry) if (order > 0) { prep_compound_page(&folio->page, order); - if (order > 1) - INIT_LIST_HEAD(&folio->_deferred_list); WARN_ON_ONCE(folio_ref_count(folio)); } } @@ -775,24 +773,23 @@ fallback: /** * dax_layout_busy_page_range - find first pinned page in @mapping - * @mapping: address space to scan for a page with ref count > 1 + * @mapping: address space to scan for a pinned page * @start: Starting offset. Page containing 'start' is included. * @end: End offset. Page containing 'end' is included. If 'end' is LLONG_MAX, * pages from 'start' till the end of file are included. * - * DAX requires ZONE_DEVICE mapped pages. These pages are never - * 'onlined' to the page allocator so they are considered idle when - * page->count == 1. A filesystem uses this interface to determine if - * any page in the mapping is busy, i.e. for DMA, or other - * get_user_pages() usages. + * DAX requires ZONE_DEVICE mapped pages. A page is considered busy when + * folio_ref_count(folio) exceeds folio_mapcount(folio). This helper is + * used to determine if any page in the mapping is busy, i.e. for DMA, + * or other get_user_pages() usages. * * It is expected that the filesystem is holding locks to block the * establishment of new mappings in this address_space. I.e. it expects - * to be able to run unmap_mapping_range() and subsequently not race + * to be able to run unmap_mapping_pages() and subsequently not race * mapping_mapped() becoming true. */ -struct page *dax_layout_busy_page_range(struct address_space *mapping, - loff_t start, loff_t end) +static struct page *dax_layout_busy_page_range(struct address_space *mapping, + loff_t start, loff_t end) { void *entry; unsigned int scanned = 0; @@ -844,13 +841,6 @@ struct page *dax_layout_busy_page_range(struct address_space *mapping, xas_unlock_irq(&xas); return page; } -EXPORT_SYMBOL_GPL(dax_layout_busy_page_range); - -struct page *dax_layout_busy_page(struct address_space *mapping) -{ - return dax_layout_busy_page_range(mapping, 0, LLONG_MAX); -} -EXPORT_SYMBOL_GPL(dax_layout_busy_page); static int __dax_invalidate_entry(struct address_space *mapping, pgoff_t index, bool trunc) diff --git a/fs/dcache.c b/fs/dcache.c index a66be85f9d01..7a9346c4f2e4 100644 --- a/fs/dcache.c +++ b/fs/dcache.c @@ -32,6 +32,7 @@ #include <linux/bit_spinlock.h> #include <linux/rculist_bl.h> #include <linux/list_lru.h> +#include <linux/namei.h> #include "internal.h" #include "mount.h" @@ -451,6 +452,17 @@ static void dentry_free(struct dentry *dentry) } /* + * If inode is unlinked and doesn't have any aliases (i.e., all fds pointing to + * it are closed), it is pretty much dead. Except that file handle lookup could + * still revive it which causes issues to fsnotify. So once inode reaches this + * state we make sure to block creating any new aliases. + */ +static bool inode_notify_dead(struct inode *inode) +{ + return !inode->i_nlink && hlist_empty(&inode->i_dentry); +} + +/* * Release the dentry's inode, using the filesystem * d_iput() operation if defined. */ @@ -459,6 +471,7 @@ static void dentry_unlink_inode(struct dentry * dentry) __releases(dentry->d_inode->i_lock) { struct inode *inode = dentry->d_inode; + bool notify_dead; raw_write_seqcount_begin(&dentry->d_seq); __d_clear_type_and_inode(dentry); @@ -469,9 +482,10 @@ static void dentry_unlink_inode(struct dentry * dentry) */ dentry->waiters = NULL; raw_write_seqcount_end(&dentry->d_seq); + notify_dead = inode_notify_dead(inode); spin_unlock(&dentry->d_lock); spin_unlock(&inode->i_lock); - if (!inode->i_nlink) + if (notify_dead) fsnotify_inoderemove(inode); if (dentry->d_op && dentry->d_op->d_iput) dentry->d_op->d_iput(dentry, inode); @@ -830,7 +844,7 @@ static struct dentry *dentry_kill(struct dentry *dentry) if (dentry->d_op && dentry->d_op->d_release) dentry->d_op->d_release(dentry); - cond_resched(); + cond_resched_tasks_rcu_qs(); /* now that it's negative, ->d_parent is stable */ if (!IS_ROOT(dentry)) { parent = dentry->d_parent; @@ -1900,6 +1914,7 @@ EXPORT_SYMBOL(d_invalidate); static struct dentry *__d_alloc(struct super_block *sb, const struct qstr *name) { + static struct lock_class_key __lookup_key; struct dentry *dentry; char *dname; int err; @@ -1961,6 +1976,8 @@ static struct dentry *__d_alloc(struct super_block *sb, const struct qstr *name) dentry->waiters = NULL; INIT_HLIST_NODE(&dentry->d_sib); + lockdep_init_map(&dentry->lookup_map, "DCACHE_PAR_LOOKUP", &__lookup_key, 0); + if (dentry->d_op && dentry->d_op->d_init) { err = dentry->d_op->d_init(dentry); if (err) { @@ -2003,6 +2020,58 @@ struct dentry *d_alloc(struct dentry * parent, const struct qstr *name) } EXPORT_SYMBOL(d_alloc); +/** + * d_duplicate - duplicate a dentry for combined atomic operation + * @dentry: the dentry to duplicate + * + * Some rename operations need to be combined with another operation + * inside the filesystem. + * 1/ A cluster filesystem when renaming to an in-use file might need to + * first "silly-rename" that target out of the way before the main rename + * 2/ A filesystem that supports white-out might want to create a whiteout + * in place of the file being moved. + * + * For this they need two dentries which temporarily have the same name, + * before one is renamed. d_duplicate() provides for this. Given a + * positive hashed dentry, it creates a second in-lookup dentry. + * Because the original dentry exists, no other thread will try to + * create an in-lookup dentry, so there can be no race in this create. + * + * The caller should d_move() the original to a new name, often via a + * rename request, and should call d_lookup_done() on the newly created + * dentry. If the new is instantiated then the old MUST either be moved + * or dropped. + * + * Parent must be locked. + * + * Returns: an in-lookup dentry, or -ENOMEM. + */ +struct dentry *d_duplicate(struct dentry *dentry) +{ + unsigned int hash = dentry->d_name.hash; + struct dentry *parent = dentry->d_parent; + struct hlist_bl_head *b = in_lookup_hash(parent, hash); + struct dentry *new = __d_alloc(parent->d_sb, &dentry->d_name); + + if (unlikely(!new)) + return ERR_PTR(-ENOMEM); + + new->d_flags |= DCACHE_PAR_LOOKUP; + lock_map_acquire_try(&new->lookup_map); + spin_lock(&parent->d_lock); + new->d_parent = dget_dlock(parent); + hlist_add_head(&new->d_sib, &parent->d_children); + if (parent->d_flags & DCACHE_DISCONNECTED) + new->d_flags |= DCACHE_DISCONNECTED; + spin_unlock(&parent->d_lock); + + hlist_bl_lock(b); + hlist_bl_add_head(&new->d_in_lookup_hash, b); + hlist_bl_unlock(b); + return new; +} +EXPORT_SYMBOL(d_duplicate); + struct dentry *d_alloc_anon(struct super_block *sb) { return __d_alloc(sb, NULL); @@ -2172,7 +2241,6 @@ static void __d_instantiate(struct dentry *dentry, struct inode *inode) * (or otherwise set) by the caller to indicate that it is now * in use by the dcache. */ - void d_instantiate(struct dentry *entry, struct inode * inode) { BUG_ON(d_really_is_positive(entry)); @@ -2241,7 +2309,12 @@ static struct dentry *__d_obtain_alias(struct inode *inode, bool disconnected) sb = inode->i_sb; - res = d_find_any_alias(inode); /* existing alias? */ + spin_lock(&inode->i_lock); + if (!inode_notify_dead(inode)) + res = __d_find_any_alias(inode); /* existing alias? */ + else + res = ERR_PTR(-ESTALE); + spin_unlock(&inode->i_lock); if (res) goto out; @@ -2253,7 +2326,10 @@ static struct dentry *__d_obtain_alias(struct inode *inode, bool disconnected) security_d_instantiate(new, inode); spin_lock(&inode->i_lock); - res = __d_find_any_alias(inode); /* recheck under lock */ + if (!inode_notify_dead(inode)) + res = __d_find_any_alias(inode); /* recheck under lock */ + else + res = ERR_PTR(-ESTALE); if (likely(!res)) { /* still no alias, attach a disconnected dentry */ unsigned add_flags = d_flags_for_inode(inode); @@ -2754,6 +2830,15 @@ static inline void end_dir_add(struct inode *dir, unsigned int n) static void d_wait_lookup(struct dentry *dentry) { if (likely(d_in_lookup(dentry))) { + /* + * Tell lockdep we will wait for the lookup lock, after + * dropping ->d_lock, but won't actually take it. + */ + spin_release(&dentry->d_lock.dep_map, _THIS_IP_); + lock_map_acquire(&dentry->lookup_map); + lock_map_release(&dentry->lookup_map); + spin_acquire(&dentry->d_lock.dep_map, 0, 1, _THIS_IP_); + dentry->d_flags |= DCACHE_LOOKUP_WAITERS; wait_var_event_spinlock(&dentry->d_flags, !d_in_lookup(dentry), @@ -2761,8 +2846,16 @@ static void d_wait_lookup(struct dentry *dentry) } } -struct dentry *d_alloc_parallel(struct dentry *parent, - const struct qstr *name) +/* What to do when __d_alloc_parallel finds a d_in_lookup dentry */ +enum alloc_para { + ALLOC_PARA_WAIT, + ALLOC_PARA_FAIL, +}; + +static inline +struct dentry *__d_alloc_parallel(struct dentry *parent, + const struct qstr *name, + enum alloc_para how) { unsigned int hash = name->hash; struct hlist_bl_head *b = in_lookup_hash(parent, hash); @@ -2835,6 +2928,12 @@ retry: spin_unlock(&dentry->d_lock); goto retry; } + if (unlikely(how == ALLOC_PARA_FAIL)) { + /* mustn't wait for concurrent lookup to complete */ + spin_unlock(&dentry->d_lock); + dput(new); + return ERR_PTR(-EWOULDBLOCK); + } /* * somebody is likely to be still doing lookup for it; * pin it and wait for them to finish @@ -2862,14 +2961,77 @@ retry: } hlist_bl_add_head(&new->d_in_lookup_hash, b); hlist_bl_unlock(b); + lock_map_acquire_try(&new->lookup_map); return new; mismatch: spin_unlock(&dentry->d_lock); dput(dentry); goto retry; } + +/** + * d_alloc_parallel() - allocate a new dentry and ensure uniqueness + * @parent: dentry of the parent + * @name: name of the dentry within that parent. + * + * A new dentry is allocated and, providing it is unique, added to the + * relevant index. + * If an existing dentry is found with the same parent/name that is + * not d_in_lookup(), then that is returned instead. + * If the existing dentry is d_in_lookup(), d_alloc_parallel() waits for + * that lookup to complete before returning the dentry and then ensures the + * match is still valid. + * Thus if the returned dentry is d_in_lookup() then the caller has + * exclusive access until it completes the lookup. + * If the returned dentry is not d_in_lookup() then a lookup has + * already completed. + * + * The @name must already have ->hash set, as can be achieved + * by e.g. try_lookup_noperm(). + * + * Returns: the dentry, whether found or allocated, or an error %-ENOMEM. + */ +struct dentry *d_alloc_parallel(struct dentry *parent, + const struct qstr *name) +{ + return __d_alloc_parallel(parent, name, ALLOC_PARA_WAIT); +} EXPORT_SYMBOL(d_alloc_parallel); +/** + * d_alloc_trylock() - find or allocate a new dentry + * @parent: dentry of the parent + * @name: name of the dentry within that parent. + * + * A new dentry is allocated and, providing it is unique, added to the + * relevant index. + * If an existing dentry is found with the same parent/name that is + * not d_in_lookup() then that is returned instead. + * If the existing dentry is d_in_lookup(), d_alloc_trylock() + * returns with error %-EWOULDBLOCK. + * Thus if the returned dentry is d_in_lookup() then the caller has + * exclusive access until it completes the lookup. + * If the returned dentry is not d_in_lookup() then a lookup has + * already completed. + * + * The @name need not already have ->hash set. + * + * Returns: the dentry, whether found or allocated, or an error + * %-ENOMEM, %-EWOULDBLOCK, %-EACCES (for a bad name) or + * anything returned by ->d_hash(). + */ +struct dentry *d_alloc_trylock(struct dentry *parent, + struct qstr *name) +{ + struct dentry *de; + + de = try_lookup_noperm(name, parent); + if (!de) + de = __d_alloc_parallel(parent, name, ALLOC_PARA_FAIL); + return de; +} +EXPORT_SYMBOL(d_alloc_trylock); + /* * Move dentry from in-lookup state to busy-negative one. * @@ -2898,6 +3060,7 @@ static void __d_lookup_unhash(struct dentry *dentry) b = in_lookup_hash(dentry->d_parent, dentry->d_name.hash); hlist_bl_lock(b); dentry->d_flags &= ~DCACHE_PAR_LOOKUP; + lock_map_release(&dentry->lookup_map); __hlist_bl_del(&dentry->d_in_lookup_hash); hlist_bl_unlock(b); dentry->waiters = NULL; @@ -2935,15 +3098,10 @@ static inline void __d_add(struct dentry *dentry, struct inode *inode, } if (unlikely(ops)) d_set_d_op(dentry, ops); - if (inode) { - unsigned add_flags = d_flags_for_inode(inode); - hlist_add_head(&dentry->d_alias, &inode->i_dentry); - raw_write_seqcount_begin(&dentry->d_seq); - __d_set_inode_and_type(dentry, inode, add_flags); - raw_write_seqcount_end(&dentry->d_seq); - fsnotify_update_flags(dentry); - } - __d_rehash(dentry); + if (inode) + __d_instantiate(dentry, inode); + if (d_unhashed(dentry)) + __d_rehash(dentry); if (dir) { end_dir_add(dir, n); __d_wake_in_lookup_waiters(dentry); @@ -3245,7 +3403,7 @@ struct dentry *d_splice_alias_ops(struct inode *inode, struct dentry *dentry, if (IS_ERR(inode)) return ERR_CAST(inode); - BUG_ON(!d_unhashed(dentry)); + BUG_ON(d_really_is_positive(dentry)); if (!inode) goto out; @@ -3301,6 +3459,8 @@ out: * @inode: the inode which may have a disconnected dentry * @dentry: a negative dentry which we want to point to the inode. * + * @dentry must be negative and may be in-lookup or unhashed or hashed. + * * If inode is a directory and has an IS_ROOT alias, then d_move that in * place of the given dentry and return it, else simply d_add the inode * to the dentry and return NULL. @@ -3308,16 +3468,14 @@ out: * If a non-IS_ROOT directory is found, the filesystem is corrupt, and * we should error out: directories can't have multiple aliases. * - * This is needed in the lookup routine of any filesystem that is exportable - * (via knfsd) so that we can build dcache paths to directories effectively. + * This should be used to return the result of ->lookup() and to + * instantiate the result of ->mkdir(), is often useful for + * ->atomic_open, and may be used to instantiate other objects. * * If a dentry was found and moved, then it is returned. Otherwise NULL - * is returned. This matches the expected return value of ->lookup. + * is returned. This matches the expected return value of ->lookup and + * ->mkdir. * - * Cluster filesystems may call this function with a negative, hashed dentry. - * In that case, we know that the inode will be a regular file, and also this - * will only occur during atomic_open. So we need to check for the dentry - * being already hashed only in the final case. */ struct dentry *d_splice_alias(struct inode *inode, struct dentry *dentry) { diff --git a/fs/debugfs/inode.c b/fs/debugfs/inode.c index a4d08bd3743b..b4915551ad31 100644 --- a/fs/debugfs/inode.c +++ b/fs/debugfs/inode.c @@ -42,7 +42,7 @@ static bool debugfs_enabled __ro_after_init = IS_ENABLED(CONFIG_DEBUG_FS_ALLOW_A * so that we can use the file mode as part of a heuristic to determine whether * to lock down individual files. */ -static int debugfs_setattr(struct mnt_idmap *idmap, +static int debugfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { int ret; diff --git a/fs/devpts/inode.c b/fs/devpts/inode.c index 9844dcf354ee..bd1e8eb26edb 100644 --- a/fs/devpts/inode.c +++ b/fs/devpts/inode.c @@ -249,6 +249,8 @@ static int devpts_parse_param(struct fs_context *fc, struct fs_parameter *param) case Opt_max: if (result.uint_32 > NR_UNIX98_PTY_MAX) return invalf(fc, "max out of range"); + if (result.uint_32 == 0) + return invalf(fc, "max must be greater than 0"); opts->max = result.uint_32; break; } diff --git a/fs/ecryptfs/inode.c b/fs/ecryptfs/inode.c index 525297c7ebd8..48e520960d66 100644 --- a/fs/ecryptfs/inode.c +++ b/fs/ecryptfs/inode.c @@ -266,7 +266,7 @@ out: * Returns zero on success; non-zero on error condition */ static int -ecryptfs_create(struct mnt_idmap *idmap, +ecryptfs_create(const struct mnt_idmap *idmap, struct inode *directory_inode, struct dentry *ecryptfs_dentry, umode_t mode) { @@ -462,7 +462,7 @@ static int ecryptfs_unlink(struct inode *dir, struct dentry *dentry) return ecryptfs_do_unlink(dir, dentry, d_inode(dentry)); } -static int ecryptfs_symlink(struct mnt_idmap *idmap, +static int ecryptfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { @@ -503,7 +503,7 @@ out_lock: return rc; } -static struct dentry *ecryptfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ecryptfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int rc; @@ -562,7 +562,7 @@ static int ecryptfs_rmdir(struct inode *dir, struct dentry *dentry) } static int -ecryptfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +ecryptfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { int rc; @@ -590,7 +590,7 @@ out: } static int -ecryptfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +ecryptfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -849,7 +849,7 @@ int ecryptfs_truncate(struct dentry *dentry, loff_t new_length) } static int -ecryptfs_permission(struct mnt_idmap *idmap, struct inode *inode, +ecryptfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return inode_permission(&nop_mnt_idmap, @@ -869,7 +869,7 @@ ecryptfs_permission(struct mnt_idmap *idmap, struct inode *inode, * All other metadata changes will be passed right to the lower filesystem, * and we will just update our inode to look like the lower. */ -static int ecryptfs_setattr(struct mnt_idmap *idmap, +static int ecryptfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { struct inode *inode = d_inode(dentry); @@ -939,7 +939,7 @@ out: return rc; } -static int ecryptfs_getattr_link(struct mnt_idmap *idmap, +static int ecryptfs_getattr_link(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -965,7 +965,7 @@ static int ecryptfs_getattr_link(struct mnt_idmap *idmap, return rc; } -static int ecryptfs_getattr(struct mnt_idmap *idmap, +static int ecryptfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1078,7 +1078,7 @@ static int ecryptfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return vfs_fileattr_get(ecryptfs_dentry_to_lower(dentry), fa); } -static int ecryptfs_fileattr_set(struct mnt_idmap *idmap, +static int ecryptfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct dentry *lower_dentry = ecryptfs_dentry_to_lower(dentry); @@ -1090,14 +1090,14 @@ static int ecryptfs_fileattr_set(struct mnt_idmap *idmap, return rc; } -static struct posix_acl *ecryptfs_get_acl(struct mnt_idmap *idmap, +static struct posix_acl *ecryptfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { return vfs_get_acl(idmap, ecryptfs_dentry_to_lower(dentry), posix_acl_xattr_name(type)); } -static int ecryptfs_set_acl(struct mnt_idmap *idmap, +static int ecryptfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { @@ -1158,7 +1158,7 @@ static int ecryptfs_xattr_get(const struct xattr_handler *handler, } static int ecryptfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/efivarfs/inode.c b/fs/efivarfs/inode.c index f0d009555fc6..07602cd5d33c 100644 --- a/fs/efivarfs/inode.c +++ b/fs/efivarfs/inode.c @@ -74,7 +74,7 @@ static bool efivarfs_valid_name(const char *str, int len) return uuid_is_valid(s); } -static int efivarfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int efivarfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode = NULL; @@ -150,7 +150,7 @@ efivarfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) } static int -efivarfs_fileattr_set(struct mnt_idmap *idmap, +efivarfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { unsigned int i_flags = 0; @@ -170,7 +170,7 @@ efivarfs_fileattr_set(struct mnt_idmap *idmap, } /* copy of simple_setattr except that it doesn't do i_size updates */ -static int efivarfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int efivarfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/erofs/inode.c b/fs/erofs/inode.c index 45afe5c50de8..26ea3790ff21 100644 --- a/fs/erofs/inode.c +++ b/fs/erofs/inode.c @@ -311,7 +311,7 @@ struct inode *erofs_iget(struct super_block *sb, erofs_nid_t nid) return inode; } -int erofs_getattr(struct mnt_idmap *idmap, const struct path *path, +int erofs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/erofs/internal.h b/fs/erofs/internal.h index 12e3a5b80a5a..ab817091bd29 100644 --- a/fs/erofs/internal.h +++ b/fs/erofs/internal.h @@ -417,7 +417,7 @@ void erofs_onlinefolio_init(struct folio *folio); void erofs_onlinefolio_split(struct folio *folio); void erofs_onlinefolio_end(struct folio *folio, int err, bool dirty); struct inode *erofs_iget(struct super_block *sb, erofs_nid_t nid); -int erofs_getattr(struct mnt_idmap *idmap, const struct path *path, +int erofs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); int erofs_namei(struct inode *dir, const struct qstr *name, diff --git a/fs/eventfd.c b/fs/eventfd.c index 9d33a02757d5..52426795752e 100644 --- a/fs/eventfd.c +++ b/fs/eventfd.c @@ -403,8 +403,8 @@ static int do_eventfd(unsigned int count, int flags) FD_PREPARE(fdf, flags, anon_inode_getfile_fmode("[eventfd]", &eventfd_fops, ctx, flags, FMODE_NOWAIT)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; ctx->id = ida_alloc(&eventfd_ida, GFP_KERNEL); retain_and_null_ptr(ctx); diff --git a/fs/eventpoll.c b/fs/eventpoll.c index e0c4bf88a838..f48b829a710f 100644 --- a/fs/eventpoll.c +++ b/fs/eventpoll.c @@ -2514,11 +2514,11 @@ static int do_epoll_create(int flags) FD_PREPARE(fdf, O_RDWR | (flags & O_CLOEXEC), anon_inode_getfile("[eventpoll]", &eventpoll_fops, ep, O_RDWR | (flags & O_CLOEXEC))); - if (fdf.err) { + if (fdf->fd < 0) { ep_clear_and_put(ep); - return fdf.err; + return fdf->fd; } - ep->file = fd_prepare_file(fdf); + ep->file = fdf->file; return fd_publish(fdf); } diff --git a/fs/exec.c b/fs/exec.c index a5269b5e00df..33a1e4689e49 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -1136,6 +1136,7 @@ static void posixtimer_exec(struct task_struct *me) int begin_new_exec(struct linux_binprm * bprm) { struct task_struct *me = current; + struct files_struct *files = NULL; int retval; /* A pending PT_INTERP substitution this format cannot consume. */ @@ -1160,6 +1161,13 @@ int begin_new_exec(struct linux_binprm * bprm) */ bprm->point_of_no_return = true; + /* + * Cancel any io_uring activity across execve. This runs task work + * that may still create an io-wq worker, so do it while de_thread() + * can still zap it. + */ + io_uring_task_cancel(); + /* Make this the only thread in the thread group */ retval = de_thread(me); if (retval) @@ -1176,15 +1184,13 @@ int begin_new_exec(struct linux_binprm * bprm) /* see the comment in check_unsafe_exec() */ current->fs->in_exec = 0; - /* - * Cancel any io_uring activity across execve - */ - io_uring_task_cancel(); /* Ensure the files table is not shared. */ - retval = unshare_files(); + retval = unshare_fd(CLONE_FILES, &files); if (retval) goto out; + if (files) + switch_files_struct(me, files); /* * We have to apply CLOEXEC before we change whether the process is @@ -1192,13 +1198,13 @@ int begin_new_exec(struct linux_binprm * bprm) * trying to access the should-be-closed file descriptors of a process * undergoing exec(2). * - * This can block on filesystem ->flush() handlers, including waiting - * for FUSE daemons, so do it before exec_mmap takes the - * exec_update_lock. + * This can block on filesystem ->flush() and ->release() handlers, + * including waiting for FUSE daemons, so do it before exec_mmap + * takes the exec_update_lock. * This must happen after the point of no return, and after unsharing * the FD table. */ - do_close_on_exec(me->files); + close_cloexec_files(me->files); /* * Must be called _before_ exec_mmap() as bprm->mm is @@ -1359,7 +1365,7 @@ EXPORT_SYMBOL(begin_new_exec); void would_dump(struct linux_binprm *bprm, struct file *file) { struct inode *inode = file_inode(file); - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); if (inode_permission(idmap, inode, MAY_READ) < 0) { struct user_namespace *old, *user_ns; bprm->interp_flags |= BINPRM_FLAGS_ENFORCE_NONDUMP; @@ -1643,7 +1649,7 @@ static void check_unsafe_exec(struct linux_binprm *bprm) static void bprm_fill_uid(struct linux_binprm *bprm, struct file *file) { /* Handle suid and sgid on files */ - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode = file_inode(file); unsigned int mode; vfsuid_t vfsuid; diff --git a/fs/exfat/exfat_fs.h b/fs/exfat/exfat_fs.h index 41a2c7dfc479..5f258e96fce9 100644 --- a/fs/exfat/exfat_fs.h +++ b/fs/exfat/exfat_fs.h @@ -556,9 +556,9 @@ int exfat_trim_fs(struct inode *inode, struct fstrim_range *range); /* file.c */ extern const struct file_operations exfat_file_operations; int __exfat_truncate(struct inode *inode); -int exfat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int exfat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int exfat_getattr(struct mnt_idmap *idmap, const struct path *path, +int exfat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags); struct file_kattr; diff --git a/fs/exfat/file.c b/fs/exfat/file.c index a2a9ee1a2004..3867e78c2312 100644 --- a/fs/exfat/file.c +++ b/fs/exfat/file.c @@ -143,7 +143,7 @@ error: return err; } -static bool exfat_allow_set_time(struct mnt_idmap *idmap, +static bool exfat_allow_set_time(const struct mnt_idmap *idmap, struct exfat_sb_info *sbi, struct inode *inode) { mode_t allow_utime = sbi->options.allow_utime; @@ -319,7 +319,7 @@ write_size: mutex_unlock(&sbi->s_lock); } -int exfat_getattr(struct mnt_idmap *idmap, const struct path *path, +int exfat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags) { @@ -347,7 +347,7 @@ int exfat_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int exfat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int exfat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct exfat_sb_info *sbi = EXFAT_SB(dentry->d_sb); diff --git a/fs/exfat/misc.c b/fs/exfat/misc.c index 6f11a96a4ffa..dfd0bbf31c94 100644 --- a/fs/exfat/misc.c +++ b/fs/exfat/misc.c @@ -187,7 +187,7 @@ int exfat_update_bhs(struct buffer_head **bhs, int nr_bhs, int sync) for (i = 0; i < nr_bhs && sync; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; diff --git a/fs/exfat/namei.c b/fs/exfat/namei.c index 3c5746fc57d9..d116c89d724e 100644 --- a/fs/exfat/namei.c +++ b/fs/exfat/namei.c @@ -552,7 +552,7 @@ out: return ret; } -static int exfat_create(struct mnt_idmap *idmap, struct inode *dir, +static int exfat_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -826,7 +826,7 @@ unlock: return err; } -static struct dentry *exfat_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *exfat_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1264,7 +1264,7 @@ out: return ret; } -static int exfat_rename(struct mnt_idmap *idmap, +static int exfat_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/ext2/acl.c b/fs/ext2/acl.c index 7e54c31589c7..b2746657fc53 100644 --- a/fs/ext2/acl.c +++ b/fs/ext2/acl.c @@ -219,7 +219,7 @@ __ext2_set_acl(struct inode *inode, struct posix_acl *acl, int type) * inode->i_mutex: down */ int -ext2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ext2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; diff --git a/fs/ext2/acl.h b/fs/ext2/acl.h index 4a8443a2b8ec..e68bc3545608 100644 --- a/fs/ext2/acl.h +++ b/fs/ext2/acl.h @@ -56,7 +56,7 @@ static inline int ext2_acl_count(size_t size) /* acl.c */ extern struct posix_acl *ext2_get_acl(struct inode *inode, int type, bool rcu); -extern int ext2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int ext2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ext2_init_acl (struct inode *, struct inode *); diff --git a/fs/ext2/ext2.h b/fs/ext2/ext2.h index 7aeb7cfb0ceb..7bdada93dd06 100644 --- a/fs/ext2/ext2.h +++ b/fs/ext2/ext2.h @@ -741,8 +741,8 @@ extern int ext2_sync_inode_metadata(struct inode *, struct writeback_control *); extern void ext2_evict_inode(struct inode *); void ext2_write_failed(struct address_space *mapping, loff_t to); extern int ext2_get_block(struct inode *, sector_t, struct buffer_head *, int); -extern int ext2_setattr (struct mnt_idmap *, struct dentry *, struct iattr *); -extern int ext2_getattr (struct mnt_idmap *, const struct path *, +extern int ext2_setattr (const struct mnt_idmap *, struct dentry *, struct iattr *); +extern int ext2_getattr (const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext2_set_inode_flags(struct inode *inode); extern int ext2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, @@ -750,7 +750,7 @@ extern int ext2_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, /* ioctl.c */ extern int ext2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -extern int ext2_fileattr_set(struct mnt_idmap *idmap, +extern int ext2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); extern long ext2_ioctl(struct file *, unsigned int, unsigned long); extern long ext2_compat_ioctl(struct file *, unsigned int, unsigned long); diff --git a/fs/ext2/inode.c b/fs/ext2/inode.c index 1a1ea1fd485b..12ac4cfe1500 100644 --- a/fs/ext2/inode.c +++ b/fs/ext2/inode.c @@ -1598,7 +1598,7 @@ out: return err; } -int ext2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ext2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -1624,7 +1624,7 @@ int ext2_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int ext2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ext2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ext2/ioctl.c b/fs/ext2/ioctl.c index c3fea55b8efa..f2218455fa47 100644 --- a/fs/ext2/ioctl.c +++ b/fs/ext2/ioctl.c @@ -27,7 +27,7 @@ int ext2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ext2_fileattr_set(struct mnt_idmap *idmap, +int ext2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ext2/namei.c b/fs/ext2/namei.c index 8666233ec63b..bfb6a463a95e 100644 --- a/fs/ext2/namei.c +++ b/fs/ext2/namei.c @@ -97,7 +97,7 @@ struct dentry *ext2_get_parent(struct dentry *child) * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ext2_create (struct mnt_idmap * idmap, +static int ext2_create (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { @@ -117,7 +117,7 @@ static int ext2_create (struct mnt_idmap * idmap, return ext2_add_nondir(dentry, inode); } -static int ext2_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ext2_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = ext2_new_inode(dir, mode, NULL); @@ -131,7 +131,7 @@ static int ext2_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int ext2_mknod (struct mnt_idmap * idmap, struct inode * dir, +static int ext2_mknod (const struct mnt_idmap * idmap, struct inode * dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode * inode; @@ -152,7 +152,7 @@ static int ext2_mknod (struct mnt_idmap * idmap, struct inode * dir, return err; } -static int ext2_symlink (struct mnt_idmap * idmap, struct inode * dir, +static int ext2_symlink (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, const char * symname) { struct super_block * sb = dir->i_sb; @@ -223,7 +223,7 @@ static int ext2_link (struct dentry * old_dentry, struct inode * dir, return err; } -static struct dentry *ext2_mkdir(struct mnt_idmap * idmap, +static struct dentry *ext2_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { @@ -316,7 +316,7 @@ static int ext2_rmdir (struct inode * dir, struct dentry *dentry) return err; } -static int ext2_rename (struct mnt_idmap * idmap, +static int ext2_rename (const struct mnt_idmap * idmap, struct inode * old_dir, struct dentry * old_dentry, struct inode * new_dir, struct dentry * new_dentry, unsigned int flags) diff --git a/fs/ext2/xattr.c b/fs/ext2/xattr.c index 9b68c490ab26..8f608930a48c 100644 --- a/fs/ext2/xattr.c +++ b/fs/ext2/xattr.c @@ -769,7 +769,7 @@ ext2_xattr_set2(struct inode *inode, struct buffer_head *old_bh, if (IS_SYNC(inode)) { sync_dirty_buffer(new_bh); error = -EIO; - if (buffer_req(new_bh) && !buffer_uptodate(new_bh)) + if (buffer_write_io_error(new_bh)) goto cleanup; } } diff --git a/fs/ext2/xattr_security.c b/fs/ext2/xattr_security.c index db47b8ab153e..ade074354258 100644 --- a/fs/ext2/xattr_security.c +++ b/fs/ext2/xattr_security.c @@ -19,7 +19,7 @@ ext2_xattr_security_get(const struct xattr_handler *handler, static int ext2_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext2/xattr_trusted.c b/fs/ext2/xattr_trusted.c index 995f931228ce..0f12d634d6d0 100644 --- a/fs/ext2/xattr_trusted.c +++ b/fs/ext2/xattr_trusted.c @@ -26,7 +26,7 @@ ext2_xattr_trusted_get(const struct xattr_handler *handler, static int ext2_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext2/xattr_user.c b/fs/ext2/xattr_user.c index dd1507231081..48002c033e9c 100644 --- a/fs/ext2/xattr_user.c +++ b/fs/ext2/xattr_user.c @@ -30,7 +30,7 @@ ext2_xattr_user_get(const struct xattr_handler *handler, static int ext2_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/acl.c b/fs/ext4/acl.c index 3bffe862f954..59fac55a2426 100644 --- a/fs/ext4/acl.c +++ b/fs/ext4/acl.c @@ -225,7 +225,7 @@ __ext4_set_acl(handle_t *handle, struct inode *inode, int type, } int -ext4_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ext4_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { handle_t *handle; diff --git a/fs/ext4/acl.h b/fs/ext4/acl.h index 0c5a79c3b5d4..a14838c5bc42 100644 --- a/fs/ext4/acl.h +++ b/fs/ext4/acl.h @@ -56,7 +56,7 @@ static inline int ext4_acl_count(size_t size) /* acl.c */ struct posix_acl *ext4_get_acl(struct inode *inode, int type, bool rcu); -int ext4_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ext4_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ext4_init_acl(handle_t *, struct inode *, struct inode *); diff --git a/fs/ext4/ext4.h b/fs/ext4/ext4.h index 724a27e8be61..cbc59d03ca81 100644 --- a/fs/ext4/ext4.h +++ b/fs/ext4/ext4.h @@ -3044,7 +3044,7 @@ extern int ext4fs_dirhash(const struct inode *dir, const char *name, int len, /* ialloc.c */ extern int ext4_mark_inode_used(struct super_block *sb, int ino); -extern struct inode *__ext4_new_inode(struct mnt_idmap *, handle_t *, +extern struct inode *__ext4_new_inode(const struct mnt_idmap *, handle_t *, struct inode *, umode_t, const struct qstr *qstr, __u32 goal, uid_t *owner, __u32 i_flags, @@ -3179,14 +3179,14 @@ extern struct inode *__ext4_iget(struct super_block *sb, unsigned long ino, extern int ext4_write_inode(struct inode *, struct writeback_control *); extern int ext4_sync_inode_metadata(struct inode *, struct writeback_control *); -extern int ext4_setattr(struct mnt_idmap *, struct dentry *, +extern int ext4_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern u32 ext4_dio_alignment(struct inode *inode); -extern int ext4_getattr(struct mnt_idmap *, const struct path *, +extern int ext4_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext4_evict_inode(struct inode *); extern void ext4_clear_inode(struct inode *); -extern int ext4_file_getattr(struct mnt_idmap *, const struct path *, +extern int ext4_file_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); extern void ext4_dirty_inode(struct inode *, int); extern int ext4_change_inode_journal_flag(struct inode *, int); @@ -3246,7 +3246,7 @@ extern int ext4_ind_remove_space(handle_t *handle, struct inode *inode, /* ioctl.c */ extern long ext4_ioctl(struct file *, unsigned int, unsigned long); extern long ext4_compat_ioctl(struct file *, unsigned int, unsigned long); -int ext4_fileattr_set(struct mnt_idmap *idmap, +int ext4_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); int ext4_fileattr_get(struct dentry *dentry, struct file_kattr *fa); extern void ext4_reset_inode_seed(struct inode *inode); diff --git a/fs/ext4/ext4_jbd2.c b/fs/ext4/ext4_jbd2.c index 53ddedb52a6f..c241f50b97bc 100644 --- a/fs/ext4/ext4_jbd2.c +++ b/fs/ext4/ext4_jbd2.c @@ -421,7 +421,7 @@ int __ext4_handle_dirty_metadata(const char *where, unsigned int line, } if (inode && inode_needs_sync(inode)) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ext4_error_inode_err(inode, where, line, bh->b_blocknr, EIO, "IO error syncing itable block"); diff --git a/fs/ext4/ialloc.c b/fs/ext4/ialloc.c index a5831fc536db..529623103ae7 100644 --- a/fs/ext4/ialloc.c +++ b/fs/ext4/ialloc.c @@ -930,7 +930,7 @@ static int ext4_xattr_credits_for_new_inode(struct inode *dir, mode_t mode, * For other inodes, search forward from the parent directory's block * group to find a free inode. */ -struct inode *__ext4_new_inode(struct mnt_idmap *idmap, +struct inode *__ext4_new_inode(const struct mnt_idmap *idmap, handle_t *handle, struct inode *dir, umode_t mode, const struct qstr *qstr, __u32 goal, uid_t *owner, __u32 i_flags, diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c index 26f0f9714f03..cb68bf50a3d6 100644 --- a/fs/ext4/inode.c +++ b/fs/ext4/inode.c @@ -6006,7 +6006,7 @@ static void ext4_wait_for_tail_page_commit(struct inode *inode) * * Called with inode->i_rwsem down. */ -int ext4_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ext4_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -6263,7 +6263,7 @@ u32 ext4_dio_alignment(struct inode *inode) return 1; /* use the iomap defaults */ } -int ext4_getattr(struct mnt_idmap *idmap, const struct path *path, +int ext4_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -6332,7 +6332,7 @@ int ext4_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int ext4_file_getattr(struct mnt_idmap *idmap, +int ext4_file_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/ext4/ioctl.c b/fs/ext4/ioctl.c index c8387e6a2c6e..0a54b00e5be5 100644 --- a/fs/ext4/ioctl.c +++ b/fs/ext4/ioctl.c @@ -373,7 +373,7 @@ void ext4_reset_inode_seed(struct inode *inode) * */ static long swap_inode_boot_loader(struct super_block *sb, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode) { handle_t *handle; @@ -1008,7 +1008,7 @@ int ext4_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ext4_fileattr_set(struct mnt_idmap *idmap, +int ext4_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -1539,7 +1539,7 @@ static long __ext4_ioctl(struct file *filp, unsigned int cmd, unsigned long arg) { struct inode *inode = file_inode(filp); struct super_block *sb = inode->i_sb; - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); ext4_debug("cmd = %u, arg = %lu\n", cmd, arg); diff --git a/fs/ext4/mmp.c b/fs/ext4/mmp.c index 7ce361484b38..4b18ddef468d 100644 --- a/fs/ext4/mmp.c +++ b/fs/ext4/mmp.c @@ -49,7 +49,7 @@ static int write_mmp_block_thawed(struct super_block *sb, bh_submit(bh, REQ_OP_WRITE | REQ_SYNC | REQ_META | REQ_PRIO, bh_end_write); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) return -EIO; return 0; } diff --git a/fs/ext4/namei.c b/fs/ext4/namei.c index a6386c1d237f..6e0630a49e48 100644 --- a/fs/ext4/namei.c +++ b/fs/ext4/namei.c @@ -2812,7 +2812,7 @@ static int ext4_add_nondir(handle_t *handle, * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ext4_create(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { handle_t *handle; @@ -2847,7 +2847,7 @@ retry: return err; } -static int ext4_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { handle_t *handle; @@ -2881,7 +2881,7 @@ retry: return err; } -static int ext4_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { handle_t *handle; @@ -2994,7 +2994,7 @@ out: return err; } -static struct dentry *ext4_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ext4_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { handle_t *handle; @@ -3360,7 +3360,7 @@ out: return err; } -static int ext4_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ext4_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { handle_t *handle; @@ -3753,7 +3753,7 @@ static void ext4_update_dir_count(handle_t *handle, struct ext4_renament *ent) } } -static struct inode *ext4_whiteout_for_rename(struct mnt_idmap *idmap, +static struct inode *ext4_whiteout_for_rename(const struct mnt_idmap *idmap, struct ext4_renament *ent, int credits, handle_t **h) { @@ -3796,7 +3796,7 @@ retry: * while new_{dentry,inode) refers to the destination dentry/inode * This comes from rename(const char *oldpath, const char *newpath) */ -static int ext4_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ext4_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -4191,7 +4191,7 @@ end_rename: return retval; } -static int ext4_rename2(struct mnt_idmap *idmap, +static int ext4_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/ext4/symlink.c b/fs/ext4/symlink.c index b612262719ed..e680d1e45b47 100644 --- a/fs/ext4/symlink.c +++ b/fs/ext4/symlink.c @@ -55,7 +55,7 @@ static const char *ext4_encrypted_get_link(struct dentry *dentry, return paddr; } -static int ext4_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int ext4_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) diff --git a/fs/ext4/xattr_hurd.c b/fs/ext4/xattr_hurd.c index 8a5842e4cd95..a3ecbff72b10 100644 --- a/fs/ext4/xattr_hurd.c +++ b/fs/ext4/xattr_hurd.c @@ -32,7 +32,7 @@ ext4_xattr_hurd_get(const struct xattr_handler *handler, static int ext4_xattr_hurd_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_security.c b/fs/ext4/xattr_security.c index 776cf11d24ca..af5b8a93fed1 100644 --- a/fs/ext4/xattr_security.c +++ b/fs/ext4/xattr_security.c @@ -23,7 +23,7 @@ ext4_xattr_security_get(const struct xattr_handler *handler, static int ext4_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_trusted.c b/fs/ext4/xattr_trusted.c index 9811eb0ab276..458e1982ef83 100644 --- a/fs/ext4/xattr_trusted.c +++ b/fs/ext4/xattr_trusted.c @@ -30,7 +30,7 @@ ext4_xattr_trusted_get(const struct xattr_handler *handler, static int ext4_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/ext4/xattr_user.c b/fs/ext4/xattr_user.c index 4b70bf4e7626..ad35215f6610 100644 --- a/fs/ext4/xattr_user.c +++ b/fs/ext4/xattr_user.c @@ -31,7 +31,7 @@ ext4_xattr_user_get(const struct xattr_handler *handler, static int ext4_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/f2fs/acl.c b/fs/f2fs/acl.c index 34c9aa279040..a7485bc38252 100644 --- a/fs/f2fs/acl.c +++ b/fs/f2fs/acl.c @@ -219,7 +219,7 @@ struct posix_acl *f2fs_get_acl(struct inode *inode, int type, bool rcu) return __f2fs_get_acl(inode, type, NULL); } -static int f2fs_acl_update_mode(struct mnt_idmap *idmap, +static int f2fs_acl_update_mode(const struct mnt_idmap *idmap, struct inode *inode, umode_t *mode_p, struct posix_acl **acl) { @@ -240,7 +240,7 @@ static int f2fs_acl_update_mode(struct mnt_idmap *idmap, return 0; } -static int __f2fs_set_acl(struct mnt_idmap *idmap, +static int __f2fs_set_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, struct posix_acl *acl, struct f2fs_cached_block *ientry) { @@ -289,7 +289,7 @@ static int __f2fs_set_acl(struct mnt_idmap *idmap, return error; } -int f2fs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/f2fs/acl.h b/fs/f2fs/acl.h index 0f639367a0ab..b1085efcc05e 100644 --- a/fs/f2fs/acl.h +++ b/fs/f2fs/acl.h @@ -34,7 +34,7 @@ struct f2fs_acl_header { #ifdef CONFIG_F2FS_FS_POSIX_ACL struct posix_acl *f2fs_get_acl(struct inode *, int, bool); -int f2fs_set_acl(struct mnt_idmap *, struct dentry *, +int f2fs_set_acl(const struct mnt_idmap *, struct dentry *, struct posix_acl *, int); int f2fs_init_acl(struct inode *inode, struct inode *dir, struct f2fs_cached_block *ientry, diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 089a62c054ea..c1f1e339f085 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -3972,9 +3972,9 @@ int f2fs_sync_file(struct file *file, loff_t start, loff_t end, int datasync); int f2fs_do_truncate_blocks(struct inode *inode, u64 from, bool lock); int f2fs_truncate_blocks(struct inode *inode, u64 from, bool lock); int f2fs_truncate(struct inode *inode); -int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, +int f2fs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int f2fs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int f2fs_truncate_hole(struct inode *inode, pgoff_t pg_start, pgoff_t pg_end); void f2fs_truncate_data_blocks_range(struct dnode_of_data *dn, int count); @@ -3982,7 +3982,7 @@ int f2fs_do_shutdown(struct f2fs_sb_info *sbi, unsigned int flag, bool readonly, bool need_lock); int f2fs_precache_extents(struct inode *inode); int f2fs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int f2fs_fileattr_set(struct mnt_idmap *idmap, +int f2fs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long f2fs_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); long f2fs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); @@ -4014,7 +4014,7 @@ void f2fs_destroy_evict_inode_work(void); int f2fs_update_extension_list(struct f2fs_sb_info *sbi, const char *name, bool hot, bool set); struct dentry *f2fs_get_parent(struct dentry *child); -int f2fs_get_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int f2fs_get_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct inode **new_inode); /* diff --git a/fs/f2fs/file.c b/fs/f2fs/file.c index ef4d218e694b..626c6f97b6f8 100644 --- a/fs/f2fs/file.c +++ b/fs/f2fs/file.c @@ -1033,7 +1033,7 @@ static bool f2fs_force_buffered_io(struct inode *inode, int rw) return false; } -int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, +int f2fs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -1097,7 +1097,7 @@ int f2fs_getattr(struct mnt_idmap *idmap, const struct path *path, } #ifdef CONFIG_F2FS_FS_POSIX_ACL -static void __setattr_copy(struct mnt_idmap *idmap, +static void __setattr_copy(const struct mnt_idmap *idmap, struct inode *inode, const struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; @@ -1122,7 +1122,7 @@ static void __setattr_copy(struct mnt_idmap *idmap, #define __setattr_copy setattr_copy #endif -int f2fs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int f2fs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -2364,7 +2364,7 @@ static int f2fs_ioc_getversion(struct file *filp, unsigned long arg) static int f2fs_ioc_start_atomic_write(struct file *filp, bool truncate) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); struct f2fs_inode_info *fi = F2FS_I(inode); struct f2fs_sb_info *sbi = F2FS_I_SB(inode); loff_t isize; @@ -2476,7 +2476,7 @@ out: static int f2fs_ioc_commit_atomic_write(struct file *filp) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); int ret; if (!(filp->f_mode & FMODE_WRITE)) @@ -2511,7 +2511,7 @@ static int f2fs_ioc_commit_atomic_write(struct file *filp) static int f2fs_ioc_abort_atomic_write(struct file *filp) { struct inode *inode = file_inode(filp); - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); int ret; if (!(filp->f_mode & FMODE_WRITE)) @@ -3585,7 +3585,7 @@ int f2fs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int f2fs_fileattr_set(struct mnt_idmap *idmap, +int f2fs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/f2fs/namei.c b/fs/f2fs/namei.c index 38fcea8b72bf..ce5d536d892b 100644 --- a/fs/f2fs/namei.c +++ b/fs/f2fs/namei.c @@ -231,7 +231,7 @@ static void set_file_temperature(struct f2fs_sb_info *sbi, struct inode *inode, file_set_hot(inode); } -static struct inode *f2fs_new_inode(struct mnt_idmap *idmap, +static struct inode *f2fs_new_inode(const struct mnt_idmap *idmap, struct inode *dir, umode_t mode, const char *name) { @@ -365,7 +365,7 @@ fail_drop: return ERR_PTR(err); } -static int f2fs_create(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -663,7 +663,7 @@ static const char *f2fs_get_link(struct dentry *dentry, return link; } -static int f2fs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -752,7 +752,7 @@ free_inode: goto out; } -static struct dentry *f2fs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *f2fs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -811,7 +811,7 @@ static int f2fs_rmdir(struct inode *dir, struct dentry *dentry) return -ENOTEMPTY; } -static int f2fs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -858,7 +858,7 @@ out: return err; } -static int __f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int __f2fs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode, bool is_whiteout, struct inode **new_inode, struct f2fs_filename *fname) { @@ -929,7 +929,7 @@ out: return err; } -static int f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int f2fs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct f2fs_sb_info *sbi = F2FS_I_SB(dir); @@ -945,7 +945,7 @@ static int f2fs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, err); } -static int f2fs_create_whiteout(struct mnt_idmap *idmap, +static int f2fs_create_whiteout(const struct mnt_idmap *idmap, struct inode *dir, struct inode **whiteout, struct f2fs_filename *fname) { @@ -953,14 +953,14 @@ static int f2fs_create_whiteout(struct mnt_idmap *idmap, true, whiteout, fname); } -int f2fs_get_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int f2fs_get_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct inode **new_inode) { return __f2fs_tmpfile(idmap, dir, NULL, S_IFREG, false, new_inode, NULL); } -static int f2fs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int f2fs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1345,7 +1345,7 @@ out: return err; } -static int f2fs_rename2(struct mnt_idmap *idmap, +static int f2fs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1398,7 +1398,7 @@ static const char *f2fs_encrypted_get_link(struct dentry *dentry, return target; } -static int f2fs_encrypted_symlink_getattr(struct mnt_idmap *idmap, +static int f2fs_encrypted_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) diff --git a/fs/f2fs/xattr.c b/fs/f2fs/xattr.c index 4328c9d9de45..0ed879fc3076 100644 --- a/fs/f2fs/xattr.c +++ b/fs/f2fs/xattr.c @@ -67,7 +67,7 @@ static int f2fs_xattr_generic_get(const struct xattr_handler *handler, } static int f2fs_xattr_generic_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -111,7 +111,7 @@ static int f2fs_xattr_advise_get(const struct xattr_handler *handler, } static int f2fs_xattr_advise_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/failfs.c b/fs/failfs.c index 66a36da3d236..437cdc981c2d 100644 --- a/fs/failfs.c +++ b/fs/failfs.c @@ -22,7 +22,7 @@ bool failfs_mnt(const struct vfsmount *mnt) return mnt->mnt_sb == failfs_root_path.mnt->mnt_sb; } -static int failfs_permission(struct mnt_idmap *idmap, struct inode *inode, +static int failfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { return -EOPNOTSUPP; @@ -35,7 +35,7 @@ static struct dentry *failfs_lookup(struct inode *dir, struct dentry *dentry, return ERR_PTR(-EOPNOTSUPP); } -static int failfs_getattr(struct mnt_idmap *idmap, const struct path *path, +static int failfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/fat/fat.h b/fs/fat/fat.h index 61338413d9f3..dbbcfc90a9c2 100644 --- a/fs/fat/fat.h +++ b/fs/fat/fat.h @@ -404,10 +404,10 @@ extern long fat_generic_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); extern const struct file_operations fat_file_operations; extern const struct inode_operations fat_file_inode_operations; -extern int fat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int fat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void fat_truncate_blocks(struct inode *inode, loff_t offset); -extern int fat_getattr(struct mnt_idmap *idmap, +extern int fat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); int fat_fileattr_get(struct dentry *dentry, struct file_kattr *fa); diff --git a/fs/fat/file.c b/fs/fat/file.c index 1c835ca5f21a..2c6aee9f0305 100644 --- a/fs/fat/file.c +++ b/fs/fat/file.c @@ -432,7 +432,7 @@ int fat_fileattr_get(struct dentry *dentry, struct file_kattr *fa) } EXPORT_SYMBOL_GPL(fat_fileattr_get); -int fat_getattr(struct mnt_idmap *idmap, const struct path *path, +int fat_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); @@ -493,7 +493,7 @@ static int fat_sanitize_mode(const struct msdos_sb_info *sbi, return 0; } -static int fat_allow_set_time(struct mnt_idmap *idmap, +static int fat_allow_set_time(const struct mnt_idmap *idmap, struct msdos_sb_info *sbi, struct inode *inode) { umode_t allow_utime = sbi->options.allow_utime; @@ -514,7 +514,7 @@ static int fat_allow_set_time(struct mnt_idmap *idmap, /* valid file mode bits */ #define FAT_VALID_MODE (S_IFREG | S_IFDIR | S_IRWXUGO) -int fat_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fat_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct msdos_sb_info *sbi = MSDOS_SB(dentry->d_sb); diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf1975..0d04228f916e 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -360,7 +360,7 @@ int fat_sync_bhs(struct buffer_head **bhs, int nr_bhs) for (i = 0; i < nr_bhs; i++) { wait_on_buffer(bhs[i]); - if (!err && !buffer_uptodate(bhs[i])) + if (!err && buffer_write_io_error(bhs[i])) err = -EIO; } return err; diff --git a/fs/fat/namei_msdos.c b/fs/fat/namei_msdos.c index d46d1a3851f2..dde4215616f9 100644 --- a/fs/fat/namei_msdos.c +++ b/fs/fat/namei_msdos.c @@ -263,7 +263,7 @@ static int msdos_add_entry(struct inode *dir, const unsigned char *name, } /***** Create a file */ -static int msdos_create(struct mnt_idmap *idmap, struct inode *dir, +static int msdos_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -345,7 +345,7 @@ out: } /***** Make a directory */ -static struct dentry *msdos_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *msdos_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -600,7 +600,7 @@ error_inode: } /***** Rename, a wrapper for rename_same_dir & rename_diff_dir */ -static int msdos_rename(struct mnt_idmap *idmap, +static int msdos_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/fat/namei_vfat.c b/fs/fat/namei_vfat.c index da3e89c0b16a..3dc063ba0a73 100644 --- a/fs/fat/namei_vfat.c +++ b/fs/fat/namei_vfat.c @@ -753,7 +753,7 @@ error: return ERR_PTR(err); } -static int vfat_create(struct mnt_idmap *idmap, struct inode *dir, +static int vfat_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -846,7 +846,7 @@ out: return err; } -static struct dentry *vfat_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *vfat_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1160,7 +1160,7 @@ error_exchange: goto out; } -static int vfat_rename2(struct mnt_idmap *idmap, struct inode *old_dir, +static int vfat_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/fhandle.c b/fs/fhandle.c index f8829231e3d7..2aa55b8a878a 100644 --- a/fs/fhandle.c +++ b/fs/fhandle.c @@ -201,7 +201,7 @@ static int vfs_dentry_acceptable(void *context, struct dentry *dentry) struct handle_to_path_ctx *ctx = context; struct user_namespace *user_ns = current_user_ns(); struct dentry *d, *root = ctx->root.dentry; - struct mnt_idmap *idmap = mnt_idmap(ctx->root.mnt); + const struct mnt_idmap *idmap = mnt_idmap(ctx->root.mnt); int retval = 0; if (!root) diff --git a/fs/file.c b/fs/file.c index 628ca07dc4b1..88b7a5340815 100644 --- a/fs/file.c +++ b/fs/file.c @@ -352,24 +352,78 @@ static inline bool fd_is_open(unsigned int fd, const struct fdtable *fdt) return test_bit(fd, fdt->open_fds); } +/* Bits of [range->from, range->to] that fall into word @i of a bitmap. */ +static unsigned long fd_range_word(struct fd_range *range, unsigned int i) +{ + unsigned int first = i * BITS_PER_LONG; + unsigned int last = first + BITS_PER_LONG - 1; + + if (range->to < first || range->from > last) + return 0; + return GENMASK(min(range->to, last) - first, + max(range->from, first) - first); +} + +/* Bits of word @i that dup_fd() leaves behind and __range_close() closes. */ +static unsigned long dup_fd_dropped_word(struct fdtable *fdt, unsigned int i, + struct fd_range *range) +{ + unsigned long dropped; + + if (!range) + return 0; + dropped = fd_range_word(range, i); + if (range->flags & FD_RANGE_EXCEPT) + dropped = ~dropped; + if (range->flags & FD_RANGE_CLOEXEC_ONLY) + dropped &= fdt->close_on_exec[i]; + return dropped; +} + /* * Note that a sane fdtable size always has to be a multiple of * BITS_PER_LONG, since we have bitmaps that are sized by this. * - * punch_hole is optional - when close_range() is asked to unshare - * and close, we don't need to copy descriptors in that range, so - * a smaller cloned descriptor table might suffice if the last - * currently opened descriptor falls into that range. + * range is optional. When close_range() is asked to unshare dup_fd() + * will leave any files behind according to the range and its flags. The + * cloned table only has to reach the last open descriptor that is + * carried over. */ -static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punch_hole) +static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *range) { unsigned int last = find_last_bit(fdt->open_fds, fdt->max_fds); + unsigned int i; if (last == fdt->max_fds) return NR_OPEN_DEFAULT; - if (punch_hole && punch_hole->to >= last && punch_hole->from <= last) { - last = find_last_bit(fdt->open_fds, punch_hole->from); - if (last == punch_hole->from) + if (!range) + return ALIGN(last + 1, BITS_PER_LONG); + + if (range->flags & FD_RANGE_CLOEXEC_ONLY) { + /* The close-on-exec bits decide what is dropped, walk the words. */ + i = last / BITS_PER_LONG + 1; + while (i--) { + unsigned long dropped = dup_fd_dropped_word(fdt, i, range); + + if (fdt->open_fds[i] & ~dropped) + return (i + 1) * BITS_PER_LONG; + } + return NR_OPEN_DEFAULT; + } + + if (range->flags & FD_RANGE_EXCEPT) { + /* Only the range is carried over. */ + if (last > range->to) { + last = find_last_bit(fdt->open_fds, range->to + 1); + if (last > range->to) + return NR_OPEN_DEFAULT; + } + if (last < range->from) + return NR_OPEN_DEFAULT; + } else if (last >= range->from && last <= range->to) { + /* The last open descriptor goes, the kept ones sit below the range. */ + last = find_last_bit(fdt->open_fds, range->from); + if (last == range->from) return NR_OPEN_DEFAULT; } return ALIGN(last + 1, BITS_PER_LONG); @@ -378,13 +432,14 @@ static unsigned int sane_fdtable_size(struct fdtable *fdt, struct fd_range *punc /* * Allocate a new descriptor table and copy contents from the passed in * instance. Returns a pointer to cloned table on success, ERR_PTR() - * on failure. For 'punch_hole' see sane_fdtable_size(). + * on failure. For 'range' see sane_fdtable_size(). */ -struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_hole) +struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *range) { struct files_struct *newf; struct file **old_fds, **new_fds; - unsigned int open_files, i; + unsigned int open_files, fd; + unsigned long dropped = 0; struct fdtable *old_fdt, *new_fdt; newf = kmem_cache_alloc(files_cachep, GFP_KERNEL); @@ -406,7 +461,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); /* * Check whether we need to allocate a larger fd array and fd set. @@ -430,7 +485,7 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho */ spin_lock(&oldf->file_lock); old_fdt = files_fdtable(oldf); - open_files = sane_fdtable_size(old_fdt, punch_hole); + open_files = sane_fdtable_size(old_fdt, range); } copy_fd_bitmaps(new_fdt, old_fdt, open_files / BITS_PER_LONG); @@ -451,13 +506,18 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho * * Instead of trying to placate userspace racing with itself, we * ref the file if we see it and mark the fd slot as unused otherwise. + * Descriptors dup_fd() is asked to leave behind get the same treatment. */ - for (i = open_files; i != 0; i--) { + for (fd = 0; fd < open_files; fd++) { struct file *f = rcu_dereference_raw(*old_fds++); - if (f) { + + if (!(fd % BITS_PER_LONG)) + dropped = dup_fd_dropped_word(old_fdt, fd / BITS_PER_LONG, range); + if (f && !(dropped & BIT_MASK(fd))) { get_file(f); } else { - __clear_open_fd(open_files - i, new_fdt); + f = NULL; + __clear_open_fd(fd, new_fdt); } rcu_assign_pointer(*new_fds++, f); } @@ -471,7 +531,25 @@ struct files_struct *dup_fd(struct files_struct *oldf, struct fd_range *punch_ho return newf; } -static struct fdtable *close_files(struct files_struct * files) +/* + * Unshare file descriptor table if it is being shared + */ +int unshare_fd(unsigned long unshare_flags, struct files_struct **new_fdp) +{ + struct files_struct *fd = current->files; + + if ((unshare_flags & CLONE_FILES) && + (fd && atomic_read(&fd->count) > 1)) { + fd = dup_fd(fd, NULL); + if (IS_ERR(fd)) + return PTR_ERR(fd); + *new_fdp = fd; + } + + return 0; +} + +static struct fdtable *close_files(struct files_struct *files) { /* * It is safe to dereference the fd table without RCU or @@ -479,24 +557,21 @@ static struct fdtable *close_files(struct files_struct * files) * files structure. */ struct fdtable *fdt = rcu_dereference_raw(files->fdt); - unsigned int i, j = 0; + unsigned int j = fdt->max_fds / BITS_PER_LONG; + + /* Highest fd first, the order the deferred puts ran in. */ + while (j--) { + unsigned long set = fdt->open_fds[j]; - for (;;) { - unsigned long set; - i = j * BITS_PER_LONG; - if (i >= fdt->max_fds) - break; - set = fdt->open_fds[j++]; while (set) { - if (set & 1) { - struct file *file = fdt->fd[i]; - if (file) { - filp_close(file, files); - cond_resched(); - } + unsigned int bit = __fls(set); + struct file *file = fdt->fd[j * BITS_PER_LONG + bit]; + + set ^= 1UL << bit; + if (file) { + filp_close_sync(file, files); + cond_resched(); } - i++; - set >>= 1; } } @@ -515,16 +590,18 @@ void put_files_struct(struct files_struct *files) } } -void exit_files(struct task_struct *tsk) +/* Install @files on @tsk, consuming the reference, and put the old table. */ +void switch_files_struct(struct task_struct *tsk, struct files_struct *files) { - struct files_struct * files = tsk->files; + scoped_guard(task_lock, tsk) + swap(tsk->files, files); + put_files_struct(files); +} - if (files) { - task_lock(tsk); - tsk->files = NULL; - task_unlock(tsk); - put_files_struct(files); - } +void exit_files(struct task_struct *tsk) +{ + if (tsk->files) + switch_files_struct(tsk, NULL); } struct files_struct init_files = { @@ -732,16 +809,13 @@ struct file *file_close_fd_locked(struct files_struct *files, unsigned fd) int close_fd(unsigned fd) { - struct files_struct *files = current->files; struct file *file; - spin_lock(&files->file_lock); - file = file_close_fd_locked(files, fd); - spin_unlock(&files->file_lock); + file = file_close_fd(fd); if (!file) return -EBADF; - return filp_close(file, files); + return filp_close(file, current->files); } EXPORT_SYMBOL(close_fd); @@ -759,38 +833,90 @@ static inline unsigned last_fd(struct fdtable *fdt) } static inline void __range_cloexec(struct files_struct *cur_fds, - unsigned int fd, unsigned int max_fd) + struct fd_range *range) { struct fdtable *fdt; + unsigned int last; - /* make sure we're using the correct maximum value */ spin_lock(&cur_fds->file_lock); fdt = files_fdtable(cur_fds); - max_fd = min(last_fd(fdt), max_fd); - if (fd <= max_fd) - bitmap_set(fdt->close_on_exec, fd, max_fd - fd + 1); + /* make sure we're using the correct maximum value */ + last = last_fd(fdt); + if (!(range->flags & FD_RANGE_EXCEPT)) { + if (range->from <= last) + bitmap_set(fdt->close_on_exec, range->from, + min(range->to, last) - range->from + 1); + } else { + if (range->from > 0) + bitmap_set(fdt->close_on_exec, 0, + min(range->from - 1, last) + 1); + if (range->to < last) + bitmap_set(fdt->close_on_exec, range->to + 1, + last - range->to); + } spin_unlock(&cur_fds->file_lock); } -static inline void __range_close(struct files_struct *files, unsigned int fd, - unsigned int max_fd) +/* Highest open descriptor below @n that @range selects, or @n. */ +static inline unsigned int last_fd_to_close(struct fdtable *fdt, unsigned int n, + struct fd_range *range) +{ + unsigned int i, lo = 0; + + if (!(range->flags & FD_RANGE_EXCEPT)) + lo = range->from / BITS_PER_LONG; + for (i = n ? (n - 1) / BITS_PER_LONG + 1 : 0; i-- > lo; ) { + unsigned long set = fdt->open_fds[i]; + + if (!set) { + /* Skip the empty stretch at find_last_bit() speed. */ + unsigned int last = find_last_bit(fdt->open_fds, i * BITS_PER_LONG); + + if (last >= i * BITS_PER_LONG) + break; + i = last / BITS_PER_LONG + 1; + continue; + } + /* Hop below the kept window in one step. */ + if ((range->flags & FD_RANGE_EXCEPT) && + i * BITS_PER_LONG >= range->from && + i * BITS_PER_LONG + BITS_PER_LONG - 1 <= range->to) { + if (!range->from) + break; + i = (range->from - 1) / BITS_PER_LONG + 1; + continue; + } + set &= dup_fd_dropped_word(fdt, i, range); + if (i == (n - 1) / BITS_PER_LONG) + set &= BITMAP_LAST_WORD_MASK(n); + if (set) + return i * BITS_PER_LONG + __fls(set); + } + return n; +} + +static inline void __range_close(struct files_struct *files, + struct fd_range *range) { struct file *file; struct fdtable *fdt; - unsigned n; + unsigned int fd, n; spin_lock(&files->file_lock); fdt = files_fdtable(files); - n = last_fd(fdt); - max_fd = min(max_fd, n); + if (range->flags & FD_RANGE_EXCEPT) + /* Outside of the range means the whole table. */ + n = fdt->max_fds; + else + n = min(range->to, last_fd(fdt)) + 1; - for (fd = find_next_bit(fdt->open_fds, max_fd + 1, fd); - fd <= max_fd; - fd = find_next_bit(fdt->open_fds, max_fd + 1, fd + 1)) { + /* Highest fd first, see close_files(). */ + while ((fd = last_fd_to_close(fdt, n, range)) < n) { + n = fd; file = file_close_fd_locked(files, fd); if (file) { spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); fdt = files_fdtable(files); @@ -814,21 +940,43 @@ static inline void __range_close(struct files_struct *files, unsigned int fd, * This closes a range of file descriptors. All file descriptors * from @fd up to and including @max_fd are closed. * Currently, errors to close a given file descriptor are ignored. + * + * With CLOSE_RANGE_EXCEPT the range names what to leave alone instead: + * every open file descriptor outside of [@fd, @max_fd] is closed, or + * marked close-on-exec with CLOSE_RANGE_CLOEXEC. + * + * With CLOSE_RANGE_CLOEXEC_ONLY only file descriptors that have + * close-on-exec set are closed. Together with CLOSE_RANGE_EXCEPT the + * range names the close-on-exec file descriptors to keep. To keep none + * of them, name a range that cannot hold an open file descriptor, e.g. + * close_range(~0U, ~0U, ...). */ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, unsigned int, flags) { struct task_struct *me = current; struct files_struct *cur_fds = me->files, *fds = NULL; + struct fd_range range = {fd, max_fd}; + + if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_EXCEPT | CLOSE_RANGE_CLOEXEC_ONLY)) + return -EINVAL; - if (flags & ~(CLOSE_RANGE_UNSHARE | CLOSE_RANGE_CLOEXEC)) + /* One marks close-on-exec, the other closes what is marked. */ + if (hweight32(flags & (CLOSE_RANGE_CLOEXEC | + CLOSE_RANGE_CLOEXEC_ONLY)) > 1) return -EINVAL; if (fd > max_fd) return -EINVAL; + if (flags & CLOSE_RANGE_EXCEPT) + range.flags |= FD_RANGE_EXCEPT; + if (flags & CLOSE_RANGE_CLOEXEC_ONLY) + range.flags |= FD_RANGE_CLOEXEC_ONLY; + if ((flags & CLOSE_RANGE_UNSHARE) && atomic_read(&cur_fds->count) > 1) { - struct fd_range range = {fd, max_fd}, *punch_hole = ⦥ + struct fd_range *drop = ⦥ /* * If the caller requested all fds to be made cloexec we always @@ -836,9 +984,9 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, * use them. */ if (flags & CLOSE_RANGE_CLOEXEC) - punch_hole = NULL; + drop = NULL; - fds = dup_fd(cur_fds, punch_hole); + fds = dup_fd(cur_fds, drop); if (IS_ERR(fds)) return PTR_ERR(fds); /* @@ -848,20 +996,19 @@ SYSCALL_DEFINE3(close_range, unsigned int, fd, unsigned int, max_fd, swap(cur_fds, fds); } - if (flags & CLOSE_RANGE_CLOEXEC) - __range_cloexec(cur_fds, fd, max_fd); - else - __range_close(cur_fds, fd, max_fd); + if (flags & CLOSE_RANGE_CLOEXEC) { + __range_cloexec(cur_fds, &range); + } else if (!fds) { + /* If we unshared, dup_fd() already left behind what we'd close. */ + __range_close(cur_fds, &range); + } if (fds) { /* * We're done closing the files we were supposed to. Time to install * the new file descriptor table and drop the old one. */ - task_lock(me); - me->files = cur_fds; - task_unlock(me); - put_files_struct(fds); + switch_files_struct(me, cur_fds); } return 0; @@ -887,34 +1034,36 @@ struct file *file_close_fd(unsigned int fd) return file; } -void do_close_on_exec(struct files_struct *files) +void close_cloexec_files(struct files_struct *files) { unsigned i; struct fdtable *fdt; /* exec unshares first */ spin_lock(&files->file_lock); - for (i = 0; ; i++) { + fdt = files_fdtable(files); + /* Highest fd first, see close_files(). */ + for (i = fdt->max_fds / BITS_PER_LONG; i--; ) { unsigned long set; - unsigned fd = i * BITS_PER_LONG; + fdt = files_fdtable(files); - if (fd >= fdt->max_fds) - break; set = fdt->close_on_exec[i]; if (!set) continue; fdt->close_on_exec[i] = 0; - for ( ; set ; fd++, set >>= 1) { + while (set) { + unsigned int bit = __fls(set); + unsigned fd = i * BITS_PER_LONG + bit; struct file *file; - if (!(set & 1)) - continue; + + set ^= 1UL << bit; file = fdt->fd[fd]; if (!file) continue; rcu_assign_pointer(fdt->fd[fd], NULL); __put_unused_fd(files, fd); spin_unlock(&files->file_lock); - filp_close(file, files); + filp_close_sync(file, files); cond_resched(); spin_lock(&files->file_lock); } @@ -1391,17 +1540,17 @@ int receive_fd(struct file *file, int __user *ufd, unsigned int o_flags) return error; FD_PREPARE(fdf, o_flags, file); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; get_file(file); if (ufd) { - error = put_user(fd_prepare_fd(fdf), ufd); + error = put_user(fdf->fd, ufd); if (error) return error; } - __receive_sock(fd_prepare_file(fdf)); + __receive_sock(fdf->file); return fd_publish(fdf); } EXPORT_SYMBOL_GPL(receive_fd); @@ -1529,3 +1678,7 @@ int iterate_fd(struct files_struct *files, unsigned n, return res; } EXPORT_SYMBOL(iterate_fd); + +#ifdef CONFIG_FDTABLE_KUNIT_TEST +#include "tests/fdtable_kunit.c" +#endif diff --git a/fs/file_attr.c b/fs/file_attr.c index bfb00d256dd5..81af4364e33a 100644 --- a/fs/file_attr.c +++ b/fs/file_attr.c @@ -265,7 +265,7 @@ static int fileattr_set_prepare(struct inode *inode, * * Return: 0 on success, or a negative error on failure. */ -int vfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -323,7 +323,7 @@ int ioctl_getflags(struct file *file, unsigned int __user *argp) int ioctl_setflags(struct file *file, unsigned int __user *argp) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct dentry *dentry = file->f_path.dentry; struct file_kattr fa = {}; unsigned int flags; @@ -355,7 +355,7 @@ int ioctl_fsgetxattr(struct file *file, void __user *argp) int ioctl_fssetxattr(struct file *file, void __user *argp) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct dentry *dentry = file->f_path.dentry; struct file_kattr fa = {}; int err; diff --git a/fs/fs-writeback.c b/fs/fs-writeback.c index ea3eb40bf828..58a6403780aa 100644 --- a/fs/fs-writeback.c +++ b/fs/fs-writeback.c @@ -233,7 +233,7 @@ void wb_wait_for_completion(struct wb_completion *done) * Parameters for foreign inode detection, see wbc_detach_inode() to see * how they're used. * - * These paramters are inherently heuristical as the detection target + * These parameters are inherently heuristical as the detection target * itself is fuzzy. All we want to do is detaching an inode from the * current owner if it's being written to by some other cgroups too much. * @@ -248,7 +248,7 @@ void wb_wait_for_completion(struct wb_completion *done) * to 16 slots. To avoid tiny writes from swinging the decision too much, * writes smaller than 1/8 of avg size are ignored. */ -#define WB_FRN_TIME_SHIFT 13 /* 1s = 2^13, upto 8 secs w/ 16bit */ +#define WB_FRN_TIME_SHIFT 13 /* 1s = 2^13, up to 8 secs w/ 16bit */ #define WB_FRN_TIME_AVG_SHIFT 3 /* avg = avg * 7/8 + new * 1/8 */ #define WB_FRN_TIME_CUT_DIV 8 /* ignore rounds < avg / 8 */ #define WB_FRN_TIME_PERIOD (2 * (1 << WB_FRN_TIME_SHIFT)) /* 2s */ @@ -259,7 +259,7 @@ void wb_wait_for_completion(struct wb_completion *done) #define WB_FRN_HIST_THR_SLOTS (WB_FRN_HIST_SLOTS / 2) /* if foreign slots >= 8, switch */ #define WB_FRN_HIST_MAX_SLOTS (WB_FRN_HIST_THR_SLOTS / 2 + 1) - /* one round can affect upto 5 slots */ + /* one round can affect up to 5 slots */ #define WB_FRN_MAX_IN_FLIGHT 1024 /* don't queue too many concurrently */ /* @@ -1181,7 +1181,7 @@ int cgroup_writeback_by_id(u64 bdi_id, int memcg_id, struct cgroup_subsys_state *memcg_css; struct bdi_writeback *wb; struct wb_writeback_work *work; - unsigned long dirty; + long dirty; int ret; /* lookup bdi and memcg */ @@ -1210,16 +1210,13 @@ int cgroup_writeback_by_id(u64 bdi_id, int memcg_id, } /* - * The caller is attempting to write out most of - * the currently dirty pages. Let's take the current dirty page - * count and inflate it by 25% which should be large enough to - * flush out most dirty pages while avoiding getting livelocked by - * concurrent dirtiers. - * - * BTW the memcg stats are flushed periodically and this is best-effort - * estimation, so some potential error is ok. + * The caller is attempting to write out most of the target wb's + * currently dirty pages. Size the work from the wb's reclaimable pages + * and inflate the count by 25%, which should be large enough to flush + * out most dirty pages while avoiding getting livelocked by concurrent + * dirtiers. */ - dirty = memcg_page_state(mem_cgroup_from_css(memcg_css), NR_FILE_DIRTY); + dirty = wb_stat_sum(wb, WB_RECLAIMABLE); dirty = dirty * 10 / 8; /* issue the writeback work */ diff --git a/fs/fs_pin.c b/fs/fs_pin.c index 47ef3c71ce90..1a508f2167e0 100644 --- a/fs/fs_pin.c +++ b/fs/fs_pin.c @@ -1,5 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 #include <linux/fs.h> +#include <linux/rculist.h> #include <linux/sched.h> #include <linux/slab.h> #include "internal.h" @@ -22,8 +23,8 @@ void pin_remove(struct fs_pin *pin) void pin_insert(struct fs_pin *pin, struct vfsmount *m) { spin_lock(&pin_lock); - hlist_add_head(&pin->s_list, &m->mnt_sb->s_pins); - hlist_add_head(&pin->m_list, &real_mount(m)->mnt_pins); + hlist_add_head_rcu(&pin->s_list, &m->mnt_sb->s_pins); + hlist_add_head_rcu(&pin->m_list, &real_mount(m)->mnt_pins); spin_unlock(&pin_lock); } @@ -73,7 +74,7 @@ void mnt_pin_kill(struct mount *m) while (1) { struct hlist_node *p; rcu_read_lock(); - p = READ_ONCE(m->mnt_pins.first); + p = rcu_dereference(hlist_first_rcu(&m->mnt_pins)); if (!p) { rcu_read_unlock(); break; @@ -87,7 +88,7 @@ void group_pin_kill(struct hlist_head *p) while (1) { struct hlist_node *q; rcu_read_lock(); - q = READ_ONCE(p->first); + q = rcu_dereference(hlist_first_rcu(p)); if (!q) { rcu_read_unlock(); break; diff --git a/fs/fuse/acl.c b/fs/fuse/acl.c index 31fb50e16aed..738abed9a816 100644 --- a/fs/fuse/acl.c +++ b/fs/fuse/acl.c @@ -62,7 +62,7 @@ static inline bool fuse_no_acl(const struct fuse_conn *fc, return !fc->posix_acl && (i_user_ns(inode) != &init_user_ns); } -struct posix_acl *fuse_get_acl(struct mnt_idmap *idmap, +struct posix_acl *fuse_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -90,7 +90,7 @@ struct posix_acl *fuse_get_inode_acl(struct inode *inode, int type, bool rcu) return __fuse_get_acl(fc, inode, type, rcu); } -int fuse_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index f48fafccce4b..d11e2aabbfd5 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -751,7 +751,7 @@ static u32 fuse_ext_size(size_t size) /* * This adds just a single supplementary group that matches the parent's group. */ -static int get_create_supp_group(struct mnt_idmap *idmap, +static int get_create_supp_group(const struct mnt_idmap *idmap, struct inode *dir, struct fuse_in_arg *ext) { @@ -782,7 +782,7 @@ static int get_create_supp_group(struct mnt_idmap *idmap, return 0; } -static int get_create_ext(struct mnt_idmap *idmap, +static int get_create_ext(const struct mnt_idmap *idmap, struct fuse_args *args, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -820,7 +820,7 @@ static void free_ext_value(struct fuse_args *args) * If the filesystem doesn't support this, then fall back to separate * 'mknod' + 'open' requests. */ -static int fuse_create_open(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_create_open(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, struct file *file, unsigned int flags, umode_t mode, u32 opcode) { @@ -934,14 +934,14 @@ out_err: return err; } -static int fuse_mknod(struct mnt_idmap *, struct inode *, struct dentry *, +static int fuse_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t); static int fuse_atomic_open(struct inode *dir, struct dentry *entry, struct file *file, unsigned flags, umode_t mode) { int err; - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct fuse_conn *fc = get_fuse_conn(dir); if (fuse_is_bad(dir)) @@ -980,7 +980,7 @@ mknod: /* * Code shared between mknod, mkdir, symlink and link */ -static struct dentry *create_new_entry(struct mnt_idmap *idmap, struct fuse_mount *fm, +static struct dentry *create_new_entry(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args, struct inode *dir, struct dentry *entry, umode_t mode) { @@ -1053,7 +1053,7 @@ static struct dentry *create_new_entry(struct mnt_idmap *idmap, struct fuse_moun return ERR_PTR(err); } -static int create_new_nondir(struct mnt_idmap *idmap, struct fuse_mount *fm, +static int create_new_nondir(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args, struct inode *dir, struct dentry *entry, umode_t mode) { @@ -1069,7 +1069,7 @@ static int create_new_nondir(struct mnt_idmap *idmap, struct fuse_mount *fm, return PTR_ERR(create_new_entry(idmap, fm, args, dir, entry, mode)); } -static int fuse_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode, dev_t rdev) { struct fuse_mknod_in inarg; @@ -1092,13 +1092,13 @@ static int fuse_mknod(struct mnt_idmap *idmap, struct inode *dir, return create_new_nondir(idmap, fm, &args, dir, entry, mode); } -static int fuse_create(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode) { return fuse_mknod(idmap, dir, entry, mode, 0); } -static int fuse_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct fuse_conn *fc = get_fuse_conn(dir); @@ -1116,7 +1116,7 @@ static int fuse_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return err; } -static struct dentry *fuse_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *fuse_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, umode_t mode) { struct fuse_mkdir_in inarg; @@ -1146,7 +1146,7 @@ static struct dentry *fuse_mkdir(struct mnt_idmap *idmap, struct inode *dir, return create_new_entry(idmap, fm, &args, dir, entry, S_IFDIR); } -static int fuse_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int fuse_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *entry, const char *link) { struct fuse_mount *fm = get_fuse_mount(dir); @@ -1256,9 +1256,10 @@ static int fuse_rmdir(struct inode *dir, struct dentry *entry) return err; } -static int fuse_rename_common(struct mnt_idmap *idmap, struct inode *olddir, struct dentry *oldent, - struct inode *newdir, struct dentry *newent, - unsigned int flags, int opcode, size_t argsize) +static int fuse_rename_common(const struct mnt_idmap *idmap, struct inode *olddir, + struct dentry *oldent, struct inode *newdir, + struct dentry *newent, unsigned int flags, + int opcode, size_t argsize) { int err; struct fuse_rename2_in inarg; @@ -1306,7 +1307,7 @@ static int fuse_rename_common(struct mnt_idmap *idmap, struct inode *olddir, str return err; } -static int fuse_rename2(struct mnt_idmap *idmap, struct inode *olddir, +static int fuse_rename2(const struct mnt_idmap *idmap, struct inode *olddir, struct dentry *oldent, struct inode *newdir, struct dentry *newent, unsigned int flags) { @@ -1375,7 +1376,7 @@ out: return err; } -static void fuse_fillattr(struct mnt_idmap *idmap, struct inode *inode, +static void fuse_fillattr(const struct mnt_idmap *idmap, struct inode *inode, struct fuse_attr *attr, struct kstat *stat) { unsigned int blkbits; @@ -1429,7 +1430,7 @@ static void fuse_statx_to_attr(struct fuse_statx *sx, struct fuse_attr *attr) attr->blksize = sx->blksize; } -static int fuse_do_statx(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_do_statx(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, struct kstat *stat) { int err; @@ -1490,7 +1491,7 @@ static int fuse_do_statx(struct mnt_idmap *idmap, struct inode *inode, return 0; } -static int fuse_do_getattr(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_do_getattr(const struct mnt_idmap *idmap, struct inode *inode, struct kstat *stat, struct file *file) { int err; @@ -1536,7 +1537,7 @@ static int fuse_do_getattr(struct mnt_idmap *idmap, struct inode *inode, return err; } -static int fuse_update_get_attr(struct mnt_idmap *idmap, struct inode *inode, +static int fuse_update_get_attr(const struct mnt_idmap *idmap, struct inode *inode, struct file *file, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -1762,7 +1763,7 @@ static int fuse_perm_getattr(struct inode *inode, int mask) * access request is sent. Execute permission is still checked * locally based on file mode. */ -static int fuse_permission(struct mnt_idmap *idmap, +static int fuse_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct fuse_conn *fc = get_fuse_conn(inode); @@ -2000,7 +2001,7 @@ static bool update_mtime(unsigned ivalid, bool trust_local_mtime) return true; } -static void iattr_to_fattr(struct mnt_idmap *idmap, struct fuse_conn *fc, +static void iattr_to_fattr(const struct mnt_idmap *idmap, struct fuse_conn *fc, struct iattr *iattr, struct fuse_setattr_in *arg, bool trust_local_cmtime) { @@ -2142,7 +2143,7 @@ int fuse_flush_times(struct inode *inode, struct fuse_file *ff) * vmtruncate() doesn't allow for this case, so do the rlimit checking * and the actual truncation by hand. */ -int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_do_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct file *file) { struct inode *inode = d_inode(dentry); @@ -2323,7 +2324,7 @@ unlock: return err; } -static int fuse_setattr(struct mnt_idmap *idmap, struct dentry *entry, +static int fuse_setattr(const struct mnt_idmap *idmap, struct dentry *entry, struct iattr *attr) { struct inode *inode = d_inode(entry); @@ -2386,7 +2387,7 @@ static int fuse_setattr(struct mnt_idmap *idmap, struct dentry *entry, return ret; } -static int fuse_getattr(struct mnt_idmap *idmap, +static int fuse_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/fuse/file.c b/fs/fuse/file.c index d73afcbc1eb2..3d209e2b71ba 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -1507,7 +1507,7 @@ static const struct iomap_write_ops fuse_iomap_write_ops = { static ssize_t fuse_cache_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct file *file = iocb->ki_filp; - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct address_space *mapping = file->f_mapping; ssize_t written = 0; struct inode *inode = mapping->host; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 8546855386b5..87e2bd9d4bb1 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1012,7 +1012,7 @@ void __exit fuse_ctl_cleanup(void); /* * Simple request sending that does request allocation and freeing */ -ssize_t __fuse_simple_request(struct mnt_idmap *idmap, +ssize_t __fuse_simple_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args); @@ -1021,7 +1021,7 @@ static inline ssize_t fuse_simple_request(struct fuse_mount *fm, struct fuse_arg return __fuse_simple_request(&invalid_mnt_idmap, fm, args); } -static inline ssize_t fuse_simple_idmap_request(struct mnt_idmap *idmap, +static inline ssize_t fuse_simple_idmap_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args) { @@ -1198,7 +1198,7 @@ bool fuse_write_update_attr(struct inode *inode, loff_t pos, ssize_t written); int fuse_flush_times(struct inode *inode, struct fuse_file *ff); int fuse_write_inode(struct inode *inode, struct writeback_control *wbc); -int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int fuse_do_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr, struct file *file); void fuse_unlock_inode(struct inode *inode, bool locked); @@ -1214,9 +1214,9 @@ extern const struct xattr_handler * const fuse_xattr_handlers[]; struct posix_acl; struct posix_acl *fuse_get_inode_acl(struct inode *inode, int type, bool rcu); -struct posix_acl *fuse_get_acl(struct mnt_idmap *idmap, +struct posix_acl *fuse_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int fuse_set_acl(struct mnt_idmap *, struct dentry *dentry, +int fuse_set_acl(const struct mnt_idmap *, struct dentry *dentry, struct posix_acl *acl, int type); /* readdir.c */ @@ -1247,7 +1247,7 @@ long fuse_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg); long fuse_file_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); int fuse_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int fuse_fileattr_set(struct mnt_idmap *idmap, +int fuse_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); /* iomode.c */ diff --git a/fs/fuse/ioctl.c b/fs/fuse/ioctl.c index dc3a188f5d72..ce1807704da6 100644 --- a/fs/fuse/ioctl.c +++ b/fs/fuse/ioctl.c @@ -537,7 +537,7 @@ cleanup: return err; } -int fuse_fileattr_set(struct mnt_idmap *idmap, +int fuse_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/fuse/req.c b/fs/fuse/req.c index 51cfd64ea1ba..6274a23af445 100644 --- a/fs/fuse/req.c +++ b/fs/fuse/req.c @@ -3,7 +3,8 @@ #include "dev.h" #include "fuse_i.h" -static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, struct mnt_idmap *idmap) +static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, + const struct mnt_idmap *idmap) { struct fuse_conn *fc = fm->fc; bool no_idmap = !fm->sb || (fm->sb->s_iflags & SB_I_NOIDMAP); @@ -48,7 +49,8 @@ static int fuse_fill_creds(struct fuse_mount *fm, struct fuse_args *args, struct return 0; } -static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, struct mnt_idmap *idmap) +static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, + const struct mnt_idmap *idmap) { if (!args->force && fm->fc->conn_error) return -ECONNREFUSED; @@ -56,7 +58,7 @@ static int fuse_req_prep(struct fuse_mount *fm, struct fuse_args *args, struct m return fuse_fill_creds(fm, args, idmap); } -ssize_t __fuse_simple_request(struct mnt_idmap *idmap, struct fuse_mount *fm, +ssize_t __fuse_simple_request(const struct mnt_idmap *idmap, struct fuse_mount *fm, struct fuse_args *args) { struct fuse_conn *fc = fm->fc; diff --git a/fs/fuse/xattr.c b/fs/fuse/xattr.c index cab2685acc65..53e3c5e6fff0 100644 --- a/fs/fuse/xattr.c +++ b/fs/fuse/xattr.c @@ -188,7 +188,7 @@ static int fuse_xattr_get(const struct xattr_handler *handler, } static int fuse_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/gfs2/acl.c b/fs/gfs2/acl.c index 49e489fe27ef..f1c6a5e392b4 100644 --- a/fs/gfs2/acl.c +++ b/fs/gfs2/acl.c @@ -102,7 +102,7 @@ out: return error; } -int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int gfs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct inode *inode = d_inode(dentry); diff --git a/fs/gfs2/acl.h b/fs/gfs2/acl.h index 82f5b09c04e6..d19d41755936 100644 --- a/fs/gfs2/acl.h +++ b/fs/gfs2/acl.h @@ -13,7 +13,7 @@ struct posix_acl *gfs2_get_acl(struct inode *inode, int type, bool rcu); int __gfs2_set_acl(struct inode *inode, struct posix_acl *acl, int type); -int gfs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int gfs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); #endif /* __ACL_DOT_H__ */ diff --git a/fs/gfs2/file.c b/fs/gfs2/file.c index 6fb2adeef274..1efd0679badd 100644 --- a/fs/gfs2/file.c +++ b/fs/gfs2/file.c @@ -277,7 +277,7 @@ out: return error; } -int gfs2_fileattr_set(struct mnt_idmap *idmap, +int gfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c index f748f8b3c73a..2c2dc459c037 100644 --- a/fs/gfs2/inode.c +++ b/fs/gfs2/inode.c @@ -973,7 +973,7 @@ fail: * Returns: errno */ -static int gfs2_create(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return gfs2_create_inode(dir, dentry, NULL, S_IFREG | mode, 0, NULL, 0, 1); @@ -1329,7 +1329,7 @@ out_inodes: * Returns: errno */ -static int gfs2_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { unsigned int size; @@ -1351,7 +1351,7 @@ static int gfs2_symlink(struct mnt_idmap *idmap, struct inode *dir, * Returns: the dentry, or ERR_PTR(errno) */ -static struct dentry *gfs2_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *gfs2_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { unsigned dsize = gfs2_max_stuffed_size(GFS2_I(dir)); @@ -1369,7 +1369,7 @@ static struct dentry *gfs2_mkdir(struct mnt_idmap *idmap, struct inode *dir, * */ -static int gfs2_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int gfs2_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { return gfs2_create_inode(dir, dentry, NULL, mode, dev, NULL, 0, 1); @@ -1887,7 +1887,7 @@ out: return error; } -static int gfs2_rename2(struct mnt_idmap *idmap, struct inode *odir, +static int gfs2_rename2(const struct mnt_idmap *idmap, struct inode *odir, struct dentry *odentry, struct inode *ndir, struct dentry *ndentry, unsigned int flags) { @@ -1974,7 +1974,7 @@ out: * Returns: errno */ -int gfs2_permission(struct mnt_idmap *idmap, struct inode *inode, +int gfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int may_not_block = mask & MAY_NOT_BLOCK; @@ -2106,7 +2106,7 @@ out: * Returns: errno */ -static int gfs2_setattr(struct mnt_idmap *idmap, +static int gfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -2168,7 +2168,7 @@ out: * Returns: errno */ -static int gfs2_getattr(struct mnt_idmap *idmap, +static int gfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/gfs2/inode.h b/fs/gfs2/inode.h index 2fcd96dd1361..99196d3114e4 100644 --- a/fs/gfs2/inode.h +++ b/fs/gfs2/inode.h @@ -97,7 +97,7 @@ int gfs2_dinode_dealloc(struct gfs2_inode *ip); struct inode *gfs2_lookupi(struct inode *dir, const struct qstr *name, int is_root); -int gfs2_permission(struct mnt_idmap *idmap, +int gfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); struct inode *gfs2_lookup_meta(struct inode *dip, const char *name); void gfs2_dinode_out(const struct gfs2_inode *ip, void *buf); @@ -109,7 +109,7 @@ extern const struct file_operations gfs2_file_fops_nolock; extern const struct file_operations gfs2_dir_fops_nolock; int gfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int gfs2_fileattr_set(struct mnt_idmap *idmap, +int gfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); void gfs2_set_inode_flags(struct inode *inode); diff --git a/fs/gfs2/log.c b/fs/gfs2/log.c index a92c84146de9..b55daf1e6381 100644 --- a/fs/gfs2/log.c +++ b/fs/gfs2/log.c @@ -107,7 +107,7 @@ __acquires(&sdp->sd_ail_lock) gfs2_assert(sdp, bd->bd_tr == tr); if (!buffer_busy(bh)) { - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { list_move(&bd->bd_ail_st_list, &tr->tr_ail2_list); continue; @@ -321,7 +321,7 @@ static int gfs2_ail1_empty_one(struct gfs2_sbd *sdp, struct gfs2_trans *tr, active_count++; continue; } - if (!buffer_uptodate(bh) && + if (buffer_write_io_error(bh) && !cmpxchg(&sdp->sd_log_error, 0, -EIO)) gfs2_io_error_bh(sdp, bh); /* diff --git a/fs/gfs2/lops.c b/fs/gfs2/lops.c index 77ef22eab368..88c84895ec6f 100644 --- a/fs/gfs2/lops.c +++ b/fs/gfs2/lops.c @@ -48,7 +48,7 @@ void gfs2_pin(struct gfs2_sbd *sdp, struct buffer_head *bh) clear_buffer_dirty(bh); if (test_set_buffer_pinned(bh)) gfs2_assert_withdraw(sdp, 0); - if (!buffer_uptodate(bh)) + if (!buffer_uptodate(bh) || buffer_write_io_error(bh)) gfs2_io_error_bh(sdp, bh); bd = bh->b_private; /* If this buffer is in the AIL and it has already been written @@ -179,6 +179,8 @@ static void gfs2_end_log_write_bh(struct gfs2_sbd *sdp, struct folio *folio, do { if (error) mark_buffer_write_io_error(bh); + else + clear_buffer_write_io_error(bh); unlock_buffer(bh); next = bh->b_this_page; size -= bh->b_size; diff --git a/fs/gfs2/xattr.c b/fs/gfs2/xattr.c index db38d972debd..c26f180419d9 100644 --- a/fs/gfs2/xattr.c +++ b/fs/gfs2/xattr.c @@ -1239,7 +1239,7 @@ int __gfs2_xattr_set(struct inode *inode, const char *name, } static int gfs2_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/hfs/attr.c b/fs/hfs/attr.c index f8395cdd1adf..6d737a085461 100644 --- a/fs/hfs/attr.c +++ b/fs/hfs/attr.c @@ -121,7 +121,7 @@ static int hfs_xattr_get(const struct xattr_handler *handler, } static int hfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/hfs/dir.c b/fs/hfs/dir.c index e1f1fb351464..f6b97da19788 100644 --- a/fs/hfs/dir.c +++ b/fs/hfs/dir.c @@ -183,7 +183,7 @@ static int hfs_dir_release(struct inode *inode, struct file *file) * a directory and return a corresponding inode, given the inode for * the directory and the name (and its length) of the new file. */ -static int hfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -213,7 +213,7 @@ static int hfs_create(struct mnt_idmap *idmap, struct inode *dir, * in a directory, given the inode for the parent directory and the * name (and its length) of the new directory. */ -static struct dentry *hfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -281,7 +281,7 @@ static int hfs_remove(struct inode *dir, struct dentry *dentry) * new file/directory. * XXX: how do you handle must_be dir? */ -static int hfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int hfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/hfs/hfs_fs.h b/fs/hfs/hfs_fs.h index e250f87a5e33..fdfa5d303d5e 100644 --- a/fs/hfs/hfs_fs.h +++ b/fs/hfs/hfs_fs.h @@ -212,7 +212,7 @@ extern struct inode *hfs_new_inode(struct inode *dir, const struct qstr *name, extern void hfs_inode_write_fork(struct inode *inode, struct hfs_extent *ext, __be32 *log_size, __be32 *phys_size); extern int hfs_write_inode(struct inode *inode, struct writeback_control *wbc); -extern int hfs_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int hfs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void hfs_inode_read_fork(struct inode *inode, struct hfs_extent *ext, __be32 __log_size, __be32 phys_size, diff --git a/fs/hfs/inode.c b/fs/hfs/inode.c index 2aef3c36a150..c81314f668ac 100644 --- a/fs/hfs/inode.c +++ b/fs/hfs/inode.c @@ -643,7 +643,7 @@ static int hfs_file_release(struct inode *inode, struct file *file) return 0; } -int hfs_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int hfs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hfsplus/dir.c b/fs/hfsplus/dir.c index 51fcba2e6d40..b3a1193a491f 100644 --- a/fs/hfsplus/dir.c +++ b/fs/hfsplus/dir.c @@ -460,7 +460,7 @@ out: return res; } -static int hfsplus_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct hfsplus_sb_info *sbi = HFSPLUS_SB(dir->i_sb); @@ -511,7 +511,7 @@ out: return res; } -static int hfsplus_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct hfsplus_sb_info *sbi = HFSPLUS_SB(dir->i_sb); @@ -561,19 +561,19 @@ out: return res; } -static int hfsplus_create(struct mnt_idmap *idmap, struct inode *dir, +static int hfsplus_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); } -static struct dentry *hfsplus_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hfsplus_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(hfsplus_mknod(&nop_mnt_idmap, dir, dentry, mode, 0)); } -static int hfsplus_rename(struct mnt_idmap *idmap, +static int hfsplus_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/hfsplus/hfsplus_fs.h b/fs/hfsplus/hfsplus_fs.h index 1e5b58e6a13f..d55cb16e3897 100644 --- a/fs/hfsplus/hfsplus_fs.h +++ b/fs/hfsplus/hfsplus_fs.h @@ -459,13 +459,13 @@ void hfsplus_inode_write_fork(struct inode *inode, struct hfsplus_fork_raw *fork); int hfsplus_cat_read_inode(struct inode *inode, struct hfs_find_data *fd); int hfsplus_cat_write_inode(struct inode *inode); -int hfsplus_getattr(struct mnt_idmap *idmap, const struct path *path, +int hfsplus_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); int hfsplus_file_fsync(struct file *file, loff_t start, loff_t end, int datasync); int hfsplus_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int hfsplus_fileattr_set(struct mnt_idmap *idmap, +int hfsplus_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); /* ioctl.c */ diff --git a/fs/hfsplus/inode.c b/fs/hfsplus/inode.c index 2ce6de574fa6..aed0866499b2 100644 --- a/fs/hfsplus/inode.c +++ b/fs/hfsplus/inode.c @@ -305,7 +305,7 @@ static int hfsplus_file_release(struct inode *inode, struct file *file) return 0; } -static int hfsplus_setattr(struct mnt_idmap *idmap, +static int hfsplus_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -335,7 +335,7 @@ static int hfsplus_setattr(struct mnt_idmap *idmap, return 0; } -int hfsplus_getattr(struct mnt_idmap *idmap, const struct path *path, +int hfsplus_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -797,7 +797,7 @@ int hfsplus_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int hfsplus_fileattr_set(struct mnt_idmap *idmap, +int hfsplus_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/hfsplus/xattr.c b/fs/hfsplus/xattr.c index 21a1c196c71f..71364e093fa7 100644 --- a/fs/hfsplus/xattr.c +++ b/fs/hfsplus/xattr.c @@ -1008,7 +1008,7 @@ static int hfsplus_osx_getxattr(const struct xattr_handler *handler, } static int hfsplus_osx_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_security.c b/fs/hfsplus/xattr_security.c index 90f68ec119cd..1969919c12cb 100644 --- a/fs/hfsplus/xattr_security.c +++ b/fs/hfsplus/xattr_security.c @@ -23,7 +23,7 @@ static int hfsplus_security_getxattr(const struct xattr_handler *handler, } static int hfsplus_security_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_trusted.c b/fs/hfsplus/xattr_trusted.c index fdbaebc1c49a..c140a95ab3f0 100644 --- a/fs/hfsplus/xattr_trusted.c +++ b/fs/hfsplus/xattr_trusted.c @@ -22,7 +22,7 @@ static int hfsplus_trusted_getxattr(const struct xattr_handler *handler, } static int hfsplus_trusted_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hfsplus/xattr_user.c b/fs/hfsplus/xattr_user.c index 6464b6c3d58d..7e5da15f9937 100644 --- a/fs/hfsplus/xattr_user.c +++ b/fs/hfsplus/xattr_user.c @@ -22,7 +22,7 @@ static int hfsplus_user_getxattr(const struct xattr_handler *handler, } static int hfsplus_user_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/hostfs/hostfs_kern.c b/fs/hostfs/hostfs_kern.c index 7add056d47d8..613146e76dec 100644 --- a/fs/hostfs/hostfs_kern.c +++ b/fs/hostfs/hostfs_kern.c @@ -592,7 +592,7 @@ static struct inode *hostfs_iget(struct super_block *sb, char *name) return inode; } -static int hostfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hostfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -673,7 +673,7 @@ static int hostfs_unlink(struct inode *ino, struct dentry *dentry) return err; } -static int hostfs_symlink(struct mnt_idmap *idmap, struct inode *ino, +static int hostfs_symlink(const struct mnt_idmap *idmap, struct inode *ino, struct dentry *dentry, const char *to) { char *file; @@ -686,7 +686,7 @@ static int hostfs_symlink(struct mnt_idmap *idmap, struct inode *ino, return err; } -static struct dentry *hostfs_mkdir(struct mnt_idmap *idmap, struct inode *ino, +static struct dentry *hostfs_mkdir(const struct mnt_idmap *idmap, struct inode *ino, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -719,7 +719,7 @@ static int hostfs_rmdir(struct inode *ino, struct dentry *dentry) return err; } -static int hostfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hostfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode *inode; @@ -745,7 +745,7 @@ static int hostfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return 0; } -static int hostfs_rename2(struct mnt_idmap *idmap, +static int hostfs_rename2(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -774,7 +774,7 @@ static int hostfs_rename2(struct mnt_idmap *idmap, return err; } -static int hostfs_permission(struct mnt_idmap *idmap, +static int hostfs_permission(const struct mnt_idmap *idmap, struct inode *ino, int desired) { char *name; @@ -801,7 +801,7 @@ static int hostfs_permission(struct mnt_idmap *idmap, return err; } -static int hostfs_setattr(struct mnt_idmap *idmap, +static int hostfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hpfs/hpfs_fn.h b/fs/hpfs/hpfs_fn.h index 237c1c23e855..a398dd8bdf30 100644 --- a/fs/hpfs/hpfs_fn.h +++ b/fs/hpfs/hpfs_fn.h @@ -280,7 +280,7 @@ void hpfs_init_inode(struct inode *); void hpfs_read_inode(struct inode *); void hpfs_write_inode(struct inode *); void hpfs_write_inode_nolock(struct inode *); -int hpfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int hpfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); void hpfs_write_if_changed(struct inode *); void hpfs_evict_inode(struct inode *); diff --git a/fs/hpfs/inode.c b/fs/hpfs/inode.c index 1b4fcf760aad..396773d0b669 100644 --- a/fs/hpfs/inode.c +++ b/fs/hpfs/inode.c @@ -257,7 +257,7 @@ void hpfs_write_inode_nolock(struct inode *i) brelse(bh); } -int hpfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int hpfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/hpfs/namei.c b/fs/hpfs/namei.c index 9446f4038874..ac9b5e3e83fa 100644 --- a/fs/hpfs/namei.c +++ b/fs/hpfs/namei.c @@ -19,7 +19,7 @@ static void hpfs_update_directory_times(struct inode *dir) hpfs_write_inode_nolock(dir); } -static struct dentry *hpfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hpfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { const unsigned char *name = dentry->d_name.name; @@ -128,7 +128,7 @@ bail: return ERR_PTR(err); } -static int hpfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { const unsigned char *name = dentry->d_name.name; @@ -215,7 +215,7 @@ bail: return err; } -static int hpfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { const unsigned char *name = dentry->d_name.name; @@ -289,7 +289,7 @@ bail: return err; } -static int hpfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int hpfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symlink) { const unsigned char *name = dentry->d_name.name; @@ -500,7 +500,7 @@ const struct address_space_operations hpfs_symlink_aops = { .read_folio = hpfs_symlink_read_folio }; -static int hpfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int hpfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index 7611a8470ea2..6656807ce437 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -828,7 +828,7 @@ out_nolock: return error; } -static int hugetlbfs_setattr(struct mnt_idmap *idmap, +static int hugetlbfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -892,7 +892,7 @@ static struct inode *hugetlbfs_get_root(struct super_block *sb, static struct lock_class_key hugetlbfs_i_mmap_rwsem_key; static struct inode *hugetlbfs_get_inode(struct super_block *sb, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, umode_t mode, dev_t dev) { @@ -954,7 +954,7 @@ static struct inode *hugetlbfs_get_inode(struct super_block *sb, /* * File creation. Allocate an inode, and we're done.. */ -static int hugetlbfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int hugetlbfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode *inode; @@ -967,7 +967,7 @@ static int hugetlbfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return 0; } -static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *hugetlbfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int retval = hugetlbfs_mknod(idmap, dir, dentry, @@ -977,14 +977,14 @@ static struct dentry *hugetlbfs_mkdir(struct mnt_idmap *idmap, struct inode *dir return ERR_PTR(retval); } -static int hugetlbfs_create(struct mnt_idmap *idmap, +static int hugetlbfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return hugetlbfs_mknod(idmap, dir, dentry, mode | S_IFREG, 0); } -static int hugetlbfs_tmpfile(struct mnt_idmap *idmap, +static int hugetlbfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { @@ -998,7 +998,7 @@ static int hugetlbfs_tmpfile(struct mnt_idmap *idmap, return finish_open_simple(file, 0); } -static int hugetlbfs_symlink(struct mnt_idmap *idmap, +static int hugetlbfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { diff --git a/fs/inode.c b/fs/inode.c index a9d37be390a1..1cb6293b237a 100644 --- a/fs/inode.c +++ b/fs/inode.c @@ -1770,7 +1770,7 @@ EXPORT_SYMBOL(ilookup); * function must never block --- find_inode() can block in * __wait_on_freeing_inode() --- or when the caller can not increment * the reference count because the resulting iput() might cause an - * inode eviction. The tradeoff is that the @match funtion must be + * inode eviction. The tradeoff is that the @match function must be * very carefully implemented. */ struct inode *find_inode_nowait(struct super_block *sb, @@ -2336,7 +2336,7 @@ EXPORT_SYMBOL(touch_atime); * response to write or truncate. Return 0 if nothing has to be changed. * Negative value on error (change should be denied). */ -int dentry_needs_remove_privs(struct mnt_idmap *idmap, +int dentry_needs_remove_privs(const struct mnt_idmap *idmap, struct dentry *dentry) { struct inode *inode = d_inode(dentry); @@ -2355,7 +2355,7 @@ int dentry_needs_remove_privs(struct mnt_idmap *idmap, return mask; } -static int __remove_privs(struct mnt_idmap *idmap, +static int __remove_privs(const struct mnt_idmap *idmap, struct dentry *dentry, int kill) { struct iattr newattrs; @@ -2715,7 +2715,7 @@ EXPORT_SYMBOL(init_special_inode); * and initializing i_uid and i_gid. On non-idmapped mounts or if permission * checking is to be performed on the raw inode simply pass @nop_mnt_idmap. */ -void inode_init_owner(struct mnt_idmap *idmap, struct inode *inode, +void inode_init_owner(const struct mnt_idmap *idmap, struct inode *inode, const struct inode *dir, umode_t mode) { inode_fsuid_set(inode, idmap); @@ -2745,7 +2745,7 @@ EXPORT_SYMBOL(inode_init_owner); * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -bool inode_owner_or_capable(struct mnt_idmap *idmap, +bool inode_owner_or_capable(const struct mnt_idmap *idmap, const struct inode *inode) { vfsuid_t vfsuid; @@ -3032,7 +3032,7 @@ EXPORT_SYMBOL(inode_set_ctime_deleg); * * Return: true if the caller is sufficiently privileged, false if not. */ -bool in_group_or_capable(struct mnt_idmap *idmap, +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid) { if (vfsgid_in_group_p(vfsgid)) @@ -3057,7 +3057,7 @@ EXPORT_SYMBOL(in_group_or_capable); * * Return: the new mode to use for the file */ -umode_t mode_strip_sgid(struct mnt_idmap *idmap, +umode_t mode_strip_sgid(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode) { if ((mode & (S_ISGID | S_IXGRP)) != (S_ISGID | S_IXGRP)) diff --git a/fs/internal.h b/fs/internal.h index c658c8a5ebd5..e833c7e6e14f 100644 --- a/fs/internal.h +++ b/fs/internal.h @@ -55,7 +55,7 @@ extern int filename_lookup(int dfd, struct filename *name, unsigned flags, struct path *path, const struct path *root); int filename_rmdir(int dfd, struct filename *name); int filename_unlinkat(int dfd, struct filename *name); -int may_linkat(struct mnt_idmap *idmap, const struct path *link); +int may_linkat(const struct mnt_idmap *idmap, const struct path *link); int filename_renameat2(int olddfd, struct filename *oldname, int newdfd, struct filename *newname, unsigned int flags); int filename_mkdirat(int dfd, struct filename *name, umode_t mode); @@ -63,7 +63,7 @@ int filename_mknodat(int dfd, struct filename *name, umode_t mode, unsigned int int filename_symlinkat(struct filename *from, int newdfd, struct filename *to); int filename_linkat(int olddfd, struct filename *old, int newdfd, struct filename *new, int flags); -int vfs_tmpfile(struct mnt_idmap *idmap, +int vfs_tmpfile(const struct mnt_idmap *idmap, const struct path *parentpath, struct file *file, umode_t mode); struct dentry *d_hash_and_lookup(struct dentry *, struct qstr *); @@ -198,6 +198,7 @@ extern struct file *do_file_open_root(const struct path *, extern struct open_how build_open_how(int flags, umode_t mode); extern int build_open_flags(const struct open_how *how, struct open_flags *op); struct file *file_close_fd_locked(struct files_struct *files, unsigned fd); +int filp_close_sync(struct file *filp, fl_owner_t id); int do_ftruncate(struct file *file, loff_t length, unsigned int flags); int chmod_common(const struct path *path, umode_t mode); @@ -205,13 +206,14 @@ int do_fchownat(int dfd, const char __user *filename, uid_t user, gid_t group, int flag); int chown_common(const struct path *path, uid_t user, gid_t group); extern int vfs_open(const struct path *, struct file *); +int vfs_open_consume(struct path *, struct file *); /* * inode.c */ extern long prune_icache_sb(struct super_block *sb, struct shrink_control *sc); -int dentry_needs_remove_privs(struct mnt_idmap *, struct dentry *dentry); -bool in_group_or_capable(struct mnt_idmap *idmap, +int dentry_needs_remove_privs(const struct mnt_idmap *, struct dentry *dentry); +bool in_group_or_capable(const struct mnt_idmap *idmap, const struct inode *inode, vfsgid_t vfsgid); /* @@ -299,21 +301,21 @@ int filename_setxattr(int dfd, struct filename *filename, int setxattr_copy(const char __user *name, struct kernel_xattr_ctx *ctx); int import_xattr_name(struct xattr_name *kname, const char __user *name); -int may_write_xattr(struct mnt_idmap *idmap, struct inode *inode); +int may_write_xattr(const struct mnt_idmap *idmap, struct inode *inode); #ifdef CONFIG_FS_POSIX_ACL -int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size); -ssize_t do_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size); #else -static inline int do_set_acl(struct mnt_idmap *idmap, +static inline int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size) { return -EOPNOTSUPP; } -static inline ssize_t do_get_acl(struct mnt_idmap *idmap, +static inline ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size) { @@ -327,8 +329,8 @@ ssize_t __kernel_write_iter(struct file *file, struct iov_iter *from, loff_t *po * fs/attr.c */ struct mnt_idmap *alloc_mnt_idmap(struct user_namespace *mnt_userns); -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap); -void mnt_idmap_put(struct mnt_idmap *idmap); +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap); +void mnt_idmap_put(const struct mnt_idmap *idmap); struct stashed_operations { struct dentry *(*stash_dentry)(struct dentry **stashed, struct dentry *dentry); @@ -354,12 +356,12 @@ static inline bool path_mounted(const struct path *path) } void file_f_owner_release(struct file *file); bool file_seek_cur_needs_f_lock(struct file *file); -int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map); +int statmount_mnt_idmap(const struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map); struct dentry *find_next_child(struct dentry *parent, struct dentry *prev); -int anon_inode_getattr(struct mnt_idmap *idmap, const struct path *path, +int anon_inode_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); -int anon_inode_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int anon_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); void pidfs_get_root(struct path *path); void nsfs_get_root(struct path *path); diff --git a/fs/iomap/bio.c b/fs/iomap/bio.c index 48100c614431..d46c2f8ea18c 100644 --- a/fs/iomap/bio.c +++ b/fs/iomap/bio.c @@ -169,6 +169,7 @@ int iomap_bio_read_folio_range_sync(const struct iomap_iter *iter, { const struct iomap *srcmap = iomap_iter_srcmap(iter); sector_t sector = iomap_sector(srcmap, pos); + struct bvec_iter saved_iter; struct bio_vec bvec; struct bio bio; int error; @@ -178,10 +179,11 @@ int iomap_bio_read_folio_range_sync(const struct iomap_iter *iter, bio_add_folio_nofail(&bio, folio, len, offset_in_folio(folio, pos)); if (srcmap->flags & IOMAP_F_INTEGRITY) fs_bio_integrity_alloc(&bio); + saved_iter = bio.bi_iter; error = submit_bio_wait(&bio); if (bio_integrity(&bio)) { if (!error) - error = fs_bio_integrity_verify(&bio, sector, len); + error = fs_bio_integrity_verify(&bio, &saved_iter); fs_bio_integrity_free(&bio); } bio_uninit(&bio); diff --git a/fs/iomap/buffered-io.c b/fs/iomap/buffered-io.c index 0a5ebfda90f1..678f4c5329e7 100644 --- a/fs/iomap/buffered-io.c +++ b/fs/iomap/buffered-io.c @@ -143,8 +143,8 @@ static unsigned ifs_next_clean_block(struct folio *folio, blks + start_blk) - blks; } -static unsigned ifs_find_dirty_range(struct folio *folio, - struct iomap_folio_state *ifs, u64 *range_start, u64 range_end) +static unsigned ifs_find_dirty_range(struct folio *folio, u64 *range_start, + u64 range_end) { struct inode *inode = folio->mapping->host; unsigned start_blk = @@ -176,7 +176,7 @@ static unsigned iomap_find_dirty_range(struct folio *folio, u64 *range_start, return 0; if (ifs) - return ifs_find_dirty_range(folio, ifs, range_start, range_end); + return ifs_find_dirty_range(folio, range_start, range_end); return range_end - *range_start; } @@ -1708,7 +1708,7 @@ static int iomap_zero_iter(struct iomap_iter *iter, bool *did_zero, * @iomap_flags: Flags to set on the associated iomap to track the batch. * * Returns the folio count directly. Also returns the associated control flag if - * the the batch lookup is performed and the expected offset of a subsequent + * the batch lookup is performed and the expected offset of a subsequent * lookup via out params. The caller is responsible to set the flag on the * associated iomap. */ diff --git a/fs/iomap/direct-io.c b/fs/iomap/direct-io.c index 8b4039d16ce8..a431ceda9ebd 100644 --- a/fs/iomap/direct-io.c +++ b/fs/iomap/direct-io.c @@ -76,10 +76,19 @@ static void iomap_dio_submit_bio(const struct iomap_iter *iter, if (dio->dops && dio->dops->submit_io) { dio->dops->submit_io(iter, bio, pos); - } else { - WARN_ON_ONCE(iter->iomap.flags & IOMAP_F_ANON_WRITE); - blk_crypto_submit_bio(bio); + return; + } + + WARN_ON_ONCE(iter->iomap.flags & IOMAP_F_ANON_WRITE); + + if (iter->iomap.flags & IOMAP_F_INTEGRITY) { + if (dio->flags & IOMAP_DIO_WRITE) + fs_bio_integrity_generate(bio); + else + fs_bio_integrity_alloc(bio); } + + blk_crypto_submit_bio(bio); } static inline enum fserror_type iomap_dio_err_type(const struct iomap_dio *dio) @@ -246,8 +255,7 @@ static void __iomap_dio_bio_end_io(struct bio *bio, bool inline_completion) fs_bio_integrity_free(bio); if (dio->flags & IOMAP_DIO_BOUNCE) { - bio_iov_iter_unbounce(bio, !!dio->error, - dio->flags & IOMAP_DIO_USER_BACKED); + bio_free_folios(bio); bio_put(bio); } else if (dio->flags & IOMAP_DIO_USER_BACKED) { bio_check_pages_dirty(bio); @@ -336,6 +344,7 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, struct iomap_dio *dio, loff_t pos, unsigned int alignment, blk_opf_t op) { + unsigned int maxsize = iomap_max_bio_size(&iter->iomap); unsigned int nr_vecs; struct bio *bio; ssize_t ret; @@ -353,14 +362,12 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, bio->bi_private = dio; bio->bi_end_io = iomap_dio_bio_end_io; - if (dio->flags & IOMAP_DIO_BOUNCE) - ret = bio_iov_iter_bounce(bio, dio->submit.iter, - iomap_max_bio_size(&iter->iomap), alignment); + ret = bio_iov_iter_bounce_write(bio, dio->submit.iter, maxsize, + alignment); else - ret = bio_iov_iter_get_pages(bio, dio->submit.iter, - bdev_dma_alignment(bio->bi_bdev), - alignment - 1); + ret = bio_iov_iter_get_pages(bio, dio->submit.iter, maxsize, + bdev_dma_alignment(bio->bi_bdev), alignment - 1); if (unlikely(ret)) goto out_put_bio; ret = bio->bi_iter.bi_size; @@ -374,13 +381,6 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, goto out_bio_release_pages; } - if (iter->iomap.flags & IOMAP_F_INTEGRITY) { - if (dio->flags & IOMAP_DIO_WRITE) - fs_bio_integrity_generate(bio); - else - fs_bio_integrity_alloc(bio); - } - if (dio->flags & IOMAP_DIO_WRITE) task_io_account_write(ret); else if ((dio->flags & IOMAP_DIO_USER_BACKED) && @@ -397,7 +397,7 @@ static ssize_t iomap_dio_bio_iter_one(struct iomap_iter *iter, out_bio_release_pages: if (dio->flags & IOMAP_DIO_BOUNCE) - bio_iov_iter_unbounce(bio, true, false); + bio_free_folios(bio); else bio_release_pages(bio, false); out_put_bio: @@ -505,7 +505,7 @@ static int iomap_dio_bio_iter(struct iomap_iter *iter, struct iomap_dio *dio) * We can only do inline completion for pure overwrites that * don't require additional I/O at completion time. * - * This rules out writes that need zeroing or metdata updates to + * This rules out writes that need zeroing or metadata updates to * convert unwritten or shared extents. * * Writes that extend i_size are also not supported, but this is @@ -1034,9 +1034,9 @@ ssize_t __iomap_dio_read_simple(struct kiocb *iocb, struct iov_iter *iter, bio->bi_iter.bi_sector = iomap_sector(&iomi->iomap, iomi->pos); bio->bi_ioprio = iocb->ki_ioprio; - ret = bio_iov_iter_get_pages(bio, iter, - bdev_dma_alignment(bio->bi_bdev), - alignment - 1); + ret = bio_iov_iter_get_pages(bio, iter, BIO_MAX_SIZE, + bdev_dma_alignment(bio->bi_bdev), + alignment - 1); if (unlikely(ret)) goto out_bio_put; diff --git a/fs/iomap/ioend.c b/fs/iomap/ioend.c index 7bbbb417f915..bbebecc31670 100644 --- a/fs/iomap/ioend.c +++ b/fs/iomap/ioend.c @@ -25,6 +25,7 @@ struct iomap_ioend *iomap_init_ioend(struct inode *inode, ioend->io_parent = NULL; INIT_LIST_HEAD(&ioend->io_list); ioend->io_flags = ioend_flags; + ioend->io_bvec_offset = bio->bi_iter.bi_offset; ioend->io_inode = inode; ioend->io_offset = file_offset; ioend->io_size = bio->bi_iter.bi_size; @@ -149,7 +150,7 @@ int iomap_ioend_writeback_submit(struct iomap_writepage_ctx *wpc, int error) return error; } - if (wpc->iomap.flags & IOMAP_F_INTEGRITY) + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) fs_bio_integrity_generate(&ioend->io_bio); submit_bio(&ioend->io_bio); return 0; @@ -215,7 +216,7 @@ ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, { struct iomap_ioend *ioend = wpc->wb_ctx; size_t poff = offset_in_folio(folio, pos); - unsigned int ioend_flags = 0; + unsigned int ioend_flags = iomap_ioend_flags(&wpc->iomap); unsigned int map_len = min_t(u64, dirty_len, wpc->iomap.offset + wpc->iomap.length - pos); int error; @@ -225,20 +226,16 @@ ssize_t iomap_add_to_ioend(struct iomap_writepage_ctx *wpc, struct folio *folio, WARN_ON_ONCE(!folio->private && map_len < dirty_len); switch (wpc->iomap.type) { + case IOMAP_HOLE: + return map_len; case IOMAP_UNWRITTEN: - ioend_flags |= IOMAP_IOEND_UNWRITTEN; - break; case IOMAP_MAPPED: break; - case IOMAP_HOLE: - return map_len; default: WARN_ON_ONCE(1); return -EIO; } - if (wpc->iomap.flags & IOMAP_F_SHARED) - ioend_flags |= IOMAP_IOEND_SHARED; if (pos == wpc->iomap.offset && (wpc->iomap.flags & IOMAP_F_BOUNDARY)) ioend_flags |= IOMAP_IOEND_BOUNDARY; @@ -312,6 +309,16 @@ new_ioend: } EXPORT_SYMBOL_GPL(iomap_add_to_ioend); +#ifdef CONFIG_BLK_DEV_INTEGRITY +int iomap_ioend_integrity_verify(struct iomap_ioend *ioend) +{ + struct bvec_iter data_iter = BVEC_ITER_IOEND(ioend); + + return fs_bio_integrity_verify(&ioend->io_bio, &data_iter); +} +EXPORT_SYMBOL_GPL(iomap_ioend_integrity_verify); +#endif /* CONFIG_BLK_DEV_INTEGRITY */ + static u32 iomap_finish_ioend(struct iomap_ioend *ioend, int error) { if (ioend->io_parent) { @@ -327,13 +334,6 @@ static u32 iomap_finish_ioend(struct iomap_ioend *ioend, int error) if (!atomic_dec_and_test(&ioend->io_remaining)) return 0; - if (!ioend->io_error && - bio_integrity(&ioend->io_bio) && - bio_op(&ioend->io_bio) == REQ_OP_READ) { - ioend->io_error = fs_bio_integrity_verify(&ioend->io_bio, - ioend->io_sector, ioend->io_size); - } - if (ioend->io_flags & IOMAP_IOEND_DIRECT) return iomap_finish_ioend_direct(ioend); if (bio_op(&ioend->io_bio) == REQ_OP_READ) @@ -512,6 +512,96 @@ struct iomap_ioend *iomap_split_ioend(struct iomap_ioend *ioend, } EXPORT_SYMBOL_GPL(iomap_split_ioend); +void iomap_bounce_read(struct iomap_ioend *orig_ioend, unsigned int minsize, + void (*submit_ioend)(struct iomap_ioend *ioend)) +{ + struct inode *inode = orig_ioend->io_inode; + struct bio *orig_bio = &orig_ioend->io_bio; + loff_t file_offset = orig_ioend->io_offset; + sector_t sector = orig_ioend->io_sector; + size_t total_len = round_up(orig_ioend->io_size, minsize); + + WARN_ON_ONCE(!(orig_ioend->io_flags & IOMAP_IOEND_DIRECT)); + + /* We can't poll a bio that is not passed on to hardware */ + orig_bio->bi_opf &= ~REQ_POLLED; + + do { + struct iomap_ioend *ioend; + struct bio *bio; + int error; + + bio = bio_alloc_bioset(orig_bio->bi_bdev, + min(total_len / minsize, BIO_MAX_VECS), + orig_bio->bi_opf, GFP_KERNEL, + &iomap_ioend_split_bioset); + error = bio_alloc_bounce_folios(bio, total_len, minsize); + if (error) { + bio_put(bio); + orig_bio->bi_status = errno_to_blk_status(error); + break; + } + bio->bi_ioprio = orig_bio->bi_ioprio; + bio->bi_write_hint = orig_bio->bi_write_hint; + bio->bi_write_stream = orig_bio->bi_write_stream; + bio->bi_iter.bi_sector = sector; + + ioend = iomap_init_ioend(inode, bio, file_offset, + orig_ioend->io_flags); + + total_len -= bio->bi_iter.bi_size; + file_offset += bio->bi_iter.bi_size; + sector += (bio->bi_iter.bi_size >> SECTOR_SHIFT); + + bio->bi_private = orig_bio; + bio_inc_remaining(orig_bio); + submit_ioend(ioend); + } while (total_len > 0); + + bio_endio(&orig_ioend->io_bio); +} +EXPORT_SYMBOL_GPL(iomap_bounce_read); + +static void iomap_ioend_unbounce(struct iomap_ioend *orig_ioend, + struct iomap_ioend *ioend) +{ + struct bio *orig_bio = &orig_ioend->io_bio; + struct iov_iter to; + struct bio_vec *bv; + int i; + + iov_iter_bvec(&to, ITER_DEST, orig_bio->bi_io_vec, orig_bio->bi_vcnt, + orig_ioend->io_size); + to.iov_offset = orig_ioend->io_bvec_offset; + + if (ioend->io_offset != orig_ioend->io_offset) { + WARN_ON_ONCE(ioend->io_offset < orig_ioend->io_offset); + iov_iter_advance(&to, ioend->io_offset - orig_ioend->io_offset); + } + + /* copying to pinned pages should always work */ + bio_for_each_bvec_all(bv, &ioend->io_bio, i) + WARN_ON_ONCE(copy_to_iter(bvec_virt(bv), bv->bv_len, &to) != + bv->bv_len); +} + +void iomap_bounce_read_end_io(struct iomap_ioend *ioend, struct bio *orig_bio, + int error) +{ + if (error) + orig_bio->bi_status = errno_to_blk_status(error); + else + iomap_ioend_unbounce(iomap_ioend_from_bio(orig_bio), ioend); + + bio_free_folios(&ioend->io_bio); + if (bio_integrity(&ioend->io_bio)) + fs_bio_integrity_free(&ioend->io_bio); + bio_put(&ioend->io_bio); + + bio_endio(orig_bio); +} +EXPORT_SYMBOL_GPL(iomap_bounce_read_end_io); + static int __init iomap_ioend_init(void) { const unsigned int nr_mempool_entries = 4 * (PAGE_SIZE / SECTOR_SIZE); diff --git a/fs/jbd2/commit.c b/fs/jbd2/commit.c index 3029cb6f6d64..ebf6ba58ff4d 100644 --- a/fs/jbd2/commit.c +++ b/fs/jbd2/commit.c @@ -32,14 +32,14 @@ static void journal_end_buffer_io_sync(struct bio *bio) { struct buffer_head *bh; - bool uptodate = bio_endio_bh(bio, &bh); + bool success = bio_endio_bh(bio, &bh); struct buffer_head *orig_bh = bh->b_private; BUFFER_TRACE(bh, ""); - if (uptodate) - set_buffer_uptodate(bh); + if (success) + clear_buffer_write_io_error(bh); else - clear_buffer_uptodate(bh); + mark_buffer_write_io_error(bh); if (orig_bh) { clear_and_wake_up_bit(BH_Shadow, &orig_bh->b_state); } @@ -169,7 +169,7 @@ static int journal_wait_on_commit_record(journal_t *journal, clear_buffer_dirty(bh); wait_on_buffer(bh); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) ret = -EIO; put_bh(bh); /* One for getblk() */ @@ -330,9 +330,9 @@ static __u32 jbd2_checksum_data(__u32 crc32_sum, struct buffer_head *bh) char *addr; __u32 checksum; - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); checksum = crc32_be(crc32_sum, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); return checksum; } @@ -357,10 +357,10 @@ static void jbd2_block_tag_csum_set(journal_t *j, journal_block_tag_t *tag, return; seq = cpu_to_be32(sequence); - addr = kmap_local_folio(bh->b_folio, bh_offset(bh)); + addr = kmap_local_bh(bh); csum32 = jbd2_chksum(j->j_csum_seed, (__u8 *)&seq, sizeof(seq)); csum32 = jbd2_chksum(csum32, addr, bh->b_size); - kunmap_local(addr); + kunmap_local_bh(bh, addr); if (jbd2_has_feature_csum3(j)) tag3->t_checksum = cpu_to_be32(csum32); @@ -834,7 +834,7 @@ start_journal_io: wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; jbd2_unfile_log_bh(bh); stats.run.rs_blocks_logged++; @@ -877,7 +877,7 @@ start_journal_io: wait_on_buffer(bh); cond_resched(); - if (unlikely(!buffer_uptodate(bh))) + if (unlikely(buffer_write_io_error(bh))) err = -EIO; BUFFER_TRACE(bh, "ph5: control buffer writeout done: unfile"); diff --git a/fs/jbd2/journal.c b/fs/jbd2/journal.c index 00f5a98f3d4f..cda1ff8851dc 100644 --- a/fs/jbd2/journal.c +++ b/fs/jbd2/journal.c @@ -328,8 +328,6 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, { int do_escape = 0; struct buffer_head *new_bh; - struct folio *new_folio; - unsigned int new_offset; struct buffer_head *bh_in = jh2bh(jh_in); journal_t *journal = transaction->t_journal; @@ -349,24 +347,31 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* keep subsequent assertions sane */ atomic_set(&new_bh->b_count, 1); + /* + * b_frozen_data is slab memory, not page cache, so when we use it the + * shadow buffer gets no folio at all: b_folio stays NULL from the + * allocation and b_data points straight at the copy. Pointing it at + * the slab folio instead would hand its overloaded ->mapping to + * anything that goes looking for an address_space. + */ + spin_lock(&jh_in->b_state_lock); /* * If a new transaction has already done a buffer copy-out, then * we use that version of the data for the commit. */ if (jh_in->b_frozen_data) { - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); do_escape = jbd2_data_needs_escaping(jh_in->b_frozen_data); if (do_escape) jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } else { + struct folio *folio = bh_in->b_folio; + unsigned int offset = offset_in_folio(folio, bh_in->b_data); char *tmp; char *mapped_data; - new_folio = bh_in->b_folio; - new_offset = offset_in_folio(new_folio, bh_in->b_data); - mapped_data = kmap_local_folio(new_folio, new_offset); + mapped_data = kmap_local_folio(folio, offset); /* * Fire data frozen trigger if data already wasn't frozen. Do * this before checking for escaping, as the trigger may modify @@ -380,8 +385,10 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, /* * Do we need to do a data copy? */ - if (!do_escape) + if (!do_escape) { + folio_set_bh(new_bh, folio, offset); goto escape_done; + } spin_unlock(&jh_in->b_state_lock); tmp = kmalloc(bh_in->b_size, GFP_NOFS | __GFP_NOFAIL); @@ -392,7 +399,7 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, } jh_in->b_frozen_data = tmp; - memcpy_from_folio(tmp, new_folio, new_offset, bh_in->b_size); + memcpy_from_folio(tmp, folio, offset, bh_in->b_size); /* * This isn't strictly necessary, as we're using frozen * data for the escaping, but it keeps consistency with @@ -401,13 +408,11 @@ int jbd2_journal_write_metadata_buffer(transaction_t *transaction, jh_in->b_frozen_triggers = jh_in->b_triggers; copy_done: - new_folio = virt_to_folio(jh_in->b_frozen_data); - new_offset = offset_in_folio(new_folio, jh_in->b_frozen_data); jbd2_data_do_escape(jh_in->b_frozen_data); + new_bh->b_data = jh_in->b_frozen_data; } escape_done: - folio_set_bh(new_bh, new_folio, new_offset); new_bh->b_size = bh_in->b_size; new_bh->b_bdev = journal->j_dev; new_bh->b_blocknr = blocknr; @@ -882,7 +887,7 @@ int jbd2_fc_wait_bufs(journal_t *journal, int num_blks) * Update j_fc_off so jbd2_fc_release_bufs can release remain * buffer head. */ - if (unlikely(!buffer_uptodate(bh))) { + if (unlikely(buffer_write_io_error(bh))) { journal->j_fc_off = i + 1; return -EIO; } diff --git a/fs/jbd2/transaction.c b/fs/jbd2/transaction.c index 5cc7d097b2ac..85d84d909f78 100644 --- a/fs/jbd2/transaction.c +++ b/fs/jbd2/transaction.c @@ -920,7 +920,7 @@ static void jbd2_freeze_jh_data(struct journal_head *jh) char *source; struct buffer_head *bh = jh2bh(jh); - J_EXPECT_JH(jh, buffer_uptodate(bh), "Possible IO failure.\n"); + J_EXPECT_JH(jh, buffer_uptodate(bh), "Buffer not uptodate!\n"); source = kmap_local_folio(bh->b_folio, bh_offset(bh)); /* Fire data frozen trigger just before we copy the data */ jbd2_buffer_frozen_trigger(jh, source, jh->b_triggers); diff --git a/fs/jffs2/acl.c b/fs/jffs2/acl.c index f0f8a4f57add..7548f44bf327 100644 --- a/fs/jffs2/acl.c +++ b/fs/jffs2/acl.c @@ -228,7 +228,7 @@ static int __jffs2_set_acl(struct inode *inode, int xprefix, struct posix_acl *a return rc; } -int jffs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int rc, xprefix; diff --git a/fs/jffs2/acl.h b/fs/jffs2/acl.h index e976b8cb82cf..bc5df521633f 100644 --- a/fs/jffs2/acl.h +++ b/fs/jffs2/acl.h @@ -28,7 +28,7 @@ struct jffs2_acl_header { #ifdef CONFIG_JFFS2_FS_POSIX_ACL struct posix_acl *jffs2_get_acl(struct inode *inode, int type, bool rcu); -int jffs2_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int jffs2_init_acl_pre(struct inode *, struct inode *, umode_t *); extern int jffs2_init_acl_post(struct inode *); diff --git a/fs/jffs2/dir.c b/fs/jffs2/dir.c index 656c920864c5..23813f191281 100644 --- a/fs/jffs2/dir.c +++ b/fs/jffs2/dir.c @@ -25,20 +25,20 @@ static int jffs2_readdir (struct file *, struct dir_context *); -static int jffs2_create (struct mnt_idmap *, struct inode *, +static int jffs2_create (const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); static struct dentry *jffs2_lookup (struct inode *,struct dentry *, unsigned int); static int jffs2_link (struct dentry *,struct inode *,struct dentry *); static int jffs2_unlink (struct inode *,struct dentry *); -static int jffs2_symlink (struct mnt_idmap *, struct inode *, +static int jffs2_symlink (const struct mnt_idmap *, struct inode *, struct dentry *, const char *); -static struct dentry *jffs2_mkdir (struct mnt_idmap *, struct inode *,struct dentry *, +static struct dentry *jffs2_mkdir (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t); static int jffs2_rmdir (struct inode *,struct dentry *); -static int jffs2_mknod (struct mnt_idmap *, struct inode *,struct dentry *, +static int jffs2_mknod (const struct mnt_idmap *, struct inode *,struct dentry *, umode_t,dev_t); -static int jffs2_rename (struct mnt_idmap *, struct inode *, +static int jffs2_rename (const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); @@ -162,7 +162,7 @@ static int jffs2_readdir(struct file *file, struct dir_context *ctx) /***********************************************************************/ -static int jffs2_create(struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_create(const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode) { struct jffs2_raw_inode *ri; @@ -284,7 +284,7 @@ static int jffs2_link (struct dentry *old_dentry, struct inode *dir_i, struct de /***********************************************************************/ -static int jffs2_symlink (struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_symlink (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, const char *target) { struct jffs2_inode_info *f, *dir_f; @@ -448,7 +448,7 @@ static int jffs2_symlink (struct mnt_idmap *idmap, struct inode *dir_i, } -static struct dentry *jffs2_mkdir (struct mnt_idmap *idmap, struct inode *dir_i, +static struct dentry *jffs2_mkdir (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode) { struct jffs2_inode_info *f, *dir_f; @@ -620,7 +620,7 @@ static int jffs2_rmdir (struct inode *dir_i, struct dentry *dentry) return ret; } -static int jffs2_mknod (struct mnt_idmap *idmap, struct inode *dir_i, +static int jffs2_mknod (const struct mnt_idmap *idmap, struct inode *dir_i, struct dentry *dentry, umode_t mode, dev_t rdev) { struct jffs2_inode_info *f, *dir_f; @@ -769,7 +769,7 @@ static int jffs2_mknod (struct mnt_idmap *idmap, struct inode *dir_i, return ret; } -static int jffs2_rename (struct mnt_idmap *idmap, +static int jffs2_rename (const struct mnt_idmap *idmap, struct inode *old_dir_i, struct dentry *old_dentry, struct inode *new_dir_i, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/jffs2/fs.c b/fs/jffs2/fs.c index 6ada8369a762..05cf860307c7 100644 --- a/fs/jffs2/fs.c +++ b/fs/jffs2/fs.c @@ -190,7 +190,7 @@ int jffs2_do_setattr (struct inode *inode, struct iattr *iattr) return 0; } -int jffs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int jffs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/jffs2/os-linux.h b/fs/jffs2/os-linux.h index 86ab014a349c..bff2134d771d 100644 --- a/fs/jffs2/os-linux.h +++ b/fs/jffs2/os-linux.h @@ -164,7 +164,7 @@ long jffs2_ioctl(struct file *, unsigned int, unsigned long); extern const struct inode_operations jffs2_symlink_inode_operations; /* fs.c */ -int jffs2_setattr (struct mnt_idmap *, struct dentry *, struct iattr *); +int jffs2_setattr (const struct mnt_idmap *, struct dentry *, struct iattr *); int jffs2_do_setattr (struct inode *, struct iattr *); struct inode *jffs2_iget(struct super_block *, unsigned long); void jffs2_evict_inode (struct inode *); diff --git a/fs/jffs2/security.c b/fs/jffs2/security.c index 437f3a2c1b54..67330aeb8ae8 100644 --- a/fs/jffs2/security.c +++ b/fs/jffs2/security.c @@ -57,7 +57,7 @@ static int jffs2_security_getxattr(const struct xattr_handler *handler, } static int jffs2_security_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jffs2/xattr_trusted.c b/fs/jffs2/xattr_trusted.c index b7c5da2d89bd..85133ad8b449 100644 --- a/fs/jffs2/xattr_trusted.c +++ b/fs/jffs2/xattr_trusted.c @@ -25,7 +25,7 @@ static int jffs2_trusted_getxattr(const struct xattr_handler *handler, } static int jffs2_trusted_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jffs2/xattr_user.c b/fs/jffs2/xattr_user.c index f64edce4927b..dcfd3caf1d8b 100644 --- a/fs/jffs2/xattr_user.c +++ b/fs/jffs2/xattr_user.c @@ -25,7 +25,7 @@ static int jffs2_user_getxattr(const struct xattr_handler *handler, } static int jffs2_user_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *buffer, size_t size, int flags) diff --git a/fs/jfs/acl.c b/fs/jfs/acl.c index 16b71a23ff1e..6e0a7feb6c80 100644 --- a/fs/jfs/acl.c +++ b/fs/jfs/acl.c @@ -89,7 +89,7 @@ static int __jfs_set_acl(tid_t tid, struct inode *inode, int type, return rc; } -int jfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int rc; diff --git a/fs/jfs/file.c b/fs/jfs/file.c index 246568cb9a6e..2f5bb79c0591 100644 --- a/fs/jfs/file.c +++ b/fs/jfs/file.c @@ -89,7 +89,7 @@ static int jfs_release(struct inode *inode, struct file *file) return 0; } -int jfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); diff --git a/fs/jfs/ioctl.c b/fs/jfs/ioctl.c index 563f148be8af..27d39cddaca6 100644 --- a/fs/jfs/ioctl.c +++ b/fs/jfs/ioctl.c @@ -70,7 +70,7 @@ int jfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int jfs_fileattr_set(struct mnt_idmap *idmap, +int jfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/jfs/jfs_acl.h b/fs/jfs/jfs_acl.h index f892e54d0fcd..bda26b333519 100644 --- a/fs/jfs/jfs_acl.h +++ b/fs/jfs/jfs_acl.h @@ -8,7 +8,7 @@ #ifdef CONFIG_JFS_POSIX_ACL struct posix_acl *jfs_get_acl(struct inode *inode, int type, bool rcu); -int jfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int jfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int jfs_init_acl(tid_t, struct inode *, struct inode *); diff --git a/fs/jfs/jfs_inode.h b/fs/jfs/jfs_inode.h index 2c6c81c8cb9f..5a118b07fbff 100644 --- a/fs/jfs/jfs_inode.h +++ b/fs/jfs/jfs_inode.h @@ -10,7 +10,7 @@ struct fid; extern struct inode *ialloc(struct inode *, umode_t); extern int jfs_fsync(struct file *, loff_t, loff_t, int); extern int jfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -extern int jfs_fileattr_set(struct mnt_idmap *idmap, +extern int jfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); extern long jfs_ioctl(struct file *, unsigned int, unsigned long); extern struct inode *jfs_iget(struct super_block *, unsigned long); @@ -28,7 +28,7 @@ extern struct dentry *jfs_fh_to_parent(struct super_block *sb, struct fid *fid, int fh_len, int fh_type); extern void jfs_set_inode_flags(struct inode *); extern int jfs_get_block(struct inode *, sector_t, struct buffer_head *, int); -extern int jfs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +extern int jfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern const struct address_space_operations jfs_aops; extern const struct inode_operations jfs_dir_inode_operations; diff --git a/fs/jfs/namei.c b/fs/jfs/namei.c index 8a36c218f0f7..8ab2e952ce16 100644 --- a/fs/jfs/namei.c +++ b/fs/jfs/namei.c @@ -60,7 +60,7 @@ static inline void free_ea_wmap(struct inode *inode) * RETURN: Errors from subroutines * */ -static int jfs_create(struct mnt_idmap *idmap, struct inode *dip, +static int jfs_create(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, umode_t mode) { int rc = 0; @@ -193,7 +193,7 @@ static int jfs_create(struct mnt_idmap *idmap, struct inode *dip, * note: * EACCES: user needs search+write permission on the parent directory */ -static struct dentry *jfs_mkdir(struct mnt_idmap *idmap, struct inode *dip, +static struct dentry *jfs_mkdir(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, umode_t mode) { int rc = 0; @@ -876,7 +876,7 @@ static int jfs_link(struct dentry *old_dentry, * an intermediate result whose length exceeds PATH_MAX [XPG4.2] */ -static int jfs_symlink(struct mnt_idmap *idmap, struct inode *dip, +static int jfs_symlink(const struct mnt_idmap *idmap, struct inode *dip, struct dentry *dentry, const char *name) { int rc; @@ -1066,7 +1066,7 @@ static int jfs_symlink(struct mnt_idmap *idmap, struct inode *dip, * * FUNCTION: rename a file or directory */ -static int jfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int jfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1355,7 +1355,7 @@ static int jfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, * * FUNCTION: Create a special file (device) */ -static int jfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int jfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct jfs_inode_info *jfs_ip; diff --git a/fs/jfs/xattr.c b/fs/jfs/xattr.c index 11d7f74d207b..dcc4a69d44fe 100644 --- a/fs/jfs/xattr.c +++ b/fs/jfs/xattr.c @@ -956,7 +956,7 @@ static int jfs_xattr_get(const struct xattr_handler *handler, } static int jfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -975,7 +975,7 @@ static int jfs_xattr_get_os2(const struct xattr_handler *handler, } static int jfs_xattr_set_os2(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/kernfs/dir.c b/fs/kernfs/dir.c index 82bbaeb326aa..324d61a00545 100644 --- a/fs/kernfs/dir.c +++ b/fs/kernfs/dir.c @@ -30,6 +30,8 @@ static char kernfs_pr_cont_buf[PATH_MAX]; /* protected by pr_cont_lock */ #define rb_to_kn(X) rb_entry((X), struct kernfs_node, rb) +static void kernfs_activate_one(struct kernfs_node *kn); + static bool __kernfs_active(struct kernfs_node *kn) { return atomic_read(&kn->active) >= 0; @@ -736,13 +738,19 @@ struct kernfs_node *kernfs_new_node(struct kernfs_node *parent, { struct kernfs_node *kn; - if (parent->mode & S_ISGID) { + /* + * The mode and the gid below are read unlocked on purpose: they feed + * a node that does not exist yet, so nothing orders a racing chmod or + * chown against this creation. + */ + if (READ_ONCE(parent->mode) & S_ISGID) { /* this code block imitates inode_init_owner() for * kernfs */ + struct kernfs_iattrs *attrs = READ_ONCE(parent->iattr); - if (parent->iattr) - gid = parent->iattr->ia_gid; + if (attrs) + gid = READ_ONCE(attrs->ia_gid); if (flags & KERNFS_DIR) mode |= S_ISGID; @@ -855,7 +863,6 @@ int kernfs_add_one(struct kernfs_node *kn) } up_write(&root->kernfs_iattr_rwsem); - up_write(&root->kernfs_rwsem); /* * Activate the new node unless CREATE_DEACTIVATED is requested. @@ -863,9 +870,15 @@ int kernfs_add_one(struct kernfs_node *kn) * activating the node with kernfs_activate(). A node which hasn't * been activated is not visible to userland and its removal won't * trigger deactivation. + * + * @kn has no children yet, so kernfs_activate() would walk only @kn. + * Do it here rather than dropping the write lock and taking it again + * for every new node. */ - if (!(kernfs_root(kn)->flags & KERNFS_ROOT_CREATE_DEACTIVATED)) - kernfs_activate(kn); + if (!(root->flags & KERNFS_ROOT_CREATE_DEACTIVATED)) + kernfs_activate_one(kn); + + up_write(&root->kernfs_rwsem); return 0; out_unlock: @@ -1171,23 +1184,18 @@ struct kernfs_node *kernfs_create_empty_dir(struct kernfs_node *parent, static int kernfs_dop_revalidate(struct inode *dir, const struct qstr *name, struct dentry *dentry, unsigned int flags) { - struct kernfs_node *kn, *parent; - struct kernfs_root *root; + struct kernfs_node *parent = dir->i_private; + struct kernfs_node *kn; + const char *kn_name; if (flags & LOOKUP_RCU) return -ECHILD; /* Negative hashed dentry? */ if (d_really_is_negative(dentry)) { - /* If the kernfs parent node has changed discard and - * proceed to ->lookup. - * - * There's nothing special needed here when getting the - * dentry parent, even if a concurrent rename is in - * progress. That's because the dentry is negative so - * it can only be the target of the rename and it will - * be doing a d_move() not a replace. Consequently the - * dentry d_parent won't change over the d_move(). + /* + * If the kernfs parent node has changed discard and proceed to + * ->lookup. * * Also kernfs negative dentries transitioning from * negative to positive during revalidate won't happen @@ -1195,50 +1203,41 @@ static int kernfs_dop_revalidate(struct inode *dir, const struct qstr *name, * changes and the lookup re-done so that a new positive * dentry can be properly created. */ - root = kernfs_root_from_sb(dentry->d_sb); - down_read(&root->kernfs_rwsem); - parent = kernfs_dentry_node(dentry->d_parent); - if (parent) { - if (kernfs_dir_changed(parent, dentry)) { - up_read(&root->kernfs_rwsem); - return 0; - } - } - up_read(&root->kernfs_rwsem); - - /* The kernfs parent node hasn't changed, leave the - * dentry negative and return success. - */ - return 1; + return !kernfs_dir_changed(parent, dentry); } kn = kernfs_dentry_node(dentry); - root = kernfs_root(kn); - down_read(&root->kernfs_rwsem); + + guard(rcu)(); /* The kernfs node has been deactivated */ - if (!kernfs_active(kn)) - goto out_bad; + if (!__kernfs_active(kn)) + return 0; - parent = kernfs_parent(kn); /* The kernfs node has been moved? */ - if (kernfs_dentry_node(dentry->d_parent) != parent) - goto out_bad; + if (kernfs_parent(kn) != parent) + return 0; /* The kernfs node has been renamed */ - if (strcmp(dentry->d_name.name, kernfs_rcu_name(kn)) != 0) - goto out_bad; + kn_name = kernfs_rcu_name(kn); + if (name->len != strlen(kn_name) || + memcmp(name->name, kn_name, name->len)) + return 0; - /* The kernfs node has been moved to a different namespace */ - if (parent && kernfs_ns_enabled(parent) && - kernfs_ns_id(kernfs_info(dentry->d_sb)->ns) != kernfs_ns_id(kn->ns)) - goto out_bad; + /* + * The kernfs node has been moved to a different namespace. + * + * KERNFS_NS is set by kernfs_enable_ns() while @parent still has no + * children, so it cannot change while a child of @parent is being + * revalidated. The other bits in that word, KERNFS_ACTIVATED and + * KERNFS_REMOVING, are updated under kernfs_rwsem and are not read + * here, so racing with them is intentional and harmless. + */ + if (data_race(kernfs_ns_enabled(parent)) && + kernfs_info(dir->i_sb)->ns != READ_ONCE(kn->ns)) + return 0; - up_read(&root->kernfs_rwsem); return 1; -out_bad: - up_read(&root->kernfs_rwsem); - return 0; } const struct dentry_operations kernfs_dops = { @@ -1288,7 +1287,7 @@ static struct dentry *kernfs_iop_lookup(struct inode *dir, return d_splice_alias(inode, dentry); } -static struct dentry *kernfs_iop_mkdir(struct mnt_idmap *idmap, +static struct dentry *kernfs_iop_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { @@ -1326,7 +1325,7 @@ static int kernfs_iop_rmdir(struct inode *dir, struct dentry *dentry) return ret; } -static int kernfs_iop_rename(struct mnt_idmap *idmap, +static int kernfs_iop_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1820,14 +1819,20 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, const char *new_name, const struct ns_common *new_ns) { struct kernfs_node *old_parent; + const char *dup_name = NULL; + const char *put_name = NULL; struct kernfs_root *root; const char *old_name; + bool reparent; int error; /* can't move or rename root */ if (!rcu_access_pointer(kn->__parent)) return -EINVAL; + if (new_name) + dup_name = kstrdup_const(new_name, GFP_KERNEL); + root = kernfs_root(kn); down_write(&root->kernfs_rwsem); @@ -1859,9 +1864,10 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, /* rename kernfs_node */ if (strcmp(old_name, new_name) != 0) { error = -ENOMEM; - new_name = kstrdup_const(new_name, GFP_KERNEL); - if (!new_name) + if (!dup_name) goto out; + new_name = dup_name; + dup_name = NULL; } else { new_name = NULL; } @@ -1871,35 +1877,39 @@ int kernfs_rename_ns(struct kernfs_node *kn, struct kernfs_node *new_parent, */ kernfs_unlink_sibling(kn); - /* rename_lock protects ->parent accessors */ - if (old_parent != new_parent) { + reparent = old_parent != new_parent; + if (reparent) kernfs_get(new_parent); - write_lock_irq(&root->kernfs_rename_lock); + /* + * kernfs_rename_lock protects ->__parent, ->ns and ->name, so take it + * even when the parent does not change. + */ + write_lock_irq(&root->kernfs_rename_lock); + + if (reparent) rcu_assign_pointer(kn->__parent, new_parent); + WRITE_ONCE(kn->ns, new_ns); + if (new_name) + rcu_assign_pointer(kn->name, new_name); - kn->ns = new_ns; - if (new_name) - rcu_assign_pointer(kn->name, new_name); + write_unlock_irq(&root->kernfs_rename_lock); - write_unlock_irq(&root->kernfs_rename_lock); + if (reparent) kernfs_put(old_parent); - } else { - /* name assignment is RCU protected, parent is the same */ - kn->ns = new_ns; - if (new_name) - rcu_assign_pointer(kn->name, new_name); - } kn->hash = kernfs_name_hash(new_name ?: old_name, kn->ns); kernfs_link_sibling(kn); if (new_name && !is_kernel_rodata((unsigned long)old_name)) - kfree_rcu_mightsleep(old_name); + put_name = old_name; error = 0; out: up_write(&root->kernfs_rwsem); + kfree_const(dup_name); + if (put_name) + kfree_rcu_mightsleep(put_name); return error; } @@ -1909,33 +1919,49 @@ static int kernfs_dir_fop_release(struct inode *inode, struct file *filp) return 0; } +/* + * Find where a listing left off. @resumed says whether @pos is still that + * entry; if not, the search falls back to @hash, keyed by @name if given. + */ static struct kernfs_node *kernfs_dir_pos(const struct ns_common *ns, - struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos) + struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos, + const char *name, bool *resumed) { + if (resumed) + *resumed = false; if (pos) { + /* + * A rename keeps the hash if the new name hashes the same, so + * check @name too. Otherwise the caller would step over the + * entry now sitting where @pos used to be. + */ int valid = kernfs_active(pos) && rcu_access_pointer(pos->__parent) == parent && - hash == pos->hash; + hash == pos->hash && + (!name || !strcmp(name, kernfs_rcu_name(pos))); kernfs_put(pos); if (!valid) pos = NULL; + else if (resumed) + *resumed = true; } if (!pos && (hash > 1) && (hash < INT_MAX)) { struct rb_node *node = parent->dir.children.rb_node; - u64 ns_id = kernfs_ns_id(ns); + + /* + * Keep a node only on the way left, so the search ends on the + * first entry after the key. An empty @name sorts before all + * entries sharing the hash, so it lands on the first of them. + */ while (node) { - pos = rb_to_kn(node); + struct kernfs_node *kn = rb_to_kn(node); - if (hash < pos->hash) + if (kernfs_name_compare(hash, name ?: "", ns, kn) < 0) { + pos = kn; node = node->rb_left; - else if (hash > pos->hash) + } else { node = node->rb_right; - else if (ns_id < kernfs_ns_id(pos->ns)) - node = node->rb_left; - else if (ns_id > kernfs_ns_id(pos->ns)) - node = node->rb_right; - else - break; + } } } /* Skip over entries which are dying/dead or in the wrong namespace */ @@ -1951,10 +1977,14 @@ static struct kernfs_node *kernfs_dir_pos(const struct ns_common *ns, } static struct kernfs_node *kernfs_dir_next_pos(const struct ns_common *ns, - struct kernfs_node *parent, ino_t ino, struct kernfs_node *pos) + struct kernfs_node *parent, loff_t hash, struct kernfs_node *pos, + const char *name) { - pos = kernfs_dir_pos(ns, parent, ino, pos); - if (pos) { + bool resumed; + + pos = kernfs_dir_pos(ns, parent, hash, pos, name, &resumed); + /* Step over @pos only if it survived; @name finds the spot if not. */ + if (pos && resumed) { do { struct rb_node *node = rb_next(&pos->rb); if (!node) @@ -1972,34 +2002,55 @@ static int kernfs_fop_readdir(struct file *file, struct dir_context *ctx) struct dentry *dentry = file->f_path.dentry; struct kernfs_node *parent = kernfs_dentry_node(dentry); struct kernfs_node *pos = file->private_data; + char *name __free(kfree) = NULL; struct kernfs_root *root; const struct ns_common *ns = NULL; if (!dir_emit_dots(file, ctx)) return 0; + /* + * One buffer for the call, holding the name of the entry the listing + * is on. PATH_MAX: kernfs bounds no single name. + */ + name = kmalloc(PATH_MAX, GFP_KERNEL); + if (!name) + return -ENOMEM; + root = kernfs_root(parent); down_read(&root->kernfs_rwsem); if (kernfs_ns_enabled(parent)) ns = kernfs_info(dentry->d_sb)->ns; - for (pos = kernfs_dir_pos(ns, parent, ctx->pos, pos); + for (pos = kernfs_dir_pos(ns, parent, ctx->pos, pos, NULL, NULL); pos; - pos = kernfs_dir_next_pos(ns, parent, ctx->pos, pos)) { - const char *name = kernfs_rcu_name(pos); + pos = kernfs_dir_next_pos(ns, parent, ctx->pos, pos, name)) { unsigned int type = fs_umode_to_dtype(pos->mode); - int len = strlen(name); ino_t ino = kernfs_ino(pos); + int len; + + /* + * The copy is also the resume key, so a truncated name would + * resume here again. getname() caps a path, so only an + * in-kernel caller can get here; end the listing instead. + */ + len = strscpy(name, kernfs_rcu_name(pos), PATH_MAX); + if (WARN_ON_ONCE(len < 0)) + break; ctx->pos = pos->hash; file->private_data = pos; kernfs_get(pos); - if (!dir_emit(ctx, name, len, ino, type)) { - up_read(&root->kernfs_rwsem); + /* + * dir_emit() can fault, so run it unlocked. @pos is pinned + * above and kernfs_dir_pos() rechecks it on the way back. + */ + up_read(&root->kernfs_rwsem); + if (!dir_emit(ctx, name, len, ino, type)) return 0; - } + down_read(&root->kernfs_rwsem); } up_read(&root->kernfs_rwsem); file->private_data = NULL; diff --git a/fs/kernfs/file.c b/fs/kernfs/file.c index 8e0e90c93372..cca9f83fc9b5 100644 --- a/fs/kernfs/file.c +++ b/fs/kernfs/file.c @@ -525,18 +525,31 @@ out_unlock: static int kernfs_get_open_node(struct kernfs_node *kn, struct kernfs_open_file *of) { - struct kernfs_open_node *on; + struct kernfs_open_node *on, *new_on = NULL; struct mutex *mutex; + /* + * Peek without the mutex: if nothing has this open, we will need a + * node and can allocate before taking a mutex shared by every node + * hashing to it. + */ + if (!rcu_access_pointer(kn->attr.open)) + new_on = kzalloc_obj(*new_on); + mutex = kernfs_open_file_mutex_lock(kn); on = kernfs_deref_open_node_locked(kn); if (!on) { /* not there, initialize a new one */ - on = kzalloc_obj(*on); + on = new_on; + new_on = NULL; if (!on) { - mutex_unlock(mutex); - return -ENOMEM; + /* the peek raced; rare, so allocate here */ + on = kzalloc_obj(*on); + if (!on) { + mutex_unlock(mutex); + return -ENOMEM; + } } atomic_set(&on->event, 1); init_waitqueue_head(&on->poll); @@ -549,6 +562,7 @@ static int kernfs_get_open_node(struct kernfs_node *kn, on->nr_to_release++; mutex_unlock(mutex); + kfree(new_on); return 0; } @@ -904,9 +918,12 @@ static loff_t kernfs_fop_llseek(struct file *file, loff_t offset, int whence) static void kernfs_notify_workfn(struct work_struct *work) { - struct kernfs_node *kn; + char name_buf[NAME_MAX + 1]; struct kernfs_super_info *info; + struct kernfs_node *kn; struct kernfs_root *root; + struct qstr name; + bool have_name; repeat: /* pop one off the notify_list */ spin_lock_irq(&kernfs_notify_lock); @@ -922,14 +939,20 @@ repeat: root = kernfs_root(kn); /* kick fsnotify */ + /* + * Sample the name once so kernfs_rwsem need not be held across the + * loop. A name that does not fit is reported without one; fsnotify() + * takes the name as optional, so a watcher loses the name and not the + * event. + */ + have_name = kernfs_name(kn, name_buf, sizeof(name_buf)) >= 0; + name = QSTR(name_buf); + down_read(&root->kernfs_supers_rwsem); - down_read(&root->kernfs_rwsem); - list_for_each_entry(info, &kernfs_root(kn)->supers, node) { + list_for_each_entry(info, &root->supers, node) { struct kernfs_node *parent; struct inode *p_inode = NULL; - const char *kn_name; struct inode *inode; - struct qstr name; /* * We want fsnotify_modify() on @kn but as the @@ -941,15 +964,14 @@ repeat: if (!inode) continue; - kn_name = kernfs_rcu_name(kn); - name = QSTR(kn_name); parent = kernfs_get_parent(kn); if (parent) { p_inode = ilookup(info->sb, kernfs_ino(parent)); if (p_inode) { fsnotify(FS_MODIFY | FS_EVENT_ON_CHILD, inode, FSNOTIFY_EVENT_INODE, - p_inode, &name, inode, 0); + p_inode, have_name ? &name : NULL, + inode, 0); iput(p_inode); } @@ -962,7 +984,6 @@ repeat: iput(inode); } - up_read(&root->kernfs_rwsem); up_read(&root->kernfs_supers_rwsem); kernfs_put(kn); goto repeat; diff --git a/fs/kernfs/inode.c b/fs/kernfs/inode.c index abb286bc3474..6630b29d7c07 100644 --- a/fs/kernfs/inode.c +++ b/fs/kernfs/inode.c @@ -107,7 +107,7 @@ int kernfs_setattr(struct kernfs_node *kn, const struct iattr *iattr) return ret; } -int kernfs_iop_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int kernfs_iop_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -179,7 +179,7 @@ static void kernfs_refresh_inode(struct kernfs_node *kn, struct inode *inode) set_nlink(inode, kn->dir.subdirs + 2); } -int kernfs_iop_getattr(struct mnt_idmap *idmap, +int kernfs_iop_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -270,7 +270,7 @@ void kernfs_evict_inode(struct inode *inode) kernfs_put(kn); } -int kernfs_iop_permission(struct mnt_idmap *idmap, +int kernfs_iop_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct kernfs_node *kn; @@ -342,7 +342,7 @@ static int kernfs_vfs_xattr_get(const struct xattr_handler *handler, } static int kernfs_vfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) @@ -354,7 +354,7 @@ static int kernfs_vfs_xattr_set(const struct xattr_handler *handler, } static int kernfs_vfs_user_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) diff --git a/fs/kernfs/kernfs-internal.h b/fs/kernfs/kernfs-internal.h index aa784b540b36..f1e93b09e27d 100644 --- a/fs/kernfs/kernfs-internal.h +++ b/fs/kernfs/kernfs-internal.h @@ -117,7 +117,14 @@ static inline bool kernfs_rename_is_locked(const struct kernfs_node *kn) static inline const char *kernfs_rcu_name(const struct kernfs_node *kn) { - return rcu_dereference_check(kn->name, kernfs_root_is_locked(kn)); + /* + * Like kernfs_node::__parent below, the name is only replaced under + * both kernfs_root::kernfs_rwsem and kernfs_root::kernfs_rename_lock, + * so either one keeps it, and the string it points at, stable. + */ + return rcu_dereference_check(kn->name, + kernfs_root_is_locked(kn) || + kernfs_rename_is_locked(kn)); } static inline struct kernfs_node *kernfs_parent(const struct kernfs_node *kn) @@ -147,20 +154,19 @@ static inline struct kernfs_node *kernfs_dentry_node(struct dentry *dentry) static inline void kernfs_set_rev(struct kernfs_node *parent, struct dentry *dentry) { - dentry->d_time = parent->dir.rev; + WRITE_ONCE(dentry->d_time, READ_ONCE(parent->dir.rev)); } static inline void kernfs_inc_rev(struct kernfs_node *parent) { - parent->dir.rev++; + lockdep_assert_held_write(&parent->dir.root->kernfs_rwsem); + WRITE_ONCE(parent->dir.rev, parent->dir.rev + 1); } static inline bool kernfs_dir_changed(struct kernfs_node *parent, struct dentry *dentry) { - if (parent->dir.rev != dentry->d_time) - return true; - return false; + return READ_ONCE(parent->dir.rev) != READ_ONCE(dentry->d_time); } extern const struct super_operations kernfs_sops; @@ -171,11 +177,11 @@ extern struct kmem_cache *kernfs_node_cache, *kernfs_iattrs_cache; */ extern const struct xattr_handler * const kernfs_xattr_handlers[]; void kernfs_evict_inode(struct inode *inode); -int kernfs_iop_permission(struct mnt_idmap *idmap, +int kernfs_iop_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); -int kernfs_iop_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int kernfs_iop_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); -int kernfs_iop_getattr(struct mnt_idmap *idmap, +int kernfs_iop_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags); ssize_t kernfs_iop_listxattr(struct dentry *dentry, char *buf, size_t size); diff --git a/fs/kernfs/mount.c b/fs/kernfs/mount.c index a57399021c8b..a0be784bfb06 100644 --- a/fs/kernfs/mount.c +++ b/fs/kernfs/mount.c @@ -124,22 +124,32 @@ static struct dentry *__kernfs_fh_to_dentry(struct super_block *sb, return NULL; } - kn = kernfs_find_and_get_node_by_id(info->root, id); - if (!kn) - return ERR_PTR(-ESTALE); + /* + * Hold kernfs_rwsem across the lookup as well as kernfs_get_inode(). + * __kernfs_remove() deactivates the subtree and clears i_nlink on its + * inodes under the write lock, so under the read lock either + * kernfs_find_and_get_node_by_id() refuses the node, or the inode is + * in the inode hash before the ilookup() pass goes looking for it. + */ + scoped_guard(rwsem_read, &info->root->kernfs_rwsem) { + kn = kernfs_find_and_get_node_by_id(info->root, id); + if (!kn) + return ERR_PTR(-ESTALE); - if (get_parent) { - struct kernfs_node *parent; + if (get_parent) { + struct kernfs_node *parent; - parent = kernfs_get_parent(kn); + parent = kernfs_get_parent(kn); + kernfs_put(kn); + kn = parent; + if (!kn) + return ERR_PTR(-ESTALE); + } + + inode = kernfs_get_inode(sb, kn); kernfs_put(kn); - kn = parent; - if (!kn) - return ERR_PTR(-ESTALE); } - inode = kernfs_get_inode(sb, kn); - kernfs_put(kn); return d_obtain_alias(inode); } diff --git a/fs/kernfs/symlink.c b/fs/kernfs/symlink.c index 90e2b3221b83..3e53105d3abf 100644 --- a/fs/kernfs/symlink.c +++ b/fs/kernfs/symlink.c @@ -31,9 +31,20 @@ struct kernfs_node *kernfs_create_link(struct kernfs_node *parent, kuid_t uid = GLOBAL_ROOT_UID; kgid_t gid = GLOBAL_ROOT_GID; - if (target->iattr) { - uid = target->iattr->ia_uid; - gid = target->iattr->ia_gid; + /* + * A symlink takes its owner from its target, so both fields have to + * come from the same moment: read them under kernfs_iattr_rwsem, or + * a chown of the target racing this could leave the link with the + * old uid and the new gid. The section ends before kernfs_add_one() + * takes kernfs_rwsem. + */ + scoped_guard(rwsem_read, &kernfs_root(target)->kernfs_iattr_rwsem) { + struct kernfs_iattrs *attrs = READ_ONCE(target->iattr); + + if (attrs) { + uid = attrs->ia_uid; + gid = attrs->ia_gid; + } } kn = kernfs_new_node(parent, name, S_IFLNK|0777, uid, gid, KERNFS_LINK); diff --git a/fs/libfs.c b/fs/libfs.c index 27d7dc16fcb0..8e2cc627bc7f 100644 --- a/fs/libfs.c +++ b/fs/libfs.c @@ -29,7 +29,7 @@ #include "internal.h" -int simple_getattr(struct mnt_idmap *idmap, const struct path *path, +int simple_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -867,7 +867,7 @@ int simple_rename_exchange(struct inode *old_dir, struct dentry *old_dentry, } EXPORT_SYMBOL_GPL(simple_rename_exchange); -int simple_rename(struct mnt_idmap *idmap, struct inode *old_dir, +int simple_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -913,7 +913,7 @@ EXPORT_SYMBOL(simple_rename); * on simple regular filesystems. Anything that needs to change on-disk * or wire state on size changes needs its own setattr method. */ -int simple_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int simple_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -1727,7 +1727,7 @@ static struct dentry *empty_dir_lookup(struct inode *dir, struct dentry *dentry, return ERR_PTR(-ENOENT); } -static int empty_dir_setattr(struct mnt_idmap *idmap, +static int empty_dir_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return -EPERM; diff --git a/fs/minix/file.c b/fs/minix/file.c index 02aabbdb5dea..c0929fc38fbc 100644 --- a/fs/minix/file.c +++ b/fs/minix/file.c @@ -23,7 +23,7 @@ const struct file_operations minix_file_operations = { .splice_read = filemap_splice_read, }; -static int minix_setattr(struct mnt_idmap *idmap, +static int minix_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/minix/inode.c b/fs/minix/inode.c index daf83e4ff25c..670179173645 100644 --- a/fs/minix/inode.c +++ b/fs/minix/inode.c @@ -724,7 +724,7 @@ out: return err; } -int minix_getattr(struct mnt_idmap *idmap, const struct path *path, +int minix_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct super_block *sb = path->dentry->d_sb; diff --git a/fs/minix/minix.h b/fs/minix/minix.h index 78722ce22e1e..db92cf9e0e1b 100644 --- a/fs/minix/minix.h +++ b/fs/minix/minix.h @@ -55,7 +55,7 @@ unsigned long minix_count_free_inodes(struct super_block *sb); int minix_new_block(struct inode *inode); void minix_free_block(struct inode *inode, unsigned long block); unsigned long minix_count_free_blocks(struct super_block *sb); -int minix_getattr(struct mnt_idmap *, const struct path *, +int minix_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); int minix_prepare_chunk(struct folio *folio, loff_t pos, unsigned len); struct mapping_metadata_bhs *minix_get_metadata_bhs(struct inode *inode); diff --git a/fs/minix/namei.c b/fs/minix/namei.c index 5525ba367ed7..f450b11b9860 100644 --- a/fs/minix/namei.c +++ b/fs/minix/namei.c @@ -33,7 +33,7 @@ static struct dentry *minix_lookup(struct inode * dir, struct dentry *dentry, un return d_splice_alias(inode, dentry); } -static int minix_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int minix_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -50,7 +50,7 @@ static int minix_mknod(struct mnt_idmap *idmap, struct inode *dir, return add_nondir(dentry, inode); } -static int minix_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int minix_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = minix_new_inode(dir, mode); @@ -63,13 +63,13 @@ static int minix_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int minix_create(struct mnt_idmap *idmap, struct inode *dir, +static int minix_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return minix_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); } -static int minix_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int minix_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { int i = strlen(symname)+1; @@ -104,7 +104,7 @@ static int minix_link(struct dentry * old_dentry, struct inode * dir, return add_nondir(dentry, inode); } -static struct dentry *minix_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *minix_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode * inode; @@ -187,7 +187,7 @@ out: return err; } -static int minix_rename(struct mnt_idmap *idmap, +static int minix_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/mnt_idmapping.c b/fs/mnt_idmapping.c index cb61fbdb52e9..bed57094cef0 100644 --- a/fs/mnt_idmapping.c +++ b/fs/mnt_idmapping.c @@ -28,7 +28,7 @@ struct mnt_idmap { * mapping. This means that {g,u}id 0 is mapped to {g,u}id 0, {g,u}id 1 is * mapped to {g,u}id 1, [...], {g,u}id 1000 to {g,u}id 1000, [...]. */ -struct mnt_idmap nop_mnt_idmap = { +const struct mnt_idmap nop_mnt_idmap = { .count = REFCOUNT_INIT(1), }; EXPORT_SYMBOL_GPL(nop_mnt_idmap); @@ -37,7 +37,7 @@ EXPORT_SYMBOL_GPL(nop_mnt_idmap); * Carries the invalid idmapping of a full 0-4294967295 {g,u}id range. * This means that all {g,u}ids are mapped to INVALID_VFS{G,U}ID. */ -struct mnt_idmap invalid_mnt_idmap = { +const struct mnt_idmap invalid_mnt_idmap = { .count = REFCOUNT_INIT(1), }; EXPORT_SYMBOL_GPL(invalid_mnt_idmap); @@ -77,7 +77,7 @@ static inline bool initial_idmapping(const struct user_namespace *ns) * returned. */ -vfsuid_t make_vfsuid(struct mnt_idmap *idmap, +vfsuid_t make_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kuid_t kuid) { @@ -117,7 +117,7 @@ EXPORT_SYMBOL_GPL(make_vfsuid); * If @kgid has no mapping in either @idmap or @fs_userns INVALID_GID is * returned. */ -vfsgid_t make_vfsgid(struct mnt_idmap *idmap, +vfsgid_t make_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, kgid_t kgid) { gid_t gid; @@ -147,7 +147,7 @@ EXPORT_SYMBOL_GPL(make_vfsgid); * * Return: @vfsuid mapped into the filesystem idmapping */ -kuid_t from_vfsuid(struct mnt_idmap *idmap, +kuid_t from_vfsuid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsuid_t vfsuid) { uid_t uid; @@ -176,7 +176,7 @@ EXPORT_SYMBOL_GPL(from_vfsuid); * * Return: @vfsgid mapped into the filesystem idmapping */ -kgid_t from_vfsgid(struct mnt_idmap *idmap, +kgid_t from_vfsgid(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, vfsgid_t vfsgid) { gid_t gid; @@ -312,10 +312,12 @@ struct mnt_idmap *alloc_mnt_idmap(struct user_namespace *mnt_userns) * * Return: @idmap with reference count bumped if @not_mnt_idmap isn't passed. */ -struct mnt_idmap *mnt_idmap_get(struct mnt_idmap *idmap) +const struct mnt_idmap *mnt_idmap_get(const struct mnt_idmap *idmap) { + struct mnt_idmap *nonconst_idmap = (struct mnt_idmap *)idmap; + if (idmap != &nop_mnt_idmap && idmap != &invalid_mnt_idmap) - refcount_inc(&idmap->count); + refcount_inc(&nonconst_idmap->count); return idmap; } @@ -328,17 +330,20 @@ EXPORT_SYMBOL_GPL(mnt_idmap_get); * If this is a non-initial idmapping, put the reference count when a mount is * released and free it if we're the last user. */ -void mnt_idmap_put(struct mnt_idmap *idmap) +void mnt_idmap_put(const struct mnt_idmap *idmap) { + struct mnt_idmap *nonconst_idmap = (struct mnt_idmap *)idmap; + if (idmap != &nop_mnt_idmap && idmap != &invalid_mnt_idmap && - refcount_dec_and_test(&idmap->count)) - free_mnt_idmap(idmap); + refcount_dec_and_test(&nonconst_idmap->count)) + free_mnt_idmap(nonconst_idmap); } EXPORT_SYMBOL_GPL(mnt_idmap_put); -int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map) +int statmount_mnt_idmap(const struct mnt_idmap *idmap, struct seq_file *seq, bool uid_map) { - struct uid_gid_map *map, *map_up; + const struct uid_gid_map *map; + struct uid_gid_map *map_up; u32 idx, nr_mappings; if (!is_valid_mnt_idmap(idmap)) @@ -358,7 +363,7 @@ int statmount_mnt_idmap(struct mnt_idmap *idmap, struct seq_file *seq, bool uid_ for (idx = 0, nr_mappings = 0; idx < map->nr_extents; idx++) { uid_t lower; - struct uid_gid_extent *extent; + const struct uid_gid_extent *extent; if (map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS) extent = &map->extent[idx]; diff --git a/fs/mount.h b/fs/mount.h index 94fcc306d21e..85f136786bbc 100644 --- a/fs/mount.h +++ b/fs/mount.h @@ -33,7 +33,8 @@ struct mnt_namespace { } __randomize_layout; struct mnt_pcp { - int mnt_count; + unsigned int mnt_gets; + unsigned int mnt_puts; int mnt_writers; }; diff --git a/fs/namei.c b/fs/namei.c index d95249dd527c..59c8a669081a 100644 --- a/fs/namei.c +++ b/fs/namei.c @@ -371,7 +371,7 @@ struct filename *complete_getname(struct delayed_filename *v) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -static int check_acl(struct mnt_idmap *idmap, +static int check_acl(const struct mnt_idmap *idmap, struct inode *inode, int mask) { #ifdef CONFIG_FS_POSIX_ACL @@ -435,7 +435,7 @@ static inline bool no_acl_inode(struct inode *inode) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -static int acl_permission_check(struct mnt_idmap *idmap, +static int acl_permission_check(const struct mnt_idmap *idmap, struct inode *inode, int mask) { unsigned int mode = inode->i_mode; @@ -518,7 +518,7 @@ static int acl_permission_check(struct mnt_idmap *idmap, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int generic_permission(struct mnt_idmap *idmap, struct inode *inode, +int generic_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret; @@ -575,7 +575,7 @@ EXPORT_SYMBOL(generic_permission); * flag in inode->i_opflags, that says "this has not special * permission function, use the fast case". */ -static inline int do_inode_permission(struct mnt_idmap *idmap, +static inline int do_inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { if (unlikely(!(inode->i_opflags & IOP_FASTPERM))) { @@ -625,7 +625,7 @@ static int sb_permission(struct super_block *sb, struct inode *inode, int mask) * * When checking for MAY_APPEND, MAY_WRITE must also be set in @mask. */ -int inode_permission(struct mnt_idmap *idmap, +int inode_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int retval; @@ -680,7 +680,7 @@ EXPORT_SYMBOL(inode_permission); * on IOP_FASTPERM can still get the optimization if they set IOP_FASTPERM_MAY_EXEC * on their directory inodes. */ -static __always_inline int lookup_inode_permission_may_exec(struct mnt_idmap *idmap, +static __always_inline int lookup_inode_permission_may_exec(const struct mnt_idmap *idmap, struct inode *inode, int mask) { /* Lookup already checked this to return -ENOTDIR */ @@ -1273,7 +1273,7 @@ fs_initcall(init_fs_namei_sysctls); */ static inline int may_follow_link(struct nameidata *nd, const struct inode *inode) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; vfsuid_t vfsuid; if (!sysctl_protected_symlinks) @@ -1314,7 +1314,7 @@ static inline int may_follow_link(struct nameidata *nd, const struct inode *inod * * Otherwise returns true. */ -static bool safe_hardlink_source(struct mnt_idmap *idmap, +static bool safe_hardlink_source(const struct mnt_idmap *idmap, struct inode *inode) { umode_t mode = inode->i_mode; @@ -1357,7 +1357,7 @@ static bool safe_hardlink_source(struct mnt_idmap *idmap, * * Returns 0 if successful, -ve on error. */ -int may_linkat(struct mnt_idmap *idmap, const struct path *link) +int may_linkat(const struct mnt_idmap *idmap, const struct path *link) { struct inode *inode = link->dentry->d_inode; @@ -1407,7 +1407,7 @@ int may_linkat(struct mnt_idmap *idmap, const struct path *link) * * Returns 0 if the open is allowed, -ve on error. */ -static int may_create_in_sticky(struct mnt_idmap *idmap, struct nameidata *nd, +static int may_create_in_sticky(const struct mnt_idmap *idmap, struct nameidata *nd, struct inode *const inode) { umode_t dir_mode = nd->dir_mode; @@ -1933,7 +1933,7 @@ static noinline struct dentry *lookup_slow(const struct qstr *name, struct inode *inode = dir->d_inode; struct dentry *res; inode_lock_shared(inode); - res = __lookup_slow(name, dir, flags); + res = __lookup_slow(name, dir, flags | LOOKUP_SHARED); inode_unlock_shared(inode); return res; } @@ -1947,12 +1947,12 @@ static struct dentry *lookup_slow_killable(const struct qstr *name, if (inode_lock_shared_killable(inode)) return ERR_PTR(-EINTR); - res = __lookup_slow(name, dir, flags); + res = __lookup_slow(name, dir, flags | LOOKUP_SHARED); inode_unlock_shared(inode); return res; } -static inline int may_lookup(struct mnt_idmap *idmap, +static inline int may_lookup(const struct mnt_idmap *idmap, struct nameidata *restrict nd) { int err, mask; @@ -2596,7 +2596,7 @@ static int link_path_walk(const char *name, struct nameidata *nd) /* At this point we know we have a real path component. */ for(;;) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; const char *link; unsigned long lastword; @@ -2946,8 +2946,8 @@ struct dentry *start_dirop(struct dentry *parent, struct qstr *name, * end_dirop - signal completion of a dirop * @de: the dentry which was returned by start_dirop or similar. * - * If the de is an error, nothing happens. Otherwise any lock taken to - * protect the dentry is dropped and the dentry itself is release (dput()). + * If the @de is an error, nothing happens. Otherwise any lock taken to + * protect the dentry is dropped and the dentry itself is released (dput()). */ void end_dirop(struct dentry *de) { @@ -3111,7 +3111,7 @@ int lookup_noperm_common(struct qstr *qname, struct dentry *base) return 0; } -static int lookup_one_common(struct mnt_idmap *idmap, +static int lookup_one_common(const struct mnt_idmap *idmap, struct qstr *qname, struct dentry *base) { int err; @@ -3190,7 +3190,7 @@ EXPORT_SYMBOL(lookup_noperm); * * The caller must hold base->i_rwsem. */ -struct dentry *lookup_one(struct mnt_idmap *idmap, struct qstr *name, +struct dentry *lookup_one(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { struct dentry *dentry; @@ -3210,7 +3210,7 @@ EXPORT_SYMBOL(lookup_one); /** * lookup_one_unlocked - lookup single pathname component * @idmap: idmap of the mount the lookup is performed from - * @name: qstr olding pathname component to lookup + * @name: qstr holding pathname component to lookup * @base: base directory to lookup from * * This can be used for in-kernel filesystem clients such as file servers. @@ -3223,7 +3223,7 @@ EXPORT_SYMBOL(lookup_one); * - ERR_PTR(-ENOENT) if parent has been removed, or * - ERR_PTR(-EACCES) if parent directory is not searchable. */ -struct dentry *lookup_one_unlocked(struct mnt_idmap *idmap, struct qstr *name, +struct dentry *lookup_one_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { int err; @@ -3243,7 +3243,7 @@ EXPORT_SYMBOL(lookup_one_unlocked); /** * lookup_one_positive_killable - lookup single pathname component * @idmap: idmap of the mount the lookup is performed from - * @name: qstr olding pathname component to lookup + * @name: qstr holding pathname component to lookup * @base: base directory to lookup from * * This helper will yield ERR_PTR(-ENOENT) on negatives. The helper returns @@ -3259,11 +3259,11 @@ EXPORT_SYMBOL(lookup_one_unlocked); * the i_rwsem itself if necessary. If a fatal signal is pending or * delivered, it will return %-EINTR if the lock is needed. * - * Returns: A dentry, possibly negative, or + * Returns: A positive dentry, or * - same errors as lookup_one_unlocked() or * - ERR_PTR(-EINTR) if a fatal signal is pending. */ -struct dentry *lookup_one_positive_killable(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_killable(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { @@ -3306,7 +3306,7 @@ EXPORT_SYMBOL(lookup_one_positive_killable); * - ERR_PTR(-ENOENT) if the name could not be found, or * - same errors as lookup_one_unlocked(). */ -struct dentry *lookup_one_positive_unlocked(struct mnt_idmap *idmap, +struct dentry *lookup_one_positive_unlocked(const struct mnt_idmap *idmap, struct qstr *name, struct dentry *base) { @@ -3381,7 +3381,7 @@ struct dentry *lookup_noperm_positive_unlocked(struct qstr *name, EXPORT_SYMBOL(lookup_noperm_positive_unlocked); /** - * start_creating - prepare to create a given name with permission checking + * start_creating - prepare to access or create a given name with permission checking * @idmap: idmap of the mount * @parent: directory in which to prepare to create the name * @name: the name to be created @@ -3396,7 +3396,7 @@ EXPORT_SYMBOL(lookup_noperm_positive_unlocked); * * Returns: a negative or positive dentry, or an error. */ -struct dentry *start_creating(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_creating(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { int err = lookup_one_common(idmap, name, parent); @@ -3413,8 +3413,8 @@ EXPORT_SYMBOL(start_creating); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing - * an object from a directory. Permission checking (MAY_EXEC) is performed + * Locks are taken and a lookup is performed prior to removing an object + * from a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * * If the name doesn't exist, an error is returned. @@ -3423,7 +3423,7 @@ EXPORT_SYMBOL(start_creating); * * Returns: a positive dentry, or an error. */ -struct dentry *start_removing(struct mnt_idmap *idmap, struct dentry *parent, +struct dentry *start_removing(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { int err = lookup_one_common(idmap, name, parent); @@ -3440,7 +3440,7 @@ EXPORT_SYMBOL(start_removing); * @parent: directory in which to prepare to create the name * @name: the name to be created * - * Locks are taken and a lookup in performed prior to creating + * Locks are taken and a lookup is performed prior to creating * an object in a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * @@ -3451,7 +3451,7 @@ EXPORT_SYMBOL(start_removing); * * Returns: a negative or positive dentry, or an error. */ -struct dentry *start_creating_killable(struct mnt_idmap *idmap, +struct dentry *start_creating_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { @@ -3469,7 +3469,7 @@ EXPORT_SYMBOL(start_creating_killable); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing + * Locks are taken and a lookup is performed prior to removing * an object from a directory. Permission checking (MAY_EXEC) is performed * against @idmap. * @@ -3482,7 +3482,7 @@ EXPORT_SYMBOL(start_creating_killable); * * Returns: a positive dentry, or an error. */ -struct dentry *start_removing_killable(struct mnt_idmap *idmap, +struct dentry *start_removing_killable(const struct mnt_idmap *idmap, struct dentry *parent, struct qstr *name) { @@ -3499,7 +3499,7 @@ EXPORT_SYMBOL(start_removing_killable); * @parent: directory in which to prepare to create the name * @name: the name to be created * - * Locks are taken and a lookup in performed prior to creating + * Locks are taken and a lookup is performed prior to creating * an object in a directory. * * If the name already exists, a positive dentry is returned. @@ -3522,7 +3522,7 @@ EXPORT_SYMBOL(start_creating_noperm); * @parent: directory in which to find the name * @name: the name to be removed * - * Locks are taken and a lookup in performed prior to removing + * Locks are taken and a lookup is performed prior to removing * an object from a directory. * * If the name doesn't exist, an error is returned. @@ -3543,11 +3543,11 @@ struct dentry *start_removing_noperm(struct dentry *parent, EXPORT_SYMBOL(start_removing_noperm); /** - * start_creating_dentry - prepare to create a given dentry - * @parent: directory from which dentry should be removed - * @child: the dentry to be removed + * start_creating_dentry - prepare to access or create a given dentry + * @parent: directory of dentry + * @child: the dentry to be prepared * - * A lock is taken to protect the dentry again other dirops and + * A lock is taken to protect the dentry against other dirops and * the validity of the dentry is checked: correct parent and still hashed. * * If the dentry is valid and negative a reference is taken and @@ -3580,7 +3580,7 @@ EXPORT_SYMBOL(start_creating_dentry); * @parent: directory from which dentry should be removed * @child: the dentry to be removed * - * A lock is taken to protect the dentry again other dirops and + * A lock is taken to protect the dentry against other dirops and * the validity of the dentry is checked: correct parent and still hashed. * * If the dentry is valid and positive, a reference is taken and @@ -3642,7 +3642,7 @@ int user_path_at(int dfd, const char __user *name, unsigned flags, } EXPORT_SYMBOL(user_path_at); -int __check_sticky(struct mnt_idmap *idmap, struct inode *dir, +int __check_sticky(const struct mnt_idmap *idmap, struct inode *dir, struct inode *inode) { kuid_t fsuid = current_fsuid(); @@ -3675,7 +3675,7 @@ EXPORT_SYMBOL(__check_sticky); * 11. We don't allow removal of NFS sillyrenamed files; it's handled by * nfs_async_unlink(). */ -int may_delete_dentry(struct mnt_idmap *idmap, struct inode *dir, +int may_delete_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *victim, bool isdir) { struct inode *inode = d_backing_inode(victim); @@ -3728,7 +3728,7 @@ EXPORT_SYMBOL(may_delete_dentry); * 4. We should have write and exec permissions on dir * 5. We can't do it if dir is immutable (done in permission()) */ -int may_create_dentry(struct mnt_idmap *idmap, +int may_create_dentry(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *child) { audit_inode_child(dir, child, AUDIT_TYPE_CHILD_CREATE); @@ -4142,7 +4142,7 @@ EXPORT_SYMBOL(end_renaming); * * Returns: mode to be passed to the filesystem */ -static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap, +static inline umode_t vfs_prepare_mode(const struct mnt_idmap *idmap, const struct inode *dir, umode_t mode, umode_t mask_perms, umode_t type) { @@ -4174,7 +4174,7 @@ static inline umode_t vfs_prepare_mode(struct mnt_idmap *idmap, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_create(struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode, +int vfs_create(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode, struct delegated_inode *di) { struct inode *dir = d_inode(dentry->d_parent); @@ -4228,7 +4228,7 @@ bool may_open_dev(const struct path *path) !(path->mnt->mnt_sb->s_iflags & SB_I_NODEV); } -static int may_open(struct mnt_idmap *idmap, const struct path *path, +static int may_open(const struct mnt_idmap *idmap, const struct path *path, int acc_mode, int flag) { struct dentry *dentry = path->dentry; @@ -4287,7 +4287,7 @@ static int may_open(struct mnt_idmap *idmap, const struct path *path, return 0; } -static int handle_truncate(struct mnt_idmap *idmap, struct file *filp) +static int handle_truncate(const struct mnt_idmap *idmap, struct file *filp) { const struct path *path = &filp->f_path; struct inode *inode = path->dentry->d_inode; @@ -4312,7 +4312,7 @@ static inline int open_to_namei_flags(int flag) return flag; } -static int may_o_create(struct mnt_idmap *idmap, +static int may_o_create(const struct mnt_idmap *idmap, const struct path *dir, struct dentry *dentry, umode_t mode) { @@ -4432,7 +4432,7 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, const struct open_flags *op) { struct delegated_inode delegated_inode = { }; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *dir = nd->path.dentry; struct inode *dir_inode = dir->d_inode; int open_flag; @@ -4440,12 +4440,14 @@ static struct dentry *lookup_open(struct nameidata *nd, struct file *file, int error, create_error; umode_t mode; bool got_write; + unsigned int shared_flag; retry: open_flag = op->open_flag; got_write = false; mode = op->mode; create_error = 0; + shared_flag = (open_flag & O_CREAT) ? 0 : LOOKUP_SHARED; if (open_flag & (O_CREAT | O_TRUNC | O_WRONLY | O_RDWR)) { got_write = !mnt_want_write(nd->path.mnt); @@ -4454,10 +4456,10 @@ retry: * a different error; we'll be dropping this one anyway. */ } - if (open_flag & O_CREAT) - inode_lock(dir_inode); - else + if (shared_flag) inode_lock_shared(dir_inode); + else + inode_lock(dir_inode); if (unlikely(IS_DEADDIR(dir_inode))) { dentry = ERR_PTR(-ENOENT); @@ -4526,7 +4528,7 @@ retry: if (d_in_lookup(dentry)) { struct dentry *res = dir_inode->i_op->lookup(dir_inode, dentry, - nd->flags); + nd->flags | shared_flag); d_lookup_done(dentry); if (unlikely(res)) { if (IS_ERR(res)) { @@ -4574,10 +4576,10 @@ out: if (file->f_mode & FMODE_OPENED) fsnotify_open(file); } - if ((open_flag & O_CREAT) || create_error) - inode_unlock(dir_inode); - else + if (shared_flag) inode_unlock_shared(dir_inode); + else + inode_unlock(dir_inode); if (got_write) mnt_drop_write(nd->path.mnt); @@ -4789,7 +4791,8 @@ finish_lookup: static int do_open(struct nameidata *nd, struct file *file, const struct open_flags *op) { - struct mnt_idmap *idmap; + struct vfsmount *mnt; + const struct mnt_idmap *idmap; int open_flag = op->open_flag; bool do_truncate; int acc_mode; @@ -4830,11 +4833,17 @@ static int do_open(struct nameidata *nd, error = mnt_want_write(nd->path.mnt); if (error) return error; + /* + * A dedicated reference is needed because after the call to + * vfs_open_consume() we no longer own the reference in nd->path.mnt + * while we need to undo write acess below. + */ + mnt = mntget(nd->path.mnt); do_truncate = true; } error = may_open(idmap, &nd->path, acc_mode, open_flag); if (!error && !(file->f_mode & FMODE_OPENED)) - error = vfs_open(&nd->path, file); + error = vfs_open_consume(&nd->path, file); if (!error) error = security_file_post_open(file, op->acc_mode); if (!error && do_truncate) @@ -4843,8 +4852,10 @@ static int do_open(struct nameidata *nd, WARN_ON(1); error = -EINVAL; } - if (do_truncate) - mnt_drop_write(nd->path.mnt); + if (do_truncate) { + mnt_drop_write(mnt); + mntput(mnt); + } return error; } @@ -4863,7 +4874,7 @@ static int do_open(struct nameidata *nd, * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_tmpfile(struct mnt_idmap *idmap, +int vfs_tmpfile(const struct mnt_idmap *idmap, const struct path *parentpath, struct file *file, umode_t mode) { @@ -4921,7 +4932,7 @@ int vfs_tmpfile(struct mnt_idmap *idmap, * hence this is only for kernel internal use, and must not be installed into * file tables or such. */ -struct file *kernel_tmpfile_open(struct mnt_idmap *idmap, +struct file *kernel_tmpfile_open(const struct mnt_idmap *idmap, const struct path *parentpath, umode_t mode, int open_flag, const struct cred *cred) @@ -5169,7 +5180,7 @@ struct file *dentry_create(struct path *path, int flags, umode_t mode, struct dentry *orig_dentry = dentry; struct dentry *dir = dentry->d_parent; struct inode *dir_inode = d_inode(dir); - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; int error, create_error; file = alloc_empty_file(flags, cred); @@ -5238,7 +5249,7 @@ EXPORT_SYMBOL(dentry_create); * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +int vfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev, struct delegated_inode *delegated_inode) { @@ -5296,7 +5307,7 @@ int filename_mknodat(int dfd, struct filename *name, umode_t mode, unsigned int dev) { struct delegated_inode di = { }; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *dentry; struct path path; int error; @@ -5380,7 +5391,7 @@ SYSCALL_DEFINE3(mknod, const char __user *, filename, umode_t, mode, unsigned, d * * In case of an error the dentry is dput() and an ERR_PTR() is returned. */ -struct dentry *vfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +struct dentry *vfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, struct delegated_inode *delegated_inode) { @@ -5487,7 +5498,7 @@ SYSCALL_DEFINE2(mkdir, const char __user *, pathname, umode_t, mode) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_rmdir(struct mnt_idmap *idmap, struct inode *dir, +int vfs_rmdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct delegated_inode *delegated_inode) { int error = may_delete_dentry(idmap, dir, dentry, true); @@ -5622,7 +5633,7 @@ SYSCALL_DEFINE1(rmdir, const char __user *, pathname) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_unlink(struct mnt_idmap *idmap, struct inode *dir, +int vfs_unlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct delegated_inode *delegated_inode) { struct inode *target = dentry->d_inode; @@ -5772,7 +5783,7 @@ SYSCALL_DEFINE1(unlink, const char __user *, pathname) * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int vfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *oldname, struct delegated_inode *delegated_inode) { @@ -5874,7 +5885,7 @@ SYSCALL_DEFINE2(symlink, const char __user *, oldname, const char __user *, newn * On non-idmapped mounts or if permission checking is to be performed on the * raw inode simply pass @nop_mnt_idmap. */ -int vfs_link(struct dentry *old_dentry, struct mnt_idmap *idmap, +int vfs_link(struct dentry *old_dentry, const struct mnt_idmap *idmap, struct inode *dir, struct dentry *new_dentry, struct delegated_inode *delegated_inode) { @@ -5951,7 +5962,7 @@ EXPORT_SYMBOL(vfs_link); int filename_linkat(int olddfd, struct filename *old, int newdfd, struct filename *new, int flags) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *new_dentry; struct path old_path, new_path; struct delegated_inode delegated_inode = { }; diff --git a/fs/namespace.c b/fs/namespace.c index 580877e46b1a..973efee4b968 100644 --- a/fs/namespace.c +++ b/fs/namespace.c @@ -109,7 +109,7 @@ struct mount_kattr { unsigned int lookup_flags; enum mount_kattr_flags_t kflags; struct user_namespace *mnt_userns; - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; }; /* /sys/fs */ @@ -249,16 +249,24 @@ void mnt_release_group_id(struct mount *mnt) mnt->mnt_group_id = 0; } -/* - * vfsmount lock must be held for read - */ -static inline void mnt_add_count(struct mount *mnt, int n) +static inline void mnt_inc_count(struct mount *mnt) { #ifdef CONFIG_SMP - this_cpu_add(mnt->mnt_pcp->mnt_count, n); + this_cpu_inc(mnt->mnt_pcp->mnt_gets); #else preempt_disable(); - mnt->mnt_count += n; + mnt->mnt_count++; + preempt_enable(); +#endif +} + +static inline void mnt_dec_count(struct mount *mnt) +{ +#ifdef CONFIG_SMP + this_cpu_inc(mnt->mnt_pcp->mnt_puts); +#else + preempt_disable(); + mnt->mnt_count--; preempt_enable(); #endif } @@ -269,14 +277,17 @@ static inline void mnt_add_count(struct mount *mnt, int n) int mnt_get_count(struct mount *mnt) { #ifdef CONFIG_SMP - int count = 0; + unsigned int gets = 0, puts = 0; int cpu; - for_each_possible_cpu(cpu) { - count += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_count; - } + /* puts first, so a put counted here has its get counted below */ + for_each_possible_cpu(cpu) + puts += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_puts; + smp_mb(); /* pairs with the smp_wmb() in mntput_no_expire() */ + for_each_possible_cpu(cpu) + gets += per_cpu_ptr(mnt->mnt_pcp, cpu)->mnt_gets; - return count; + return gets - puts; #else return mnt->mnt_count; #endif @@ -305,7 +316,7 @@ static struct mount *alloc_vfsmnt(const char *name) if (!mnt->mnt_pcp) goto out_free_devname; - this_cpu_add(mnt->mnt_pcp->mnt_count, 1); + this_cpu_inc(mnt->mnt_pcp->mnt_gets); #else mnt->mnt_count = 1; mnt->mnt_writers = 0; @@ -746,13 +757,13 @@ int __legitimize_mnt(struct vfsmount *bastard, unsigned seq) if (bastard == NULL) return 0; mnt = real_mount(bastard); - mnt_add_count(mnt, 1); - smp_mb(); // see mntput_no_expire() and do_umount() + mnt_inc_count(mnt); + smp_mb(); /* see mntput_no_expire_slowpath() and do_umount() */ if (likely(!read_seqretry(&mount_lock, seq))) return 0; lock_mount_hash(); if (unlikely(bastard->mnt_flags & (MNT_SYNC_UMOUNT | MNT_DOOMED))) { - mnt_add_count(mnt, -1); + mnt_dec_count(mnt); unlock_mount_hash(); return 1; } @@ -1254,6 +1265,7 @@ static struct mount *clone_mnt(struct mount *old, struct dentry *root, mnt->mnt.mnt_flags = READ_ONCE(old->mnt.mnt_flags) & ~MNT_INTERNAL_FLAGS; + mnt->mnt_t_flags = old->mnt_t_flags & T_UNBINDABLE; if (flag & (CL_SLAVE | CL_PRIVATE)) mnt->mnt_group_id = 0; /* not a peer of original */ @@ -1347,7 +1359,7 @@ static void noinline mntput_no_expire_slowpath(struct mount *mnt) * mount_lock, we'll see their refcount increment here. */ smp_mb(); - mnt_add_count(mnt, -1); + mnt_dec_count(mnt); count = mnt_get_count(mnt); if (count != 0) { WARN_ON(count < 0); @@ -1404,7 +1416,8 @@ static void mntput_no_expire(struct mount *mnt) * non-NULL under rcu_read_lock(), the reference * we are dropping is not the final one. */ - mnt_add_count(mnt, -1); + smp_wmb(); /* pairs with the smp_mb() in mnt_get_count() */ + mnt_dec_count(mnt); rcu_read_unlock(); return; } @@ -1426,7 +1439,7 @@ EXPORT_SYMBOL(mntput); struct vfsmount *mntget(struct vfsmount *mnt) { if (mnt) - mnt_add_count(real_mount(mnt), 1); + mnt_inc_count(real_mount(mnt)); return mnt; } EXPORT_SYMBOL(mntget); @@ -3467,7 +3480,7 @@ static int do_set_group(const struct path *from_path, const struct path *to_path return -EINVAL; /* Setting sharing groups is only allowed on private mounts */ - if (IS_MNT_SHARED(to) || IS_MNT_SLAVE(to)) + if (IS_MNT_SHARED(to) || IS_MNT_SLAVE(to) || IS_MNT_UNBINDABLE(to)) return -EINVAL; /* From should not be private */ @@ -4115,7 +4128,7 @@ int path_mount(const char *dev_name, const struct path *path, if (flags & SB_MANDLOCK) warn_mandlock(); - /* Default to relatime unless overriden */ + /* Default to relatime unless overridden */ if (!(flags & MS_NOATIME)) mnt_flags |= MNT_RELATIME; @@ -4247,8 +4260,6 @@ struct mnt_namespace *copy_mnt_ns(u64 flags, struct mnt_namespace *ns, struct mount *new; int copy_flags; - BUG_ON(!ns); - if (likely(!(flags & CLONE_NEWNS))) { get_mnt_ns(ns); return ns; @@ -4544,16 +4555,16 @@ SYSCALL_DEFINE3(fsmount, int, fs_fd, unsigned int, flags, FD_PREPARE(fdf, (flags & FSMOUNT_CLOEXEC) ? O_CLOEXEC : 0, dentry_open(&new_path, O_PATH, fc->cred)); - if (fdf.err) { + if (fdf->fd < 0) { dissolve_on_fput(new_path.mnt); - return fdf.err; + return fdf->fd; } /* * Attach to an apparent O_PATH fd with a note that we * need to unmount it, not just simply put it. */ - fd_prepare_file(fdf)->f_mode |= FMODE_NEED_UNMOUNT; + fdf->file->f_mode |= FMODE_NEED_UNMOUNT; return fd_publish(fdf); } @@ -4898,7 +4909,7 @@ static int mount_setattr_prepare(struct mount_kattr *kattr, struct mount *mnt) static void do_idmap_mount(const struct mount_kattr *kattr, struct mount *mnt) { - struct mnt_idmap *old_idmap; + const struct mnt_idmap *old_idmap; if (!kattr->mnt_idmap) return; @@ -4941,7 +4952,7 @@ static int do_mount_setattr(const struct path *path, struct mount_kattr *kattr) return -EINVAL; if (kattr->mnt_userns) { - struct mnt_idmap *mnt_idmap; + const struct mnt_idmap *mnt_idmap; mnt_idmap = alloc_mnt_idmap(kattr->mnt_userns); if (IS_ERR(mnt_idmap)) @@ -5198,12 +5209,12 @@ SYSCALL_DEFINE5(open_tree_attr, int, dfd, const char __user *, filename, return -EINVAL; FD_PREPARE(fdf, flags, vfs_open_tree(dfd, filename, flags)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; if (uattr) { struct mount_kattr kattr = {}; - struct file *file = fd_prepare_file(fdf); + struct file *file = fdf->file; int ret; if (flags & OPEN_TREE_CLONE) @@ -5246,7 +5257,7 @@ struct kstatmount { struct statmount __user *buf; size_t bufsize; struct vfsmount *mnt; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; u64 mask; struct path root; struct seq_file seq; diff --git a/fs/netfs/Kconfig b/fs/netfs/Kconfig index 7701c037c328..d0e7b0971fa3 100644 --- a/fs/netfs/Kconfig +++ b/fs/netfs/Kconfig @@ -22,6 +22,9 @@ config NETFS_STATS between CPUs. On the other hand, the stats are very useful for debugging purposes. Saying 'Y' here is recommended. +config NETFS_PGPRIV2 + bool + config NETFS_DEBUG bool "Enable dynamic debugging netfslib and FS-Cache" depends on NETFS_SUPPORT diff --git a/fs/netfs/Makefile b/fs/netfs/Makefile index b43188d64bd8..54834cde7e56 100644 --- a/fs/netfs/Makefile +++ b/fs/netfs/Makefile @@ -11,7 +11,6 @@ netfs-y := \ misc.o \ objects.o \ read_collect.o \ - read_pgpriv2.o \ read_retry.o \ read_single.o \ rolling_buffer.o \ @@ -19,6 +18,7 @@ netfs-y := \ write_issue.o \ write_retry.o +netfs-$(CONFIG_NETFS_PGPRIV2) += read_pgpriv2.o netfs-$(CONFIG_NETFS_STATS) += stats.o netfs-$(CONFIG_FSCACHE) += \ diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 105194de6e13..e30bde80276a 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -10,9 +10,9 @@ #include "internal.h" static void netfs_cache_expand_readahead(struct netfs_io_request *rreq, - unsigned long long *_start, - unsigned long long *_len, - unsigned long long i_size) + uoff_t *_start, + uoff_t *_len, + uoff_t i_size) { struct netfs_cache_resources *cres = &rreq->cache_resources; @@ -137,21 +137,6 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq) return subreq->len; } -static enum netfs_io_source netfs_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq, - loff_t i_size) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - enum netfs_io_source source; - - if (!cres->ops) - return NETFS_DOWNLOAD_FROM_SERVER; - source = cres->ops->prepare_read(subreq, i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - return source; - -} - /* * Issue a read against the cache. * - Eats the caller's ref on subreq. @@ -166,6 +151,19 @@ static void netfs_read_cache_to_pagecache(struct netfs_io_request *rreq, netfs_cache_read_terminated, subreq); } +int netfs_read_query_cache(struct netfs_io_request *rreq, struct fscache_occupancy *occ) +{ + struct netfs_cache_resources *cres = &rreq->cache_resources; + + occ->granularity = PAGE_SIZE; + if (occ->query_from >= occ->query_to) + return 0; + if (!cres->ops) + return 0; + occ->query_from = round_up(occ->query_from, occ->granularity); + return cres->ops->query_occupancy(cres, occ); +} + void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -242,7 +240,7 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, if (overlap > 0 && copy) { folio = folioq_folio(*fq, *slot); - if (unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags))) { + if (netfs_using_pgpriv2(rreq)) { if (!folio_test_private_2(folio)) folio_start_private_2(folio); } else { @@ -268,18 +266,113 @@ static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq, */ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) { + struct fscache_occupancy _occ = { + .query_from = rreq->start, + .query_to = rreq->start + rreq->len, + .cached_from[0] = 0, + .cached_to[0] = 0, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; + struct fscache_occupancy *occ = &_occ; struct folio_queue *fq = rreq->buffer.tail; - unsigned long long start = rreq->start; unsigned int offset = 0; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret = 0, slot = 0; do { + int (*prepare_read)(struct netfs_io_subrequest *subreq) = NULL; struct netfs_io_subrequest *subreq; - enum netfs_io_source source = NETFS_SOURCE_UNKNOWN; + enum netfs_io_source source; ssize_t slice; + uoff_t hole_to, cache_to; + size_t len = size; + bool copy = false; + + /* If we don't have any, find out the next couple of data + * extents from the cache, containing of following the + * specified start offset. Holes have to be fetched from the + * server; data regions from the cache. + */ + hole_to = occ->cached_from[0]; + cache_to = occ->cached_to[0]; + if (start >= cache_to) { + /* Extent exhausted; shuffle down. */ + int i; + + for (i = 0; i < ARRAY_SIZE(occ->cached_from) - 1; i++) { + occ->cached_from[i] = occ->cached_from[i + 1]; + occ->cached_to[i] = occ->cached_to[i + 1]; + occ->cached_type[i] = occ->cached_type[i + 1]; + } + occ->cached_from[i] = ULLONG_MAX; + occ->cached_to[i] = ULLONG_MAX; + + if (occ->cached_from[0] != ULLONG_MAX) + continue; + + /* Get new extents */ + ret = netfs_read_query_cache(rreq, occ); + if (ret < 0) + break; + continue; + } - subreq = netfs_alloc_subrequest(rreq); + uoff_t zero_point = netfs_read_zero_point(rreq->inode); + uoff_t zlimit = umin(zero_point, rreq->i_size); + + _debug("rsub %llx %llx-%llx", start, hole_to, cache_to); + + if (start >= hole_to && start < cache_to) { + /* Overlap with a cached region, where the cache may + * record a block of zeroes. + */ + _debug("cached s=%llx c=%llx l=%zx", start, cache_to, size); + len = umin(cache_to - start, size); + len = round_up(len, occ->granularity); + if (occ->cached_type[0] == FSCACHE_EXTENT_ZERO) { + source = NETFS_FILL_WITH_ZEROES; + netfs_stat(&netfs_n_rh_zero); + } else { + source = NETFS_READ_FROM_CACHE; + prepare_read = rreq->cache_resources.ops->prepare_read; + } + } else if (start >= zlimit && size > 0) { + /* If this range lies beyond the zero-point, that part + * can just be cleared locally. + */ + _debug("zero %llx-%llx", start, start + size); + len = size; + source = NETFS_FILL_WITH_ZEROES; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_zero); + } else { + /* Read a cache hole from the server. If any part of + * this range lies beyond the zero-point or the EOF, + * that part can just be cleared locally. + */ + uoff_t limit = min3(zlimit, start + size, hole_to); + + _debug("limit %llx %llx", rreq->i_size, zero_point); + _debug("download %llx-%llx", start, start + size); + len = umin(limit - start, ULONG_MAX); + source = NETFS_DOWNLOAD_FROM_SERVER; + prepare_read = rreq->netfs_ops->prepare_read; + if (rreq->cache_resources.ops) + copy = true; + netfs_stat(&netfs_n_rh_download); + } + + if (len == 0) { + pr_err("ZERO-LEN READ: R=%08x l=%zx/%zx s=%llx z=%llx i=%llx", + rreq->debug_id, len, size, + start, zero_point, rreq->i_size); + break; + } + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) { ret = -ENOMEM; break; @@ -287,66 +380,23 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) subreq->start = start; subreq->len = size; + if (copy) + __set_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_queue_read(rreq, subreq); - source = netfs_cache_prepare_read(rreq, subreq, rreq->i_size); - subreq->source = source; - if (source == NETFS_DOWNLOAD_FROM_SERVER) { - unsigned long long zero_point = netfs_read_zero_point(rreq->inode); - unsigned long long zp = umin(zero_point, rreq->i_size); - size_t len = subreq->len; - - if (unlikely(rreq->origin == NETFS_READ_SINGLE)) - zp = rreq->i_size; - if (subreq->start >= zp) { - subreq->source = source = NETFS_FILL_WITH_ZEROES; - goto fill_with_zeroes; - } + rreq->io_streams[0].sreq_max_len = MAX_RW_COUNT; + rreq->io_streams[0].sreq_max_segs = INT_MAX; - if (len > zp - subreq->start) - len = zp - subreq->start; - if (len == 0) { - pr_err("ZERO-LEN READ: R=%08x[%x] l=%zx/%zx s=%llx z=%llx i=%llx", - rreq->debug_id, subreq->debug_index, - subreq->len, size, - subreq->start, zero_point, rreq->i_size); + if (prepare_read) { + ret = prepare_read(subreq); + if (ret < 0) { netfs_cancel_read(subreq, ret); break; } - subreq->len = len; - - netfs_stat(&netfs_n_rh_download); - if (rreq->netfs_ops->prepare_read) { - ret = rreq->netfs_ops->prepare_read(subreq); - if (ret < 0) { - netfs_cancel_read(subreq, ret); - break; - } - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - } - goto issue; - } - - fill_with_zeroes: - if (source == NETFS_FILL_WITH_ZEROES) { - subreq->source = NETFS_FILL_WITH_ZEROES; - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - netfs_stat(&netfs_n_rh_zero); - goto issue; - } - - if (source == NETFS_READ_FROM_CACHE) { - trace_netfs_sreq(subreq, netfs_sreq_trace_submit); - goto issue; + trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); } - pr_err("Unexpected read source %u\n", source); - WARN_ON_ONCE(1); - netfs_cancel_read(subreq, ret); - break; - - issue: slice = netfs_prepare_read_iterator(subreq); if (slice < 0) { ret = slice; @@ -355,18 +405,16 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } start += slice; size -= slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); if (fq) { /* See if the cache indicated this should be cached. */ - bool copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); - + copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags); netfs_mark_copy_to_cache(rreq, &fq, &slot, &offset, slice, copy); } + trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_issue_read(rreq, subreq); netfs_maybe_bulk_drop_ra_refs(rreq); @@ -378,8 +426,7 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } @@ -646,11 +693,11 @@ EXPORT_SYMBOL(netfs_read_folio); * If any of these criteria are met, then zero out the unwritten parts * of the folio and return true. Otherwise, return false. */ -static bool netfs_skip_folio_read(struct folio *folio, loff_t pos, size_t len, +static bool netfs_skip_folio_read(struct folio *folio, uoff_t pos, size_t len, bool always_fill) { struct inode *inode = folio_inode(folio); - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); size_t offset = offset_in_folio(folio, pos); size_t plen = folio_size(folio); @@ -715,7 +762,7 @@ zero_out: */ int netfs_write_begin(struct netfs_inode *ctx, struct file *file, struct address_space *mapping, - loff_t pos, unsigned int len, struct folio **_folio, + uoff_t pos, unsigned int len, struct folio **_folio, void **_fsdata) { struct netfs_io_request *rreq; @@ -811,7 +858,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, struct netfs_io_request *rreq; struct address_space *mapping = folio->mapping; struct netfs_inode *ctx = netfs_inode(mapping->host); - unsigned long long start = folio_pos(folio); + uoff_t start = folio_pos(folio); size_t flen = folio_size(folio); int ret; diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index 2cdb68e6b16f..49b47252f675 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -17,7 +17,7 @@ * as possible to hold as much of the remaining length as possible in one go. */ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, - loff_t pos, size_t part) + uoff_t pos, size_t part) { pgoff_t index = pos / PAGE_SIZE; fgf_t fgp_flags = FGP_WRITEBEGIN; @@ -35,9 +35,9 @@ static struct folio *netfs_grab_folio_for_write(struct address_space *mapping, * the values actually are. */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied) + uoff_t pos, size_t copied) { - loff_t i_size, end = pos + copied; + uoff_t i_size, end = pos + copied; blkcnt_t add; size_t gap; @@ -54,9 +54,6 @@ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, i_size = i_size_read(inode); if (end > i_size) { i_size_write(inode, end); -#if IS_ENABLED(CONFIG_FSCACHE) - fscache_update_cookie(ctx->cache, NULL, &end); -#endif gap = SECTOR_SIZE - (i_size & (SECTOR_SIZE - 1)); if (copied > gap) { @@ -91,50 +88,20 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, struct inode *inode = file_inode(file); struct address_space *mapping = inode->i_mapping; struct netfs_inode *ctx = netfs_inode(inode); - struct writeback_control wbc = { - .sync_mode = WB_SYNC_NONE, - .for_sync = true, - .nr_to_write = LONG_MAX, - .range_start = iocb->ki_pos, - .range_end = iocb->ki_pos + iter->count, - }; - struct netfs_io_request *wreq = NULL; - struct folio *folio = NULL, *writethrough = NULL; + struct folio *folio = NULL; unsigned int bdp_flags = (iocb->ki_flags & IOCB_NOWAIT) ? BDP_ASYNC : 0; - ssize_t written = 0, ret, ret2; - loff_t pos = iocb->ki_pos; + ssize_t written = 0, ret; + uoff_t pos = iocb->ki_pos; size_t max_chunk = mapping_max_folio_size(mapping); bool maybe_trouble = false; - if (unlikely(iocb->ki_flags & (IOCB_DSYNC | IOCB_SYNC)) - ) { - wbc_attach_fdatawrite_inode(&wbc, mapping->host); - - ret = filemap_write_and_wait_range(mapping, pos, pos + iter->count); - if (ret < 0) { - wbc_detach_inode(&wbc); - goto out; - } - - wreq = netfs_begin_writethrough(iocb, iter->count); - if (IS_ERR(wreq)) { - wbc_detach_inode(&wbc); - ret = PTR_ERR(wreq); - wreq = NULL; - goto out; - } - if (!is_sync_kiocb(iocb)) - wreq->iocb = iocb; - netfs_stat(&netfs_n_wh_writethrough); - } else { - netfs_stat(&netfs_n_wh_buffered_write); - } + netfs_stat(&netfs_n_wh_buffered_write); do { enum netfs_folio_trace trace; struct netfs_folio *finfo; struct netfs_group *group; - unsigned long long fpos; + uoff_t fpos; size_t flen; size_t offset; /* Offset into pagecache folio */ size_t part; /* Bytes to write to folio */ @@ -390,15 +357,8 @@ ssize_t netfs_perform_write(struct kiocb *iocb, struct iov_iter *iter, pos += copied; written += copied; - if (likely(!wreq)) { - folio_mark_dirty(folio); - folio_unlock(folio); - } else { - netfs_advance_writethrough(wreq, &wbc, folio, copied, - offset + copied == flen, - &writethrough); - /* Folio unlocked */ - } + folio_mark_dirty(folio); + folio_unlock(folio); retry: folio_put(folio); folio = NULL; @@ -420,15 +380,6 @@ out: ctx->ops->post_modify(inode); } - if (unlikely(wreq)) { - ret2 = netfs_end_writethrough(wreq, &wbc, writethrough); - wbc_detach_inode(&wbc); - if (ret2 == -EIOCBQUEUED) - return ret2; - if (ret == 0 && ret2 < 0) - ret = ret2; - } - iocb->ki_pos += written; _leave(" = %zd [%zd]", written, ret); return written ? written : ret; diff --git a/fs/netfs/direct_read.c b/fs/netfs/direct_read.c index 6a8fb0d55e04..8c15f3079723 100644 --- a/fs/netfs/direct_read.c +++ b/fs/netfs/direct_read.c @@ -47,15 +47,15 @@ static void netfs_prepare_dio_read_iterator(struct netfs_io_subrequest *subreq) */ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) { - unsigned long long start = rreq->start; ssize_t size = rreq->len; + uoff_t start = rreq->start; int ret; do { struct netfs_io_subrequest *subreq; ssize_t slice; - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { /* Stash the error in the request if there's not * already an error set. @@ -64,7 +64,6 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) break; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = size; @@ -84,10 +83,8 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) size -= slice; start += slice; rreq->submitted += slice; - if (size <= 0) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); - } + if (size <= 0) + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); @@ -99,8 +96,7 @@ static void netfs_dispatch_unbuffered_reads(struct netfs_io_request *rreq) } while (size > 0); if (unlikely(size > 0)) { - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); } } diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c index 2361277416c7..32200c10d2a4 100644 --- a/fs/netfs/direct_write.c +++ b/fs/netfs/direct_write.c @@ -225,9 +225,9 @@ ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter * struct netfs_group *netfs_group) { struct netfs_io_request *wreq; - unsigned long long start = iocb->ki_pos; - unsigned long long end = start + iov_iter_count(iter); ssize_t ret, n; + uoff_t start = iocb->ki_pos; + uoff_t end = start + iov_iter_count(iter); size_t len = iov_iter_count(iter); bool async = !is_sync_kiocb(iocb); @@ -336,8 +336,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from) struct inode *inode = mapping->host; struct netfs_inode *ictx = netfs_inode(inode); ssize_t ret; - loff_t pos = iocb->ki_pos; - unsigned long long end = pos + iov_iter_count(from) - 1; + uoff_t pos = iocb->ki_pos; + uoff_t end = pos + iov_iter_count(from) - 1; _enter("%llx,%zx,%llx", pos, iov_iter_count(from), i_size_read(inode)); diff --git a/fs/netfs/fscache_cookie.c b/fs/netfs/fscache_cookie.c index 3d56fc73435f..5a226f9cbdea 100644 --- a/fs/netfs/fscache_cookie.c +++ b/fs/netfs/fscache_cookie.c @@ -327,7 +327,7 @@ static struct fscache_cookie *fscache_alloc_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -452,7 +452,7 @@ struct fscache_cookie *__fscache_acquire_cookie( u8 advice, const void *index_key, size_t index_key_len, const void *aux_data, size_t aux_data_len, - loff_t object_size) + uoff_t object_size) { struct fscache_cookie *cookie; @@ -663,7 +663,7 @@ static void fscache_unuse_cookie_locked(struct fscache_cookie *cookie) * Stop using the cookie for I/O. */ void __fscache_unuse_cookie(struct fscache_cookie *cookie, - const void *aux_data, const loff_t *object_size) + const void *aux_data, const uoff_t *object_size) { unsigned int debug_id = cookie->debug_id; unsigned int r = refcount_read(&cookie->ref); @@ -1049,7 +1049,7 @@ static void fscache_perform_invalidation(struct fscache_cookie *cookie) * Invalidate an object. */ void __fscache_invalidate(struct fscache_cookie *cookie, - const void *aux_data, loff_t new_size, + const void *aux_data, uoff_t new_size, unsigned int flags) { bool is_caching; diff --git a/fs/netfs/fscache_internal.h b/fs/netfs/fscache_internal.h deleted file mode 100644 index a09b948fcef2..000000000000 --- a/fs/netfs/fscache_internal.h +++ /dev/null @@ -1,14 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0-or-later */ -/* Internal definitions for FS-Cache - * - * Copyright (C) 2021 Red Hat, Inc. All Rights Reserved. - * Written by David Howells (dhowells@redhat.com) - */ - -#include "internal.h" - -#ifdef pr_fmt -#undef pr_fmt -#endif - -#define pr_fmt(fmt) "FS-Cache: " fmt diff --git a/fs/netfs/fscache_io.c b/fs/netfs/fscache_io.c index 37f05b4d3469..056a2bae5d99 100644 --- a/fs/netfs/fscache_io.c +++ b/fs/netfs/fscache_io.c @@ -79,7 +79,7 @@ static int fscache_begin_operation(struct netfs_cache_resources *cres, cres->ops = NULL; cres->cache_priv = cookie; cres->cache_priv2 = NULL; - cres->debug_id = cookie->debug_id; + cres->cookie_id = cookie->debug_id; cres->inval_counter = cookie->inval_counter; if (!fscache_begin_cookie_access(cookie, why)) { @@ -162,7 +162,7 @@ EXPORT_SYMBOL(__fscache_begin_write_operation); struct fscache_write_request { struct netfs_cache_resources cache_resources; struct address_space *mapping; - loff_t start; + uoff_t start; size_t len; bool set_bits; bool using_pgpriv2; @@ -171,7 +171,7 @@ struct fscache_write_request { }; void __fscache_clear_page_bits(struct address_space *mapping, - loff_t start, size_t len) + uoff_t start, size_t len) { pgoff_t first = start / PAGE_SIZE; pgoff_t last = (start + len - 1) / PAGE_SIZE; @@ -208,7 +208,7 @@ static void fscache_wreq_done(void *priv, ssize_t transferred_or_error) void __fscache_write_to_cache(struct fscache_cookie *cookie, struct address_space *mapping, - loff_t start, size_t len, loff_t i_size, + uoff_t start, size_t len, uoff_t i_size, netfs_io_terminated_t term_func, void *term_func_priv, bool using_pgpriv2, bool cond) @@ -267,7 +267,7 @@ EXPORT_SYMBOL(__fscache_write_to_cache); /* * Change the size of a backing object. */ -void __fscache_resize_cookie(struct fscache_cookie *cookie, loff_t new_size) +void __fscache_resize_cookie(struct fscache_cookie *cookie, uoff_t new_size) { struct netfs_cache_resources cres; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index c79c8e69d60c..b8591abc90a9 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -23,6 +23,8 @@ /* * buffered_read.c */ +int netfs_read_query_cache(struct netfs_io_request *rreq, + struct fscache_occupancy *occ); void netfs_queue_read(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq); void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); @@ -33,7 +35,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio, * buffered_write.c */ void netfs_update_i_size(struct netfs_inode *ctx, struct inode *inode, - loff_t pos, size_t copied); + uoff_t pos, size_t copied); /* * main.c @@ -86,13 +88,14 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin); void netfs_get_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_clear_subrequests(struct netfs_io_request *rreq); void netfs_put_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what); void netfs_put_failed_request(struct netfs_io_request *rreq); -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq); +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source); static inline void netfs_see_request(struct netfs_io_request *rreq, enum netfs_rreq_ref_trace what) @@ -120,9 +123,37 @@ void netfs_cache_read_terminated(void *priv, ssize_t transferred_or_error); /* * read_pgpriv2.c */ +#ifdef CONFIG_NETFS_PGPRIV2 +int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs); void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio); void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq); bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq); +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)); +} +#else +static inline int netfs_prepare_pgpriv2_write_buffer(struct netfs_io_subrequest *subreq, + unsigned int max_segs) +{ + return -EIO; +} +static inline void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) +{ +} +static inline void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) +{ +} +static inline bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *wreq) +{ + return true; +} +static inline bool netfs_using_pgpriv2(const struct netfs_io_request *rreq) +{ + return false; +} +#endif /* * read_retry.c @@ -157,7 +188,6 @@ extern atomic_t netfs_n_rh_write_zskip; extern atomic_t netfs_n_rh_retry_read_req; extern atomic_t netfs_n_rh_retry_read_subreq; extern atomic_t netfs_n_wh_buffered_write; -extern atomic_t netfs_n_wh_writethrough; extern atomic_t netfs_n_wh_dio_write; extern atomic_t netfs_n_wh_writepages; extern atomic_t netfs_n_wh_copy_to_cache; @@ -203,11 +233,11 @@ void netfs_write_collection_worker(struct work_struct *work); */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin); void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start); + uoff_t start); void netfs_reissue_write(struct netfs_io_stream *stream, struct netfs_io_subrequest *subreq, struct iov_iter *source); @@ -215,13 +245,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream); size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof); -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len); -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache); -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache); + uoff_t start, size_t len, bool to_eof); /* * write_retry.c @@ -321,6 +345,26 @@ static inline bool netfs_check_subreq_in_progress(const struct netfs_io_subreque } /* + * Indicate that we've generated and queued all the subrequests we're going to. + */ +static inline void netfs_all_subreqs_queued(struct netfs_io_request *rreq) +{ + smp_wmb(); /* Write lists before ALL_QUEUED. */ + set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + smp_mb__after_atomic(); + trace_netfs_rreq(rreq, netfs_rreq_trace_all_queued); +} + +/* + * Query if all subrequests are queued. + */ +static inline bool netfs_are_all_subreqs_queued(const struct netfs_io_request *rreq) +{ + /* Read lists after ALL_QUEUED. */ + return test_bit_acquire(NETFS_RREQ_ALL_QUEUED, &rreq->flags); +} + +/* * fscache-cache.c */ #ifdef CONFIG_PROC_FS diff --git a/fs/netfs/iterator.c b/fs/netfs/iterator.c index b375567e0520..eb1efb17f53a 100644 --- a/fs/netfs/iterator.c +++ b/fs/netfs/iterator.c @@ -209,7 +209,7 @@ static size_t netfs_limit_xarray(const struct iov_iter *iter, size_t start_offse { struct folio *folio; unsigned int nsegs = 0; - loff_t pos = iter->xarray_start + iter->iov_offset; + uoff_t pos = iter->xarray_start + iter->iov_offset; pgoff_t index = pos / PAGE_SIZE; size_t span = 0, n = iter->count; diff --git a/fs/netfs/main.c b/fs/netfs/main.c index 927badf3989d..609e22e8f76a 100644 --- a/fs/netfs/main.c +++ b/fs/netfs/main.c @@ -44,7 +44,6 @@ static const char *netfs_origins[nr__netfs_io_origin] = { [NETFS_DIO_READ] = "DR", [NETFS_WRITEBACK] = "WB", [NETFS_WRITEBACK_SINGLE] = "W1", - [NETFS_WRITETHROUGH] = "WT", [NETFS_UNBUFFERED_WRITE] = "UW", [NETFS_DIO_WRITE] = "DW", [NETFS_PGPRIV2_COPY_TO_CACHE] = "2C", diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c index f5c1c463f4ff..a3cd76d584b8 100644 --- a/fs/netfs/misc.c +++ b/fs/netfs/misc.c @@ -193,7 +193,7 @@ void netfs_clear_inode_writeback(struct inode *inode, const void *aux) struct fscache_cookie *cookie = netfs_i_cookie(netfs_inode(inode)); if (inode_state_read_once(inode) & I_PINNING_NETFS_WB) { - loff_t i_size = i_size_read(inode); + uoff_t i_size = i_size_read(inode); fscache_unuse_cookie(cookie, aux, &i_size); } } @@ -218,8 +218,8 @@ void netfs_invalidate_folio(struct folio *folio, size_t offset, size_t length) _enter("{%lx},%zx,%zx", folio->index, offset, length); if (offset == 0 && length == flen) { - unsigned long long i_size, remote_i_size, zero_point; - unsigned long long fpos = folio_pos(folio), end; + uoff_t i_size, remote_i_size, zero_point; + uoff_t fpos = folio_pos(folio), end; netfs_read_sizes(inode, &i_size, &remote_i_size, &zero_point); end = umin(fpos + flen, i_size); @@ -305,7 +305,7 @@ bool netfs_release_folio(struct folio *folio, gfp_t gfp) { struct inode *inode = folio_inode(folio); struct netfs_inode *ctx = netfs_inode(inode); - unsigned long long i_size, remote_i_size, zero_point, end; + uoff_t i_size, remote_i_size, zero_point, end; if (folio_test_dirty(folio)) return false; @@ -424,7 +424,7 @@ static int netfs_collect_in_app(struct netfs_io_request *rreq, need_collect = true; break; } - if (subreq || !test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (subreq || !netfs_are_all_subreqs_queued(rreq)) done = false; } diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index ad549daa9c79..4b8d20559b0e 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -16,7 +16,7 @@ static void netfs_free_request(struct work_struct *work); */ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, struct file *file, - loff_t start, size_t len, + uoff_t start, size_t len, enum netfs_io_origin origin) { static atomic_t debug_ids; @@ -44,6 +44,7 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, rreq->gfp = gfp; rreq->start = start; rreq->collected_to = start; + rreq->cache_coll_to = start; rreq->cleaned_to = start; rreq->len = len; rreq->progress_at = 0; @@ -207,7 +208,8 @@ void netfs_put_failed_request(struct netfs_io_request *rreq) /* * Allocate and partially initialise an I/O request structure. */ -struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq) +struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq, + enum netfs_io_source source) { struct netfs_io_subrequest *subreq; mempool_t *mempool = rreq->netfs_ops->subrequest_pool ?: &netfs_subrequest_pool; @@ -224,6 +226,7 @@ struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq INIT_WORK(&subreq->work, NULL); INIT_LIST_HEAD(&subreq->rreq_link); refcount_set(&subreq->ref, 2); + subreq->source = source; subreq->rreq = rreq; subreq->debug_index = atomic_inc_return(&rreq->subreq_counter); netfs_get_request(rreq, netfs_rreq_trace_get_subreq); diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index a94197ef0181..2625efd48a9b 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -38,7 +38,7 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq) */ void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio) { - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (folio_get_private(folio) == NETFS_FOLIO_COPY_TO_CACHE) { folio_detach_private(folio); trace_netfs_folio(folio, netfs_folio_trace_cancel_copy); @@ -81,7 +81,7 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq, if (unlikely(test_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags))) netfs_cancel_copy_to_cache(rreq, folio); - if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) { + if (!netfs_using_pgpriv2(rreq)) { if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) { trace_netfs_folio(folio, netfs_folio_trace_sched_copy); folio_mark_dirty(folio); @@ -153,8 +153,8 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, unsigned int *notes) { struct folio_queue *folioq = rreq->buffer.tail; - unsigned long long collected_to = rreq->collected_to; unsigned int slot = rreq->buffer.first_tail_slot; + uoff_t collected_to = rreq->collected_to; if (rreq->cleaned_to >= rreq->collected_to) return; @@ -179,7 +179,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize; folio = folioq_folio(folioq, slot); @@ -192,7 +192,7 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq, fpos = folio_pos(folio); fend = fpos + fsize; - trace_netfs_collect_folio(rreq, folio, fend, collected_to); + trace_netfs_collect_folio(rreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) @@ -467,10 +467,8 @@ bool netfs_read_collection(struct netfs_io_request *rreq) /* We're done when the app thread has finished posting subreqs and the * queue is empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags)) + if (!netfs_are_all_subreqs_queued(rreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before subreq lists. */ - if (!list_empty(&stream->subrequests)) return false; diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c index a4b7bb88cbdb..5280b606fda4 100644 --- a/fs/netfs/read_pgpriv2.c +++ b/fs/netfs/read_pgpriv2.c @@ -20,7 +20,7 @@ static void netfs_pgpriv2_copy_folio(struct netfs_io_request *creq, struct folio { struct netfs_io_stream *cache = &creq->io_streams[1]; size_t fsize = folio_size(folio), flen = fsize; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false; _enter(""); @@ -158,8 +158,7 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) return; netfs_issue_write(creq, &creq->io_streams[1]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &creq->flags); + netfs_all_subreqs_queued(creq); trace_netfs_rreq(rreq, netfs_rreq_trace_end_copy_to_cache); if (list_empty_careful(&creq->io_streams[1].subrequests)) netfs_wake_collector(creq); @@ -175,8 +174,8 @@ void netfs_pgpriv2_end_copy_to_cache(struct netfs_io_request *rreq) bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) { struct folio_queue *folioq = creq->buffer.tail; - unsigned long long collected_to = creq->collected_to; unsigned int slot = creq->buffer.first_tail_slot; + uoff_t collected_to = creq->collected_to; bool made_progress = false; if (slot >= folioq_nr_slots(folioq)) { @@ -186,7 +185,7 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) for (;;) { struct folio *folio; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -199,9 +198,9 @@ bool netfs_pgpriv2_unlock_copied_folios(struct netfs_io_request *creq) fsize = folio_size(folio); flen = fsize; - fend = min_t(unsigned long long, fpos + flen, creq->i_size); + fend = min_t(uoff_t, fpos + flen, creq->i_size); - trace_netfs_collect_folio(creq, folio, fend, collected_to); + trace_netfs_collect_folio(creq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c index 4f6a36c6e214..5bd8dee5a834 100644 --- a/fs/netfs/read_retry.c +++ b/fs/netfs/read_retry.c @@ -75,7 +75,7 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) do { struct netfs_io_subrequest *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false, subreq_superfluous = false; @@ -195,12 +195,11 @@ static void netfs_retry_read_subrequests(struct netfs_io_request *rreq) * and insert them after. */ do { - subreq = netfs_alloc_subrequest(rreq); + subreq = netfs_alloc_subrequest(rreq, NETFS_DOWNLOAD_FROM_SERVER); if (!subreq) { subreq = to; goto abandon_after; } - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; subreq->start = start; subreq->len = len; subreq->stream_nr = stream->stream_nr; @@ -272,6 +271,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) struct netfs_io_stream *stream = &rreq->io_streams[0]; netfs_stat(&netfs_n_rh_retry_read_req); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -282,6 +282,7 @@ void netfs_retry_reads(struct netfs_io_request *rreq) trace_netfs_rreq(rreq, netfs_rreq_trace_resubmit); netfs_retry_read_subrequests(rreq); + trace_netfs_rreq(rreq, netfs_rreq_trace_retry_end); } /* diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c index de67ac41548d..b248e34bd0c8 100644 --- a/fs/netfs/read_single.c +++ b/fs/netfs/read_single.c @@ -58,20 +58,6 @@ static int netfs_single_begin_cache_read(struct netfs_io_request *rreq, struct n return fscache_begin_read_operation(&rreq->cache_resources, netfs_i_cookie(ctx)); } -static void netfs_single_cache_prepare_read(struct netfs_io_request *rreq, - struct netfs_io_subrequest *subreq) -{ - struct netfs_cache_resources *cres = &rreq->cache_resources; - - if (!cres->ops) { - subreq->source = NETFS_DOWNLOAD_FROM_SERVER; - return; - } - subreq->source = cres->ops->prepare_read(subreq, rreq->i_size); - trace_netfs_sreq(subreq, netfs_sreq_trace_prepare); - -} - static void netfs_single_read_cache(struct netfs_io_request *rreq, struct netfs_io_subrequest *subreq) { @@ -89,21 +75,36 @@ static void netfs_single_read_cache(struct netfs_io_request *rreq, */ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) { + struct fscache_occupancy occ = { + .query_from = 0, + .query_to = rreq->len, + .cached_from[0] = ULLONG_MAX, + .cached_to[0] = ULLONG_MAX, + .cached_from[1] = ULLONG_MAX, + .cached_to[1] = ULLONG_MAX, + }; struct netfs_io_subrequest *subreq; + enum netfs_io_source source = NETFS_DOWNLOAD_FROM_SERVER; int ret = 0; - subreq = netfs_alloc_subrequest(rreq); + /* Try to use the cache if the cache content matches the size of the + * remote file. + */ + netfs_read_query_cache(rreq, &occ); + if (occ.cached_from[0] == 0 && + occ.cached_to[0] >= rreq->len) + source = NETFS_READ_FROM_CACHE; + + subreq = netfs_alloc_subrequest(rreq, source); if (!subreq) return -ENOMEM; - subreq->source = NETFS_SOURCE_UNKNOWN; subreq->start = 0; subreq->len = rreq->len; subreq->io_iter = rreq->buffer.iter; netfs_queue_read(rreq, subreq); - netfs_single_cache_prepare_read(rreq, subreq); switch (subreq->source) { case NETFS_DOWNLOAD_FROM_SERVER: netfs_stat(&netfs_n_rh_download); @@ -113,14 +114,18 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) goto cancel; } - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); rreq->netfs_ops->issue_read(subreq); rreq->submitted += subreq->len; break; case NETFS_READ_FROM_CACHE: - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + if (rreq->cache_resources.ops->prepare_read) { + ret = rreq->cache_resources.ops->prepare_read(subreq); + if (ret < 0) + goto cancel; + } + + netfs_all_subreqs_queued(rreq); trace_netfs_sreq(subreq, netfs_sreq_trace_submit); netfs_single_read_cache(rreq, subreq); rreq->submitted += subreq->len; @@ -136,8 +141,7 @@ static int netfs_single_dispatch_read(struct netfs_io_request *rreq) return ret; cancel: netfs_cancel_read(subreq, ret); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags); + netfs_all_subreqs_queued(rreq); netfs_wake_collector(rreq); return ret; } diff --git a/fs/netfs/stats.c b/fs/netfs/stats.c index ab6b916addc4..9a607c4e62dd 100644 --- a/fs/netfs/stats.c +++ b/fs/netfs/stats.c @@ -32,7 +32,6 @@ atomic_t netfs_n_rh_write_zskip; atomic_t netfs_n_rh_retry_read_req; atomic_t netfs_n_rh_retry_read_subreq; atomic_t netfs_n_wh_buffered_write; -atomic_t netfs_n_wh_writethrough; atomic_t netfs_n_wh_dio_write; atomic_t netfs_n_wh_writepages; atomic_t netfs_n_wh_copy_to_cache; @@ -58,9 +57,8 @@ int netfs_stats_show(struct seq_file *m, void *v) atomic_read(&netfs_n_rh_read_single), atomic_read(&netfs_n_rh_write_begin), atomic_read(&netfs_n_rh_write_zskip)); - seq_printf(m, "Writes : BW=%u WT=%u DW=%u WP=%u 2C=%u\n", + seq_printf(m, "Writes : BW=%u DW=%u WP=%u 2C=%u\n", atomic_read(&netfs_n_wh_buffered_write), - atomic_read(&netfs_n_wh_writethrough), atomic_read(&netfs_n_wh_dio_write), atomic_read(&netfs_n_wh_writepages), atomic_read(&netfs_n_wh_copy_to_cache)); diff --git a/fs/netfs/write_collect.c b/fs/netfs/write_collect.c index 210eb8f3958d..6e8ea534230d 100644 --- a/fs/netfs/write_collect.c +++ b/fs/netfs/write_collect.c @@ -56,7 +56,7 @@ static void netfs_dump_request(const struct netfs_io_request *rreq) */ int netfs_folio_written_back(struct folio *folio) { - enum netfs_folio_trace why = netfs_folio_trace_clear; + enum netfs_folio_trace why = netfs_folio_trace_endwb; struct inode *inode = folio_inode(folio); struct netfs_inode *ictx = netfs_inode(inode); struct netfs_folio *finfo; @@ -67,7 +67,7 @@ int netfs_folio_written_back(struct folio *folio) /* Streaming writes cannot be redirtied whilst under writeback, * so discard the streaming record. */ - unsigned long long fend; + uoff_t fend; fend = folio_pos(folio) + finfo->dirty_offset + finfo->dirty_len; spin_lock(&ictx->inode.i_lock); @@ -79,13 +79,13 @@ int netfs_folio_written_back(struct folio *folio) group = finfo->netfs_group; gcount++; kfree(finfo); - why = netfs_folio_trace_clear_s; + why = netfs_folio_trace_endwb_s; goto end_wb; } if ((group = netfs_folio_group(folio))) { if (group == NETFS_FOLIO_COPY_TO_CACHE) { - why = netfs_folio_trace_clear_cc; + why = netfs_folio_trace_endwb_cc; folio_detach_private(folio); goto end_wb; } @@ -98,7 +98,7 @@ int netfs_folio_written_back(struct folio *folio) if (!folio_test_dirty(folio)) { folio_detach_private(folio); gcount++; - why = netfs_folio_trace_clear_g; + why = netfs_folio_trace_endwb_g; } } @@ -115,8 +115,8 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, unsigned int *notes) { struct folio_queue *folioq = wreq->buffer.tail; - unsigned long long collected_to = wreq->collected_to; unsigned int slot = wreq->buffer.first_tail_slot; + uoff_t collected_to = wreq->collected_to; if (WARN_ON_ONCE(!folioq)) { pr_err("[!] Writeback unlock found empty rolling buffer!\n"); @@ -140,7 +140,7 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, for (;;) { struct folio *folio; struct netfs_folio *finfo; - unsigned long long fpos, fend; + uoff_t fpos, fend; size_t fsize, flen; folio = folioq_folio(folioq, slot); @@ -154,9 +154,9 @@ static void netfs_writeback_unlock_folios(struct netfs_io_request *wreq, finfo = netfs_folio_info(folio); flen = finfo ? finfo->dirty_offset + finfo->dirty_len : fsize; - fend = min_t(unsigned long long, fpos + flen, wreq->i_size); + fend = min_t(uoff_t, fpos + flen, wreq->i_size); - trace_netfs_collect_folio(wreq, folio, fend, collected_to); + trace_netfs_collect_folio(wreq, folio); /* Unlock any folio we've transferred all of. */ if (collected_to < fend) @@ -189,6 +189,26 @@ done: } /* + * Collect cache results. + */ +static void netfs_cache_collect(struct netfs_io_request *wreq, + struct netfs_io_stream *stream, + enum netfs_cache_collect block_type) +{ + struct netfs_cache_resources *cres = &wreq->cache_resources; + + if (stream->source != NETFS_WRITE_TO_CACHE || + wreq->cache_coll_to >= stream->collected_to) + return; + + if (cres->ops && cres->ops->collect_write) + cres->ops->collect_write(wreq, wreq->cache_coll_to, + stream->collected_to - wreq->cache_coll_to, + block_type); + wreq->cache_coll_to = stream->collected_to; +} + +/* * Collect and assess the results of various write subrequests. We may need to * retry some of the results - or even do an RMW cycle for content crypto. * @@ -201,8 +221,8 @@ static void netfs_collect_write_results(struct netfs_io_request *wreq) { struct netfs_io_subrequest *front, *remove; struct netfs_io_stream *stream; - unsigned long long collected_to, issued_to; unsigned int notes; + uoff_t collected_to, issued_to; int s; _enter("%llx-%llx", wreq->start, wreq->start + wreq->len); @@ -214,7 +234,6 @@ reassess_streams: smp_rmb(); collected_to = ULLONG_MAX; if (wreq->origin == NETFS_WRITEBACK || - wreq->origin == NETFS_WRITETHROUGH || wreq->origin == NETFS_PGPRIV2_COPY_TO_CACHE) notes = NEED_UNLOCK; else @@ -236,13 +255,19 @@ reassess_streams: /* Read first subreq pointer before IN_PROGRESS flag. */ while (front) { + enum netfs_cache_collect cache_collect; + trace_netfs_collect_sreq(wreq, front); //_debug("sreq [%x] %llx %zx/%zx", // front->debug_index, front->start, front->transferred, front->len); if (stream->collected_to < front->start) { trace_netfs_collect_gap(wreq, stream, issued_to, 'F'); + if (stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) + netfs_cache_collect(wreq, stream, stream->cache_collect); stream->collected_to = front->start; + netfs_cache_collect(wreq, stream, NETFS_CACHE_COLLECT_WRITE_GAP); + stream->cache_collect = NETFS_CACHE_COLLECT_WRITE_GAP; } /* Stall if the front is still undergoing I/O. */ @@ -250,7 +275,6 @@ reassess_streams: notes |= HIT_PENDING; break; } - smp_rmb(); /* Read counters after I-P flag. */ if (stream->failed) { stream->collected_to = front->start + front->len; @@ -263,15 +287,44 @@ reassess_streams: stream->transferred_valid = true; notes |= MADE_PROGRESS; } - if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { - stream->failed = true; - stream->error = front->error; - if (stream->source == NETFS_UPLOAD_TO_SERVER) - mapping_set_error(wreq->mapping, front->error); - notes |= NEED_REASSESS | SAW_FAILURE; + + /* Handle failed or cancelled subreqs. Failure of + * cache writes are handled differently to upload + * failures. Cache writes aren't fatal, provided we're + * not doing disconnected operation, and so we can kind + * of treat them as if they had succeeded - except that + * we need to log any holes they cause. + */ + switch (stream->source) { + case NETFS_UPLOAD_TO_SERVER: + if (test_bit(NETFS_SREQ_FAILED, &front->flags)) { + if (!stream->failed) { + stream->failed = true; + stream->error = front->error; + mapping_set_error(wreq->mapping, front->error); + break; + } + notes |= NEED_REASSESS | SAW_FAILURE; + } + break; + + case NETFS_WRITE_TO_CACHE: + cache_collect = test_bit(NETFS_SREQ_CANCELLED, &front->flags) ? + NETFS_CACHE_COLLECT_WRITE_CANCEL : + NETFS_CACHE_COLLECT_WRITE_DATA; + if (cache_collect != stream->cache_collect && + stream->cache_collect != NETFS_CACHE_COLLECT_WRITE_GAP) { + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_fail_collect); + netfs_cache_collect(wreq, stream, stream->cache_collect); + } + stream->cache_collect = cache_collect; + break; + + default: + WARN_ON(1); break; } - if (front->transferred < front->len) { + if (test_bit(NETFS_SREQ_NEED_RETRY, &front->flags)) { stream->need_retry = true; notes |= NEED_RETRY | MADE_PROGRESS; break; @@ -360,6 +413,7 @@ need_retry: */ bool netfs_write_collection(struct netfs_io_request *wreq) { + struct netfs_io_stream *cstream = &wreq->io_streams[1]; struct netfs_inode *ictx = netfs_inode(wreq->inode); size_t transferred; bool transferred_valid = false; @@ -372,9 +426,8 @@ bool netfs_write_collection(struct netfs_io_request *wreq) /* We're done when the app thread has finished posting subreqs and all * the queues in all the streams are empty. */ - if (!test_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags)) + if (!netfs_are_all_subreqs_queued(wreq)) return false; - smp_rmb(); /* Read ALL_QUEUED before lists. */ transferred = LONG_MAX; for (s = 0; s < NR_IO_STREAMS; s++) { @@ -395,13 +448,19 @@ bool netfs_write_collection(struct netfs_io_request *wreq) wreq->transferred = transferred; trace_netfs_rreq(wreq, netfs_rreq_trace_write_done); - if (wreq->io_streams[1].active && - wreq->io_streams[1].failed && - ictx->ops->invalidate_cache) { - /* Cache write failure doesn't prevent writeback completion - * unless we're in disconnected mode. - */ - ictx->ops->invalidate_cache(wreq); + if (cstream->active) { + if (test_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) { + if (ictx->ops->invalidate_cache) { + /* Cache write failure doesn't prevent + * writeback completion unless we're in + * disconnected mode. + */ + trace_netfs_rreq(wreq, netfs_rreq_trace_inval_cache); + ictx->ops->invalidate_cache(wreq); + } + } else if (!cstream->failed) { + netfs_cache_collect(wreq, cstream, cstream->cache_collect); + } } _debug("finished"); @@ -411,7 +470,6 @@ bool netfs_write_collection(struct netfs_io_request *wreq) switch (wreq->origin) { case NETFS_WRITEBACK: case NETFS_WRITEBACK_SINGLE: - case NETFS_WRITETHROUGH: netfs_wb_end(ictx); break; default: @@ -486,24 +544,51 @@ void netfs_write_subrequest_terminated(void *_op, ssize_t transferred_or_error) if (IS_ERR_VALUE(transferred_or_error)) { subreq->error = transferred_or_error; - /* if need retry is set, error should not matter */ - if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { - set_bit(NETFS_SREQ_FAILED, &subreq->flags); - trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); - } switch (subreq->source) { case NETFS_WRITE_TO_CACHE: + /* We don't mark a cache-write subreq as failed. + * Instead we tell the issuer to produce dummy subreqs + * instead and make a note if we need to invalidate the + * cache at the end. We also don't pause the loop that + * grabs pages and launches upload subreqs. + * + * Note that we need to distinguish between -ENOBUFS + * (no space available in the cache) and other errors. + * In the former case, we can keep the data we have, + * though we might have to change the way the on-disk + * data is tracked. + */ netfs_stat(&netfs_n_wh_write_failed); + if (test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) + break; + + trace_netfs_failure(wreq, subreq, transferred_or_error, netfs_fail_write); + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + set_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags); + if (transferred_or_error == -ENOBUFS) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_no_space); + else if (!test_and_set_bit(NETFS_RREQ_CACHE_ERROR, &wreq->flags)) + trace_netfs_rreq(wreq, netfs_rreq_trace_cache_failed); + subreq->transferred = subreq->len; break; + case NETFS_UPLOAD_TO_SERVER: + /* If need_retry is set, error should not matter */ + if (!test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) { + set_bit(NETFS_SREQ_FAILED, &subreq->flags); + trace_netfs_failure(wreq, subreq, transferred_or_error, + netfs_fail_upload); + } + + set_bit(NETFS_RREQ_PAUSE, &wreq->flags); + trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); netfs_stat(&netfs_n_wh_upload_failed); break; + default: break; } - trace_netfs_rreq(wreq, netfs_rreq_trace_set_pause); - set_bit(NETFS_RREQ_PAUSE, &wreq->flags); } else { if (WARN(transferred_or_error > subreq->len - subreq->transferred, "Subreq excess write: R=%x[%x] %zd > %zu - %zu", diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c index 851f6f93ad45..3989b4ec0c4b 100644 --- a/fs/netfs/write_issue.c +++ b/fs/netfs/write_issue.c @@ -89,14 +89,13 @@ static void netfs_kill_dirty_pages(struct address_space *mapping, */ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, struct file *file, - loff_t start, + uoff_t start, enum netfs_io_origin origin) { struct netfs_io_request *wreq; struct netfs_inode *ictx; bool is_cacheable = (origin == NETFS_WRITEBACK || origin == NETFS_WRITEBACK_SINGLE || - origin == NETFS_WRITETHROUGH || origin == NETFS_PGPRIV2_COPY_TO_CACHE); wreq = netfs_alloc_request(mapping, file, start, 0, origin); @@ -112,6 +111,8 @@ struct netfs_io_request *netfs_create_write_req(struct address_space *mapping, goto nomem; wreq->cleaned_to = wreq->start; + if (wreq->cache_resources.dio_size > 1) + wreq->cache_coll_to = round_down(wreq->start, wreq->cache_resources.dio_size); wreq->io_streams[0].stream_nr = 0; wreq->io_streams[0].source = NETFS_UPLOAD_TO_SERVER; @@ -156,7 +157,7 @@ EXPORT_SYMBOL(netfs_prepare_write_failed); */ void netfs_prepare_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start) + uoff_t start) { struct netfs_io_subrequest *subreq; struct iov_iter *wreq_iter = &wreq->buffer.iter; @@ -169,10 +170,9 @@ void netfs_prepare_write(struct netfs_io_request *wreq, wreq_iter->folioq_slot >= folioq_nr_slots(wreq_iter->folioq)) rolling_buffer_make_space(&wreq->buffer, wreq->gfp); - subreq = netfs_alloc_subrequest(wreq); + subreq = netfs_alloc_subrequest(wreq, stream->source); if (!subreq) return; - subreq->source = stream->source; subreq->start = start; subreq->stream_nr = stream->stream_nr; subreq->io_iter = *wreq_iter; @@ -233,6 +233,21 @@ static void netfs_do_issue_write(struct netfs_io_stream *stream, _enter("R=%x[%x],%zx", wreq->debug_id, subreq->debug_index, subreq->len); + if (stream->source == NETFS_WRITE_TO_CACHE && + unlikely(test_bit(NETFS_RREQ_CACHE_STOP, &wreq->flags))) { + size_t dio_size = wreq->cache_resources.dio_size; + size_t len, disp; + + disp = subreq->start & (dio_size - 1); + len = round_up(subreq->len + disp, dio_size); + + subreq->start -= disp; + subreq->len = len; + + __set_bit(NETFS_SREQ_CANCELLED, &subreq->flags); + return netfs_write_subrequest_terminated(subreq, subreq->len); + } + if (test_bit(NETFS_SREQ_FAILED, &subreq->flags)) return netfs_write_subrequest_terminated(subreq, subreq->error); @@ -266,6 +281,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, if (!subreq) return; + stream->construct = NULL; subreq->io_iter.count = subreq->len; netfs_do_issue_write(stream, subreq); @@ -279,7 +295,7 @@ void netfs_issue_write(struct netfs_io_request *wreq, */ size_t netfs_advance_write(struct netfs_io_request *wreq, struct netfs_io_stream *stream, - loff_t start, size_t len, bool to_eof) + uoff_t start, size_t len, bool to_eof) { struct netfs_io_subrequest *subreq = stream->construct; size_t part; @@ -330,7 +346,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, struct netfs_folio *finfo; size_t iter_off = 0; size_t fsize = folio_size(folio), flen = fsize, foff = 0; - loff_t fpos = folio_pos(folio), i_size; + uoff_t fpos = folio_pos(folio), i_size; bool to_eof = false, streamw = false; bool debug = false; @@ -367,11 +383,7 @@ static int netfs_write_folio(struct netfs_io_request *wreq, streamw = true; } - if (wreq->origin == NETFS_WRITETHROUGH) { - to_eof = false; - if (flen > i_size - fpos) - flen = i_size - fpos; - } else if (flen > i_size - fpos) { + if (flen > i_size - fpos) { flen = i_size - fpos; if (!streamw) folio_zero_segment(folio, flen, fsize); @@ -525,8 +537,7 @@ static void netfs_end_issue_write(struct netfs_io_request *wreq) { bool needs_poke = true; - smp_wmb(); /* Write subreq lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); for (int s = 0; s < NR_IO_STREAMS; s++) { struct netfs_io_stream *stream = &wreq->io_streams[s]; @@ -614,103 +625,6 @@ out: EXPORT_SYMBOL(netfs_writepages); /* - * Begin a write operation for writing through the pagecache. - */ -struct netfs_io_request *netfs_begin_writethrough(struct kiocb *iocb, size_t len) -{ - struct netfs_io_request *wreq = NULL; - struct netfs_inode *ictx = netfs_inode(file_inode(iocb->ki_filp)); - - netfs_wb_begin(ictx, false); - - wreq = netfs_create_write_req(iocb->ki_filp->f_mapping, iocb->ki_filp, - iocb->ki_pos, NETFS_WRITETHROUGH); - if (IS_ERR(wreq)) { - netfs_wb_end(ictx); - return wreq; - } - - wreq->io_streams[0].avail = true; - __set_bit(NETFS_RREQ_OFFLOAD_COLLECTION, &wreq->flags); - trace_netfs_write(wreq, netfs_write_trace_writethrough); - return wreq; -} - -/* - * Advance the state of the write operation used when writing through the - * pagecache. Data has been copied into the pagecache that we need to append - * to the request. If we've added more than wsize then we need to create a new - * subrequest. - */ -int netfs_advance_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *folio, size_t copied, bool to_page_end, - struct folio **writethrough_cache) -{ - int ret; - - _enter("R=%x ic=%zu ws=%u cp=%zu tp=%u", - wreq->debug_id, wreq->buffer.iter.count, wreq->wsize, copied, to_page_end); - - /* The folio is locked. */ - - if (*writethrough_cache != folio) { - if (*writethrough_cache) { - /* Did the folio get moved? */ - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - } - /* We can make multiple writes to the folio... */ - if (wreq->len == 0) - trace_netfs_folio(folio, netfs_folio_trace_wthru); - else - trace_netfs_folio(folio, netfs_folio_trace_wthru_plus); - *writethrough_cache = folio; - folio_get(folio); - } - - wreq->len += copied; - - if (!to_page_end) { - folio_mark_dirty(folio); - folio_unlock(folio); - return 0; - } - - ret = netfs_write_folio(wreq, wbc, folio); - folio_put(*writethrough_cache); - *writethrough_cache = NULL; - wreq->submitted = wreq->len; - return ret; -} - -/* - * End a write operation used when writing through the pagecache. - */ -ssize_t netfs_end_writethrough(struct netfs_io_request *wreq, struct writeback_control *wbc, - struct folio *writethrough_cache) -{ - ssize_t ret; - - _enter("R=%x", wreq->debug_id); - - if (writethrough_cache) { - folio_lock(writethrough_cache); - netfs_write_folio(wreq, wbc, writethrough_cache); - folio_put(writethrough_cache); - wreq->submitted = wreq->len; - } - - netfs_end_issue_write(wreq); - - if (wreq->iocb) - ret = -EIOCBQUEUED; - else - ret = netfs_wait_for_write(wreq); - netfs_put_request(wreq, netfs_rreq_trace_put_return); - return ret; -} - -/* * Write some of a pending folio data back to the server and/or the cache. */ static int netfs_write_folio_single(struct netfs_io_request *wreq, @@ -721,7 +635,7 @@ static int netfs_write_folio_single(struct netfs_io_request *wreq, struct netfs_io_stream *stream; size_t iter_off = 0; size_t fsize = folio_size(folio), flen; - loff_t fpos = folio_pos(folio); + uoff_t fpos = folio_pos(folio); ssize_t ret; bool to_eof = false; bool no_debug = false; @@ -891,8 +805,7 @@ int netfs_writeback_single(struct address_space *mapping, stop: for (int s = 0; s < NR_IO_STREAMS; s++) netfs_issue_write(wreq, &wreq->io_streams[s]); - smp_wmb(); /* Write lists before ALL_QUEUED. */ - set_bit(NETFS_RREQ_ALL_QUEUED, &wreq->flags); + netfs_all_subreqs_queued(wreq); netfs_wake_collector(wreq); diff --git a/fs/netfs/write_retry.c b/fs/netfs/write_retry.c index 058bc7a166a5..2f20577563e1 100644 --- a/fs/netfs/write_retry.c +++ b/fs/netfs/write_retry.c @@ -55,7 +55,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, do { struct netfs_io_subrequest *subreq = NULL, *from, *to, *tmp; struct iov_iter source; - unsigned long long start, len; + uoff_t start, len; size_t part; bool boundary = false; @@ -149,8 +149,7 @@ static void netfs_retry_write_stream(struct netfs_io_request *wreq, * and insert them after. */ do { - subreq = netfs_alloc_subrequest(wreq); - subreq->source = to->source; + subreq = netfs_alloc_subrequest(wreq, stream->source); subreq->start = start; subreq->stream_nr = to->stream_nr; subreq->retry_count = 1; @@ -211,6 +210,7 @@ void netfs_retry_writes(struct netfs_io_request *wreq) int s; netfs_stat(&netfs_n_wh_retry_write_req); + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_begin); /* Wait for all outstanding I/O to quiesce before performing retries as * we may need to renegotiate the I/O sizes. @@ -235,4 +235,6 @@ void netfs_retry_writes(struct netfs_io_request *wreq) netfs_retry_write_stream(wreq, stream); } } + + trace_netfs_rreq(wreq, netfs_rreq_trace_retry_end); } diff --git a/fs/nfs/Kconfig b/fs/nfs/Kconfig index 64c249f800a9..fd430718d02b 100644 --- a/fs/nfs/Kconfig +++ b/fs/nfs/Kconfig @@ -189,6 +189,7 @@ config NFS_FSCACHE bool "Provide NFS client caching support" depends on NFS_FS select NETFS_SUPPORT + select NETFS_PGPRIV2 select FSCACHE help Say Y here if you want NFS data to be cached locally on disc through diff --git a/fs/nfs/dir.c b/fs/nfs/dir.c index 49394123bd09..354f986e60c4 100644 --- a/fs/nfs/dir.c +++ b/fs/nfs/dir.c @@ -2437,7 +2437,7 @@ out_err: return error; } -int nfs_create(struct mnt_idmap *idmap, struct inode *dir, +int nfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return nfs_do_create(dir, dentry, mode, O_EXCL); @@ -2448,7 +2448,7 @@ EXPORT_SYMBOL_GPL(nfs_create); * See comments for nfs_proc_create regarding failed operations. */ int -nfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +nfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct iattr attr; @@ -2475,7 +2475,7 @@ EXPORT_SYMBOL_GPL(nfs_mknod); /* * See comments for nfs_proc_create regarding failed operations. */ -struct dentry *nfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +struct dentry *nfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct iattr attr; @@ -2641,7 +2641,7 @@ EXPORT_SYMBOL_GPL(nfs_unlink); * now have a new file handle and can instantiate an in-core NFS inode * and move the raw page into its mapping. */ -int nfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +int nfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct folio *folio; @@ -2771,7 +2771,7 @@ static bool nfs_rename_is_unsafe_cross_dir(struct dentry *old_dentry, * If these conditions are met, we can drop the dentries before doing * the rename. */ -int nfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +int nfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -3393,7 +3393,7 @@ static int nfs_execute_ok(struct inode *inode, int mask) return ret; } -int nfs_permission(struct mnt_idmap *idmap, +int nfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { diff --git a/fs/nfs/inode.c b/fs/nfs/inode.c index 3022454f7698..3c9b2ec4e244 100644 --- a/fs/nfs/inode.c +++ b/fs/nfs/inode.c @@ -690,7 +690,7 @@ EXPORT_SYMBOL_GPL(nfs_update_delegated_mtime); #define NFS_VALID_ATTRS (ATTR_MODE|ATTR_UID|ATTR_GID|ATTR_SIZE|ATTR_ATIME|ATTR_ATIME_SET|ATTR_MTIME|ATTR_MTIME_SET|ATTR_FILE|ATTR_OPEN) int -nfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +nfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -955,7 +955,7 @@ static u32 nfs_get_valid_attrmask(struct inode *inode) return reply_mask; } -int nfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int nfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h index 48f7c0e25da1..d6ea41a3f9b4 100644 --- a/fs/nfs/internal.h +++ b/fs/nfs/internal.h @@ -397,18 +397,18 @@ extern unsigned long nfs_access_cache_scan(struct shrinker *shrink, struct shrink_control *sc); struct dentry *nfs_lookup(struct inode *, struct dentry *, unsigned int); void nfs_d_prune_case_insensitive_aliases(struct inode *inode); -int nfs_create(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_create(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); -struct dentry *nfs_mkdir(struct mnt_idmap *, struct inode *, struct dentry *, +struct dentry *nfs_mkdir(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t); int nfs_rmdir(struct inode *, struct dentry *); int nfs_unlink(struct inode *, struct dentry *); -int nfs_symlink(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_symlink(const struct mnt_idmap *, struct inode *, struct dentry *, const char *); int nfs_link(struct dentry *, struct inode *, struct dentry *); -int nfs_mknod(struct mnt_idmap *, struct inode *, struct dentry *, umode_t, +int nfs_mknod(const struct mnt_idmap *, struct inode *, struct dentry *, umode_t, dev_t); -int nfs_rename(struct mnt_idmap *, struct inode *, struct dentry *, +int nfs_rename(const struct mnt_idmap *, struct inode *, struct dentry *, struct inode *, struct dentry *, unsigned int); #ifdef CONFIG_NFS_V4_2 diff --git a/fs/nfs/namespace.c b/fs/nfs/namespace.c index 6d0073c24771..c50d59c52c50 100644 --- a/fs/nfs/namespace.c +++ b/fs/nfs/namespace.c @@ -222,7 +222,7 @@ out_fc: } static int -nfs_namespace_getattr(struct mnt_idmap *idmap, +nfs_namespace_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -235,7 +235,7 @@ nfs_namespace_getattr(struct mnt_idmap *idmap, } static int -nfs_namespace_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +nfs_namespace_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { if (NFS_FH(d_inode(dentry))->size != 0) diff --git a/fs/nfs/nfs3_fs.h b/fs/nfs/nfs3_fs.h index b333ea119ef5..ffcabadb3546 100644 --- a/fs/nfs/nfs3_fs.h +++ b/fs/nfs/nfs3_fs.h @@ -12,7 +12,7 @@ */ #ifdef CONFIG_NFS_V3_ACL extern struct posix_acl *nfs3_get_acl(struct inode *inode, int type, bool rcu); -extern int nfs3_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int nfs3_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int nfs3_proc_setacls(struct inode *inode, struct posix_acl *acl, struct posix_acl *dfacl); diff --git a/fs/nfs/nfs3acl.c b/fs/nfs/nfs3acl.c index a126eb31f62f..2549a1985b9a 100644 --- a/fs/nfs/nfs3acl.c +++ b/fs/nfs/nfs3acl.c @@ -254,7 +254,7 @@ int nfs3_proc_setacls(struct inode *inode, struct posix_acl *acl, } -int nfs3_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int nfs3_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct posix_acl *orig = acl, *dfacl = NULL, *alloc; diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c index 518348e87dd8..beb659744760 100644 --- a/fs/nfs/nfs4proc.c +++ b/fs/nfs/nfs4proc.c @@ -7866,7 +7866,7 @@ int nfs4_lock_delegation_recall(struct file_lock *fl, struct nfs4_state *state, #define XATTR_NAME_NFSV4_ACL "system.nfs4_acl" static int nfs4_xattr_set_nfs4_acl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7889,7 +7889,7 @@ static bool nfs4_xattr_list_nfs4_acl(struct dentry *dentry) #define XATTR_NAME_NFSV4_DACL "system.nfs4_dacl" static int nfs4_xattr_set_nfs4_dacl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7912,7 +7912,7 @@ static bool nfs4_xattr_list_nfs4_dacl(struct dentry *dentry) #define XATTR_NAME_NFSV4_SACL "system.nfs4_sacl" static int nfs4_xattr_set_nfs4_sacl(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7935,7 +7935,7 @@ static bool nfs4_xattr_list_nfs4_sacl(struct dentry *dentry) #ifdef CONFIG_NFS_V4_SECURITY_LABEL static int nfs4_xattr_set_nfs4_label(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) @@ -7965,7 +7965,7 @@ static const struct xattr_handler nfs4_xattr_nfs4_label_handler = { #ifdef CONFIG_NFS_V4_2 static int nfs4_xattr_set_nfs4_user(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *key, const void *buf, size_t buflen, int flags) diff --git a/fs/nfs/unlink.c b/fs/nfs/unlink.c index b57cfaa4d516..c8d712204e64 100644 --- a/fs/nfs/unlink.c +++ b/fs/nfs/unlink.c @@ -67,6 +67,7 @@ static void nfs_async_unlink_release(void *calldata) struct super_block *sb = dentry->d_sb; up_read_non_owner(&NFS_I(d_inode(dentry->d_parent))->rmdir_sem); + d_lookup_acquire(dentry); d_lookup_done(dentry); nfs_free_unlinkdata(data); dput(dentry); @@ -159,6 +160,8 @@ static int nfs_call_unlink(struct dentry *dentry, struct inode *inode, struct nf return ret; } data->dentry = alias; + d_lookup_release(alias); + nfs_do_call_unlink(inode, data); return 1; } diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c index f9131827d391..4789f2ec2078 100644 --- a/fs/nfsd/vfs.c +++ b/fs/nfsd/vfs.c @@ -1046,7 +1046,6 @@ static __be32 nfsd_finish_read(struct svc_rqst *rqstp, struct svc_fh *fhp, nfsd_stats_io_read_add(nn, fhp->fh_export, host_err); *eof = nfsd_eof_on_read(file, offset, host_err, *count); *count = host_err; - fsnotify_access(file); trace_nfsd_read_io_done(rqstp, fhp, offset, *count); return 0; } else { @@ -1071,19 +1070,11 @@ __be32 nfsd_splice_read(struct svc_rqst *rqstp, struct svc_fh *fhp, struct file *file, loff_t offset, unsigned long *count, u32 *eof) { - struct splice_desc sd = { - .len = 0, - .total_len = *count, - .pos = offset, - .u.data = rqstp, - }; ssize_t host_err; trace_nfsd_read_splice(rqstp, fhp, offset, *count); - host_err = rw_verify_area(READ, file, &offset, *count); - if (!host_err) - host_err = splice_direct_to_actor(file, &sd, - nfsd_direct_splice_actor); + host_err = vfs_splice_to_actor(file, offset, *count, + nfsd_direct_splice_actor, rqstp); return nfsd_finish_read(rqstp, fhp, file, offset, count, eof, host_err); } diff --git a/fs/nilfs2/inode.c b/fs/nilfs2/inode.c index 34e6096069ad..8953719c4950 100644 --- a/fs/nilfs2/inode.c +++ b/fs/nilfs2/inode.c @@ -904,7 +904,7 @@ void nilfs_evict_inode(struct inode *inode) */ } -int nilfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int nilfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct nilfs_transaction_info ti; @@ -943,7 +943,7 @@ out_err: return err; } -int nilfs_permission(struct mnt_idmap *idmap, struct inode *inode, +int nilfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct nilfs_root *root = NILFS_I(inode)->i_root; diff --git a/fs/nilfs2/ioctl.c b/fs/nilfs2/ioctl.c index 01a04080ef70..f56a79767c69 100644 --- a/fs/nilfs2/ioctl.c +++ b/fs/nilfs2/ioctl.c @@ -135,7 +135,7 @@ int nilfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) * * Return: 0 on success, or a negative error code on failure. */ -int nilfs_fileattr_set(struct mnt_idmap *idmap, +int nilfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/nilfs2/namei.c b/fs/nilfs2/namei.c index e037e0c6e31a..d0ae24f37854 100644 --- a/fs/nilfs2/namei.c +++ b/fs/nilfs2/namei.c @@ -85,7 +85,7 @@ nilfs_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags) * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int nilfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int nilfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -113,7 +113,7 @@ static int nilfs_create(struct mnt_idmap *idmap, struct inode *dir, } static int -nilfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +nilfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -138,7 +138,7 @@ nilfs_mknod(struct mnt_idmap *idmap, struct inode *dir, return err; } -static int nilfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int nilfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct nilfs_transaction_info ti; @@ -218,7 +218,7 @@ static int nilfs_link(struct dentry *old_dentry, struct inode *dir, return err; } -static struct dentry *nilfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *nilfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -358,7 +358,7 @@ static int nilfs_rmdir(struct inode *dir, struct dentry *dentry) return err; } -static int nilfs_rename(struct mnt_idmap *idmap, +static int nilfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) diff --git a/fs/nilfs2/nilfs.h b/fs/nilfs2/nilfs.h index 4fc42d3787a4..28d35b2fa7f7 100644 --- a/fs/nilfs2/nilfs.h +++ b/fs/nilfs2/nilfs.h @@ -270,7 +270,7 @@ extern int nilfs_sync_file(struct file *, loff_t, loff_t, int); /* ioctl.c */ int nilfs_fileattr_get(struct dentry *dentry, struct file_kattr *m); -int nilfs_fileattr_set(struct mnt_idmap *idmap, +int nilfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long nilfs_ioctl(struct file *, unsigned int, unsigned long); long nilfs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg); @@ -299,10 +299,10 @@ struct inode *nilfs_iget_for_shadow(struct inode *inode); extern void nilfs_update_inode(struct inode *, struct buffer_head *, int); extern void nilfs_truncate(struct inode *); extern void nilfs_evict_inode(struct inode *); -extern int nilfs_setattr(struct mnt_idmap *, struct dentry *, +extern int nilfs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); extern void nilfs_write_failed(struct address_space *mapping, loff_t to); -int nilfs_permission(struct mnt_idmap *idmap, struct inode *inode, +int nilfs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int nilfs_load_inode_block(struct inode *inode, struct buffer_head **pbh); extern int nilfs_inode_dirty(struct inode *); diff --git a/fs/nls/nls_iso8859-14.c b/fs/nls/nls_iso8859-14.c index c789eccb8a69..60b9400f915a 100644 --- a/fs/nls/nls_iso8859-14.c +++ b/fs/nls/nls_iso8859-14.c @@ -138,24 +138,23 @@ static const unsigned char page00[256] = { }; static const unsigned char page01[256] = { - 0x00, 0x00, 0xa1, 0xa2, 0x00, 0x00, 0x00, 0x00, /* 0x00-0x07 */ - 0x00, 0x00, 0xa6, 0xab, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x00-0x07 */ + 0x00, 0x00, 0xa4, 0xa5, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x10-0x17 */ - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb0, 0xb1, /* 0x18-0x1f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x18-0x1f */ 0xb2, 0xb3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x28-0x2f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x30-0x37 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x38-0x3f */ - 0xb4, 0xb5, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x40-0x47 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x40-0x47 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x48-0x4f */ - 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb7, 0xb9, /* 0x50-0x57 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x50-0x57 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x58-0x5f */ - 0xbb, 0xbf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ - 0x00, 0x00, 0xd7, 0xf7, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ 0x00, 0x00, 0x00, 0x00, 0xd0, 0xf0, 0xde, 0xfe, /* 0x70-0x77 */ 0xaf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ - - 0xa8, 0xb8, 0xaa, 0xba, 0xbd, 0xbe, 0x00, 0x00, /* 0x80-0x87 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x80-0x87 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x88-0x8f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x90-0x97 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x98-0x9f */ @@ -169,7 +168,7 @@ static const unsigned char page01[256] = { 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xd8-0xdf */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xe0-0xe7 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xe8-0xef */ - 0x00, 0x00, 0xac, 0xbc, 0x00, 0x00, 0x00, 0x00, /* 0xf0-0xf7 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xf0-0xf7 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0xf8-0xff */ }; @@ -178,7 +177,7 @@ static const unsigned char page1e[256] = { 0x00, 0x00, 0xa6, 0xab, 0x00, 0x00, 0x00, 0x00, /* 0x08-0x0f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x10-0x17 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb0, 0xb1, /* 0x18-0x1f */ - 0xb2, 0xb3, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x20-0x27 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x28-0x2f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x30-0x37 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x38-0x3f */ @@ -188,9 +187,8 @@ static const unsigned char page1e[256] = { 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x58-0x5f */ 0xbb, 0xbf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x60-0x67 */ 0x00, 0x00, 0xd7, 0xf7, 0x00, 0x00, 0x00, 0x00, /* 0x68-0x6f */ - 0x00, 0x00, 0x00, 0x00, 0xd0, 0xf0, 0xde, 0xfe, /* 0x70-0x77 */ - 0xaf, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ - + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x70-0x77 */ + 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x78-0x7f */ 0xa8, 0xb8, 0xaa, 0xba, 0xbd, 0xbe, 0x00, 0x00, /* 0x80-0x87 */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x88-0x8f */ 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, /* 0x90-0x97 */ diff --git a/fs/nsfs.c b/fs/nsfs.c index c3b6ae76594a..56ea0bb9ef3a 100644 --- a/fs/nsfs.c +++ b/fs/nsfs.c @@ -348,8 +348,8 @@ static long ns_ioctl(struct file *filp, unsigned int ioctl, return ret; FD_PREPARE(fdf, O_CLOEXEC, dentry_open(&path, O_RDONLY, current_cred())); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; /* * If @uinfo is passed return all information about the * mount namespace as well. diff --git a/fs/ntfs/ea.c b/fs/ntfs/ea.c index b4fcfbe2da4c..ddc201e3e2aa 100644 --- a/fs/ntfs/ea.c +++ b/fs/ntfs/ea.c @@ -875,7 +875,7 @@ static int ntfs_validate_fattr(struct ntfs_inode *ni, __le32 fattr) } static int ntfs_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) { @@ -977,7 +977,7 @@ const struct xattr_handler * const ntfs_xattr_handlers[] = { // clang-format on #ifdef CONFIG_NTFS_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -1021,7 +1021,7 @@ struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, return acl; } -static noinline int ntfs_set_acl_ex(struct mnt_idmap *idmap, +static noinline int ntfs_set_acl_ex(const struct mnt_idmap *idmap, struct inode *inode, struct posix_acl *acl, int type, bool init_acl) { @@ -1103,13 +1103,13 @@ out: return err; } -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { return ntfs_set_acl_ex(idmap, d_inode(dentry), acl, type, false); } -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir) { struct posix_acl *default_acl, *acl; diff --git a/fs/ntfs/ea.h b/fs/ntfs/ea.h index acb39c2a6fbc..690fafe181fb 100644 --- a/fs/ntfs/ea.h +++ b/fs/ntfs/ea.h @@ -17,11 +17,11 @@ int ntfs_ea_set_wsl_inode(struct inode *inode, dev_t rdev, __le16 *ea_size, ssize_t ntfs_listxattr(struct dentry *dentry, char *buffer, size_t size); #ifdef CONFIG_NTFS_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir); #else #define ntfs_get_acl NULL diff --git a/fs/ntfs/file.c b/fs/ntfs/file.c index 3ec82715a588..bb8641103d26 100644 --- a/fs/ntfs/file.c +++ b/fs/ntfs/file.c @@ -318,7 +318,7 @@ static int ntfs_setattr_size(struct inode *vi, struct iattr *attr) * NOTE: Changes in inode size are not supported yet for compressed or * encrypted files. */ -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *vi = d_inode(dentry); @@ -387,7 +387,7 @@ out: return err; } -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags) { diff --git a/fs/ntfs/inode.h b/fs/ntfs/inode.h index ff61bd402df0..45a396c97846 100644 --- a/fs/ntfs/inode.h +++ b/fs/ntfs/inode.h @@ -330,9 +330,9 @@ int ntfs_read_inode_mount(struct inode *vi); int ntfs_show_options(struct seq_file *sf, struct dentry *root); int ntfs_truncate_vfs(struct inode *vi, loff_t new_size, loff_t i_size); -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, unsigned int request_mask, unsigned int query_flags); diff --git a/fs/ntfs/namei.c b/fs/ntfs/namei.c index 75e201096525..028efad7d9c0 100644 --- a/fs/ntfs/namei.c +++ b/fs/ntfs/namei.c @@ -391,7 +391,7 @@ static int ntfs_sd_add_everyone(struct ntfs_inode *ni) return ret; } -static struct ntfs_inode *__ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static struct ntfs_inode *__ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, __le16 *name, u8 name_len, mode_t mode, dev_t dev, const char *target, int target_len) { @@ -731,7 +731,7 @@ err_out: return ERR_PTR(err); } -static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct ntfs_volume *vol = NTFS_SB(dir->i_sb); @@ -1046,7 +1046,7 @@ out: return err; } -static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ntfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct super_block *sb = dir->i_sb; @@ -1243,7 +1243,7 @@ err_out: return err; } -static int ntfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ntfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { @@ -1393,7 +1393,7 @@ err_out: return err; } -static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct super_block *sb = dir->i_sb; @@ -1440,7 +1440,7 @@ out: return err; } -static int ntfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct super_block *sb = dir->i_sb; diff --git a/fs/ntfs3/file.c b/fs/ntfs3/file.c index 95dfe6878860..4cbdd9e4222b 100644 --- a/fs/ntfs3/file.c +++ b/fs/ntfs3/file.c @@ -129,7 +129,7 @@ int ntfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) /* * ntfs_fileattr_set - inode_operations::fileattr_set */ -int ntfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); @@ -264,7 +264,7 @@ long ntfs_compat_ioctl(struct file *filp, u32 cmd, unsigned long arg) /* * ntfs_getattr - inode_operations::getattr */ -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, u32 flags) { struct inode *inode = d_inode(path->dentry); @@ -706,7 +706,7 @@ out: /* * ntfs_setattr - inode_operations::setattr */ -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ntfs3/inode.c b/fs/ntfs3/inode.c index 5276c7a00db9..366af3038248 100644 --- a/fs/ntfs3/inode.c +++ b/fs/ntfs3/inode.c @@ -1385,7 +1385,7 @@ out: * * NOTE: if fnd != NULL (ntfs_atomic_open) then @dir is locked */ -int ntfs_create_inode(struct mnt_idmap *idmap, struct inode *dir, +int ntfs_create_inode(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const struct cpu_str *uni, umode_t mode, dev_t dev, const char *symname, u32 size, struct ntfs_fnd *fnd) diff --git a/fs/ntfs3/namei.c b/fs/ntfs3/namei.c index ec59bbabd3c5..ae88cb66fb7e 100644 --- a/fs/ntfs3/namei.c +++ b/fs/ntfs3/namei.c @@ -111,7 +111,7 @@ static struct dentry *ntfs_lookup(struct inode *dir, struct dentry *dentry, /* * ntfs_create - inode_operations::create */ -static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ntfs_create_inode(idmap, dir, dentry, NULL, S_IFREG | mode, 0, @@ -121,7 +121,7 @@ static int ntfs_create(struct mnt_idmap *idmap, struct inode *dir, /* * ntfs_mknod - inode_operations::mknod */ -static int ntfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { return ntfs_create_inode(idmap, dir, dentry, NULL, mode, rdev, NULL, 0, @@ -209,7 +209,7 @@ static int ntfs_unlink(struct inode *dir, struct dentry *dentry) /* * ntfs_symlink - inode_operations::symlink */ -static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { u32 size = strlen(symname); @@ -228,7 +228,7 @@ static int ntfs_symlink(struct mnt_idmap *idmap, struct inode *dir, /* * ntfs_mkdir - inode_operations::mkdir */ -static struct dentry *ntfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ntfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(ntfs_create_inode(idmap, dir, dentry, NULL, @@ -262,7 +262,7 @@ static int ntfs_rmdir(struct inode *dir, struct dentry *dentry) /* * ntfs_rename - inode_operations::rename */ -static int ntfs_rename(struct mnt_idmap *idmap, struct inode *dir, +static int ntfs_rename(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, struct inode *new_dir, struct dentry *new_dentry, u32 flags) { diff --git a/fs/ntfs3/ntfs_fs.h b/fs/ntfs3/ntfs_fs.h index 5811d89d67b3..ea24126c70db 100644 --- a/fs/ntfs3/ntfs_fs.h +++ b/fs/ntfs3/ntfs_fs.h @@ -551,11 +551,11 @@ extern const struct file_operations ntfs_dir_operations; /* Globals from file.c */ int ntfs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ntfs_fileattr_set(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); -int ntfs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ntfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, u32 flags); -int ntfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int ntfs_file_open(struct inode *inode, struct file *file); int ntfs_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo, @@ -803,7 +803,7 @@ int ntfs_set_size(struct inode *inode, u64 new_size); int ntfs3_write_inode(struct inode *inode, struct writeback_control *wbc); int ntfs_sync_inode(struct inode *inode); int inode_read_data(struct inode *inode, void *data, size_t bytes); -int ntfs_create_inode(struct mnt_idmap *idmap, struct inode *dir, +int ntfs_create_inode(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const struct cpu_str *uni, umode_t mode, dev_t dev, const char *symname, u32 size, struct ntfs_fnd *fnd); @@ -963,18 +963,18 @@ unsigned long ntfs_names_hash(const u16 *name, size_t len, const u16 *upcase, /* globals from xattr.c */ #ifdef CONFIG_NTFS3_FS_POSIX_ACL -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir); #else #define ntfs_get_acl NULL #define ntfs_set_acl NULL #endif -int ntfs_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry); +int ntfs_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry); ssize_t ntfs_listxattr(struct dentry *dentry, char *buffer, size_t size); extern const struct xattr_handler *const ntfs_xattr_handlers[]; diff --git a/fs/ntfs3/xattr.c b/fs/ntfs3/xattr.c index 7f77df9c46f8..e9824bd80322 100644 --- a/fs/ntfs3/xattr.c +++ b/fs/ntfs3/xattr.c @@ -545,7 +545,7 @@ out: /* * ntfs_get_acl - inode_operations::get_acl */ -struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *ntfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { struct inode *inode = d_inode(dentry); @@ -597,7 +597,7 @@ struct posix_acl *ntfs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, return acl; } -static noinline int ntfs_set_acl_ex(struct mnt_idmap *idmap, +static noinline int ntfs_set_acl_ex(const struct mnt_idmap *idmap, struct inode *inode, struct posix_acl *acl, int type, bool init_acl) { @@ -679,7 +679,7 @@ out: /* * ntfs_set_acl - inode_operations::set_acl */ -int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ntfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { return ntfs_set_acl_ex(idmap, d_inode(dentry), acl, type, false); @@ -690,7 +690,7 @@ int ntfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, * * Called from ntfs_create_inode(). */ -int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, +int ntfs_init_acl(const struct mnt_idmap *idmap, struct inode *inode, struct inode *dir) { struct posix_acl *default_acl, *acl; @@ -724,7 +724,7 @@ int ntfs_init_acl(struct mnt_idmap *idmap, struct inode *inode, /* * ntfs_acl_chmod - Helper for ntfs_setattr(). */ -int ntfs_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry) +int ntfs_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry) { struct inode *inode = d_inode(dentry); struct super_block *sb = inode->i_sb; @@ -865,7 +865,7 @@ static bool ntfs_is_reserved_lxattr(const char *name) * ntfs_setxattr - inode_operations::setxattr */ static noinline int ntfs_setxattr(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *de, + const struct mnt_idmap *idmap, struct dentry *de, struct inode *inode, const char *name, const void *value, size_t size, int flags) { diff --git a/fs/ocfs2/acl.c b/fs/ocfs2/acl.c index 090ec60fb576..801a2f56ad08 100644 --- a/fs/ocfs2/acl.c +++ b/fs/ocfs2/acl.c @@ -260,7 +260,7 @@ static int ocfs2_set_acl(handle_t *handle, return ret; } -int ocfs2_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { struct buffer_head *bh = NULL; diff --git a/fs/ocfs2/acl.h b/fs/ocfs2/acl.h index a91f9ce278d6..1ed05899cce1 100644 --- a/fs/ocfs2/acl.h +++ b/fs/ocfs2/acl.h @@ -17,7 +17,7 @@ struct ocfs2_acl_entry { }; struct posix_acl *ocfs2_iop_get_acl(struct inode *inode, int type, bool rcu); -int ocfs2_iop_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_iop_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int ocfs2_acl_chmod(struct inode *, struct buffer_head *); struct ocfs2_acl_state { diff --git a/fs/ocfs2/buffer_head_io.c b/fs/ocfs2/buffer_head_io.c index 7bfe377af2df..733ceda79ca1 100644 --- a/fs/ocfs2/buffer_head_io.c +++ b/fs/ocfs2/buffer_head_io.c @@ -66,12 +66,14 @@ int ocfs2_write_block(struct ocfs2_super *osb, struct buffer_head *bh, wait_on_buffer(bh); - if (buffer_uptodate(bh)) { + if (!buffer_write_io_error(bh)) { ocfs2_set_buffer_uptodate(ci, bh); } else { - /* We don't need to remove the clustered uptodate - * information for this bh as it's not marked locally - * uptodate. */ + /* + * The buffer still holds what we tried to write, but it did + * not reach the disk, so don't advertise it to the cluster + * as up to date. + */ ret = -EIO; mlog_errno(ret); } @@ -446,7 +448,7 @@ int ocfs2_write_super_or_backup(struct ocfs2_super *osb, wait_on_buffer(bh); - if (!buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { ret = -EIO; mlog_errno(ret); } diff --git a/fs/ocfs2/dlmfs/dlmfs.c b/fs/ocfs2/dlmfs/dlmfs.c index 53df5dd10ad0..d3bfcada3e3b 100644 --- a/fs/ocfs2/dlmfs/dlmfs.c +++ b/fs/ocfs2/dlmfs/dlmfs.c @@ -188,7 +188,7 @@ static int dlmfs_file_release(struct inode *inode, * We do ->setattr() just to override size changes. Our size is the size * of the LVB and nothing else. */ -static int dlmfs_file_setattr(struct mnt_idmap *idmap, +static int dlmfs_file_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int error; @@ -402,7 +402,7 @@ static struct inode *dlmfs_get_inode(struct inode *parent, * File creation. Allocate an inode, and we're done.. */ /* SMP-safe */ -static struct dentry *dlmfs_mkdir(struct mnt_idmap * idmap, +static struct dentry *dlmfs_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) @@ -450,7 +450,7 @@ bail: return ERR_PTR(status); } -static int dlmfs_create(struct mnt_idmap *idmap, +static int dlmfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) diff --git a/fs/ocfs2/file.c b/fs/ocfs2/file.c index d6e977ba6565..62f45a1b5ca1 100644 --- a/fs/ocfs2/file.c +++ b/fs/ocfs2/file.c @@ -1117,7 +1117,7 @@ out: return ret; } -int ocfs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int status = 0, size_change; @@ -1317,7 +1317,7 @@ bail: return status; } -int ocfs2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ocfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct inode *inode = d_inode(path->dentry); @@ -1349,7 +1349,7 @@ bail: return err; } -int ocfs2_permission(struct mnt_idmap *idmap, struct inode *inode, +int ocfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret, had_lock; diff --git a/fs/ocfs2/file.h b/fs/ocfs2/file.h index 41e65e45a9f3..97492ee5789e 100644 --- a/fs/ocfs2/file.h +++ b/fs/ocfs2/file.h @@ -50,11 +50,11 @@ int ocfs2_extend_no_holes(struct inode *inode, struct buffer_head *di_bh, u64 new_i_size, u64 zero_to); int ocfs2_zero_extend(struct inode *inode, struct buffer_head *di_bh, loff_t zero_to); -int ocfs2_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ocfs2_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ocfs2_getattr(struct mnt_idmap *idmap, const struct path *path, +int ocfs2_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int ocfs2_permission(struct mnt_idmap *idmap, +int ocfs2_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); diff --git a/fs/ocfs2/ioctl.c b/fs/ocfs2/ioctl.c index cbe59d231666..36c7c9ac8b5d 100644 --- a/fs/ocfs2/ioctl.c +++ b/fs/ocfs2/ioctl.c @@ -82,7 +82,7 @@ int ocfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return status; } -int ocfs2_fileattr_set(struct mnt_idmap *idmap, +int ocfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ocfs2/ioctl.h b/fs/ocfs2/ioctl.h index 4a1c2313b429..b1cb529fc5f9 100644 --- a/fs/ocfs2/ioctl.h +++ b/fs/ocfs2/ioctl.h @@ -12,7 +12,7 @@ #define OCFS2_IOCTL_PROTO_H int ocfs2_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ocfs2_fileattr_set(struct mnt_idmap *idmap, +int ocfs2_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long ocfs2_ioctl(struct file *filp, unsigned int cmd, unsigned long arg); long ocfs2_compat_ioctl(struct file *file, unsigned cmd, unsigned long arg); diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb..ea6802d894c2 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -676,19 +676,20 @@ static int __ocfs2_journal_access(handle_t *handle, mlog(ML_ERROR, "giving me a buffer that's not uptodate!\n"); mlog(ML_ERROR, "b_blocknr=%llu, b_state=0x%lx\n", (unsigned long long)bh->b_blocknr, bh->b_state); - + } + /* + * A previous transaction with a couple of buffer heads fail + * to checkpoint, so all the bhs are marked as BH_Write_EIO. + * For current transaction, the bh is just among those error + * bhs which previous transaction handle. We can't just clear + * its BH_Write_EIO and reuse directly, since other bhs are + * not written to disk yet and that will cause metadata + * inconsistency. So we should set fs read-only to avoid + * further damage. + */ + if (buffer_write_io_error(bh)) { lock_buffer(bh); - /* - * A previous transaction with a couple of buffer heads fail - * to checkpoint, so all the bhs are marked as BH_Write_EIO. - * For current transaction, the bh is just among those error - * bhs which previous transaction handle. We can't just clear - * its BH_Write_EIO and reuse directly, since other bhs are - * not written to disk yet and that will cause metadata - * inconsistency. So we should set fs read-only to avoid - * further damage. - */ - if (buffer_write_io_error(bh) && !buffer_uptodate(bh)) { + if (buffer_write_io_error(bh)) { unlock_buffer(bh); return ocfs2_error(osb->sb, "A previous attempt to " "write this buffer head failed\n"); diff --git a/fs/ocfs2/namei.c b/fs/ocfs2/namei.c index 58c6061ed983..fce9a31a3671 100644 --- a/fs/ocfs2/namei.c +++ b/fs/ocfs2/namei.c @@ -227,7 +227,7 @@ static void ocfs2_cleanup_add_entry_failure(struct ocfs2_super *osb, iput(inode); } -static int ocfs2_mknod(struct mnt_idmap *idmap, +static int ocfs2_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -650,7 +650,7 @@ static int ocfs2_mknod_locked(struct ocfs2_super *osb, suballoc_loc, suballoc_bit); } -static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap, +static struct dentry *ocfs2_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -666,7 +666,7 @@ static struct dentry *ocfs2_mkdir(struct mnt_idmap *idmap, return ERR_PTR(ret); } -static int ocfs2_create(struct mnt_idmap *idmap, +static int ocfs2_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -1206,7 +1206,7 @@ static void ocfs2_double_unlock(struct inode *inode1, struct inode *inode2) ocfs2_inode_unlock(inode2, 1); } -static int ocfs2_rename(struct mnt_idmap *idmap, +static int ocfs2_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, @@ -1810,7 +1810,7 @@ bail: return status; } -static int ocfs2_symlink(struct mnt_idmap *idmap, +static int ocfs2_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index bfafe059bedf..5274f4d571b6 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -7500,7 +7500,7 @@ static int ocfs2_xattr_security_get(const struct xattr_handler *handler, } static int ocfs2_xattr_security_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -7595,7 +7595,7 @@ static int ocfs2_xattr_trusted_get(const struct xattr_handler *handler, } static int ocfs2_xattr_trusted_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -7626,7 +7626,7 @@ static int ocfs2_xattr_user_get(const struct xattr_handler *handler, } static int ocfs2_xattr_user_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/omfs/dir.c b/fs/omfs/dir.c index 692297cf84e7..18f4b4543cc8 100644 --- a/fs/omfs/dir.c +++ b/fs/omfs/dir.c @@ -279,13 +279,13 @@ out_free_inode: return err; } -static struct dentry *omfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *omfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(omfs_add_node(dir, dentry, mode)); } -static int omfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int omfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return omfs_add_node(dir, dentry, mode | S_IFREG); @@ -370,7 +370,7 @@ static bool omfs_fill_chain(struct inode *dir, struct dir_context *ctx, return true; } -static int omfs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int omfs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/omfs/file.c b/fs/omfs/file.c index 28f3b113340e..79a413f1dc0d 100644 --- a/fs/omfs/file.c +++ b/fs/omfs/file.c @@ -338,7 +338,7 @@ const struct file_operations omfs_file_operations = { .splice_read = filemap_splice_read, }; -static int omfs_setattr(struct mnt_idmap *idmap, +static int omfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/omfs/inode.c b/fs/omfs/inode.c index 1d915ef72119..bc37029a4afb 100644 --- a/fs/omfs/inode.c +++ b/fs/omfs/inode.c @@ -145,7 +145,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh); if (wait) { sync_dirty_buffer(bh); - if (buffer_req(bh) && !buffer_uptodate(bh)) + if (buffer_write_io_error(bh)) sync_failed = 1; } @@ -159,7 +159,7 @@ static int __omfs_write_inode(struct inode *inode, int wait) mark_buffer_dirty(bh2); if (wait) { sync_dirty_buffer(bh2); - if (buffer_req(bh2) && !buffer_uptodate(bh2)) + if (buffer_write_io_error(bh2)) sync_failed = 1; } brelse(bh2); diff --git a/fs/open.c b/fs/open.c index 6b1c14e684a9..e43f02ff64ac 100644 --- a/fs/open.c +++ b/fs/open.c @@ -36,7 +36,7 @@ #include "internal.h" -int do_truncate(struct mnt_idmap *idmap, struct dentry *dentry, +int do_truncate(const struct mnt_idmap *idmap, struct dentry *dentry, loff_t length, unsigned int time_attrs, struct file *filp) { int ret; @@ -72,7 +72,7 @@ int do_truncate(struct mnt_idmap *idmap, struct dentry *dentry, int vfs_truncate(const struct path *path, loff_t length) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode; int error; @@ -787,7 +787,7 @@ static inline bool setattr_vfsgid(struct iattr *attr, kgid_t kgid) int chown_common(const struct path *path, uid_t user, gid_t group) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct user_namespace *fs_userns; struct inode *inode = path->dentry->d_inode; struct delegated_inode delegated_inode = { }; @@ -931,6 +931,11 @@ cleanup_inode: return error; } +/* + * Populate struct file + * + * NOTE: it assumes f_path is populated and consumes the caller's reference. + */ static int do_dentry_open(struct file *f, int (*open)(struct inode *, struct file *)) { @@ -938,7 +943,6 @@ static int do_dentry_open(struct file *f, struct inode *inode = f->f_path.dentry->d_inode; int error; - path_get(&f->f_path); f->f_inode = inode; f->f_mapping = inode->i_mapping; f->f_wb_err = filemap_sample_wb_err(f->f_mapping); @@ -1055,6 +1059,7 @@ int finish_open(struct file *file, struct dentry *dentry, BUG_ON(file->f_mode & FMODE_OPENED); /* once it's opened, it's opened */ file->__f_path.dentry = dentry; + path_get(&file->f_path); return do_dentry_open(file, open); } EXPORT_SYMBOL(finish_open); @@ -1098,6 +1103,7 @@ int vfs_open(const struct path *path, struct file *file) int ret; file->__f_path = *path; + path_get(&file->f_path); ret = do_dentry_open(file, NULL); if (!ret) { /* @@ -1110,6 +1116,25 @@ int vfs_open(const struct path *path, struct file *file) return ret; } +/** + * vfs_open_consume - open the file at the given path and consume the reference + * @path: path to open + * @file: newly allocated file with f_flag initialized + */ +int vfs_open_consume(struct path *path, struct file *file) +{ + int ret; + + file->__f_path = *path; + path->mnt = NULL; + path->dentry = NULL; + ret = do_dentry_open(file, NULL); + if (!ret) { + fsnotify_open(file); + } + return ret; +} + struct file *dentry_open(const struct path *path, int flags, const struct cred *cred) { @@ -1537,6 +1562,19 @@ int filp_close(struct file *filp, fl_owner_t id) } EXPORT_SYMBOL(filp_close); +/* Like filp_close() but the last reference is put right here. */ +int filp_close_sync(struct file *filp, fl_owner_t id) +{ + int retval; + + /* Kernel threads must never put their final reference here. */ + VFS_WARN_ON_ONCE(current->flags & PF_KTHREAD); + retval = filp_flush(filp, id); + fput_close_sync(filp); + + return retval; +} + /* * Careful here! We test whether the file pointer is NULL before * releasing the fd. This ensures that one clone task can't release @@ -1551,13 +1589,11 @@ SYSCALL_DEFINE1(close, unsigned int, fd) if (!file) return -EBADF; - retval = filp_flush(file, current->files); - /* * We're returning to user space. Don't bother * with any delayed fput() cases. */ - fput_close_sync(file); + retval = filp_close_sync(file, current->files); if (likely(retval == 0)) return 0; diff --git a/fs/orangefs/acl.c b/fs/orangefs/acl.c index a01ef0c1b1bf..f31196e4bfaa 100644 --- a/fs/orangefs/acl.c +++ b/fs/orangefs/acl.c @@ -112,7 +112,7 @@ int __orangefs_set_acl(struct inode *inode, struct posix_acl *acl, int type) return error; } -int orangefs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int orangefs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; diff --git a/fs/orangefs/inode.c b/fs/orangefs/inode.c index c088a02e8215..b1fed1c81a4d 100644 --- a/fs/orangefs/inode.c +++ b/fs/orangefs/inode.c @@ -847,7 +847,7 @@ int __orangefs_setattr_mode(struct dentry *dentry, struct iattr *iattr) /* * Change attributes of an object referenced by dentry. */ -int orangefs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int orangefs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { int ret; @@ -867,7 +867,7 @@ out: /* * Obtain attributes of an object given a dentry */ -int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, +int orangefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { int ret; @@ -891,7 +891,7 @@ int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, return ret; } -int orangefs_permission(struct mnt_idmap *idmap, +int orangefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { int ret; @@ -953,7 +953,7 @@ static int orangefs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -static int orangefs_fileattr_set(struct mnt_idmap *idmap, +static int orangefs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { u64 val = 0; diff --git a/fs/orangefs/namei.c b/fs/orangefs/namei.c index 8ebc34e112d5..32b7769ea49c 100644 --- a/fs/orangefs/namei.c +++ b/fs/orangefs/namei.c @@ -15,7 +15,7 @@ /* * Get a newly allocated inode to go with a negative dentry. */ -static int orangefs_create(struct mnt_idmap *idmap, +static int orangefs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -211,7 +211,7 @@ static int orangefs_unlink(struct inode *dir, struct dentry *dentry) return ret; } -static int orangefs_symlink(struct mnt_idmap *idmap, +static int orangefs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) @@ -296,7 +296,7 @@ out: return ret; } -static struct dentry *orangefs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *orangefs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct orangefs_inode_s *parent = ORANGEFS_I(dir); @@ -364,7 +364,7 @@ out: return ret ? ERR_PTR(ret) : NULL; } -static int orangefs_rename(struct mnt_idmap *idmap, +static int orangefs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, diff --git a/fs/orangefs/orangefs-kernel.h b/fs/orangefs/orangefs-kernel.h index 1451fc2c1917..348fe340c5d5 100644 --- a/fs/orangefs/orangefs-kernel.h +++ b/fs/orangefs/orangefs-kernel.h @@ -98,7 +98,7 @@ enum orangefs_vfs_op_states { extern const struct xattr_handler * const orangefs_xattr_handlers[]; extern struct posix_acl *orangefs_get_acl(struct inode *inode, int type, bool rcu); -extern int orangefs_set_acl(struct mnt_idmap *idmap, +extern int orangefs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int __orangefs_set_acl(struct inode *inode, struct posix_acl *acl, int type); @@ -352,12 +352,12 @@ struct inode *orangefs_new_inode(struct super_block *sb, int __orangefs_setattr(struct inode *, struct iattr *); int __orangefs_setattr_mode(struct dentry *dentry, struct iattr *iattr); -int orangefs_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +int orangefs_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); -int orangefs_getattr(struct mnt_idmap *idmap, const struct path *path, +int orangefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int orangefs_permission(struct mnt_idmap *idmap, +int orangefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); int orangefs_update_time(struct inode *inode, enum fs_update_time type, diff --git a/fs/orangefs/xattr.c b/fs/orangefs/xattr.c index 885fd3bd5a3d..a49e64566e1e 100644 --- a/fs/orangefs/xattr.c +++ b/fs/orangefs/xattr.c @@ -527,7 +527,7 @@ out_unlock: } static int orangefs_xattr_set_default(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, diff --git a/fs/overlayfs/dir.c b/fs/overlayfs/dir.c index 7beb0af26498..1194ccf981c3 100644 --- a/fs/overlayfs/dir.c +++ b/fs/overlayfs/dir.c @@ -688,7 +688,7 @@ static int ovl_create_or_link(struct dentry *dentry, struct inode *inode, return err; } -static int ovl_create_object(struct mnt_idmap *idmap, struct dentry *dentry, +static int ovl_create_object(const struct mnt_idmap *idmap, struct dentry *dentry, int mode, dev_t rdev, const char *link) { int err; @@ -730,19 +730,19 @@ out: return err; } -static int ovl_create(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ovl_create_object(idmap, dentry, (mode & 07777) | S_IFREG, 0, NULL); } -static struct dentry *ovl_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ovl_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ERR_PTR(ovl_create_object(idmap, dentry, (mode & 07777) | S_IFDIR, 0, NULL)); } -static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { /* Don't allow creation of "whiteout" on overlay */ @@ -752,7 +752,7 @@ static int ovl_mknod(struct mnt_idmap *idmap, struct inode *dir, return ovl_create_object(idmap, dentry, mode, rdev, NULL); } -static int ovl_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *link) { return ovl_create_object(idmap, dentry, S_IFLNK, 0, link); @@ -1344,7 +1344,7 @@ static void ovl_rename_end(struct ovl_renamedata *ovlrd) ovl_drop_write(ovlrd->old_dentry); } -static int ovl_rename(struct mnt_idmap *idmap, struct inode *olddir, +static int ovl_rename(const struct mnt_idmap *idmap, struct inode *olddir, struct dentry *old, struct inode *newdir, struct dentry *new, unsigned int flags) { @@ -1420,7 +1420,7 @@ static int ovl_dummy_open(struct inode *inode, struct file *file) return 0; } -static int ovl_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ovl_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { int err; diff --git a/fs/overlayfs/file.c b/fs/overlayfs/file.c index f3d97eb146e8..7433220d4ad6 100644 --- a/fs/overlayfs/file.c +++ b/fs/overlayfs/file.c @@ -30,7 +30,7 @@ static struct file *ovl_open_realfile(const struct file *file, { struct inode *realinode = d_inode(realpath->dentry); struct inode *inode = file_inode(file); - struct mnt_idmap *real_idmap; + const struct mnt_idmap *real_idmap; struct file *realfile; int flags = file->f_flags | OVL_OPEN_FLAGS; int acc_mode = ACC_MODE(flags); diff --git a/fs/overlayfs/inode.c b/fs/overlayfs/inode.c index 401cb8c75520..70183d516e5e 100644 --- a/fs/overlayfs/inode.c +++ b/fs/overlayfs/inode.c @@ -18,7 +18,7 @@ #include "overlayfs.h" -int ovl_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int err; @@ -168,7 +168,7 @@ static inline int ovl_real_getattr_nosec(struct super_block *sb, return vfs_getattr_nosec(path, stat, request_mask, flags); } -int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, +int ovl_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct dentry *dentry = path->dentry; @@ -303,7 +303,7 @@ int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, return err; } -int ovl_permission(struct mnt_idmap *idmap, +int ovl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct inode *upperinode = ovl_inode_upper(inode); @@ -355,7 +355,7 @@ static const char *ovl_get_link(struct dentry *dentry, * alter the POSIX ACLs for the underlying filesystem. */ static void ovl_idmap_posix_acl(const struct inode *realinode, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct posix_acl *acl) { struct user_namespace *fs_userns = i_user_ns(realinode); @@ -406,7 +406,7 @@ struct posix_acl *ovl_get_acl_path(const struct path *path, const char *acl_name, bool noperm) { struct posix_acl *real_acl, *clone; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *realinode = d_inode(path->dentry); idmap = mnt_idmap(path->mnt); @@ -447,7 +447,7 @@ struct posix_acl *ovl_get_acl_path(const struct path *path, * * This is obviously only relevant when idmapped layers are used. */ -struct posix_acl *do_ovl_get_acl(struct mnt_idmap *idmap, +struct posix_acl *do_ovl_get_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, bool rcu, bool noperm) { @@ -536,7 +536,7 @@ out: return err; } -int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int err; @@ -650,7 +650,7 @@ int ovl_real_fileattr_set(const struct path *realpath, struct file_kattr *fa) return vfs_fileattr_set(mnt_idmap(realpath->mnt), realpath->dentry, fa); } -int ovl_fileattr_set(struct mnt_idmap *idmap, +int ovl_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/overlayfs/overlayfs.h b/fs/overlayfs/overlayfs.h index 7f3558372c59..53fbbe15c31d 100644 --- a/fs/overlayfs/overlayfs.h +++ b/fs/overlayfs/overlayfs.h @@ -804,11 +804,11 @@ int ovl_set_nlink_lower(struct dentry *dentry); unsigned int ovl_get_nlink(struct ovl_fs *ofs, struct dentry *lowerdentry, struct dentry *upperdentry, unsigned int fallback); -int ovl_permission(struct mnt_idmap *idmap, struct inode *inode, +int ovl_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); #ifdef CONFIG_FS_POSIX_ACL -struct posix_acl *do_ovl_get_acl(struct mnt_idmap *idmap, +struct posix_acl *do_ovl_get_acl(const struct mnt_idmap *idmap, struct inode *inode, int type, bool rcu, bool noperm); static inline struct posix_acl *ovl_get_inode_acl(struct inode *inode, int type, @@ -816,12 +816,12 @@ static inline struct posix_acl *ovl_get_inode_acl(struct inode *inode, int type, { return do_ovl_get_acl(&nop_mnt_idmap, inode, type, rcu, true); } -static inline struct posix_acl *ovl_get_acl(struct mnt_idmap *idmap, +static inline struct posix_acl *ovl_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { return do_ovl_get_acl(idmap, d_inode(dentry), type, false, false); } -int ovl_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); struct posix_acl *ovl_get_acl_path(const struct path *path, const char *acl_name, bool noperm); @@ -916,7 +916,7 @@ extern const struct file_operations ovl_file_operations; int ovl_real_fileattr_get(const struct path *realpath, struct file_kattr *fa); int ovl_real_fileattr_set(const struct path *realpath, struct file_kattr *fa); int ovl_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ovl_fileattr_set(struct mnt_idmap *idmap, +int ovl_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); struct ovl_file; struct ovl_file *ovl_file_alloc(struct file *realfile); @@ -950,8 +950,8 @@ static inline bool ovl_force_readonly(struct ovl_fs *ofs) /* xattr.c */ const struct xattr_handler * const *ovl_xattr_handlers(struct ovl_fs *ofs); -int ovl_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ovl_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); -int ovl_getattr(struct mnt_idmap *idmap, const struct path *path, +int ovl_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); ssize_t ovl_listxattr(struct dentry *dentry, char *list, size_t size); diff --git a/fs/overlayfs/ovl_entry.h b/fs/overlayfs/ovl_entry.h index 80cad4ea96a3..ac07e8769f9b 100644 --- a/fs/overlayfs/ovl_entry.h +++ b/fs/overlayfs/ovl_entry.h @@ -105,7 +105,7 @@ static inline struct vfsmount *ovl_upper_mnt(struct ovl_fs *ofs) return ofs->layers[0].mnt; } -static inline struct mnt_idmap *ovl_upper_mnt_idmap(struct ovl_fs *ofs) +static inline const struct mnt_idmap *ovl_upper_mnt_idmap(struct ovl_fs *ofs) { return mnt_idmap(ovl_upper_mnt(ofs)); } diff --git a/fs/overlayfs/util.c b/fs/overlayfs/util.c index b41f4788e4f0..521717209b2e 100644 --- a/fs/overlayfs/util.c +++ b/fs/overlayfs/util.c @@ -657,7 +657,7 @@ bool ovl_path_is_whiteout(struct ovl_fs *ofs, const struct path *path) struct file *ovl_path_open(const struct path *path, int flags) { struct inode *inode = d_inode(path->dentry); - struct mnt_idmap *real_idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *real_idmap = mnt_idmap(path->mnt); int err, acc_mode; if (flags & ~(O_ACCMODE | O_LARGEFILE)) @@ -1496,7 +1496,7 @@ void ovl_copyattr(struct inode *inode) { struct path realpath; struct inode *realinode; - struct mnt_idmap *real_idmap; + const struct mnt_idmap *real_idmap; vfsuid_t vfsuid; vfsgid_t vfsgid; diff --git a/fs/overlayfs/xattrs.c b/fs/overlayfs/xattrs.c index 5ae44b9c8790..acc54f6138ef 100644 --- a/fs/overlayfs/xattrs.c +++ b/fs/overlayfs/xattrs.c @@ -190,7 +190,7 @@ static int ovl_own_xattr_get(const struct xattr_handler *handler, } static int ovl_own_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) @@ -217,7 +217,7 @@ static int ovl_other_xattr_get(const struct xattr_handler *handler, } static int ovl_other_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/pidfs.c b/fs/pidfs.c index a6a643f15d08..c37c17bcdbdc 100644 --- a/fs/pidfs.c +++ b/fs/pidfs.c @@ -823,13 +823,13 @@ static struct vfsmount *pidfs_mnt __ro_after_init; * implemented. Let's reject it completely until we have a clean * permission concept for pidfds. */ -static int pidfs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int pidfs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { return anon_inode_setattr(idmap, dentry, attr); } -static int pidfs_getattr(struct mnt_idmap *idmap, const struct path *path, +static int pidfs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { @@ -1102,7 +1102,7 @@ static int pidfs_xattr_get(const struct xattr_handler *handler, } static int pidfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *suffix, const void *value, size_t size, int flags) { diff --git a/fs/pipe.c b/fs/pipe.c index 292425a834dd..5791db016e24 100644 --- a/fs/pipe.c +++ b/fs/pipe.c @@ -1433,7 +1433,7 @@ int pipe_resize_ring(struct pipe_inode_info *pipe, unsigned int nr_slots) spin_unlock_irq(&pipe->rd_wait.lock); /* This might have made more room for writers */ - wake_up_interruptible(&pipe->wr_wait); + wake_up_interruptible_poll(&pipe->wr_wait, EPOLLOUT | EPOLLWRNORM); return 0; } diff --git a/fs/pnode.c b/fs/pnode.c index 5d91c3e58d2a..2cd667958efe 100644 --- a/fs/pnode.c +++ b/fs/pnode.c @@ -410,19 +410,99 @@ bool propagation_would_overmount(const struct mount *from, return false; } +/* Does @m receive propagation from @parent? */ +static bool receives_from(struct mount *m, struct mount *parent) +{ + if (m == parent) + return false; + for (; m; m = m->mnt_master) + if (m == parent || peers(m, parent)) + return true; + return false; +} + +/* + * Does @m receive propagation from the victim's parent as well? If so, then + * the mount at the victim's mountpoint inside of @m is a umount candidate as + * well. So it's the next candidate in the chain. Otherwise the chain ends at + * @m. + */ +static struct mount *next_candidate(struct mount *m, struct mount *victim) +{ + if (!receives_from(m, victim->mnt_parent)) + return NULL; + return __lookup_mnt(&m->mnt, victim->mnt_mountpoint); +} + +/* + * Would propagate_umount() pull out a mount of the chain of candidates that + * starts at @c, and does that mount have references beyond its own? + * + * This mirrors how trim_one(), trim_ancestors() and handle_locked() handle a + * synchronous umount: + * + * - single victim + * - without children + * - with MNT_LOCKED already cleared on every candidate by propagate_mount_unlock() + * + * A copy of the victim gets unmounted when each of its children is + * the next candidate in the chain or its overmount, unless the next + * unmount candidate is not its overmount and some unmount candidate further + * down has a child outside the chain. Keep this in sync with + * Documentation/filesystems/propagate_umount.txt. + */ +static bool chain_busy(struct mount *c, struct mount *victim) +{ + struct mount *m, *n, *next, *deepest = NULL; + bool above; + + /* the deepest candidate with a child outside the chain */ + for (m = c; m; m = next) { + next = next_candidate(m, victim); + list_for_each_entry(n, &m->mnt_mounts, mnt_child) { + if (n != next && n != victim) { + deepest = m; + break; + } + } + } + + above = deepest != NULL; /* @deepest is at or below @m */ + for (m = c; m; m = next) { + bool goes = true; + + next = next_candidate(m, victim); + list_for_each_entry(n, &m->mnt_mounts, mnt_child) { + if (n != next && n != m->overmount && n != victim) { + goes = false; + break; + } + } + if (goes && next && next != m->overmount && above && m != deepest) + goes = false; + if (m == deepest) + above = false; + if (goes && do_refcount_check(m, 1)) + return true; + } + return false; +} + /* * check if the mount 'mnt' can be unmounted successfully. * @mnt: the mount to be checked for unmount * NOTE: unmounting 'mnt' would naturally propagate to all * other mounts its parent propagates to. - * Check if any of these mounts that **do not have submounts** - * have more references than 'refcnt'. If so return busy. + * Check if any of the mounts that propagate_umount() would pull out + * along with it have more references than their own. If so return busy. * * vfsmount lock must be held for write */ int propagate_mount_busy(struct mount *mnt, int refcnt) { struct mount *parent = mnt->mnt_parent; + struct dentry *mp = mnt->mnt_mountpoint; + struct mount *m; /* * quickly check if the current mount can be unmounted. @@ -435,24 +515,16 @@ int propagate_mount_busy(struct mount *mnt, int refcnt) if (mnt == parent) return 0; - for (struct mount *m = propagation_next(parent, parent); m; - m = propagation_next(m, parent)) { - struct list_head *head; - struct mount *child = __lookup_mnt(&m->mnt, mnt->mnt_mountpoint); + /* the candidates are the mounts at @mp below the receivers */ + for (m = propagation_next(parent, parent); m; + m = propagation_next(m, parent)) { + struct mount *c = __lookup_mnt(&m->mnt, mp); - if (!child) + /* each chain once, from its top: skip receivers that are candidates */ + if (!c || (mnt_has_parent(m) && m->mnt_mountpoint == mp && + receives_from(m->mnt_parent, parent))) continue; - - head = &child->mnt_mounts; - if (!list_empty(head)) { - /* - * a mount that covers child completely wouldn't prevent - * it being pulled out; any other would. - */ - if (!list_is_singular(head) || !child->overmount) - continue; - } - if (do_refcount_check(child, 1)) + if (chain_busy(c, mnt)) return 1; } return 0; diff --git a/fs/posix_acl.c b/fs/posix_acl.c index 18b302f94174..fe77934ea8f2 100644 --- a/fs/posix_acl.c +++ b/fs/posix_acl.c @@ -118,7 +118,7 @@ void forget_all_cached_acls(struct inode *inode) } EXPORT_SYMBOL(forget_all_cached_acls); -static struct posix_acl *__get_acl(struct mnt_idmap *idmap, +static struct posix_acl *__get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, int type) { @@ -378,7 +378,7 @@ EXPORT_SYMBOL(posix_acl_from_mode); * by the acl. Returns -E... otherwise. */ int -posix_acl_permission(struct mnt_idmap *idmap, struct inode *inode, +posix_acl_permission(const struct mnt_idmap *idmap, struct inode *inode, const struct posix_acl *acl, int want) { const struct posix_acl_entry *pa, *pe, *mask_obj; @@ -608,7 +608,7 @@ EXPORT_SYMBOL(__posix_acl_chmod); * performed on the raw inode simply pass @nop_mnt_idmap. */ int - posix_acl_chmod(struct mnt_idmap *idmap, struct dentry *dentry, + posix_acl_chmod(const struct mnt_idmap *idmap, struct dentry *dentry, umode_t mode) { struct inode *inode = d_inode(dentry); @@ -709,7 +709,7 @@ EXPORT_SYMBOL_GPL(posix_acl_create); * * Called from set_acl inode operations. */ -int posix_acl_update_mode(struct mnt_idmap *idmap, +int posix_acl_update_mode(const struct mnt_idmap *idmap, struct inode *inode, umode_t *mode_p, struct posix_acl **acl) { @@ -889,7 +889,7 @@ EXPORT_SYMBOL (posix_acl_to_xattr); * Return: On success, the size of the stored uapi posix acls, on error a * negative errno. */ -static ssize_t vfs_posix_acl_to_xattr(struct mnt_idmap *idmap, +static ssize_t vfs_posix_acl_to_xattr(const struct mnt_idmap *idmap, struct inode *inode, const struct posix_acl *acl, void *buffer, size_t size) @@ -937,7 +937,7 @@ static ssize_t vfs_posix_acl_to_xattr(struct mnt_idmap *idmap, } int -set_posix_acl(struct mnt_idmap *idmap, struct dentry *dentry, +set_posix_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type, struct posix_acl *acl) { struct inode *inode = d_inode(dentry); @@ -1018,7 +1018,7 @@ const struct xattr_handler nop_posix_acl_default = { }; EXPORT_SYMBOL_GPL(nop_posix_acl_default); -int simple_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int simple_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { int error; @@ -1057,7 +1057,7 @@ int simple_acl_create(struct inode *dir, struct inode *inode) return 0; } -static int vfs_set_acl_idmapped_mnt(struct mnt_idmap *idmap, +static int vfs_set_acl_idmapped_mnt(const struct mnt_idmap *idmap, struct user_namespace *fs_userns, struct posix_acl *acl) { @@ -1091,7 +1091,7 @@ static int vfs_set_acl_idmapped_mnt(struct mnt_idmap *idmap, * * Return: On success 0, on error negative errno. */ -int vfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, struct posix_acl *kacl) { int acl_type; @@ -1168,7 +1168,7 @@ EXPORT_SYMBOL_GPL(vfs_set_acl); * * Return: On success POSIX ACLs in VFS format, on error negative errno. */ -struct posix_acl *vfs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *vfs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { struct inode *inode = d_inode(dentry); @@ -1212,7 +1212,7 @@ EXPORT_SYMBOL_GPL(vfs_get_acl); * * Return: On success 0, on error negative errno. */ -int vfs_remove_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int vfs_remove_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name) { int acl_type; @@ -1265,7 +1265,7 @@ out_inode_unlock: } EXPORT_SYMBOL_GPL(vfs_remove_acl); -int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int do_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, const void *kvalue, size_t size) { int error; @@ -1286,7 +1286,7 @@ int do_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, return error; } -ssize_t do_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +ssize_t do_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, const char *acl_name, void *kvalue, size_t size) { ssize_t error; diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62..c9ee0946ecaf 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -702,7 +702,7 @@ static int proc_pid_syscall(struct seq_file *m, struct pid_namespace *ns, /* Here the fs part begins */ /************************************************************************/ -int proc_nochmod_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int proc_nochmod_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int error; @@ -743,7 +743,7 @@ static bool has_pid_permissions(struct proc_fs_info *fs_info, } -static int proc_pid_permission(struct mnt_idmap *idmap, +static int proc_pid_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct proc_fs_info *fs_info = proc_sb_info(inode->i_sb); @@ -848,15 +848,35 @@ static int __mem_open(struct inode *inode, struct file *file, unsigned int mode) return 0; } +/* private_data for proc_mem_operations */ +struct mem_private { + struct mm_struct *mm; + /* + * Was the ptrace access check on open bypassed because the opener used + * the same MM (introspection)? + */ + bool opened_by_owner; +}; + static int mem_open(struct inode *inode, struct file *file) { + struct mem_private *priv __free(kfree) = kmalloc_obj(struct mem_private); + + if (!priv) + return -ENOMEM; if (WARN_ON_ONCE(!(file->f_op->fop_flags & FOP_UNSIGNED_OFFSET))) return -EINVAL; - return __mem_open(inode, file, PTRACE_MODE_ATTACH); + priv->mm = proc_mem_open(inode, PTRACE_MODE_ATTACH); + if (IS_ERR_OR_NULL(priv->mm)) + return priv->mm ? PTR_ERR(priv->mm) : -ESRCH; + priv->opened_by_owner = priv->mm == current->mm; + file->private_data = no_free_ptr(priv); + return 0; } static bool proc_mem_foll_force(struct file *file, struct mm_struct *mm) { + struct mem_private *priv = file->private_data; struct task_struct *task; bool ptrace_active = false; @@ -871,16 +891,20 @@ static bool proc_mem_foll_force(struct file *file, struct mm_struct *mm) READ_ONCE(task->parent) == current; put_task_struct(task); } - return ptrace_active; + if (!ptrace_active) + return false; + break; default: - return true; + break; } + return security_mem_foll_force(file->f_cred, priv->opened_by_owner) == 0; } static ssize_t mem_rw(struct file *file, char __user *buf, size_t count, loff_t *ppos, int write) { - struct mm_struct *mm = file->private_data; + struct mem_private *priv = file->private_data; + struct mm_struct *mm = priv->mm; unsigned long addr = *ppos; ssize_t copied; char *page; @@ -970,12 +994,21 @@ static int mem_release(struct inode *inode, struct file *file) return 0; } +static int mem_release_with_private(struct inode *inode, struct file *file) +{ + struct mem_private *priv = file->private_data; + + mmdrop(priv->mm); + kfree(priv); + return 0; +} + static const struct file_operations proc_mem_operations = { .llseek = mem_lseek, .read = mem_read, .write = mem_write, .open = mem_open, - .release = mem_release, + .release = mem_release_with_private, .fop_flags = FOP_UNSIGNED_OFFSET, }; @@ -1994,7 +2027,7 @@ static struct inode *proc_pid_make_base_inode(struct super_block *sb, return inode; } -int pid_getattr(struct mnt_idmap *idmap, const struct path *path, +int pid_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { struct inode *inode = d_inode(path->dentry); @@ -3607,7 +3640,7 @@ int proc_pid_readdir(struct file *file, struct dir_context *ctx) * This function makes sure that the node is always accessible for members of * same thread group. */ -static int proc_tid_comm_permission(struct mnt_idmap *idmap, +static int proc_tid_comm_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { bool is_same_tgroup; @@ -3936,7 +3969,7 @@ static int proc_task_readdir(struct file *file, struct dir_context *ctx) return 0; } -static int proc_task_getattr(struct mnt_idmap *idmap, +static int proc_task_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/fd.c b/fs/proc/fd.c index 0f9a1556f2a3..7214a7495380 100644 --- a/fs/proc/fd.c +++ b/fs/proc/fd.c @@ -82,7 +82,7 @@ static int seq_fdinfo_open(struct inode *inode, struct file *file) * that the current task has PTRACE_MODE_READ in addition to the normal * POSIX-like checks. */ -static int proc_fdinfo_permission(struct mnt_idmap *idmap, struct inode *inode, +static int proc_fdinfo_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { bool allowed = false; @@ -323,7 +323,7 @@ static struct dentry *proc_lookupfd(struct inode *dir, struct dentry *dentry, * /proc/pid/fd needs a special permission handler so that a process can still * access /proc/self/fd after it has executed a setuid(). */ -int proc_fd_permission(struct mnt_idmap *idmap, +int proc_fd_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { struct task_struct *p; @@ -342,7 +342,7 @@ int proc_fd_permission(struct mnt_idmap *idmap, return rv; } -static int proc_fd_getattr(struct mnt_idmap *idmap, +static int proc_fd_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/fd.h b/fs/proc/fd.h index 7e7265f7e06f..77f2e4f38592 100644 --- a/fs/proc/fd.h +++ b/fs/proc/fd.h @@ -10,7 +10,7 @@ extern const struct inode_operations proc_fd_inode_operations; extern const struct file_operations proc_fdinfo_operations; extern const struct inode_operations proc_fdinfo_inode_operations; -extern int proc_fd_permission(struct mnt_idmap *idmap, +extern int proc_fd_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask); static inline unsigned int proc_fd(struct inode *inode) diff --git a/fs/proc/generic.c b/fs/proc/generic.c index 26086a283672..2b1971da4a85 100644 --- a/fs/proc/generic.c +++ b/fs/proc/generic.c @@ -117,7 +117,7 @@ static bool pde_subdir_insert(struct proc_dir_entry *dir, return true; } -static int proc_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int proc_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); @@ -135,7 +135,7 @@ static int proc_setattr(struct mnt_idmap *idmap, struct dentry *dentry, return 0; } -static int proc_getattr(struct mnt_idmap *idmap, +static int proc_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 04bd6c9e65a7..b9aaac41c283 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -258,9 +258,9 @@ extern int proc_pid_statm(struct seq_file *, struct pid_namespace *, * base.c */ extern const struct dentry_operations pid_dentry_operations; -extern int pid_getattr(struct mnt_idmap *, const struct path *, +extern int pid_getattr(const struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); -int proc_nochmod_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int proc_nochmod_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); extern void proc_pid_evict_inode(struct proc_inode *); extern struct inode *proc_pid_make_inode(struct super_block *, struct task_struct *, umode_t); diff --git a/fs/proc/proc_net.c b/fs/proc/proc_net.c index 00cc385bce21..b1f5eafb069a 100644 --- a/fs/proc/proc_net.c +++ b/fs/proc/proc_net.c @@ -308,7 +308,7 @@ static struct dentry *proc_tgid_net_lookup(struct inode *dir, return de; } -static int proc_tgid_net_getattr(struct mnt_idmap *idmap, +static int proc_tgid_net_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/proc_sysctl.c b/fs/proc/proc_sysctl.c index 04a382178c65..d1cfd2941359 100644 --- a/fs/proc/proc_sysctl.c +++ b/fs/proc/proc_sysctl.c @@ -788,7 +788,7 @@ out: return 0; } -static int proc_sys_permission(struct mnt_idmap *idmap, +static int proc_sys_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { /* @@ -817,7 +817,7 @@ static int proc_sys_permission(struct mnt_idmap *idmap, return error; } -static int proc_sys_setattr(struct mnt_idmap *idmap, +static int proc_sys_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); @@ -834,7 +834,7 @@ static int proc_sys_setattr(struct mnt_idmap *idmap, return 0; } -static int proc_sys_getattr(struct mnt_idmap *idmap, +static int proc_sys_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/root.c b/fs/proc/root.c index 99adddfeb4a4..7fbbe92bf73a 100644 --- a/fs/proc/root.c +++ b/fs/proc/root.c @@ -402,7 +402,7 @@ void __init proc_root_init(void) register_filesystem(&proc_fs_type); } -static int proc_root_getattr(struct mnt_idmap *idmap, +static int proc_root_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/proc/vmcore.c b/fs/proc/vmcore.c index 44d15436439f..406898247d7a 100644 --- a/fs/proc/vmcore.c +++ b/fs/proc/vmcore.c @@ -1709,6 +1709,24 @@ static void vmcore_free_device_dumps(void) #endif /* CONFIG_PROC_VMCORE_DEVICE_DUMP */ } +#define VMCOREINFO_OSRELEASE_KEY "OSRELEASE=" + +static void __init vmcore_report_crashed_release(void) +{ + const char *ver, *eol; + + ver = strnstr(elfnotes_buf, VMCOREINFO_OSRELEASE_KEY, elfnotes_sz); + if (!ver) + return; + + ver += sizeof(VMCOREINFO_OSRELEASE_KEY) - 1; + eol = memchr(ver, '\n', elfnotes_buf + elfnotes_sz - ver); + if (!eol) + return; + + pr_notice("dump is from kernel %.*s\n", (int)(eol - ver), ver); +} + /* Init function for vmcore module. */ static int __init vmcore_init(void) { @@ -1733,6 +1751,8 @@ static int __init vmcore_init(void) elfcorehdr_free(elfcorehdr_addr); elfcorehdr_addr = ELFCORE_ADDR_ERR; + vmcore_report_crashed_release(); + proc_vmcore = proc_create("vmcore", S_IRUSR, NULL, &vmcore_proc_ops); if (proc_vmcore) proc_vmcore->size = vmcore_size; diff --git a/fs/quota/dquot.c b/fs/quota/dquot.c index cde707fad46f..3ad19d7e472f 100644 --- a/fs/quota/dquot.c +++ b/fs/quota/dquot.c @@ -2095,7 +2095,7 @@ EXPORT_SYMBOL(__dquot_transfer); /* Wrapper for transferring ownership of an inode for uid/gid only * Called from FSXXX_setattr() */ -int dquot_transfer(struct mnt_idmap *idmap, struct inode *inode, +int dquot_transfer(const struct mnt_idmap *idmap, struct inode *inode, struct iattr *iattr) { struct dquot *transfer_to[MAXQUOTAS] = {}; diff --git a/fs/ramfs/file-nommu.c b/fs/ramfs/file-nommu.c index fb471bf88ab7..7ed6fba134c6 100644 --- a/fs/ramfs/file-nommu.c +++ b/fs/ramfs/file-nommu.c @@ -22,7 +22,7 @@ #include <linux/uaccess.h> #include "internal.h" -static int ramfs_nommu_setattr(struct mnt_idmap *, struct dentry *, struct iattr *); +static int ramfs_nommu_setattr(const struct mnt_idmap *, struct dentry *, struct iattr *); static unsigned long ramfs_nommu_get_unmapped_area(struct file *file, unsigned long addr, unsigned long len, @@ -161,7 +161,7 @@ static int ramfs_nommu_resize(struct inode *inode, loff_t newsize, loff_t size) * handle a change of attributes * - we're specifically interested in a change of size */ -static int ramfs_nommu_setattr(struct mnt_idmap *idmap, +static int ramfs_nommu_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *ia) { struct inode *inode = d_inode(dentry); diff --git a/fs/ramfs/inode.c b/fs/ramfs/inode.c index 0a88ede48e0a..ef91b933d4e6 100644 --- a/fs/ramfs/inode.c +++ b/fs/ramfs/inode.c @@ -95,7 +95,7 @@ struct inode *ramfs_get_inode(struct super_block *sb, */ /* SMP-safe */ static int -ramfs_mknod(struct mnt_idmap *idmap, struct inode *dir, +ramfs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t dev) { struct inode * inode = ramfs_get_inode(dir->i_sb, dir, mode, dev); @@ -118,7 +118,7 @@ out: return error; } -static struct dentry *ramfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ramfs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { int retval = ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode, 0); @@ -127,13 +127,13 @@ static struct dentry *ramfs_mkdir(struct mnt_idmap *idmap, struct inode *dir, return ERR_PTR(retval); } -static int ramfs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ramfs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { return ramfs_mknod(&nop_mnt_idmap, dir, dentry, mode | S_IFREG, 0); } -static int ramfs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ramfs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -163,7 +163,7 @@ out: return error; } -static int ramfs_tmpfile(struct mnt_idmap *idmap, +static int ramfs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode; diff --git a/fs/read_write.c b/fs/read_write.c index e8c14e2760b2..4da846a2bf17 100644 --- a/fs/read_write.c +++ b/fs/read_write.c @@ -274,7 +274,7 @@ loff_t fixed_size_llseek(struct file *file, loff_t offset, int whence, loff_t si EXPORT_SYMBOL(fixed_size_llseek); /** - * no_seek_end_llseek - llseek implementation for fixed-sized devices + * no_seek_end_llseek - llseek implementation for files without SEEK_END * @file: file structure to seek on * @offset: file offset to seek to * @whence: type of seek @@ -293,7 +293,7 @@ loff_t no_seek_end_llseek(struct file *file, loff_t offset, int whence) EXPORT_SYMBOL(no_seek_end_llseek); /** - * no_seek_end_llseek_size - llseek implementation for fixed-sized devices + * no_seek_end_llseek_size - llseek implementation for files without SEEK_END * @file: file structure to seek on * @offset: file offset to seek to * @whence: type of seek @@ -1761,8 +1761,8 @@ EXPORT_SYMBOL(generic_write_checks_count); * Performs necessary checks before doing a write * * Can adjust writing position or amount of bytes to write. - * Returns appropriate error code that caller should return or - * zero in case that write should be allowed. + * Returns the number of bytes that may be written on success (which + * may be less than requested if truncated), or a negative error code. */ ssize_t generic_write_checks(struct kiocb *iocb, struct iov_iter *from) { diff --git a/fs/remap_range.c b/fs/remap_range.c index 26afbbbfb10c..6eb7d845de5d 100644 --- a/fs/remap_range.c +++ b/fs/remap_range.c @@ -415,7 +415,7 @@ EXPORT_SYMBOL(vfs_clone_file_range); /* Check whether we are allowed to dedupe the destination file */ static bool may_dedupe_file(struct file *file) { - struct mnt_idmap *idmap = file_mnt_idmap(file); + const struct mnt_idmap *idmap = file_mnt_idmap(file); struct inode *inode = file_inode(file); if (capable(CAP_SYS_ADMIN)) diff --git a/fs/smb/client/cifsacl.c b/fs/smb/client/cifsacl.c index c5e47a835f99..d1a92bb4d9a5 100644 --- a/fs/smb/client/cifsacl.c +++ b/fs/smb/client/cifsacl.c @@ -1904,7 +1904,7 @@ id_mode_to_cifs_acl_exit: return rc; } -struct posix_acl *cifs_get_acl(struct mnt_idmap *idmap, +struct posix_acl *cifs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type) { #if defined(CONFIG_CIFS_ALLOW_INSECURE_LEGACY) && defined(CONFIG_CIFS_POSIX) @@ -1968,7 +1968,7 @@ out: #endif } -int cifs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int cifs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { #if defined(CONFIG_CIFS_ALLOW_INSECURE_LEGACY) && defined(CONFIG_CIFS_POSIX) diff --git a/fs/smb/client/cifsfs.c b/fs/smb/client/cifsfs.c index 7ecd70efdfea..b1ecbcfb154e 100644 --- a/fs/smb/client/cifsfs.c +++ b/fs/smb/client/cifsfs.c @@ -402,7 +402,7 @@ out_unlock: return rc; } -static int cifs_permission(struct mnt_idmap *idmap, +static int cifs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { unsigned int sbflags = cifs_sb_flags(CIFS_SB(inode)); diff --git a/fs/smb/client/cifsfs.h b/fs/smb/client/cifsfs.h index 0c85daa8386e..51692e14c4dd 100644 --- a/fs/smb/client/cifsfs.h +++ b/fs/smb/client/cifsfs.h @@ -53,23 +53,23 @@ void cifs_sb_deactive(struct super_block *sb); /* Functions related to inodes */ extern const struct inode_operations cifs_dir_inode_ops; struct inode *cifs_root_iget(struct super_block *sb); -int cifs_create(struct mnt_idmap *idmap, struct inode *dir, +int cifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *direntry, umode_t mode); int cifs_atomic_open(struct inode *dir, struct dentry *direntry, struct file *file, unsigned int oflags, umode_t mode); -int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int cifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode); struct dentry *cifs_lookup(struct inode *parent_dir_inode, struct dentry *direntry, unsigned int flags); int cifs_unlink(struct inode *dir, struct dentry *dentry); int cifs_hardlink(struct dentry *old_file, struct inode *inode, struct dentry *direntry); -int cifs_mknod(struct mnt_idmap *idmap, struct inode *inode, +int cifs_mknod(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode, dev_t device_number); -struct dentry *cifs_mkdir(struct mnt_idmap *idmap, struct inode *inode, +struct dentry *cifs_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode); int cifs_rmdir(struct inode *inode, struct dentry *direntry); -int cifs_rename2(struct mnt_idmap *idmap, struct inode *source_dir, +int cifs_rename2(const struct mnt_idmap *idmap, struct inode *source_dir, struct dentry *source_dentry, struct inode *target_dir, struct dentry *target_dentry, unsigned int flags); int cifs_revalidate_file_attr(struct file *filp); @@ -78,9 +78,9 @@ int cifs_revalidate_file(struct file *filp); int cifs_revalidate_dentry(struct dentry *dentry); int cifs_revalidate_mapping(struct inode *inode); int cifs_zap_mapping(struct inode *inode); -int cifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int cifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); -int cifs_setattr(struct mnt_idmap *idmap, struct dentry *direntry, +int cifs_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs); int cifs_fiemap(struct inode *inode, struct fiemap_extent_info *fei, u64 start, u64 len); @@ -129,7 +129,7 @@ struct vfsmount *cifs_d_automount(struct path *path); /* Functions related to symlinks */ const char *cifs_get_link(struct dentry *dentry, struct inode *inode, struct delayed_call *done); -int cifs_symlink(struct mnt_idmap *idmap, struct inode *inode, +int cifs_symlink(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, const char *symname); #ifdef CONFIG_CIFS_XATTR diff --git a/fs/smb/client/cifsproto.h b/fs/smb/client/cifsproto.h index 00168839c123..e6beff8aafe0 100644 --- a/fs/smb/client/cifsproto.h +++ b/fs/smb/client/cifsproto.h @@ -212,9 +212,9 @@ struct smb_ntsd *get_cifs_acl(struct cifs_sb_info *cifs_sb, struct smb_ntsd *get_cifs_acl_by_fid(struct cifs_sb_info *cifs_sb, const struct cifs_fid *cifsfid, u32 *pacllen, u32 info); -struct posix_acl *cifs_get_acl(struct mnt_idmap *idmap, struct dentry *dentry, +struct posix_acl *cifs_get_acl(const struct mnt_idmap *idmap, struct dentry *dentry, int type); -int cifs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +int cifs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); int set_cifs_acl(struct smb_ntsd *pnntsd, __u32 acllen, struct inode *inode, const char *path, int aclflag); diff --git a/fs/smb/client/dir.c b/fs/smb/client/dir.c index 1a56fa4d0e89..a2cf3d35ca5d 100644 --- a/fs/smb/client/dir.c +++ b/fs/smb/client/dir.c @@ -664,7 +664,7 @@ out_free_xid: * The initial dentry state is hashed-negative. On success, dentry will become * hashed-positive by calling d_instantiate(). */ -int cifs_create(struct mnt_idmap *idmap, struct inode *dir, +int cifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *direntry, umode_t mode) { struct cifs_sb_info *cifs_sb = CIFS_SB(dir); @@ -721,7 +721,7 @@ out_free_xid: return rc; } -int cifs_mknod(struct mnt_idmap *idmap, struct inode *inode, +int cifs_mknod(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode, dev_t device_number) { int rc = -EPERM; @@ -1084,7 +1084,7 @@ static int set_tmpfile_attr(const unsigned int xid, unsigned int oflags, * The initial dentry state is unhashed-negative. On success, dentry will * become unhashed-positive by calling d_instantiate(). */ -int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +int cifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct dentry *dentry = file->f_path.dentry; diff --git a/fs/smb/client/inode.c b/fs/smb/client/inode.c index 1fe0ef0a95db..4d7a87c7b73a 100644 --- a/fs/smb/client/inode.c +++ b/fs/smb/client/inode.c @@ -2281,7 +2281,7 @@ posix_mkdir_get_info: } #endif /* CONFIG_CIFS_ALLOW_INSECURE_LEGACY */ -struct dentry *cifs_mkdir(struct mnt_idmap *idmap, struct inode *inode, +struct dentry *cifs_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, umode_t mode) { int rc = 0; @@ -2531,7 +2531,7 @@ do_rename_exit: } int -cifs_rename2(struct mnt_idmap *idmap, struct inode *source_dir, +cifs_rename2(const struct mnt_idmap *idmap, struct inode *source_dir, struct dentry *source_dentry, struct inode *target_dir, struct dentry *target_dentry, unsigned int flags) { @@ -2937,7 +2937,7 @@ int cifs_revalidate_dentry(struct dentry *dentry) return cifs_revalidate_mapping(inode); } -int cifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int cifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { struct cifs_sb_info *cifs_sb = CIFS_SB(path->dentry); @@ -3554,7 +3554,7 @@ cifs_setattr_exit: } int -cifs_setattr(struct mnt_idmap *idmap, struct dentry *direntry, +cifs_setattr(const struct mnt_idmap *idmap, struct dentry *direntry, struct iattr *attrs) { struct cifs_sb_info *cifs_sb = CIFS_SB(direntry->d_sb); diff --git a/fs/smb/client/link.c b/fs/smb/client/link.c index 8d5d6aca742a..76df31abeaea 100644 --- a/fs/smb/client/link.c +++ b/fs/smb/client/link.c @@ -533,7 +533,7 @@ cifs_hl_exit: } int -cifs_symlink(struct mnt_idmap *idmap, struct inode *inode, +cifs_symlink(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *direntry, const char *symname) { struct cifs_sb_info *cifs_sb = CIFS_SB(inode); diff --git a/fs/smb/client/transport.c b/fs/smb/client/transport.c index 6e21b5f8754a..93ff6a4dbb35 100644 --- a/fs/smb/client/transport.c +++ b/fs/smb/client/transport.c @@ -22,7 +22,6 @@ #include <linux/mempool.h> #include <linux/sched/signal.h> #include <linux/task_io_accounting_ops.h> -#include <linux/task_work.h> #include "cifsglob.h" #include "cifsproto.h" #include "cifs_debug.h" @@ -171,15 +170,11 @@ smb_send_kvec(struct TCP_Server_Info *server, struct msghdr *smb_msg, * after the retries we will kill the socket and * reconnect which may clear the network problem. * - * Even if regular signals are masked, EINTR might be - * propagated from sk_stream_wait_memory() to here when - * TIF_NOTIFY_SIGNAL is used for task work. For example, - * certain io_uring completions will use that. Treat - * having EINTR with pending task work the same as EAGAIN - * to avoid unnecessary reconnects. + * Task work must not abort the send, see signal_pending(). */ - rc = sock_sendmsg(ssocket, smb_msg); - if (rc == -EAGAIN || unlikely(rc == -EINTR && task_work_pending(current))) { + scoped_guard(no_notify_signal) + rc = sock_sendmsg(ssocket, smb_msg); + if (rc == -EAGAIN) { retries++; if (retries >= 14 || (!server->noblocksnd && (retries > 2))) { diff --git a/fs/smb/client/xattr.c b/fs/smb/client/xattr.c index 5091f6c0d7fe..f6c9343016f7 100644 --- a/fs/smb/client/xattr.c +++ b/fs/smb/client/xattr.c @@ -91,7 +91,7 @@ static int cifs_creation_time_set(unsigned int xid, struct cifs_tcon *pTcon, } static int cifs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/smb/server/ndr.c b/fs/smb/server/ndr.c index 58d71560f626..7e546c22e284 100644 --- a/fs/smb/server/ndr.c +++ b/fs/smb/server/ndr.c @@ -338,7 +338,7 @@ static int ndr_encode_posix_acl_entry(struct ndr *n, struct xattr_smb_acl *acl) } int ndr_encode_posix_acl(struct ndr *n, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode, struct xattr_smb_acl *acl, struct xattr_smb_acl *def_acl) diff --git a/fs/smb/server/ndr.h b/fs/smb/server/ndr.h index f3c108c8cf4d..646568c42e4d 100644 --- a/fs/smb/server/ndr.h +++ b/fs/smb/server/ndr.h @@ -14,7 +14,7 @@ struct ndr { int ndr_encode_dos_attr(struct ndr *n, struct xattr_dos_attrib *da); int ndr_decode_dos_attr(struct ndr *n, struct xattr_dos_attrib *da); -int ndr_encode_posix_acl(struct ndr *n, struct mnt_idmap *idmap, +int ndr_encode_posix_acl(struct ndr *n, const struct mnt_idmap *idmap, struct inode *inode, struct xattr_smb_acl *acl, struct xattr_smb_acl *def_acl); int ndr_encode_v4_ntacl(struct ndr *n, struct xattr_ntacl *acl); diff --git a/fs/smb/server/oplock.c b/fs/smb/server/oplock.c index 1b8c3482d1e4..d0f18ebf471b 100644 --- a/fs/smb/server/oplock.c +++ b/fs/smb/server/oplock.c @@ -2270,7 +2270,7 @@ void create_posix_rsp_buf(char *cc, struct ksmbd_file *fp) { struct create_posix_rsp *buf; struct inode *inode = file_inode(fp->filp); - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c index 2f19ec9afa50..45dce9c30b6b 100644 --- a/fs/smb/server/smb2pdu.c +++ b/fs/smb/server/smb2pdu.c @@ -3297,7 +3297,7 @@ static bool smb2_is_private_ea(const char *name, size_t name_len) static int smb2_set_ea(struct smb2_ea_info *eabuf, unsigned int buf_len, const struct path *path, bool get_write) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); char *attr_name = NULL, *value; int rc = 0; unsigned int next = 0; @@ -3398,7 +3398,7 @@ static noinline int smb2_set_stream_name_xattr(const struct path *path, struct ksmbd_file *fp, char *stream_name, int s_type) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); size_t xattr_stream_size; char *xattr_stream_name; int rc; @@ -3475,7 +3475,7 @@ static loff_t ksmbd_stream_eof(struct ksmbd_file *fp) static int smb2_remove_smb_xattrs(const struct path *path) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); char *name, *xattr_list = NULL; ssize_t xattr_list_len; int err = 0; @@ -3668,7 +3668,7 @@ static int smb2_create_sd_buffer(struct ksmbd_work *work, } static int ksmbd_acls_fattr(struct smb_fattr *fattr, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *inode) { struct posix_acl *acl; @@ -4171,7 +4171,7 @@ int smb2_open(struct ksmbd_work *work) struct ksmbd_share_config *share = tcon->share_conf; struct ksmbd_file *fp = NULL; struct file *filp = NULL; - struct mnt_idmap *idmap = NULL; + const struct mnt_idmap *idmap = NULL; struct kstat stat; struct create_context *context; struct lease_ctx_info *lc = NULL; @@ -5839,7 +5839,7 @@ struct smb2_query_dir_private { static int process_query_dir_entries(struct smb2_query_dir_private *priv) { - struct mnt_idmap *idmap = file_mnt_idmap(priv->dir_fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(priv->dir_fp->filp); struct kstat kstat; struct ksmbd_kstat ksmbd_kstat; int rc; @@ -6432,7 +6432,7 @@ static int smb2_get_ea(struct ksmbd_work *work, struct ksmbd_file *fp, ssize_t buf_free_len, alignment_bytes, next_offset, rsp_data_cnt = 0; struct smb2_ea_info_req *ea_req = NULL; const struct path *path; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); if (!(fp->daccess & FILE_READ_EA_LE)) { pr_err("Not permitted to read ext attr : 0x%x\n", @@ -7141,7 +7141,7 @@ static int find_file_posix_info(struct smb2_query_info_rsp *rsp, { struct smb311_posix_qinfo *file_info; struct inode *inode = file_inode(fp->filp); - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); vfsgid_t vfsgid = i_gid_into_vfsgid(idmap, inode); struct kstat stat; @@ -7633,7 +7633,7 @@ static int smb2_get_info_sec(struct ksmbd_work *work, struct smb2_query_info_rsp *rsp) { struct ksmbd_file *fp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct smb_ntsd *pntsd = NULL, *ppntsd = NULL; struct smb_fattr fattr = {{0}}; struct inode *inode; @@ -8175,7 +8175,7 @@ static int set_file_basic_info(struct ksmbd_file *fp, struct iattr attrs; struct file *filp; struct inode *inode; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; __le32 attrs_mask = FILE_ATTRIBUTE_DIRECTORY_LE | FILE_ATTRIBUTE_COMPRESSED_LE; int rc = 0; @@ -10793,7 +10793,7 @@ static inline int fsctl_set_sparse(struct ksmbd_work *work, u64 id, struct file_sparse *sparse) { struct ksmbd_file *fp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; int ret = 0; __le32 old_fattr; diff --git a/fs/smb/server/smb_common.c b/fs/smb/server/smb_common.c index 086a1b85e5f4..7dbcfa658edd 100644 --- a/fs/smb/server/smb_common.c +++ b/fs/smb/server/smb_common.c @@ -467,7 +467,7 @@ int ksmbd_populate_dot_dotdot_entries(struct ksmbd_work *work, int info_level, { int i, rc = 0; struct ksmbd_conn *conn = work->conn; - struct mnt_idmap *idmap = file_mnt_idmap(dir->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(dir->filp); for (i = 0; i < 2; i++) { struct kstat kstat; diff --git a/fs/smb/server/smbacl.c b/fs/smb/server/smbacl.c index e75247915c87..96428df33b43 100644 --- a/fs/smb/server/smbacl.c +++ b/fs/smb/server/smbacl.c @@ -258,7 +258,7 @@ void id_to_sid(unsigned int cid, uint sidtype, struct smb_sid *ssid) ssid->num_subauth++; } -static int sid_to_id(struct mnt_idmap *idmap, +static int sid_to_id(const struct mnt_idmap *idmap, struct smb_sid *psid, uint sidtype, struct smb_fattr *fattr) { @@ -384,7 +384,7 @@ void free_acl_state(struct posix_acl_state *state) kfree(state->groups); } -static int parse_dacl(struct mnt_idmap *idmap, +static int parse_dacl(const struct mnt_idmap *idmap, struct smb_acl *pdacl, char *end_of_acl, struct smb_sid *pownersid, struct smb_sid *pgrpsid, struct smb_fattr *fattr) @@ -620,7 +620,7 @@ out: return ret; } -static void set_posix_acl_entries_dacl(struct mnt_idmap *idmap, +static void set_posix_acl_entries_dacl(const struct mnt_idmap *idmap, struct smb_ace *pndace, struct smb_fattr *fattr, u16 *num_aces, u16 *size, u16 existing_nt_aces, @@ -751,7 +751,7 @@ posix_default_acl: } } -static void set_ntacl_dacl(struct mnt_idmap *idmap, +static void set_ntacl_dacl(const struct mnt_idmap *idmap, struct smb_acl *pndacl, struct smb_acl *nt_dacl, unsigned int aces_size, @@ -810,7 +810,7 @@ next_ace: pndacl->size = cpu_to_le16(le16_to_cpu(pndacl->size) + size); } -static void set_mode_dacl(struct mnt_idmap *idmap, +static void set_mode_dacl(const struct mnt_idmap *idmap, struct smb_acl *pndacl, struct smb_fattr *fattr) { struct smb_ace *pace, *pndace; @@ -896,7 +896,7 @@ static int parse_sid(struct smb_sid *psid, char *end_of_acl) } /* Convert CIFS ACL to POSIX form */ -int parse_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int parse_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, int acl_len, struct smb_fattr *fattr) { int rc = 0; @@ -1031,7 +1031,7 @@ size_t smb_acl_sec_desc_scratch_len(struct smb_fattr *fattr, } /* Convert permission bits from mode to equivalent CIFS ACL */ -int build_sec_desc(struct mnt_idmap *idmap, +int build_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info, __u32 *secdesclen, struct smb_fattr *fattr) @@ -1200,7 +1200,7 @@ int smb_inherit_dacl(struct ksmbd_conn *conn, struct smb_ntsd *parent_pntsd = NULL; struct smb_sid owner_sid, group_sid; struct dentry *parent = path->dentry->d_parent; - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); int inherited_flags = 0, flags = 0, i, nt_size = 0, pdacl_size; int rc = 0, pntsd_type, ppntsd_size, acl_len, aces_size; unsigned int dacloffset; @@ -1455,7 +1455,7 @@ int smb_check_perm_dacl(struct ksmbd_conn *conn, const struct path *path, __le32 *pdaccess, __le32 raw_daccess, int uid, bool strict) { - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); struct smb_ntsd *pntsd = NULL; struct smb_acl *pdacl; struct posix_acl *posix_acls; @@ -1678,7 +1678,7 @@ int set_info_sec(struct ksmbd_conn *conn, struct ksmbd_tree_connect *tcon, int rc; struct smb_fattr fattr = {{0}}; struct inode *inode = d_inode(path->dentry); - struct mnt_idmap *idmap = mnt_idmap(path->mnt); + const struct mnt_idmap *idmap = mnt_idmap(path->mnt); struct iattr newattrs; fattr.cf_uid = INVALID_UID; diff --git a/fs/smb/server/smbacl.h b/fs/smb/server/smbacl.h index 01810c16cc04..28d215807faa 100644 --- a/fs/smb/server/smbacl.h +++ b/fs/smb/server/smbacl.h @@ -81,9 +81,9 @@ struct posix_acl_state { struct posix_ace_state_array *groups; }; -int parse_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int parse_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, int acl_len, struct smb_fattr *fattr); -int build_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd, +int build_sec_desc(const struct mnt_idmap *idmap, struct smb_ntsd *pntsd, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info, __u32 *secdesclen, struct smb_fattr *fattr); int init_acl_state(struct posix_acl_state *state, u16 cnt); @@ -105,7 +105,7 @@ void ksmbd_init_domain(u32 *sub_auth); size_t smb_acl_sec_desc_scratch_len(struct smb_fattr *fattr, struct smb_ntsd *ppntsd, int ppntsd_size, int addition_info); -static inline uid_t posix_acl_uid_translate(struct mnt_idmap *idmap, +static inline uid_t posix_acl_uid_translate(const struct mnt_idmap *idmap, struct posix_acl_entry *pace) { vfsuid_t vfsuid; @@ -117,7 +117,7 @@ static inline uid_t posix_acl_uid_translate(struct mnt_idmap *idmap, return from_kuid(&init_user_ns, vfsuid_into_kuid(vfsuid)); } -static inline gid_t posix_acl_gid_translate(struct mnt_idmap *idmap, +static inline gid_t posix_acl_gid_translate(const struct mnt_idmap *idmap, struct posix_acl_entry *pace) { vfsgid_t vfsgid; diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c index db0f2de2bab3..2e2c554bc2c1 100644 --- a/fs/smb/server/vfs.c +++ b/fs/smb/server/vfs.c @@ -116,7 +116,7 @@ static int ksmbd_vfs_path_lookup(struct ksmbd_share_config *share_conf, return 0; } -void ksmbd_vfs_query_maximal_access(struct mnt_idmap *idmap, +void ksmbd_vfs_query_maximal_access(const struct mnt_idmap *idmap, struct dentry *dentry, __le32 *daccess) { *daccess = cpu_to_le32(FILE_READ_ATTRIBUTES | READ_CONTROL); @@ -184,7 +184,7 @@ int ksmbd_vfs_create(struct ksmbd_work *work, const char *name, umode_t mode) */ int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct path path; struct dentry *dentry, *d; int err = 0; @@ -217,7 +217,7 @@ int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode) return err; } -ssize_t ksmbd_vfs_getcasexattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getcasexattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len, char **attr_value) { @@ -387,7 +387,7 @@ static int ksmbd_vfs_stream_write(struct ksmbd_file *fp, char *buf, loff_t *pos, { const struct cred *saved_cred; char *stream_buf = NULL, *wbuf; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); size_t size; ssize_t v_len; int err = 0; @@ -578,7 +578,7 @@ int ksmbd_vfs_fsync(struct ksmbd_work *work, u64 fid, u64 p_id) */ int ksmbd_vfs_remove_file(struct ksmbd_work *work, const struct path *path) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct dentry *parent = path->dentry->d_parent; int err; @@ -842,7 +842,7 @@ ssize_t ksmbd_vfs_listxattr(struct dentry *dentry, char **list) return size; } -ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_xattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name) { return vfs_getxattr(idmap, dentry, xattr_name, NULL, 0); @@ -857,7 +857,7 @@ ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, * * Return: read xattr value length on success, otherwise error */ -ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name, char **xattr_buf) { @@ -894,7 +894,7 @@ ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, * * Return: 0 on success, otherwise error */ -int ksmbd_vfs_setxattr(struct mnt_idmap *idmap, +int ksmbd_vfs_setxattr(const struct mnt_idmap *idmap, const struct path *path, const char *attr_name, void *attr_value, size_t attr_size, int flags, bool get_write) @@ -1178,7 +1178,7 @@ int ksmbd_vfs_query_allocated_ranges(struct ksmbd_file *fp, loff_t start, return ret; } -int ksmbd_vfs_remove_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_xattr(const struct mnt_idmap *idmap, const struct path *path, char *attr_name, bool get_write) { @@ -1203,7 +1203,7 @@ int ksmbd_vfs_unlink(struct file *filp) const struct cred *saved_cred; int err = 0; struct dentry *dir, *dentry = filp->f_path.dentry; - struct mnt_idmap *idmap = file_mnt_idmap(filp); + const struct mnt_idmap *idmap = file_mnt_idmap(filp); saved_cred = override_creds(filp->f_cred); err = mnt_want_write(filp->f_path.mnt); @@ -1472,7 +1472,7 @@ struct dentry *ksmbd_vfs_kern_path_create(struct ksmbd_work *work, return dent; } -int ksmbd_vfs_remove_acl_xattrs(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_acl_xattrs(const struct mnt_idmap *idmap, const struct path *path) { char *name, *xattr_list = NULL; @@ -1512,7 +1512,7 @@ out: return err; } -int ksmbd_vfs_remove_sd_xattrs(struct mnt_idmap *idmap, const struct path *path) +int ksmbd_vfs_remove_sd_xattrs(const struct mnt_idmap *idmap, const struct path *path) { char *name, *xattr_list = NULL; ssize_t xattr_list_len; @@ -1541,7 +1541,7 @@ out: return err; } -static struct xattr_smb_acl *ksmbd_vfs_make_xattr_posix_acl(struct mnt_idmap *idmap, +static struct xattr_smb_acl *ksmbd_vfs_make_xattr_posix_acl(const struct mnt_idmap *idmap, struct inode *inode, int acl_type) { @@ -1607,7 +1607,7 @@ out: } int ksmbd_vfs_set_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct smb_ntsd *pntsd, int len, bool get_write) @@ -1675,7 +1675,7 @@ out: EXPORT_SYMBOL_IF_KUNIT(ksmbd_vfs_set_sd_xattr); int ksmbd_vfs_get_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct smb_ntsd **pntsd) { @@ -1744,7 +1744,7 @@ out_free: return rc; } -int ksmbd_vfs_set_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_set_dos_attrib_xattr(const struct mnt_idmap *idmap, const struct path *path, struct xattr_dos_attrib *da, bool get_write) @@ -1766,7 +1766,7 @@ out: return err; } -int ksmbd_vfs_get_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_get_dos_attrib_xattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct xattr_dos_attrib *da) { @@ -1822,7 +1822,7 @@ void *ksmbd_vfs_init_kstat(char **p, struct ksmbd_kstat *ksmbd_kstat) } int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct ksmbd_kstat *ksmbd_kstat) { @@ -1897,7 +1897,7 @@ int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, return 0; } -ssize_t ksmbd_vfs_casexattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_casexattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len) { @@ -2212,7 +2212,7 @@ void ksmbd_vfs_posix_lock_unblock(struct file_lock *flock) locks_delete_block(flock); } -int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_set_init_posix_acl(const struct mnt_idmap *idmap, const struct path *path) { struct posix_acl_state acl_state; @@ -2265,7 +2265,7 @@ int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, return rc; } -int ksmbd_vfs_inherit_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_inherit_posix_acl(const struct mnt_idmap *idmap, const struct path *path, struct inode *parent_inode) { struct posix_acl *acls; @@ -2328,7 +2328,7 @@ static int __ksmbd_vfs_set_compression(struct ksmbd_work *work, const struct cred *saved_cred = NULL; struct file_kattr fa; struct dentry *dentry = fp->filp->f_path.dentry; - struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); + const struct mnt_idmap *idmap = file_mnt_idmap(fp->filp); u32 flags; __le32 old_fattr; int rc; diff --git a/fs/smb/server/vfs.h b/fs/smb/server/vfs.h index 566c670c90be..216c76291fc1 100644 --- a/fs/smb/server/vfs.h +++ b/fs/smb/server/vfs.h @@ -74,7 +74,7 @@ struct ksmbd_kstat { }; int ksmbd_vfs_lock_parent(struct dentry *parent, struct dentry *child); -void ksmbd_vfs_query_maximal_access(struct mnt_idmap *idmap, +void ksmbd_vfs_query_maximal_access(const struct mnt_idmap *idmap, struct dentry *dentry, __le32 *daccess); int ksmbd_vfs_create(struct ksmbd_work *work, const char *name, umode_t mode); int ksmbd_vfs_mkdir(struct ksmbd_work *work, const char *name, umode_t mode); @@ -104,25 +104,25 @@ int ksmbd_vfs_copy_file_ranges(struct ksmbd_work *work, unsigned int *chunk_size_written, loff_t *total_size_written); ssize_t ksmbd_vfs_listxattr(struct dentry *dentry, char **list); -ssize_t ksmbd_vfs_getxattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name, char **xattr_buf); -ssize_t ksmbd_vfs_xattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_xattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *xattr_name); -ssize_t ksmbd_vfs_getcasexattr(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_getcasexattr(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len, char **attr_value); -ssize_t ksmbd_vfs_casexattr_len(struct mnt_idmap *idmap, +ssize_t ksmbd_vfs_casexattr_len(const struct mnt_idmap *idmap, struct dentry *dentry, char *attr_name, int attr_name_len); -int ksmbd_vfs_setxattr(struct mnt_idmap *idmap, +int ksmbd_vfs_setxattr(const struct mnt_idmap *idmap, const struct path *path, const char *attr_name, void *attr_value, size_t attr_size, int flags, bool get_write); int ksmbd_vfs_xattr_stream_name(char *stream_name, char **xattr_stream_name, size_t *xattr_stream_name_size, int s_type); -int ksmbd_vfs_remove_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_xattr(const struct mnt_idmap *idmap, const struct path *path, char *attr_name, bool get_write); int ksmbd_vfs_kern_path(struct ksmbd_work *work, char *name, @@ -152,33 +152,33 @@ int ksmbd_vfs_query_allocated_ranges(struct ksmbd_file *fp, loff_t start, int ksmbd_vfs_unlink(struct file *filp); void *ksmbd_vfs_init_kstat(char **p, struct ksmbd_kstat *ksmbd_kstat); int ksmbd_vfs_fill_dentry_attrs(struct ksmbd_work *work, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct ksmbd_kstat *ksmbd_kstat); void ksmbd_vfs_posix_lock_wait(struct file_lock *flock); void ksmbd_vfs_posix_lock_unblock(struct file_lock *flock); -int ksmbd_vfs_remove_acl_xattrs(struct mnt_idmap *idmap, +int ksmbd_vfs_remove_acl_xattrs(const struct mnt_idmap *idmap, const struct path *path); -int ksmbd_vfs_remove_sd_xattrs(struct mnt_idmap *idmap, const struct path *path); +int ksmbd_vfs_remove_sd_xattrs(const struct mnt_idmap *idmap, const struct path *path); int ksmbd_vfs_set_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct smb_ntsd *pntsd, int len, bool get_write); int ksmbd_vfs_get_sd_xattr(struct ksmbd_conn *conn, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct smb_ntsd **pntsd); -int ksmbd_vfs_set_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_set_dos_attrib_xattr(const struct mnt_idmap *idmap, const struct path *path, struct xattr_dos_attrib *da, bool get_write); -int ksmbd_vfs_get_dos_attrib_xattr(struct mnt_idmap *idmap, +int ksmbd_vfs_get_dos_attrib_xattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct xattr_dos_attrib *da); -int ksmbd_vfs_set_init_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_set_init_posix_acl(const struct mnt_idmap *idmap, const struct path *path); -int ksmbd_vfs_inherit_posix_acl(struct mnt_idmap *idmap, +int ksmbd_vfs_inherit_posix_acl(const struct mnt_idmap *idmap, const struct path *path, struct inode *parent_inode); void ksmbd_vfs_update_compressed_fattr(struct dentry *dentry, __le32 *fattr); diff --git a/fs/splice.c b/fs/splice.c index 9d8f63e2fd1a..bc243ab8dc43 100644 --- a/fs/splice.c +++ b/fs/splice.c @@ -177,9 +177,9 @@ static const struct pipe_buf_operations user_page_pipe_buf_ops = { static void wakeup_pipe_readers(struct pipe_inode_info *pipe) { - smp_mb(); - if (waitqueue_active(&pipe->rd_wait)) - wake_up_interruptible(&pipe->rd_wait); + if (wq_has_sleeper(&pipe->rd_wait)) + wake_up_interruptible_poll(&pipe->rd_wait, + EPOLLIN | EPOLLRDNORM); kill_fasync(&pipe->fasync_readers, SIGIO, POLL_IN); } @@ -413,9 +413,9 @@ EXPORT_SYMBOL(nosteal_pipe_buf_ops); static void wakeup_pipe_writers(struct pipe_inode_info *pipe) { - smp_mb(); - if (waitqueue_active(&pipe->wr_wait)) - wake_up_interruptible(&pipe->wr_wait); + if (wq_has_sleeper(&pipe->wr_wait)) + wake_up_interruptible_poll(&pipe->wr_wait, + EPOLLOUT | EPOLLWRNORM); kill_fasync(&pipe->fasync_writers, SIGIO, POLL_OUT); } @@ -1009,21 +1009,14 @@ ssize_t vfs_splice_read(struct file *in, loff_t *ppos, } EXPORT_SYMBOL_GPL(vfs_splice_read); -/** - * splice_direct_to_actor - splices data directly between two non-pipes - * @in: file to splice from - * @sd: actor information on where to splice to - * @actor: handles the data splicing - * - * Description: - * This is a special case helper to splice directly between two - * points, without requiring an explicit pipe. Internally an allocated - * pipe is cached in the process, and reused during the lifetime of - * that process. - * +/* + * This is a special case helper to splice directly between two + * points, without requiring an explicit pipe. Internally an allocated + * pipe is cached in the process, and reused during the lifetime of + * that process. */ -ssize_t splice_direct_to_actor(struct file *in, struct splice_desc *sd, - splice_direct_actor *actor) +static ssize_t splice_direct_to_actor(struct file *in, struct splice_desc *sd, + splice_direct_actor *actor) { struct pipe_inode_info *pipe; ssize_t ret, bytes; @@ -1147,7 +1140,42 @@ out_release: goto done; } -EXPORT_SYMBOL(splice_direct_to_actor); + +/** + * vfs_splice_to_actor - call an actor on data read from a file + * @in: file to read from + * @pos: file offset + * @count: maximum number of bytes to read + * @actor: callback to process a pipe's worth of data + * @private: private data passed to @actor + * + * Read up to @count worth of data from @in at @pos, and call @actor + * when the hidden pipe used to buffer the data is full. Ensures the + * read is allowed using rw_verify_area() and emits fsnotify access + * events. @in must be seekable (FMODE_LSEEK). + * + * Return: The number of bytes spliced, or a negative errno. + */ +ssize_t vfs_splice_to_actor(struct file *in, loff_t pos, size_t count, + splice_direct_actor *actor, void *private) +{ + struct splice_desc sd = { + .total_len = count, + .pos = pos, + .u.data = private, + }; + ssize_t ret; + + ret = rw_verify_area(READ, in, &sd.pos, sd.total_len); + if (ret < 0) + return ret; + + ret = splice_direct_to_actor(in, &sd, actor); + if (ret >= 0) + fsnotify_access(in); + return ret; +} +EXPORT_SYMBOL(vfs_splice_to_actor); static int direct_splice_actor(struct pipe_inode_info *pipe, struct splice_desc *sd) diff --git a/fs/stat.c b/fs/stat.c index c461c3054234..a9b7383d538d 100644 --- a/fs/stat.c +++ b/fs/stat.c @@ -79,7 +79,7 @@ EXPORT_SYMBOL(fill_mg_cmtime); * uid and gid filds. On non-idmapped mounts or if permission checking is to be * performed on the raw inode simply pass @nop_mnt_idmap. */ -void generic_fillattr(struct mnt_idmap *idmap, u32 request_mask, +void generic_fillattr(const struct mnt_idmap *idmap, u32 request_mask, struct inode *inode, struct kstat *stat) { vfsuid_t vfsuid = i_uid_into_vfsuid(idmap, inode); @@ -181,7 +181,7 @@ EXPORT_SYMBOL_GPL(generic_fill_statx_atomic_writes); int vfs_getattr_nosec(const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct inode *inode = d_backing_inode(path->dentry); memset(stat, 0, sizeof(*stat)); diff --git a/fs/super.c b/fs/super.c index 1d5ccf540a9b..b1d4add11b77 100644 --- a/fs/super.c +++ b/fs/super.c @@ -1374,7 +1374,17 @@ static int test_single_super(struct super_block *s, struct fs_context *fc) return 1; } -static int vfs_get_super(struct fs_context *fc, +/** + * get_tree_super - Get a superblock, optionally sharing an existing one + * @fc: The filesystem context holding the parameters + * @test: Comparison function to find a matching existing superblock, or NULL + * @fill_super: Helper to initialise a new superblock + * + * If @test is non-NULL and matches an existing superblock, that superblock is + * reused; otherwise a new anonymous superblock is created and initialised with + * @fill_super. Passing NULL for @test always creates a new superblock. + */ +int get_tree_super(struct fs_context *fc, int (*test)(struct super_block *, struct fs_context *), int (*fill_super)(struct super_block *sb, struct fs_context *fc)) @@ -1401,12 +1411,13 @@ error: deactivate_locked_super(sb); return err; } +EXPORT_SYMBOL(get_tree_super); int get_tree_nodev(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)) { - return vfs_get_super(fc, NULL, fill_super); + return get_tree_super(fc, NULL, fill_super); } EXPORT_SYMBOL(get_tree_nodev); @@ -1414,7 +1425,7 @@ int get_tree_single(struct fs_context *fc, int (*fill_super)(struct super_block *sb, struct fs_context *fc)) { - return vfs_get_super(fc, test_single_super, fill_super); + return get_tree_super(fc, test_single_super, fill_super); } EXPORT_SYMBOL(get_tree_single); @@ -1424,7 +1435,7 @@ int get_tree_keyed(struct fs_context *fc, void *key) { fc->s_fs_info = key; - return vfs_get_super(fc, test_keyed_super, fill_super); + return get_tree_super(fc, test_keyed_super, fill_super); } EXPORT_SYMBOL(get_tree_keyed); diff --git a/fs/tests/.kunitconfig b/fs/tests/.kunitconfig new file mode 100644 index 000000000000..de67125a9421 --- /dev/null +++ b/fs/tests/.kunitconfig @@ -0,0 +1,2 @@ +CONFIG_KUNIT=y +CONFIG_FDTABLE_KUNIT_TEST=y diff --git a/fs/tests/fdtable_kunit.c b/fs/tests/fdtable_kunit.c new file mode 100644 index 000000000000..c5b028264557 --- /dev/null +++ b/fs/tests/fdtable_kunit.c @@ -0,0 +1,72 @@ +// SPDX-License-Identifier: GPL-2.0-only +#include <kunit/test.h> +#include <linux/fdtable.h> +#include <linux/file.h> + +static void test_alloc_fdtable(struct kunit *test) +{ + struct fdtable *fdt; + unsigned int slots = 64; + + fdt = alloc_fdtable(slots); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt); + + /* Check that max_fds is set correctly and is >= slots */ + KUNIT_EXPECT_GE(test, fdt->max_fds, slots); + + /* Check that fd is allocated */ + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt->fd); + + /* + * Check dynamic object size of fdt->fd if compiler supports + * __counted_by_ptr. + */ +#ifdef CONFIG_CC_HAS_COUNTED_BY_PTR + KUNIT_EXPECT_EQ(test, __struct_size(fdt->fd), + fdt->max_fds * sizeof(struct file *)); +#endif + + __free_fdtable(fdt); +} + +static void test_dup_fd(struct kunit *test) +{ + struct files_struct *newf; + struct fdtable *fdt; + + newf = dup_fd(&init_files, NULL); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, newf); + + fdt = rcu_dereference_raw(newf->fdt); + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt); + + /* Check that max_fds is set correctly and is >= NR_OPEN_DEFAULT */ + KUNIT_EXPECT_GE(test, fdt->max_fds, NR_OPEN_DEFAULT); + + /* Check that fd is allocated */ + KUNIT_ASSERT_NOT_ERR_OR_NULL(test, fdt->fd); + + /* + * Check dynamic object size of fdt->fd if compiler supports + * __counted_by_ptr. + */ +#ifdef CONFIG_CC_HAS_COUNTED_BY_PTR + KUNIT_EXPECT_EQ(test, __struct_size(fdt->fd), + fdt->max_fds * sizeof(struct file *)); +#endif + + put_files_struct(newf); +} + +static struct kunit_case fdtable_test_cases[] = { + KUNIT_CASE(test_alloc_fdtable), + KUNIT_CASE(test_dup_fd), + {} +}; + +static struct kunit_suite fdtable_test_suite = { + .name = "fdtable", + .test_cases = fdtable_test_cases, +}; + +kunit_test_suite(fdtable_test_suite); diff --git a/fs/tracefs/event_inode.c b/fs/tracefs/event_inode.c index 6e3513b13cfa..6f6daac88621 100644 --- a/fs/tracefs/event_inode.c +++ b/fs/tracefs/event_inode.c @@ -182,7 +182,7 @@ static void update_attr(struct eventfs_attr *attr, struct iattr *iattr) } } -static int eventfs_set_attr(struct mnt_idmap *idmap, struct dentry *dentry, +static int eventfs_set_attr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { const struct eventfs_entry *entry; diff --git a/fs/tracefs/inode.c b/fs/tracefs/inode.c index f3d6188a3b7b..5020365ca704 100644 --- a/fs/tracefs/inode.c +++ b/fs/tracefs/inode.c @@ -94,7 +94,7 @@ static struct tracefs_dir_ops { int (*rmdir)(const char *name); } tracefs_ops __ro_after_init; -static struct dentry *tracefs_syscall_mkdir(struct mnt_idmap *idmap, +static struct dentry *tracefs_syscall_mkdir(const struct mnt_idmap *idmap, struct inode *inode, struct dentry *dentry, umode_t mode) { @@ -189,14 +189,14 @@ static void set_tracefs_inode_owner(struct inode *inode) inode->i_gid = gid; } -static int tracefs_permission(struct mnt_idmap *idmap, +static int tracefs_permission(const struct mnt_idmap *idmap, struct inode *inode, int mask) { set_tracefs_inode_owner(inode); return generic_permission(idmap, inode, mask); } -static int tracefs_getattr(struct mnt_idmap *idmap, +static int tracefs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { @@ -207,7 +207,7 @@ static int tracefs_getattr(struct mnt_idmap *idmap, return 0; } -static int tracefs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int tracefs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { unsigned int ia_valid = attr->ia_valid; diff --git a/fs/ubifs/dir.c b/fs/ubifs/dir.c index 23ec924162d6..c67954f6bee0 100644 --- a/fs/ubifs/dir.c +++ b/fs/ubifs/dir.c @@ -302,7 +302,7 @@ static int ubifs_prepare_create(struct inode *dir, struct dentry *dentry, return fscrypt_setup_filename(dir, &dentry->d_name, 0, nm); } -static int ubifs_create(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -440,7 +440,7 @@ static void unlock_2_inodes(struct inode *inode1, struct inode *inode2) mutex_unlock(&ubifs_inode(inode1)->ui_mutex); } -static int ubifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct dentry *dentry = file->f_path.dentry; @@ -1002,7 +1002,7 @@ out_fname: return err; } -static struct dentry *ubifs_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *ubifs_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -1077,7 +1077,7 @@ out_budg: return ERR_PTR(err); } -static int ubifs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -1170,7 +1170,7 @@ out_budg: return err; } -static int ubifs_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int ubifs_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -1642,7 +1642,7 @@ out: return err; } -static int ubifs_rename(struct mnt_idmap *idmap, +static int ubifs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) @@ -1667,7 +1667,7 @@ static int ubifs_rename(struct mnt_idmap *idmap, return do_rename(old_dir, old_dentry, new_dir, new_dentry, flags); } -int ubifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ubifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { loff_t size; diff --git a/fs/ubifs/file.c b/fs/ubifs/file.c index e73c28b12f97..244b835fc82a 100644 --- a/fs/ubifs/file.c +++ b/fs/ubifs/file.c @@ -1251,7 +1251,7 @@ static int do_setattr(struct ubifs_info *c, struct inode *inode, return err; } -int ubifs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ubifs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { int err; @@ -1611,7 +1611,7 @@ static const char *ubifs_get_link(struct dentry *dentry, return fscrypt_get_symlink(inode, ui->data, ui->data_len, done); } -static int ubifs_symlink_getattr(struct mnt_idmap *idmap, +static int ubifs_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int query_flags) { diff --git a/fs/ubifs/ioctl.c b/fs/ubifs/ioctl.c index 79536b2e3d7a..5c34f895bd4e 100644 --- a/fs/ubifs/ioctl.c +++ b/fs/ubifs/ioctl.c @@ -144,7 +144,7 @@ int ubifs_fileattr_get(struct dentry *dentry, struct file_kattr *fa) return 0; } -int ubifs_fileattr_set(struct mnt_idmap *idmap, +int ubifs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { struct inode *inode = d_inode(dentry); diff --git a/fs/ubifs/ubifs.h b/fs/ubifs/ubifs.h index 00db0d19a85e..b85b45a0564b 100644 --- a/fs/ubifs/ubifs.h +++ b/fs/ubifs/ubifs.h @@ -2020,7 +2020,7 @@ int ubifs_calc_dark(const struct ubifs_info *c, int spc); /* file.c */ int ubifs_fsync(struct file *file, loff_t start, loff_t end, int datasync); -int ubifs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ubifs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); int ubifs_update_time(struct inode *inode, enum fs_update_time type, unsigned int flags); @@ -2028,7 +2028,7 @@ int ubifs_update_time(struct inode *inode, enum fs_update_time type, /* dir.c */ struct inode *ubifs_new_inode(struct ubifs_info *c, struct inode *dir, umode_t mode, bool is_xattr); -int ubifs_getattr(struct mnt_idmap *idmap, const struct path *path, +int ubifs_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags); int ubifs_check_dir_empty(struct inode *dir); @@ -2083,7 +2083,7 @@ void ubifs_destroy_size_tree(struct ubifs_info *c); /* ioctl.c */ int ubifs_fileattr_get(struct dentry *dentry, struct file_kattr *fa); -int ubifs_fileattr_set(struct mnt_idmap *idmap, +int ubifs_fileattr_set(const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); long ubifs_ioctl(struct file *file, unsigned int cmd, unsigned long arg); void ubifs_set_inode_flags(struct inode *inode); diff --git a/fs/ubifs/xattr.c b/fs/ubifs/xattr.c index b5a9ab9d8a10..3b0e8a270f59 100644 --- a/fs/ubifs/xattr.c +++ b/fs/ubifs/xattr.c @@ -660,7 +660,7 @@ static int xattr_get(const struct xattr_handler *handler, } static int xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) diff --git a/fs/udf/file.c b/fs/udf/file.c index 57d11606a2a7..02e9314818dc 100644 --- a/fs/udf/file.c +++ b/fs/udf/file.c @@ -212,7 +212,7 @@ const struct file_operations udf_file_operations = { .setlease = generic_setlease, }; -static int udf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int udf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/udf/namei.c b/fs/udf/namei.c index b90841ac0a40..42d000fe9d6b 100644 --- a/fs/udf/namei.c +++ b/fs/udf/namei.c @@ -370,7 +370,7 @@ static int udf_add_nondir(struct dentry *dentry, struct inode *inode) return 0; } -static int udf_create(struct mnt_idmap *idmap, struct inode *dir, +static int udf_create(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode = udf_new_inode(dir, mode); @@ -386,7 +386,7 @@ static int udf_create(struct mnt_idmap *idmap, struct inode *dir, return udf_add_nondir(dentry, inode); } -static int udf_tmpfile(struct mnt_idmap *idmap, struct inode *dir, +static int udf_tmpfile(const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) { struct inode *inode = udf_new_inode(dir, mode); @@ -403,7 +403,7 @@ static int udf_tmpfile(struct mnt_idmap *idmap, struct inode *dir, return finish_open_simple(file, 0); } -static int udf_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int udf_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -419,7 +419,7 @@ static int udf_mknod(struct mnt_idmap *idmap, struct inode *dir, return udf_add_nondir(dentry, inode); } -static struct dentry *udf_mkdir(struct mnt_idmap *idmap, struct inode *dir, +static struct dentry *udf_mkdir(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) { struct inode *inode; @@ -567,7 +567,7 @@ out: return ret; } -static int udf_symlink(struct mnt_idmap *idmap, struct inode *dir, +static int udf_symlink(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) { struct inode *inode; @@ -762,7 +762,7 @@ static int udf_link(struct dentry *old_dentry, struct inode *dir, /* Anybody can rename anything with this: the permission checks are left to the * higher-level routines. */ -static int udf_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int udf_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/udf/symlink.c b/fs/udf/symlink.c index a05d1888a2ba..df41bf5a05b1 100644 --- a/fs/udf/symlink.c +++ b/fs/udf/symlink.c @@ -133,7 +133,7 @@ out: return err; } -static int udf_symlink_getattr(struct mnt_idmap *idmap, +static int udf_symlink_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, unsigned int flags) { diff --git a/fs/ufs/dir.c b/fs/ufs/dir.c index ce43cf20b07c..e96174b738b4 100644 --- a/fs/ufs/dir.c +++ b/fs/ufs/dir.c @@ -213,7 +213,7 @@ fail: static unsigned ufs_last_byte(struct inode *inode, unsigned long page_nr) { - unsigned last_byte = inode->i_size; + u64 last_byte = inode->i_size; last_byte -= page_nr << PAGE_SHIFT; if (last_byte > PAGE_SIZE) diff --git a/fs/ufs/inode.c b/fs/ufs/inode.c index 440d014cc5ed..c9ff8673fa66 100644 --- a/fs/ufs/inode.c +++ b/fs/ufs/inode.c @@ -1195,7 +1195,7 @@ out: return err; } -int ufs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int ufs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr) { struct inode *inode = d_inode(dentry); diff --git a/fs/ufs/namei.c b/fs/ufs/namei.c index 6703f3bcf76f..d45347a25741 100644 --- a/fs/ufs/namei.c +++ b/fs/ufs/namei.c @@ -69,7 +69,7 @@ static struct dentry *ufs_lookup(struct inode * dir, struct dentry *dentry, unsi * If the create succeeds, we fill in the inode information * with d_instantiate(). */ -static int ufs_create (struct mnt_idmap * idmap, +static int ufs_create (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { struct inode *inode; @@ -85,7 +85,7 @@ static int ufs_create (struct mnt_idmap * idmap, return ufs_add_nondir(dentry, inode); } -static int ufs_mknod(struct mnt_idmap *idmap, struct inode *dir, +static int ufs_mknod(const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, dev_t rdev) { struct inode *inode; @@ -105,7 +105,7 @@ static int ufs_mknod(struct mnt_idmap *idmap, struct inode *dir, return err; } -static int ufs_symlink (struct mnt_idmap * idmap, struct inode * dir, +static int ufs_symlink (const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, const char * symname) { struct super_block * sb = dir->i_sb; @@ -165,7 +165,7 @@ static int ufs_link (struct dentry * old_dentry, struct inode * dir, return error; } -static struct dentry *ufs_mkdir(struct mnt_idmap * idmap, struct inode * dir, +static struct dentry *ufs_mkdir(const struct mnt_idmap * idmap, struct inode * dir, struct dentry * dentry, umode_t mode) { struct inode * inode; @@ -240,7 +240,7 @@ static int ufs_rmdir (struct inode * dir, struct dentry *dentry) return err; } -static int ufs_rename(struct mnt_idmap *idmap, struct inode *old_dir, +static int ufs_rename(const struct mnt_idmap *idmap, struct inode *old_dir, struct dentry *old_dentry, struct inode *new_dir, struct dentry *new_dentry, unsigned int flags) { diff --git a/fs/ufs/ufs.h b/fs/ufs/ufs.h index 788e025056b2..541566f5b5fb 100644 --- a/fs/ufs/ufs.h +++ b/fs/ufs/ufs.h @@ -120,7 +120,7 @@ extern struct inode *ufs_iget(struct super_block *, unsigned long); extern int ufs_write_inode (struct inode *, struct writeback_control *); extern int ufs_sync_inode (struct inode *); extern void ufs_evict_inode (struct inode *); -extern int ufs_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +extern int ufs_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *attr); /* namei.c */ diff --git a/fs/vboxsf/dir.c b/fs/vboxsf/dir.c index 0b9eab157432..b9c46c769c4e 100644 --- a/fs/vboxsf/dir.c +++ b/fs/vboxsf/dir.c @@ -296,14 +296,14 @@ out: return err; } -static int vboxsf_dir_mkfile(struct mnt_idmap *idmap, +static int vboxsf_dir_mkfile(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, umode_t mode) { return vboxsf_dir_create(parent, dentry, mode, false, true, NULL); } -static struct dentry *vboxsf_dir_mkdir(struct mnt_idmap *idmap, +static struct dentry *vboxsf_dir_mkdir(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, umode_t mode) { @@ -382,7 +382,7 @@ static int vboxsf_dir_unlink(struct inode *parent, struct dentry *dentry) return 0; } -static int vboxsf_dir_rename(struct mnt_idmap *idmap, +static int vboxsf_dir_rename(const struct mnt_idmap *idmap, struct inode *old_parent, struct dentry *old_dentry, struct inode *new_parent, @@ -425,7 +425,7 @@ err_put_old_path: return err; } -static int vboxsf_dir_symlink(struct mnt_idmap *idmap, +static int vboxsf_dir_symlink(const struct mnt_idmap *idmap, struct inode *parent, struct dentry *dentry, const char *symname) { diff --git a/fs/vboxsf/utils.c b/fs/vboxsf/utils.c index 298bfc93255c..8775fbee1ce6 100644 --- a/fs/vboxsf/utils.c +++ b/fs/vboxsf/utils.c @@ -233,7 +233,7 @@ int vboxsf_inode_revalidate(struct dentry *dentry) return 0; } -int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, +int vboxsf_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *kstat, u32 request_mask, unsigned int flags) { int err; @@ -258,7 +258,7 @@ int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, return 0; } -int vboxsf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int vboxsf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct vboxsf_inode *sf_i = VBOXSF_I(d_inode(dentry)); diff --git a/fs/vboxsf/vfsmod.h b/fs/vboxsf/vfsmod.h index b61afd0ce842..59a4e44c4005 100644 --- a/fs/vboxsf/vfsmod.h +++ b/fs/vboxsf/vfsmod.h @@ -98,10 +98,10 @@ int vboxsf_stat(struct vboxsf_sbi *sbi, struct shfl_string *path, struct shfl_fsobjinfo *info); int vboxsf_stat_dentry(struct dentry *dentry, struct shfl_fsobjinfo *info); int vboxsf_inode_revalidate(struct dentry *dentry); -int vboxsf_getattr(struct mnt_idmap *idmap, const struct path *path, +int vboxsf_getattr(const struct mnt_idmap *idmap, const struct path *path, struct kstat *kstat, u32 request_mask, unsigned int query_flags); -int vboxsf_setattr(struct mnt_idmap *idmap, struct dentry *dentry, +int vboxsf_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr); struct shfl_string *vboxsf_path_from_dentry(struct vboxsf_sbi *sbi, struct dentry *dentry); diff --git a/fs/xattr.c b/fs/xattr.c index d58979115200..d9f035610f0b 100644 --- a/fs/xattr.c +++ b/fs/xattr.c @@ -100,7 +100,7 @@ xattr_resolve_name(struct inode *inode, const char **name) * * Return: On success zero is returned. On error a negative errno is returned. */ -int may_write_xattr(struct mnt_idmap *idmap, struct inode *inode) +int may_write_xattr(const struct mnt_idmap *idmap, struct inode *inode) { if (IS_IMMUTABLE(inode)) return -EPERM; @@ -123,7 +123,7 @@ static inline int xattr_permission_error(int mask) * because different namespaces have very different rules. */ static int -xattr_permission(struct mnt_idmap *idmap, struct inode *inode, +xattr_permission(const struct mnt_idmap *idmap, struct inode *inode, const char *name, int mask) { if (mask & MAY_WRITE) { @@ -204,7 +204,7 @@ xattr_supports_user_prefix(struct inode *inode) EXPORT_SYMBOL(xattr_supports_user_prefix); int -__vfs_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct inode *inode, const char *name, const void *value, size_t size, int flags) { @@ -242,7 +242,7 @@ EXPORT_SYMBOL(__vfs_setxattr); * is executed. It also assumes that the caller will make the appropriate * permission checks. */ -int __vfs_setxattr_noperm(struct mnt_idmap *idmap, +int __vfs_setxattr_noperm(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { @@ -295,7 +295,7 @@ int __vfs_setxattr_noperm(struct mnt_idmap *idmap, * a delegation was broken on, NULL if none. */ int -__vfs_setxattr_locked(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_setxattr_locked(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags, struct delegated_inode *delegated_inode) { @@ -324,7 +324,7 @@ out: EXPORT_SYMBOL_GPL(__vfs_setxattr_locked); int -vfs_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, const void *value, size_t size, int flags) { struct inode *inode = dentry->d_inode; @@ -358,7 +358,7 @@ retry_deleg: EXPORT_SYMBOL_GPL(vfs_setxattr); static ssize_t -xattr_getsecurity(struct mnt_idmap *idmap, struct inode *inode, +xattr_getsecurity(const struct mnt_idmap *idmap, struct inode *inode, const char *name, void *value, size_t size) { void *buffer = NULL; @@ -395,7 +395,7 @@ out_noalloc: * Returns the result of alloc, if failed, or the getxattr operation. */ int -vfs_getxattr_alloc(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_getxattr_alloc(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, char **xattr_value, size_t xattr_size, gfp_t flags) { @@ -448,7 +448,7 @@ __vfs_getxattr(struct dentry *dentry, struct inode *inode, const char *name, EXPORT_SYMBOL(__vfs_getxattr); ssize_t -vfs_getxattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_getxattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, void *value, size_t size) { struct inode *inode = dentry->d_inode; @@ -527,7 +527,7 @@ vfs_listxattr(struct dentry *dentry, char *list, size_t size) EXPORT_SYMBOL_GPL(vfs_listxattr); int -__vfs_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +__vfs_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct inode *inode = d_inode(dentry); @@ -557,7 +557,7 @@ EXPORT_SYMBOL(__vfs_removexattr); * a delegation was broken on, NULL if none. */ int -__vfs_removexattr_locked(struct mnt_idmap *idmap, +__vfs_removexattr_locked(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name, struct delegated_inode *delegated_inode) { @@ -589,7 +589,7 @@ out: EXPORT_SYMBOL_GPL(__vfs_removexattr_locked); int -vfs_removexattr(struct mnt_idmap *idmap, struct dentry *dentry, +vfs_removexattr(const struct mnt_idmap *idmap, struct dentry *dentry, const char *name) { struct inode *inode = dentry->d_inode; @@ -652,7 +652,7 @@ int setxattr_copy(const char __user *name, struct kernel_xattr_ctx *ctx) return error; } -static int do_setxattr(struct mnt_idmap *idmap, struct dentry *dentry, +static int do_setxattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct kernel_xattr_ctx *ctx) { if (is_posix_acl_xattr(ctx->kname->name)) @@ -787,7 +787,7 @@ SYSCALL_DEFINE5(fsetxattr, int, fd, const char __user *, name, * Extended attribute GET operations */ static ssize_t -do_getxattr(struct mnt_idmap *idmap, struct dentry *d, +do_getxattr(const struct mnt_idmap *idmap, struct dentry *d, struct kernel_xattr_ctx *ctx) { ssize_t error; @@ -1029,7 +1029,7 @@ SYSCALL_DEFINE3(flistxattr, int, fd, char __user *, list, size_t, size) * Extended attribute REMOVE operations */ static long -removexattr(struct mnt_idmap *idmap, struct dentry *d, const char *name) +removexattr(const struct mnt_idmap *idmap, struct dentry *d, const char *name) { if (is_posix_acl_xattr(name)) return vfs_remove_acl(idmap, d, name); diff --git a/fs/xfs/libxfs/xfs_errortag.h b/fs/xfs/libxfs/xfs_errortag.h index f0c83f1f0b3b..d14aa289699f 100644 --- a/fs/xfs/libxfs/xfs_errortag.h +++ b/fs/xfs/libxfs/xfs_errortag.h @@ -75,7 +75,8 @@ #define XFS_ERRTAG_METAFILE_RESV_CRITICAL 45 #define XFS_ERRTAG_FORCE_ZERO_RANGE 46 #define XFS_ERRTAG_ZONE_RESET 47 -#define XFS_ERRTAG_MAX 48 +#define XFS_ERRTAG_BOUNCE_REREAD 48 +#define XFS_ERRTAG_MAX 49 /* * Random factors for above tags, 1 means always, 2 means 1/2 time, etc. @@ -137,7 +138,8 @@ XFS_ERRTAG(WRITE_DELAY_MS, write_delay_ms, 3000) \ XFS_ERRTAG(EXCHMAPS_FINISH_ONE, exchmaps_finish_one, 1) \ XFS_ERRTAG(METAFILE_RESV_CRITICAL, metafile_resv_crit, 4) \ XFS_ERRTAG(FORCE_ZERO_RANGE, force_zero_range, 4) \ -XFS_ERRTAG(ZONE_RESET, zone_reset, 1) +XFS_ERRTAG(ZONE_RESET, zone_reset, 1) \ +XFS_ERRTAG(BOUNCE_REREAD, bounce_reread, XFS_RANDOM_DEFAULT) #endif /* XFS_ERRTAG */ #endif /* __XFS_ERRORTAG_H_ */ diff --git a/fs/xfs/libxfs/xfs_inode_util.h b/fs/xfs/libxfs/xfs_inode_util.h index 060242998a23..e9eac35159c3 100644 --- a/fs/xfs/libxfs/xfs_inode_util.h +++ b/fs/xfs/libxfs/xfs_inode_util.h @@ -27,7 +27,7 @@ prid_t xfs_get_initial_prid(struct xfs_inode *dp); * idmap to NULL. To create a tree root, set pip to NULL. */ struct xfs_icreate_args { - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; struct xfs_inode *pip; /* parent inode or null */ dev_t rdev; umode_t mode; diff --git a/fs/xfs/xfs_acl.c b/fs/xfs/xfs_acl.c index fdfca6fc75b6..20d87b52c4fc 100644 --- a/fs/xfs/xfs_acl.c +++ b/fs/xfs/xfs_acl.c @@ -243,7 +243,7 @@ xfs_acl_set_mode( } int -xfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +xfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type) { umode_t mode; diff --git a/fs/xfs/xfs_acl.h b/fs/xfs/xfs_acl.h index bf7f960997d3..183526bec32c 100644 --- a/fs/xfs/xfs_acl.h +++ b/fs/xfs/xfs_acl.h @@ -11,7 +11,7 @@ struct posix_acl; #ifdef CONFIG_XFS_POSIX_ACL extern struct posix_acl *xfs_get_acl(struct inode *inode, int type, bool rcu); -extern int xfs_set_acl(struct mnt_idmap *idmap, struct dentry *dentry, +extern int xfs_set_acl(const struct mnt_idmap *idmap, struct dentry *dentry, struct posix_acl *acl, int type); extern int __xfs_set_acl(struct inode *inode, struct posix_acl *acl, int type); void xfs_forget_acl(struct inode *inode, const char *name); diff --git a/fs/xfs/xfs_aops.c b/fs/xfs/xfs_aops.c index 8b6119776fb3..c30e688cfc9f 100644 --- a/fs/xfs/xfs_aops.c +++ b/fs/xfs/xfs_aops.c @@ -23,7 +23,6 @@ #include "xfs_ioend.h" #include "xfs_zone_alloc.h" #include "xfs_rtgroup.h" -#include <linux/bio-integrity.h> struct xfs_writepage_ctx { struct iomap_writepage_ctx ctx; @@ -498,8 +497,7 @@ xfs_zoned_writeback_submit( bio_endio(&ioend->io_bio); return error; } - if (wpc->iomap.flags & IOMAP_F_INTEGRITY) - fs_bio_integrity_generate(&ioend->io_bio); + xfs_zone_alloc_and_submit(ioend, &XFS_ZWPC(wpc)->open_zone); return 0; } @@ -585,11 +583,10 @@ xfs_bio_submit_read( const struct iomap_iter *iter, struct iomap_read_folio_ctx *ctx) { - struct bio *bio = ctx->read_ctx; - - /* defer read completions to the ioend workqueue */ - iomap_init_ioend(iter->inode, bio, ctx->read_ctx_file_offset, 0); - iomap_bio_submit_read_endio(iter, ctx, xfs_end_bio); + xfs_ioend_submit_read(iter->inode, ctx->read_ctx, + ctx->read_ctx_file_offset, + iomap_ioend_flags(&iter->iomap)); + ctx->read_ctx = NULL; } static const struct iomap_read_ops xfs_iomap_read_ops = { diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c index 8256c1d13ce2..6c93b4f5629c 100644 --- a/fs/xfs/xfs_buf.c +++ b/fs/xfs/xfs_buf.c @@ -5,6 +5,7 @@ */ #include "xfs_platform.h" #include <linux/backing-dev.h> +#include <linux/blk-integrity.h> #include <linux/dax.h> #include "xfs_shared.h" @@ -1694,6 +1695,7 @@ xfs_configure_buftarg( struct xfs_mount *mp = btp->bt_mount; if (btp->bt_bdev) { + struct blk_integrity *bi = bdev_get_integrity(btp->bt_bdev); int error; error = bdev_validate_blocksize(btp->bt_bdev, sectorsize); @@ -1706,6 +1708,15 @@ xfs_configure_buftarg( if (bdev_can_atomic_write(btp->bt_bdev)) xfs_configure_buftarg_atomic_writes(btp); + + if (!bi) + ; + else if (btp->bt_bdev == btp->bt_mount->m_super->s_bdev) + xfs_info(mp, "using %s integrity profile", + blk_integrity_profile_name(bi)); + else + xfs_info(mp, "using %s integrity profile for %pg", + blk_integrity_profile_name(bi), btp->bt_bdev); } btp->bt_meta_sectorsize = sectorsize; diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c index dd6d2e08faff..d164de6ff98b 100644 --- a/fs/xfs/xfs_file.c +++ b/fs/xfs/xfs_file.c @@ -37,6 +37,7 @@ #include <linux/fadvise.h> #include <linux/mount.h> #include <linux/filelock.h> +#include <linux/bio-integrity.h> static const struct vm_operations_struct xfs_file_vm_ops; @@ -222,9 +223,8 @@ xfs_dio_read_bounce_submit_io( struct bio *bio, loff_t file_offset) { - iomap_init_ioend(iter->inode, bio, file_offset, IOMAP_IOEND_DIRECT); - bio->bi_end_io = xfs_end_bio; - submit_bio(bio); + xfs_ioend_submit_read(iter->inode, bio, file_offset, + iomap_ioend_flags(&iter->iomap) | IOMAP_IOEND_DIRECT); } static const struct iomap_dio_ops xfs_dio_read_bounce_ops = { @@ -252,8 +252,7 @@ xfs_file_dio_read( return ret; if (mapping_stable_writes(iocb->ki_filp->f_mapping)) { ret = iomap_dio_rw(iocb, to, &xfs_read_iomap_ops, - &xfs_dio_read_bounce_ops, IOMAP_DIO_BOUNCE, - NULL, 0); + &xfs_dio_read_bounce_ops, 0, NULL, 0); } else { ret = iomap_dio_read_simple(iocb, to, xfs_read_iomap_begin); if (ret == -ENOTBLK) @@ -713,7 +712,7 @@ xfs_dio_zoned_submit_io( bio->bi_end_io = xfs_end_bio; ioend = iomap_init_ioend(iter->inode, bio, file_offset, - IOMAP_IOEND_DIRECT); + iomap_ioend_flags(&iter->iomap) | IOMAP_IOEND_DIRECT); xfs_zone_alloc_and_submit(ioend, &ac->open_zone); } diff --git a/fs/xfs/xfs_handle.c b/fs/xfs/xfs_handle.c index fd9d4d8258ff..4924e676ae91 100644 --- a/fs/xfs/xfs_handle.c +++ b/fs/xfs/xfs_handle.c @@ -272,11 +272,11 @@ xfs_open_by_handle( path.mnt = mntget(parfilp->f_path.mnt); FD_PREPARE(fdf, 0, dentry_open(&path, hreq->oflags, cred)); - if (fdf.err) - return fdf.err; + if (fdf->fd < 0) + return fdf->fd; if (S_ISREG(inode->i_mode)) { - struct file *filp = fd_prepare_file(fdf); + struct file *filp = fdf->file; filp->f_flags |= O_NOATIME; filp->f_mode |= FMODE_NOCMTIME; diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c index 15b62574b8d4..05a14da28031 100644 --- a/fs/xfs/xfs_inode.c +++ b/fs/xfs/xfs_inode.c @@ -2084,7 +2084,7 @@ xfs_sort_inodes( */ static int xfs_rename_alloc_whiteout( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_name *src_name, struct xfs_inode *dp, struct xfs_inode **wip) @@ -2130,7 +2130,7 @@ xfs_rename_alloc_whiteout( */ int xfs_rename( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_inode *src_dp, struct xfs_name *src_name, struct xfs_inode *src_ip, diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h index 1602027cd0aa..ca96ba096359 100644 --- a/fs/xfs/xfs_inode.h +++ b/fs/xfs/xfs_inode.h @@ -568,7 +568,7 @@ int xfs_remove(struct xfs_inode *dp, struct xfs_name *name, struct xfs_inode *ip); int xfs_link(struct xfs_inode *tdp, struct xfs_inode *sip, struct xfs_name *target_name); -int xfs_rename(struct mnt_idmap *idmap, +int xfs_rename(const struct mnt_idmap *idmap, struct xfs_inode *src_dp, struct xfs_name *src_name, struct xfs_inode *src_ip, struct xfs_inode *target_dp, struct xfs_name *target_name, diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c index c0fc9b34f393..f81b6e52ac40 100644 --- a/fs/xfs/xfs_ioctl.c +++ b/fs/xfs/xfs_ioctl.c @@ -748,7 +748,7 @@ xfs_ioctl_setattr_check_projid( int xfs_fileattr_set( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa) { diff --git a/fs/xfs/xfs_ioctl.h b/fs/xfs/xfs_ioctl.h index e57d8f5148bf..6e55cc847654 100644 --- a/fs/xfs/xfs_ioctl.h +++ b/fs/xfs/xfs_ioctl.h @@ -19,7 +19,7 @@ xfs_fileattr_get( extern int xfs_fileattr_set( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct file_kattr *fa); diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c index 40695d18dac0..e70be5b86f0b 100644 --- a/fs/xfs/xfs_ioend.c +++ b/fs/xfs/xfs_ioend.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * Copyright (c) 2016-2025 Christoph Hellwig. + * Copyright (c) 2016-2026 Christoph Hellwig. * All Rights Reserved. */ #include "xfs_platform.h" @@ -16,6 +16,135 @@ #include "xfs_reflink.h" #include "xfs_zone_alloc.h" #include "xfs_ioend.h" +#include "xfs_error.h" +#include "xfs_errortag.h" +#include <linux/bio-integrity.h> + +static void +xfs_dio_bounce_end_io( + struct bio *bio) +{ + struct iomap_ioend *ioend = iomap_ioend_from_bio(bio); + int error = blk_status_to_errno(bio->bi_status); + struct bio *orig_bio = bio->bi_private; + + if ((ioend->io_flags & IOMAP_IOEND_INTEGRITY) && !bio->bi_status) + error = iomap_ioend_integrity_verify(ioend); + iomap_bounce_read_end_io(ioend, orig_bio, error); +} + +static void +xfs_bounce_submit_ioend( + struct iomap_ioend *ioend) +{ + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_alloc(&ioend->io_bio); + ioend->io_bio.bi_end_io = xfs_dio_bounce_end_io; + bio_set_flag(&ioend->io_bio, BIO_COMPLETE_IN_TASK); + submit_bio(&ioend->io_bio); +} + +static void +xfs_end_bio_bounced( + struct bio *bio) +{ + /* + * Just complete the original ioends as all verification is done by the + * end_io handlers for the clone bio(s). + */ + iomap_finish_ioends(iomap_ioend_from_bio(bio), + blk_status_to_errno(bio->bi_status)); +} + +static void +xfs_read_bounce_and_resubmit( + struct iomap_ioend *ioend) +{ + struct bio *bio = &ioend->io_bio; + struct xfs_inode *ip = XFS_I(ioend->io_inode); + unsigned int nofs_flag = memalloc_nofs_save(); + + trace_xfs_bounce_reread(ip, ioend->io_offset, ioend->io_size); + + /* + * Free the bio integrity data for the original bio, as we'll allocate + * a new one for each sub-I/O, which could deadlock if we keep the + * integrity data for the original bio around. + */ + if (bio_integrity(bio)) + fs_bio_integrity_free(bio); + + /* + * Resubmit the bio through the iomap bounce machinery. The original + * bio itself is not resubmitted to the block layer, but just used to + * track I/O completion of the cloned bios. + */ + bio_prepare_reissue(bio, xfs_inode_buftarg(ip)->bt_bdev); + bio->bi_iter = (struct bvec_iter) { + .bi_sector = ioend->io_sector, + .bi_size = ioend->io_size, + .bi_offset = ioend->io_bvec_offset, + }; + bio->bi_end_io = xfs_end_bio_bounced; + iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev), + xfs_bounce_submit_ioend); + memalloc_nofs_restore(nofs_flag); +} + +static void +xfs_end_io_read( + struct bio *bio) +{ + struct iomap_ioend *ioend = iomap_ioend_from_bio(bio); + struct xfs_inode *ip = XFS_I(ioend->io_inode); + struct xfs_mount *mp = ip->i_mount; + int error = blk_status_to_errno(bio->bi_status); + + if (!error && (ioend->io_flags & IOMAP_IOEND_INTEGRITY)) { + error = iomap_ioend_integrity_verify(ioend); + if ((ioend->io_flags & IOMAP_IOEND_DIRECT) && + READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_LAZY) { + /* + * We only really need to retry for guard tag errors, + * but right now we can't distinguish them from other + * (i.e, reftag) errors. + */ + if (error || + XFS_TEST_ERROR(mp, XFS_ERRTAG_BOUNCE_REREAD)) { + xfs_read_bounce_and_resubmit(ioend); + return; + } + } + } + + iomap_finish_ioends(ioend, error); +} + +void +xfs_ioend_submit_read( + struct inode *inode, + struct bio *bio, + loff_t file_offset, + u16 ioend_flags) +{ + struct xfs_inode *ip = XFS_I(inode); + struct xfs_mount *mp = ip->i_mount; + struct iomap_ioend *ioend; + + ioend = iomap_init_ioend(inode, bio, file_offset, ioend_flags); + if ((ioend_flags & IOMAP_IOEND_DIRECT) && + READ_ONCE(mp->m_read_bounce) == XFS_READ_BOUNCE_ALWAYS) { + iomap_bounce_read(ioend, bdev_logical_block_size(bio->bi_bdev), + xfs_bounce_submit_ioend); + return; + } + + if (ioend_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_alloc(bio); + bio->bi_end_io = xfs_end_io_read; + bio_set_flag(bio, BIO_COMPLETE_IN_TASK); + submit_bio(bio); +} static void xfs_ioend_put_open_zones( @@ -148,11 +277,7 @@ xfs_end_io( io_list))) { list_del_init(&ioend->io_list); iomap_ioend_try_merge(ioend, &tmp); - if (bio_op(&ioend->io_bio) == REQ_OP_READ) - iomap_finish_ioends(ioend, - blk_status_to_errno(ioend->io_bio.bi_status)); - else - xfs_end_ioend_write(ioend); + xfs_end_ioend_write(ioend); cond_resched(); } } diff --git a/fs/xfs/xfs_ioend.h b/fs/xfs/xfs_ioend.h index 525865767fca..7c2a1ea3e6ed 100644 --- a/fs/xfs/xfs_ioend.h +++ b/fs/xfs/xfs_ioend.h @@ -12,5 +12,7 @@ static inline bool xfs_ioend_is_append(struct iomap_ioend *ioend) } void xfs_end_bio(struct bio *bio); +void xfs_ioend_submit_read(struct inode *inode, struct bio *bio, + loff_t file_offset, u16 ioend_flags); #endif /* __XFS_IOEND_H */ diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c index d1306e723899..f67a541ef35a 100644 --- a/fs/xfs/xfs_iops.c +++ b/fs/xfs/xfs_iops.c @@ -169,7 +169,7 @@ xfs_create_need_xattr( STATIC int xfs_generic_create( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -279,7 +279,7 @@ xfs_generic_create( STATIC int xfs_vn_mknod( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode, @@ -290,7 +290,7 @@ xfs_vn_mknod( STATIC int xfs_vn_create( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -300,7 +300,7 @@ xfs_vn_create( STATIC struct dentry * xfs_vn_mkdir( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, umode_t mode) @@ -425,7 +425,7 @@ xfs_vn_unlink( STATIC int xfs_vn_symlink( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct dentry *dentry, const char *symname) @@ -466,7 +466,7 @@ xfs_vn_symlink( STATIC int xfs_vn_rename( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *odir, struct dentry *odentry, struct inode *ndir, @@ -679,7 +679,7 @@ xfs_report_atomic_write( STATIC int xfs_vn_getattr( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, const struct path *path, struct kstat *stat, u32 request_mask, @@ -754,7 +754,7 @@ xfs_vn_getattr( static int xfs_vn_change_ok( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -777,7 +777,7 @@ xfs_vn_change_ok( */ static int xfs_setattr_nonsize( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct xfs_inode *ip, struct iattr *iattr) @@ -903,7 +903,7 @@ out_dqrele: */ int xfs_vn_setattr_size( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -1130,7 +1130,7 @@ out_trans_cancel: STATIC int xfs_vn_setattr( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { @@ -1250,7 +1250,7 @@ xfs_vn_fiemap( STATIC int xfs_vn_tmpfile( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct inode *dir, struct file *file, umode_t mode) diff --git a/fs/xfs/xfs_iops.h b/fs/xfs/xfs_iops.h index 0896f6b8b3b8..328305bba19d 100644 --- a/fs/xfs/xfs_iops.h +++ b/fs/xfs/xfs_iops.h @@ -10,7 +10,7 @@ struct xfs_inode; extern ssize_t xfs_vn_listxattr(struct dentry *, char *data, size_t size); -int xfs_vn_setattr_size(struct mnt_idmap *idmap, +int xfs_vn_setattr_size(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *vap); int xfs_inode_init_security(struct inode *inode, struct inode *dir, diff --git a/fs/xfs/xfs_itable.c b/fs/xfs/xfs_itable.c index 159295c63e8f..a4cf1effa5e6 100644 --- a/fs/xfs/xfs_itable.c +++ b/fs/xfs/xfs_itable.c @@ -63,7 +63,7 @@ want_metadir_file( STATIC int xfs_bulkstat_one_int( struct xfs_mount *mp, - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_trans *tp, xfs_ino_t ino, struct xfs_bstat_chunk *bc) diff --git a/fs/xfs/xfs_itable.h b/fs/xfs/xfs_itable.h index 2d0612f14d6e..c0567bfc30fb 100644 --- a/fs/xfs/xfs_itable.h +++ b/fs/xfs/xfs_itable.h @@ -8,7 +8,7 @@ /* In-memory representation of a userspace request for batch inode data. */ struct xfs_ibulk { struct xfs_mount *mp; - struct mnt_idmap *idmap; + const struct mnt_idmap *idmap; void __user *ubuffer; /* user output buffer */ xfs_ino_t startino; /* start with this inode */ unsigned int icount; /* number of elements in ubuffer */ diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h index 216a38a354e7..894ff2f4ecbd 100644 --- a/fs/xfs/xfs_mount.h +++ b/fs/xfs/xfs_mount.h @@ -142,6 +142,12 @@ struct xfs_freecounter { uint64_t res_saved; }; +enum xfs_read_bounce { + XFS_READ_BOUNCE_NEVER, + XFS_READ_BOUNCE_ALWAYS, + XFS_READ_BOUNCE_LAZY, +}; + /* * The struct xfsmount layout is optimised to separate read-mostly variables * from variables that are frequently modified. We put the read-mostly variables @@ -177,6 +183,7 @@ typedef struct xfs_mount { struct workqueue_struct *m_sync_workqueue; struct workqueue_struct *m_blockgc_wq; struct workqueue_struct *m_inodegc_wq; + enum xfs_read_bounce m_read_bounce; int m_bsize; /* fs logical block size */ uint8_t m_blkbit_log; /* blocklog + NBBY */ @@ -291,6 +298,7 @@ typedef struct xfs_mount { struct xfs_zone_info *m_zone_info; /* zone allocator information */ struct dentry *m_debugfs; /* debugfs parent */ struct xfs_kobj m_kobj; + struct xfs_kobj m_csum_kobj; struct xfs_kobj m_error_kobj; struct xfs_kobj m_error_meta_kobj; struct xfs_error_cfg m_error_cfg[XFS_ERR_CLASS_MAX][XFS_ERR_ERRNO_MAX]; diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c index 2edc2a497883..5a06132aa384 100644 --- a/fs/xfs/xfs_super.c +++ b/fs/xfs/xfs_super.c @@ -2317,6 +2317,7 @@ xfs_init_fs_context( mp->m_logbufs = -1; mp->m_logbsize = -1; mp->m_allocsize_log = 16; /* 64k */ + mp->m_read_bounce = XFS_READ_BOUNCE_LAZY; xfs_hooks_init(&mp->m_dir_update_hooks); diff --git a/fs/xfs/xfs_symlink.c b/fs/xfs/xfs_symlink.c index cc13819df6f2..709cd22248f8 100644 --- a/fs/xfs/xfs_symlink.c +++ b/fs/xfs/xfs_symlink.c @@ -82,7 +82,7 @@ xfs_readlink( int xfs_symlink( - struct mnt_idmap *idmap, + const struct mnt_idmap *idmap, struct xfs_inode *dp, struct xfs_name *link_name, const char *target_path, diff --git a/fs/xfs/xfs_symlink.h b/fs/xfs/xfs_symlink.h index 0d29a50e66fd..3c5a969f9fc5 100644 --- a/fs/xfs/xfs_symlink.h +++ b/fs/xfs/xfs_symlink.h @@ -7,7 +7,7 @@ /* Kernel only symlink definitions */ -int xfs_symlink(struct mnt_idmap *idmap, struct xfs_inode *dp, +int xfs_symlink(const struct mnt_idmap *idmap, struct xfs_inode *dp, struct xfs_name *link_name, const char *target_path, umode_t mode, struct xfs_inode **ipp); int xfs_readlink(struct xfs_inode *ip, char *link); diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c index b62712187324..e77917ac179d 100644 --- a/fs/xfs/xfs_sysfs.c +++ b/fs/xfs/xfs_sysfs.c @@ -392,6 +392,71 @@ const struct kobj_type xfs_stats_ktype = { .default_groups = xfs_stats_groups, }; +static inline struct xfs_mount *csum_to_mp(struct kobject *kobj) +{ + return container_of(to_kobj(kobj), struct xfs_mount, m_csum_kobj); +} + +static bool +xfs_has_read_bounce( + struct xfs_mount *mp) +{ + if (bdev_has_integrity_csum(mp->m_ddev_targp->bt_bdev)) + return true; + if (mp->m_rtdev_targp && + bdev_has_integrity_csum(mp->m_rtdev_targp->bt_bdev)) + return true; + return false; +} + +static const char * const bounce_modes[] = { + [XFS_READ_BOUNCE_NEVER] = "never", + [XFS_READ_BOUNCE_ALWAYS] = "always", + [XFS_READ_BOUNCE_LAZY] = "lazy", +}; + +static ssize_t +read_bounce_show( + struct kobject *kobj, + char *buf) +{ + struct xfs_mount *mp = csum_to_mp(kobj); + + return sysfs_emit(buf, "%s\n", + bounce_modes[READ_ONCE(mp->m_read_bounce)]); +} + +static ssize_t +read_bounce_store( + struct kobject *kobj, + const char *buf, + size_t count) +{ + struct xfs_mount *mp = csum_to_mp(kobj); + int ret; + + if (!xfs_has_read_bounce(mp)) + return -EINVAL; + ret = sysfs_match_string(bounce_modes, buf); + if (ret < 0) + return ret; + WRITE_ONCE(mp->m_read_bounce, ret); + return count; +} +XFS_SYSFS_ATTR_RW(read_bounce); + +static struct attribute *xfs_csum_attrs[] = { + ATTR_LIST(read_bounce), + NULL, +}; +ATTRIBUTE_GROUPS(xfs_csum); + +static const struct kobj_type xfs_csum_ktype = { + .release = xfs_sysfs_release, + .sysfs_ops = &xfs_sysfs_ops, + .default_groups = xfs_csum_groups, +}; + /* xlog */ static inline struct xlog * @@ -817,11 +882,17 @@ xfs_mount_sysfs_init( if (error) goto out_remove_fsdir; + /* .../xfs/<dev>/csum/ */ + error = xfs_sysfs_init(&mp->m_csum_kobj, &xfs_csum_ktype, &mp->m_kobj, + "csum"); + if (error) + goto out_remove_stats_dir; + /* .../xfs/<dev>/error/ */ error = xfs_sysfs_init(&mp->m_error_kobj, &xfs_error_ktype, &mp->m_kobj, "error"); if (error) - goto out_remove_stats_dir; + goto out_remove_csum_dir; /* .../xfs/<dev>/error/fail_at_unmount */ error = sysfs_create_file(&mp->m_error_kobj.kobject, @@ -835,12 +906,14 @@ xfs_mount_sysfs_init( "metadata", &mp->m_error_meta_kobj, xfs_error_meta_init); if (error) - goto out_remove_error_dir; + goto out_remove_csum_dir; return 0; out_remove_error_dir: xfs_sysfs_del(&mp->m_error_kobj); +out_remove_csum_dir: + xfs_sysfs_del(&mp->m_csum_kobj); out_remove_stats_dir: xfs_sysfs_del(&mp->m_stats.xs_kobj); out_remove_fsdir: @@ -864,6 +937,7 @@ xfs_mount_sysfs_del( } xfs_sysfs_del(&mp->m_error_meta_kobj); xfs_sysfs_del(&mp->m_error_kobj); + xfs_sysfs_del(&mp->m_csum_kobj); xfs_sysfs_del(&mp->m_stats.xs_kobj); xfs_sysfs_del(&mp->m_kobj); } diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h index 0fc8927339b5..28d49f158b22 100644 --- a/fs/xfs/xfs_trace.h +++ b/fs/xfs/xfs_trace.h @@ -1896,6 +1896,7 @@ DEFINE_SIMPLE_IO_EVENT(xfs_zero_eof); DEFINE_SIMPLE_IO_EVENT(xfs_end_io_direct_write); DEFINE_SIMPLE_IO_EVENT(xfs_file_splice_read); DEFINE_SIMPLE_IO_EVENT(xfs_zoned_map_blocks); +DEFINE_SIMPLE_IO_EVENT(xfs_bounce_reread); DECLARE_EVENT_CLASS(xfs_itrunc_class, TP_PROTO(struct xfs_inode *ip, xfs_fsize_t new_size), diff --git a/fs/xfs/xfs_xattr.c b/fs/xfs/xfs_xattr.c index 1efe6c8139b2..b059a9714d11 100644 --- a/fs/xfs/xfs_xattr.c +++ b/fs/xfs/xfs_xattr.c @@ -169,7 +169,7 @@ xfs_xattr_flags_to_op( static int xfs_xattr_set(const struct xattr_handler *handler, - struct mnt_idmap *idmap, struct dentry *unused, + const struct mnt_idmap *idmap, struct dentry *unused, struct inode *inode, const char *name, const void *value, size_t size, int flags) { diff --git a/fs/xfs/xfs_zone_alloc.c b/fs/xfs/xfs_zone_alloc.c index b75cf3bfe33c..5f0af0c2c5e5 100644 --- a/fs/xfs/xfs_zone_alloc.c +++ b/fs/xfs/xfs_zone_alloc.c @@ -26,6 +26,7 @@ #include "xfs_zones.h" #include "xfs_trace.h" #include "xfs_mru_cache.h" +#include <linux/bio-integrity.h> static void xfs_open_zone_free_rcu( @@ -911,6 +912,9 @@ xfs_zone_alloc_and_submit( if (xfs_is_shutdown(mp)) goto out_error; + if (ioend->io_flags & IOMAP_IOEND_INTEGRITY) + fs_bio_integrity_generate(&ioend->io_bio); + /* * If we don't have a locally cached zone in this write context, see if * the inode is still associated with a zone and use that if so. diff --git a/fs/zonefs/super.c b/fs/zonefs/super.c index ff43d6d1ea30..b97f1f2b8dda 100644 --- a/fs/zonefs/super.c +++ b/fs/zonefs/super.c @@ -533,7 +533,7 @@ static int zonefs_show_options(struct seq_file *seq, struct dentry *root) return 0; } -static int zonefs_inode_setattr(struct mnt_idmap *idmap, +static int zonefs_inode_setattr(const struct mnt_idmap *idmap, struct dentry *dentry, struct iattr *iattr) { struct inode *inode = d_inode(dentry); |
