diff options
Diffstat (limited to 'fs')
73 files changed, 520 insertions, 272 deletions
diff --git a/fs/autofs/inode.c b/fs/autofs/inode.c index 6b15a3717ba7..066c16f2ea56 100644 --- a/fs/autofs/inode.c +++ b/fs/autofs/inode.c @@ -51,6 +51,10 @@ void autofs_kill_sb(struct super_block *sb) if (sbi) { /* Free wait queues, close pipe */ autofs_catatonic_mode(sbi); + if (sbi->pipe) { + fput(sbi->pipe); + sbi->pipe = NULL; + } put_pid(sbi->oz_pgrp); } diff --git a/fs/binfmt_misc_bpf.c b/fs/binfmt_misc_bpf.c index 91576ff05911..a3e26e8a4027 100644 --- a/fs/binfmt_misc_bpf.c +++ b/fs/binfmt_misc_bpf.c @@ -141,8 +141,6 @@ __bpf_kfunc int bpf_binprm_set_interp(struct linux_binprm *bprm, len = strnlen(path, path__sz); if (len == path__sz) return -EINVAL; - if (path[0] != '/') - return -EINVAL; if (len >= PATH_MAX) return -ENAMETOOLONG; @@ -150,6 +148,15 @@ __bpf_kfunc int bpf_binprm_set_interp(struct linux_binprm *bprm, if (!interp) return -ENOMEM; + /* + * The program may pass memory that is written to while this runs, + * so check the private copy and not the buffer it was made from. + */ + if (interp[0] != '/') { + kfree(interp); + return -EINVAL; + } + bm_bpf_stage_selection(bprm, interp, NULL); return 0; } @@ -176,6 +183,7 @@ __bpf_kfunc int bpf_binprm_select_interp(struct linux_binprm *bprm, const char *name, size_t name__sz) { const struct binfmt_misc_interp *interp; + char buf[BINFMT_MISC_INTERP_NAME_MAX + 1]; size_t len; char *path; @@ -184,8 +192,20 @@ __bpf_kfunc int bpf_binprm_select_interp(struct linux_binprm *bprm, len = strnlen(name, name__sz); if (len == name__sz || !len) return -EINVAL; + /* No entry binds a longer name, so it cannot be found. */ + if (len > BINFMT_MISC_INTERP_NAME_MAX) + return -ENOENT; + + /* + * The program may pass memory that is written to while this runs, + * so look the name up in a private copy and check that instead. + */ + memcpy(buf, name, len); + buf[len] = '\0'; + if (!buf[0]) + return -EINVAL; - interp = binfmt_misc_find_interp(bprm->bpf_interps, name); + interp = binfmt_misc_find_interp(bprm->bpf_interps, buf); if (!interp) return -ENOENT; @@ -228,6 +248,15 @@ __bpf_kfunc int bpf_binprm_set_interp_arg(struct linux_binprm *bprm, if (!val) return -ENOMEM; + /* + * The program may pass memory that is written to while this runs, + * so check the private copy and not the buffer it was made from. + */ + if (!val[0]) { + kfree(val); + return -EINVAL; + } + kfree(bprm->bpf_interp_arg); bprm->bpf_interp_arg = val; return 0; diff --git a/fs/bpf_fs_kfuncs.c b/fs/bpf_fs_kfuncs.c index 6cb877267978..357a379ef92a 100644 --- a/fs/bpf_fs_kfuncs.c +++ b/fs/bpf_fs_kfuncs.c @@ -472,10 +472,6 @@ BTF_ID(func, bpf_lsm_inode_rmdir) BTF_ID(func, bpf_lsm_inode_setattr) BTF_ID(func, bpf_lsm_inode_setxattr) BTF_ID(func, bpf_lsm_inode_unlink) -#ifdef CONFIG_SECURITY_PATH -BTF_ID(func, bpf_lsm_path_unlink) -BTF_ID(func, bpf_lsm_path_rmdir) -#endif /* CONFIG_SECURITY_PATH */ BTF_SET_END(d_inode_locked_hooks) bool bpf_lsm_has_d_inode_locked(const struct bpf_prog *prog) diff --git a/fs/dcache.c b/fs/dcache.c index 1b1a81f10da6..a66be85f9d01 100644 --- a/fs/dcache.c +++ b/fs/dcache.c @@ -1916,6 +1916,10 @@ static struct dentry *__d_alloc(struct super_block *sb, const struct qstr *name) * be overwriting an internal NUL character */ dentry->d_shortname.string[DNAME_INLINE_LEN-1] = 0; + + /* Racy __d_lookup_rcu() walk may read past the NUL; harmless */ + kmsan_unpoison_memory(dentry->d_shortname.string, DNAME_INLINE_LEN); + if (unlikely(!name)) { name = &slash_name; dname = dentry->d_shortname.string; diff --git a/fs/debugfs/inode.c b/fs/debugfs/inode.c index e054e62919ec..a4d08bd3743b 100644 --- a/fs/debugfs/inode.c +++ b/fs/debugfs/inode.c @@ -368,6 +368,9 @@ static struct dentry *debugfs_start_creating(const char *name, if (!debugfs_enabled) return ERR_PTR(-EPERM); + if (IS_ERR(parent)) + return parent; + if (!debugfs_initialized()) { pr_err("Unable to create file '%s', debugfs is not initialized yet\n", name); @@ -376,9 +379,6 @@ static struct dentry *debugfs_start_creating(const char *name, pr_debug("creating file '%s'\n", name); - if (IS_ERR(parent)) - return parent; - error = simple_pin_fs(&debug_fs_type, &debugfs_mount, &debugfs_mount_count); if (error) { diff --git a/fs/exec.c b/fs/exec.c index 819643408e6d..a5269b5e00df 100644 --- a/fs/exec.c +++ b/fs/exec.c @@ -882,6 +882,7 @@ static int exec_mmap(struct linux_binprm *bprm) active_mm = tsk->active_mm; tsk->active_mm = mm; tsk->mm = mm; + sched_cache_exec_mmap(tsk, mm); mm_init_cid(mm, tsk); exec_state = task_exec_state_replace(tsk, exec_state); /* diff --git a/fs/fs-writeback.c b/fs/fs-writeback.c index e744f9f9d43f..ea3eb40bf828 100644 --- a/fs/fs-writeback.c +++ b/fs/fs-writeback.c @@ -727,19 +727,34 @@ static bool isw_prepare_wbs_switch(struct bdi_writeback *new_wb, struct inode_switch_wbs_context *isw, struct list_head *list, int *nr) { - struct inode *inode; + struct inode *inode, *tmp; + LIST_HEAD(scanned); + bool full = false; + + /* + * Walk from the oldest end and move scanned inodes to the newest + * end, so the next scan resumes at unscanned inodes instead of + * re-walking an ever-growing run of prepared and skipped ones. + * For b_dirty_time this keeps the oldest unscanned inode at the + * end move_expired_inodes() picks from; b_attached is unordered. + */ + list_for_each_entry_safe_reverse(inode, tmp, list, i_io_list) { + list_move(&inode->i_io_list, &scanned); - list_for_each_entry(inode, list, i_io_list) { if (!inode_prepare_wbs_switch(inode, new_wb)) continue; isw->inodes[*nr] = inode; (*nr)++; - if (*nr >= WB_MAX_INODES_PER_ISW - 1) - return true; + if (*nr >= WB_MAX_INODES_PER_ISW - 1) { + full = true; + break; + } } - return false; + list_splice(&scanned, list); + + return full; } /** diff --git a/fs/inode.c b/fs/inode.c index ba7da39be4a3..a9d37be390a1 100644 --- a/fs/inode.c +++ b/fs/inode.c @@ -880,7 +880,6 @@ void evict_inodes(struct super_block *sb) struct inode *inode; LIST_HEAD(dispose); -again: spin_lock(&sb->s_inode_list_lock); list_for_each_entry(inode, &sb->s_inodes, i_sb_list) { if (icount_read_once(inode)) @@ -899,19 +898,19 @@ again: inode_state_set(inode, I_FREEING); inode_lru_list_del(inode); spin_unlock(&inode->i_lock); - list_add(&inode->i_lru, &dispose); /* - * We can have a ton of inodes to evict at unmount time given - * enough memory, check to see if we need to go to sleep for a - * bit so we don't livelock. + * Keep this inode out of dispose so it stays on s_inodes while + * the list lock is dropped. I_FREEING prevents new references + * and leaves eviction to us, so we can resume the walk from it. */ if (need_resched()) { spin_unlock(&sb->s_inode_list_lock); cond_resched(); dispose_list(&dispose); - goto again; + spin_lock(&sb->s_inode_list_lock); } + list_add(&inode->i_lru, &dispose); } spin_unlock(&sb->s_inode_list_lock); diff --git a/fs/isofs/dir.c b/fs/isofs/dir.c index c7ca7603e97a..28741251d56e 100644 --- a/fs/isofs/dir.c +++ b/fs/isofs/dir.c @@ -110,25 +110,21 @@ static int do_isofs_readdir(struct inode *inode, struct file *file, return 0; } - de = (struct iso_directory_record *) (bh->b_data + offset); - - de_len = *(unsigned char *)de; - + de = (struct iso_directory_record *)(bh->b_data + offset); /* - * If the length byte is zero, we should move on to the next - * CDROM sector. If we are at the end of the directory, we - * kick out of the while loop. + * If we are at the end of a block (or at its zero-padded + * tail), move on to the next CDROM sector. If we are at the + * end of the directory, we'll abort the while loop. */ - - if (de_len == 0) { + if (offset >= bufsize || de->length[0] == 0) { brelse(bh); bh = NULL; - ctx->pos = (ctx->pos + ISOFS_BLOCK_SIZE) & ~(ISOFS_BLOCK_SIZE - 1); - block = ctx->pos >> bufbits; + block++; + ctx->pos = (loff_t)block << bufbits; offset = 0; continue; } - + de_len = de->length[0]; block_saved = block; offset_saved = offset; offset += de_len; diff --git a/fs/isofs/namei.c b/fs/isofs/namei.c index 010682f5901a..4fba1bf7f016 100644 --- a/fs/isofs/namei.c +++ b/fs/isofs/namei.c @@ -74,18 +74,20 @@ isofs_find_entry(struct inode *dir, struct dentry *dentry, return 0; } - de = (struct iso_directory_record *) (bh->b_data + offset); - - de_len = *(unsigned char *) de; - if (!de_len) { + de = (struct iso_directory_record *)(bh->b_data + offset); + /* + * If we are at the end of the block or at its zero-padded + * tail, move to the next block. + */ + if (offset >= bufsize || de->length[0] == 0) { brelse(bh); bh = NULL; - f_pos = (f_pos + ISOFS_BLOCK_SIZE) & ~(ISOFS_BLOCK_SIZE - 1); - block = f_pos >> bufbits; + block++; + f_pos = block << bufbits; offset = 0; continue; } - + de_len = de->length[0]; block_saved = bh->b_blocknr; offset_saved = offset; offset += de_len; diff --git a/fs/kernfs/mount.c b/fs/kernfs/mount.c index f183a96778b9..a57399021c8b 100644 --- a/fs/kernfs/mount.c +++ b/fs/kernfs/mount.c @@ -434,8 +434,8 @@ void kernfs_kill_sb(struct super_block *sb) up_write(&root->kernfs_supers_rwsem); /* - * Remove the superblock from fs_supers/s_instances - * so we can't find it, before freeing kernfs_super_info. + * Mark the superblock dead so sget_fc() can't find it, + * before freeing kernfs_super_info. */ kill_anon_super(sb); kfree(info); diff --git a/fs/namespace.c b/fs/namespace.c index ae5dc64f8b45..580877e46b1a 100644 --- a/fs/namespace.c +++ b/fs/namespace.c @@ -6184,6 +6184,21 @@ struct mnt_namespace init_mnt_ns = { .poll = __WAIT_QUEUE_HEAD_INITIALIZER(init_mnt_ns.poll), }; +static void __init mount_rootfs_on_nullfs(struct vfsmount *mnt, + struct vfsmount *nullfs_mnt) +{ + struct path root = { + .mnt = nullfs_mnt, + .dentry = nullfs_mnt->mnt_root, + }; + + LOCK_MOUNT_EXACT(mp, &root); + if (unlikely(IS_ERR(mp.parent))) + panic("VFS: Failed to mount rootfs on nullfs"); + scoped_guard(mount_writer) + attach_mnt(real_mount(mnt), mp.parent, mp.mp); +} + static void __init init_mount_tree(void) { struct vfsmount *mnt, *nullfs_mnt; @@ -6215,15 +6230,7 @@ static void __init init_mount_tree(void) mnt_root = real_mount(nullfs_mnt); init_mnt_ns.root = mnt_root; - /* Mount mutable rootfs on top of nullfs. */ - root.mnt = nullfs_mnt; - root.dentry = nullfs_mnt->mnt_root; - - LOCK_MOUNT_EXACT(mp, &root); - if (unlikely(IS_ERR(mp.parent))) - panic("VFS: Failed to mount rootfs on nullfs"); - scoped_guard(mount_writer) - attach_mnt(real_mount(mnt), mp.parent, mp.mp); + mount_rootfs_on_nullfs(mnt, nullfs_mnt); pr_info("VFS: Finished mounting rootfs on nullfs\n"); @@ -6294,7 +6301,7 @@ void put_mnt_ns(struct mnt_namespace *ns) guard(namespace_excl)(); emptied_ns = ns; guard(mount_writer)(); - umount_tree(ns->root, UMOUNT_CONNECTED); + umount_tree(ns->root, 0); } struct vfsmount *kern_mount(struct file_system_type *type) diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c index 424df70a5c30..105194de6e13 100644 --- a/fs/netfs/buffered_read.c +++ b/fs/netfs/buffered_read.c @@ -482,15 +482,14 @@ static int netfs_read_gaps(struct file *file, struct folio *folio) struct netfs_group *group = netfs_folio_group(folio); struct netfs_folio *finfo = netfs_folio_info(folio); struct netfs_inode *ctx = netfs_inode(mapping->host); - struct folio *sink = NULL; - struct bio_vec *bvec; + struct bio_vec *bvec = NULL; unsigned int from = finfo->dirty_offset; unsigned int to = from + finfo->dirty_len; - unsigned int off = 0, i = 0; + unsigned int off = 0; size_t flen = folio_size(folio); size_t nr_bvec = flen / PAGE_SIZE + 2; size_t part; - int ret; + int ret, i = 0, sink_from = -1, sink_to = -1; _enter("%lx", folio->index); @@ -515,24 +514,23 @@ static int netfs_read_gaps(struct file *file, struct folio *folio) if (!bvec) goto discard; - sink = folio_alloc(GFP_KERNEL, 0); - if (!sink) { - kfree(bvec); - goto discard; - } - trace_netfs_folio(folio, netfs_folio_trace_read_gaps); - rreq->direct_bv = bvec; - rreq->direct_bv_count = nr_bvec; if (from > 0) { bvec_set_folio(&bvec[i++], folio, from, 0); off = from; } + sink_from = i; while (off < to) { + struct folio *sink = folio_alloc(GFP_KERNEL, 0); + + if (!sink) + goto discard; part = min_t(size_t, to - off, PAGE_SIZE); - bvec_set_folio(&bvec[i++], sink, part, 0); + bvec_set_folio(&bvec[i], sink, part, 0); off += part; + sink_to = i; + i++; } if (to < flen) bvec_set_folio(&bvec[i++], folio, flen - to, to); @@ -553,8 +551,10 @@ static int netfs_read_gaps(struct file *file, struct folio *folio) folio_mark_uptodate(folio); } - if (sink) - folio_put(sink); + if (sink_to >= 0) + for (; sink_from <= sink_to; sink_from++) + folio_put(bvec_folio(&bvec[sink_from])); + kfree(bvec); folio_unlock(folio); netfs_put_request(rreq, netfs_rreq_trace_put_return); return ret < 0 ? ret : 0; @@ -563,6 +563,10 @@ discard: netfs_put_failed_request(rreq); alloc_error: folio_unlock(folio); + if (sink_to >= 0) + for (; sink_from <= sink_to; sink_from++) + folio_put(bvec_folio(&bvec[sink_from])); + kfree(bvec); return ret; } diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c index 7f6a3e912602..ad549daa9c79 100644 --- a/fs/netfs/objects.c +++ b/fs/netfs/objects.c @@ -34,7 +34,7 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping, rreq = mempool_alloc(mempool, gfp); } else { - rreq = mempool->alloc(gfp, mempool->pool_data); + rreq = mempool_alloc_noreserve(mempool, gfp); if (!rreq) return ERR_PTR(-ENOMEM); } @@ -214,7 +214,7 @@ struct netfs_io_subrequest *netfs_alloc_subrequest(struct netfs_io_request *rreq struct kmem_cache *cache = mempool->pool_data; if (rreq->gfp == GFP_KERNEL) - subreq = mempool->alloc(rreq->gfp, mempool->pool_data); + subreq = mempool_alloc_noreserve(mempool, rreq->gfp); else subreq = mempool_alloc(mempool, rreq->gfp); if (!subreq) diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c index 5cf22087d243..a94197ef0181 100644 --- a/fs/netfs/read_collect.c +++ b/fs/netfs/read_collect.c @@ -435,6 +435,11 @@ static void netfs_rreq_assess_single(struct netfs_io_request *rreq) netfs_single_mark_inode_dirty(rreq->inode); } + /* To do DIO, the cache has to round the size up, so we need to undo + * the rounding. + */ + rreq->transferred = min(rreq->transferred, rreq->i_size); + if (rreq->iocb) { rreq->iocb->ki_pos += rreq->transferred; if (rreq->iocb->ki_complete) { diff --git a/fs/netfs/rolling_buffer.c b/fs/netfs/rolling_buffer.c index 424e77a9a109..d30d5ef6d86e 100644 --- a/fs/netfs/rolling_buffer.c +++ b/fs/netfs/rolling_buffer.c @@ -29,7 +29,7 @@ struct folio_queue *netfs_folioq_alloc(unsigned int rreq_id, gfp_t gfp, struct folio_queue *fq; if (gfp == GFP_KERNEL) - fq = netfs_folioq_pool.alloc(gfp, netfs_folioq_pool.pool_data); + fq = mempool_alloc_noreserve(&netfs_folioq_pool, gfp); else fq = mempool_alloc(&netfs_folioq_pool, gfp); if (fq) { diff --git a/fs/ntfs3/inode.c b/fs/ntfs3/inode.c index 56b4f6469a28..4ac26c80bd34 100644 --- a/fs/ntfs3/inode.c +++ b/fs/ntfs3/inode.c @@ -1866,10 +1866,10 @@ int ntfs_create_inode(struct mnt_idmap *idmap, struct inode *dir, goto out6; /* - * Call 'd_instantiate' after inode->i_op is set + * Call 'd_instantiate_new' after inode->i_op is set * but before finish_open. */ - d_instantiate(dentry, inode); + d_instantiate_new(dentry, inode); /* Set original time. inode times (i_ctime) may be changed in ntfs_init_acl. */ inode_set_atime_to_ts(inode, ni->i_crtime); @@ -1917,9 +1917,6 @@ out1: if (!fnd) ni_unlock(dir_ni); - if (!err) - unlock_new_inode(inode); - return err; } diff --git a/fs/ocfs2/namei.c b/fs/ocfs2/namei.c index e9c7774ccf91..58c6061ed983 100644 --- a/fs/ocfs2/namei.c +++ b/fs/ocfs2/namei.c @@ -336,13 +336,8 @@ static int ocfs2_mknod(struct mnt_idmap *idmap, goto leave; /* calculate meta data/clusters for setting security and acl xattr */ - status = ocfs2_calc_xattr_init(dir, mode, &si, &want_clusters, - &xattr_credits, &want_meta, - &acl_state); - if (status < 0) { - mlog_errno(status); - goto leave; - } + ocfs2_calc_xattr_init(dir, mode, &si, &want_clusters, &xattr_credits, + &want_meta, &acl_state); /* Reserve a cluster if creating an extent based directory. */ if (S_ISDIR(mode) && !ocfs2_supports_inline_data(osb)) { diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 35bcbb0ff607..bfafe059bedf 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -635,12 +635,11 @@ int ocfs2_calc_security_init(struct inode *dir, return ret; } -int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, - struct ocfs2_security_xattr_info *si, - int *want_clusters, int *xattr_credits, - int *want_meta, struct ocfs2_acl_state *acl_state) +void ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, + struct ocfs2_security_xattr_info *si, + int *want_clusters, int *xattr_credits, + int *want_meta, struct ocfs2_acl_state *acl_state) { - int ret = 0; struct ocfs2_super *osb = OCFS2_SB(dir->i_sb); int s_size = 0, a_size = 0, acl_len = 0, new_clusters; @@ -662,7 +661,7 @@ int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, } if (!(s_size + a_size)) - return ret; + return; /* * The max space of security xattr taken inline is @@ -728,8 +727,6 @@ int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, } } } - - return ret; } static int ocfs2_xattr_extend_allocation(struct inode *inode, diff --git a/fs/ocfs2/xattr.h b/fs/ocfs2/xattr.h index 5e18513277f1..887cc1a18b1a 100644 --- a/fs/ocfs2/xattr.h +++ b/fs/ocfs2/xattr.h @@ -59,10 +59,10 @@ int ocfs2_calc_security_init(struct inode *, int *, int *, struct ocfs2_alloc_context **); struct ocfs2_acl_state; -int ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, - struct ocfs2_security_xattr_info *si, - int *want_clusters, int *xattr_credits, - int *want_meta, struct ocfs2_acl_state *acl_state); +void ocfs2_calc_xattr_init(struct inode *dir, umode_t mode, + struct ocfs2_security_xattr_info *si, + int *want_clusters, int *xattr_credits, + int *want_meta, struct ocfs2_acl_state *acl_state); /* * xattrs can live inside an inode, as part of an external xattr block, diff --git a/fs/overlayfs/overlayfs.h b/fs/overlayfs/overlayfs.h index e0d8c6152e9f..7f3558372c59 100644 --- a/fs/overlayfs/overlayfs.h +++ b/fs/overlayfs/overlayfs.h @@ -254,8 +254,10 @@ static inline struct dentry *ovl_do_mkdir(struct ovl_fs *ofs, { struct dentry *ret; + /* vfs_mkdir() drops @dentry on failure and may replace it on success */ + pr_debug("mkdir(%pd2, 0%o)\n", dentry, mode); ret = vfs_mkdir(ovl_upper_mnt_idmap(ofs), dir, dentry, mode, NULL); - pr_debug("mkdir(%pd2, 0%o) = %i\n", dentry, mode, PTR_ERR_OR_ZERO(ret)); + pr_debug("...mkdir = %i\n", PTR_ERR_OR_ZERO(ret)); return ret; } diff --git a/fs/smb/client/cached_dir.c b/fs/smb/client/cached_dir.c index 88d5e9a32f28..647fa26da4d2 100644 --- a/fs/smb/client/cached_dir.c +++ b/fs/smb/client/cached_dir.c @@ -8,6 +8,7 @@ #include <linux/namei.h> #include "cifsglob.h" #include "cifsproto.h" +#include "../common/smb2status.h" #include "cifs_debug.h" #include "smb2proto.h" #include "cached_dir.h" @@ -323,25 +324,37 @@ replay_again: rc = compound_send_recv(xid, ses, server, flags, 2, rqst, resp_buftype, rsp_iov); - if (rc) { - if (rc == -EREMCHG) { - tcon->need_reconnect = true; - pr_warn_once("server share %s deleted\n", - tcon->tree_name); - } - goto oshr_free; + if (rc == -EREMCHG) { + tcon->need_reconnect = true; + pr_warn_once("server share %s deleted\n", + tcon->tree_name); } - cfid->is_open = true; - spin_lock(&cfids->cfid_list_lock); + if (!rsp_iov[0].iov_base || rsp_iov[0].iov_len < sizeof(*o_rsp)) { + if (!rc) + rc = -EIO; + goto oshr_free; + } o_rsp = (struct smb2_create_rsp *)rsp_iov[0].iov_base; + if (o_rsp->hdr.Status != STATUS_SUCCESS) { + if (!rc) + rc = -EIO; + goto oshr_free; + } + oparms.fid->persistent_fid = o_rsp->PersistentFileId; oparms.fid->volatile_fid = o_rsp->VolatileFileId; #ifdef CONFIG_CIFS_DEBUG2 oparms.fid->mid = le64_to_cpu(o_rsp->hdr.MessageId); #endif /* CIFS_DEBUG2 */ + cfid->is_open = true; + atomic_inc(&tcon->num_remote_opens); + if (rc) + goto oshr_free; + + spin_lock(&cfids->cfid_list_lock); if (o_rsp->OplockLevel != SMB2_OPLOCK_LEVEL_LEASE) { spin_unlock(&cfids->cfid_list_lock); @@ -408,7 +421,6 @@ out: close_cached_dir(cfid); } else { *ret_cfid = cfid; - atomic_inc(&tcon->num_remote_opens); } kfree(utf16_path); diff --git a/fs/smb/client/dfs_cache.c b/fs/smb/client/dfs_cache.c index 29dfd7595941..26956e33037e 100644 --- a/fs/smb/client/dfs_cache.c +++ b/fs/smb/client/dfs_cache.c @@ -798,13 +798,13 @@ static int get_targets(struct cache_entry *ce, struct dfs_cache_tgt_list *tl) INIT_LIST_HEAD(head); list_for_each_entry(t, &ce->tlist, list) { - it = kzalloc_obj(*it, GFP_ATOMIC); + it = kzalloc_obj(*it, GFP_KERNEL); if (!it) { rc = -ENOMEM; goto err_free_it; } - it->it_name = kstrdup(t->name, GFP_ATOMIC); + it->it_name = kstrdup(t->name, GFP_KERNEL); if (!it->it_name) { kfree(it); rc = -ENOMEM; diff --git a/fs/smb/client/dir.c b/fs/smb/client/dir.c index 6fa6d48fdfd3..1a56fa4d0e89 100644 --- a/fs/smb/client/dir.c +++ b/fs/smb/client/dir.c @@ -199,7 +199,7 @@ static int __cifs_do_create(struct inode *dir, struct dentry *direntry, struct tcon_link *tlink, unsigned int oflags, umode_t mode, __u32 *oplock, struct cifs_fid *fid, struct cifs_open_info_data *buf, - struct inode **inode) + struct inode **inode, bool *opened) { int rc = -ENOENT; int create_options = CREATE_NOT_DIR; @@ -216,6 +216,7 @@ static int __cifs_do_create(struct inode *dir, struct dentry *direntry, __le32 lease_flags = 0; *inode = NULL; + *opened = false; *oplock = 0; if (tcon->ses->server->oplocks) *oplock = REQ_OPLOCK; @@ -232,6 +233,7 @@ static int __cifs_do_create(struct inode *dir, struct dentry *direntry, oflags, oplock, &fid->netfid, xid); switch (rc) { case 0: + *opened = true; if (newinode == NULL) { /* query inode info */ goto cifs_create_get_file_info; @@ -253,11 +255,9 @@ static int __cifs_do_create(struct inode *dir, struct dentry *direntry, /* * The server may allow us to open things like * FIFOs, but the client isn't set up to deal - * with that. If it's not a regular file, just - * close it and proceed as if it were a normal - * lookup. + * with that. Keep the handle until the caller + * can finish the lookup. */ - CIFSSMBClose(xid, tcon, fid->netfid); goto cifs_create_get_file_info; } /* success, no need to query */ @@ -384,6 +384,7 @@ retry_open: } return rc; } + *opened = true; if (rdwr_for_fscache == 2) cifs_invalidate_cache(dir, FSCACHE_INVAL_DIO_WRITE); @@ -475,7 +476,7 @@ cifs_create_set_dentry: return rc; out_err: - if (server->ops->close) + if (*opened && server->ops->close) server->ops->close(xid, tcon, fid); if (newinode) iput(newinode); @@ -487,7 +488,7 @@ static int cifs_do_create(struct inode *dir, struct dentry *direntry, unsigned int oflags, umode_t mode, __u32 *oplock, struct cifs_fid *fid, struct cifs_open_info_data *buf, - struct inode **inode) + struct inode **inode, bool *opened) { void *page = alloc_dentry_path(); const char *full_path; @@ -496,10 +497,11 @@ static int cifs_do_create(struct inode *dir, struct dentry *direntry, full_path = build_path_from_dentry(direntry, page); if (IS_ERR(full_path)) { rc = PTR_ERR(full_path); + *opened = false; } else { rc = __cifs_do_create(dir, direntry, full_path, xid, tlink, oflags, mode, oplock, - fid, buf, inode); + fid, buf, inode, opened); } free_dentry_path(page); return rc; @@ -529,6 +531,8 @@ int cifs_atomic_open(struct inode *dir, struct dentry *direntry, struct inode *inode; unsigned int xid; __u32 oplock; + bool is_regular; + bool opened; int rc; if (unlikely(cifs_forced_shutdown(cifs_sb))) @@ -581,12 +585,26 @@ int cifs_atomic_open(struct inode *dir, struct dentry *direntry, cifs_add_pending_open(&fid, tlink, &open); rc = cifs_do_create(dir, direntry, xid, tlink, oflags, mode, - &oplock, &fid, &buf, &inode); + &oplock, &fid, &buf, &inode, &opened); if (rc) { cifs_del_pending_open(&open); goto out; } + is_regular = S_ISREG(inode->i_mode); + if (!is_regular || !opened) { + if (opened && server->ops->close) + server->ops->close(xid, tcon, &fid); + cifs_del_pending_open(&open); + if (S_ISLNK(inode->i_mode) && + (oflags & (O_NOFOLLOW | __O_REGULAR)) == + (O_NOFOLLOW | __O_REGULAR) && !(oflags & O_EXCL)) { + iput(inode); + rc = -ELOOP; + goto out; + } + } + if (d_in_lookup(direntry)) { alias = d_splice_alias(inode, direntry); if (!IS_ERR_OR_NULL(alias)) @@ -595,9 +613,15 @@ int cifs_atomic_open(struct inode *dir, struct dentry *direntry, d_instantiate(direntry, inode); } - if ((oflags & (O_CREAT | O_EXCL)) == (O_CREAT | O_EXCL)) + if (is_regular && opened && + (oflags & (O_CREAT | O_EXCL)) == (O_CREAT | O_EXCL)) file->f_mode |= FMODE_CREATED; + if (!is_regular || !opened) { + rc = finish_no_open(file, NULL); + goto out; + } + rc = finish_open(file, direntry, generic_file_open); if (rc) { if (server->ops->close) @@ -660,6 +684,7 @@ int cifs_create(struct mnt_idmap *idmap, struct inode *dir, struct inode *inode; struct cifs_fid fid; __u32 oplock; + bool opened; struct cifs_open_info_data buf = {}; cifs_dbg(FYI, "cifs_create parent inode = 0x%p name is: %pd and dentry = 0x%p\n", @@ -682,10 +707,10 @@ int cifs_create(struct mnt_idmap *idmap, struct inode *dir, server->ops->new_lease_key(&fid); rc = cifs_do_create(dir, direntry, xid, tlink, oflags, - mode, &oplock, &fid, &buf, &inode); + mode, &oplock, &fid, &buf, &inode, &opened); if (!rc) { d_instantiate(direntry, inode); - if (server->ops->close) + if (opened && server->ops->close) server->ops->close(xid, tcon, &fid); } @@ -1078,6 +1103,7 @@ int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, struct inode *inode; unsigned int xid; __u32 oplock; + bool opened; int namelen; int rc; @@ -1116,7 +1142,7 @@ int cifs_tmpfile(struct mnt_idmap *idmap, struct inode *dir, namelen = scnprintf(name, namesize, CIFS_TMPNAME_PREFIX "%x", atomic_inc_return(&cifs_tmpcounter)); rc = __cifs_do_create(dir, dentry, path, xid, tlink, oflags, - mode, &oplock, &fid, NULL, &inode); + mode, &oplock, &fid, NULL, &inode, &opened); if (!rc) { rc = d_mark_tmpfile_name(file, &QSTR_LEN(name, namelen)); if (rc) { diff --git a/fs/smb/client/smb2inode.c b/fs/smb/client/smb2inode.c index 13fe8e3b48f3..f46a62eae659 100644 --- a/fs/smb/client/smb2inode.c +++ b/fs/smb/client/smb2inode.c @@ -598,8 +598,10 @@ finished: /* smb2_parse_contexts() fills idata->fi.IndexNumber */ rc = smb2_parse_contexts(server, &rsp_iov[0], &oparms->fid->epoch, oparms->fid->lease_key, &oplock, &idata->fi, NULL); - if (rc) + if (rc) { cifs_dbg(VFS, "rc: %d parsing context of compound op\n", rc); + tmp_rc = rc; + } } for (i = 0; i < num_cmds; i++) { @@ -1121,7 +1123,7 @@ smb2_unlink(const unsigned int xid, struct cifs_tcon *tcon, const char *name, struct kvec close_iov; int resp_buftype[2]; struct cifs_fid fid; - int flags = 0; + int flags = CIFS_CP_CREATE_CLOSE_OP; __u8 oplock; int rc; diff --git a/fs/smb/client/smb2misc.c b/fs/smb/client/smb2misc.c index 0cfe60ae42c3..5e5cf92d1eb3 100644 --- a/fs/smb/client/smb2misc.c +++ b/fs/smb/client/smb2misc.c @@ -834,7 +834,8 @@ smb2_cancelled_close_fid(struct work_struct *work) */ static int __smb2_handle_cancelled_cmd(struct cifs_tcon *tcon, __u16 cmd, __u64 mid, - __u64 persistent_fid, __u64 volatile_fid) + __u64 persistent_fid, __u64 volatile_fid, + bool account_remote_open) { struct close_cancelled_open *cancelled; @@ -848,6 +849,8 @@ __smb2_handle_cancelled_cmd(struct cifs_tcon *tcon, __u16 cmd, __u64 mid, cancelled->cmd = cmd; cancelled->mid = mid; INIT_WORK(&cancelled->work, smb2_cancelled_close_fid); + if (account_remote_open) + atomic_inc(&tcon->num_remote_opens); WARN_ON(queue_work(cifsiod_wq, &cancelled->work) == false); return 0; @@ -884,7 +887,7 @@ smb2_handle_cancelled_close(struct cifs_tcon *tcon, __u64 persistent_fid, spin_unlock(&tcon->tc_lock); rc = __smb2_handle_cancelled_cmd(tcon, SMB2_CLOSE_HE, 0, - persistent_fid, volatile_fid); + persistent_fid, volatile_fid, false); if (rc) cifs_put_tcon(tcon, netfs_trace_tcon_ref_put_cancelled_close); @@ -912,7 +915,7 @@ smb2_handle_cancelled_mid(struct mid_q_entry *mid, struct TCP_Server_Info *serve le16_to_cpu(hdr->Command), le64_to_cpu(hdr->MessageId), rsp->PersistentFileId, - rsp->VolatileFileId); + rsp->VolatileFileId, true); if (rc) cifs_put_tcon(tcon, netfs_trace_tcon_ref_put_cancelled_mid); diff --git a/fs/smb/client/smb2ops.c b/fs/smb/client/smb2ops.c index 3464470d3297..aa142420dae2 100644 --- a/fs/smb/client/smb2ops.c +++ b/fs/smb/client/smb2ops.c @@ -4572,25 +4572,37 @@ smb3_create_lease_buf(u8 *lease_key, u8 oplock, u8 *parent_lease_key, __le32 fla static __u8 smb2_parse_lease_buf(void *buf, __u16 *epoch, char *lease_key) { - struct create_lease *lc = (struct create_lease *)buf; + struct create_context *cc = buf; + struct lease_context lc; *epoch = 0; /* not used */ - if (lc->lcontext.LeaseFlags & SMB2_LEASE_FLAG_BREAK_IN_PROGRESS_LE) + if (le32_to_cpu(cc->DataLength) != sizeof(lc)) + return 0; + + memcpy(&lc, (u8 *)cc + le16_to_cpu(cc->DataOffset), sizeof(lc)); + if (lc.LeaseFlags & SMB2_LEASE_FLAG_BREAK_IN_PROGRESS_LE) return SMB2_OPLOCK_LEVEL_NOCHANGE; - return le32_to_cpu(lc->lcontext.LeaseState); + return le32_to_cpu(lc.LeaseState); } static __u8 smb3_parse_lease_buf(void *buf, __u16 *epoch, char *lease_key) { - struct create_lease_v2 *lc = (struct create_lease_v2 *)buf; + struct create_context *cc = buf; + struct lease_context_v2 lc; + + if (le32_to_cpu(cc->DataLength) != sizeof(lc)) { + *epoch = 0; + return 0; + } - *epoch = le16_to_cpu(lc->lcontext.Epoch); - if (lc->lcontext.LeaseFlags & SMB2_LEASE_FLAG_BREAK_IN_PROGRESS_LE) + memcpy(&lc, (u8 *)cc + le16_to_cpu(cc->DataOffset), sizeof(lc)); + *epoch = le16_to_cpu(lc.Epoch); + if (lc.LeaseFlags & SMB2_LEASE_FLAG_BREAK_IN_PROGRESS_LE) return SMB2_OPLOCK_LEVEL_NOCHANGE; if (lease_key) - memcpy(lease_key, &lc->lcontext.LeaseKey, SMB2_LEASE_KEY_SIZE); - return le32_to_cpu(lc->lcontext.LeaseState); + memcpy(lease_key, lc.LeaseKey, SMB2_LEASE_KEY_SIZE); + return le32_to_cpu(lc.LeaseState); } static unsigned int diff --git a/fs/smb/client/smb2pdu.c b/fs/smb/client/smb2pdu.c index 880ce12f50c4..3d7ead36d1a0 100644 --- a/fs/smb/client/smb2pdu.c +++ b/fs/smb/client/smb2pdu.c @@ -2379,23 +2379,32 @@ create_reconnect_durable_buf(struct cifs_fid *fid) static void parse_query_id_ctxt(struct create_context *cc, struct smb2_file_all_info *buf) { - struct create_disk_id_rsp *pdisk_id = (struct create_disk_id_rsp *)cc; + u16 doff = le16_to_cpu(cc->DataOffset); + u32 dlen = le32_to_cpu(cc->DataLength); + u8 *beg; - cifs_dbg(FYI, "parse query id context 0x%llx 0x%llx\n", - pdisk_id->DiskFileId, pdisk_id->VolumeId); - buf->IndexNumber = pdisk_id->DiskFileId; + if (dlen < sizeof(__le64)) + return; + + beg = (u8 *)cc + doff; + memcpy(&buf->IndexNumber, beg, sizeof(__le64)); + cifs_dbg(FYI, "parse query id context 0x%llx\n", + le64_to_cpu(buf->IndexNumber)); } static void parse_posix_ctxt(struct create_context *cc, struct smb2_file_all_info *info, struct create_posix_rsp *posix) { - int sid_len; u8 *beg = (u8 *)cc + le16_to_cpu(cc->DataOffset); - u8 *end = beg + le32_to_cpu(cc->DataLength); + u32 dlen = le32_to_cpu(cc->DataLength); + u8 *end = beg + dlen; + int sid_len; u8 *sid; memset(posix, 0, sizeof(*posix)); + if (dlen < 3 * sizeof(__le32)) + return; posix->nlink = get_unaligned_le32(beg); posix->reparse_tag = get_unaligned_le32(beg + 4); @@ -2431,6 +2440,7 @@ int smb2_parse_contexts(struct TCP_Server_Info *server, struct smb2_create_rsp *rsp = rsp_iov->iov_base; struct create_context *cc; size_t rem, off, len; + size_t cc_len; size_t doff, dlen; size_t noff, nlen; char *name; @@ -2453,29 +2463,41 @@ int smb2_parse_contexts(struct TCP_Server_Info *server, buf->IndexNumber = 0; while (rem >= sizeof(*cc)) { + off = le32_to_cpu(cc->Next); + if (off) { + if ((off & 0x7) || off >= rem || off < sizeof(*cc)) + return -EINVAL; + cc_len = off; + } else { + cc_len = rem; + } + doff = le16_to_cpu(cc->DataOffset); dlen = le32_to_cpu(cc->DataLength); - if (check_add_overflow(doff, dlen, &len) || len > rem) + if (doff < sizeof(*cc) || + check_add_overflow(doff, dlen, &len) || len > cc_len) return -EINVAL; noff = le16_to_cpu(cc->NameOffset); nlen = le16_to_cpu(cc->NameLength); - if (noff + nlen > doff) + if (noff < sizeof(*cc) || + check_add_overflow(noff, nlen, &len) || len > cc_len || + (dlen && len > doff)) return -EINVAL; name = (char *)cc + noff; switch (nlen) { case 4: - if (!strncmp(name, SMB2_CREATE_REQUEST_LEASE, 4)) { + if (dlen && !strncmp(name, SMB2_CREATE_REQUEST_LEASE, 4)) { *oplock = server->ops->parse_lease_buf(cc, epoch, lease_key); - } else if (buf && + } else if (dlen && buf && !strncmp(name, SMB2_CREATE_QUERY_ON_DISK_ID, 4)) { parse_query_id_ctxt(cc, buf); } break; case 16: - if (posix && !memcmp(name, smb3_create_tag_posix, 16)) + if (dlen && posix && !memcmp(name, smb3_create_tag_posix, 16)) parse_posix_ctxt(cc, buf, posix); break; default: @@ -2487,13 +2509,18 @@ int smb2_parse_contexts(struct TCP_Server_Info *server, } off = le32_to_cpu(cc->Next); - if (!off) + if (!off) { + rem = 0; break; + } if (check_sub_overflow(rem, off, &rem)) return -EINVAL; cc = (struct create_context *)((u8 *)cc + off); } + if (rem) + return -EINVAL; + if (rsp->OplockLevel != SMB2_OPLOCK_LEVEL_LEASE) *oplock = rsp->OplockLevel; @@ -3389,6 +3416,9 @@ replay_again: rc = smb2_parse_contexts(server, &rsp_iov, &oparms->fid->epoch, oparms->fid->lease_key, oplock, file_info, posix); + if (rc) + SMB2_close(xid, tcon, oparms->fid->persistent_fid, + oparms->fid->volatile_fid); trace_smb3_open_done(xid, rsp->PersistentFileId, tcon->tid, ses->Suid, oparms->create_options, oparms->desired_access, diff --git a/fs/smb/client/smb2pdu.h b/fs/smb/client/smb2pdu.h index ab6c667bebc0..b37a1e3941a1 100644 --- a/fs/smb/client/smb2pdu.h +++ b/fs/smb/client/smb2pdu.h @@ -224,8 +224,7 @@ struct smb2_file_id_extd_directory_info { extern char smb2_padding[7]; /* - * See POSIX-SMB2 2.2.14.2.16 - * Link: https://gitlab.com/samba-team/smb3-posix-spec/-/blob/master/smb3_posix_extensions.md + * See POSIX-SMB2 2.1.3.2.1 */ struct create_posix_rsp { u32 nlink; @@ -238,6 +237,7 @@ struct create_posix_rsp { #define SMB2_QUERY_DIRECTORY_IOV_SIZE 2 /* + * See POSIX-FSCC 2.2.1 * SMB2-only POSIX info level for query dir * * See posix_info_sid_size(), posix_info_extra_size() and @@ -256,13 +256,17 @@ struct smb2_posix_info { __le64 Inode; __le32 DeviceId; __le32 Zero; - /* beginning of POSIX Create Context Response */ + /* + * Beginning of POSIX Create Context Response + * See POSIX-SMB2 2.1.3.2.1 + */ __le32 HardLinks; __le32 ReparseTag; __le32 Mode; /* * var sized owner SID * var sized group SID + * End of POSIX Create Context Response * le32 filenamelength * u8 filename[] */ diff --git a/fs/smb/client/transport.c b/fs/smb/client/transport.c index e266859818a4..6e21b5f8754a 100644 --- a/fs/smb/client/transport.c +++ b/fs/smb/client/transport.c @@ -805,6 +805,18 @@ cifs_cancelled_callback(struct TCP_Server_Info *server, struct mid_q_entry *mid) release_mid(server, mid); } +static void +cifs_mark_compound_mids_cancelled(struct mid_q_entry **mid, int count) +{ + int i; + + for (i = 0; i < count; i++) { + spin_lock(&mid[i]->mid_lock); + mid[i]->wait_cancelled = true; + spin_unlock(&mid[i]->mid_lock); + } +} + /* * cifs_pick_channel - pick an eligible channel for network operations * @@ -865,6 +877,7 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, int *resp_buf_type, struct kvec *resp_iov) { int i, j, optype, rc = 0; + int num_processed = 0; struct mid_q_entry *mid[MAX_COMPOUND]; bool cancelled_mid[MAX_COMPOUND] = {false}; struct cifs_credits credits[MAX_COMPOUND] = { @@ -965,6 +978,10 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, if (rc < 0) { revert_current_mid(server, num_rqst); server->sequence_number -= 2; + for (i = 0; i < num_rqst; i++) { + delete_mid(server, mid[i]); + cancelled_mid[i] = true; + } } cifs_server_unlock(server); @@ -1011,6 +1028,14 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, break; } if (rc != 0) { + /* + * A completed CREATE earlier in the compound chain may have + * opened a remote handle even though a later wait was + * interrupted. Mark it cancelled so __release_mid() invokes + * the existing unmatched-open cleanup. + */ + cifs_mark_compound_mids_cancelled(mid, i); + for (; i < num_rqst; i++) { cifs_server_dbg(FYI, "Cancelling wait for mid %llu cmd: %d\n", mid[i]->mid, le16_to_cpu(mid[i]->command)); @@ -1033,6 +1058,14 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, rc = cifs_sync_mid_result(mid[i], server); if (rc != 0) { + /* + * A previous CREATE may have completed before this + * response failed. Mark it cancelled so its remote + * handle is closed when the mid is released. + */ + cifs_mark_compound_mids_cancelled(mid, i); + /* Keep their response buffers for cancelled-mid cleanup. */ + num_processed = 0; /* mark this mid as cancelled to not free it below */ cancelled_mid[i] = true; goto out; @@ -1042,13 +1075,24 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, mid[i]->mid_state != MID_RESPONSE_READY) { rc = smb_EIO1(smb_eio_trace_rx_mid_unready, mid[i]->mid_state); cifs_dbg(FYI, "Bad MID state?\n"); + cifs_mark_compound_mids_cancelled(mid, i); + num_processed = 0; goto out; } rc = server->ops->check_receive(mid[i], server, flags & CIFS_LOG_ERROR); + num_processed = i + 1; + } - if (resp_iov) { +out: + /* + * Delay moving response buffers out of their mids until response + * synchronization completes. This lets cancelled-mid cleanup inspect + * an earlier CREATE response if a later MID fails. + */ + if (resp_iov) { + for (i = 0; i < num_processed; i++) { buf = (char *)mid[i]->resp_buf; resp_iov[i].iov_base = buf; resp_iov[i].iov_len = mid[i]->resp_buf_size; @@ -1067,21 +1111,22 @@ compound_send_recv(const unsigned int xid, struct cifs_ses *ses, /* * Compounding is never used during session establish. */ - spin_lock(&ses->ses_lock); - if ((ses->ses_status == SES_NEW) || (optype & CIFS_NEG_OP) || (optype & CIFS_SESS_OP)) { - struct kvec iov = { - .iov_base = resp_iov[0].iov_base, - .iov_len = resp_iov[0].iov_len - }; - spin_unlock(&ses->ses_lock); - cifs_server_lock(server); - smb311_update_preauth_hash(ses, server, &iov, 1); - cifs_server_unlock(server); + if (num_processed == num_rqst) { spin_lock(&ses->ses_lock); + if ((ses->ses_status == SES_NEW) || (optype & CIFS_NEG_OP) || (optype & CIFS_SESS_OP)) { + struct kvec iov = { + .iov_base = resp_iov[0].iov_base, + .iov_len = resp_iov[0].iov_len + }; + spin_unlock(&ses->ses_lock); + cifs_server_lock(server); + smb311_update_preauth_hash(ses, server, &iov, 1); + cifs_server_unlock(server); + spin_lock(&ses->ses_lock); + } + spin_unlock(&ses->ses_lock); } - spin_unlock(&ses->ses_lock); -out: /* * This will dequeue all mids. After this it is important that the * demultiplex_thread will not process any of these mids any further. diff --git a/fs/squashfs/xz_wrapper.c b/fs/squashfs/xz_wrapper.c index 0a4ff3ec9c8c..6610af241449 100644 --- a/fs/squashfs/xz_wrapper.c +++ b/fs/squashfs/xz_wrapper.c @@ -57,10 +57,10 @@ static void *squashfs_xz_comp_opts(struct squashfs_sb_info *msblk, opts->dict_size = le32_to_cpu(comp_opts->dictionary_size); - /* the dictionary size should be 2^n or 2^n+2^(n+1) */ + /* the dictionary size should be positive and 2^n or 2^n+2^(n+1) */ n = ffs(opts->dict_size) - 1; - if (opts->dict_size != (1 << n) && opts->dict_size != (1 << n) + - (1 << (n + 1))) { + if (opts->dict_size <= 0 || (opts->dict_size != (1 << n) && + opts->dict_size != (1 << n) + (1 << (n + 1)))) { err = -EIO; goto out; } diff --git a/fs/super.c b/fs/super.c index 9d4025213521..1d5ccf540a9b 100644 --- a/fs/super.c +++ b/fs/super.c @@ -419,15 +419,19 @@ fail: void put_super(struct super_block *s) { if (refcount_dec_and_test(&s->s_passive)) { + struct file_system_type *type = s->s_type; spin_lock(&sb_lock); list_del_init(&s->s_list); + hlist_del_init(&s->s_instances); spin_unlock(&sb_lock); WARN_ON(s->s_dentry_lru.node); WARN_ON(s->s_inode_lru.node); WARN_ON(s->s_mounts); call_rcu(&s->rcu, destroy_super_rcu); + /* The unlink above may touch type->fs_supers, so drop it last. */ + put_filesystem(type); } } @@ -544,17 +548,6 @@ static void kill_super_notify(struct super_block *sb) if (sb->s_flags & SB_DEAD) return; - /* - * Remove it from @fs_supers so it isn't found by new - * sget_fc() walkers anymore. Any concurrent mounter still - * managing to grab a temporary reference is guaranteed to - * already see SB_DYING and will wait until we notify them about - * SB_DEAD. - */ - spin_lock(&sb_lock); - hlist_del_init(&sb->s_instances); - spin_unlock(&sb_lock); - /* Drop sget_fc()'s claim; a never-registered entry stays with the sb. */ if (sb->s_super_dev->sd_dev) { super_dev_put(sb->s_super_dev); @@ -563,11 +556,15 @@ static void kill_super_notify(struct super_block *sb) /* * Let concurrent mounts know that this thing is really dead. - * We don't need @sb->s_umount here as every concurrent caller - * will see SB_DYING and either discard the superblock or wait - * for SB_DEAD. + * sget_fc() skips SB_DEAD superblocks and calls test() under + * sb_lock, so set it under sb_lock: once we return no test() + * runs on this superblock anymore and none will start. Everyone + * else already saw SB_DYING and either discarded the superblock + * or waits for SB_DEAD. */ + spin_lock(&sb_lock); super_wake(sb, SB_DEAD); + spin_unlock(&sb_lock); } /** @@ -594,7 +591,6 @@ void deactivate_locked_super(struct super_block *s) list_lru_destroy(&s->s_dentry_lru); list_lru_destroy(&s->s_inode_lru); - put_filesystem(fs); put_super(s); } else { super_unlock_excl(s); @@ -781,12 +777,12 @@ void generic_shutdown_super(struct super_block *sb) } /* * Broadcast to everyone that grabbed a temporary reference to this - * superblock before we removed it from @fs_supers that the superblock - * is dying. Every walker of @fs_supers outside of sget_fc() will now - * discard this superblock and treat it as dead. + * superblock that it is dying. Every walker of @fs_supers outside + * of sget_fc() will now discard this superblock and treat it as + * dead. * - * We leave the superblock on @fs_supers so it can be found by - * sget_fc() until we passed sb->kill_sb(). + * sget_fc() keeps finding the superblock until SB_DEAD is set, so + * a concurrent mounter waits until we passed sb->kill_sb(). */ super_wake(sb, SB_DYING); super_unlock_excl(sb); @@ -865,6 +861,9 @@ retry: spin_lock(&sb_lock); if (test) { hlist_for_each_entry(old, &fc->fs_type->fs_supers, s_instances) { + /* Only unlinked at the last passive reference. */ + if (super_flags(old, SB_DEAD)) + continue; if (test(old, fc)) goto share_extant_sb; } diff --git a/fs/xfs/libxfs/xfs_ag.h b/fs/xfs/libxfs/xfs_ag.h index fd22fe598931..ee636b66a72f 100644 --- a/fs/xfs/libxfs/xfs_ag.h +++ b/fs/xfs/libxfs/xfs_ag.h @@ -207,7 +207,7 @@ xfs_perag_next( } /* - * Per-ag geometry infomation and validation + * Per-ag geometry information and validation */ xfs_agblock_t xfs_ag_block_count(struct xfs_mount *mp, xfs_agnumber_t agno); void xfs_agino_range(struct xfs_mount *mp, xfs_agnumber_t agno, diff --git a/fs/xfs/libxfs/xfs_alloc.c b/fs/xfs/libxfs/xfs_alloc.c index d99602bcc16f..f762dcce8d13 100644 --- a/fs/xfs/libxfs/xfs_alloc.c +++ b/fs/xfs/libxfs/xfs_alloc.c @@ -3487,7 +3487,7 @@ xfs_alloc_read_agf( } /* - * Pre-proces allocation arguments to set initial state that we don't require + * Pre-process allocation arguments to set initial state that we don't require * callers to set up correctly, as well as bounds check the allocation args * that are set up. */ @@ -3608,7 +3608,7 @@ xfs_alloc_vextent_finish( * ABBA AGF deadlocks because a future allocation attempt in this * transaction may attempt to lock a lower number AGF. * - * We can't release the AGF until the transaction is commited, so at + * We can't release the AGF until the transaction is committed, so at * this point we must update the "first allocation" tracker to point at * this AG if the tracker is empty or points to a lower AG. This allows * the next allocation attempt to be modified appropriately to avoid diff --git a/fs/xfs/libxfs/xfs_attr_leaf.c b/fs/xfs/libxfs/xfs_attr_leaf.c index b6288395f853..2c80f4fd0b78 100644 --- a/fs/xfs/libxfs/xfs_attr_leaf.c +++ b/fs/xfs/libxfs/xfs_attr_leaf.c @@ -1715,7 +1715,7 @@ xfs_attr3_leaf_add_work( /* * This freemap entry starts at the old end of the * leaf entry array, so we need to adjust its base - * upward to accomodate the larger array. + * upward to accommodate the larger array. */ diff = sizeof(struct xfs_attr_leaf_entry); } else if (ichdr->freemap[i].size > 0 && diff --git a/fs/xfs/libxfs/xfs_errortag.h b/fs/xfs/libxfs/xfs_errortag.h index 6de207fed2d8..f0c83f1f0b3b 100644 --- a/fs/xfs/libxfs/xfs_errortag.h +++ b/fs/xfs/libxfs/xfs_errortag.h @@ -83,7 +83,7 @@ #define XFS_RANDOM_DEFAULT 100 /* - * Table of errror injection knobs. The parameters to the XFS_ERRTAG macro are: + * Table of error injection knobs. The parameters to the XFS_ERRTAG macro are: * 1. The XFS_ERRTAG_ flag but without the prefix; * 2. The name of the sysfs knob; and * 3. The default value for the knob. diff --git a/fs/xfs/libxfs/xfs_exchmaps.c b/fs/xfs/libxfs/xfs_exchmaps.c index 49eda8d0994d..6a66b6075e0a 100644 --- a/fs/xfs/libxfs/xfs_exchmaps.c +++ b/fs/xfs/libxfs/xfs_exchmaps.c @@ -395,7 +395,7 @@ xfs_exchmaps_one_step( /* * Re-add both mappings. We exchange the file offsets between the two * maps and add the opposite map, which has the effect of filling the - * logical offsets we just unmapped, but with with the physical mapping + * logical offsets we just unmapped, but with the physical mapping * information exchanged. */ swap(irec1->br_startoff, irec2->br_startoff); @@ -969,16 +969,6 @@ xmi_can_exchange_reflink_flags( if (req->flags & XFS_EXCHMAPS_INO1_WRITTEN) return false; - /* - * The INO1_WRITTEN optimization can skip exchanging hole and - * unwritten mappings, which means we cannot guarantee that all - * shared extents actually moved to the other file. Clearing the - * reflink flag of an inode that still holds shared extents breaks - * the CoW write path, so refuse to exchange the flags in that case. - */ - if (req->flags & XFS_EXCHMAPS_INO1_WRITTEN) - return false; - if (hweight32(reflink_state) != 1) return false; if (req->startoff1 != 0 || req->startoff2 != 0) diff --git a/fs/xfs/libxfs/xfs_format.h b/fs/xfs/libxfs/xfs_format.h index dd0ed046fbe9..1a7a7e60a170 100644 --- a/fs/xfs/libxfs/xfs_format.h +++ b/fs/xfs/libxfs/xfs_format.h @@ -1051,7 +1051,7 @@ enum xfs_dinode_fmt { * block is 1KB in size. * * With XFS_MAX_EXTCNT_DATA_FORK_SMALL representing maximum extent count and - * with 1KB sized blocks, a file can reach upto, + * with 1KB sized blocks, a file can reach up to, * 1KB * (2^31) = 2TB * * This is much larger than the theoretical maximum size of a directory diff --git a/fs/xfs/libxfs/xfs_inode_buf.c b/fs/xfs/libxfs/xfs_inode_buf.c index e4c3f7b24e95..0340e2189921 100644 --- a/fs/xfs/libxfs/xfs_inode_buf.c +++ b/fs/xfs/libxfs/xfs_inode_buf.c @@ -626,7 +626,7 @@ xfs_dinode_verify( * have di_nlink track the link count, even if the actual filesystem * only supported V1 inodes (i.e. di_onlink). When writing out the * ondisk inode, it would set both the ondisk di_nlink and di_onlink to - * the the incore di_nlink value, which is why we cannot check for + * the incore di_nlink value, which is why we cannot check for * di_nlink==0 on a V1 inode. V2/3 inodes would get written out with * di_onlink==0, so we can check that. */ diff --git a/fs/xfs/libxfs/xfs_metafile.c b/fs/xfs/libxfs/xfs_metafile.c index 71f004e9dc64..1f54d39003c2 100644 --- a/fs/xfs/libxfs/xfs_metafile.c +++ b/fs/xfs/libxfs/xfs_metafile.c @@ -297,14 +297,14 @@ xfs_metafile_resv_init( goto out_unlock; /* - * Space taken by the per-AG metadata btrees are accounted on-disk as - * used space. We therefore only hide the space that is reserved but - * not used by the trees. + * Space taken by metadata btrees are accounted on-disk as used space. + * We therefore only hide the space that is reserved but not used by + * the trees. */ if (used > target) target = used; else if (target > dblocks_avail) - target = dblocks_avail; + target = max(dblocks_avail, used); hidden_space = target - used; error = xfs_dec_fdblocks(mp, hidden_space, true); diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c index e2950dbe2068..dcc89b8e149b 100644 --- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c +++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c @@ -617,7 +617,7 @@ xfs_rtrefcountbt_from_disk( fpp = xfs_rtrefcount_droot_ptr_addr(dblock, 1, maxrecs); tpp = xfs_rtrefcount_broot_ptr_addr(mp, rblock, 1, rblocklen); numrecs = be16_to_cpu(dblock->bb_numrecs); - memcpy(tkp, fkp, 2 * sizeof(*fkp) * numrecs); + memcpy(tkp, fkp, sizeof(*fkp) * numrecs); memcpy(tpp, fpp, sizeof(*fpp) * numrecs); } else { frp = xfs_rtrefcount_droot_rec_addr(dblock, 1); @@ -703,7 +703,7 @@ xfs_rtrefcountbt_to_disk( fpp = xfs_rtrefcount_broot_ptr_addr(mp, rblock, 1, rblocklen); tpp = xfs_rtrefcount_droot_ptr_addr(dblock, 1, maxrecs); numrecs = be16_to_cpu(rblock->bb_numrecs); - memcpy(tkp, fkp, 2 * sizeof(*fkp) * numrecs); + memcpy(tkp, fkp, sizeof(*fkp) * numrecs); memcpy(tpp, fpp, sizeof(*fpp) * numrecs); } else { frp = xfs_rtrefcount_rec_addr(rblock, 1); diff --git a/fs/xfs/scrub/agheader_repair.c b/fs/xfs/scrub/agheader_repair.c index a66b611588c4..ff1b4b361cf2 100644 --- a/fs/xfs/scrub/agheader_repair.c +++ b/fs/xfs/scrub/agheader_repair.c @@ -1369,7 +1369,7 @@ xrep_iunlink_mark_ondisk( /* * Walk an iunlink bucket's inode list. For each inode that should be on this - * chain, clear its entry in in iunlink_bmp because it's ok and we don't need + * chain, clear its entry in iunlink_bmp because it's ok and we don't need * to touch it further. */ STATIC int diff --git a/fs/xfs/scrub/alloc_repair.c b/fs/xfs/scrub/alloc_repair.c index 95e318e4f3a6..2398e3819597 100644 --- a/fs/xfs/scrub/alloc_repair.c +++ b/fs/xfs/scrub/alloc_repair.c @@ -338,7 +338,7 @@ xrep_cntbt_extent_cmp( } /* - * Sort the free extents by length so so that we can put the records into the + * Sort the free extents by length so that we can put the records into the * cntbt in the correct order. Don't let userspace kill us if we're resorting * after allocating btree blocks. */ diff --git a/fs/xfs/scrub/bitmap.c b/fs/xfs/scrub/bitmap.c index c7fa908d92b2..08f216d26ec6 100644 --- a/fs/xfs/scrub/bitmap.c +++ b/fs/xfs/scrub/bitmap.c @@ -122,8 +122,8 @@ xbitmap64_set( uint64_t start, uint64_t len) { - struct xbitmap64_node *left; - struct xbitmap64_node *right; + struct xbitmap64_node *left = NULL; + struct xbitmap64_node *right = NULL; uint64_t last = start + len - 1; int error; @@ -131,6 +131,7 @@ xbitmap64_set( left = xbitmap64_tree_iter_first(&bitmap->xb_root, start, last); if (left && left->bn_start <= start && left->bn_last >= last) return 0; + left = NULL; /* Clear out everything in the range we want to set. */ error = xbitmap64_clear(bitmap, start, len); @@ -138,11 +139,15 @@ xbitmap64_set( return error; /* Do we have a left-adjacent extent? */ - left = xbitmap64_tree_iter_first(&bitmap->xb_root, start - 1, start - 1); + if (start > 0) + left = xbitmap64_tree_iter_first(&bitmap->xb_root, start - 1, + start - 1); ASSERT(!left || left->bn_last + 1 == start); /* Do we have a right-adjacent extent? */ - right = xbitmap64_tree_iter_first(&bitmap->xb_root, last + 1, last + 1); + if (last < U64_MAX) + right = xbitmap64_tree_iter_first(&bitmap->xb_root, last + 1, + last + 1); ASSERT(!right || right->bn_start == last + 1); if (left && right) { @@ -397,8 +402,8 @@ xbitmap32_set( uint32_t start, uint32_t len) { - struct xbitmap32_node *left; - struct xbitmap32_node *right; + struct xbitmap32_node *left = NULL; + struct xbitmap32_node *right = NULL; uint32_t last = start + len - 1; int error; @@ -406,6 +411,7 @@ xbitmap32_set( left = xbitmap32_tree_iter_first(&bitmap->xb_root, start, last); if (left && left->bn_start <= start && left->bn_last >= last) return 0; + left = NULL; /* Clear out everything in the range we want to set. */ error = xbitmap32_clear(bitmap, start, len); @@ -413,11 +419,15 @@ xbitmap32_set( return error; /* Do we have a left-adjacent extent? */ - left = xbitmap32_tree_iter_first(&bitmap->xb_root, start - 1, start - 1); + if (start > 0) + left = xbitmap32_tree_iter_first(&bitmap->xb_root, start - 1, + start - 1); ASSERT(!left || left->bn_last + 1 == start); /* Do we have a right-adjacent extent? */ - right = xbitmap32_tree_iter_first(&bitmap->xb_root, last + 1, last + 1); + if (last < U32_MAX) + right = xbitmap32_tree_iter_first(&bitmap->xb_root, last + 1, + last + 1); ASSERT(!right || right->bn_start == last + 1); if (left && right) { diff --git a/fs/xfs/scrub/dir.c b/fs/xfs/scrub/dir.c index 2a037aae904d..19d974c7e2b7 100644 --- a/fs/xfs/scrub/dir.c +++ b/fs/xfs/scrub/dir.c @@ -492,7 +492,7 @@ xchk_directory_data_bestfree( goto out; xchk_buffer_recheck(sc, bp); - if (xfs_has_crc(sc->mp)) { + if (!is_block && xfs_has_crc(sc->mp)) { struct xfs_dir3_data_hdr *hdr3 = bp->b_addr; if (hdr3->pad) diff --git a/fs/xfs/scrub/dirtree.c b/fs/xfs/scrub/dirtree.c index 9b0ab2316612..887383d2e941 100644 --- a/fs/xfs/scrub/dirtree.c +++ b/fs/xfs/scrub/dirtree.c @@ -1021,7 +1021,7 @@ out: return error; } -/* Does the directory targetted by this scrub have no parents? */ +/* Does the directory targeted by this scrub have no parents? */ bool xchk_dirtree_parentless(const struct xchk_dirtree *dl) { diff --git a/fs/xfs/scrub/findparent.c b/fs/xfs/scrub/findparent.c index eab3ac2704be..d921fe5a9b0c 100644 --- a/fs/xfs/scrub/findparent.c +++ b/fs/xfs/scrub/findparent.c @@ -473,6 +473,9 @@ xrep_findparent_from_dcache( pip = igrab(d_inode(parent)); dput(parent); + if (!pip) + goto out_dput; + if (S_ISDIR(pip->i_mode)) { ret = pip->i_ino; trace_xrep_findparent_from_dcache(sc->ip, ret); diff --git a/fs/xfs/scrub/health.c b/fs/xfs/scrub/health.c index 2171bcf0f6c1..487ecc5f9f3c 100644 --- a/fs/xfs/scrub/health.c +++ b/fs/xfs/scrub/health.c @@ -202,9 +202,9 @@ xchk_update_health( * there's no sick flag defined for it, so we branch here ahead of the * mask check. */ - if (sc->sm->sm_type == XFS_SCRUB_TYPE_HEALTHY && - !(sc->sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT)) { - xchk_mark_all_healthy(sc->mp); + if (sc->sm->sm_type == XFS_SCRUB_TYPE_HEALTHY) { + if (!(sc->sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT)) + xchk_mark_all_healthy(sc->mp); return; } diff --git a/fs/xfs/scrub/inode.c b/fs/xfs/scrub/inode.c index 65b13e311916..46e9bf4a4317 100644 --- a/fs/xfs/scrub/inode.c +++ b/fs/xfs/scrub/inode.c @@ -607,7 +607,7 @@ xchk_dinode( } /* di_forkoff */ - if (XFS_DFORK_BOFF(dip) >= mp->m_sb.sb_inodesize) + if (dip->di_forkoff >= (XFS_LITINO(mp) >> 3)) xchk_ino_set_corrupt(sc, ino); if (naextents != 0 && dip->di_forkoff == 0) xchk_ino_set_corrupt(sc, ino); diff --git a/fs/xfs/scrub/inode_repair.c b/fs/xfs/scrub/inode_repair.c index 8bc508336aa5..b87c22146233 100644 --- a/fs/xfs/scrub/inode_repair.c +++ b/fs/xfs/scrub/inode_repair.c @@ -1702,7 +1702,7 @@ xrep_inode_blockcounts( &acount); if (error) return error; - if (count >= sc->mp->m_sb.sb_dblocks) + if (acount >= sc->mp->m_sb.sb_dblocks) return -EFSCORRUPTED; error = xrep_ino_ensure_extent_count(sc, XFS_ATTR_FORK, nextents); diff --git a/fs/xfs/scrub/newbt.c b/fs/xfs/scrub/newbt.c index c82f4631fd9c..584076b2a6ee 100644 --- a/fs/xfs/scrub/newbt.c +++ b/fs/xfs/scrub/newbt.c @@ -193,9 +193,11 @@ xrep_newbt_add_blocks( struct xrep_newbt_resv *resv; int error; - resv = kmalloc_obj(struct xrep_newbt_resv, XCHK_GFP_FLAGS); - if (!resv) - return -ENOMEM; + /* + * We have no way to clean up the allocated space *and* return an + * ENOMEM if we fail to allocate this control structure. + */ + resv = kmalloc_obj(struct xrep_newbt_resv, GFP_KERNEL | __GFP_NOFAIL); INIT_LIST_HEAD(&resv->list); resv->agbno = XFS_FSB_TO_AGBNO(mp, args->fsbno); diff --git a/fs/xfs/scrub/orphanage.c b/fs/xfs/scrub/orphanage.c index 3aca66869b80..21e31eeaa042 100644 --- a/fs/xfs/scrub/orphanage.c +++ b/fs/xfs/scrub/orphanage.c @@ -192,12 +192,16 @@ xrep_orphanage_create( /* Make sure the orphanage is owned by root. */ error = xrep_chown_orphanage(sc, XFS_I(orphanage_inode)); if (error) - goto out_dput_orphanage; + goto out_rele_orphanage; /* Stash the reference for later and bail out. */ sc->orphanage = XFS_I(orphanage_inode); sc->orphanage_ilock_flags = 0; + orphanage_inode = NULL; +out_rele_orphanage: + if (orphanage_inode) + xchk_irele(sc, XFS_I(orphanage_inode)); out_dput_orphanage: end_creating(orphanage_dentry); out_dput_root: diff --git a/fs/xfs/scrub/reap.c b/fs/xfs/scrub/reap.c index f698b9be3dd1..0dfe61bafc6a 100644 --- a/fs/xfs/scrub/reap.c +++ b/fs/xfs/scrub/reap.c @@ -172,7 +172,7 @@ static inline bool xreap_is_dirty(const struct xreap_state *rs) } /* - * Decide if we need to roll the transaction to clear out the the log + * Decide if we need to roll the transaction to clear out the log * reservation that we allocated to buffer invalidations. */ static inline bool xreap_want_binval_roll(const struct xreap_state *rs) diff --git a/fs/xfs/scrub/repair.c b/fs/xfs/scrub/repair.c index 11697a8b2a1d..c2a437416227 100644 --- a/fs/xfs/scrub/repair.c +++ b/fs/xfs/scrub/repair.c @@ -399,6 +399,7 @@ xrep_calc_rtgroup_resblks( struct xfs_mount *mp = sc->mp; struct xfs_scrub_metadata *sm = sc->sm; uint64_t usedlen; + xfs_extlen_t refcbt_sz = 0; xfs_extlen_t rmapbt_sz = 0; if (!(sm->sm_flags & XFS_SCRUB_IFLAG_REPAIR)) @@ -411,13 +412,27 @@ xrep_calc_rtgroup_resblks( usedlen = xfs_rtbxlen_to_blen(mp, xfs_rtgroup_extents(mp, sm->sm_agno)); ASSERT(usedlen <= XFS_MAX_RGBLOCKS); + if (xfs_has_reflink(mp)) + refcbt_sz = xfs_rtrefcountbt_calc_size(mp, usedlen); + if (xfs_has_rmapbt(mp)) rmapbt_sz = xfs_rtrmapbt_calc_size(mp, usedlen); + /* + * Guess how many blocks we need to rebuild the rmapbt. For + * non-reflink filesystems we can't have more records than used blocks. + * However, with reflink it's possible to have more than one rmap + * record per rtgroup block. We don't know how many rmaps there could + * be in the rtgroup, so we start off with what we hope is an generous + * over-estimation. + */ + if (refcbt_sz > 0 && rmapbt_sz > 0) + rmapbt_sz *= 2; + trace_xrep_calc_rtgroup_resblks_btsize(mp, sm->sm_agno, usedlen, - rmapbt_sz); + rmapbt_sz, refcbt_sz); - return rmapbt_sz; + return max(rmapbt_sz, refcbt_sz); } #endif /* CONFIG_XFS_RT */ diff --git a/fs/xfs/scrub/rmap.c b/fs/xfs/scrub/rmap.c index 0cd3eecd2ca5..68e2847c962b 100644 --- a/fs/xfs/scrub/rmap.c +++ b/fs/xfs/scrub/rmap.c @@ -493,11 +493,18 @@ out: * If there's an error, set XFAIL and disable the bitmap * cross-referencing checks, but proceed with the scrub anyway. */ - if (error) - xchk_btree_xref_process_error(sc, sc->sa.rmap_cur, - sc->sa.rmap_cur->bc_nlevels - 1, &error); - else - cr->bitmaps_complete = true; + if (error) { + if (!xchk_btree_xref_process_error(sc, sc->sa.rmap_cur, + sc->sa.rmap_cur->bc_nlevels - 1, &error)) { + /* only set incomplete if we didn't set xfail */ + if (error) + xchk_set_incomplete(sc); + } + + return 0; + } + + cr->bitmaps_complete = true; return 0; } @@ -567,7 +574,8 @@ xchk_rmapbt( if (error) goto out; - xchk_rmapbt_check_bitmaps(sc, cr); + if (cr->bitmaps_complete) + xchk_rmapbt_check_bitmaps(sc, cr); out: xagb_bitmap_destroy(&cr->refcbt_owned); diff --git a/fs/xfs/scrub/rmap_repair.c b/fs/xfs/scrub/rmap_repair.c index 590f9f41856e..725035bf4903 100644 --- a/fs/xfs/scrub/rmap_repair.c +++ b/fs/xfs/scrub/rmap_repair.c @@ -1109,6 +1109,7 @@ xrep_rmap_try_reserve( return error; error = xfs_agfl_walk(sc->mp, agf, agfl_bp, xrep_rmap_walk_agfl, &ra); + xfs_trans_brelse(sc->tp, agfl_bp); if (error) return error; diff --git a/fs/xfs/scrub/scrub.h b/fs/xfs/scrub/scrub.h index 737a5d6db15f..b093945f3631 100644 --- a/fs/xfs/scrub/scrub.h +++ b/fs/xfs/scrub/scrub.h @@ -40,7 +40,7 @@ static inline int xchk_maybe_relax(struct xchk_relax *widget) return 0; widget->resched_nr = 0; - if (unlikely(widget->next_resched <= jiffies)) { + if (unlikely(time_after_eq(jiffies, widget->next_resched))) { cond_resched(); widget->next_resched = XCHK_RELAX_NEXT; } diff --git a/fs/xfs/scrub/trace.h b/fs/xfs/scrub/trace.h index 0f5adc293962..cb85f75ce101 100644 --- a/fs/xfs/scrub/trace.h +++ b/fs/xfs/scrub/trace.h @@ -2376,25 +2376,29 @@ TRACE_EVENT(xrep_calc_ag_resblks_btsize, #ifdef CONFIG_XFS_RT TRACE_EVENT(xrep_calc_rtgroup_resblks_btsize, TP_PROTO(struct xfs_mount *mp, xfs_rgnumber_t rgno, - xfs_rgblock_t usedlen, xfs_rgblock_t rmapbt_sz), - TP_ARGS(mp, rgno, usedlen, rmapbt_sz), + xfs_rgblock_t usedlen, xfs_rgblock_t rmapbt_sz, + xfs_rgblock_t refcbt_sz), + TP_ARGS(mp, rgno, usedlen, rmapbt_sz, refcbt_sz), TP_STRUCT__entry( __field(dev_t, dev) __field(xfs_rgnumber_t, rgno) __field(xfs_rgblock_t, usedlen) __field(xfs_rgblock_t, rmapbt_sz) + __field(xfs_rgblock_t, refcbt_sz) ), TP_fast_assign( __entry->dev = mp->m_super->s_dev; __entry->rgno = rgno; __entry->usedlen = usedlen; __entry->rmapbt_sz = rmapbt_sz; + __entry->refcbt_sz = refcbt_sz; ), - TP_printk("dev %d:%d rgno 0x%x usedlen %u rmapbt %u", + TP_printk("dev %d:%d rgno 0x%x usedlen %u rmapbt %u refcountbt %u", MAJOR(__entry->dev), MINOR(__entry->dev), __entry->rgno, __entry->usedlen, - __entry->rmapbt_sz) + __entry->rmapbt_sz, + __entry->refcbt_sz) ); #endif /* CONFIG_XFS_RT */ diff --git a/fs/xfs/xfs_bmap_item.c b/fs/xfs/xfs_bmap_item.c index 89f6e79a955f..aa5b41629747 100644 --- a/fs/xfs/xfs_bmap_item.c +++ b/fs/xfs/xfs_bmap_item.c @@ -339,7 +339,7 @@ xfs_bmap_update_get_group( /* * Bump the intent count on behalf of the deferred rmap and refcount - * intent items that that we can queue when we finish this bmap work. + * intent items that we can queue when we finish this bmap work. * This new intent item will bump the intent count before the bmap * intent drops the intent count, ensuring that the intent count * remains nonzero across the transaction roll. diff --git a/fs/xfs/xfs_dquot.c b/fs/xfs/xfs_dquot.c index b4f6c594808c..e696ee36c2e8 100644 --- a/fs/xfs/xfs_dquot.c +++ b/fs/xfs/xfs_dquot.c @@ -139,10 +139,14 @@ xfs_qm_adjust_dqlimits( dq->q_ino.softlimit = defq->ino.soft; if (!dq->q_ino.hardlimit) dq->q_ino.hardlimit = defq->ino.hard; - if (!dq->q_rtb.softlimit) + if (!dq->q_rtb.softlimit) { dq->q_rtb.softlimit = defq->rtb.soft; - if (!dq->q_rtb.hardlimit) + prealloc = 1; + } + if (!dq->q_rtb.hardlimit) { dq->q_rtb.hardlimit = defq->rtb.hard; + prealloc = 1; + } if (prealloc) xfs_dquot_set_prealloc_limits(dq); diff --git a/fs/xfs/xfs_exchrange.c b/fs/xfs/xfs_exchrange.c index c69ecd6a19de..fafb4e3f065c 100644 --- a/fs/xfs/xfs_exchrange.c +++ b/fs/xfs/xfs_exchrange.c @@ -633,6 +633,9 @@ xfs_exchrange_prep( if (error) return error; + if (fxr->flags & XFS_EXCHANGE_RANGE_DRY_RUN) + return 0; + trace_xfs_exchrange_flush(fxr, ip1, ip2); /* Flush the relevant ranges of both files. */ @@ -709,9 +712,11 @@ xfs_exchrange_contents( * other file write would do. This may involve turning on support for * logged xattrs if either file has security capabilities. */ - error = xfs_exchange_range_finish(fxr); - if (error) - goto out_unlock; + if (!(fxr->flags & XFS_EXCHANGE_RANGE_DRY_RUN)) { + error = xfs_exchange_range_finish(fxr); + if (error) + goto out_unlock; + } out_unlock: xfs_iunlock2_io_mmap(ip1, ip2); @@ -902,7 +907,7 @@ xfs_ioc_commit_range( if (copy_from_user(&args, argp, sizeof(args))) return -EFAULT; - if (args.flags & ~XFS_EXCHANGE_RANGE_ALL_FLAGS) + if (args.pad || (args.flags & ~XFS_EXCHANGE_RANGE_ALL_FLAGS)) return -EINVAL; if (kern_f->magic != XCR_FRESH_MAGIC) return -EBUSY; diff --git a/fs/xfs/xfs_healthmon.c b/fs/xfs/xfs_healthmon.c index c3749675ef19..2bdd747f1ead 100644 --- a/fs/xfs/xfs_healthmon.c +++ b/fs/xfs/xfs_healthmon.c @@ -247,7 +247,9 @@ xfs_healthmon_merge_events( case XFS_HEALTHMON_DIOWRITE: case XFS_HEALTHMON_DATALOST: /* logically adjacent file ranges can merge */ - if (existing->fino != new->fino || existing->fgen != new->fgen) + if (existing->fino != new->fino || + existing->fgen != new->fgen || + existing->error != new->error) return false; if (existing->fpos + existing->flen == new->fpos) { diff --git a/fs/xfs/xfs_icache.c b/fs/xfs/xfs_icache.c index 82dac88e3c4c..de8be344e987 100644 --- a/fs/xfs/xfs_icache.c +++ b/fs/xfs/xfs_icache.c @@ -1653,7 +1653,7 @@ xfs_blockgc_free_dquots( do_work = true; } - if (XFS_IS_UQUOTA_ENFORCED(mp) && gdqp && xfs_dquot_lowsp(gdqp)) { + if (XFS_IS_GQUOTA_ENFORCED(mp) && gdqp && xfs_dquot_lowsp(gdqp)) { icw.icw_gid = make_kgid(mp->m_super->s_user_ns, gdqp->q_id); icw.icw_flags |= XFS_ICWALK_FLAG_GID; do_work = true; diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c index 030a7c8f2c12..621513d7215e 100644 --- a/fs/xfs/xfs_inode.c +++ b/fs/xfs/xfs_inode.c @@ -2669,7 +2669,7 @@ xfs_irele( } /* - * Ensure all commited transactions touching the inode are written to the log. + * Ensure all committed transactions touching the inode are written to the log. */ int xfs_log_force_inode( diff --git a/fs/xfs/xfs_log_cil.c b/fs/xfs/xfs_log_cil.c index f9e07a32f60f..9446ac44ba88 100644 --- a/fs/xfs/xfs_log_cil.c +++ b/fs/xfs/xfs_log_cil.c @@ -1370,7 +1370,7 @@ xlog_cil_cleanup_whiteouts( * allocation context. However, we do not want to block on memory reclaim * recursing back into the filesystem because this push may have been triggered * by memory reclaim itself. Hence we really need to run under full GFP_NOFS - * contraints here. + * constraints here. */ static void xlog_cil_push_work( diff --git a/fs/xfs/xfs_log_recover.c b/fs/xfs/xfs_log_recover.c index e7e49529658b..cf0d610265fe 100644 --- a/fs/xfs/xfs_log_recover.c +++ b/fs/xfs/xfs_log_recover.c @@ -2736,12 +2736,13 @@ xlog_recover_iunlink_bucket( { struct xfs_mount *mp = pag_mount(pag); struct xfs_inode *prev_ip = NULL; - struct xfs_inode *ip; xfs_agino_t prev_agino, agino; int error = 0; agino = be32_to_cpu(agi->agi_unlinked[bucket]); while (agino != NULLAGINO) { + struct xfs_inode *ip; + error = xfs_iget(mp, NULL, xfs_agino_to_ino(pag, agino), 0, 0, &ip); if (error) @@ -2750,11 +2751,11 @@ xlog_recover_iunlink_bucket( ASSERT(VFS_I(ip)->i_nlink == 0); ASSERT(VFS_I(ip)->i_mode != 0); xfs_iflags_clear(ip, XFS_IRECOVERY); - agino = ip->i_next_unlinked; if (prev_ip) { ip->i_prev_unlinked = prev_agino; xfs_irele(prev_ip); + prev_ip = NULL; /* * Ensure the inode is removed from the unlinked list @@ -2766,18 +2767,20 @@ xlog_recover_iunlink_bucket( * complete. */ error = xfs_inodegc_flush(mp); - if (error) - break; + if (error) { + xfs_irele(ip); + return error; + } } prev_agino = agino; + agino = ip->i_next_unlinked; prev_ip = ip; } if (prev_ip) { int error2; - ip->i_prev_unlinked = prev_agino; xfs_irele(prev_ip); error2 = xfs_inodegc_flush(mp); diff --git a/fs/xfs/xfs_platform.h b/fs/xfs/xfs_platform.h index 5d542e95fe44..745d715b4c64 100644 --- a/fs/xfs/xfs_platform.h +++ b/fs/xfs/xfs_platform.h @@ -153,7 +153,7 @@ static inline void delay(long ticks) /* * XFS wrapper structure for sysfs support. It depends on external data * structures and is embedded in various internal data structures to implement - * the XFS sysfs object heirarchy. Define it here for broad access throughout + * the XFS sysfs object hierarchy. Define it here for broad access throughout * the codebase. */ struct xfs_kobj { diff --git a/fs/xfs/xfs_qm.c b/fs/xfs/xfs_qm.c index 99a82107b8e6..54d00d543b51 100644 --- a/fs/xfs/xfs_qm.c +++ b/fs/xfs/xfs_qm.c @@ -1432,16 +1432,22 @@ xfs_qm_flush_one( error = xfs_dquot_use_attached_buf(dqp, &bp); if (error) - goto out_unlock; + goto out_dqflock; if (!bp) { error = -EFSCORRUPTED; - goto out_unlock; + goto out_dqflock; } error = xfs_qm_dqflush(dqp, bp); if (!error) xfs_buf_delwri_queue(bp, buffer_list); xfs_buf_relse(bp); + mutex_unlock(&dqp->q_qlock); + xfs_qm_dqrele(dqp); + return error; + +out_dqflock: + xfs_dqfunlock(dqp); out_unlock: mutex_unlock(&dqp->q_qlock); xfs_qm_dqrele(dqp); diff --git a/fs/xfs/xfs_refcount_item.c b/fs/xfs/xfs_refcount_item.c index 8bccf89a7766..682c6e1b45e3 100644 --- a/fs/xfs/xfs_refcount_item.c +++ b/fs/xfs/xfs_refcount_item.c @@ -508,6 +508,7 @@ xfs_refcount_recover_work( struct xfs_cui_log_item *cuip = CUI_ITEM(lip); struct xfs_trans *tp; struct xfs_mount *mp = lip->li_log->l_mp; + unsigned int dblocks; bool isrt = xfs_cui_item_isrt(lip); int i; int error = 0; @@ -543,8 +544,11 @@ xfs_refcount_recover_work( * full btree split on either end of the refcount range. */ resv = xlog_recover_resv(&M_RES(mp)->tr_itruncate); - error = xfs_trans_alloc(mp, &resv, mp->m_refc_maxlevels * 2, 0, - XFS_TRANS_RESERVE, &tp); + if (isrt) + dblocks = mp->m_rtrefc_maxlevels * 2; + else + dblocks = mp->m_refc_maxlevels * 2; + error = xfs_trans_alloc(mp, &resv, dblocks, 0, XFS_TRANS_RESERVE, &tp); if (error) return error; diff --git a/fs/xfs/xfs_reflink.h b/fs/xfs/xfs_reflink.h index 9d1ed9bb0bee..683c1841e640 100644 --- a/fs/xfs/xfs_reflink.h +++ b/fs/xfs/xfs_reflink.h @@ -48,9 +48,6 @@ extern int xfs_reflink_end_cow(struct xfs_inode *ip, xfs_off_t offset, int xfs_reflink_end_atomic_cow(struct xfs_inode *ip, xfs_off_t offset, xfs_off_t count); extern int xfs_reflink_recover_cow(struct xfs_mount *mp); -extern loff_t xfs_reflink_remap_range(struct file *file_in, loff_t pos_in, - struct file *file_out, loff_t pos_out, loff_t len, - unsigned int remap_flags); extern int xfs_reflink_inode_has_shared_extents(struct xfs_trans *tp, struct xfs_inode *ip, bool *has_shared); extern int xfs_reflink_clear_inode_flag(struct xfs_inode *ip, diff --git a/fs/xfs/xfs_rmap_item.c b/fs/xfs/xfs_rmap_item.c index 2a3a73a8566d..000cff1ce324 100644 --- a/fs/xfs/xfs_rmap_item.c +++ b/fs/xfs/xfs_rmap_item.c @@ -573,6 +573,7 @@ xfs_rmap_recover_work( struct xfs_rui_log_item *ruip = RUI_ITEM(lip); struct xfs_trans *tp; struct xfs_mount *mp = lip->li_log->l_mp; + unsigned int dblocks; bool isrt = xfs_rui_item_isrt(lip); int i; int error = 0; @@ -596,8 +597,11 @@ xfs_rmap_recover_work( } resv = xlog_recover_resv(&M_RES(mp)->tr_itruncate); - error = xfs_trans_alloc(mp, &resv, mp->m_rmap_maxlevels, 0, - XFS_TRANS_RESERVE, &tp); + if (isrt) + dblocks = mp->m_rtrmap_maxlevels; + else + dblocks = mp->m_rmap_maxlevels; + error = xfs_trans_alloc(mp, &resv, dblocks, 0, XFS_TRANS_RESERVE, &tp); if (error) return error; diff --git a/fs/xfs/xfs_zone_alloc.c b/fs/xfs/xfs_zone_alloc.c index 28c1e48909fa..b75cf3bfe33c 100644 --- a/fs/xfs/xfs_zone_alloc.c +++ b/fs/xfs/xfs_zone_alloc.c @@ -820,7 +820,7 @@ xfs_get_cached_zone( spin_unlock(&ip->i_flags_lock); } - if (!atomic_inc_not_zero(&oz->oz_ref)) + if (oz && !atomic_inc_not_zero(&oz->oz_ref)) oz = NULL; out_unlock: rcu_read_unlock(); @@ -828,7 +828,7 @@ out_unlock: } /* - * Stash our zone in the inode so that is is reused for future allocations. + * Stash our zone in the inode so that it is reused for future allocations. * * The open_zone structure will be pinned until either the inode is freed or * until the cached open zone is replaced with a different one because the diff --git a/fs/xfs/xfs_zone_gc.c b/fs/xfs/xfs_zone_gc.c index 5fdcf98a2133..54b70ed2922f 100644 --- a/fs/xfs/xfs_zone_gc.c +++ b/fs/xfs/xfs_zone_gc.c @@ -46,7 +46,7 @@ * before remapping. * * Once a zone does not contain any valid data, be that through GC or user - * block removal, it is queued for for a zone reset. The reset operation + * block removal, it is queued for a zone reset. The reset operation * carefully ensures that the RT device cache is flushed and all transactions * referencing the rmap have been committed to disk. */ |
