summaryrefslogtreecommitdiff
path: root/fs
diff options
context:
space:
mode:
Diffstat (limited to 'fs')
-rw-r--r--fs/9p/vfs_addr.c28
-rw-r--r--fs/9p/vfs_dentry.c3
-rw-r--r--fs/adfs/super.c24
-rw-r--r--fs/afs/addr_list.c5
-rw-r--r--fs/afs/dir.c2
-rw-r--r--fs/afs/dir_edit.c9
-rw-r--r--fs/afs/dir_search.c11
-rw-r--r--fs/afs/fs_probe.c1
-rw-r--r--fs/afs/internal.h8
-rw-r--r--fs/afs/server.c1
-rw-r--r--fs/afs/symlink.c3
-rw-r--r--fs/autofs/inode.c4
-rw-r--r--fs/binfmt_misc.c7
-rw-r--r--fs/btrfs/block-group.c18
-rw-r--r--fs/btrfs/dev-replace.c3
-rw-r--r--fs/btrfs/disk-io.c4
-rw-r--r--fs/btrfs/file.c4
-rw-r--r--fs/btrfs/free-space-tree.c14
-rw-r--r--fs/btrfs/inode.c21
-rw-r--r--fs/btrfs/ioctl.c21
-rw-r--r--fs/btrfs/raid-stripe-tree.c25
-rw-r--r--fs/btrfs/raid-stripe-tree.h1
-rw-r--r--fs/btrfs/scrub.c24
-rw-r--r--fs/btrfs/send.c9
-rw-r--r--fs/btrfs/super.c8
-rw-r--r--fs/btrfs/tests/extent-io-tests.c5
-rw-r--r--fs/btrfs/transaction.c19
-rw-r--r--fs/btrfs/tree-checker.c33
-rw-r--r--fs/btrfs/tree-log.c16
-rw-r--r--fs/btrfs/verity.c18
-rw-r--r--fs/btrfs/volumes.c37
-rw-r--r--fs/btrfs/zoned.c21
-rw-r--r--fs/btrfs/zstd.c11
-rw-r--r--fs/cachefiles/xattr.c16
-rw-r--r--fs/ceph/addr.c2
-rw-r--r--fs/ceph/mds_client.c6
-rw-r--r--fs/ceph/mds_client.h1
-rw-r--r--fs/ceph/subvolume_metrics.c2
-rw-r--r--fs/ceph/super.c5
-rw-r--r--fs/configfs/dir.c9
-rw-r--r--fs/configfs/mount.c4
-rw-r--r--fs/configfs/symlink.c24
-rw-r--r--fs/coredump.c2
-rw-r--r--fs/dax.c9
-rw-r--r--fs/erofs/data.c2
-rw-r--r--fs/erofs/decompressor.c55
-rw-r--r--fs/erofs/decompressor_lzma.c18
-rw-r--r--fs/erofs/internal.h6
-rw-r--r--fs/erofs/sysfs.c2
-rw-r--r--fs/erofs/xattr.c9
-rw-r--r--fs/erofs/zdata.c18
-rw-r--r--fs/exec.c53
-rw-r--r--fs/ext4/fast_commit.c4
-rw-r--r--fs/ext4/inode.c11
-rw-r--r--fs/fuse/file.c3
-rw-r--r--fs/fuse/readdir.c2
-rw-r--r--fs/hfs/bnode.c2
-rw-r--r--fs/kernfs/inode.c4
-rw-r--r--fs/namespace.c2
-rw-r--r--fs/netfs/buffered_read.c164
-rw-r--r--fs/netfs/direct_write.c31
-rw-r--r--fs/netfs/internal.h3
-rw-r--r--fs/netfs/misc.c19
-rw-r--r--fs/netfs/objects.c32
-rw-r--r--fs/netfs/read_collect.c128
-rw-r--r--fs/netfs/read_pgpriv2.c15
-rw-r--r--fs/netfs/read_retry.c13
-rw-r--r--fs/netfs/read_single.c2
-rw-r--r--fs/netfs/rolling_buffer.c81
-rw-r--r--fs/netfs/write_issue.c2
-rw-r--r--fs/nfsd/export.c10
-rw-r--r--fs/nfsd/nfs4callback.c4
-rw-r--r--fs/nfsd/nfs4state.c7
-rw-r--r--fs/nfsd/nfsctl.c2
-rw-r--r--fs/ntfs/attrib.c277
-rw-r--r--fs/ntfs/attrib.h9
-rw-r--r--fs/ntfs/attrlist.c224
-rw-r--r--fs/ntfs/attrlist.h2
-rw-r--r--fs/ntfs/bdev-io.c2
-rw-r--r--fs/ntfs/bitmap.c10
-rw-r--r--fs/ntfs/compress.c12
-rw-r--r--fs/ntfs/dir.c16
-rw-r--r--fs/ntfs/ea.c51
-rw-r--r--fs/ntfs/file.c53
-rw-r--r--fs/ntfs/index.c2
-rw-r--r--fs/ntfs/inode.c75
-rw-r--r--fs/ntfs/lcnalloc.c9
-rw-r--r--fs/ntfs/logfile.c2
-rw-r--r--fs/ntfs/mft.c481
-rw-r--r--fs/ntfs/mft.h2
-rw-r--r--fs/ntfs/namei.c2
-rw-r--r--fs/ntfs/ntfs.h10
-rw-r--r--fs/ntfs/reparse.c7
-rw-r--r--fs/ntfs/runlist.c14
-rw-r--r--fs/ntfs/super.c29
-rw-r--r--fs/ntfs/volume.h12
-rw-r--r--fs/ntfs/wof.c127
-rw-r--r--fs/overlayfs/readdir.c2
-rw-r--r--fs/overlayfs/super.c2
-rw-r--r--fs/quota/dquot.c2
-rw-r--r--fs/smb/client/cifs_swn.c2
-rw-r--r--fs/smb/client/cifsacl.c95
-rw-r--r--fs/smb/client/cifssmb.c52
-rw-r--r--fs/smb/client/connect.c30
-rw-r--r--fs/smb/client/dfs_cache.c20
-rw-r--r--fs/smb/client/file.c60
-rw-r--r--fs/smb/client/inode.c17
-rw-r--r--fs/smb/client/misc.c36
-rw-r--r--fs/smb/client/readdir.c17
-rw-r--r--fs/smb/client/reparse.c79
-rw-r--r--fs/smb/client/reparse.h7
-rw-r--r--fs/smb/client/sess.c106
-rw-r--r--fs/smb/client/smb2inode.c13
-rw-r--r--fs/smb/client/smb2misc.c69
-rw-r--r--fs/smb/client/smb2ops.c246
-rw-r--r--fs/smb/client/smb2pdu.c13
-rw-r--r--fs/smb/client/trace.h2
-rw-r--r--fs/smb/client/transport.c11
-rw-r--r--fs/smb/server/connection.c80
-rw-r--r--fs/smb/server/connection.h3
-rw-r--r--fs/smb/server/ksmbd_work.c2
-rw-r--r--fs/smb/server/ksmbd_work.h2
-rw-r--r--fs/smb/server/mgmt/share_config.c40
-rw-r--r--fs/smb/server/mgmt/tree_connect.c8
-rw-r--r--fs/smb/server/mgmt/user_session.c137
-rw-r--r--fs/smb/server/mgmt/user_session.h9
-rw-r--r--fs/smb/server/oplock.c73
-rw-r--r--fs/smb/server/proc.c2
-rw-r--r--fs/smb/server/server.c2
-rw-r--r--fs/smb/server/smb2pdu.c173
-rw-r--r--fs/smb/server/smb2pdu.h3
-rw-r--r--fs/smb/server/smbacl.c67
-rw-r--r--fs/smb/server/stats.h1
-rw-r--r--fs/smb/server/transport_ipc.c21
-rw-r--r--fs/smb/server/transport_tcp.c36
-rw-r--r--fs/smb/server/vfs.c12
-rw-r--r--fs/smb/server/vfs_cache.c17
-rw-r--r--fs/smb/server/vfs_cache.h1
-rw-r--r--fs/super.c27
-rw-r--r--fs/ufs/cylinder.c10
-rw-r--r--fs/ufs/dir.c2
-rw-r--r--fs/ufs/super.c17
-rw-r--r--fs/xfs/Makefile1
-rw-r--r--fs/xfs/libxfs/xfs_btree_mem.c2
-rw-r--r--fs/xfs/libxfs/xfs_btree_staging.c4
-rw-r--r--fs/xfs/libxfs/xfs_da_btree.c7
-rw-r--r--fs/xfs/libxfs/xfs_da_btree.h2
-rw-r--r--fs/xfs/libxfs/xfs_defer.c18
-rw-r--r--fs/xfs/libxfs/xfs_exchmaps.c10
-rw-r--r--fs/xfs/libxfs/xfs_parent.c12
-rw-r--r--fs/xfs/libxfs/xfs_rtgroup.h6
-rw-r--r--fs/xfs/libxfs/xfs_rtrefcount_btree.c7
-rw-r--r--fs/xfs/libxfs/xfs_rtrmap_btree.c6
-rw-r--r--fs/xfs/libxfs/xfs_sb.c37
-rw-r--r--fs/xfs/libxfs/xfs_trans_space.c19
-rw-r--r--fs/xfs/scrub/agheader.c7
-rw-r--r--fs/xfs/scrub/agheader_repair.c39
-rw-r--r--fs/xfs/scrub/alloc_repair.c18
-rw-r--r--fs/xfs/scrub/attr_repair.c2
-rw-r--r--fs/xfs/scrub/bmap.c7
-rw-r--r--fs/xfs/scrub/common.h1
-rw-r--r--fs/xfs/scrub/dabtree.h2
-rw-r--r--fs/xfs/scrub/dir_repair.c31
-rw-r--r--fs/xfs/scrub/dirtree.c31
-rw-r--r--fs/xfs/scrub/dirtree_repair.c2
-rw-r--r--fs/xfs/scrub/findparent.c54
-rw-r--r--fs/xfs/scrub/ialloc.c4
-rw-r--r--fs/xfs/scrub/metapath.c87
-rw-r--r--fs/xfs/scrub/quota_repair.c14
-rw-r--r--fs/xfs/scrub/quotacheck.c4
-rw-r--r--fs/xfs/scrub/reap.c6
-rw-r--r--fs/xfs/scrub/refcount.c8
-rw-r--r--fs/xfs/scrub/rgsuper.c6
-rw-r--r--fs/xfs/scrub/rtrefcount.c82
-rw-r--r--fs/xfs/scrub/rtsummary_repair.c3
-rw-r--r--fs/xfs/scrub/scrub.c3
-rw-r--r--fs/xfs/scrub/scrub.h1
-rw-r--r--fs/xfs/scrub/stats.c69
-rw-r--r--fs/xfs/scrub/stats.h4
-rw-r--r--fs/xfs/scrub/symlink_repair.c2
-rw-r--r--fs/xfs/scrub/tempfile.c29
-rw-r--r--fs/xfs/scrub/tempfile.h4
-rw-r--r--fs/xfs/scrub/trace.h35
-rw-r--r--fs/xfs/xfs_aops.c188
-rw-r--r--fs/xfs/xfs_aops.h1
-rw-r--r--fs/xfs/xfs_buf.c4
-rw-r--r--fs/xfs/xfs_buf_item.h1
-rw-r--r--fs/xfs/xfs_exchmaps_item.c19
-rw-r--r--fs/xfs/xfs_exchrange.c12
-rw-r--r--fs/xfs/xfs_extent_busy.c4
-rw-r--r--fs/xfs/xfs_file.c49
-rw-r--r--fs/xfs/xfs_fsmap.c3
-rw-r--r--fs/xfs/xfs_healthmon.c128
-rw-r--r--fs/xfs/xfs_healthmon.h5
-rw-r--r--fs/xfs/xfs_icache.c19
-rw-r--r--fs/xfs/xfs_inode.h1
-rw-r--r--fs/xfs/xfs_ioctl.c2
-rw-r--r--fs/xfs/xfs_ioend.c184
-rw-r--r--fs/xfs/xfs_ioend.h16
-rw-r--r--fs/xfs/xfs_iomap.c7
-rw-r--r--fs/xfs/xfs_iomap.h16
-rw-r--r--fs/xfs/xfs_iops.c2
-rw-r--r--fs/xfs/xfs_log.c78
-rw-r--r--fs/xfs/xfs_log.h2
-rw-r--r--fs/xfs/xfs_log_cil.c10
-rw-r--r--fs/xfs/xfs_log_priv.h4
-rw-r--r--fs/xfs/xfs_mru_cache.c2
-rw-r--r--fs/xfs/xfs_platform.h11
-rw-r--r--fs/xfs/xfs_super.c2
-rw-r--r--fs/xfs/xfs_trace.h4
-rw-r--r--fs/xfs/xfs_trans_ail.c4
-rw-r--r--fs/xfs/xfs_trans_buf.c3
-rw-r--r--fs/xfs/xfs_verify_media.c24
-rw-r--r--fs/xfs/xfs_zone_alloc.c68
-rw-r--r--fs/xfs/xfs_zone_gc.c5
-rw-r--r--fs/xfs/xfs_zone_space_resv.c8
216 files changed, 4122 insertions, 1706 deletions
diff --git a/fs/9p/vfs_addr.c b/fs/9p/vfs_addr.c
index 1ac0b3dcc077..13cf87a5f90c 100644
--- a/fs/9p/vfs_addr.c
+++ b/fs/9p/vfs_addr.c
@@ -54,11 +54,37 @@ static void v9fs_begin_writeback(struct netfs_io_request *wreq)
static void v9fs_issue_write(struct netfs_io_subrequest *subreq)
{
struct p9_fid *fid = subreq->rreq->netfs_priv;
+ struct inode *inode = subreq->rreq->inode;
+ struct netfs_inode *ictx = netfs_inode(inode);
int err, len;
len = p9_client_write(fid, subreq->start, &subreq->io_iter, &err);
- if (len > 0)
+ if (len > 0) {
+ uoff_t end = subreq->start + len, i_size, remote, zp;
+ bool set = false;
+
+ spin_lock(&inode->i_lock);
+
+ /* We can read the sizes directly as we hold i_lock. */
+ i_size = inode->i_size;
+ remote = ictx->_remote_i_size;
+ zp = ictx->_zero_point;
+
+ if (end > i_size) {
+ i_size = end;
+ set = true;
+ }
+ if (end > remote) {
+ remote = end;
+ set = true;
+ }
+
+ if (set)
+ netfs_write_sizes(inode, i_size, remote, zp);
+ spin_unlock(&inode->i_lock);
+
__set_bit(NETFS_SREQ_MADE_PROGRESS, &subreq->flags);
+ }
netfs_write_subrequest_terminated(subreq, len ?: err);
}
diff --git a/fs/9p/vfs_dentry.c b/fs/9p/vfs_dentry.c
index e549e222602e..fa6b7143db98 100644
--- a/fs/9p/vfs_dentry.c
+++ b/fs/9p/vfs_dentry.c
@@ -113,8 +113,7 @@ void v9fs_dentry_fid_remove(struct dentry *dentry)
*/
static int v9fs_dentry_init(struct dentry *dentry)
{
- struct v9fs_dentry *v9fs_dentry = kzalloc(sizeof(*v9fs_dentry),
- GFP_KERNEL);
+ struct v9fs_dentry *v9fs_dentry = kzalloc_obj(*v9fs_dentry);
if (!v9fs_dentry)
return -ENOMEM;
diff --git a/fs/adfs/super.c b/fs/adfs/super.c
index a4cd0a5159dd..888aa81a6b39 100644
--- a/fs/adfs/super.c
+++ b/fs/adfs/super.c
@@ -92,10 +92,7 @@ static int adfs_checkdiscrecord(struct adfs_discrecord *dr)
static void adfs_put_super(struct super_block *sb)
{
- struct adfs_sb_info *asb = ADFS_SB(sb);
-
adfs_free_map(sb);
- kfree_rcu(asb, rcu);
}
static int adfs_show_options(struct seq_file *seq, struct dentry *root)
@@ -365,7 +362,7 @@ static int adfs_fill_super(struct super_block *sb, struct fs_context *fc)
ret = -EINVAL;
}
if (ret)
- goto error;
+ return ret;
/* set up enough so that we can read an inode */
sb->s_op = &adfs_sops;
@@ -406,15 +403,9 @@ static int adfs_fill_super(struct super_block *sb, struct fs_context *fc)
if (!sb->s_root) {
adfs_free_map(sb);
adfs_error(sb, "get root inode failed\n");
- ret = -EIO;
- goto error;
+ return -EIO;
}
return 0;
-
-error:
- sb->s_fs_info = NULL;
- kfree(asb);
- return ret;
}
static int adfs_get_tree(struct fs_context *fc)
@@ -465,10 +456,19 @@ static int adfs_init_fs_context(struct fs_context *fc)
return 0;
}
+static void adfs_kill_sb(struct super_block *sb)
+{
+ struct adfs_sb_info *asb = ADFS_SB(sb);
+
+ kill_block_super(sb);
+
+ kfree_rcu(asb, rcu);
+}
+
static struct file_system_type adfs_fs_type = {
.owner = THIS_MODULE,
.name = "adfs",
- .kill_sb = kill_block_super,
+ .kill_sb = adfs_kill_sb,
.fs_flags = FS_REQUIRES_DEV,
.init_fs_context = adfs_init_fs_context,
.parameters = adfs_param_spec,
diff --git a/fs/afs/addr_list.c b/fs/afs/addr_list.c
index 63bf096b721a..73195d76b481 100644
--- a/fs/afs/addr_list.c
+++ b/fs/afs/addr_list.c
@@ -394,8 +394,11 @@ void afs_set_peer_appdata(struct afs_server *server,
struct rxrpc_peer *pn = new_alist->addrs[n].peer;
struct rxrpc_peer *po = old_alist->addrs[o].peer;
- if (pn == po)
+ if (pn == po) {
+ n++;
+ o++;
continue;
+ }
if (pn < po) {
rxrpc_kernel_set_peer_data(pn, data);
n++;
diff --git a/fs/afs/dir.c b/fs/afs/dir.c
index 81565366d937..2db534a2c7cc 100644
--- a/fs/afs/dir.c
+++ b/fs/afs/dir.c
@@ -1801,7 +1801,7 @@ static int afs_symlink(struct mnt_idmap *idmap, struct inode *dir,
goto error;
ret = -ENOMEM;
- symlink = kmalloc_flex(struct afs_symlink, content, clen + 1, GFP_KERNEL);
+ symlink = kmalloc_flex(struct afs_symlink, content, clen + 1);
if (!symlink)
goto error;
refcount_set(&symlink->ref, 1);
diff --git a/fs/afs/dir_edit.c b/fs/afs/dir_edit.c
index 3ead36a07048..c31303059444 100644
--- a/fs/afs/dir_edit.c
+++ b/fs/afs/dir_edit.c
@@ -442,7 +442,7 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
/* Check and clear the entry. */
de = &block->dirents[slot];
if (de->u.valid != 1)
- goto error_unmap;
+ goto error;
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete, b, slot,
ntohl(de->u.vnode), ntohl(de->u.unique),
@@ -458,7 +458,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
/* Clear the constituent entries. */
next = de->u.hash_next;
memset(de, 0, sizeof(*de) * iter.nr_slots);
- kunmap_local(block);
/* Adjust the hash chain: if iter->prev_entry is 0, the hashtable head
* index is previous; otherwise it's slot number of the previous entry.
@@ -485,7 +484,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
pde = &pblock->dirents[ps];
prev_next = pde->u.hash_next;
if (prev_next != htons(entry)) {
- kunmap_local(pblock);
pr_warn("%llx:%llx:%x: not prev in chain b=%x p=%x,%x e=%x %*s",
vnode->fid.vid, vnode->fid.vnode, vnode->fid.unique,
iter.bucket, iter.prev_entry, prev_next, entry,
@@ -493,7 +491,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
goto error;
}
pde->u.hash_next = next;
- kunmap_local(pblock);
}
netfs_single_mark_inode_dirty(&vnode->netfs.inode);
@@ -503,18 +500,16 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
_debug("Remove %s from %u[%u]", name->name, b, slot);
out_unmap:
+ afs_dir_end_iter(&iter);
kunmap_local(meta);
_leave("");
return;
already_invalidated:
- kunmap_local(block);
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete_inval,
0, 0, 0, 0, name->name);
goto out_unmap;
-error_unmap:
- kunmap_local(block);
error:
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete_error,
0, 0, 0, 0, name->name);
diff --git a/fs/afs/dir_search.c b/fs/afs/dir_search.c
index 104411c0692f..11ebdfffcb1d 100644
--- a/fs/afs/dir_search.c
+++ b/fs/afs/dir_search.c
@@ -75,10 +75,7 @@ union afs_xdr_dir_block *afs_dir_find_block(struct afs_dir_iter *iter, size_t bl
_enter("%zx,%d", block, slot);
- if (iter->block) {
- kunmap_local(iter->block);
- iter->block = NULL;
- }
+ afs_dir_end_iter(iter);
if (dvnode->directory_size < blend)
goto fail;
@@ -173,12 +170,8 @@ int afs_dir_search_bucket(struct afs_dir_iter *iter, const struct qstr *name,
ret = -ENOENT;
found:
- if (iter->block) {
- kunmap_local(iter->block);
- iter->block = NULL;
- }
-
bad:
+ afs_dir_end_iter(iter);
if (ret == -ESTALE)
afs_invalidate_dir(iter->dvnode, afs_dir_invalid_iter_stale);
_leave(" = %d", ret);
diff --git a/fs/afs/fs_probe.c b/fs/afs/fs_probe.c
index a91ad1938d07..8c62334dbfe7 100644
--- a/fs/afs/fs_probe.c
+++ b/fs/afs/fs_probe.c
@@ -258,6 +258,7 @@ int afs_fs_probe_fileserver(struct afs_net *net, struct afs_server *server,
lockdep_is_held(&server->fs_lock));
if (old) {
estate->responsive_set = old->responsive_set;
+ old_alist = old->addresses;
if (!new_alist)
new_alist = old->addresses;
}
diff --git a/fs/afs/internal.h b/fs/afs/internal.h
index 290873bac89b..330654ed16ec 100644
--- a/fs/afs/internal.h
+++ b/fs/afs/internal.h
@@ -1133,6 +1133,14 @@ int afs_dir_search_bucket(struct afs_dir_iter *iter, const struct qstr *name,
int afs_dir_search(struct afs_vnode *dvnode, const struct qstr *name,
struct afs_fid *_fid, afs_dataversion_t *_dir_version);
+static inline void afs_dir_end_iter(struct afs_dir_iter *iter)
+{
+ if (iter->block) {
+ kunmap_local(iter->block);
+ iter->block = NULL;
+ }
+}
+
/*
* dir_silly.c
*/
diff --git a/fs/afs/server.c b/fs/afs/server.c
index 0fe162ea2a36..189138bd6d71 100644
--- a/fs/afs/server.c
+++ b/fs/afs/server.c
@@ -242,7 +242,6 @@ struct afs_server *afs_lookup_server(struct afs_cell *cell, struct key *key,
out:
afs_put_addrlist(alist, afs_alist_trace_put_server_create);
if (candidate) {
- kfree(rcu_access_pointer(server->endpoint_state));
kfree(candidate);
afs_dec_servers_outstanding(cell->net);
}
diff --git a/fs/afs/symlink.c b/fs/afs/symlink.c
index 16b4823cb7b7..6b8c122877ca 100644
--- a/fs/afs/symlink.c
+++ b/fs/afs/symlink.c
@@ -119,8 +119,7 @@ static ssize_t afs_do_read_symlink(struct afs_vnode *vnode)
vnode->directory_size = i_size;
/* Copy the symlink. */
- symlink = kmalloc_flex(struct afs_symlink, content, i_size + 1,
- GFP_KERNEL);
+ symlink = kmalloc_flex(struct afs_symlink, content, i_size + 1);
if (!symlink)
return -ENOMEM;
diff --git a/fs/autofs/inode.c b/fs/autofs/inode.c
index c1e210cec436..6b15a3717ba7 100644
--- a/fs/autofs/inode.c
+++ b/fs/autofs/inode.c
@@ -323,8 +323,10 @@ static int autofs_fill_super(struct super_block *s, struct fs_context *fc)
return -ENOMEM;
root_inode = autofs_get_inode(s, S_IFDIR | 0755);
- if (!root_inode)
+ if (!root_inode) {
+ autofs_free_ino(ino);
return -ENOMEM;
+ }
root_inode->i_uid = ctx->uid;
root_inode->i_gid = ctx->gid;
diff --git a/fs/binfmt_misc.c b/fs/binfmt_misc.c
index ddfd3aa57ac8..620da85948b4 100644
--- a/fs/binfmt_misc.c
+++ b/fs/binfmt_misc.c
@@ -331,8 +331,8 @@ static int entry_attach_interpreter(struct binfmt_misc_entry *e,
return -ENOSPC;
/* One allocation, both strings in it, like the entry's own buffer. */
- interp = kmalloc(struct_size(interp, name, nlen + plen + 2),
- GFP_KERNEL_ACCOUNT);
+ interp = kmalloc_flex(*interp, name, nlen + plen + 2,
+ GFP_KERNEL_ACCOUNT);
if (!interp) {
dec_ucount(ucounts, UCOUNT_BINFMT_MISC_INTERPRETERS);
return -ENOMEM;
@@ -858,8 +858,7 @@ static struct binfmt_misc_entry *create_entry(const char __user *buffer,
if ((count < 11) || (count > MAX_REGISTER_LENGTH))
return ERR_PTR(-EINVAL);
- e = kmalloc(struct_size(e, buf, count + MISC_DELIM_PAD),
- GFP_KERNEL_ACCOUNT);
+ e = kmalloc_flex(*e, buf, count + MISC_DELIM_PAD, GFP_KERNEL_ACCOUNT);
if (!e)
return ERR_PTR(-ENOMEM);
diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c
index 830460a40e86..ee182369254c 100644
--- a/fs/btrfs/block-group.c
+++ b/fs/btrfs/block-group.c
@@ -3074,21 +3074,25 @@ struct btrfs_block_group *btrfs_make_block_group(struct btrfs_trans_handle *tran
return ERR_PTR(ret);
}
- ret = btrfs_add_new_free_space(cache, chunk_offset, chunk_offset + size, NULL);
- btrfs_free_excluded_extents(cache);
- if (ret) {
- btrfs_put_block_group(cache);
- return ERR_PTR(ret);
- }
-
/*
* Ensure the corresponding space_info object is created and
* assigned to our block group. We want our bg to be added to the rbtree
* with its ->space_info set.
+ *
+ * On a zoned filesystem btrfs_add_new_free_space() ends up in
+ * __btrfs_add_free_space_zoned(), which dereferences
+ * block_group->space_info, so it has to be set beforehand.
*/
cache->space_info = space_info;
ASSERT(cache->space_info);
+ ret = btrfs_add_new_free_space(cache, chunk_offset, chunk_offset + size, NULL);
+ btrfs_free_excluded_extents(cache);
+ if (ret) {
+ btrfs_put_block_group(cache);
+ return ERR_PTR(ret);
+ }
+
ret = btrfs_add_block_group_cache(cache);
if (ret) {
btrfs_remove_free_space_cache(cache);
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
index dc0834f920c3..0284be0e4e82 100644
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -494,6 +494,7 @@ static int mark_block_group_to_copy(struct btrfs_fs_info *fs_info,
path->reada = READA_FORWARD;
path->search_commit_root = true;
path->skip_locking = true;
+ path->need_commit_sem = true;
key.objectid = src_dev->devid;
key.type = BTRFS_DEV_EXTENT_KEY;
@@ -636,7 +637,7 @@ static int btrfs_dev_replace_start(struct btrfs_fs_info *fs_info,
ret = mark_block_group_to_copy(fs_info, src_device);
if (ret)
- return ret;
+ goto leave;
down_write(&dev_replace->rwsem);
dev_replace->replace_task = current;
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index 819727460bcf..dc7ad92876c0 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -2357,6 +2357,10 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info,
key.type, cur);
return -EUCLEAN;
}
+
+ if (unlikely(cur + sizeof(*chunk) > sys_array_size))
+ goto short_read;
+
chunk = (struct btrfs_chunk *)(sb->sys_chunk_array + cur);
num_stripes = btrfs_stack_chunk_num_stripes(chunk);
if (unlikely(cur + btrfs_chunk_item_size(num_stripes) > sys_array_size))
diff --git a/fs/btrfs/file.c b/fs/btrfs/file.c
index 20e15dc30bfb..f978c6524aa0 100644
--- a/fs/btrfs/file.c
+++ b/fs/btrfs/file.c
@@ -2509,8 +2509,10 @@ int btrfs_replace_file_extents(struct btrfs_inode *inode,
inode_set_ctime_current(&inode->vfs_inode));
ret = btrfs_update_inode(trans, inode);
- if (ret)
+ if (unlikely(ret)) {
+ btrfs_abort_transaction(trans, ret);
break;
+ }
btrfs_end_transaction(trans);
btrfs_btree_balance_dirty(fs_info);
diff --git a/fs/btrfs/free-space-tree.c b/fs/btrfs/free-space-tree.c
index 1b3d82ae3de8..b7a4a6ade30f 100644
--- a/fs/btrfs/free-space-tree.c
+++ b/fs/btrfs/free-space-tree.c
@@ -1353,7 +1353,7 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info)
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
btrfs_end_transaction(trans);
- return ret;
+ goto out_clear;
}
node = rb_first_cached(&fs_info->block_group_cache_tree);
@@ -1371,14 +1371,16 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info)
if (unlikely(ret)) {
btrfs_abort_transaction(trans, ret);
btrfs_end_transaction(trans);
- return ret;
+ goto out_clear;
}
next:
if (btrfs_should_end_transaction(trans)) {
btrfs_end_transaction(trans);
trans = btrfs_start_transaction(free_space_root, 1);
- if (IS_ERR(trans))
- return PTR_ERR(trans);
+ if (IS_ERR(trans)) {
+ ret = PTR_ERR(trans);
+ goto out_clear;
+ }
}
node = rb_next(node);
}
@@ -1390,6 +1392,10 @@ next:
ret = btrfs_commit_transaction(trans);
clear_bit(BTRFS_FS_FREE_SPACE_TREE_UNTRUSTED, &fs_info->flags);
return ret;
+
+out_clear:
+ clear_bit(BTRFS_FS_CREATING_FREE_SPACE_TREE, &fs_info->flags);
+ return ret;
}
static int __add_block_group_free_space(struct btrfs_trans_handle *trans,
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index 3c10a0ef0002..558b4a3f9633 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -2339,12 +2339,27 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f
} else if (inode->prop_compress) {
compress_type = inode->prop_compress;
}
+ /*
+ * We need to pass blocksize and not i_size, otherwise we can't
+ * create compressed inline extents for data smaller than sector
+ * size with lzo.
+ */
cb = btrfs_compress_bio(inode, 0, blocksize, compress_type, compress_level, 0);
if (IS_ERR(cb)) {
cb = NULL;
/* Just fall back to non-compressed case. */
} else {
compressed_size = cb->bbio.bio.bi_iter.bi_size;
+ /*
+ * If we did not save space, it's pointless and wasteful
+ * to have an inline compressed extent, so fallback to
+ * an uncompressed inline extent.
+ */
+ if (compressed_size >= i_size) {
+ cleanup_compressed_bio(cb);
+ cb = NULL;
+ compressed_size = 0;
+ }
}
}
if (!can_cow_file_range_inline(inode, 0, i_size, compressed_size)) {
@@ -3436,6 +3451,9 @@ out:
*/
btrfs_remove_ordered_extent(ordered_extent);
+ /* Cleanup any remaining biocs attached to the OE. */
+ btrfs_cleanup_ordered_bioc_list(ordered_extent);
+
/* once for us */
btrfs_put_ordered_extent(ordered_extent);
/* once for the tree */
@@ -3874,7 +3892,8 @@ int btrfs_orphan_cleanup(struct btrfs_root *root)
if (ret)
goto out;
}
- trans = btrfs_start_transaction(root, 1);
+ /* Only deletes the orphan. */
+ trans = btrfs_start_transaction_fallback_global_rsv(root, 1);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
goto out;
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
index 72bc9d4f7708..e4b2da31a0d5 100644
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -384,6 +384,7 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags &= ~BTRFS_INODE_COMPRESS;
inode_flags |= BTRFS_INODE_NOCOMPRESS;
} else if (fsflags & FS_COMPR_FL) {
+ enum btrfs_compression_type comp_type;
if (IS_SWAPFILE(&inode->vfs_inode))
return -ETXTBSY;
@@ -391,9 +392,23 @@ int btrfs_fileattr_set(struct mnt_idmap *idmap,
inode_flags |= BTRFS_INODE_COMPRESS;
inode_flags &= ~BTRFS_INODE_NOCOMPRESS;
- comp = btrfs_compress_type2str(fs_info->compress_type);
- if (!comp || comp[0] == 0)
- comp = btrfs_compress_type2str(BTRFS_COMPRESS_ZLIB);
+ /*
+ * Keep the algorithm recorded in the compression property,
+ * otherwise changing an unrelated attribute would reset it to
+ * the mount default, since FS_IOC_SETFLAGS callers write back
+ * the whole flag set they got from FS_IOC_GETFLAGS and that
+ * includes FS_COMPR_FL for any inode carrying the property.
+ *
+ * Inodes with the compress flag set but no property keep using
+ * the mount default, so they behave as before.
+ */
+ if (inode->prop_compress)
+ comp_type = inode->prop_compress;
+ else if (fs_info->compress_type)
+ comp_type = fs_info->compress_type;
+ else
+ comp_type = BTRFS_COMPRESS_ZLIB;
+ comp = btrfs_compress_type2str(comp_type);
} else {
inode_flags &= ~(BTRFS_INODE_COMPRESS | BTRFS_INODE_NOCOMPRESS);
}
diff --git a/fs/btrfs/raid-stripe-tree.c b/fs/btrfs/raid-stripe-tree.c
index b210371ce91e..d9e660447205 100644
--- a/fs/btrfs/raid-stripe-tree.c
+++ b/fs/btrfs/raid-stripe-tree.c
@@ -310,8 +310,10 @@ static int update_raid_extent_item(struct btrfs_trans_handle *trans,
ret = btrfs_search_slot(trans, trans->fs_info->stripe_root, key, path,
0, 1);
- if (ret)
- return (ret == 1 ? ret : -EINVAL);
+ if (ret > 0)
+ ret = -ENOENT;
+ if (ret < 0)
+ return ret;
leaf = path->nodes[0];
slot = path->slots[0];
@@ -337,7 +339,6 @@ int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans,
stripe_extent = kzalloc(item_size, GFP_NOFS);
if (unlikely(!stripe_extent)) {
btrfs_abort_transaction(trans, -ENOMEM);
- btrfs_end_transaction(trans);
return -ENOMEM;
}
@@ -374,7 +375,7 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
struct btrfs_ordered_extent *ordered_extent)
{
struct btrfs_io_context *bioc;
- int ret;
+ int ret = 0;
if (!btrfs_fs_incompat(trans->fs_info, RAID_STRIPE_TREE))
return 0;
@@ -382,17 +383,23 @@ int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
list_for_each_entry(bioc, &ordered_extent->bioc_list, rst_ordered_entry) {
ret = btrfs_insert_one_raid_extent(trans, bioc);
if (ret)
- return ret;
+ break;
}
- while (!list_empty(&ordered_extent->bioc_list)) {
- bioc = list_first_entry(&ordered_extent->bioc_list,
+ btrfs_cleanup_ordered_bioc_list(ordered_extent);
+ return ret;
+}
+
+void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered)
+{
+ while (!list_empty(&ordered->bioc_list)) {
+ struct btrfs_io_context *bioc;
+
+ bioc = list_first_entry(&ordered->bioc_list,
typeof(*bioc), rst_ordered_entry);
list_del(&bioc->rst_ordered_entry);
btrfs_put_bioc(bioc);
}
-
- return 0;
}
int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info,
diff --git a/fs/btrfs/raid-stripe-tree.h b/fs/btrfs/raid-stripe-tree.h
index 69942ad43140..eb02cf48511b 100644
--- a/fs/btrfs/raid-stripe-tree.h
+++ b/fs/btrfs/raid-stripe-tree.h
@@ -28,6 +28,7 @@ int btrfs_get_raid_extent_offset(struct btrfs_fs_info *fs_info,
u32 stripe_index, struct btrfs_io_stripe *stripe);
int btrfs_insert_raid_extent(struct btrfs_trans_handle *trans,
struct btrfs_ordered_extent *ordered_extent);
+void btrfs_cleanup_ordered_bioc_list(struct btrfs_ordered_extent *ordered);
#ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS
int btrfs_insert_one_raid_extent(struct btrfs_trans_handle *trans,
diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c
index f209e75f0ff5..c09d4213ad89 100644
--- a/fs/btrfs/scrub.c
+++ b/fs/btrfs/scrub.c
@@ -1023,6 +1023,10 @@ static void scrub_stripe_report_errors(struct scrub_ctx *sctx,
skip:
for_each_set_bit(sector_nr, &extent_bitmap, stripe->nr_sectors) {
+ const u64 sector_logical = stripe->logical +
+ ((u64)sector_nr << fs_info->sectorsize_bits);
+ const u64 sector_physical = physical +
+ ((u64)sector_nr << fs_info->sectorsize_bits);
bool repaired = false;
if (scrub_bitmap_test_bit_is_metadata(stripe, sector_nr)) {
@@ -1051,12 +1055,12 @@ skip:
if (dev) {
btrfs_err_rl(fs_info,
"scrub: fixed up error at logical %llu on dev %s physical %llu",
- stripe->logical, btrfs_dev_name(dev),
- physical);
+ sector_logical, btrfs_dev_name(dev),
+ sector_physical);
} else {
btrfs_err_rl(fs_info,
"scrub: fixed up error at logical %llu on mirror %u",
- stripe->logical, stripe->mirror_num);
+ sector_logical, stripe->mirror_num);
}
continue;
}
@@ -1065,30 +1069,30 @@ skip:
if (dev) {
btrfs_err_rl(fs_info,
"scrub: unable to fixup (regular) error at logical %llu on dev %s physical %llu",
- stripe->logical, btrfs_dev_name(dev),
- physical);
+ sector_logical, btrfs_dev_name(dev),
+ sector_physical);
} else {
btrfs_err_rl(fs_info,
"scrub: unable to fixup (regular) error at logical %llu on mirror %u",
- stripe->logical, stripe->mirror_num);
+ sector_logical, stripe->mirror_num);
}
if (scrub_bitmap_test_bit_io_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("i/o error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_csum_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("checksum error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_meta_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("header error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
if (scrub_bitmap_test_bit_meta_gen_error(stripe, sector_nr))
if (__ratelimit(&rs) && dev)
scrub_print_common_warning("generation error", dev, false,
- stripe->logical, physical);
+ sector_logical, sector_physical);
}
/* Update the device stats. */
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
index dca3570168c7..5c59b9abedcd 100644
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -2065,7 +2065,7 @@ static int will_overwrite_ref(struct send_ctx *sctx, u64 dir, u64 dir_gen,
ret = is_inode_existent(sctx, dir, dir_gen, NULL, &parent_root_dir_gen);
if (ret <= 0)
- return 0;
+ return ret;
/*
* If we have a parent root we need to verify that the parent dir was
@@ -6417,6 +6417,13 @@ static int process_extent(struct send_ctx *sctx,
if (S_ISLNK(sctx->cur_inode_mode))
return 0;
+ if (unlikely(!S_ISREG(sctx->cur_inode_mode))) {
+ btrfs_crit(sctx->send_root->fs_info,
+ "send: extent for non-regular inode %llu root %llu mode 0%llo",
+ key->objectid, btrfs_root_id(sctx->send_root),
+ sctx->cur_inode_mode & S_IFMT);
+ return -EUCLEAN;
+ }
if (sctx->parent_root && !sctx->cur_inode_new) {
ret = is_extent_unchanged(sctx, path, key);
diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c
index 464129b1b0d4..ddb620ac241b 100644
--- a/fs/btrfs/super.c
+++ b/fs/btrfs/super.c
@@ -1836,8 +1836,12 @@ static int btrfs_statfs(struct dentry *dentry, struct kstatfs *buf)
f_fsid.val[0] ^= btrfs_root_id(BTRFS_I(d_inode(dentry))->root) >> 32;
f_fsid.val[1] ^= btrfs_root_id(BTRFS_I(d_inode(dentry))->root);
- /* Hash dev_t to avoid f_fsid collision with cloned filesystems. */
- if (fs_info->fs_devices->total_devices == 1) {
+ /*
+ * Hash dev_t to avoid f_fsid collisions with cloned filesystems.
+ * Only do this when a clone is present so the original filesystem
+ * (mounted first) maintains backward-compatible f_fsid behavior.
+ */
+ if (fs_info->fs_devices->temp_fsid) {
__kernel_fsid_t dev_fsid =
u64_to_fsid(huge_encode_dev(fs_info->fs_devices->latest_dev->bdev->bd_dev));
diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c
index b2aacf846c8b..23459cd4e503 100644
--- a/fs/btrfs/tests/extent-io-tests.c
+++ b/fs/btrfs/tests/extent-io-tests.c
@@ -133,14 +133,14 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize)
if (IS_ERR(root)) {
test_std_err(TEST_ALLOC_ROOT);
ret = PTR_ERR(root);
- goto out;
+ goto out_root_info;
}
inode = btrfs_new_test_inode();
if (!inode) {
test_std_err(TEST_ALLOC_INODE);
ret = -ENOMEM;
- goto out;
+ goto out_root_info;
}
tmp = &BTRFS_I(inode)->io_tree;
BTRFS_I(inode)->root = root;
@@ -333,6 +333,7 @@ out:
process_page_range(inode, 0, total_dirty - 1,
PROCESS_UNLOCK | PROCESS_RELEASE);
iput(inode);
+out_root_info:
btrfs_free_dummy_root(root);
btrfs_free_dummy_fs_info(fs_info);
return ret;
diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c
index bafc62cf5ebc..6802b94ed76f 100644
--- a/fs/btrfs/transaction.c
+++ b/fs/btrfs/transaction.c
@@ -458,8 +458,19 @@ static int record_root_in_trans(struct btrfs_trans_handle *trans,
* through btrfs_record_root_in_trans without having to take the
* lock. smp_wmb() makes sure that all the writes above are
* done before we pop in the zero below
+ *
+ * If @force is true, it means the call is from
+ * qgroup_account_snapshot(), which only requires radix tree
+ * tracking.
+ * We should not force reloc root creation here, as the root
+ * may have already been modified, and in that case
+ * root->commit_root has already been dropped.
+ *
+ * Using that commit root will cause the reloc root to refer
+ * to a deleted extent, causing extent tree corruption.
*/
- ret = btrfs_init_reloc_root(trans, root);
+ if (!force)
+ ret = btrfs_init_reloc_root(trans, root);
smp_mb__before_atomic();
clear_bit(BTRFS_ROOT_IN_TRANS_SETUP, &root->state);
}
@@ -2583,6 +2594,12 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans)
ret = btrfs_write_and_wait_transaction(trans);
if (unlikely(ret)) {
btrfs_err(fs_info, "error while writing out transaction: %pe", ERR_PTR(ret));
+ /*
+ * Abort before releasing tree_log_mutex, so a log sync waiting
+ * on it sees the fs error and skips writing super_for_commit
+ * for this failed transaction. See btrfs_sync_log().
+ */
+ btrfs_abort_transaction(trans, ret);
mutex_unlock(&fs_info->tree_log_mutex);
goto scrub_continue;
}
diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c
index 0ce91396b517..4b1e47173c63 100644
--- a/fs/btrfs/tree-checker.c
+++ b/fs/btrfs/tree-checker.c
@@ -1909,6 +1909,16 @@ static int check_inode_ref(struct extent_buffer *leaf,
return -EUCLEAN;
}
+ if (unlikely(btrfs_is_fstree(btrfs_header_owner(leaf)) &&
+ (key->offset < BTRFS_FIRST_FREE_OBJECTID ||
+ key->offset > BTRFS_LAST_FREE_OBJECTID))) {
+ inode_ref_err(leaf, slot,
+ "invalid offset for ref key, have %llu expect [%llu, %lld]",
+ key->offset, BTRFS_FIRST_FREE_OBJECTID,
+ BTRFS_LAST_FREE_OBJECTID);
+ return -EUCLEAN;
+ }
+
ptr = btrfs_item_ptr_offset(leaf, slot);
end = ptr + btrfs_item_size(leaf, slot);
while (ptr < end) {
@@ -1952,12 +1962,14 @@ static int check_inode_extref(struct extent_buffer *leaf,
{
unsigned long ptr = btrfs_item_ptr_offset(leaf, slot);
unsigned long end = ptr + btrfs_item_size(leaf, slot);
+ const bool is_fstree = btrfs_is_fstree(btrfs_header_owner(leaf));
if (unlikely(!check_prev_ino(leaf, key, slot, prev_key)))
return -EUCLEAN;
while (ptr < end) {
struct btrfs_inode_extref *extref = (struct btrfs_inode_extref *)ptr;
+ u64 parent;
u16 namelen;
if (unlikely(ptr + sizeof(*extref) > end)) {
@@ -1967,7 +1979,24 @@ static int check_inode_extref(struct extent_buffer *leaf,
return -EUCLEAN;
}
+ parent = btrfs_inode_extref_parent(leaf, extref);
+ if (unlikely(is_fstree && (parent < BTRFS_FIRST_FREE_OBJECTID ||
+ parent > BTRFS_LAST_FREE_OBJECTID))) {
+ inode_ref_err(leaf, slot,
+ "invalid parent for extref key, have %llu expect [%llu, %lld]",
+ parent, BTRFS_FIRST_FREE_OBJECTID,
+ BTRFS_LAST_FREE_OBJECTID);
+ return -EUCLEAN;
+ }
+
namelen = btrfs_inode_extref_name_len(leaf, extref);
+ if (unlikely(namelen == 0 || namelen > BTRFS_NAME_LEN)) {
+ inode_ref_err(leaf, slot,
+ "invalid inode extref name length, has %u expect [1, %u]",
+ namelen, BTRFS_NAME_LEN);
+ return -EUCLEAN;
+ }
+
if (unlikely(ptr + sizeof(*extref) + namelen > end)) {
inode_ref_err(leaf, slot,
"inode extref overflow, ptr %lu end %lu namelen %u",
@@ -2100,7 +2129,7 @@ static int check_dev_extent_item(const struct extent_buffer *leaf,
sectorsize))) {
generic_err(leaf, slot,
"invalid dev extent chunk offset, has %llu not aligned to %u",
- btrfs_dev_extent_chunk_objectid(leaf, de),
+ btrfs_dev_extent_chunk_offset(leaf, de),
sectorsize);
return -EUCLEAN;
}
@@ -2277,7 +2306,7 @@ static int check_free_space_extent(struct extent_buffer *leaf, struct btrfs_key
if (unlikely(btrfs_item_size(leaf, slot) != 0)) {
generic_err(leaf, slot,
- "invalid item size for free space info, has %u expect 0",
+ "invalid item size for free space extent, has %u expect 0",
btrfs_item_size(leaf, slot));
return -EUCLEAN;
}
diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c
index 7ba7b6098aa5..a00094604e54 100644
--- a/fs/btrfs/tree-log.c
+++ b/fs/btrfs/tree-log.c
@@ -7286,6 +7286,22 @@ static int btrfs_log_all_parents(struct btrfs_trans_handle *trans,
ret = btrfs_search_slot(NULL, root, &key, path, 0, 0);
if (ret < 0)
goto out;
+ /*
+ * There can't be an inode ref key with offset 0 because inode numbers
+ * start at BTRFS_FIRST_FREE_OBJECTID.
+ */
+ if (WARN_ON_ONCE(ret == 0)) {
+ btrfs_err(trans->fs_info,
+ "found inode ref key with offset 0 for root %llu inode %llu",
+ btrfs_root_id(root), ino);
+ ret = BTRFS_LOG_FORCE_COMMIT;
+ goto out;
+ }
+ /*
+ * Set to 0 so that in case we don't do any work below, we won't return
+ * 1 and trigger an unnecessary transaction commit.
+ */
+ ret = 0;
while (true) {
struct extent_buffer *leaf = path->nodes[0];
diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c
index 4e0ab5842274..600337a84fbe 100644
--- a/fs/btrfs/verity.c
+++ b/fs/btrfs/verity.c
@@ -94,6 +94,20 @@ static loff_t merkle_file_pos(const struct inode *inode)
}
/*
+ * Start a transaction for removing verity items or the verity orphan.
+ *
+ * Like unlink, this only deletes items and frees space in the end, so the
+ * reservation may come from the global reserve when the filesystem is full
+ * (-ENOSPC) and is not subject to the qgroup limit (-EDQUOT). Otherwise a
+ * failed enable could never be cleaned up in either situation.
+ */
+static struct btrfs_trans_handle *start_verity_cleanup_trans(struct btrfs_root *root,
+ unsigned int num_items)
+{
+ return btrfs_start_transaction_fallback_global_rsv(root, num_items);
+}
+
+/*
* Drop all the items for this inode with this key_type.
*
* @inode: inode to drop items for
@@ -120,7 +134,7 @@ static int drop_verity_items(struct btrfs_inode *inode, u8 key_type)
while (1) {
/* 1 for the item being dropped */
- trans = btrfs_start_transaction(root, 1);
+ trans = start_verity_cleanup_trans(root, 1);
if (IS_ERR(trans))
return PTR_ERR(trans);
@@ -466,7 +480,7 @@ static int rollback_verity(struct btrfs_inode *inode)
* 1 for updating the inode flag
* 1 for deleting the orphan
*/
- trans = btrfs_start_transaction(root, 2);
+ trans = start_verity_cleanup_trans(root, 2);
if (IS_ERR(trans)) {
ret = PTR_ERR(trans);
trans = NULL;
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
index 9b66eb584ece..85ea9c5d4536 100644
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -749,6 +749,36 @@ const u8 *btrfs_sb_fsid_ptr(const struct btrfs_super_block *sb)
return has_metadata_uuid ? sb->metadata_uuid : sb->fsid;
}
+static bool should_rename_device(const struct btrfs_device *dev)
+{
+ bool ret;
+ const char *old_name;
+
+ rcu_read_lock();
+ old_name = rcu_dereference(dev->name);
+ /*
+ * For systems booted without an initramfs, the rootfs has the device
+ * name "/dev/root".
+ *
+ * Although using btrfs without an initramfs is not recommended (if a
+ * new device is added to the rootfs, the system can no longer boot, as
+ * there is no way to register all devices), there is still a minority
+ * of users doing this.
+ *
+ * And after the system is up, a later device scan on the real block
+ * device file will never get this device's name updated, as the
+ * device->devt is still the same.
+ *
+ * Here we add one and only one exception for "/dev/root", to allow the
+ * device name to be updated even if the new path points to the same
+ * block device.
+ */
+ ret = (strcmp(old_name, "/dev/root") == 0);
+ rcu_read_unlock();
+
+ return ret;
+}
+
/*
* Add new device to list of registered devices
*
@@ -869,7 +899,8 @@ static noinline struct btrfs_device *device_list_add(const char *path,
MAJOR(path_devt), MINOR(path_devt),
current->comm, task_pid_nr(current));
- } else if (!device->name || device->devt != path_devt) {
+ } else if (!device->name || device->devt != path_devt ||
+ should_rename_device(device)) {
const char *old_name;
/*
@@ -3117,7 +3148,11 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path
error_sysfs:
btrfs_sysfs_remove_device(device);
mutex_lock(&fs_info->fs_devices->device_list_mutex);
+ if (seeding_dev)
+ btrfs_assign_next_active_device(device, seed_devices->latest_dev);
mutex_lock(&fs_info->chunk_mutex);
+ if (!list_empty(&device->post_commit_list))
+ list_del_init(&device->post_commit_list);
list_del_rcu(&device->dev_list);
list_del(&device->dev_alloc_list);
fs_info->fs_devices->num_devices--;
diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c
index a016cb471beb..08a15465a087 100644
--- a/fs/btrfs/zoned.c
+++ b/fs/btrfs/zoned.c
@@ -2626,16 +2626,13 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ
down_read(&dev_replace->rwsem);
map = block_group->physical_map;
for (i = 0; i < map->num_stripes; i++) {
-
ret = call_zone_finish(block_group, &map->stripes[i]);
- if (ret) {
- up_read(&dev_replace->rwsem);
- return ret;
- }
+ if (ret)
+ break;
}
up_read(&dev_replace->rwsem);
- if (!fully_written)
+ if (!ret && !fully_written)
btrfs_dec_block_group_ro(block_group);
spin_lock(&fs_info->zone_active_bgs_lock);
@@ -2648,7 +2645,7 @@ static int do_zone_finish(struct btrfs_block_group *block_group, bool fully_writ
clear_and_wake_up_bit(BTRFS_FS_NEED_ZONE_FINISH, &fs_info->flags);
- return 0;
+ return ret;
}
int btrfs_zone_finish(struct btrfs_block_group *block_group)
@@ -2691,6 +2688,11 @@ bool btrfs_can_activate_zone(struct btrfs_fs_devices *fs_devices, u64 flags)
switch (flags & BTRFS_BLOCK_GROUP_PROFILE_MASK) {
case 0: /* single */
+ case BTRFS_BLOCK_GROUP_RAID0:
+ case BTRFS_BLOCK_GROUP_RAID1:
+ case BTRFS_BLOCK_GROUP_RAID1C3:
+ case BTRFS_BLOCK_GROUP_RAID1C4:
+ case BTRFS_BLOCK_GROUP_RAID10:
ret = (atomic_read(&zinfo->active_zones_left) >= (1 + reserved));
break;
case BTRFS_BLOCK_GROUP_DUP:
@@ -2713,6 +2715,7 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
{
struct btrfs_block_group *block_group;
u64 min_alloc_bytes;
+ int ret = 0;
if (!btrfs_is_zoned(fs_info))
return 0;
@@ -2732,11 +2735,11 @@ int btrfs_zone_finish_endio(struct btrfs_fs_info *fs_info, u64 logical, u64 leng
block_group->start + block_group->zone_capacity)
goto out;
- do_zone_finish(block_group, true);
+ ret = do_zone_finish(block_group, true);
out:
btrfs_put_block_group(block_group);
- return 0;
+ return ret;
}
static void btrfs_zone_finish_endio_workfn(struct work_struct *work)
diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c
index 86919293fd54..58d9ff76fe07 100644
--- a/fs/btrfs/zstd.c
+++ b/fs/btrfs/zstd.c
@@ -307,8 +307,17 @@ again:
DEFINE_WAIT(wait);
prepare_to_wait(&zwsm->wait, &wait, TASK_UNINTERRUPTIBLE);
- schedule();
+ /*
+ * Re-check after being queued: zstd_put_workspace() only wakes
+ * a queue that already has a sleeper, so a workspace returned
+ * since the failed allocation woke nobody.
+ */
+ ws = zstd_find_workspace(fs_info, level);
+ if (!ws)
+ schedule();
finish_wait(&zwsm->wait, &wait);
+ if (ws)
+ return ws;
goto again;
}
diff --git a/fs/cachefiles/xattr.c b/fs/cachefiles/xattr.c
index f8ae78b3f7b6..c70bf67e52b0 100644
--- a/fs/cachefiles/xattr.c
+++ b/fs/cachefiles/xattr.c
@@ -13,6 +13,7 @@
#include <linux/quotaops.h>
#include <linux/xattr.h>
#include <linux/slab.h>
+#include <linux/unaligned.h>
#include "internal.h"
#define CACHEFILES_COOKIE_TYPE_DATA 1
@@ -50,7 +51,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
_enter("%x,#%d", object->debug_id, len);
- buf = kmalloc(sizeof(struct cachefiles_xattr) + len, GFP_KERNEL);
+ buf = kmalloc(sizeof(struct cachefiles_xattr) + max(len, sizeof(__be64)), GFP_KERNEL);
if (!buf)
return -ENOMEM;
@@ -60,6 +61,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
buf->content = object->content_info;
if (test_bit(FSCACHE_COOKIE_LOCAL_WRITE, &object->cookie->flags))
buf->content = CACHEFILES_CONTENT_DIRTY;
+ put_unaligned_be64(0, (__be64 *)buf->data);
if (len > 0)
memcpy(buf->data, fscache_get_aux(object->cookie), len);
@@ -77,8 +79,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
trace_cachefiles_vfs_error(object, file_inode(file), ret,
cachefiles_trace_setxattr_error);
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
- be64_to_cpup((__be64 *)buf->data),
- buf->content,
+ buf->data, buf->content,
cachefiles_coherency_set_fail);
if (ret != -ENOMEM)
cachefiles_io_error_obj(
@@ -86,8 +87,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
"Failed to set xattr with error %d", ret);
} else {
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
- be64_to_cpup((__be64 *)buf->data),
- buf->content,
+ buf->data, buf->content,
cachefiles_coherency_set_ok);
}
@@ -110,9 +110,10 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file
int ret = -ESTALE;
tlen = sizeof(struct cachefiles_xattr) + len;
- buf = kmalloc(tlen, GFP_KERNEL);
+ buf = kmalloc(sizeof(struct cachefiles_xattr) + max(len, sizeof(__be64)), GFP_KERNEL);
if (!buf)
return -ENOMEM;
+ put_unaligned_be64(0, (__be64 *)buf->data);
xlen = cachefiles_inject_read_error();
if (xlen == 0)
@@ -148,8 +149,7 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file
out:
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
- be64_to_cpup((__be64 *)buf->data),
- buf->content, why);
+ buf->data, buf->content, why);
kfree(buf);
return ret;
}
diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c
index 657c2cb0f881..e598b2d424ec 100644
--- a/fs/ceph/addr.c
+++ b/fs/ceph/addr.c
@@ -2546,7 +2546,7 @@ static int __ceph_pool_perm_get(struct ceph_inode_info *ci,
}
pool_ns_len = pool_ns ? pool_ns->len : 0;
- perm = kmalloc_flex(*perm, pool_ns, pool_ns_len + 1, GFP_KERNEL);
+ perm = kmalloc_flex(*perm, pool_ns, pool_ns_len + 1);
if (!perm) {
err = -ENOMEM;
goto out_unlock;
diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
index a091f77cedaf..085ae0cfb5f7 100644
--- a/fs/ceph/mds_client.c
+++ b/fs/ceph/mds_client.c
@@ -5492,7 +5492,7 @@ static void ceph_mdsc_reset_workfn(struct work_struct *work)
goto out_complete;
}
- sessions = kcalloc(max_sessions, sizeof(*sessions), GFP_KERNEL);
+ sessions = kzalloc_objs(*sessions, max_sessions);
if (!sessions) {
mutex_unlock(&mdsc->mutex);
ret = -ENOMEM;
@@ -6600,11 +6600,13 @@ int ceph_mds_check_access(struct ceph_mds_client *mdsc, char *tpath, int mask)
doutc(cl, "tpath '%s', mask %d, caller_uid %d, caller_gid %d\n",
tpath, mask, caller_uid, caller_gid);
+ mutex_lock(&mdsc->mutex);
for (i = 0; i < mdsc->s_cap_auths_num; i++) {
struct ceph_mds_cap_auth *s = &mdsc->s_cap_auths[i];
err = ceph_mds_auth_match(mdsc, s, cred, tpath);
if (err < 0) {
+ mutex_unlock(&mdsc->mutex);
put_cred(cred);
return err;
} else if (err > 0) {
@@ -6626,6 +6628,7 @@ int ceph_mds_check_access(struct ceph_mds_client *mdsc, char *tpath, int mask)
doutc(cl, "root_squash_perms %d, rw_perms_s %p\n", root_squash_perms,
rw_perms_s);
if (root_squash_perms && rw_perms_s == NULL) {
+ mutex_unlock(&mdsc->mutex);
doutc(cl, "access allowed\n");
return 0;
}
@@ -6640,6 +6643,7 @@ int ceph_mds_check_access(struct ceph_mds_client *mdsc, char *tpath, int mask)
!!(mask & MAY_READ), !!(mask & MAY_WRITE));
}
doutc(cl, "access denied\n");
+ mutex_unlock(&mdsc->mutex);
return -EACCES;
}
diff --git a/fs/ceph/mds_client.h b/fs/ceph/mds_client.h
index 3c62e3c3530b..e7a262c9c2ab 100644
--- a/fs/ceph/mds_client.h
+++ b/fs/ceph/mds_client.h
@@ -604,6 +604,7 @@ struct ceph_mds_client {
struct rw_semaphore pool_perm_rwsem;
struct rb_root pool_perm_tree;
+ /* protected by mutex */
u32 s_cap_auths_num;
struct ceph_mds_cap_auth *s_cap_auths;
diff --git a/fs/ceph/subvolume_metrics.c b/fs/ceph/subvolume_metrics.c
index 03fda1f9257b..01419c9482f1 100644
--- a/fs/ceph/subvolume_metrics.c
+++ b/fs/ceph/subvolume_metrics.c
@@ -245,7 +245,7 @@ int ceph_subvolume_metrics_snapshot(struct ceph_subvolume_metrics_tracker *track
return 0;
}
- snap = kcalloc(count, sizeof(*snap), GFP_NOFS);
+ snap = kzalloc_objs(*snap, count, GFP_NOFS);
if (!snap) {
atomic64_inc(&tracker->snapshot_failures);
return -ENOMEM;
diff --git a/fs/ceph/super.c b/fs/ceph/super.c
index 15edea30dc8b..72935f665f11 100644
--- a/fs/ceph/super.c
+++ b/fs/ceph/super.c
@@ -1420,6 +1420,11 @@ static int ceph_reconfigure_fc(struct fs_context *fc)
else
ceph_clear_mount_opt(fsc, SPARSEREAD);
+ if (fsopt->flags & CEPH_MOUNT_OPT_NEARFULL_SYNC)
+ ceph_set_mount_opt(fsc, NEARFULL_SYNC);
+ else
+ ceph_clear_mount_opt(fsc, NEARFULL_SYNC);
+
if (strcmp_null(fsc->mount_options->mon_addr, fsopt->mon_addr)) {
kfree(fsc->mount_options->mon_addr);
fsc->mount_options->mon_addr = fsopt->mon_addr;
diff --git a/fs/configfs/dir.c b/fs/configfs/dir.c
index 3c88f13f1ca2..eda80c2a2d38 100644
--- a/fs/configfs/dir.c
+++ b/fs/configfs/dir.c
@@ -416,6 +416,15 @@ static void configfs_remove_dir(struct dentry *d)
if (d_really_is_positive(d)) {
if (unlikely(simple_rmdir(d_inode(parent), d)))
pr_warn("remove_dir (%pd): attributes remain", d);
+ else
+ /*
+ * configfs_get_config_item() takes a hashed dentry as
+ * proof that ->s_element is still alive. Our caller
+ * is about to drop the last reference to the item and
+ * the VFS will not unhash until after we return, so
+ * unhash it here.
+ */
+ d_drop(d);
}
pr_debug(" o %pd removing done (%d)\n", d, d_count(d));
diff --git a/fs/configfs/mount.c b/fs/configfs/mount.c
index 4929f3431189..d8cac1cbf3bd 100644
--- a/fs/configfs/mount.c
+++ b/fs/configfs/mount.c
@@ -9,6 +9,7 @@
*/
#include <linux/fs.h>
+#include <linux/magic.h>
#include <linux/module.h>
#include <linux/mount.h>
#include <linux/fs_context.h>
@@ -19,9 +20,6 @@
#include <linux/configfs.h>
#include "configfs_internal.h"
-/* Random magic number */
-#define CONFIGFS_MAGIC 0x62656570
-
static struct vfsmount *configfs_mount = NULL;
struct kmem_cache *configfs_dir_cachep;
static int configfs_mnt_count = 0;
diff --git a/fs/configfs/symlink.c b/fs/configfs/symlink.c
index 31eb28b27309..3b31c714400f 100644
--- a/fs/configfs/symlink.c
+++ b/fs/configfs/symlink.c
@@ -76,9 +76,9 @@ static int configfs_get_target_path(struct config_item *item,
static int create_link(struct config_item *parent_item,
struct config_item *item,
+ struct configfs_dirent *target_sd,
struct dentry *dentry)
{
- struct configfs_dirent *target_sd = item->ci_dentry->d_fsdata;
char *body;
int ret;
@@ -115,6 +115,7 @@ static int create_link(struct config_item *parent_item,
static int get_target(const char *symname, struct config_item **target,
+ struct configfs_dirent **target_sd,
struct super_block *sb)
{
struct path path __free(path_put) = {};
@@ -125,7 +126,20 @@ static int get_target(const char *symname, struct config_item **target,
return ret;
if (path.dentry->d_sb != sb)
return -EPERM;
- *target = configfs_get_config_item(path.dentry);
+ /*
+ * A hashed dentry guarantees that neither the item nor the dirent
+ * have been released yet, as removals unhash before dropping.
+ * Grab both references here. An item reference alone would not keep
+ * ->ci_dentry alive.
+ */
+ spin_lock(&path.dentry->d_lock);
+ if (!d_unhashed(path.dentry)) {
+ struct configfs_dirent *sd = path.dentry->d_fsdata;
+
+ *target = config_item_get(sd->s_element);
+ *target_sd = configfs_get(sd);
+ }
+ spin_unlock(&path.dentry->d_lock);
if (!*target)
return -ENOENT;
return 0;
@@ -139,6 +153,7 @@ int configfs_symlink(struct mnt_idmap *idmap, struct inode *dir,
struct configfs_dirent *sd;
struct config_item *parent_item;
struct config_item *target_item = NULL;
+ struct configfs_dirent *target_sd = NULL;
const struct config_item_type *type;
sd = dentry->d_parent->d_fsdata;
@@ -182,7 +197,7 @@ int configfs_symlink(struct mnt_idmap *idmap, struct inode *dir,
* AV, a thoroughly annoyed bastard.
*/
inode_unlock(dir);
- ret = get_target(symname, &target_item, dentry->d_sb);
+ ret = get_target(symname, &target_item, &target_sd, dentry->d_sb);
inode_lock(dir);
if (ret)
goto out_put;
@@ -196,13 +211,14 @@ int configfs_symlink(struct mnt_idmap *idmap, struct inode *dir,
ret = type->ct_item_ops->allow_link(parent_item, target_item);
if (!ret) {
mutex_lock(&configfs_symlink_mutex);
- ret = create_link(parent_item, target_item, dentry);
+ ret = create_link(parent_item, target_item, target_sd, dentry);
mutex_unlock(&configfs_symlink_mutex);
if (ret && type->ct_item_ops->drop_link)
type->ct_item_ops->drop_link(parent_item,
target_item);
}
+ configfs_put(target_sd);
config_item_put(target_item);
out_put:
diff --git a/fs/coredump.c b/fs/coredump.c
index ac3cd74808c6..6114839f5178 100644
--- a/fs/coredump.c
+++ b/fs/coredump.c
@@ -1000,7 +1000,7 @@ static bool coredump_pipe(struct core_name *cn, struct coredump_params *cprm,
return false;
}
- helper_argv = kmalloc_array(argc + 1, sizeof(*helper_argv), GFP_KERNEL);
+ helper_argv = kmalloc_objs(*helper_argv, argc + 1);
if (!helper_argv) {
coredump_report_failure("%s failed to allocate memory", __func__);
return false;
diff --git a/fs/dax.c b/fs/dax.c
index 6ba50142eeb2..1fbba0d21c13 100644
--- a/fs/dax.c
+++ b/fs/dax.c
@@ -480,11 +480,12 @@ static void dax_associate_entry(void *entry, struct address_space *mapping,
unsigned long address, bool shared)
{
unsigned long size = dax_entry_size(entry), index;
- struct folio *folio = dax_to_folio(entry);
+ struct folio *folio;
if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry))
return;
+ folio = dax_to_folio(entry);
index = linear_page_index(vma, address & ~(size - 1));
if (shared && (folio->mapping || dax_folio_is_shared(folio))) {
if (folio->mapping)
@@ -505,21 +506,23 @@ static void dax_associate_entry(void *entry, struct address_space *mapping,
static void dax_disassociate_entry(void *entry, struct address_space *mapping,
bool trunc)
{
- struct folio *folio = dax_to_folio(entry);
+ struct folio *folio;
if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry))
return;
+ folio = dax_to_folio(entry);
dax_folio_put(folio);
}
static struct page *dax_busy_page(void *entry)
{
- struct folio *folio = dax_to_folio(entry);
+ struct folio *folio;
if (dax_is_zero_entry(entry) || dax_is_empty_entry(entry))
return NULL;
+ folio = dax_to_folio(entry);
if (folio_ref_count(folio) - folio_mapcount(folio))
return &folio->page;
else
diff --git a/fs/erofs/data.c b/fs/erofs/data.c
index 0885b1f2fc92..be63b89f0862 100644
--- a/fs/erofs/data.c
+++ b/fs/erofs/data.c
@@ -48,7 +48,7 @@ void *erofs_bread(struct erofs_buf *buf, erofs_off_t offset, bool need_kmap)
return NULL;
if (!buf->base)
buf->base = kmap_local_page(buf->page);
- return buf->base + (offset & ~PAGE_MASK);
+ return buf->base + ((buf->off + offset) & ~PAGE_MASK);
}
int erofs_init_metabuf(struct erofs_buf *buf, struct super_block *sb,
diff --git a/fs/erofs/decompressor.c b/fs/erofs/decompressor.c
index 27caf4bebddc..d387b27c4ee2 100644
--- a/fs/erofs/decompressor.c
+++ b/fs/erofs/decompressor.c
@@ -7,8 +7,6 @@
#include "compress.h"
#include <linux/lz4.h>
-#define LZ4_MAX_DISTANCE_PAGES (DIV_ROUND_UP(LZ4_DISTANCE_MAX, PAGE_SIZE) + 1)
-
static int z_erofs_load_lz4_config(struct super_block *sb,
struct erofs_super_block *dsb, void *data, int size)
{
@@ -21,8 +19,6 @@ static int z_erofs_load_lz4_config(struct super_block *sb,
erofs_err(sb, "invalid lz4 cfgs, size=%u", size);
return -EINVAL;
}
- distance = le16_to_cpu(lz4->max_distance);
-
sbi->lz4.max_pclusterblks = le16_to_cpu(lz4->max_pclusterblks);
if (!sbi->lz4.max_pclusterblks) {
sbi->lz4.max_pclusterblks = 1; /* reserved case */
@@ -39,45 +35,25 @@ static int z_erofs_load_lz4_config(struct super_block *sb,
sbi->lz4.max_pclusterblks = 1;
sbi->available_compr_algs = 1 << Z_EROFS_COMPRESSION_LZ4;
}
-
- sbi->lz4.max_distance_pages = distance ?
- DIV_ROUND_UP(distance, PAGE_SIZE) + 1 :
- LZ4_MAX_DISTANCE_PAGES;
return z_erofs_gbuf_growsize(sbi->lz4.max_pclusterblks);
}
/*
- * Fill all gaps with bounce pages if it's a sparse page list. Also check if
- * all physical pages are consecutive, which can be seen for moderate CR.
+ * Fill all gaps with bounce pages if it's a sparse page list (for example some
+ * folios are already uptodate and thus can be mapped into userspace). Also
+ * check if pages are physically consecutive, which can be seen for moderate CR.
*/
-static int z_erofs_lz4_prepare_dstpages(struct z_erofs_decompress_req *rq,
- struct page **pagepool)
+static int z_erofs_oneshot_prepare_dstpages(struct z_erofs_decompress_req *rq,
+ struct page **pagepool)
{
- struct page *availables[LZ4_MAX_DISTANCE_PAGES] = { NULL };
- unsigned long bounced[DIV_ROUND_UP(LZ4_MAX_DISTANCE_PAGES,
- BITS_PER_LONG)] = { 0 };
- unsigned int lz4_max_distance_pages =
- EROFS_SB(rq->sb)->lz4.max_distance_pages;
void *kaddr = NULL;
- unsigned int i, j, top;
+ unsigned int i;
- top = 0;
- for (i = j = 0; i < rq->outpages; ++i, ++j) {
- struct page *const page = rq->out[i];
- struct page *victim;
-
- if (j >= lz4_max_distance_pages)
- j = 0;
-
- /* 'valid' bounced can only be tested after a complete round */
- if (!rq->fillgaps && test_bit(j, bounced)) {
- DBG_BUGON(i < lz4_max_distance_pages);
- DBG_BUGON(top >= lz4_max_distance_pages);
- availables[top++] = rq->out[i - lz4_max_distance_pages];
- }
+ for (i = 0; i < rq->outpages; ++i) {
+ struct page *page, *victim;
+ page = rq->out[i];
if (page) {
- __clear_bit(j, bounced);
if (!PageHighMem(page)) {
if (!i) {
kaddr = page_address(page);
@@ -89,21 +65,14 @@ static int z_erofs_lz4_prepare_dstpages(struct z_erofs_decompress_req *rq,
continue;
}
}
- kaddr = NULL;
- continue;
- }
- kaddr = NULL;
- __set_bit(j, bounced);
-
- if (top) {
- victim = availables[--top];
} else {
victim = __erofs_allocpage(pagepool, rq->gfp, true);
if (!victim)
return -ENOMEM;
set_page_private(victim, Z_EROFS_SHORTLIVED_PAGE);
+ rq->out[i] = victim;
}
- rq->out[i] = victim;
+ kaddr = NULL;
}
return kaddr ? 1 : 0;
}
@@ -266,7 +235,7 @@ static const char *z_erofs_lz4_decompress(struct z_erofs_decompress_req *rq,
dst_maptype = 0;
} else {
/* general decoding path which can be used for all cases */
- ret = z_erofs_lz4_prepare_dstpages(rq, pagepool);
+ ret = z_erofs_oneshot_prepare_dstpages(rq, pagepool);
if (ret < 0)
return ERR_PTR(ret);
if (ret > 0) {
diff --git a/fs/erofs/decompressor_lzma.c b/fs/erofs/decompressor_lzma.c
index 6b0cdb446c6a..9d15f94cbee1 100644
--- a/fs/erofs/decompressor_lzma.c
+++ b/fs/erofs/decompressor_lzma.c
@@ -5,6 +5,7 @@
struct z_erofs_lzma {
struct z_erofs_lzma *next;
struct xz_dec_microlzma *state;
+ unsigned int dict_size;
u8 bounce[PAGE_SIZE];
};
@@ -128,11 +129,19 @@ again:
err = 0;
/* 2. walk each isolated stream and grow max dict_size if needed */
for (strm = head; strm; strm = strm->next) {
+ struct xz_dec_microlzma *state;
+
+ if (strm->dict_size >= dict_size)
+ continue;
+ state = xz_dec_microlzma_alloc(XZ_PREALLOC, dict_size);
+ if (!state) {
+ err = -ENOMEM;
+ break;
+ }
if (strm->state)
xz_dec_microlzma_end(strm->state);
- strm->state = xz_dec_microlzma_alloc(XZ_PREALLOC, dict_size);
- if (!strm->state)
- err = -ENOMEM;
+ strm->state = state;
+ strm->dict_size = dict_size;
}
/* 3. push back all to the global list and update max dict_size */
@@ -142,7 +151,8 @@ again:
spin_unlock(&z_erofs_lzma_lock);
wake_up_all(&z_erofs_lzma_wq);
- z_erofs_lzma_max_dictsize = dict_size;
+ if (!err)
+ z_erofs_lzma_max_dictsize = dict_size;
mutex_unlock(&lzma_resize_mutex);
return err;
}
diff --git a/fs/erofs/internal.h b/fs/erofs/internal.h
index 65974e57aebf..12e3a5b80a5a 100644
--- a/fs/erofs/internal.h
+++ b/fs/erofs/internal.h
@@ -71,12 +71,8 @@ struct erofs_dev_context {
bool flatdev;
};
-/* all filesystem-wide lz4 configurations */
struct erofs_sb_lz4_info {
- /* # of pages needed for EROFS lz4 rolling decompression */
- u16 max_distance_pages;
- /* maximum possible blocks for pclusters in the filesystem */
- u16 max_pclusterblks;
+ u16 max_pclusterblks; /* maximum physical blocks for LZ4 pclusters */
};
struct erofs_xattr_prefix_item {
diff --git a/fs/erofs/sysfs.c b/fs/erofs/sysfs.c
index 6734483a440f..dfcec9376cd5 100644
--- a/fs/erofs/sysfs.c
+++ b/fs/erofs/sysfs.c
@@ -95,6 +95,7 @@ EROFS_ATTR_FEATURE(sb_chksum);
EROFS_ATTR_FEATURE(ztailpacking);
EROFS_ATTR_FEATURE(fragments);
EROFS_ATTR_FEATURE(dedupe);
+EROFS_ATTR_FEATURE(xattr_prefixes);
EROFS_ATTR_FEATURE(48bit);
EROFS_ATTR_FEATURE(metabox);
@@ -108,6 +109,7 @@ static struct attribute *erofs_feat_attrs[] = {
ATTR_LIST(ztailpacking),
ATTR_LIST(fragments),
ATTR_LIST(dedupe),
+ ATTR_LIST(xattr_prefixes),
ATTR_LIST(48bit),
ATTR_LIST(metabox),
NULL,
diff --git a/fs/erofs/xattr.c b/fs/erofs/xattr.c
index df7ea019526d..57cfb7520782 100644
--- a/fs/erofs/xattr.c
+++ b/fs/erofs/xattr.c
@@ -620,8 +620,8 @@ int erofs_xattr_fill_inode_fingerprint(struct erofs_inode_fingerprint *fp,
{
struct erofs_sb_info *sbi = EROFS_SB(inode->i_sb);
struct erofs_xattr_prefix_item *prefix;
+ int domainlen, valuelen, base_index;
const char *infix;
- int valuelen, base_index;
if (!test_opt(&sbi->opt, INODE_SHARE))
return -EOPNOTSUPP;
@@ -633,17 +633,18 @@ int erofs_xattr_fill_inode_fingerprint(struct erofs_inode_fingerprint *fp,
valuelen = erofs_getxattr(inode, base_index, infix, NULL, 0);
if (valuelen <= 0 || valuelen > (1 << sbi->blkszbits))
return -EFSCORRUPTED;
- fp->size = valuelen + (domain_id ? strlen(domain_id) : 0);
+ domainlen = strlen(domain_id);
+ fp->size = domainlen + 1 + valuelen;
fp->opaque = kmalloc(fp->size, GFP_KERNEL);
if (!fp->opaque)
return -ENOMEM;
+ memcpy(fp->opaque, domain_id, domainlen + 1);
if (valuelen != erofs_getxattr(inode, base_index, infix,
- fp->opaque, valuelen)) {
+ fp->opaque + domainlen + 1, valuelen)) {
kfree(fp->opaque);
fp->opaque = NULL;
return -EFSCORRUPTED;
}
- memcpy(fp->opaque + valuelen, domain_id, fp->size - valuelen);
return 0;
}
#endif
diff --git a/fs/erofs/zdata.c b/fs/erofs/zdata.c
index e1e25ca0d190..6b07e73ee2aa 100644
--- a/fs/erofs/zdata.c
+++ b/fs/erofs/zdata.c
@@ -1259,7 +1259,7 @@ static int z_erofs_decompress_pcluster(struct z_erofs_backend *be, bool eio)
const struct z_erofs_decompressor *alg =
z_erofs_decomp[pcl->algorithmformat];
bool try_free = true;
- int i, j, jtop, err2, err = eio ? -EIO : 0;
+ int i, err2, err = eio ? -EIO : 0;
struct page *page;
bool overlapped;
const char *reason;
@@ -1348,7 +1348,6 @@ static int z_erofs_decompress_pcluster(struct z_erofs_backend *be, bool eio)
be->compressed_pages >= be->onstack_pages + Z_EROFS_ONSTACK_PAGES)
kvfree(be->compressed_pages);
- jtop = 0;
z_erofs_fill_other_copies(be, err);
for (i = 0; i < be->nr_pages; ++i) {
page = be->decompressed_pages[i];
@@ -1356,22 +1355,11 @@ static int z_erofs_decompress_pcluster(struct z_erofs_backend *be, bool eio)
continue;
DBG_BUGON(z_erofs_page_is_invalidated(page));
- if (!z_erofs_is_shortlived_page(page)) {
+ if (!z_erofs_is_shortlived_page(page))
erofs_onlinefolio_end(page_folio(page), err, true);
- continue;
- }
- if (pcl->algorithmformat != Z_EROFS_COMPRESSION_LZ4) {
+ else
erofs_pagepool_add(be->pagepool, page);
- continue;
- }
- for (j = 0; j < jtop && be->decompressed_pages[j] != page; ++j)
- ;
- if (j >= jtop) /* this bounce page is newly detected */
- be->decompressed_pages[jtop++] = page;
}
- while (jtop)
- erofs_pagepool_add(be->pagepool,
- be->decompressed_pages[--jtop]);
if (be->decompressed_pages != be->onstack_pages)
kvfree(be->decompressed_pages);
diff --git a/fs/exec.c b/fs/exec.c
index 745f6eb5279e..819643408e6d 100644
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1115,6 +1115,17 @@ static struct file *bprm_identity_file(const struct linux_binprm *bprm)
return bprm->file;
}
+static void posixtimer_exec(struct task_struct *me)
+{
+#ifdef CONFIG_POSIX_TIMERS
+ spin_lock_irq(&me->sighand->siglock);
+ posix_cpu_timers_exit(me);
+ spin_unlock_irq(&me->sighand->siglock);
+ exit_itimers(me);
+ flush_itimer_signals();
+#endif
+}
+
/*
* Calling this is the point of no return. None of the failures will be
* seen by userspace since either the process is already taking a fatal
@@ -1152,6 +1163,16 @@ int begin_new_exec(struct linux_binprm * bprm)
retval = de_thread(me);
if (retval)
goto out;
+
+ /*
+ * This must be done here to ensure that POSIX CPU timers which were
+ * armed on the current task are dequeued from me::posix_cputimers.
+ * Otherwise in case of a TID switch the deletion of the related POSIX
+ * timer would not remove an enqueued timer because the TID lookup
+ * of the old TID fails.
+ */
+ posixtimer_exec(me);
+
/* see the comment in check_unsafe_exec() */
current->fs->in_exec = 0;
/*
@@ -1165,6 +1186,20 @@ int begin_new_exec(struct linux_binprm * bprm)
goto out;
/*
+ * We have to apply CLOEXEC before we change whether the process is
+ * dumpable (in setup_new_exec) to avoid a race with a process in userspace
+ * trying to access the should-be-closed file descriptors of a process
+ * undergoing exec(2).
+ *
+ * This can block on filesystem ->flush() handlers, including waiting
+ * for FUSE daemons, so do it before exec_mmap takes the
+ * exec_update_lock.
+ * This must happen after the point of no return, and after unsharing
+ * the FD table.
+ */
+ do_close_on_exec(me->files);
+
+ /*
* Must be called _before_ exec_mmap() as bprm->mm is
* not visible until then. Doing it here also ensures
* we don't race against replace_mm_exe_file().
@@ -1192,14 +1227,6 @@ int begin_new_exec(struct linux_binprm * bprm)
if (retval)
goto out_unlock;
-#ifdef CONFIG_POSIX_TIMERS
- spin_lock_irq(&me->sighand->siglock);
- posix_cpu_timers_exit(me);
- spin_unlock_irq(&me->sighand->siglock);
- exit_itimers(me);
- flush_itimer_signals();
-#endif
-
/*
* Make the signal table private.
*/
@@ -1214,14 +1241,6 @@ int begin_new_exec(struct linux_binprm * bprm)
clear_syscall_work_syscall_user_dispatch(me);
- /*
- * We have to apply CLOEXEC before we change whether the process is
- * dumpable (in setup_new_exec) to avoid a race with a process in userspace
- * trying to access the should-be-closed file descriptors of a process
- * undergoing exec(2).
- */
- do_close_on_exec(me->files);
-
if (bprm->secureexec) {
/* Make sure parent cannot signal privileged process. */
me->pdeath_signal = 0;
@@ -1472,9 +1491,9 @@ static void free_bprm(struct linux_binprm *bprm)
/* exec swapped the mm but failed before setup_new_exec() freed it */
if (bprm->old_mm)
exec_mm_put_old(bprm->old_mm);
- do_close_execat(bprm->file);
/* An unconsumed PT_INTERP substitute from a binfmt_misc loader entry. */
bprm_drop_loader(bprm);
+ do_close_execat(bprm->file);
do_close_execat(bprm->executable);
/* If a binfmt changed the interp, free it. */
if (bprm->interp != bprm->filename)
diff --git a/fs/ext4/fast_commit.c b/fs/ext4/fast_commit.c
index 062103e42cd8..0cac890cf370 100644
--- a/fs/ext4/fast_commit.c
+++ b/fs/ext4/fast_commit.c
@@ -1116,7 +1116,7 @@ static int ext4_fc_snapshot_inode(struct inode *inode,
else if (EXT4_INODE_SIZE(inode->i_sb) > EXT4_GOOD_OLD_INODE_SIZE)
inode_len += ei->i_extra_isize;
- snap = kmalloc(struct_size(snap, inode_buf, inode_len), GFP_NOFS);
+ snap = kmalloc_flex(*snap, inode_buf, inode_len, GFP_NOFS);
if (!snap) {
atomic64_inc(&stats->snap_fail_nomem);
ext4_fc_set_snap_err(snap_err, EXT4_FC_SNAP_ERR_NOMEM);
@@ -1522,7 +1522,7 @@ static int ext4_fc_alloc_snapshot_inodes(struct super_block *sb,
if (nr_inodes > EXT4_FC_SNAPSHOT_MAX_INODES)
return -E2BIG;
- inodes = kvcalloc(nr_inodes, sizeof(*inodes), GFP_NOFS);
+ inodes = kvzalloc_objs(*inodes, nr_inodes, GFP_NOFS);
if (!inodes)
return -ENOMEM;
diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c
index bd4b778df9eb..26f0f9714f03 100644
--- a/fs/ext4/inode.c
+++ b/fs/ext4/inode.c
@@ -6456,9 +6456,10 @@ int ext4_chunk_trans_blocks(struct inode *inode, int nrblocks)
int ext4_mark_iloc_dirty(handle_t *handle,
struct inode *inode, struct ext4_iloc *iloc)
{
+ struct super_block *sb = inode->i_sb;
int err = 0;
- err = ext4_emergency_state(inode->i_sb);
+ err = ext4_emergency_state(sb);
if (unlikely(err)) {
put_bh(iloc->bh);
return err;
@@ -6473,9 +6474,13 @@ int ext4_mark_iloc_dirty(handle_t *handle,
put_bh(iloc->bh);
/*
* Mark that there's metadata writeout pending for the inode so that it
- * gets properly flushed on fsync(2) and similar.
+ * gets properly flushed on fsync(2) and similar. We don't bother for
+ * fastcommit replay as that flushes the whole bdev afterwards anyway.
+ * It is faster this way and we avoid entering fs writeback paths which
+ * aren't fully initialized yet.
*/
- if (!EXT4_SB(inode->i_sb)->s_journal) {
+ if (!ext4_handle_valid(handle) &&
+ !(EXT4_SB(sb)->s_mount_state & EXT4_FC_REPLAY)) {
/*
* Inode didn't need to go through dirtying, make sure it is
* attached to wb so that writeback can handle it.
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index 8d6135a6108a..9a36d0329e22 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1597,8 +1597,7 @@ static int fuse_get_user_pages(struct fuse_args_pages *ap, struct iov_iter *ii,
* manually extract pages using iov_iter_extract_pages() and then
* copy that to a folios array.
*/
- struct page **pages = kcalloc(max_pages, sizeof(struct page *),
- GFP_KERNEL);
+ struct page **pages = kzalloc_objs(struct page *, max_pages);
if (!pages) {
ret = -ENOMEM;
goto out;
diff --git a/fs/fuse/readdir.c b/fs/fuse/readdir.c
index 5ca87151d70d..d2599043f7ec 100644
--- a/fs/fuse/readdir.c
+++ b/fs/fuse/readdir.c
@@ -336,7 +336,7 @@ static int parse_dirplusfile(char *buf, size_t nbytes, struct file *file,
static struct page **fuse_readdir_alloc_buf(struct fuse_args_pages *ap, size_t *bufsize)
{
unsigned int i, nr_alloc, nr_pages = DIV_ROUND_UP(*bufsize, PAGE_SIZE);
- struct page **pages = kcalloc(nr_pages, sizeof(*pages), GFP_KERNEL);
+ struct page **pages = kzalloc_objs(*pages, nr_pages);
if (!pages)
return NULL;
diff --git a/fs/hfs/bnode.c b/fs/hfs/bnode.c
index 1b331108d9c0..fcb5b9cd17f6 100644
--- a/fs/hfs/bnode.c
+++ b/fs/hfs/bnode.c
@@ -312,7 +312,7 @@ static struct hfs_bnode *__hfs_bnode_create(struct hfs_btree *tree, u32 cnid)
return NULL;
}
- node = kzalloc_flex(*node, page, tree->pages_per_bnode, GFP_KERNEL);
+ node = kzalloc_flex(*node, page, tree->pages_per_bnode);
if (!node)
return NULL;
node->tree = tree;
diff --git a/fs/kernfs/inode.c b/fs/kernfs/inode.c
index 237dcdd73fc2..abb286bc3474 100644
--- a/fs/kernfs/inode.c
+++ b/fs/kernfs/inode.c
@@ -142,10 +142,8 @@ ssize_t kernfs_iop_listxattr(struct dentry *dentry, char *buf, size_t size)
struct kernfs_iattrs *attrs;
attrs = kernfs_iattrs_noalloc(kn);
- if (!attrs)
- return 0;
- return simple_xattr_list(d_inode(dentry), &attrs->xattrs, buf, size);
+ return simple_xattr_list(d_inode(dentry), attrs ? &attrs->xattrs : NULL, buf, size);
}
static inline void set_default_inode_attr(struct inode *inode, umode_t mode)
diff --git a/fs/namespace.c b/fs/namespace.c
index 1ecd96c918b3..ae5dc64f8b45 100644
--- a/fs/namespace.c
+++ b/fs/namespace.c
@@ -5999,7 +5999,7 @@ SYSCALL_DEFINE4(statmount, const struct mnt_id_req __user *, req,
return -EPERM;
}
- ks = kmalloc(sizeof(*ks), GFP_KERNEL_ACCOUNT);
+ ks = kmalloc_obj(*ks, GFP_KERNEL_ACCOUNT);
if (!ks)
return -ENOMEM;
diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c
index 7fdfa4f27e34..424df70a5c30 100644
--- a/fs/netfs/buffered_read.c
+++ b/fs/netfs/buffered_read.c
@@ -55,6 +55,42 @@ static void netfs_rreq_expand(struct netfs_io_request *rreq,
}
/*
+ * Drop the folio refs acquired from the readahead API.
+ */
+static void netfs_bulk_drop_ra_refs(struct netfs_io_request *rreq)
+{
+ struct folio_batch fbatch;
+ struct folio *folio;
+ pgoff_t nr_pages = DIV_ROUND_UP(rreq->len, PAGE_SIZE);
+ pgoff_t first = rreq->start / PAGE_SIZE;
+ XA_STATE(xas, &rreq->mapping->i_pages, first);
+
+ folio_batch_init(&fbatch);
+
+ rcu_read_lock();
+
+ xas_for_each(&xas, folio, first + nr_pages - 1) {
+ if (xas_retry(&xas, folio))
+ continue;
+
+ if (!folio_batch_add(&fbatch, folio))
+ folio_batch_release(&fbatch);
+ }
+
+ rcu_read_unlock();
+ folio_batch_release(&fbatch);
+ trace_netfs_rreq(rreq, netfs_rreq_trace_ra_put_ref);
+ clear_bit_unlock(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags);
+ wake_up(&rreq->waitq);
+}
+
+static void netfs_maybe_bulk_drop_ra_refs(struct netfs_io_request *rreq)
+{
+ if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
+ netfs_bulk_drop_ra_refs(rreq);
+}
+
+/*
* Begin an operation, and fetch the stored zero point value from the cookie if
* available.
*/
@@ -74,12 +110,8 @@ static int netfs_begin_cache_read(struct netfs_io_request *rreq, struct netfs_in
*
* Returns the limited size if successful and -ENOMEM if insufficient memory
* available.
- *
- * [!] NOTE: This must be run in the same thread as ->issue_read() was called
- * in as we access the readahead_control struct.
*/
-static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq,
- struct readahead_control *ractl)
+static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq)
{
struct netfs_io_request *rreq = subreq->rreq;
size_t rsize = subreq->len;
@@ -87,30 +119,6 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq,
if (subreq->source == NETFS_DOWNLOAD_FROM_SERVER)
rsize = umin(rsize, rreq->io_streams[0].sreq_max_len);
- if (ractl) {
- /* If we don't have sufficient folios in the rolling buffer,
- * extract a folioq's worth from the readahead region at a time
- * into the buffer. Note that this acquires a ref on each page
- * that we will need to release later - but we don't want to do
- * that until after we've started the I/O.
- */
- struct folio_batch put_batch;
-
- folio_batch_init(&put_batch);
- while (rreq->submitted < subreq->start + rsize) {
- ssize_t added;
-
- added = rolling_buffer_load_from_ra(&rreq->buffer, ractl,
- &put_batch);
- if (added < 0) {
- folio_batch_release(&put_batch);
- return added;
- }
- rreq->submitted += added;
- }
- folio_batch_release(&put_batch);
- }
-
subreq->len = rsize;
if (unlikely(rreq->io_streams[0].sreq_max_segs)) {
size_t limit = netfs_limit_iter(&rreq->buffer.iter, 0, rsize,
@@ -204,16 +212,67 @@ static void netfs_issue_read(struct netfs_io_request *rreq,
}
/*
+ * Mark folios that we want to copy to the cache. For filesystems that use
+ * netfslib fully, we set folio->private to NETFS_FOLIO_COPY_TO_CACHE;
+ * otherwise we set the deprecated PG_private_2.
+ */
+static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq,
+ struct folio_queue **fq,
+ unsigned int *offset,
+ int *slot,
+ size_t len,
+ bool copy)
+{
+ while (len > 0) {
+ struct folio *folio;
+ size_t fsize, overlap;
+
+ if (!*fq)
+ break;
+ if (*slot >= folioq_count(*fq)) {
+ *fq = (*fq)->next;
+ *slot = 0;
+ *offset = 0;
+ continue;
+ }
+
+ /* Determine how much the subreq overlaps the folio, if at all. */
+ fsize = folioq_folio_size(*fq, *slot);
+ overlap = min(len, fsize - *offset);
+
+ if (overlap > 0 && copy) {
+ folio = folioq_folio(*fq, *slot);
+ if (unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags))) {
+ if (!folio_test_private_2(folio))
+ folio_start_private_2(folio);
+ } else {
+ if (!folio_get_private(folio))
+ folio_attach_private(folio, NETFS_FOLIO_COPY_TO_CACHE);
+ }
+ trace_netfs_folio(folio, netfs_folio_trace_mark_copy);
+ }
+
+ len -= overlap;
+ *offset += overlap;
+ if (*offset >= fsize) {
+ *slot += 1;
+ *offset = 0;
+ }
+ }
+}
+
+/*
* Perform a read to the pagecache from a series of sources of different types,
* slicing up the region to be read according to available cache blocks and
* network rsize.
*/
-static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
- struct readahead_control *ractl)
+static void netfs_read_to_pagecache(struct netfs_io_request *rreq)
{
+ struct folio_queue *fq = rreq->buffer.tail;
unsigned long long start = rreq->start;
+ unsigned int offset = 0;
ssize_t size = rreq->len;
- int ret = 0;
+ int ret = 0, slot = 0;
do {
struct netfs_io_subrequest *subreq;
@@ -288,7 +347,7 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
break;
issue:
- slice = netfs_prepare_read_iterator(subreq, ractl);
+ slice = netfs_prepare_read_iterator(subreq);
if (slice < 0) {
ret = slice;
netfs_cancel_read(subreq, ret);
@@ -301,7 +360,15 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags);
}
+ if (fq) {
+ /* See if the cache indicated this should be cached. */
+ bool copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags);
+
+ netfs_mark_copy_to_cache(rreq, &fq, &slot, &offset, slice, copy);
+ }
+
netfs_issue_read(rreq, subreq);
+ netfs_maybe_bulk_drop_ra_refs(rreq);
if (test_bit(NETFS_RREQ_PAUSE, &rreq->flags))
netfs_wait_for_paused_read(rreq);
@@ -339,7 +406,8 @@ void netfs_readahead(struct readahead_control *ractl)
{
struct netfs_io_request *rreq;
struct netfs_inode *ictx = netfs_inode(ractl->mapping->host);
- unsigned long long start = readahead_pos(ractl);
+ ssize_t added;
+ uoff_t start = readahead_pos(ractl);
size_t size = readahead_length(ractl);
int ret;
@@ -360,11 +428,24 @@ void netfs_readahead(struct readahead_control *ractl)
netfs_rreq_expand(rreq, ractl);
- rreq->submitted = rreq->start;
- if (rolling_buffer_init(&rreq->buffer, rreq->debug_id, ITER_DEST, rreq->gfp) < 0)
+ /* Load the folios to be read into a bvecq chain. Note that this
+ * acquires a ref on each folio that we will need to release later -
+ * but we don't want to do that until after we've started the I/O.
+ */
+ added = rolling_buffer_bulk_load_from_ra(&rreq->buffer, ractl,
+ rreq->debug_id, rreq->gfp);
+ if (added < 0) {
+ ret = added;
goto cleanup_free;
- netfs_read_to_pagecache(rreq, ractl);
+ }
+ __set_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags);
+
+ rreq->submitted = rreq->start + added;
+ rreq->cleaned_to = rreq->start;
+ netfs_read_set_unlock_at(rreq);
+ netfs_read_to_pagecache(rreq);
+ netfs_maybe_bulk_drop_ra_refs(rreq);
return netfs_put_request(rreq, netfs_rreq_trace_put_return);
cleanup_free:
@@ -387,6 +468,7 @@ static int netfs_create_singular_buffer(struct netfs_io_request *rreq, struct fo
if (added < 0)
return added;
rreq->submitted = rreq->start + added;
+ rreq->progress_at = added;
return 0;
}
@@ -457,7 +539,7 @@ static int netfs_read_gaps(struct file *file, struct folio *folio)
iov_iter_bvec(&rreq->buffer.iter, ITER_DEST, bvec, i, rreq->len);
rreq->submitted = rreq->start + flen;
- netfs_read_to_pagecache(rreq, NULL);
+ netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
if (ret >= 0) {
@@ -532,7 +614,7 @@ int netfs_read_folio(struct file *file, struct folio *folio)
if (ret < 0)
goto discard;
- netfs_read_to_pagecache(rreq, NULL);
+ netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
return ret < 0 ? ret : 0;
@@ -689,7 +771,7 @@ retry:
if (ret < 0)
goto error_put;
- netfs_read_to_pagecache(rreq, NULL);
+ netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
if (ret < 0)
@@ -754,7 +836,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio,
if (ret < 0)
goto error_put;
- netfs_read_to_pagecache(rreq, NULL);
+ netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
return ret < 0 ? ret : 0;
diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c
index c16fbad286a1..2361277416c7 100644
--- a/fs/netfs/direct_write.c
+++ b/fs/netfs/direct_write.c
@@ -21,7 +21,7 @@ static void netfs_unbuffered_write_done(struct netfs_io_request *wreq)
/* Okay, declare that all I/O is complete. */
trace_netfs_rreq(wreq, netfs_rreq_trace_write_done);
- if (!wreq->error)
+ if (wreq->transferred)
netfs_update_i_size(ictx, &ictx->inode, wreq->start, wreq->transferred);
if (wreq->origin == NETFS_DIO_WRITE &&
@@ -51,7 +51,7 @@ static void netfs_unbuffered_write_done(struct netfs_io_request *wreq)
wreq->iocb->ki_pos += written;
if (wreq->iocb->ki_complete) {
trace_netfs_rreq(wreq, netfs_rreq_trace_ki_complete);
- wreq->iocb->ki_complete(wreq->iocb, wreq->error ?: written);
+ wreq->iocb->ki_complete(wreq->iocb, written ?: wreq->error);
}
wreq->iocb = VFS_PTR_POISON;
}
@@ -95,7 +95,7 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
{
struct netfs_io_subrequest *subreq = NULL;
struct netfs_io_stream *stream = &wreq->io_streams[0];
- int ret;
+ int ret = 0;
_enter("%llx", wreq->len);
@@ -110,6 +110,11 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
if (!subreq) {
netfs_prepare_write(wreq, stream, wreq->start + wreq->transferred);
subreq = stream->construct;
+ if (!subreq) {
+ wreq->error = -ENOMEM;
+ ret = -ENOMEM;
+ break;
+ }
stream->construct = NULL;
}
@@ -121,8 +126,14 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
}
iov_iter_truncate(&subreq->io_iter, wreq->len - wreq->transferred);
- if (!iov_iter_count(&subreq->io_iter))
+ if (!iov_iter_count(&subreq->io_iter)) {
+ pr_warn("netfs: Unexpected zero-length iterator R=%08x\n",
+ wreq->debug_id);
+ __set_bit(NETFS_SREQ_FAILED, &subreq->flags);
+ netfs_write_subrequest_terminated(subreq, -EIO);
+ wreq->error = -EIO;
break;
+ }
subreq->len = netfs_limit_iter(&subreq->io_iter, 0,
stream->sreq_max_len,
@@ -139,13 +150,11 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
if (test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) {
retry = true;
} else if (test_bit(NETFS_SREQ_FAILED, &subreq->flags)) {
- ret = subreq->error;
- wreq->error = ret;
+ wreq->error = subreq->error;
netfs_see_subrequest(subreq, netfs_sreq_trace_see_failed);
subreq = NULL;
break;
}
- ret = 0;
if (!retry) {
netfs_unbuffered_write_collect(wreq, stream, subreq);
@@ -288,11 +297,11 @@ ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *
ret = -EIOCBQUEUED;
} else {
ret = netfs_unbuffered_write(wreq);
- if (ret < 0) {
- _debug("begin = %zd", ret);
- } else {
+ if (wreq->transferred) {
iocb->ki_pos += wreq->transferred;
- ret = wreq->transferred ?: wreq->error;
+ ret = wreq->transferred;
+ } else if (wreq->error) {
+ ret = wreq->error;
}
netfs_put_request(wreq, netfs_rreq_trace_put_complete);
diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h
index 420ee7b26580..c79c8e69d60c 100644
--- a/fs/netfs/internal.h
+++ b/fs/netfs/internal.h
@@ -79,6 +79,7 @@ ssize_t netfs_wait_for_read(struct netfs_io_request *rreq);
ssize_t netfs_wait_for_write(struct netfs_io_request *rreq);
void netfs_wait_for_paused_read(struct netfs_io_request *rreq);
void netfs_wait_for_paused_write(struct netfs_io_request *rreq);
+void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq);
/*
* objects.c
@@ -109,6 +110,8 @@ static inline void netfs_see_subrequest(struct netfs_io_subrequest *subreq,
/*
* read_collect.c
*/
+void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio);
+void netfs_read_set_unlock_at(struct netfs_io_request *rreq);
bool netfs_read_collection(struct netfs_io_request *rreq);
void netfs_read_collection_worker(struct work_struct *work);
void netfs_cancel_read(struct netfs_io_subrequest *subreq, int error);
diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c
index 5d554512ed23..f5c1c463f4ff 100644
--- a/fs/netfs/misc.c
+++ b/fs/netfs/misc.c
@@ -563,3 +563,22 @@ void netfs_wait_for_paused_write(struct netfs_io_request *rreq)
{
return netfs_wait_for_pause(rreq, netfs_write_collection);
}
+
+/*
+ * Wait for the readahead-acquired refs to be put.
+ */
+void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq)
+{
+ DEFINE_WAIT(myself);
+
+ for (;;) {
+ trace_netfs_rreq(rreq, netfs_rreq_trace_wait_put_ra_refs);
+ prepare_to_wait(&rreq->waitq, &myself, TASK_UNINTERRUPTIBLE);
+ if (!test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
+ break;
+ schedule();
+ }
+
+ trace_netfs_rreq(rreq, netfs_rreq_trace_waited_put_ra_refs);
+ finish_wait(&rreq->waitq, &myself);
+}
diff --git a/fs/netfs/objects.c b/fs/netfs/objects.c
index 01461a74642d..7f6a3e912602 100644
--- a/fs/netfs/objects.c
+++ b/fs/netfs/objects.c
@@ -41,24 +41,32 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping,
memset(rreq, 0, kmem_cache_size(cache));
INIT_WORK(&rreq->cleanup_work, netfs_free_request);
- rreq->gfp = gfp;
- rreq->start = start;
- rreq->len = len;
- rreq->origin = origin;
- rreq->netfs_ops = ctx->ops;
- rreq->mapping = mapping;
- rreq->inode = inode;
- rreq->i_size = i_size_read(inode);
- rreq->debug_id = atomic_inc_return(&debug_ids);
- rreq->wsize = INT_MAX;
+ rreq->gfp = gfp;
+ rreq->start = start;
+ rreq->collected_to = start;
+ rreq->cleaned_to = start;
+ rreq->len = len;
+ rreq->progress_at = 0;
+ rreq->origin = origin;
+ rreq->netfs_ops = ctx->ops;
+ rreq->mapping = mapping;
+ rreq->inode = inode;
+ rreq->i_size = i_size_read(inode);
+ rreq->debug_id = atomic_inc_return(&debug_ids);
+ rreq->wsize = INT_MAX;
rreq->io_streams[0].sreq_max_len = ULONG_MAX;
rreq->io_streams[0].sreq_max_segs = 0;
spin_lock_init(&rreq->lock);
- INIT_LIST_HEAD(&rreq->io_streams[0].subrequests);
- INIT_LIST_HEAD(&rreq->io_streams[1].subrequests);
init_waitqueue_head(&rreq->waitq);
refcount_set(&rreq->ref, 2);
+ for (int s = 0; s < NR_IO_STREAMS; s++) {
+ struct netfs_io_stream *stream = &rreq->io_streams[s];
+
+ INIT_LIST_HEAD(&stream->subrequests);
+ stream->collected_to = rreq->start;
+ }
+
if (origin == NETFS_READAHEAD ||
origin == NETFS_READPAGE ||
origin == NETFS_READ_GAPS ||
diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c
index 23660a590124..5cf22087d243 100644
--- a/fs/netfs/read_collect.c
+++ b/fs/netfs/read_collect.c
@@ -19,7 +19,6 @@
#define MADE_PROGRESS 0x04 /* Made progress cleaning up a stream or the folio set */
#define BUFFERED 0x08 /* The pagecache needs cleaning up */
#define NEED_RETRY 0x10 /* A front op requests retrying */
-#define COPY_TO_CACHE 0x40 /* Need to copy subrequest to cache */
#define ABANDON_SREQ 0x80 /* Need to abandon untransferred part of subrequest */
/*
@@ -35,6 +34,30 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq)
}
/*
+ * Cancel the copy-to-cache mark on a folio.
+ */
+void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio)
+{
+ if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
+ if (folio_get_private(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
+ folio_detach_private(folio);
+ trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
+ } else if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
+ struct netfs_folio *finfo = netfs_folio_info(folio);
+
+ finfo->netfs_group = NULL;
+ trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
+ }
+ } else {
+ // TODO: Use of PG_private_2 is deprecated.
+ if (folio_test_private_2(folio)) {
+ folio_end_private_2(folio);
+ trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
+ }
+ }
+}
+
+/*
* Flush, mark and unlock a folio that's now completely read. If we want to
* cache the folio, we set the group to NETFS_FOLIO_COPY_TO_CACHE, mark it
* dirty and let writeback handle it.
@@ -48,37 +71,37 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq,
if (unlikely(folio_pos(folio) < rreq->abandon_to)) {
trace_netfs_folio(folio, netfs_folio_trace_abandon);
+ netfs_cancel_copy_to_cache(rreq, folio);
goto just_unlock;
}
flush_dcache_folio(folio);
folio_mark_uptodate(folio);
- if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
- finfo = netfs_folio_info(folio);
- if (finfo) {
- trace_netfs_folio(folio, netfs_folio_trace_filled_gaps);
- if (finfo->netfs_group)
- folio_change_private(folio, finfo->netfs_group);
- else
- folio_detach_private(folio);
- kfree(finfo);
- }
+ if (unlikely(test_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags)))
+ netfs_cancel_copy_to_cache(rreq, folio);
- if (test_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags)) {
- if (!WARN_ON_ONCE(folio_get_private(folio) != NULL)) {
- trace_netfs_folio(folio, netfs_folio_trace_copy_to_cache);
- folio_attach_private(folio, NETFS_FOLIO_COPY_TO_CACHE);
- folio_mark_dirty(folio);
- }
+ if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
+ if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
+ trace_netfs_folio(folio, netfs_folio_trace_sched_copy);
+ folio_mark_dirty(folio);
} else {
+ finfo = netfs_folio_info(folio);
+ if (finfo) {
+ trace_netfs_folio(folio, netfs_folio_trace_filled_gaps);
+ if (finfo->netfs_group)
+ folio_change_private(folio, finfo->netfs_group);
+ else
+ folio_detach_private(folio);
+ kfree(finfo);
+ }
trace_netfs_folio(folio, netfs_folio_trace_read_done);
}
folioq_clear(folioq, slot);
} else {
// TODO: Use of PG_private_2 is deprecated.
- if (test_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags))
+ if (folio_test_private_2(folio))
netfs_pgpriv2_copy_to_cache(rreq, folio);
}
@@ -95,6 +118,35 @@ just_unlock:
}
/*
+ * Determine how much to gather before unlocking more folios.
+ */
+void netfs_read_set_unlock_at(struct netfs_io_request *rreq)
+{
+ struct folio_queue *folioq = rreq->buffer.tail;
+ unsigned int slot = rreq->buffer.first_tail_slot;
+ size_t cleaned_to = rreq->cleaned_to - rreq->start;
+ size_t progress_at = cleaned_to;
+ size_t minimum = 256 * 1024;
+
+ while (progress_at < rreq->len) {
+ if (slot >= folioq_count(folioq)) {
+ folioq = folioq->next;
+ if (!folioq)
+ break;
+ slot = 0;
+ }
+
+ progress_at += folioq_folio_size(folioq, slot);
+ if (progress_at - cleaned_to >= minimum)
+ break;
+ slot++;
+ }
+
+ WRITE_ONCE(rreq->progress_at, progress_at);
+ trace_netfs_read_progress_at(rreq);
+}
+
+/*
* Unlock any folios we've finished with.
*/
static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
@@ -112,30 +164,31 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
if (slot >= folioq_nr_slots(folioq)) {
folioq = rolling_buffer_delete_spent(&rreq->buffer);
if (!folioq) {
- rreq->front_folio_order = 0;
+ WRITE_ONCE(rreq->progress_at, rreq->len);
return;
}
slot = 0;
}
+ /* We have to wait for readahead refs to have been released before we
+ * can unlock any folios as the ref-dropper walks i_pages and the only
+ * thing preventing these folios from being removed is the folio lock.
+ */
+ if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
+ netfs_wait_for_put_ra_refs(rreq);
+
for (;;) {
struct folio *folio;
unsigned long long fpos, fend;
- unsigned int order;
size_t fsize;
- if (*notes & COPY_TO_CACHE)
- set_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
-
folio = folioq_folio(folioq, slot);
if (WARN_ONCE(!folio_test_locked(folio),
"R=%08x: folio %lx is not locked\n",
rreq->debug_id, folio->index))
trace_netfs_folio(folio, netfs_folio_trace_not_locked);
- order = folioq_folio_order(folioq, slot);
- rreq->front_folio_order = order;
- fsize = PAGE_SIZE << order;
+ fsize = folioq_folio_size(folioq, slot);
fpos = folio_pos(folio);
fend = fpos + fsize;
@@ -149,8 +202,6 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
WRITE_ONCE(rreq->cleaned_to, fpos + fsize);
*notes |= MADE_PROGRESS;
- clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
-
/* Clean up the head folioq. If we clear an entire folioq, then
* we can get rid of it provided it's not also the tail folioq
* being filled by the issuer.
@@ -172,6 +223,8 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
rreq->buffer.tail = folioq;
done:
rreq->buffer.first_tail_slot = slot;
+
+ netfs_read_set_unlock_at(rreq);
}
/*
@@ -232,7 +285,7 @@ reassess:
* subreqs.
*/
if (notes & BUFFERED) {
- size_t fsize = PAGE_SIZE << rreq->front_folio_order;
+ uoff_t unlock_at = rreq->start + rreq->progress_at;
/* Clear the tail of a short read. */
if (!(notes & HIT_PENDING) &&
@@ -248,16 +301,13 @@ reassess:
stream->collected_to = front->start + transferred;
rreq->collected_to = stream->collected_to;
- if (test_bit(NETFS_SREQ_COPY_TO_CACHE, &front->flags))
- notes |= COPY_TO_CACHE;
-
if (test_bit(NETFS_SREQ_FAILED, &front->flags)) {
rreq->abandon_to = front->start + front->len;
front->transferred = front->len;
transferred = front->len;
trace_netfs_rreq(rreq, netfs_rreq_trace_set_abandon);
}
- if (front->start + transferred >= rreq->cleaned_to + fsize ||
+ if (front->start + transferred >= unlock_at ||
test_bit(NETFS_SREQ_HIT_EOF, &front->flags))
netfs_read_unlock_folios(rreq, &notes);
} else {
@@ -477,20 +527,22 @@ void netfs_read_collection_worker(struct work_struct *work)
void netfs_read_subreq_progress(struct netfs_io_subrequest *subreq)
{
struct netfs_io_request *rreq = subreq->rreq;
- struct netfs_io_stream *stream = &rreq->io_streams[0];
- size_t fsize = PAGE_SIZE << rreq->front_folio_order;
-
- trace_netfs_sreq(subreq, netfs_sreq_trace_progress);
+ struct netfs_io_stream *stream = &rreq->io_streams[subreq->stream_nr];
+ size_t progress_at = READ_ONCE(rreq->progress_at);
+ uoff_t update_at = rreq->start + progress_at;
+ uoff_t transferred_to = subreq->start + subreq->transferred;
/* If we are at the head of the queue, wake up the collector,
* getting a ref to it if we were the ones to do so.
*/
- if (subreq->start + subreq->transferred > rreq->cleaned_to + fsize &&
+ if (progress_at < rreq->len &&
+ transferred_to >= update_at &&
(rreq->origin == NETFS_READAHEAD ||
rreq->origin == NETFS_READPAGE ||
rreq->origin == NETFS_READ_FOR_WRITE) &&
list_is_first(&subreq->rreq_link, &stream->subrequests)
) {
+ trace_netfs_sreq(subreq, netfs_sreq_trace_progress);
__set_bit(NETFS_SREQ_MADE_PROGRESS, &subreq->flags);
netfs_wake_collector(rreq);
}
diff --git a/fs/netfs/read_pgpriv2.c b/fs/netfs/read_pgpriv2.c
index c31190993b76..a4b7bb88cbdb 100644
--- a/fs/netfs/read_pgpriv2.c
+++ b/fs/netfs/read_pgpriv2.c
@@ -54,8 +54,8 @@ static void netfs_pgpriv2_copy_folio(struct netfs_io_request *creq, struct folio
/* Attach the folio to the rolling buffer. */
if (rolling_buffer_append(&creq->buffer, folio, 0, creq->gfp) < 0) {
+ set_bit(NETFS_RREQ_CANCEL_CACHING, &creq->flags);
folio_end_private_2(folio);
- clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &creq->flags);
return;
}
@@ -122,13 +122,14 @@ cancel_put:
netfs_put_failed_request(creq);
cancel:
rreq->copy_to_cache = ERR_PTR(-ENOBUFS);
- clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
+ set_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags);
return ERR_PTR(-ENOBUFS);
}
/*
* [DEPRECATED] Mark page as requiring copy-to-cache using PG_private_2 and add
- * it to the copy write request.
+ * it to the copy write request. PG_private_2 should already be set on the
+ * folio.
*/
void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio)
{
@@ -136,11 +137,13 @@ void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *fo
if (!creq)
creq = netfs_pgpriv2_begin_copy_to_cache(rreq, folio);
- if (IS_ERR(creq))
+ if (IS_ERR(creq)) {
+ set_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags);
+ netfs_cancel_copy_to_cache(rreq, folio);
return;
+ }
- trace_netfs_folio(folio, netfs_folio_trace_copy_to_cache);
- folio_start_private_2(folio);
+ trace_netfs_folio(folio, netfs_folio_trace_pgpriv2_copy);
netfs_pgpriv2_copy_folio(creq, folio);
}
diff --git a/fs/netfs/read_retry.c b/fs/netfs/read_retry.c
index 2b42758e01ec..4f6a36c6e214 100644
--- a/fs/netfs/read_retry.c
+++ b/fs/netfs/read_retry.c
@@ -292,11 +292,22 @@ void netfs_unlock_abandoned_read_pages(struct netfs_io_request *rreq)
{
struct folio_queue *p;
+ /* We have to wait for readahead refs to have been released before we
+ * can unlock any folios as the ref-dropper walks i_pages and the only
+ * thing preventing these folios from being removed is the folio lock.
+ */
+ if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
+ netfs_wait_for_put_ra_refs(rreq);
+
for (p = rreq->buffer.tail; p; p = p->next) {
for (int slot = 0; slot < folioq_count(p); slot++) {
struct folio *folio = folioq_folio(p, slot);
- if (folio && !folioq_is_marked2(p, slot)) {
+ if (!folio)
+ continue;
+ netfs_cancel_copy_to_cache(rreq, folio);
+
+ if (!folioq_is_marked2(p, slot)) {
if (folio == rreq->no_unlock_folio &&
test_bit(NETFS_RREQ_NO_UNLOCK_FOLIO,
&rreq->flags)) {
diff --git a/fs/netfs/read_single.c b/fs/netfs/read_single.c
index 8833550d2eb6..de67ac41548d 100644
--- a/fs/netfs/read_single.c
+++ b/fs/netfs/read_single.c
@@ -170,6 +170,8 @@ ssize_t netfs_read_single(struct inode *inode, struct file *file, struct iov_ite
if (IS_ERR(rreq))
return PTR_ERR(rreq);
+ rreq->progress_at = rreq->len;
+
ret = netfs_single_begin_cache_read(rreq, ictx);
if (ret == -ENOMEM || ret == -EINTR || ret == -ERESTARTSYS)
goto cleanup_free;
diff --git a/fs/netfs/rolling_buffer.c b/fs/netfs/rolling_buffer.c
index 8c0026836f9c..424e77a9a109 100644
--- a/fs/netfs/rolling_buffer.c
+++ b/fs/netfs/rolling_buffer.c
@@ -115,42 +115,65 @@ int rolling_buffer_make_space(struct rolling_buffer *roll, gfp_t gfp)
}
/*
- * Decant the list of folios to read into a rolling buffer.
+ * Decant the entire list of folios to read into a rolling buffer.
*/
-ssize_t rolling_buffer_load_from_ra(struct rolling_buffer *roll,
- struct readahead_control *ractl,
- struct folio_batch *put_batch)
+ssize_t rolling_buffer_bulk_load_from_ra(struct rolling_buffer *roll,
+ struct readahead_control *ractl,
+ unsigned int rreq_id, gfp_t gfp)
{
struct folio_queue *fq;
- struct page **vec;
- int nr, ix, to;
- ssize_t size = 0;
+ ssize_t loaded = 0;
- if (rolling_buffer_make_space(roll, GFP_KERNEL) < 0)
- return -ENOMEM;
+ while (ractl->_nr_pages - ractl->_batch_count > 0) {
+ unsigned int nr;
- fq = roll->head;
- vec = (struct page **)fq->vec.folios;
- nr = __readahead_batch(ractl, vec + folio_batch_count(&fq->vec),
- folio_batch_space(&fq->vec));
- ix = fq->vec.nr;
- to = ix + nr;
- fq->vec.nr = to;
- for (; ix < to; ix++) {
- struct folio *folio = folioq_folio(fq, ix);
- unsigned int order = folio_order(folio);
-
- fq->orders[ix] = order;
- size += PAGE_SIZE << order;
- trace_netfs_folio(folio, netfs_folio_trace_read);
- if (!folio_batch_add(put_batch, folio))
- folio_batch_release(put_batch);
+ /* Allocate a folioq to put some folios into and attach it to
+ * the rolling buffer.
+ */
+ fq = netfs_folioq_alloc(rreq_id, gfp,
+ netfs_trace_folioq_make_space);
+ if (!fq)
+ goto nomem_unlock;
+ fq->prev = roll->head;
+ if (!roll->tail)
+ roll->tail = fq;
+ else
+ roll->head->next = fq;
+ roll->head = fq;
+
+ /* Get a batch of folios and note their orders. */
+ nr = __readahead_batch(ractl, (struct page **)fq->vec.folios,
+ folioq_nr_slots(fq));
+ if (WARN_ON_ONCE(!nr))
+ break;
+ fq->vec.nr = nr;
+
+ for (int slot = 0; slot < nr; slot++) {
+ struct folio *folio = folioq_folio(fq, slot);
+ unsigned int order;
+
+ order = folio_order(folio);
+ fq->orders[slot] = order;
+ loaded += PAGE_SIZE << order;
+ trace_netfs_folio(folio, netfs_folio_trace_read);
+ }
}
- WRITE_ONCE(roll->iter.count, roll->iter.count + size);
- /* Store the counter after setting the slot. */
- smp_store_release(&roll->next_head_slot, to);
- return size;
+ WRITE_ONCE(roll->iter.count, loaded);
+ iov_iter_folio_queue(&roll->iter, ITER_DEST, roll->tail, 0, 0, loaded);
+ return loaded;
+
+nomem_unlock:
+ for (fq = roll->tail; fq; fq = fq->next) {
+ for (int slot = 0; slot < folioq_count(fq); slot++) {
+ folio_unlock(fq->vec.folios[slot]);
+ folioq_mark(fq, slot);
+ }
+ }
+ rolling_buffer_clear(roll);
+ roll->head = NULL;
+ roll->tail = NULL;
+ return -ENOMEM;
}
/*
diff --git a/fs/netfs/write_issue.c b/fs/netfs/write_issue.c
index 2d9cfcd43658..851f6f93ad45 100644
--- a/fs/netfs/write_issue.c
+++ b/fs/netfs/write_issue.c
@@ -170,6 +170,8 @@ void netfs_prepare_write(struct netfs_io_request *wreq,
rolling_buffer_make_space(&wreq->buffer, wreq->gfp);
subreq = netfs_alloc_subrequest(wreq);
+ if (!subreq)
+ return;
subreq->source = stream->source;
subreq->start = start;
subreq->stream_nr = stream->stream_nr;
diff --git a/fs/nfsd/export.c b/fs/nfsd/export.c
index a47c90f40422..b6e0c543e028 100644
--- a/fs/nfsd/export.c
+++ b/fs/nfsd/export.c
@@ -358,7 +358,7 @@ int nfsd_nl_expkey_get_reqs_dumpit(struct sk_buff *skb,
goto out_unlock;
}
- items = kcalloc(cnt, sizeof(*items), GFP_KERNEL);
+ items = kzalloc_objs(*items, cnt);
seqnos = kcalloc(cnt, sizeof(*seqnos), GFP_KERNEL);
if (!items || !seqnos) {
ret = -ENOMEM;
@@ -685,7 +685,7 @@ int nfsd_nl_svc_export_get_reqs_dumpit(struct sk_buff *skb,
goto out_unlock;
}
- items = kcalloc(cnt, sizeof(*items), GFP_KERNEL);
+ items = kzalloc_objs(*items, cnt);
seqnos = kcalloc(cnt, sizeof(*seqnos), GFP_KERNEL);
pathbuf = kmalloc(PATH_MAX, GFP_KERNEL);
if (!items || !seqnos || !pathbuf) {
@@ -786,8 +786,7 @@ static int nfsd_nl_parse_fslocations(struct nlattr *attr,
if (!count)
return 0;
- fsloc->locations = kcalloc(count, sizeof(struct nfsd4_fs_location),
- GFP_KERNEL);
+ fsloc->locations = kzalloc_objs(struct nfsd4_fs_location, count);
if (!fsloc->locations)
return -ENOMEM;
@@ -1006,7 +1005,8 @@ static int nfsd_nl_parse_one_export(struct cache_detail *cd,
goto out_uuid;
err = 0;
- nfsd4_setup_layout_type(&exp);
+ if (exp.ex_flags & NFSEXP_PNFS)
+ nfsd4_setup_layout_type(&exp);
}
expp = svc_export_lookup(&exp);
diff --git a/fs/nfsd/nfs4callback.c b/fs/nfsd/nfs4callback.c
index a901bbe67e03..19dc337502ca 100644
--- a/fs/nfsd/nfs4callback.c
+++ b/fs/nfsd/nfs4callback.c
@@ -1981,12 +1981,12 @@ int nfsd_net_cb_init(struct nfsd_net *nn)
{
struct nfsd_net_cb *cb;
- cb = kzalloc(sizeof(*cb), GFP_KERNEL);
+ cb = kzalloc_obj(*cb);
if (!cb)
return -ENOMEM;
cb->version4.counts = kzalloc_objs(unsigned int,
- ARRAY_SIZE(nfs4_cb_procedures), GFP_KERNEL);
+ ARRAY_SIZE(nfs4_cb_procedures));
if (!cb->version4.counts) {
kfree(cb);
return -ENOMEM;
diff --git a/fs/nfsd/nfs4state.c b/fs/nfsd/nfs4state.c
index 18e17232cf94..9c4adf3110ae 100644
--- a/fs/nfsd/nfs4state.c
+++ b/fs/nfsd/nfs4state.c
@@ -1341,7 +1341,7 @@ alloc_init_dir_deleg(struct nfs4_client *clp, struct nfs4_file *fp)
return NULL;
}
- ncn->ncn_nf = kcalloc(NOTIFY4_EVENT_QUEUE_SIZE, sizeof(*ncn->ncn_nf), GFP_KERNEL);
+ ncn->ncn_nf = kzalloc_objs(*ncn->ncn_nf, NOTIFY4_EVENT_QUEUE_SIZE);
if (!ncn->ncn_nf) {
nfs4_put_stid(&dp->dl_stid);
return NULL;
@@ -10419,8 +10419,9 @@ alloc_nfsd_notify_event(u32 mask, const struct qstr *q, struct dentry *dentry,
newnamelen = newname.name.len;
}
- ne = kmalloc(struct_size(ne, ne_name, q->len + 1 +
- (newnamelen ? newnamelen + 1 : 0)), GFP_NOFS);
+ ne = kmalloc_flex(*ne, ne_name,
+ q->len + 1 + (newnamelen ? newnamelen + 1 : 0),
+ GFP_NOFS);
if (!ne)
goto out;
diff --git a/fs/nfsd/nfsctl.c b/fs/nfsd/nfsctl.c
index adb032b7311a..5abb2d4274c9 100644
--- a/fs/nfsd/nfsctl.c
+++ b/fs/nfsd/nfsctl.c
@@ -1647,7 +1647,7 @@ static int nfsd_nl_fh_key_set(const struct nlattr *attr, struct nfsd_net *nn)
k1 = get_unaligned_le64(nla_data(attr) + 8);
if (!fh_key) {
- fh_key = kmalloc(sizeof(siphash_key_t), GFP_KERNEL);
+ fh_key = kmalloc_obj(siphash_key_t);
if (!fh_key) {
trace_nfsd_ctl_fh_key_set(false, -ENOMEM);
return -ENOMEM;
diff --git a/fs/ntfs/attrib.c b/fs/ntfs/attrib.c
index 60264833bb63..c949ff765075 100644
--- a/fs/ntfs/attrib.c
+++ b/fs/ntfs/attrib.c
@@ -1737,8 +1737,8 @@ static struct attr_def *ntfs_attr_find_in_attrdef(const struct ntfs_volume *vol,
struct attr_def *ad;
WARN_ON(!type);
- for (ad = vol->attrdef; (u8 *)ad - (u8 *)vol->attrdef <
- vol->attrdef_size && ad->type; ++ad) {
+ for (ad = vol->attrdef; (u8 *)ad - (u8 *)vol->attrdef <=
+ vol->attrdef_size - (s32)sizeof(*ad) && ad->type; ++ad) {
/* We have not found it yet, carry on searching. */
if (likely(le32_to_cpu(ad->type) < le32_to_cpu(type)))
continue;
@@ -2500,7 +2500,7 @@ int ntfs_resident_attr_record_add(struct ntfs_inode *ni, __le32 type,
return offset;
put_err_out:
ntfs_attr_put_search_ctx(ctx);
- return -EIO;
+ return err;
}
/*
@@ -2639,7 +2639,7 @@ static int ntfs_non_resident_attr_record_add(struct ntfs_inode *ni, __le32 type,
return offset;
put_err_out:
ntfs_attr_put_search_ctx(ctx);
- return -1;
+ return err;
}
/*
@@ -2917,7 +2917,7 @@ retry:
attr_ni = NULL;
/* Allocate new extent. */
- err = ntfs_mft_record_alloc(ni->vol, 0, &attr_ni, ni, NULL);
+ err = ntfs_mft_record_alloc(ni->vol, 0, &attr_ni, ni, NULL, -1);
if (err) {
ntfs_error(sb, "Failed to allocate extent record");
goto err_out;
@@ -3550,7 +3550,7 @@ int ntfs_attr_record_move_away(struct ntfs_attr_search_ctx *ctx, int extra)
* new extent and move attribute to it.
*/
ni = NULL;
- err = ntfs_mft_record_alloc(base_ni->vol, 0, &ni, base_ni, NULL);
+ err = ntfs_mft_record_alloc(base_ni->vol, 0, &ni, base_ni, NULL, -1);
if (err) {
ntfs_error(sb, "Couldn't allocate MFT record, err : %d", err);
return err;
@@ -3574,7 +3574,8 @@ int ntfs_attr_record_move_away(struct ntfs_attr_search_ctx *ctx, int extra)
* update allocated and compressed size.
*/
static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni,
- struct mft_record *m, struct ntfs_attr_search_ctx *ctx)
+ struct mft_record *m, struct ntfs_attr_search_ctx *ctx,
+ struct ntfs_inode *locked_ni, bool defer_attrlist)
{
int sparse, err = 0;
struct ntfs_inode *base_ni;
@@ -3610,6 +3611,8 @@ static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni,
le16_to_cpu(a->data.non_resident.mapping_pairs_offset) == 8) &&
!(le32_to_cpu(m->bytes_allocated) - le32_to_cpu(m->bytes_in_use))) {
+ if (defer_attrlist)
+ return -ENOSPC;
if (!NInoAttrList(base_ni)) {
err = ntfs_inode_add_attrlist(base_ni);
if (err)
@@ -3623,7 +3626,7 @@ static int ntfs_attr_update_meta(struct attr_record *a, struct ntfs_inode *ni,
goto out;
}
- err = ntfs_attrlist_update(base_ni);
+ err = ntfs_attrlist_update_locked(base_ni, locked_ni);
if (err)
goto out;
err = -EAGAIN;
@@ -3703,6 +3706,8 @@ out:
* ntfs_attr_update_mapping_pairs - update mapping pairs for ntfs attribute
* @ni: non-resident ntfs inode for which we need update
* @from_vcn: update runlist starting this VCN
+ * @locked_ni: inode whose runlist write lock is already held
+ * @defer_attrlist: return -ENOSPC instead of updating an attribute list
*
* Build mapping pairs from @na->rl and write them to the disk. Also, this
* function updates sparse bit, allocated and compressed size (allocates/frees
@@ -3712,7 +3717,10 @@ out:
* call to this function. Vice-versa @na->compressed_size will be calculated and
* set to correct value during this function.
*/
-int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn)
+static int __ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni,
+ s64 from_vcn,
+ struct ntfs_inode *locked_ni,
+ bool defer_attrlist)
{
struct ntfs_attr_search_ctx *ctx;
struct ntfs_inode *base_ni;
@@ -3804,7 +3812,8 @@ retry:
continue;
}
- err = ntfs_attr_update_meta(a, ni, m, ctx);
+ err = ntfs_attr_update_meta(a, ni, m, ctx, locked_ni,
+ defer_attrlist);
if (err < 0) {
if (err == -EAGAIN) {
ntfs_attr_put_search_ctx(ctx);
@@ -3844,18 +3853,28 @@ retry:
*/
if (ni->type == AT_ATTRIBUTE_LIST) {
ntfs_attr_put_search_ctx(ctx);
- if (ntfs_inode_free_space(base_ni, mp_size -
- cur_max_mp_size)) {
- ntfs_debug("Attribute list is too big. Defragment the volume\n");
- return -ENOSPC;
+ ctx = NULL;
+ if (locked_ni == ni || defer_attrlist) {
+ err = -ENOSPC;
+ goto put_err_out;
}
- if (ntfs_attrlist_update(base_ni))
- return -EIO;
+ err = ntfs_inode_free_space(base_ni, mp_size -
+ cur_max_mp_size);
+ if (err)
+ return err;
+ err = ntfs_attrlist_update_locked(
+ base_ni, locked_ni);
+ if (err)
+ return err;
goto retry;
}
/* Add attribute list if it isn't present, and retry. */
if (!NInoAttrList(base_ni)) {
+ if (defer_attrlist) {
+ err = -ENOSPC;
+ goto put_err_out;
+ }
ntfs_attr_put_search_ctx(ctx);
if (ntfs_inode_add_attrlist(base_ni)) {
ntfs_error(sb, "Can not add attrlist");
@@ -3883,13 +3902,21 @@ retry:
}
}
+ if (defer_attrlist &&
+ (ctx->ntfs_ino->nr_extents == -1 ||
+ NInoAttrList(ctx->ntfs_ino)) &&
+ ctx->attr->type != AT_ATTRIBUTE_LIST) {
+ err = -ENOSPC;
+ goto put_err_out;
+ }
+
/* Update lowest vcn. */
a->data.non_resident.lowest_vcn = cpu_to_le64(stop_vcn);
mark_mft_record_dirty(ctx->ntfs_ino);
if ((ctx->ntfs_ino->nr_extents == -1 || NInoAttrList(ctx->ntfs_ino)) &&
ctx->attr->type != AT_ATTRIBUTE_LIST) {
ctx->al_entry->lowest_vcn = cpu_to_le64(stop_vcn);
- err = ntfs_attrlist_update(base_ni);
+ err = ntfs_attrlist_update_locked(base_ni, locked_ni);
if (err)
goto put_err_out;
}
@@ -3976,7 +4003,10 @@ retry:
unsigned int de_cnt = 0;
/* Allocate new mft record. */
- err = ntfs_mft_record_alloc(ni->vol, 0, &ext_ni, base_ni, NULL);
+ err = ntfs_mft_record_alloc(ni->vol, 0, &ext_ni, base_ni, NULL,
+ base_ni->mft_no == FILE_MFT &&
+ ni->type == AT_DATA &&
+ ni->name == AT_UNNAMED ? stop_vcn : -1);
if (err) {
ntfs_error(sb, "Failed to allocate extent record");
goto put_err_out;
@@ -4061,6 +4091,19 @@ put_err_out:
return err;
}
+int ntfs_attr_update_mapping_pairs_locked(struct ntfs_inode *ni,
+ s64 from_vcn,
+ struct ntfs_inode *locked_ni)
+{
+ return __ntfs_attr_update_mapping_pairs(ni, from_vcn, locked_ni,
+ false);
+}
+
+int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn)
+{
+ return ntfs_attr_update_mapping_pairs_locked(ni, from_vcn, NULL);
+}
+
/*
* ntfs_attr_make_resident - convert a non-resident to a resident attribute
* @ni: open ntfs attribute to make resident
@@ -4194,7 +4237,9 @@ static int ntfs_attr_make_resident(struct ntfs_inode *ni, struct ntfs_attr_searc
*
* Reduce the size of a non-resident, open ntfs attribute @na to @newsize bytes.
*/
-static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsize)
+static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni,
+ const s64 newsize,
+ struct ntfs_inode *locked_ni)
{
struct ntfs_volume *vol;
struct ntfs_attr_search_ctx *ctx;
@@ -4202,6 +4247,7 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
s64 nr_freed_clusters;
int err;
struct ntfs_inode *base_ni;
+ bool runlist_locked = locked_ni == ni;
ntfs_debug("Inode 0x%llx attr 0x%x new size %lld\n",
(unsigned long long)ni->mft_no, ni->type, (long long)newsize);
@@ -4247,18 +4293,24 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
* clusters if there is a change.
*/
if (ntfs_bytes_to_cluster(vol, ni->allocated_size) != first_free_vcn) {
- struct ntfs_attr_search_ctx *ctx;
+ /*
+ * ntfs_cluster_free() and ntfs_rl_truncate_nolock()
+ * both require this lock.
+ */
+ if (!runlist_locked)
+ down_write(&ni->runlist.lock);
err = ntfs_attr_map_whole_runlist(ni);
if (err) {
ntfs_debug("Eeek! ntfs_attr_map_whole_runlist failed.\n");
- return err;
+ goto unlock_runlist;
}
ctx = ntfs_attr_get_search_ctx(ni, NULL);
if (!ctx) {
ntfs_error(vol->sb, "%s: Failed to get search context", __func__);
- return -ENOMEM;
+ err = -ENOMEM;
+ goto unlock_runlist;
}
/* Deallocate all clusters starting with the first free one. */
@@ -4266,7 +4318,8 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
if (nr_freed_clusters < 0) {
ntfs_debug("Eeek! Freeing of clusters failed. Aborting...\n");
ntfs_attr_put_search_ctx(ctx);
- return (int)nr_freed_clusters;
+ err = (int)nr_freed_clusters;
+ goto unlock_runlist;
}
ntfs_attr_put_search_ctx(ctx);
@@ -4279,7 +4332,8 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
kvfree(ni->runlist.rl);
ni->runlist.rl = NULL;
ntfs_error(vol->sb, "Eeek! Run list truncation failed.\n");
- return -EIO;
+ err = -EIO;
+ goto unlock_runlist;
}
/* Prepare to mapping pairs update. */
@@ -4295,11 +4349,13 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
VFS_I(base_ni)->i_blocks = ni->allocated_size >> 9;
/* Write mapping pairs for new runlist. */
- err = ntfs_attr_update_mapping_pairs(ni, 0 /*first_free_vcn*/);
+ err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
if (err) {
ntfs_debug("Eeek! Mapping pairs update failed. Leaving inconstant metadata. Run chkdsk.\n");
- return err;
+ goto unlock_runlist;
}
+ if (!runlist_locked)
+ up_write(&ni->runlist.lock);
}
/* Get the first attribute record. */
@@ -4341,7 +4397,11 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
/* If the attribute now has zero size, make it resident. */
if (!newsize && !NInoEncrypted(ni) && !NInoCompressed(ni)) {
+ if (!runlist_locked)
+ down_write(&ni->runlist.lock);
err = ntfs_attr_make_resident(ni, ctx);
+ if (!runlist_locked)
+ up_write(&ni->runlist.lock);
if (err) {
/* If couldn't make resident, just continue. */
if (err != -EPERM)
@@ -4358,6 +4418,11 @@ static int ntfs_non_resident_attr_shrink(struct ntfs_inode *ni, const s64 newsiz
put_err_out:
ntfs_attr_put_search_ctx(ctx);
return err;
+
+unlock_runlist:
+ if (!runlist_locked)
+ up_write(&ni->runlist.lock);
+ return err;
}
/*
@@ -4366,13 +4431,14 @@ put_err_out:
* @prealloc_size: preallocation size (in bytes) to which to expand the attribute
* @newsize: new size (in bytes) to which to expand the attribute
* @holes: how to create a hole if expanding
- * @need_lock: whether mrec lock is needed or not
+ * @locked_ni: inode whose runlist lock is already held
*
* Expand the size of a non-resident, open ntfs attribute @na to @newsize bytes,
* by allocating new clusters.
*/
static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsize,
- const s64 prealloc_size, unsigned int holes, bool need_lock)
+ const s64 prealloc_size, unsigned int holes,
+ struct ntfs_inode *locked_ni)
{
s64 lcn_seek_from;
s64 first_free_vcn;
@@ -4519,13 +4585,39 @@ static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsiz
ntfs_bytes_to_cluster(vol, ni->allocated_size),
first_free_vcn -
ntfs_bytes_to_cluster(vol, ni->allocated_size),
- lcn_seek_from, DATA_ZONE, false, false, false);
+ lcn_seek_from, DATA_ZONE, false,
+ ni->type == AT_ATTRIBUTE_LIST, false);
if (IS_ERR(rl)) {
ntfs_debug("Cluster allocation failed (%lld)",
(long long)first_free_vcn -
ntfs_bytes_to_cluster(vol, ni->allocated_size));
return PTR_ERR(rl);
}
+ /*
+ * A contiguous ATTRIBUTE_LIST allocation keeps its mapping
+ * pairs small enough to fit in the base MFT record. The
+ * allocator can return a short run when contiguity was
+ * requested, so discard it and retry normally if necessary.
+ */
+ if (ni->type == AT_ATTRIBUTE_LIST &&
+ (rl->vcn != ntfs_bytes_to_cluster(vol,
+ ni->allocated_size) ||
+ rl->length != first_free_vcn -
+ ntfs_bytes_to_cluster(vol, ni->allocated_size) ||
+ rl[1].length)) {
+ ntfs_cluster_free_from_rl(vol, rl);
+ kvfree(rl);
+ rl = ntfs_cluster_alloc(vol,
+ ntfs_bytes_to_cluster(vol,
+ ni->allocated_size),
+ first_free_vcn -
+ ntfs_bytes_to_cluster(vol,
+ ni->allocated_size),
+ lcn_seek_from, DATA_ZONE, false,
+ false, false);
+ if (IS_ERR(rl))
+ return PTR_ERR(rl);
+ }
}
if (!NInoCompressed(ni)) {
@@ -4544,7 +4636,8 @@ static int ntfs_non_resident_attr_expand(struct ntfs_inode *ni, const s64 newsiz
/* Prepare to mapping pairs update. */
ni->allocated_size = ntfs_cluster_to_bytes(vol, first_free_vcn);
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(
+ ni, 0, locked_ni);
if (err) {
ntfs_debug("Mapping pairs update failed");
goto rollback;
@@ -4588,11 +4681,11 @@ rollback:
ntfs_debug("Leaking clusters");
/* Now, truncate the runlist itself. */
- if (need_lock)
+ if (ni != locked_ni)
down_write(&ni->runlist.lock);
err2 = ntfs_rl_truncate_nolock(vol, &ni->runlist,
ntfs_bytes_to_cluster(vol, org_alloc_size));
- if (need_lock)
+ if (ni != locked_ni)
up_write(&ni->runlist.lock);
if (err2) {
/*
@@ -4606,11 +4699,11 @@ rollback:
/* Prepare to mapping pairs update. */
ni->allocated_size = org_alloc_size;
/* Restore mapping pairs. */
- if (need_lock)
+ if (ni != locked_ni)
down_read(&ni->runlist.lock);
- if (ntfs_attr_update_mapping_pairs(ni, 0))
+ if (__ntfs_attr_update_mapping_pairs(ni, 0, locked_ni, true))
ntfs_error(sb, "Failed to restore old mapping pairs");
- if (need_lock)
+ if (ni != locked_ni)
up_read(&ni->runlist.lock);
if (NInoSparse(ni) || NInoCompressed(ni)) {
@@ -4715,7 +4808,8 @@ attr_resize_again:
mark_mft_record_dirty(ctx->ntfs_ino);
ntfs_attr_put_search_ctx(ctx);
/* Resize non-resident attribute */
- return ntfs_non_resident_attr_expand(attr_ni, newsize, prealloc_size, holes, true);
+ return ntfs_non_resident_attr_expand(
+ attr_ni, newsize, prealloc_size, holes, NULL);
} else if (err != -ENOSPC && err != -EPERM) {
ntfs_error(sb, "Failed to make attribute non-resident");
goto put_err_out;
@@ -4836,7 +4930,7 @@ attr_resize_again:
}
/* Allocate new mft record. */
- err = ntfs_mft_record_alloc(base_ni->vol, 0, &ext_ni, base_ni, NULL);
+ err = ntfs_mft_record_alloc(base_ni->vol, 0, &ext_ni, base_ni, NULL, -1);
if (err) {
ntfs_error(sb, "Couldn't allocate MFT record");
goto put_err_out;
@@ -4890,13 +4984,14 @@ int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize,
if (NInoNonResident(ni)) {
if (newsize > i_size) {
down_write(&ni->runlist.lock);
- err = ntfs_non_resident_attr_expand(ni, newsize, 0,
- NVolDisableSparse(ni->vol) ?
- HOLES_NO : HOLES_OK,
- false);
+ err = ntfs_non_resident_attr_expand(
+ ni, newsize, 0,
+ NVolDisableSparse(ni->vol) ?
+ HOLES_NO : HOLES_OK, ni);
up_write(&ni->runlist.lock);
} else
- err = ntfs_non_resident_attr_shrink(ni, newsize);
+ err = ntfs_non_resident_attr_shrink(
+ ni, newsize, NULL);
} else
err = ntfs_resident_attr_resize(ni, newsize, 0,
NVolDisableSparse(ni->vol) ?
@@ -4905,7 +5000,9 @@ int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize,
return err;
}
-int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 prealloc_size)
+int ntfs_attr_expand_locked(struct ntfs_inode *ni, const s64 newsize,
+ const s64 prealloc_size,
+ struct ntfs_inode *locked_ni)
{
int err = 0;
@@ -4918,7 +5015,8 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo
ntfs_debug("Entering for inode 0x%llx, attr 0x%x, size %lld\n",
(unsigned long long)ni->mft_no, ni->type, newsize);
- if (ni->data_size == newsize) {
+ if (ni->data_size == newsize &&
+ (!prealloc_size || prealloc_size <= ni->allocated_size)) {
ntfs_debug("Size is already ok\n");
return 0;
}
@@ -4933,10 +5031,11 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo
}
if (NInoNonResident(ni)) {
- if (newsize > ni->data_size)
- err = ntfs_non_resident_attr_expand(ni, newsize, prealloc_size,
- NVolDisableSparse(ni->vol) ?
- HOLES_NO : HOLES_OK, true);
+ if (newsize > ni->data_size || prealloc_size > ni->allocated_size)
+ err = ntfs_non_resident_attr_expand(
+ ni, newsize, prealloc_size,
+ NVolDisableSparse(ni->vol) ?
+ HOLES_NO : HOLES_OK, locked_ni);
} else
err = ntfs_resident_attr_resize(ni, newsize, prealloc_size,
NVolDisableSparse(ni->vol) ?
@@ -4947,6 +5046,12 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo
return err;
}
+int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize,
+ const s64 prealloc_size)
+{
+ return ntfs_attr_expand_locked(ni, newsize, prealloc_size, NULL);
+}
+
/*
* ntfs_attr_truncate_i - resize an ntfs attribute
* @ni: open ntfs inode to resize
@@ -4959,7 +5064,9 @@ int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 preallo
* newly allocated space is marked as not initialised and no real allocation
* on disk is performed.
*/
-int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int holes)
+int ntfs_attr_truncate_i_locked(struct ntfs_inode *ni, const s64 newsize,
+ unsigned int holes,
+ struct ntfs_inode *locked_ni)
{
int err;
@@ -4993,15 +5100,23 @@ int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int
if (NInoNonResident(ni)) {
if (newsize > ni->data_size)
- err = ntfs_non_resident_attr_expand(ni, newsize, 0, holes, true);
+ err = ntfs_non_resident_attr_expand(
+ ni, newsize, 0, holes, locked_ni);
else
- err = ntfs_non_resident_attr_shrink(ni, newsize);
+ err = ntfs_non_resident_attr_shrink(
+ ni, newsize, locked_ni);
} else
err = ntfs_resident_attr_resize(ni, newsize, 0, holes);
ntfs_debug("Return status %d\n", err);
return err;
}
+int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize,
+ unsigned int holes)
+{
+ return ntfs_attr_truncate_i_locked(ni, newsize, holes, NULL);
+}
+
/*
* Resize an attribute, creating a hole if relevant
*/
@@ -5019,10 +5134,11 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start,
struct ntfs_volume *vol = ni->vol;
struct ntfs_attr_search_ctx *ctx;
struct runlist_element *rl, *rlc;
+ struct runlist_element *old_rl = NULL;
s64 vcn = vcn_start, lcn, clu_count;
s64 lcn_seek_from = -1;
int err = 0;
- size_t new_rl_count;
+ size_t new_rl_count, old_rl_count;
err = ntfs_attr_map_whole_runlist(ni);
if (err)
@@ -5115,6 +5231,19 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start,
WARN_ON(rlc->vcn != vcn);
lcn = rlc->lcn;
clu_count = rlc->length;
+ old_rl_count = ni->runlist.count;
+ old_rl = kmemdup(ni->runlist.rl,
+ old_rl_count * sizeof(*old_rl), GFP_NOFS);
+ if (!old_rl) {
+ err = -ENOMEM;
+ if (ntfs_cluster_free_from_rl(vol, rlc)) {
+ ntfs_error(vol->sb,
+ "Failed to free cluster allocation after runlist backup failure.");
+ NVolSetErrors(vol);
+ }
+ kvfree(rlc);
+ goto out;
+ }
rl = ntfs_runlists_merge(&ni->runlist, rlc, 0, &new_rl_count);
if (IS_ERR(rl)) {
@@ -5138,15 +5267,32 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start,
if (update_mp) {
ntfs_attr_reinit_search_ctx(ctx);
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
if (err) {
int err2;
err2 = ntfs_cluster_free(ni, vcn, clu_count, ctx);
- if (err2 < 0)
+ if (err2 < 0 || err2 != clu_count) {
ntfs_error(vol->sb,
- "Failed to free cluster allocation. Leaving inconstant metadata.\n");
- goto out;
+ "Failed to free cluster allocation. Leaving inconsistent metadata.\n");
+ NVolSetErrors(vol);
+ goto out;
+ }
+
+ /*
+ * Restore the runlist before repairing the on-disk
+ * mapping pairs.
+ */
+ kvfree(ni->runlist.rl);
+ ni->runlist.rl = old_rl;
+ ni->runlist.count = old_rl_count;
+ old_rl = NULL;
+ if (ntfs_attr_update_mapping_pairs_locked(
+ ni, 0, ni)) {
+ ntfs_error(vol->sb,
+ "Failed to restore mapping pairs after allocation rollback.\n");
+ NVolSetErrors(vol);
+ }
}
} else {
VFS_I(ni)->i_blocks += clu_count << (vol->cluster_size_bits - 9);
@@ -5158,6 +5304,7 @@ int ntfs_attr_map_cluster(struct ntfs_inode *ni, s64 vcn_start, s64 *lcn_start,
*lcn_count = clu_count;
*balloc = true;
out:
+ kvfree(old_rl);
ntfs_attr_put_search_ctx(ctx);
return err;
}
@@ -5401,7 +5548,7 @@ int ntfs_non_resident_attr_insert_range(struct ntfs_inode *ni, s64 start_vcn, s6
ni->data_size += ntfs_cluster_to_bytes(vol, len);
if (ntfs_cluster_to_bytes(vol, start_vcn) < ni->initialized_size)
ni->initialized_size += ntfs_cluster_to_bytes(vol, len);
- ret = ntfs_attr_update_mapping_pairs(ni, 0);
+ ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
up_write(&ni->runlist.lock);
if (ret)
return ret;
@@ -5486,7 +5633,7 @@ int ntfs_non_resident_attr_collapse_range(struct ntfs_inode *ni, s64 start_vcn,
}
if (ni->allocated_size > 0) {
- ret = ntfs_attr_update_mapping_pairs(ni, 0);
+ ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
if (ret) {
up_write(&ni->runlist.lock);
goto out_rl;
@@ -5564,7 +5711,7 @@ int ntfs_non_resident_attr_punch_hole(struct ntfs_inode *ni, s64 start_vcn, s64
ni->runlist.rl = rl;
ni->runlist.count = new_rl_count;
- ret = ntfs_attr_update_mapping_pairs(ni, 0);
+ ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
up_write(&ni->runlist.lock);
if (ret) {
kvfree(punch_rl);
@@ -5704,12 +5851,12 @@ int ntfs_attr_fallocate(struct ntfs_inode *ni, loff_t start, loff_t byte_len, bo
lcn << vol->cluster_size_bits,
alloc_cnt <<
vol->cluster_size_bits);
- if (err > 0)
+ if (err)
goto out;
}
if (signal_pending(current))
- goto out;
+ goto signal_out;
vcn += alloc_cnt;
try_alloc_cnt -= alloc_cnt;
@@ -5730,7 +5877,7 @@ int ntfs_attr_fallocate(struct ntfs_inode *ni, loff_t start, loff_t byte_len, bo
up_write(&ni->runlist.lock);
mutex_unlock(&ni->mrec_lock);
if (err || signal_pending(current))
- goto out;
+ goto signal_out;
vcn += alloc_cnt;
try_alloc_cnt -= alloc_cnt;
@@ -5740,7 +5887,7 @@ int ntfs_attr_fallocate(struct ntfs_inode *ni, loff_t start, loff_t byte_len, bo
if (NInoRunlistDirty(ni)) {
mutex_lock_nested(&ni->mrec_lock, NTFS_INODE_MUTEX_NORMAL);
down_write(&ni->runlist.lock);
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
if (err)
ntfs_error(ni->vol->sb, "Updating mapping pairs failed");
else
@@ -5756,4 +5903,8 @@ out_unmap:
mutex_unlock(&ni->mrec_lock);
out:
return err >= 0 ? 0 : err;
+signal_out:
+ if (!err)
+ err = -EINTR;
+ goto out;
}
diff --git a/fs/ntfs/attrib.h b/fs/ntfs/attrib.h
index e2224fbfaabe..6b4fa9f57640 100644
--- a/fs/ntfs/attrib.h
+++ b/fs/ntfs/attrib.h
@@ -112,7 +112,13 @@ int ntfs_non_resident_attr_punch_hole(struct ntfs_inode *ni, s64 start_vcn, s64
int __ntfs_attr_truncate_vfs(struct ntfs_inode *ni, const s64 newsize,
const s64 i_size);
int ntfs_attr_expand(struct ntfs_inode *ni, const s64 newsize, const s64 prealloc_size);
+int ntfs_attr_expand_locked(struct ntfs_inode *ni, const s64 newsize,
+ const s64 prealloc_size,
+ struct ntfs_inode *locked_ni);
int ntfs_attr_truncate_i(struct ntfs_inode *ni, const s64 newsize, unsigned int holes);
+int ntfs_attr_truncate_i_locked(struct ntfs_inode *ni, const s64 newsize,
+ unsigned int holes,
+ struct ntfs_inode *locked_ni);
int ntfs_attr_truncate(struct ntfs_inode *ni, const s64 newsize);
int ntfs_attr_rm(struct ntfs_inode *ni);
int ntfs_attr_exist(struct ntfs_inode *ni, const __le32 type, __le16 *name,
@@ -133,6 +139,9 @@ int ntfs_resident_attr_record_add(struct ntfs_inode *ni, __le32 type,
__le16 *name, u8 name_len, u8 *val, u32 size,
__le16 flags);
int ntfs_attr_update_mapping_pairs(struct ntfs_inode *ni, s64 from_vcn);
+int ntfs_attr_update_mapping_pairs_locked(struct ntfs_inode *ni,
+ s64 from_vcn,
+ struct ntfs_inode *locked_ni);
struct runlist_element *ntfs_attr_vcn_to_rl(struct ntfs_inode *ni, s64 vcn, s64 *lcn);
/*
diff --git a/fs/ntfs/attrlist.c b/fs/ntfs/attrlist.c
index be3086d34338..bb191953dcb1 100644
--- a/fs/ntfs/attrlist.c
+++ b/fs/ntfs/attrlist.c
@@ -12,6 +12,9 @@
#include "mft.h"
#include "attrib.h"
#include "attrlist.h"
+#include "lcnalloc.h"
+
+#define NTFS_MAX_ATTR_LIST_SIZE (256 * 1024)
/*
* ntfs_attrlist_need - check whether inode need attribute list
@@ -51,11 +54,155 @@ int ntfs_attrlist_need(struct ntfs_inode *ni)
return 0;
}
-int ntfs_attrlist_update(struct ntfs_inode *base_ni)
+/*
+ * Repack the $MFT/$ATTRIBUTE_LIST data into one run.
+ *
+ * The mapping pairs for an $ATTRIBUTE_LIST must remain in the base MFT
+ * record. Once that record has no room left, extending a fragmented list
+ * can require one more mapping-pairs byte than the record can hold. There
+ * is no attribute that can legally be moved out in that state: $STANDARD_
+ * INFORMATION, $ATTRIBUTE_LIST, and the first $MFT/$DATA extent all have to
+ * stay in the base record. Move the list data to one contiguous run. The
+ * caller supplies the minimum allocation size so a recovery can use the
+ * smallest useful run while normal updates can still request the maximum
+ * legal list size as a reserve.
+ */
+static int ntfs_attrlist_repack(struct inode *attr_vi,
+ struct ntfs_inode *attr_ni, s64 min_alloc_size,
+ struct ntfs_inode *locked_ni)
+{
+ struct ntfs_volume *vol = attr_ni->vol;
+ struct runlist_element *old_rl, *new_rl;
+ u8 *data = NULL;
+ s64 data_size, alloc_size, nr_clusters, written;
+ s64 old_alloc_size;
+ size_t old_rl_count, new_rl_count;
+ unsigned long flags;
+ int err, restore_err;
+ if (attr_ni->mft_no != FILE_MFT || !NInoNonResident(attr_ni) ||
+ min_alloc_size < 0)
+ return -EINVAL;
+ /* The buffered I/O below can reacquire the attribute runlist lock. */
+ if (attr_ni == locked_ni)
+ return -ENOSPC;
+
+ err = ntfs_attr_map_whole_runlist(attr_ni);
+ if (err)
+ return err;
+
+ data_size = attr_ni->data_size;
+ if (data_size < 0)
+ return -EIO;
+
+ if (data_size) {
+ data = kvmalloc(data_size, GFP_NOFS);
+ if (!data)
+ return -ENOMEM;
+
+ written = ntfs_inode_attr_pread(attr_vi, 0, data_size, data);
+ if (written != data_size) {
+ err = written < 0 ? (int)written : -EIO;
+ goto out_free_data;
+ }
+ }
+
+ old_alloc_size = attr_ni->allocated_size;
+ alloc_size = max_t(s64, old_alloc_size, min_alloc_size);
+ nr_clusters = ntfs_bytes_to_cluster(vol,
+ alloc_size + vol->cluster_size - 1);
+ if (nr_clusters <= 0) {
+ err = -EFBIG;
+ goto out_free_data;
+ }
+
+ /* A single run keeps the mapping pairs at the minimum size. */
+ new_rl = ntfs_cluster_alloc(vol, 0, nr_clusters, -1, DATA_ZONE,
+ true, true, false);
+ if (IS_ERR(new_rl)) {
+ err = PTR_ERR(new_rl);
+ goto out_free_data;
+ }
+
+ new_rl_count = 0;
+ if (new_rl->vcn == 0 && new_rl->length == nr_clusters &&
+ !new_rl[1].length)
+ new_rl_count = 2;
+
+ if (new_rl_count != 2) {
+ ntfs_cluster_free_from_rl(vol, new_rl);
+ kvfree(new_rl);
+ err = -ENOSPC;
+ goto out_free_data;
+ }
+ old_rl = attr_ni->runlist.rl;
+ old_rl_count = attr_ni->runlist.count;
+ down_write(&attr_ni->runlist.lock);
+ attr_ni->runlist.rl = new_rl;
+ attr_ni->runlist.count = new_rl_count;
+ up_write(&attr_ni->runlist.lock);
+
+ write_lock_irqsave(&attr_ni->size_lock, flags);
+ attr_ni->allocated_size = ntfs_cluster_to_bytes(vol, nr_clusters);
+ write_unlock_irqrestore(&attr_ni->size_lock, flags);
+
+ /* Populate the replacement extent before publishing its mapping pairs. */
+ if (data_size) {
+ written = ntfs_inode_attr_pwrite(attr_vi, 0, data_size, data, true);
+ if (written != data_size) {
+ err = written < 0 ? (int)written : -EIO;
+ goto restore_old_runlist;
+ }
+ }
+
+ err = ntfs_attr_update_mapping_pairs_locked(attr_ni, 0, locked_ni);
+ if (err)
+ goto restore_old_runlist;
+
+ /* The new mapping is now authoritative; release the old data runs. */
+ if (ntfs_cluster_free_from_rl(vol, old_rl)) {
+ ntfs_error(vol->sb,
+ "Failed to free old ATTRIBUTE_LIST extent: inode %#llx",
+ (long long)attr_ni->mft_no);
+ NVolSetErrors(vol);
+ }
+ kvfree(old_rl);
+ kvfree(data);
+ return 0;
+
+restore_old_runlist:
+ down_write(&attr_ni->runlist.lock);
+ attr_ni->runlist.rl = old_rl;
+ attr_ni->runlist.count = old_rl_count;
+ up_write(&attr_ni->runlist.lock);
+
+ write_lock_irqsave(&attr_ni->size_lock, flags);
+ attr_ni->allocated_size = old_alloc_size;
+ write_unlock_irqrestore(&attr_ni->size_lock, flags);
+
+ restore_err = ntfs_attr_update_mapping_pairs_locked(
+ attr_ni, 0, locked_ni);
+ if (restore_err) {
+ ntfs_error(vol->sb, "Failed to restore ATTRIBUTE_LIST mapping pairs (%d)",
+ restore_err);
+ NVolSetErrors(vol);
+ }
+
+ ntfs_cluster_free_from_rl(vol, new_rl);
+ kvfree(new_rl);
+ err = err ? err : restore_err;
+
+out_free_data:
+ kvfree(data);
+ return err;
+}
+
+int ntfs_attrlist_update_locked(struct ntfs_inode *base_ni,
+ struct ntfs_inode *locked_ni)
{
struct inode *attr_vi;
struct ntfs_inode *attr_ni;
- int err;
+ s64 written;
+ int err, retry_err;
/*
* generic_shutdown_super() clears SB_ACTIVE before evicting cached
@@ -72,23 +219,66 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni)
return err;
}
attr_ni = NTFS_I(attr_vi);
+ /* Truncation and page-cache writes can reacquire this runlist lock. */
+ if (attr_ni == locked_ni) {
+ iput(attr_vi);
+ return -ENOSPC;
+ }
- err = ntfs_attr_truncate_i(attr_ni, base_ni->attr_list_size, HOLES_NO);
- if (err == -ENOSPC && attr_ni->mft_no == FILE_MFT) {
- err = ntfs_attr_truncate(attr_ni, 0);
- if (err || ntfs_attr_truncate_i(attr_ni, base_ni->attr_list_size, HOLES_NO) != 0) {
+ err = ntfs_attr_truncate_i_locked(
+ attr_ni, base_ni->attr_list_size, HOLES_NO, locked_ni);
+ if (err == -ENOSPC && attr_ni->mft_no == FILE_MFT &&
+ NInoNonResident(attr_ni)) {
+ retry_err = ntfs_attrlist_repack(attr_vi, attr_ni,
+ base_ni->attr_list_size, locked_ni);
+ if (retry_err) {
+ ntfs_error(base_ni->vol->sb, "Failed to repack attribute list");
iput(attr_vi);
+ return retry_err;
+ }
+
+ retry_err = ntfs_attr_truncate_i_locked(
+ attr_ni, base_ni->attr_list_size,
+ HOLES_NO, locked_ni);
+ if (retry_err) {
ntfs_error(base_ni->vol->sb,
- "Failed to truncate attribute list of inode %#llx",
- (long long)base_ni->mft_no);
- return -EIO;
+ "Failed to resize attribute list after repack");
+ iput(attr_vi);
+ return retry_err;
}
} else if (err) {
iput(attr_vi);
ntfs_error(base_ni->vol->sb,
"Failed to truncate attribute list of inode %#llx",
(long long)base_ni->mft_no);
- return -EIO;
+ return err;
+ }
+
+ /*
+ * Reserve the maximum legal list size while the MFT metadata area is
+ * still easy to allocate contiguously. This prevents a later list entry
+ * from needing another mapping-pairs byte in the full base MFT record.
+ * Failure to obtain the optional reserve must not reject the current
+ * metadata update; the repack retry above remains available if needed.
+ */
+ if (base_ni->mft_no == FILE_MFT && NInoNonResident(attr_ni) &&
+ attr_ni->allocated_size < NTFS_MAX_ATTR_LIST_SIZE) {
+ retry_err = ntfs_attr_expand_locked(
+ attr_ni, base_ni->attr_list_size,
+ NTFS_MAX_ATTR_LIST_SIZE, locked_ni);
+ if (retry_err == -ENOSPC) {
+ retry_err = ntfs_attrlist_repack(
+ attr_vi, attr_ni,
+ NTFS_MAX_ATTR_LIST_SIZE, locked_ni);
+ if (retry_err == -ENOSPC)
+ retry_err = 0;
+ }
+ if (retry_err) {
+ ntfs_error(base_ni->vol->sb,
+ "Failed to reserve attribute list space");
+ iput(attr_vi);
+ return retry_err;
+ }
}
i_size_write(attr_vi, base_ni->attr_list_size);
@@ -96,14 +286,15 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni)
if (NInoNonResident(attr_ni) && !NInoAttrListNonResident(base_ni))
NInoSetAttrListNonResident(base_ni);
- if (ntfs_inode_attr_pwrite(attr_vi, 0, base_ni->attr_list_size,
- base_ni->attr_list, false) !=
- base_ni->attr_list_size) {
+ written = ntfs_inode_attr_pwrite(attr_vi, 0, base_ni->attr_list_size,
+ base_ni->attr_list, false);
+ if (written != base_ni->attr_list_size) {
+ err = written < 0 ? (int)written : -EIO;
iput(attr_vi);
ntfs_error(base_ni->vol->sb,
"Failed to write attribute list of inode %#llx",
(long long)base_ni->mft_no);
- return -EIO;
+ return err;
}
NInoSetAttrListDirty(base_ni);
@@ -111,6 +302,11 @@ int ntfs_attrlist_update(struct ntfs_inode *base_ni)
return 0;
}
+int ntfs_attrlist_update(struct ntfs_inode *base_ni)
+{
+ return ntfs_attrlist_update_locked(base_ni, NULL);
+}
+
/*
* ntfs_attrlist_entry_add - add an attribute list attribute entry
* @ni: opened ntfs inode, which contains that attribute
diff --git a/fs/ntfs/attrlist.h b/fs/ntfs/attrlist.h
index 1892a3934d3a..10cc2cc8e208 100644
--- a/fs/ntfs/attrlist.h
+++ b/fs/ntfs/attrlist.h
@@ -16,5 +16,7 @@ int ntfs_attrlist_need(struct ntfs_inode *ni);
int ntfs_attrlist_entry_add(struct ntfs_inode *ni, struct attr_record *attr);
int ntfs_attrlist_entry_rm(struct ntfs_attr_search_ctx *ctx);
int ntfs_attrlist_update(struct ntfs_inode *base_ni);
+int ntfs_attrlist_update_locked(struct ntfs_inode *base_ni,
+ struct ntfs_inode *locked_ni);
#endif /* defined _NTFS_ATTRLIST_H */
diff --git a/fs/ntfs/bdev-io.c b/fs/ntfs/bdev-io.c
index 86db4d9298ed..4f27eed3b072 100644
--- a/fs/ntfs/bdev-io.c
+++ b/fs/ntfs/bdev-io.c
@@ -34,7 +34,7 @@ int ntfs_bdev_read(struct block_device *bdev, char *data, loff_t start, size_t s
int error;
struct bio *bio;
blk_opf_t op;
- sector_t sector = start >> SECTOR_SHIFT;
+ sector_t sector = ntfs_bytes_to_bio_sector(start);
if (start & (SECTOR_SIZE - 1))
return -EINVAL;
diff --git a/fs/ntfs/bitmap.c b/fs/ntfs/bitmap.c
index b1436b3151b9..5a4457551306 100644
--- a/fs/ntfs/bitmap.c
+++ b/fs/ntfs/bitmap.c
@@ -40,7 +40,7 @@ int ntfs_trim_fs(struct ntfs_volume *vol, struct fstrim_range *range)
end_cluster = vol->nr_clusters;
}
- ra = kzalloc(sizeof(*ra), GFP_NOFS);
+ ra = kzalloc_obj(*ra, GFP_NOFS);
if (!ra)
return -ENOMEM;
@@ -64,7 +64,7 @@ int ntfs_trim_fs(struct ntfs_volume *vol, struct fstrim_range *range)
end = start_buf;
while (end < end_buf) {
- u64 aligned_start, aligned_count;
+ u64 aligned_start, aligned_end, aligned_count;
u64 start = find_next_zero_bit(bitmap, end_buf - start_buf,
end - start_buf) + start_buf;
if (start >= end_buf)
@@ -74,8 +74,10 @@ int ntfs_trim_fs(struct ntfs_volume *vol, struct fstrim_range *range)
start - start_buf) + start_buf;
aligned_start = ALIGN(ntfs_cluster_to_bytes(vol, start), dq);
- aligned_count =
- ALIGN_DOWN(ntfs_cluster_to_bytes(vol, end - start), dq);
+ aligned_end = ALIGN_DOWN(ntfs_cluster_to_bytes(vol, end), dq);
+ if (aligned_start >= aligned_end)
+ continue;
+ aligned_count = aligned_end - aligned_start;
if (aligned_count >= range->minlen) {
ret = blkdev_issue_discard(vol->sb->s_bdev, aligned_start >> 9,
aligned_count >> 9, GFP_NOFS);
diff --git a/fs/ntfs/compress.c b/fs/ntfs/compress.c
index 2225630b19d7..075b57fc1de6 100644
--- a/fs/ntfs/compress.c
+++ b/fs/ntfs/compress.c
@@ -514,8 +514,8 @@ int ntfs_read_compressed_block(struct folio *folio)
return -EIO;
}
- pages = kmalloc_array(nr_pages, sizeof(struct page *), GFP_NOFS);
- completed_pages = kmalloc_array(nr_pages + 1, sizeof(int), GFP_NOFS);
+ pages = kmalloc_objs(struct page *, nr_pages, GFP_NOFS);
+ completed_pages = kmalloc_objs(int, nr_pages + 1, GFP_NOFS);
if (unlikely(!pages || !completed_pages)) {
kfree(pages);
@@ -1262,7 +1262,7 @@ static int ntfs_compress_workspace_init(struct ntfs_inode *ni,
size = ni->itype.compressed.block_size + 2 *
(ni->itype.compressed.block_size / NTFS_SB_SIZE) + 2;
ws->nr_pages = DIV_ROUND_UP(size, PAGE_SIZE);
- ws->pages = kcalloc(ws->nr_pages, sizeof(*ws->pages), GFP_NOFS);
+ ws->pages = kzalloc_objs(*ws->pages, ws->nr_pages, GFP_NOFS);
if (!ws->pages)
return -ENOMEM;
@@ -1414,7 +1414,7 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
bio_pos = ntfs_cluster_to_bytes(vol, bio_lcn);
bio = bio_alloc(vol->sb->s_bdev, DIV_ROUND_UP(bio_size, PAGE_SIZE),
REQ_OP_WRITE, GFP_NOIO);
- bio->bi_iter.bi_sector = ntfs_bytes_to_sector(vol, bio_pos);
+ bio->bi_iter.bi_sector = ntfs_bytes_to_bio_sector(bio_pos);
for (i = 0; bio_size; i++) {
unsigned int len = min_t(unsigned int, bio_size, PAGE_SIZE);
@@ -1450,7 +1450,7 @@ static int ntfs_write_cb(struct ntfs_inode *ni, loff_t pos, struct page **pages,
ni->runlist.rl = rl;
rlc = NULL;
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
up_write(&ni->runlist.lock);
if (err)
err = -EIO;
@@ -1483,7 +1483,7 @@ int ntfs_compress_write(struct ntfs_inode *ni, loff_t pos, size_t count,
pages_per_cb = DIV_ROUND_UP(offset_in_page(pos & ~(cb_size - 1)) +
cb_size, PAGE_SIZE);
- pages = kmalloc_array(pages_per_cb, sizeof(struct page *), GFP_NOFS);
+ pages = kmalloc_objs(struct page *, pages_per_cb, GFP_NOFS);
if (!pages)
return -ENOMEM;
ctx = kvzalloc_obj(*ctx, GFP_NOFS);
diff --git a/fs/ntfs/dir.c b/fs/ntfs/dir.c
index 2d594cbb4ebe..df60138f9b2d 100644
--- a/fs/ntfs/dir.c
+++ b/fs/ntfs/dir.c
@@ -166,8 +166,8 @@ found_it:
*/
if (ie->key.file_name.file_name_type == FILE_NAME_DOS) {
if (!name) {
- name = kmalloc(sizeof(struct ntfs_name),
- GFP_NOFS);
+ name = kmalloc_obj(struct ntfs_name,
+ GFP_NOFS);
if (!name) {
err = -ENOMEM;
goto err_out;
@@ -401,8 +401,8 @@ found_it2:
*/
if (ie->key.file_name.file_name_type == FILE_NAME_DOS) {
if (!name) {
- name = kmalloc(sizeof(struct ntfs_name),
- GFP_NOFS);
+ name = kmalloc_obj(struct ntfs_name,
+ GFP_NOFS);
if (!name) {
err = -ENOMEM;
goto unm_err_out;
@@ -700,7 +700,7 @@ static int ntfs_ia_blocks_readahead(struct ntfs_inode *ia_ni, loff_t pos)
if (dir_start_index >= dir_end_index)
return 0;
- dir_ra = kzalloc(sizeof(*dir_ra), GFP_NOFS);
+ dir_ra = kzalloc_obj(*dir_ra, GFP_NOFS);
if (!dir_ra)
return -ENOMEM;
@@ -777,7 +777,7 @@ static int ntfs_readdir(struct file *file, struct dir_context *actor)
return -ENOMEM;
}
- ra = kzalloc(sizeof(struct file_ra_state), GFP_NOFS);
+ ra = kzalloc_obj(struct file_ra_state, GFP_NOFS);
if (!ra) {
kfree(name);
ntfs_index_ctx_put(ictx);
@@ -813,7 +813,7 @@ static int ntfs_readdir(struct file *file, struct dir_context *actor)
goto out;
}
} else if (!private) {
- private = kzalloc(sizeof(struct ntfs_file_private), GFP_KERNEL);
+ private = kzalloc_obj(struct ntfs_file_private);
if (!private) {
err = -ENOMEM;
goto out;
@@ -949,7 +949,7 @@ nextdir:
}
if (!nir) {
- nir = kzalloc(sizeof(struct ntfs_index_ra), GFP_KERNEL);
+ nir = kzalloc_obj(struct ntfs_index_ra);
if (nir) {
nir->start_index = index;
nir->count = 1;
diff --git a/fs/ntfs/ea.c b/fs/ntfs/ea.c
index cdd306933d73..b4fcfbe2da4c 100644
--- a/fs/ntfs/ea.c
+++ b/fs/ntfs/ea.c
@@ -235,7 +235,7 @@ static int ntfs_set_ea(struct inode *inode, const char *name, size_t name_len,
ea_info_qsize = le32_to_cpu(p_ea_info->ea_query_length);
} else {
create_ea_info:
- p_ea_info = kzalloc(sizeof(struct ea_information), GFP_NOFS);
+ p_ea_info = kzalloc_obj(struct ea_information, GFP_NOFS);
if (!p_ea_info)
return -ENOMEM;
@@ -404,10 +404,12 @@ alloc_new_ea:
*packed_ea_size = p_ea_info->ea_length;
mark_mft_record_dirty(ni);
out:
- if (ea_info_qsize > 0)
- NInoSetHasEA(ni);
- else
- NInoClearHasEA(ni);
+ if (!err) {
+ if (ea_info_qsize > 0)
+ NInoSetHasEA(ni);
+ else
+ NInoClearHasEA(ni);
+ }
kvfree(ea_buf);
kvfree(old_ea_buf);
@@ -615,7 +617,7 @@ static int ntfs_getxattr(const struct xattr_handler *handler,
if (!buffer) {
err = sizeof(u8);
} else if (size < sizeof(u8)) {
- err = -ENODATA;
+ err = -ERANGE;
} else {
err = sizeof(u8);
*(u8 *)buffer = (u8)(le32_to_cpu(ni->flags) & 0x3F);
@@ -628,7 +630,7 @@ static int ntfs_getxattr(const struct xattr_handler *handler,
if (!buffer) {
err = sizeof(u32);
} else if (size < sizeof(u32)) {
- err = -ENODATA;
+ err = -ERANGE;
} else {
err = sizeof(u32);
*(u32 *)buffer = le32_to_cpu(ni->flags);
@@ -753,18 +755,39 @@ static int ntfs_new_attr_flags(struct ntfs_inode *ni, __le32 fattr)
old_arec_size = le32_to_cpu(a->length);
/*
- * Move payloads before shrinking the record. Otherwise resizing moves
+ * Move payloads before shrinking the record. Otherwise resizing moves
* the following attribute over the old payload before it can be copied.
+ *
+ * When offsets increase, move mapping_pairs first to avoid name
+ * overwriting the start of mapping_pairs.
*/
if (arec_size < old_arec_size) {
- if (a->name_length && name_ofs != old_name_ofs)
- memmove((u8 *)a + name_ofs, (u8 *)a + old_name_ofs,
- a->name_length * sizeof(__le16));
- if (mp_ofs != old_mp_ofs)
- memmove((u8 *)a + mp_ofs, (u8 *)a + old_mp_ofs, mp_size);
+ if (name_ofs > old_name_ofs) {
+ /* Payload offsets increased: move mapping pairs first. */
+ if (mp_ofs != old_mp_ofs)
+ memmove((u8 *)a + mp_ofs,
+ (u8 *)a + old_mp_ofs,
+ mp_size);
+ if (a->name_length && name_ofs != old_name_ofs)
+ memmove((u8 *)a + name_ofs,
+ (u8 *)a + old_name_ofs,
+ a->name_length *
+ sizeof(__le16));
+ } else {
+ /* Payload offsets decreased or unchanged: move name first. */
+ if (a->name_length && name_ofs != old_name_ofs)
+ memmove((u8 *)a + name_ofs,
+ (u8 *)a + old_name_ofs,
+ a->name_length *
+ sizeof(__le16));
+ if (mp_ofs != old_mp_ofs)
+ memmove((u8 *)a + mp_ofs,
+ (u8 *)a + old_mp_ofs,
+ mp_size);
+ }
}
- err = ntfs_attr_record_resize(m, a, arec_size);
+ err = ntfs_attr_record_resize(ctx->mrec, a, arec_size);
if (unlikely(err))
goto err_out;
diff --git a/fs/ntfs/file.c b/fs/ntfs/file.c
index 88747217ba61..007d1614b9ac 100644
--- a/fs/ntfs/file.c
+++ b/fs/ntfs/file.c
@@ -111,7 +111,8 @@ static int ntfs_trim_prealloc(struct inode *vi)
ntfs_error(vol->sb, "Preallocated block rollback failed");
} else {
ni->allocated_size = ntfs_cluster_to_bytes(vol, vcn_tr);
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(
+ ni, 0, ni);
if (err)
ntfs_error(vol->sb,
"Failed to rollback mapping pairs for prealloc");
@@ -270,18 +271,25 @@ static int ntfs_setattr_size(struct inode *vi, struct iattr *attr)
return err;
inode_dio_wait(vi);
+
+ /*
+ * Serialize with page faults and pagecache instantiation so that
+ * readers cannot observe the size change until the attribute
+ * updates below have completed.
+ */
+ filemap_invalidate_lock(vi->i_mapping);
if (attr->ia_size > old_size) {
truncate_pagecache(vi, old_size);
i_size_write(vi, attr->ia_size);
pagecache_isize_extended(vi, old_size, attr->ia_size);
- } else
+ } else {
truncate_setsize(vi, attr->ia_size);
+ }
err = ntfs_truncate_vfs(vi, attr->ia_size, old_size);
- if (err) {
+ if (err)
i_size_write(vi, old_size);
- return err;
- }
+ filemap_invalidate_unlock(vi->i_mapping);
return err;
}
@@ -669,6 +677,7 @@ out_lock:
static vm_fault_t ntfs_filemap_page_mkwrite(struct vm_fault *vmf)
{
struct inode *inode = file_inode(vmf->vma->vm_file);
+ struct address_space *mapping = inode->i_mapping;
vm_fault_t ret;
if (NInoWofCompressed(NTFS_I(inode)))
@@ -677,7 +686,14 @@ static vm_fault_t ntfs_filemap_page_mkwrite(struct vm_fault *vmf)
sb_start_pagefault(inode->i_sb);
file_update_time(vmf->vma->vm_file);
+ /*
+ * Serialize against truncate/fallocate which hold the lock
+ * exclusively while invalidating pagecache and changing extents.
+ */
+ filemap_invalidate_lock_shared(mapping);
ret = iomap_page_mkwrite(vmf, &ntfs_page_mkwrite_iomap_ops, NULL);
+ filemap_invalidate_unlock_shared(mapping);
+
sb_end_pagefault(inode->i_sb);
return ret;
}
@@ -1116,7 +1132,6 @@ static long ntfs_fallocate(struct file *file, int mode, loff_t offset, loff_t le
struct ntfs_volume *vol = ni->vol;
int err = 0;
loff_t old_size;
- bool map_locked = false;
if (mode & ~(NTFS_FALLOC_FL_SUPPORTED))
return -EOPNOTSUPP;
@@ -1148,16 +1163,13 @@ static long ntfs_fallocate(struct file *file, int mode, loff_t offset, loff_t le
inode_lock(vi);
if (NInoCompressed(ni) || NInoEncrypted(ni) || NInoWofCompressed(ni)) {
- err = -EOPNOTSUPP;
- goto out;
+ inode_unlock(vi);
+ return -EOPNOTSUPP;
}
inode_dio_wait(vi);
- if (mode & (FALLOC_FL_PUNCH_HOLE | FALLOC_FL_COLLAPSE_RANGE |
- FALLOC_FL_INSERT_RANGE)) {
- filemap_invalidate_lock(vi->i_mapping);
- map_locked = true;
- }
+ /* Take invalidate_lock for all fallocate operations to prevent races */
+ filemap_invalidate_lock(vi->i_mapping);
switch (mode & FALLOC_FL_MODE_MASK) {
case FALLOC_FL_ALLOCATE_RANGE:
@@ -1182,14 +1194,15 @@ static long ntfs_fallocate(struct file *file, int mode, loff_t offset, loff_t le
err = file_modified(file);
out:
- if (map_locked)
- filemap_invalidate_unlock(vi->i_mapping);
+ if (!err && mode == 0 && NInoNonResident(ni) &&
+ offset > old_size) {
+ truncate_pagecache(vi, old_size);
+ pagecache_isize_extended(vi, old_size, offset);
+ }
+
+ filemap_invalidate_unlock(vi->i_mapping);
+
if (!err) {
- if (mode == 0 && NInoNonResident(ni) &&
- offset > old_size) {
- truncate_pagecache(vi, old_size);
- pagecache_isize_extended(vi, old_size, offset);
- }
NInoSetFileNameDirty(ni);
inode_set_mtime_to_ts(vi, inode_set_ctime_current(vi));
mark_inode_dirty(vi);
diff --git a/fs/ntfs/index.c b/fs/ntfs/index.c
index 46a8b19c0723..580998990bc9 100644
--- a/fs/ntfs/index.c
+++ b/fs/ntfs/index.c
@@ -1660,7 +1660,7 @@ resplit:
goto out;
}
} else {
- si = kzalloc(sizeof(struct split_info), GFP_NOFS);
+ si = kzalloc_obj(struct split_info, GFP_NOFS);
if (!si) {
ntfs_ibm_clear(icx, new_vcn);
ret = -ENOMEM;
diff --git a/fs/ntfs/inode.c b/fs/ntfs/inode.c
index 32edb4045178..a777de8a80c7 100644
--- a/fs/ntfs/inode.c
+++ b/fs/ntfs/inode.c
@@ -170,17 +170,19 @@ struct inode *ntfs_iget(struct super_block *sb, u64 mft_no)
/* If this is a freshly allocated inode, need to read it now. */
if (inode_state_read_once(vi) & I_NEW) {
err = ntfs_read_locked_inode(vi);
- unlock_new_inode(vi);
+ if (err) {
+ remove_inode_hash(vi);
+ discard_new_inode(vi);
+ } else
+ unlock_new_inode(vi);
}
/*
* There is no point in keeping bad inodes around. This also
* simplifies things in that we never need to check for bad inodes
* elsewhere.
*/
- if (unlikely(err)) {
- iput(vi);
+ if (unlikely(err))
vi = ERR_PTR(err);
- }
return vi;
}
@@ -231,17 +233,19 @@ struct inode *ntfs_attr_iget(struct inode *base_vi, __le32 type,
/* If this is a freshly allocated inode, need to read it now. */
if (inode_state_read_once(vi) & I_NEW) {
err = ntfs_read_locked_attr_inode(base_vi, vi);
- unlock_new_inode(vi);
+ if (err) {
+ remove_inode_hash(vi);
+ discard_new_inode(vi);
+ } else
+ unlock_new_inode(vi);
}
/*
* There is no point in keeping bad attribute inodes around. This also
* simplifies things in that we never need to check for bad attribute
* inodes elsewhere.
*/
- if (unlikely(err)) {
- iput(vi);
+ if (unlikely(err))
vi = ERR_PTR(err);
- }
return vi;
}
@@ -286,17 +290,19 @@ struct inode *ntfs_index_iget(struct inode *base_vi, __le16 *name,
/* If this is a freshly allocated inode, need to read it now. */
if (inode_state_read_once(vi) & I_NEW) {
err = ntfs_read_locked_index_inode(base_vi, vi);
- unlock_new_inode(vi);
+ if (err) {
+ remove_inode_hash(vi);
+ discard_new_inode(vi);
+ } else
+ unlock_new_inode(vi);
}
/*
* There is no point in keeping bad index inodes around. This also
* simplifies things in that we never need to check for bad index
* inodes elsewhere.
*/
- if (unlikely(err)) {
- iput(vi);
+ if (unlikely(err))
vi = ERR_PTR(err);
- }
return vi;
}
@@ -1241,7 +1247,8 @@ unm_err_out:
if (m)
unmap_mft_record(ni);
err_out:
- if (err != -EOPNOTSUPP && err != -ENOMEM && vol_err == true) {
+ if (err != -EOPNOTSUPP && err != -ENOMEM &&
+ err != -EINTR && err != -ERESTARTSYS && vol_err == true) {
ntfs_error(vol->sb,
"Failed with error code %i. Marking corrupt inode 0x%llx as bad. Run chkdsk.",
err, ni->mft_no);
@@ -1467,12 +1474,13 @@ unm_err_out:
ntfs_attr_put_search_ctx(ctx);
unmap_mft_record(base_ni);
err_out:
- if (err != -ENOENT)
+ if (err != -ENOENT && err != -EINTR && err != -ERESTARTSYS)
ntfs_error(vol->sb,
"Failed with error code %i while reading attribute inode (mft_no 0x%llx, type 0x%x, name_len %i). Marking corrupt inode and base inode 0x%llx as bad. Run chkdsk.",
err, ni->mft_no, ni->type, ni->name_len,
base_ni->mft_no);
- if (err != -ENOENT && err != -ENOMEM)
+ if (err != -ENOENT && err != -ENOMEM &&
+ err != -EINTR && err != -ERESTARTSYS)
NVolSetErrors(vol);
return err;
}
@@ -1676,8 +1684,9 @@ static int ntfs_read_locked_index_inode(struct inode *base_vi, struct inode *vi)
/* Get the index bitmap attribute inode. */
bvi = ntfs_attr_iget(base_vi, AT_BITMAP, ni->name, ni->name_len);
if (IS_ERR(bvi)) {
- ntfs_error(vi->i_sb, "Failed to get bitmap attribute.");
err = PTR_ERR(bvi);
+ if (err != -EINTR && err != -ERESTARTSYS)
+ ntfs_error(vi->i_sb, "Failed to get bitmap attribute.");
goto unm_err_out;
}
bni = NTFS_I(bvi);
@@ -1721,10 +1730,12 @@ unm_err_out:
if (m)
unmap_mft_record(base_ni);
err_out:
- ntfs_error(vi->i_sb,
- "Failed with error code %i while reading index inode (mft_no 0x%llx, name_len %i.",
- err, ni->mft_no, ni->name_len);
- if (err != -EOPNOTSUPP && err != -ENOMEM)
+ if (err != -EINTR && err != -ERESTARTSYS)
+ ntfs_error(vi->i_sb,
+ "Failed with error code %i while reading index inode (mft_no 0x%llx, name_len %i.",
+ err, ni->mft_no, ni->name_len);
+ if (err != -EOPNOTSUPP && err != -ENOMEM &&
+ err != -EINTR && err != -ERESTARTSYS)
NVolSetErrors(vol);
return err;
}
@@ -1852,7 +1863,7 @@ int ntfs_read_inode_mount(struct inode *vi)
struct mft_record *m = NULL;
struct attr_record *a;
struct ntfs_attr_search_ctx *ctx;
- unsigned int i, nr_blocks;
+ unsigned int i;
int err;
size_t new_rl_count;
@@ -1896,11 +1907,6 @@ int ntfs_read_inode_mount(struct inode *vi)
goto err_out;
}
- /* Determine the first block of the $MFT/$DATA attribute. */
- nr_blocks = ntfs_bytes_to_sector(vol, vol->mft_record_size);
- if (!nr_blocks)
- nr_blocks = 1;
-
/* Load $MFT/$DATA's first mft record. */
err = ntfs_bdev_read(sb->s_bdev, (char *)m,
ntfs_cluster_to_bytes(vol, vol->mft_lcn), i);
@@ -2777,7 +2783,7 @@ int __ntfs_write_inode(struct inode *vi, int sync)
if (NInoNonResident(ni) && NInoRunlistDirty(ni)) {
down_write(&ni->runlist.lock);
- err = ntfs_attr_update_mapping_pairs(ni, 0);
+ err = ntfs_attr_update_mapping_pairs_locked(ni, 0, ni);
if (!err)
NInoClearRunlistDirty(ni);
up_write(&ni->runlist.lock);
@@ -3718,7 +3724,7 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi,
FGP_CREAT | FGP_LOCK,
mapping_gfp_mask(mapping));
if (IS_ERR(folio)) {
- ret = -ENOMEM;
+ ret = PTR_ERR(folio);
break;
}
} else {
@@ -3750,6 +3756,7 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi,
u64 rl_length = 0;
s64 vcn;
struct runlist_element *rl;
+ int bio_err;
lcn_count = max_t(s64, 1, ntfs_bytes_to_cluster(vol, attr_len));
vcn = ntfs_pidx_to_cluster(vol, folio->index);
@@ -3780,8 +3787,7 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi,
bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE,
GFP_NOIO);
bio->bi_iter.bi_sector =
- ntfs_bytes_to_sector(vol,
- ntfs_cluster_to_bytes(vol, lcn) +
+ ntfs_bytes_to_bio_sector(ntfs_cluster_to_bytes(vol, lcn) +
lcn_folio_off);
length = min_t(unsigned long,
@@ -3793,8 +3799,15 @@ static s64 __ntfs_inode_non_resident_attr_pwrite(struct inode *vi,
goto err_unlock_folio;
}
- submit_bio_wait(bio);
+ bio_err = submit_bio_wait(bio);
bio_put(bio);
+ if (bio_err) {
+ ntfs_error(vi->i_sb,
+ "Synchronous attribute write failed (%d)",
+ bio_err);
+ ret = bio_err;
+ goto err_unlock_folio;
+ }
vcn += rl_length;
offset += length;
} while (lcn_count != 0);
diff --git a/fs/ntfs/lcnalloc.c b/fs/ntfs/lcnalloc.c
index aa2e017a4384..0d6cd08ee2e7 100644
--- a/fs/ntfs/lcnalloc.c
+++ b/fs/ntfs/lcnalloc.c
@@ -53,10 +53,10 @@ int ntfs_cluster_free_from_rl_nolock(struct ntfs_volume *vol,
if (rl->lcn < 0)
continue;
err = ntfs_bitmap_clear_run(lcnbmp_vi, rl->lcn, rl->length);
- if (unlikely(err && (!ret || ret == -ENOMEM) && ret != err))
- ret = err;
- else
+ if (likely(!err))
nr_freed += rl->length;
+ else if (!ret || ret == -ENOMEM)
+ ret = err;
}
ntfs_inc_free_clusters(vol, nr_freed);
ntfs_debug("Done.");
@@ -1045,8 +1045,9 @@ err_out:
"Failed to rollback (error %i). Leaving inconsistent metadata! Unmount and run chkdsk.",
(int)delta);
NVolSetErrors(vol);
+ } else {
+ ntfs_dec_free_clusters(vol, delta);
}
- ntfs_dec_free_clusters(vol, delta);
up_write(&vol->lcnbmp_lock);
memalloc_nofs_restore(memalloc_flags);
ntfs_error(vol->sb, "Aborting (error %i).", err);
diff --git a/fs/ntfs/logfile.c b/fs/ntfs/logfile.c
index 024ddee42dc8..1404664dacc0 100644
--- a/fs/ntfs/logfile.c
+++ b/fs/ntfs/logfile.c
@@ -691,7 +691,7 @@ map_vcn:
memset(empty_buf, 0xff, vol->cluster_size);
- ra = kzalloc(sizeof(*ra), GFP_NOFS);
+ ra = kzalloc_obj(*ra, GFP_NOFS);
if (!ra)
goto err;
diff --git a/fs/ntfs/mft.c b/fs/ntfs/mft.c
index 984a0827f9ac..4b7449495375 100644
--- a/fs/ntfs/mft.c
+++ b/fs/ntfs/mft.c
@@ -213,7 +213,8 @@ struct mft_record *map_mft_record(struct ntfs_inode *ni)
return m;
atomic_dec(&ni->count);
- ntfs_error(ni->vol->sb, "Failed with error code %lu.", -PTR_ERR(m));
+ if (PTR_ERR(m) != -EINTR && PTR_ERR(m) != -ERESTARTSYS)
+ ntfs_error(ni->vol->sb, "Failed with error code %lu.", -PTR_ERR(m));
return m;
}
@@ -462,7 +463,7 @@ int ntfs_sync_mft_mirror(struct ntfs_volume *vol, const u64 mft_no,
{
u8 *kmirr;
struct folio *folio;
- unsigned int folio_ofs, lcn_folio_off = 0;
+ unsigned int folio_ofs;
int err = 0;
struct bio *bio;
@@ -492,15 +493,11 @@ int ntfs_sync_mft_mirror(struct ntfs_volume *vol, const u64 mft_no,
memcpy(kmirr, m, vol->mft_record_size);
kunmap_local(kmirr);
- if (vol->cluster_size_bits > PAGE_SHIFT) {
- lcn_folio_off = folio->index << PAGE_SHIFT;
- lcn_folio_off &= vol->cluster_size_mask;
- }
-
bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE, GFP_NOIO);
bio->bi_iter.bi_sector =
- NTFS_B_TO_SECTOR(vol, NTFS_CLU_TO_B(vol, vol->mftmirr_lcn) +
- lcn_folio_off + folio_ofs);
+ ntfs_bytes_to_bio_sector(NTFS_CLU_TO_B(vol, vol->mftmirr_lcn) +
+ ((u64)folio->index << PAGE_SHIFT) +
+ folio_ofs);
if (bio_add_folio(bio, folio, vol->mft_record_size, folio_ofs))
err = submit_bio_wait(bio);
@@ -580,7 +577,7 @@ int write_mft_record_nolock(struct ntfs_inode *ni, struct mft_record *m, int syn
err = pre_write_mst_fixup((struct ntfs_record *)fixup_m, vol->mft_record_size);
if (err) {
ntfs_error(vol->sb, "Failed to apply mst fixups!");
- goto err_out;
+ goto unmap_err_out;
}
folio_size = vol->mft_record_size / ni->mft_lcn_count;
@@ -592,8 +589,8 @@ int write_mft_record_nolock(struct ntfs_inode *ni, struct mft_record *m, int syn
bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE, GFP_NOIO);
bio->bi_iter.bi_sector =
- NTFS_B_TO_SECTOR(vol, NTFS_CLU_TO_B(vol, ni->mft_lcn[i]) +
- clu_off);
+ ntfs_bytes_to_bio_sector(NTFS_CLU_TO_B(vol, ni->mft_lcn[i]) +
+ clu_off);
if (!bio_add_folio(bio, folio, folio_size,
ni->folio_ofs + offset)) {
@@ -645,6 +642,8 @@ done:
return 0;
put_bio_out:
bio_put(bio);
+unmap_err_out:
+ kunmap_local(kaddr);
err_out:
/*
* The caller should mark the base inode as bad so no more I/O
@@ -839,12 +838,126 @@ static bool ntfs_may_write_mft_record(struct ntfs_volume *vol, const u64 mft_no,
static const char *es = " Leaving inconsistent metadata. Unmount and run chkdsk.";
-#define RESERVED_MFT_RECORDS 64
+#define FIRST_NORMAL_MFT_RECORD 24
+#define MFT_RECORD_RESERVE 4
+
+/*
+ * Records 12-15 are marked in use by Windows but normally have no name
+ * and no links. Keep them as the last bootstrap option when a volume
+ * mounted without an in-memory tail reserve needs its first $MFT metadata
+ * extent.
+ */
+static bool mft_reserved_is_free(struct ntfs_volume *vol,
+ struct ntfs_inode *mft_ni, s64 mft_no)
+{
+ struct attr_record *a;
+ struct mft_record *m;
+ struct folio *folio;
+ void *mapped;
+ pgoff_t index = NTFS_MFT_NR_TO_PIDX(vol, mft_no);
+ unsigned int ofs = NTFS_MFT_NR_TO_POFS(vol, mft_no);
+ u32 attrs_offset, bytes_in_use;
+ bool available = false, have_std = false;
+ int i;
+
+ for (i = 0; i < mft_ni->nr_extents; i++) {
+ if (mft_ni->ext.extent_ntfs_inos[i] &&
+ mft_ni->ext.extent_ntfs_inos[i]->mft_no == mft_no)
+ return false;
+ }
+ m = kmalloc(vol->mft_record_size, GFP_NOFS);
+ if (!m)
+ return false;
+
+ folio = read_mapping_folio(vol->mft_ino->i_mapping, index, NULL);
+ if (IS_ERR(folio))
+ goto free_m;
+
+ folio_lock(folio);
+ mapped = kmap_local_folio(folio, 0);
+ memcpy(m, (u8 *)mapped + ofs, vol->mft_record_size);
+ kunmap_local(mapped);
+ folio_unlock(folio);
+ folio_put(folio);
+ if (post_read_mst_fixup((struct ntfs_record *)m, vol->mft_record_size))
+ goto free_m;
+
+ if (!ntfs_is_mft_record(m->magic) ||
+ !(m->flags & MFT_RECORD_IN_USE) || m->base_mft_record ||
+ m->link_count)
+ goto out;
+
+ attrs_offset = le16_to_cpu(m->attrs_offset);
+ bytes_in_use = le32_to_cpu(m->bytes_in_use);
+ if (attrs_offset > bytes_in_use || bytes_in_use > vol->mft_record_size ||
+ bytes_in_use - attrs_offset < sizeof(a->type))
+ goto out;
+
+ for (a = (struct attr_record *)((u8 *)m + attrs_offset);
+ (u8 *)a + sizeof(a->type) <= (u8 *)m + bytes_in_use;) {
+ u32 len;
+
+ if (a->type == AT_END) {
+ if ((u8 *)a + sizeof(a->type) + sizeof(a->length) >
+ (u8 *)m + bytes_in_use)
+ break;
+ /* Also accept a record emptied by an earlier bootstrap. */
+ available = have_std ||
+ (u8 *)a == (u8 *)m + attrs_offset;
+ break;
+ }
+ if (a->type == AT_FILE_NAME)
+ break;
+ len = le32_to_cpu(a->length);
+ if (len < offsetof(struct attr_record, data) ||
+ (u8 *)a + len > (u8 *)m + bytes_in_use)
+ break;
+ if (a->type == AT_STANDARD_INFORMATION) {
+ u32 value_len, value_ofs;
+
+ if (have_std || a->non_resident ||
+ len < offsetof(struct attr_record,
+ data.resident.reserved) + 1)
+ break;
+ value_len = le32_to_cpu(a->data.resident.value_length);
+ value_ofs = le16_to_cpu(a->data.resident.value_offset);
+ if (value_ofs > len || value_len > len - value_ofs)
+ break;
+ have_std = true;
+ }
+ a = (struct attr_record *)((u8 *)a + len);
+ }
+out:
+ kfree(m);
+ return available;
+free_m:
+ kfree(m);
+ return false;
+}
+
+static s64 mft_reserve_end(const u8 *buf, s64 buf_start, s64 buf_end,
+ s64 start, s64 pass_end, s64 initialized_mft_records)
+{
+ s64 end = start + 1;
+ s64 limit = min_t(s64, start + MFT_RECORD_RESERVE, pass_end);
+
+ if (limit > initialized_mft_records)
+ limit = initialized_mft_records;
+ if (limit > buf_end)
+ limit = buf_end;
+ while (end < limit &&
+ !(buf[(end - buf_start) >> 3] &
+ (1 << ((end - buf_start) & 7))))
+ end++;
+ return end;
+}
/*
- * ntfs_mft_bitmap_find_and_alloc_free_rec_nolock - see name
+ * mft_bitmap_alloc_free_rec - find and allocate a free MFT record
* @vol: volume on which to search for a free mft record
* @base_ni: open base inode if allocating an extent mft record or NULL
+ * @max_mft_no: first record which must not be allocated, or -1
+ * @new_reserve_end: if not NULL, end of a free run starting after the result
*
* Search for a free mft record in the mft bitmap attribute on the ntfs volume
* @vol.
@@ -860,10 +973,12 @@ static const char *es = " Leaving inconsistent metadata. Unmount and run chkds
*
* Locking: Caller must hold vol->mftbmp_lock for writing.
*/
-static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vol,
- struct ntfs_inode *base_ni)
+static s64 mft_bitmap_alloc_free_rec(struct ntfs_volume *vol,
+ struct ntfs_inode *base_ni,
+ s64 max_mft_no, s64 *new_reserve_end)
{
s64 pass_end, ll, data_pos, pass_start, ofs, bit;
+ s64 initialized_mft_records;
unsigned long flags;
struct address_space *mftbmp_mapping;
u8 *buf = NULL, *byte;
@@ -880,30 +995,36 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo
read_lock_irqsave(&NTFS_I(vol->mft_ino)->size_lock, flags);
pass_end = NTFS_I(vol->mft_ino)->allocated_size >>
vol->mft_record_size_bits;
+ initialized_mft_records = NTFS_I(vol->mft_ino)->initialized_size >>
+ vol->mft_record_size_bits;
read_unlock_irqrestore(&NTFS_I(vol->mft_ino)->size_lock, flags);
read_lock_irqsave(&NTFS_I(vol->mftbmp_ino)->size_lock, flags);
ll = NTFS_I(vol->mftbmp_ino)->initialized_size << 3;
read_unlock_irqrestore(&NTFS_I(vol->mftbmp_ino)->size_lock, flags);
if (pass_end > ll)
pass_end = ll;
- pass = 1;
- if (!base_ni)
- data_pos = vol->mft_data_pos;
- else
- data_pos = base_ni->mft_no + 1;
- if (data_pos < RESERVED_MFT_RECORDS)
- data_pos = RESERVED_MFT_RECORDS;
- if (data_pos >= pass_end) {
- data_pos = RESERVED_MFT_RECORDS;
+ if (max_mft_no >= 0 && pass_end > max_mft_no)
+ pass_end = max_mft_no;
+ if (base_ni && base_ni->mft_no == FILE_MFT) {
+ data_pos = FILE_first_user;
pass = 2;
- /* This happens on a freshly formatted volume. */
if (data_pos >= pass_end)
return -ENOSPC;
- }
-
- if (base_ni && base_ni->mft_no == FILE_MFT) {
- data_pos = 0;
- pass = 2;
+ } else {
+ pass = 1;
+ if (!base_ni)
+ data_pos = vol->mft_data_pos;
+ else
+ data_pos = base_ni->mft_no + 1;
+ if (data_pos < FIRST_NORMAL_MFT_RECORD)
+ data_pos = FIRST_NORMAL_MFT_RECORD;
+ if (data_pos >= pass_end) {
+ data_pos = FIRST_NORMAL_MFT_RECORD;
+ pass = 2;
+ /* This happens on a freshly formatted volume. */
+ if (data_pos >= pass_end)
+ return -ENOSPC;
+ }
}
pass_start = data_pos;
@@ -938,38 +1059,28 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo
size, data_pos, bit);
for (; bit < size && data_pos + bit < pass_end;
bit &= ~7ull, bit += 8) {
- /*
- * If we're extending $MFT and running out of the first
- * mft record (base record) then give up searching since
- * no guarantee that the found record will be accessible.
- */
- if (base_ni && base_ni->mft_no == FILE_MFT && bit > 400) {
- folio_unlock(folio);
- kunmap_local(buf);
- folio_put(folio);
- return -ENOSPC;
- }
-
byte = buf + (bit >> 3);
if (*byte == 0xff)
continue;
- b = ffz((unsigned long)*byte);
- if (b < 8 && b >= (bit & 7)) {
+ b = bit & 7;
+ for (; b < 8; b++) {
+ if (*byte & (1 << b))
+ continue;
ll = data_pos + (bit & ~7ull) + b;
+ if (ll >= pass_end)
+ break;
+ /* Keep the dynamic tail reserve for $MFT metadata. */
+ if ((!base_ni || base_ni->mft_no != FILE_MFT) &&
+ ll >= vol->mft_record_reserve_pos &&
+ ll < vol->mft_record_reserve_end)
+ continue;
if (unlikely(ll >= (1ll << 32))) {
folio_unlock(folio);
kunmap_local(buf);
folio_put(folio);
return -ENOSPC;
}
- *byte |= 1 << b;
- folio_mark_dirty(folio);
- folio_unlock(folio);
- kunmap_local(buf);
- folio_put(folio);
- ntfs_debug("Done. (Found and allocated mft record 0x%llx.)",
- ll);
- return ll;
+ goto found;
}
}
ntfs_debug("After inner for loop: size 0x%x, data_pos 0x%llx, bit 0x%llx",
@@ -992,7 +1103,8 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo
* part of the zone which we omitted earlier.
*/
pass_end = pass_start;
- data_pos = pass_start = RESERVED_MFT_RECORDS;
+ data_pos = FIRST_NORMAL_MFT_RECORD;
+ pass_start = FIRST_NORMAL_MFT_RECORD;
ntfs_debug("pass %i, pass_start 0x%llx, pass_end 0x%llx.",
pass, pass_start, pass_end);
if (data_pos >= pass_end)
@@ -1002,9 +1114,22 @@ static s64 ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(struct ntfs_volume *vo
/* No free mft records in currently initialized mft bitmap. */
ntfs_debug("Done. (No free mft records left in currently initialized mft bitmap.)");
return -ENOSPC;
+found:
+ if (new_reserve_end)
+ *new_reserve_end = mft_reserve_end(buf, data_pos,
+ data_pos + size, ll, pass_end,
+ initialized_mft_records);
+ *byte |= 1 << b;
+ folio_mark_dirty(folio);
+ folio_unlock(folio);
+ kunmap_local(buf);
+ folio_put(folio);
+ ntfs_debug("Done. (Found and allocated mft record 0x%llx.)", ll);
+ return ll;
}
-static int ntfs_mft_attr_extend(struct ntfs_inode *ni)
+static int ntfs_mft_attr_extend(struct ntfs_inode *ni,
+ struct ntfs_inode *locked_ni)
{
int ret = 0;
struct ntfs_inode *base_ni;
@@ -1025,7 +1150,7 @@ static int ntfs_mft_attr_extend(struct ntfs_inode *ni)
}
}
- ret = ntfs_attr_update_mapping_pairs(ni, 0);
+ ret = ntfs_attr_update_mapping_pairs_locked(ni, 0, locked_ni);
if (ret)
pr_err("MP update failed\n");
@@ -1213,7 +1338,7 @@ static int ntfs_mft_bitmap_extend_allocation_nolock(struct ntfs_volume *vol)
ret = ntfs_attr_record_resize(ctx->mrec, a, mp_size +
le16_to_cpu(a->data.non_resident.mapping_pairs_offset));
if (unlikely(ret)) {
- ret = ntfs_mft_attr_extend(mftbmp_ni);
+ ret = ntfs_mft_attr_extend(mftbmp_ni, mftbmp_ni);
if (!ret)
goto extended_ok;
if (ret != -EAGAIN)
@@ -1324,7 +1449,9 @@ undo_alloc:
NVolSetErrors(vol);
}
mark_mft_record_dirty(ctx->ntfs_ino);
- } else if (status.mp_extended && ntfs_attr_update_mapping_pairs(mftbmp_ni, 0)) {
+ } else if (status.mp_extended &&
+ ntfs_attr_update_mapping_pairs_locked(mftbmp_ni, 0,
+ mftbmp_ni)) {
ntfs_error(vol->sb, "Failed to restore mapping pairs.%s", es);
NVolSetErrors(vol);
}
@@ -1412,7 +1539,6 @@ static int ntfs_mft_bitmap_extend_initialized_nolock(struct ntfs_volume *vol)
ret = ntfs_attr_set(mftbmp_ni, old_initialized_size, 8, 0);
if (likely(!ret)) {
ntfs_debug("Done. (Wrote eight initialized bytes to mft bitmap.");
- ntfs_inc_free_mft_records(vol, 8 * 8);
return 0;
}
ntfs_error(vol->sb, "Failed to write to mft bitmap.");
@@ -1469,8 +1595,9 @@ err_out:
* @vol: volume on which to extend the mft data attribute
*
* Extend the mft data attribute on the ntfs volume @vol by 16 mft records
- * worth of clusters or if not enough space for this by one mft record worth
- * of clusters.
+ * worth of clusters or if not enough space for this by two mft records worth
+ * of clusters. Keeping at least two new records breaks the recursion between
+ * extending $MFT and allocating a record for a new $MFT attribute extent.
*
* Note: Only changes allocated_size, i.e. does not touch initialized_size or
* data_size.
@@ -1524,10 +1651,8 @@ static int ntfs_mft_data_extend_allocation_nolock(struct ntfs_volume *vol)
}
lcn = rl->lcn + rl->length;
ntfs_debug("Last lcn of mft data attribute is 0x%llx.", lcn);
- /* Minimum allocation is one mft record worth of clusters. */
- min_nr = NTFS_B_TO_CLU(vol, vol->mft_record_size);
- if (!min_nr)
- min_nr = 1;
+ /* Keep room for the allocating record and at least one MFT reserve. */
+ min_nr = DIV_ROUND_UP_ULL((u64)vol->mft_record_size * 2, vol->cluster_size);
/* Want to allocate 16 mft records worth of clusters. */
nr = vol->mft_record_size << 4 >> vol->cluster_size_bits;
if (!nr)
@@ -1651,7 +1776,7 @@ static int ntfs_mft_data_extend_allocation_nolock(struct ntfs_volume *vol)
ret = ntfs_attr_record_resize(ctx->mrec, a, mp_size +
le16_to_cpu(a->data.non_resident.mapping_pairs_offset));
if (unlikely(ret)) {
- ret = ntfs_mft_attr_extend(mft_ni);
+ ret = ntfs_mft_attr_extend(mft_ni, NULL);
if (!ret)
goto extended_ok;
if (ret != -EAGAIN)
@@ -1926,6 +2051,7 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n
* @ni: [OUT] on success, set to the allocated ntfs inode
* @base_ni: [IN] open base inode if allocating an extent mft record or NULL
* @ni_mrec: [OUT] on successful return this is the mapped mft record
+ * @mft_data_vcn: [IN] lowest VCN of a new $MFT/$DATA extent, or -1
*
* Allocate an mft record in $MFT/$DATA of an open ntfs volume @vol.
*
@@ -1953,30 +2079,23 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n
* optimize this we start scanning at the place specified by @base_ni or if
* @base_ni is NULL we start where we last stopped and we perform wrap around
* when we reach the end. Note, we do not try to allocate mft records below
- * number 64 because numbers 0 to 15 are the defined system files anyway and 16
- * to 64 are special in that they are used for storing extension mft records
- * for the $DATA attribute of $MFT. This is required to avoid the possibility
- * of creating a runlist with a circular dependency which once written to disk
- * can never be read in again. Windows will only use records 16 to 24 for
- * normal files if the volume is completely out of space. We never use them
- * which means that when the volume is really out of space we cannot create any
- * more files while Windows can still create up to 8 small files. We can start
- * doing this at some later time, it does not matter much for now.
+ * number 24 because numbers 0 to 15 are the defined system files and records
+ * 16 to 23 are kept for metadata compatibility. Records reserved dynamically
+ * at the initialized MFT tail are skipped by normal allocation and consumed by
+ * $MFT metadata extent allocation.
*
* When scanning the mft bitmap, we only search up to the last allocated mft
- * record. If there are no free records left in the range 64 to number of
+ * record. If there are no free records left in the range 24 to number of
* allocated mft records, then we extend the $MFT/$DATA attribute in order to
* create free mft records. We extend the allocated size of $MFT/$DATA by 16
* records at a time or one cluster, if cluster size is above 16kiB. If there
- * is not sufficient space to do this, we try to extend by a single mft record
- * or one cluster, if cluster size is above the mft record size.
+ * is not sufficient space to do this, we try to extend by two mft records or
+ * one cluster, if a cluster already contains at least two mft records.
*
- * No matter how many mft records we allocate, we initialize only the first
- * allocated mft record, incrementing mft data size and initialized size
- * accordingly, open an struct ntfs_inode for it and return it to the caller, unless
- * there are less than 64 mft records, in which case we allocate and initialize
- * mft records until we reach record 64 which we consider as the first free mft
- * record for use by normal files.
+ * When extending the initialized MFT tail, we also initialize up to four
+ * additional records and reserve them in memory for future $MFT metadata
+ * extents. If there are less than 24 mft records, records are initialized
+ * until record 24, which is the first record used for normal files.
*
* If during any stage we overflow the initialized data in the mft bitmap, we
* extend the initialized size (and data size) by 8 bytes, allocating another
@@ -2012,9 +2131,13 @@ static int ntfs_mft_record_format(const struct ntfs_volume *vol, const s64 mft_n
*/
int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode,
struct ntfs_inode **ni, struct ntfs_inode *base_ni,
- struct mft_record **ni_mrec)
+ struct mft_record **ni_mrec, const s64 mft_data_vcn)
{
s64 ll, bit, old_data_initialized, old_data_size;
+ s64 nr_new_mft_records = 0;
+ s64 max_mft_no = -1, reserve_start = -1, reserve_end = -1;
+ s64 candidate_reserve_end = -1;
+ s64 *reserve_endp;
unsigned long flags;
struct folio *folio;
struct ntfs_inode *mft_ni, *mftbmp_ni;
@@ -2025,7 +2148,9 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode,
unsigned int ofs;
int err;
__le16 seq_no, usn;
- bool record_formatted = false;
+ bool record_formatted = false, from_reserve = false, tail_alloc = false;
+ bool reserve_created = false;
+ bool forced_reserved_record = false;
unsigned int memalloc_flags;
if (base_ni && *ni)
@@ -2034,6 +2159,21 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode,
/* @mode and @base_ni are mutually exclusive. */
if (mode && base_ni)
return -EINVAL;
+ if (mft_data_vcn >= 0 &&
+ (!base_ni || base_ni->mft_no != FILE_MFT))
+ return -EINVAL;
+ if (mft_data_vcn >= 0) {
+ u64 vbo;
+
+ if ((u64)mft_data_vcn > (U64_MAX >> vol->cluster_size_bits))
+ return -EOVERFLOW;
+ vbo = (u64)mft_data_vcn << vol->cluster_size_bits;
+ /*
+ * The whole extent record must be reachable without this
+ * extent, including when an MFT record spans multiple clusters.
+ */
+ max_mft_no = vbo >> vol->mft_record_size_bits;
+ }
if (base_ni)
ntfs_debug("Entering (allocating an extent mft record for base mft record 0x%llx).",
@@ -2048,10 +2188,39 @@ int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode,
mutex_lock(&mft_ni->mrec_lock);
mftbmp_ni = NTFS_I(vol->mftbmp_ino);
search_free_rec:
+ from_reserve = false;
+ reserve_created = false;
+ candidate_reserve_end = -1;
if (!base_ni || base_ni->mft_no != FILE_MFT)
down_write(&vol->mftbmp_lock);
- bit = ntfs_mft_bitmap_find_and_alloc_free_rec_nolock(vol, base_ni);
+ if (base_ni && base_ni->mft_no == FILE_MFT &&
+ vol->mft_record_reserve_pos < vol->mft_record_reserve_end &&
+ (max_mft_no < 0 || vol->mft_record_reserve_pos < max_mft_no)) {
+ bit = vol->mft_record_reserve_pos;
+ err = ntfs_bitmap_set_bit(vol->mftbmp_ino, bit);
+ if (unlikely(err)) {
+ ntfs_error(vol->sb,
+ "Failed to allocate reserved MFT record 0x%llx.",
+ bit);
+ goto err_out;
+ }
+ vol->mft_record_reserve_pos++;
+ from_reserve = true;
+ ntfs_debug("Allocated MFT metadata record 0x%llx from tail reserve.",
+ bit);
+ goto have_alloc_rec;
+ }
+ reserve_endp = vol->mft_record_reserve_pos >=
+ vol->mft_record_reserve_end ? &candidate_reserve_end : NULL;
+ bit = mft_bitmap_alloc_free_rec(vol, base_ni, max_mft_no, reserve_endp);
if (bit >= 0) {
+ if (candidate_reserve_end > bit + 1) {
+ vol->mft_record_reserve_pos = bit + 1;
+ vol->mft_record_reserve_end = candidate_reserve_end;
+ reserve_created = true;
+ ntfs_debug("Reserved free MFT records [0x%llx, 0x%llx) for metadata.",
+ bit + 1, candidate_reserve_end);
+ }
ntfs_debug("Found and allocated free record (#1), bit 0x%llx.",
(long long)bit);
goto have_alloc_rec;
@@ -2066,6 +2235,24 @@ search_free_rec:
}
if (base_ni && base_ni->mft_no == FILE_MFT) {
+ static const u8 bootstrap_records[] = {
+ FILE_reserved15, FILE_reserved12, FILE_reserved13,
+ FILE_reserved14,
+ };
+ int i;
+
+ for (i = 0; i < ARRAY_SIZE(bootstrap_records); i++) {
+ if (max_mft_no >= 0 && bootstrap_records[i] >= max_mft_no)
+ continue;
+ if (!mft_reserved_is_free(vol, mft_ni,
+ bootstrap_records[i]))
+ continue;
+ bit = bootstrap_records[i];
+ forced_reserved_record = true;
+ ntfs_debug("Using reserved MFT record %lld to bootstrap metadata extension.",
+ bit);
+ goto have_alloc_rec;
+ }
memalloc_nofs_restore(memalloc_flags);
return bit;
}
@@ -2085,10 +2272,10 @@ search_free_rec:
old_data_initialized = mftbmp_ni->initialized_size;
read_unlock_irqrestore(&mftbmp_ni->size_lock, flags);
if (old_data_initialized << 3 > ll &&
- old_data_initialized > RESERVED_MFT_RECORDS / 8) {
+ old_data_initialized << 3 > FIRST_NORMAL_MFT_RECORD) {
bit = ll;
- if (bit < RESERVED_MFT_RECORDS)
- bit = RESERVED_MFT_RECORDS;
+ if (bit < FIRST_NORMAL_MFT_RECORD)
+ bit = FIRST_NORMAL_MFT_RECORD;
if (unlikely(bit >= (1ll << 32)))
goto max_err_out;
ntfs_debug("Found free record (#2), bit 0x%llx.",
@@ -2174,6 +2361,11 @@ have_alloc_rec:
read_lock_irqsave(&mft_ni->size_lock, flags);
old_data_initialized = mft_ni->initialized_size;
read_unlock_irqrestore(&mft_ni->size_lock, flags);
+ tail_alloc = (!base_ni || base_ni->mft_no != FILE_MFT) &&
+ bit >= (old_data_initialized >> vol->mft_record_size_bits) &&
+ vol->mft_record_reserve_pos >= vol->mft_record_reserve_end;
+ if (tail_alloc)
+ ll = (bit + 2) << vol->mft_record_size_bits;
if (ll <= old_data_initialized) {
ntfs_debug("Allocated mft record already initialized.");
goto mft_rec_already_initialized;
@@ -2206,6 +2398,29 @@ have_alloc_rec:
mft_ni->initialized_size);
}
read_unlock_irqrestore(&mft_ni->size_lock, flags);
+ if (tail_alloc) {
+ s64 bitmap_records;
+
+ read_lock_irqsave(&mft_ni->size_lock, flags);
+ reserve_end = mft_ni->allocated_size >>
+ vol->mft_record_size_bits;
+ read_unlock_irqrestore(&mft_ni->size_lock, flags);
+ read_lock_irqsave(&mftbmp_ni->size_lock, flags);
+ bitmap_records = mftbmp_ni->initialized_size << 3;
+ read_unlock_irqrestore(&mftbmp_ni->size_lock, flags);
+ if (reserve_end > bitmap_records)
+ reserve_end = bitmap_records;
+ if (reserve_end > bit + 1 + MFT_RECORD_RESERVE)
+ reserve_end = bit + 1 + MFT_RECORD_RESERVE;
+ reserve_start = bit + 1;
+ if (reserve_end > reserve_start) {
+ ll = reserve_end << vol->mft_record_size_bits;
+ } else {
+ reserve_start = -1;
+ reserve_end = -1;
+ ll = (bit + 1) << vol->mft_record_size_bits;
+ }
+ }
} else if (ll > mft_ni->allocated_size) {
err = -ENOSPC;
goto undo_mftbmp_alloc_nolock;
@@ -2274,14 +2489,25 @@ have_alloc_rec:
mark_mft_record_dirty(ctx->ntfs_ino);
ntfs_attr_put_search_ctx(ctx);
unmap_mft_record(mft_ni);
+ if (reserve_start >= 0 && reserve_end > reserve_start) {
+ vol->mft_record_reserve_pos = reserve_start;
+ vol->mft_record_reserve_end = reserve_end;
+ ntfs_debug("Reserved MFT records [0x%llx, 0x%llx) for metadata.",
+ reserve_start, reserve_end);
+ }
read_lock_irqsave(&mft_ni->size_lock, flags);
ntfs_debug("Status of mft data after mft record initialization: allocated_size 0x%llx, data_size 0x%llx, initialized_size 0x%llx.",
mft_ni->allocated_size, i_size_read(vol->mft_ino),
mft_ni->initialized_size);
WARN_ON(i_size_read(vol->mft_ino) > mft_ni->allocated_size);
WARN_ON(mft_ni->initialized_size > i_size_read(vol->mft_ino));
+ nr_new_mft_records = (i_size_read(vol->mft_ino) - old_data_size) >>
+ vol->mft_record_size_bits;
read_unlock_irqrestore(&mft_ni->size_lock, flags);
mft_rec_already_initialized:
+ /* Account for newly visible MFT records before dropping the lock. */
+ if (nr_new_mft_records > 0)
+ ntfs_inc_free_mft_records(vol, nr_new_mft_records);
/*
* We can finally drop the mft bitmap lock as the mft data attribute
* has been fully updated. The only disparity left is that the
@@ -2313,8 +2539,8 @@ mft_rec_already_initialized:
/* If we just formatted the mft record no need to do it again. */
if (!record_formatted) {
/* Sanity check that the mft record is really not in use. */
- if (ntfs_is_file_record(m->magic) &&
- (m->flags & MFT_RECORD_IN_USE)) {
+ if (!forced_reserved_record && ntfs_is_file_record(m->magic) &&
+ (m->flags & MFT_RECORD_IN_USE)) {
ntfs_warning(vol->sb,
"Mft record 0x%llx was marked free in mft bitmap but is marked used itself. Unmount and run chkdsk.",
bit);
@@ -2387,9 +2613,13 @@ mft_rec_already_initialized:
ntfs_error(vol->sb, "Failed to map allocated extent mft record 0x%llx.",
bit);
err = PTR_ERR(m_tmp);
- /* Set the mft record itself not in use. */
- m->flags &= cpu_to_le16(
- ~le16_to_cpu(MFT_RECORD_IN_USE));
+ if (forced_reserved_record) {
+ m->base_mft_record = 0;
+ m->flags |= MFT_RECORD_IN_USE;
+ } else {
+ /* Set the mft record itself not in use. */
+ m->flags &= cpu_to_le16(~le16_to_cpu(MFT_RECORD_IN_USE));
+ }
/* Make sure the mft record is written out to disk. */
ntfs_mft_mark_dirty(folio);
folio_unlock(folio);
@@ -2461,7 +2691,8 @@ mft_rec_already_initialized:
(*ni)->mft_no = bit;
if (ni_mrec)
*ni_mrec = (*ni)->mrec;
- ntfs_dec_free_mft_records(vol, 1);
+ if (!forced_reserved_record)
+ ntfs_dec_free_mft_records(vol, 1);
return 0;
undo_data_init:
write_lock_irqsave(&mft_ni->size_lock, flags);
@@ -2473,10 +2704,13 @@ undo_mftbmp_alloc:
if (!base_ni || base_ni->mft_no != FILE_MFT)
down_write(&vol->mftbmp_lock);
undo_mftbmp_alloc_nolock:
- if (ntfs_bitmap_clear_bit(vol->mftbmp_ino, bit)) {
+ if (!forced_reserved_record && ntfs_bitmap_clear_bit(vol->mftbmp_ino, bit)) {
ntfs_error(vol->sb, "Failed to clear bit in mft bitmap.%s", es);
NVolSetErrors(vol);
}
+ if ((from_reserve || reserve_created) &&
+ vol->mft_record_reserve_pos == bit + 1)
+ vol->mft_record_reserve_pos = bit;
if (!base_ni || base_ni->mft_no != FILE_MFT)
up_write(&vol->mftbmp_lock);
err_out:
@@ -2512,9 +2746,11 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni)
int err;
u16 seq_no;
__le16 old_seq_no;
+ __le64 old_base_mft_record;
struct mft_record *ni_mrec;
unsigned int memalloc_flags;
struct ntfs_inode *base_ni;
+ bool keep_reserved;
if (!vol || !ni)
return -EINVAL;
@@ -2527,9 +2763,23 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni)
/* Cache the mft reference for later. */
mft_no = ni->mft_no;
-
- /* Mark the mft record as not in use. */
- ni_mrec->flags &= ~MFT_RECORD_IN_USE;
+ if (likely(ni->nr_extents >= 0))
+ base_ni = ni;
+ else
+ base_ni = ni->ext.base_ntfs_ino;
+ keep_reserved = mft_no >= FILE_reserved12 &&
+ mft_no <= FILE_reserved15 &&
+ base_ni->mft_no == FILE_MFT;
+
+ old_base_mft_record = ni_mrec->base_mft_record;
+ if (keep_reserved) {
+ /* Restore the special, unnamed form used by reserved records. */
+ ni_mrec->base_mft_record = 0;
+ ni_mrec->flags |= MFT_RECORD_IN_USE;
+ } else {
+ /* Mark the mft record as not in use. */
+ ni_mrec->flags &= ~MFT_RECORD_IN_USE;
+ }
/* Increment the sequence number, skipping zero, if it is not zero. */
old_seq_no = ni_mrec->sequence_number;
@@ -2558,24 +2808,28 @@ int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni)
if (err)
goto sync_rollback;
- if (likely(ni->nr_extents >= 0))
- base_ni = ni;
- else
- base_ni = ni->ext.base_ntfs_ino;
+ if (keep_reserved) {
+ unmap_mft_record(ni);
+ return 0;
+ }
/* Clear the bit in the $MFT/$BITMAP corresponding to this record. */
memalloc_flags = memalloc_nofs_save();
if (base_ni->mft_no != FILE_MFT)
down_write(&vol->mftbmp_lock);
err = ntfs_bitmap_clear_bit(vol->mftbmp_ino, mft_no);
+ if (!err)
+ ntfs_inc_free_mft_records(vol, 1);
+ if (!err && base_ni->mft_no == FILE_MFT &&
+ mft_no + 1 == vol->mft_record_reserve_pos &&
+ mft_no < vol->mft_record_reserve_end)
+ vol->mft_record_reserve_pos = mft_no;
if (base_ni->mft_no != FILE_MFT)
up_write(&vol->mftbmp_lock);
memalloc_nofs_restore(memalloc_flags);
if (err)
goto bitmap_rollback;
-
unmap_mft_record(ni);
- ntfs_inc_free_mft_records(vol, 1);
return 0;
/* Rollback what we did... */
@@ -2593,6 +2847,7 @@ sync_rollback:
"Eeek! Rollback failed in %s. Leaving inconsistent metadata!\n", __func__);
ni_mrec->flags |= MFT_RECORD_IN_USE;
ni_mrec->sequence_number = old_seq_no;
+ ni_mrec->base_mft_record = old_base_mft_record;
NInoSetDirty(ni);
write_mft_record(ni, ni_mrec, 0);
unmap_mft_record(ni);
@@ -2633,11 +2888,13 @@ static int ntfs_write_mft_block(struct folio *folio, struct writeback_control *w
struct ntfs_inode *ni = NTFS_I(vi);
struct ntfs_volume *vol = ni->vol;
u8 *kaddr;
- struct ntfs_inode **locked_nis __free(kfree) = kmalloc_array(PAGE_SIZE / NTFS_BLOCK_SIZE,
- sizeof(struct ntfs_inode *), GFP_NOFS);
+ struct ntfs_inode **locked_nis __free(kfree) = kmalloc_objs(struct ntfs_inode *,
+ PAGE_SIZE / NTFS_BLOCK_SIZE,
+ GFP_NOFS);
int nr_locked_nis = 0, err = 0, mft_ofs, prev_mft_ofs;
- struct inode **ref_inos __free(kfree) = kmalloc_array(PAGE_SIZE / NTFS_BLOCK_SIZE,
- sizeof(struct inode *), GFP_NOFS);
+ struct inode **ref_inos __free(kfree) = kmalloc_objs(struct inode *,
+ PAGE_SIZE / NTFS_BLOCK_SIZE,
+ GFP_NOFS);
int nr_ref_inos = 0;
struct bio *bio = NULL;
u64 mft_no;
@@ -2740,8 +2997,8 @@ flush_bio:
bio = bio_alloc(vol->sb->s_bdev, 1, REQ_OP_WRITE,
GFP_NOIO);
bio->bi_iter.bi_sector =
- ntfs_bytes_to_sector(vol,
- ntfs_cluster_to_bytes(vol, lcn) + off);
+ ntfs_bytes_to_bio_sector(
+ ntfs_cluster_to_bytes(vol, lcn) + off);
}
if (vol->cluster_size == NTFS_BLOCK_SIZE &&
diff --git a/fs/ntfs/mft.h b/fs/ntfs/mft.h
index 75a51a98d0f6..ed5c1d595c0d 100644
--- a/fs/ntfs/mft.h
+++ b/fs/ntfs/mft.h
@@ -78,7 +78,7 @@ static inline int write_mft_record(struct ntfs_inode *ni, struct mft_record *m,
int ntfs_mft_record_alloc(struct ntfs_volume *vol, const int mode,
struct ntfs_inode **ni, struct ntfs_inode *base_ni,
- struct mft_record **ni_mrec);
+ struct mft_record **ni_mrec, const s64 mft_data_vcn);
int ntfs_mft_record_free(struct ntfs_volume *vol, struct ntfs_inode *ni);
int ntfs_mft_records_write(const struct ntfs_volume *vol, const u64 mref,
const s64 count, struct mft_record *b);
diff --git a/fs/ntfs/namei.c b/fs/ntfs/namei.c
index 7091b2496fac..fdf52fac4329 100644
--- a/fs/ntfs/namei.c
+++ b/fs/ntfs/namei.c
@@ -480,7 +480,7 @@ static struct ntfs_inode *__ntfs_create(struct mnt_idmap *idmap, struct inode *d
mark_inode_dirty(dir);
err = ntfs_mft_record_alloc(dir_ni->vol, mode, &ni, NULL,
- &ni_mrec);
+ &ni_mrec, -1);
if (err) {
iput(vi);
return ERR_PTR(err);
diff --git a/fs/ntfs/ntfs.h b/fs/ntfs/ntfs.h
index df5a75d506f6..45f77848a9cf 100644
--- a/fs/ntfs/ntfs.h
+++ b/fs/ntfs/ntfs.h
@@ -19,6 +19,7 @@
#include <linux/nls.h>
#include <linux/smp.h>
#include <linux/pagemap.h>
+#include <linux/blk_types.h>
#include <linux/uidgid.h>
#include "volume.h"
@@ -71,8 +72,6 @@
#define NTFS_CLU_TO_POFS(vol, clu) (((u64)(clu) << (vol)->cluster_size_bits) & \
~PAGE_MASK)
-#define NTFS_B_TO_SECTOR(vol, b) ((b) >> ((vol)->sb)->s_blocksize_bits)
-
enum {
NTFS_BLOCK_SIZE = 512,
NTFS_BLOCK_SIZE_BITS = 9,
@@ -154,11 +153,10 @@ static inline u64 ntfs_cluster_to_poff(const struct ntfs_volume *vol,
return (clu << vol->cluster_size_bits) & ~PAGE_MASK;
}
-/* Convert byte offset to sector (block) number. */
-static inline sector_t ntfs_bytes_to_sector(const struct ntfs_volume *vol,
- u64 bytes)
+/* Convert a byte offset on the volume to a bio sector number. */
+static inline sector_t ntfs_bytes_to_bio_sector(u64 bytes)
{
- return bytes >> vol->sb->s_blocksize_bits;
+ return bytes >> SECTOR_SHIFT;
}
/* Global variables. */
diff --git a/fs/ntfs/reparse.c b/fs/ntfs/reparse.c
index 5e483a2f9060..1a6073e22677 100644
--- a/fs/ntfs/reparse.c
+++ b/fs/ntfs/reparse.c
@@ -405,7 +405,7 @@ unsigned int ntfs_reparse_tag_dt_types(struct ntfs_volume *vol, unsigned long mr
vi = ntfs_iget(vol->sb, mref);
if (IS_ERR(vi))
- return PTR_ERR(vi);
+ return DT_UNKNOWN;
reparse_attr = (struct reparse_point *)ntfs_attr_readall(NTFS_I(vi),
AT_REPARSE_POINT, NULL, 0, &attr_size);
@@ -694,8 +694,9 @@ static int update_reparse_data(struct ntfs_inode *ni, struct ntfs_index_context
goto put_rp_inode;
}
- if (set_reparse_index(ni, xr, ((const struct reparse_point *)value)->reparse_tag) &&
- oldsize > 0) {
+ err = set_reparse_index(ni, xr,
+ ((const struct reparse_point *)value)->reparse_tag);
+ if (err && oldsize > 0) {
/*
* If cannot index, try to remove the reparse
* data and log the error. There will be an
diff --git a/fs/ntfs/runlist.c b/fs/ntfs/runlist.c
index 00373e450ea7..3a61f19bcbee 100644
--- a/fs/ntfs/runlist.c
+++ b/fs/ntfs/runlist.c
@@ -1804,7 +1804,7 @@ merge_src_rle:
new_2nd_cnt = src_cnt;
new_cnt = new_1st_cnt + new_2nd_cnt + new_3rd_cnt;
new_cnt += dst_rl_split.lcn >= LCN_HOLE ? 1 : 0;
- new_rl = kvcalloc(new_cnt, sizeof(*new_rl), GFP_NOFS);
+ new_rl = kvzalloc_objs(*new_rl, new_cnt, GFP_NOFS);
if (!new_rl)
return ERR_PTR(-ENOMEM);
@@ -1888,13 +1888,13 @@ struct runlist_element *ntfs_rl_punch_hole(struct runlist_element *dst_rl, int d
punch_cnt = (int)(e_rl - s_rl) + 1;
- *punch_rl = kvcalloc(punch_cnt + 1, sizeof(struct runlist_element),
- GFP_NOFS);
+ *punch_rl = kvzalloc_objs(struct runlist_element, punch_cnt + 1,
+ GFP_NOFS);
if (!*punch_rl)
return ERR_PTR(-ENOMEM);
new_cnt = dst_cnt - (int)(e_rl - s_rl + 1) + 3;
- new_rl = kvcalloc(new_cnt, sizeof(struct runlist_element), GFP_NOFS);
+ new_rl = kvzalloc_objs(struct runlist_element, new_cnt, GFP_NOFS);
if (!new_rl) {
kvfree(*punch_rl);
*punch_rl = NULL;
@@ -2038,13 +2038,13 @@ struct runlist_element *ntfs_rl_collapse_range(struct runlist_element *dst_rl, i
one_split_3 = e_rl == s_rl && begin_split && end_split;
punch_cnt = (int)(e_rl - s_rl) + 1;
- *punch_rl = kvcalloc(punch_cnt + 1, sizeof(struct runlist_element),
- GFP_NOFS);
+ *punch_rl = kvzalloc_objs(struct runlist_element, punch_cnt + 1,
+ GFP_NOFS);
if (!*punch_rl)
return ERR_PTR(-ENOMEM);
new_cnt = dst_cnt - (int)(e_rl - s_rl + 1) + 3;
- new_rl = kvcalloc(new_cnt, sizeof(struct runlist_element), GFP_NOFS);
+ new_rl = kvzalloc_objs(struct runlist_element, new_cnt, GFP_NOFS);
if (!new_rl) {
kvfree(*punch_rl);
*punch_rl = NULL;
diff --git a/fs/ntfs/super.c b/fs/ntfs/super.c
index 30481e5d5dd4..4066bacabe37 100644
--- a/fs/ntfs/super.c
+++ b/fs/ntfs/super.c
@@ -557,8 +557,8 @@ static bool is_boot_sector_ntfs(const struct super_block *sb,
* Check sectors per cluster value is valid and the cluster size
* is not above the maximum (2MB).
*/
- if (b->bpb.sectors_per_cluster > 0x80 &&
- b->bpb.sectors_per_cluster < 0xf4)
+ if (b->bpb.sectors_per_cluster < 0xf4 &&
+ !is_power_of_2(b->bpb.sectors_per_cluster))
goto not_ntfs;
/* Check reserved/unused fields are really zero. */
@@ -695,7 +695,7 @@ static bool parse_ntfs_boot_sector(struct ntfs_volume *vol,
* = -log2(mft_record_size) bytes. mft_record_size normaly is
* 1024 bytes, which is encoded as 0xF6 (-10 in decimal).
*/
- vol->mft_record_size = 1 << -clusters_per_mft_record;
+ vol->mft_record_size = 1U << -clusters_per_mft_record;
vol->mft_record_size_mask = vol->mft_record_size - 1;
vol->mft_record_size_bits = ffs(vol->mft_record_size) - 1;
ntfs_debug("vol->mft_record_size = %i (0x%x)", vol->mft_record_size,
@@ -732,7 +732,7 @@ static bool parse_ntfs_boot_sector(struct ntfs_volume *vol,
* index_record_size normaly equals 4096 bytes, which is
* encoded as 0xF4 (-12 in decimal).
*/
- vol->index_record_size = 1 << -clusters_per_index_record;
+ vol->index_record_size = 1U << -clusters_per_index_record;
vol->index_record_size_mask = vol->index_record_size - 1;
vol->index_record_size_bits = ffs(vol->index_record_size) - 1;
ntfs_debug("vol->index_record_size = %i (0x%x)",
@@ -1241,9 +1241,9 @@ static bool load_and_init_attrdef(struct ntfs_volume *vol)
goto failed;
}
NInoSetSparseDisabled(NTFS_I(ino));
- /* The size of FILE_AttrDef must be above 0 and fit inside 31 bits. */
+ /* FILE_AttrDef must hold at least one entry and fit inside 31 bits. */
i_size = i_size_read(ino);
- if (i_size <= 0 || i_size > 0x7fffffff)
+ if (i_size < (s64)sizeof(struct attr_def) || i_size > 0x7fffffff)
goto iput_failed;
vol->attrdef = kvzalloc(i_size, GFP_NOFS);
if (!vol->attrdef)
@@ -1862,7 +1862,8 @@ static int ntfs_sync_fs(struct super_block *sb, int wait)
return 0;
/* If there are some dirty buffers in the bdev inode */
- if (ntfs_clear_volume_flags(vol, VOLUME_IS_DIRTY)) {
+ if (!NVolErrors(vol) &&
+ ntfs_clear_volume_flags(vol, VOLUME_IS_DIRTY)) {
ntfs_warning(sb, "Failed to clear dirty bit in volume information flags. Run chkdsk.");
err = -EIO;
}
@@ -2063,8 +2064,7 @@ static unsigned long __get_nr_free_mft_records(struct ntfs_volume *vol,
/* If errors occurred we may well have gone below zero, fix this. */
if (nr_free < 0)
nr_free = 0;
- else
- atomic64_set(&vol->free_mft_records, nr_free);
+ atomic64_set(&vol->free_mft_records, nr_free);
ntfs_debug("Exiting.");
return nr_free;
@@ -2130,7 +2130,14 @@ static int ntfs_statfs(struct dentry *dentry, struct kstatfs *sfs)
read_unlock_irqrestore(&mft_ni->size_lock, flags);
/* Free inodes in fs (based on current total count). */
- sfs->f_ffree = atomic64_read(&vol->free_mft_records);
+ size = atomic64_read(&vol->free_mft_records);
+ if (unlikely(size < 0 || size > (s64)sfs->f_files))
+ ntfs_warning(vol->sb, "Invalid free MFT record count %lld.", size);
+ if (size < 0)
+ size = 0;
+ else if (size > (s64)sfs->f_files)
+ size = sfs->f_files;
+ sfs->f_ffree = size;
/*
* File system id. This is extremely *nix flavour dependent and even
@@ -2538,7 +2545,7 @@ static int ntfs_init_fs_context(struct fs_context *fc)
struct ntfs_volume *vol;
/* Allocate a new struct ntfs_volume and place it in sb->s_fs_info. */
- vol = kmalloc(sizeof(struct ntfs_volume), GFP_NOFS);
+ vol = kmalloc_obj(struct ntfs_volume, GFP_NOFS);
if (!vol)
return -ENOMEM;
diff --git a/fs/ntfs/volume.h b/fs/ntfs/volume.h
index 65fd3908af26..bc85a9592245 100644
--- a/fs/ntfs/volume.h
+++ b/fs/ntfs/volume.h
@@ -55,6 +55,10 @@
* @attrdef_size: Size of the attribute definition table in bytes.
* @attrdef: Table of attribute definitions. Obtained from FILE_AttrDef.
* @mft_data_pos: Mft record number at which to allocate the next mft record.
+ * @mft_record_reserve_pos: First record in the in-memory MFT metadata reserve
+ * (protected by mftbmp_lock).
+ * @mft_record_reserve_end: First record beyond the MFT metadata reserve
+ * (protected by mftbmp_lock).
* @mft_zone_start: First cluster of the mft zone.
* @mft_zone_end: First cluster beyond the mft zone.
* @mft_zone_pos: Current position in the mft zone.
@@ -119,6 +123,8 @@ struct ntfs_volume {
s32 attrdef_size;
struct attr_def *attrdef;
s64 mft_data_pos;
+ s64 mft_record_reserve_pos;
+ s64 mft_record_reserve_end;
s64 mft_zone_start;
s64 mft_zone_end;
s64 mft_zone_pos;
@@ -252,17 +258,11 @@ static inline void ntfs_dec_free_clusters(struct ntfs_volume *vol, s64 nr)
static inline void ntfs_inc_free_mft_records(struct ntfs_volume *vol, s64 nr)
{
- if (!NVolFreeClusterKnown(vol))
- return;
-
atomic64_add(nr, &vol->free_mft_records);
}
static inline void ntfs_dec_free_mft_records(struct ntfs_volume *vol, s64 nr)
{
- if (!NVolFreeClusterKnown(vol))
- return;
-
atomic64_sub(nr, &vol->free_mft_records);
}
diff --git a/fs/ntfs/wof.c b/fs/ntfs/wof.c
index 8f84c2212eee..9847259e5b1a 100644
--- a/fs/ntfs/wof.c
+++ b/fs/ntfs/wof.c
@@ -39,8 +39,6 @@ struct ntfs_wof_workspace {
struct mutex *lock;
const struct ntfs_codec_ops *codec;
u32 comp_unit;
- void *input;
- size_t input_size;
void *output;
void *scratch;
};
@@ -97,30 +95,36 @@ static struct ntfs_wof_workspace *ntfs_wof_workspace(u8 block_size_bits)
}
}
+/*
+ * Size of the buffer a chunk is read into. A chunk is read straight off the
+ * device, so the buffer has to hold @comp_unit bytes plus the leading partial
+ * sector.
+ */
+static size_t ntfs_wof_input_size(const struct ntfs_wof_workspace *ws)
+{
+ return round_up((size_t)ws->comp_unit + 511, 512);
+}
+
static int ntfs_wof_workspace_prepare(struct ntfs_wof_workspace *ws)
{
- void *input, *output, *scratch;
+ void *output, *scratch;
size_t scratch_size;
- if (ws->input)
+ if (ws->output)
return 0;
- ws->input_size = round_up((size_t)ws->comp_unit + 511, 512);
scratch_size = ws->codec->scratch_size(ws->comp_unit);
if (!scratch_size)
return -EINVAL;
- input = kvmalloc(ws->input_size, GFP_NOFS);
output = kvmalloc(ws->comp_unit, GFP_NOFS);
scratch = kvzalloc(scratch_size, GFP_NOFS);
- if (!input || !output || !scratch) {
- kvfree(input);
+ if (!output || !scratch) {
kvfree(output);
kvfree(scratch);
return -ENOMEM;
}
- ws->input = input;
ws->output = output;
ws->scratch = scratch;
return 0;
@@ -134,10 +138,8 @@ void ntfs_wof_free_workspaces(void)
struct ntfs_wof_workspace *ws = ntfs_wof_workspaces[i];
mutex_lock(ws->lock);
- kvfree(ws->input);
kvfree(ws->output);
kvfree(ws->scratch);
- ws->input = NULL;
ws->output = NULL;
ws->scratch = NULL;
mutex_unlock(ws->lock);
@@ -602,6 +604,51 @@ static int ntfs_wof_try_direct(struct ntfs_wof_workspace *ws,
chunk_end, src, src_len, dst_len);
}
+/*
+ * Decompress one chunk into @folio. Only this step needs the workspace, so it
+ * is the only step that takes the workspace lock.
+ */
+static int ntfs_wof_decompress_chunk(struct ntfs_wof_workspace *ws,
+ struct ntfs_volume *vol,
+ struct address_space *mapping,
+ struct folio *folio, loff_t folio_start,
+ loff_t folio_end, u64 chunk_file_offset,
+ char *chunk_mem, u32 chunk_size,
+ u32 decomp_size)
+{
+ loff_t chunk_end = chunk_file_offset + decomp_size;
+ loff_t copy_start, copy_end;
+ int err;
+
+ mutex_lock(ws->lock);
+ err = ntfs_wof_workspace_prepare(ws);
+ if (err)
+ goto out_unlock;
+
+ err = ntfs_wof_try_direct(ws, mapping, folio, chunk_file_offset,
+ chunk_end, chunk_mem, chunk_size,
+ decomp_size);
+ if (err != -EAGAIN)
+ goto out_unlock;
+
+ err = ntfs_wof_decode(ws, chunk_mem, chunk_size, ws->output,
+ decomp_size);
+ if (err) {
+ ntfs_error(vol->sb, "Decompression failed: %d", err);
+ err = -EINVAL;
+ goto out_unlock;
+ }
+
+ copy_start = max_t(loff_t, folio_start, chunk_file_offset);
+ copy_end = min_t(loff_t, folio_end, chunk_file_offset + decomp_size);
+ memcpy_to_folio(folio, copy_start - folio_start,
+ ws->output + copy_start - chunk_file_offset,
+ copy_end - copy_start);
+out_unlock:
+ mutex_unlock(ws->lock);
+ return err;
+}
+
int ntfs_read_wof_compressed_block(struct folio *folio)
{
struct address_space *mapping = folio->mapping;
@@ -613,6 +660,8 @@ int ntfs_read_wof_compressed_block(struct folio *folio)
loff_t folio_start = folio_pos(folio);
loff_t folio_end = folio_next_pos(folio);
char *chunk_mem;
+ void *input;
+ size_t input_size;
u32 decomp_size;
u64 chunk_count, chunk_idx, last_chunk, chunk_offset;
int err = 0;
@@ -652,10 +701,12 @@ int ntfs_read_wof_compressed_block(struct folio *folio)
goto out_iput;
}
- mutex_lock(ws->lock);
- err = ntfs_wof_workspace_prepare(ws);
- if (err)
- goto out_unlock_ws;
+ input_size = ntfs_wof_input_size(ws);
+ input = kvmalloc(input_size, GFP_NOFS);
+ if (!input) {
+ err = -ENOMEM;
+ goto out_iput;
+ }
chunk_idx = div_u64(folio_start, ws->comp_unit);
last_chunk =
@@ -663,55 +714,35 @@ int ntfs_read_wof_compressed_block(struct folio *folio)
chunk_count = DIV_ROUND_UP_ULL(i_size, ws->comp_unit);
for (; chunk_idx <= last_chunk; chunk_idx++) {
u32 chunk_size;
- u64 chunk_file_offset;
- loff_t chunk_end, copy_start, copy_end;
decomp_size = chunk_idx + 1 == chunk_count ?
i_size - chunk_idx * ws->comp_unit :
ws->comp_unit;
err = parse_wof_chunk_table(ni, wof_ni, chunk_idx, chunk_count,
decomp_size, &chunk_offset,
- &chunk_size, ws->input,
- ws->input_size);
+ &chunk_size, input, input_size);
if (err)
- goto out_unlock_ws;
+ goto out_free_input;
err = ntfs_read_wof_chunk(vol, wof_ni, chunk_offset, chunk_size,
- ws->input, ws->input_size,
- &chunk_mem);
+ input, input_size, &chunk_mem);
if (err)
- goto out_unlock_ws;
-
- chunk_file_offset = chunk_idx * ws->comp_unit;
- chunk_end = chunk_file_offset + decomp_size;
- err = ntfs_wof_try_direct(ws, mapping, folio, chunk_file_offset,
- chunk_end, chunk_mem, chunk_size,
- decomp_size);
- if (!err)
- continue;
- if (err != -EAGAIN)
- goto out_unlock_ws;
+ goto out_free_input;
- err = ntfs_wof_decode(ws, chunk_mem, chunk_size, ws->output,
- decomp_size);
- if (err) {
- ntfs_error(vol->sb, "Decompression failed: %d", err);
- err = -EINVAL;
- goto out_unlock_ws;
- }
- copy_start = max_t(loff_t, folio_start, chunk_file_offset);
- copy_end = min_t(loff_t, folio_end,
- chunk_file_offset + decomp_size);
- memcpy_to_folio(folio, copy_start - folio_start,
- ws->output + copy_start - chunk_file_offset,
- copy_end - copy_start);
+ err = ntfs_wof_decompress_chunk(ws, vol, mapping, folio,
+ folio_start, folio_end,
+ chunk_idx * ws->comp_unit,
+ chunk_mem, chunk_size,
+ decomp_size);
+ if (err)
+ goto out_free_input;
}
if (folio_end > i_size)
folio_zero_segment(folio, i_size - folio_start,
folio_size(folio));
-out_unlock_ws:
- mutex_unlock(ws->lock);
+out_free_input:
+ kvfree(input);
out_iput:
iput(wof_inode);
out:
diff --git a/fs/overlayfs/readdir.c b/fs/overlayfs/readdir.c
index e7fe29cb6028..7d6f7f6022eb 100644
--- a/fs/overlayfs/readdir.c
+++ b/fs/overlayfs/readdir.c
@@ -1044,7 +1044,7 @@ static int ovl_dir_open(struct inode *inode, struct file *file)
struct ovl_dir_file *od;
enum ovl_path_type type;
- od = kzalloc(sizeof(struct ovl_dir_file), GFP_KERNEL);
+ od = kzalloc_obj(struct ovl_dir_file);
if (!od)
return -ENOMEM;
diff --git a/fs/overlayfs/super.c b/fs/overlayfs/super.c
index e487597337e8..bd0a3f9039d2 100644
--- a/fs/overlayfs/super.c
+++ b/fs/overlayfs/super.c
@@ -1543,7 +1543,7 @@ int ovl_fill_super(struct super_block *sb, struct fs_context *fc)
struct ovl_fs *ofs = sb->s_fs_info;
int err;
- err = -EIO;
+ err = -EINVAL;
/* The fscontext fd may have been passed to another user namespace. */
if (fc->user_ns != current_user_ns())
goto out_err;
diff --git a/fs/quota/dquot.c b/fs/quota/dquot.c
index 204afc5e984b..1c78c695d0dd 100644
--- a/fs/quota/dquot.c
+++ b/fs/quota/dquot.c
@@ -1240,7 +1240,7 @@ static int ignore_hardlimit(struct dquot *dquot)
{
struct mem_dqinfo *info = &sb_dqopt(dquot->dq_sb)->info[dquot->dq_id.type];
- return capable(CAP_SYS_RESOURCE) &&
+ return capable_noaudit(CAP_SYS_RESOURCE) &&
(info->dqi_format->qf_fmt_id != QFMT_VFS_OLD ||
!(info->dqi_flags & DQF_ROOT_SQUASH));
}
diff --git a/fs/smb/client/cifs_swn.c b/fs/smb/client/cifs_swn.c
index fe10719e627e..c49ecddf4a33 100644
--- a/fs/smb/client/cifs_swn.c
+++ b/fs/smb/client/cifs_swn.c
@@ -443,7 +443,7 @@ static struct cifs_swn_reg *cifs_get_swn_reg(struct cifs_tcon *tcon)
goto unlock;
}
- reg = kmalloc_obj(struct cifs_swn_reg, GFP_KERNEL);
+ reg = kmalloc_obj(struct cifs_swn_reg);
if (reg == NULL) {
ret = -ENOMEM;
goto fail_unlock;
diff --git a/fs/smb/client/cifsacl.c b/fs/smb/client/cifsacl.c
index 12005f46307d..c5e47a835f99 100644
--- a/fs/smb/client/cifsacl.c
+++ b/fs/smb/client/cifsacl.c
@@ -100,8 +100,23 @@ cifs_idmap_key_destroy(struct key *key)
kfree(key->payload.data[0]);
}
+static int
+cifs_idmap_key_vet_description(const char *description)
+{
+ /*
+ * cifs.idmap descriptions are authority-bearing inputs to the
+ * cifs.idmap upcall helper. Only allow the kernel to create this
+ * type of key using the private root_cred installed in
+ * init_cifs_idmap; reject userspace request_key(2)/add_key(2).
+ */
+ if (current_cred() != root_cred)
+ return -EPERM;
+ return 0;
+}
+
static struct key_type cifs_idmap_key_type = {
.name = "cifs.idmap",
+ .vet_description = cifs_idmap_key_vet_description,
.instantiate = cifs_idmap_key_instantiate,
.destroy = cifs_idmap_key_destroy,
.describe = user_describe,
@@ -1081,13 +1096,13 @@ unsigned int setup_special_user_owner_ACE(struct smb_ace *pntace)
static void populate_new_aces(char *nacl_base,
struct smb_sid *pownersid,
struct smb_sid *pgrpsid,
- __u64 *pnmode, u16 *pnum_aces, u16 *pnsize,
+ __u64 *pnmode, u16 *pnum_aces, u32 *pnsize,
bool modefromsid,
bool posix)
{
__u64 nmode;
u16 num_aces = 0;
- u16 nsize = 0;
+ u32 nsize = 0;
__u64 user_mode;
__u64 group_mode;
__u64 other_mode;
@@ -1186,17 +1201,17 @@ set_size:
*pnsize = nsize;
}
-static __u16 replace_sids_and_copy_aces(struct smb_acl *pdacl, struct smb_acl *pndacl,
- struct smb_sid *pownersid, struct smb_sid *pgrpsid,
- struct smb_sid *pnownersid, struct smb_sid *pngrpsid,
- int *aclflag)
+static int replace_sids_and_copy_aces(struct smb_acl *pdacl, struct smb_acl *pndacl,
+ struct smb_sid *pownersid, struct smb_sid *pgrpsid,
+ struct smb_sid *pnownersid, struct smb_sid *pngrpsid,
+ int *aclflag, u16 *pnsize)
{
int i;
u16 size = 0;
struct smb_ace *pntace = NULL;
char *acl_base = NULL;
u16 src_num_aces = 0;
- u16 nsize = 0;
+ u32 nsize = 0;
struct smb_ace *pnntace = NULL;
char *nacl_base = NULL;
u16 ace_size = 0;
@@ -1225,9 +1240,12 @@ static __u16 replace_sids_and_copy_aces(struct smb_acl *pdacl, struct smb_acl *p
size += le16_to_cpu(pntace->size);
nsize += ace_size;
+ if (nsize > U16_MAX)
+ return -EOVERFLOW;
}
- return nsize;
+ *pnsize = nsize;
+ return 0;
}
static int set_chmod_dacl(struct smb_acl *pdacl, struct smb_acl *pndacl,
@@ -1239,7 +1257,7 @@ static int set_chmod_dacl(struct smb_acl *pdacl, struct smb_acl *pndacl,
struct smb_ace *pntace = NULL;
char *acl_base = NULL;
u16 src_num_aces = 0;
- u16 nsize = 0;
+ u32 nsize = 0;
struct smb_ace *pnntace = NULL;
char *nacl_base = NULL;
u16 num_aces = 0;
@@ -1290,6 +1308,8 @@ static int set_chmod_dacl(struct smb_acl *pdacl, struct smb_acl *pndacl,
nsize += cifs_copy_ace(pnntace, pntace, NULL);
num_aces++;
+ if (nsize > U16_MAX)
+ return -EOVERFLOW;
next_ace:
size += le16_to_cpu(pntace->size);
@@ -1306,6 +1326,10 @@ next_ace:
}
finalize_dacl:
+ /* The DACL size field is 16-bit on the wire, see MS-DTYP 2.4.5 */
+ if (nsize > U16_MAX)
+ return -EOVERFLOW;
+
pndacl->num_aces = cpu_to_le16(num_aces);
pndacl->size = cpu_to_le16(nsize);
@@ -1331,6 +1355,7 @@ static int parse_sec_desc(struct cifs_sb_info *cifs_sb,
{
int rc = 0;
struct smb_sid *owner_sid_ptr, *group_sid_ptr;
+ unsigned int sbflags = cifs_sb_flags(cifs_sb);
struct smb_acl *dacl_ptr; /* no need for SACL ptr */
char *end_of_acl;
__u32 dacloffset, osidoffset, gsidoffset;
@@ -1349,17 +1374,21 @@ static int parse_sec_desc(struct cifs_sb_info *cifs_sb,
cifs_dbg(NOISY, "revision %d type 0x%x ooffset 0x%x goffset 0x%x sacloffset 0x%x dacloffset 0x%x\n",
pntsd->revision, pntsd->type, osidoffset, gsidoffset,
le32_to_cpu(pntsd->sacloffset), dacloffset);
-/* cifs_dump_mem("owner_sid: ", owner_sid_ptr, 64); */
+ fattr->cf_uid = cifs_sb->ctx->linux_uid;
+ fattr->cf_gid = cifs_sb->ctx->linux_gid;
+
rc = sid_from_sd(pntsd, acl_len, osidoffset, &owner_sid_ptr);
if (rc) {
cifs_dbg(FYI, "%s: Error %d parsing Owner SID\n", __func__, rc);
return rc;
}
- rc = sid_to_id(cifs_sb, owner_sid_ptr, fattr, SIDOWNER);
- if (rc) {
- cifs_dbg(FYI, "%s: Error %d mapping Owner SID to uid\n",
- __func__, rc);
- return rc;
+ if (!(sbflags & CIFS_MOUNT_OVERR_UID)) {
+ rc = sid_to_id(cifs_sb, owner_sid_ptr, fattr, SIDOWNER);
+ if (rc) {
+ cifs_dbg(FYI, "%s: Error %d mapping Owner SID to uid\n",
+ __func__, rc);
+ return rc;
+ }
}
rc = sid_from_sd(pntsd, acl_len, gsidoffset, &group_sid_ptr);
@@ -1368,11 +1397,13 @@ static int parse_sec_desc(struct cifs_sb_info *cifs_sb,
__func__, rc);
return rc;
}
- rc = sid_to_id(cifs_sb, group_sid_ptr, fattr, SIDGROUP);
- if (rc) {
- cifs_dbg(FYI, "%s: Error %d mapping Group SID to gid\n",
- __func__, rc);
- return rc;
+ if (!(sbflags & CIFS_MOUNT_OVERR_GID)) {
+ rc = sid_to_id(cifs_sb, group_sid_ptr, fattr, SIDGROUP);
+ if (rc) {
+ cifs_dbg(FYI, "%s: Error %d mapping Group SID to gid\n",
+ __func__, rc);
+ return rc;
+ }
}
if (dacloffset) {
@@ -1451,6 +1482,8 @@ static int build_sec_desc(struct smb_ntsd *pntsd, struct smb_ntsd *pnntsd,
rc = set_chmod_dacl(dacl_ptr, ndacl_ptr, owner_sid_ptr, group_sid_ptr,
pnmode, mode_from_sid, posix);
+ if (rc)
+ return rc;
sidsoffset = ndacloffset + le16_to_cpu(ndacl_ptr->size);
/* copy the non-dacl portion of secdesc */
@@ -1526,10 +1559,12 @@ static int build_sec_desc(struct smb_ntsd *pntsd, struct smb_ntsd *pnntsd,
if (dacloffset) {
/* Replace ACEs for old owner with new one */
- size = replace_sids_and_copy_aces(dacl_ptr, ndacl_ptr,
- owner_sid_ptr, group_sid_ptr,
- nowner_sid_ptr, ngroup_sid_ptr,
- aclflag);
+ rc = replace_sids_and_copy_aces(dacl_ptr, ndacl_ptr,
+ owner_sid_ptr, group_sid_ptr,
+ nowner_sid_ptr, ngroup_sid_ptr,
+ aclflag, &size);
+ if (rc)
+ goto chown_chgrp_exit;
ndacl_ptr->size = cpu_to_le16(size);
}
@@ -1815,11 +1850,13 @@ id_mode_to_cifs_acl(struct inode *inode, const char *path, __u64 *pnmode,
cifs_put_tlink(tlink);
return rc;
}
- if (mode_from_sid)
- nsecdesclen +=
- le16_to_cpu(dacl_ptr->num_aces) * sizeof(struct smb_ace);
- else /* cifsacl */
- nsecdesclen += le16_to_cpu(dacl_ptr->size);
+ /*
+ * Worst case: every ACE is rewritten with a new SID of
+ * SID_MAX_SUB_AUTHORITIES sub-auths -> sizeof(smb_ace) each,
+ * plus the smb_acl header replace_sids_and_copy_aces() emits.
+ */
+ nsecdesclen += sizeof(struct smb_acl) +
+ le16_to_cpu(dacl_ptr->num_aces) * sizeof(struct smb_ace);
}
}
diff --git a/fs/smb/client/cifssmb.c b/fs/smb/client/cifssmb.c
index f5aad5f61dce..6dddbd84b93b 100644
--- a/fs/smb/client/cifssmb.c
+++ b/fs/smb/client/cifssmb.c
@@ -1719,8 +1719,17 @@ CIFSSMBRead(const unsigned int xid, struct cifs_io_parms *io_parms,
pSMBr = (READ_RSP *)rsp_iov.iov_base;
if (rc) {
cifs_dbg(VFS, "Send error in read = %d\n", rc);
+ } else if (rsp_iov.iov_len < tcon->ses->server->vals->read_rsp_size) {
+ /* check that the received response can hold a whole READ_RSP */
+ cifs_dbg(FYI, "%s: server returned short header. got=%zu expected=%zu\n",
+ __func__, rsp_iov.iov_len,
+ tcon->ses->server->vals->read_rsp_size);
+ rc = smb_EIO2(smb_eio_trace_read_rsp_short,
+ rsp_iov.iov_len, tcon->ses->server->vals->read_rsp_size);
+ *nbytes = 0;
} else {
- int data_length = le16_to_cpu(pSMBr->DataLengthHigh);
+ unsigned int data_length = le16_to_cpu(pSMBr->DataLengthHigh);
+ __u16 data_offset = le16_to_cpu(pSMBr->DataOffset);
data_length = data_length << 16;
data_length += le16_to_cpu(pSMBr->DataLength);
*nbytes = data_length;
@@ -1728,14 +1737,21 @@ CIFSSMBRead(const unsigned int xid, struct cifs_io_parms *io_parms,
/*check that DataLength would not go beyond end of SMB */
if ((data_length > CIFSMaxBufSize)
|| (data_length > count)) {
- cifs_dbg(FYI, "bad length %d for count %d\n",
- data_length, count);
+ cifs_dbg(FYI, "%s: bad length %u for count %u\n",
+ __func__, data_length, count);
rc = smb_EIO2(smb_eio_trace_read_overlarge,
data_length, count);
*nbytes = 0;
+ } else if (data_offset < sizeof(*pSMBr) ||
+ (size_t)data_offset + data_length > rsp_iov.iov_len) {
+ /* check that the data lies within the received response */
+ cifs_dbg(FYI, "%s: bad data offset %u length %u for response of %zu\n",
+ __func__, data_offset, data_length, rsp_iov.iov_len);
+ rc = smb_EIO2(smb_eio_trace_read_bad_offset,
+ data_offset, data_length);
+ *nbytes = 0;
} else {
- pReadData = (char *) (&pSMBr->hdr.Protocol) +
- le16_to_cpu(pSMBr->DataOffset);
+ pReadData = (char *) (&pSMBr->hdr.Protocol) + data_offset;
/* if (rc = copy_to_user(buf, pReadData, data_length)) {
cifs_dbg(VFS, "Faulting on read rc = %d\n",rc);
rc = -EFAULT;
@@ -3064,7 +3080,7 @@ int cifs_query_reparse_point(const unsigned int xid,
end = 2 + get_bcc(&io_rsp->hdr) + (__u8 *)&io_rsp->ByteCount;
start = (__u8 *)&io_rsp->hdr.Protocol + data_offset;
- if (start >= end) {
+ if (start >= end || (size_t)(end - start) < sizeof(*buf)) {
rc = smb_EIO2(smb_eio_trace_qreparse_data_area,
(unsigned long)start - (unsigned long)io_rsp,
(unsigned long)end - (unsigned long)io_rsp);
@@ -3555,6 +3571,7 @@ int cifs_do_set_acl(const unsigned int xid, struct cifs_tcon *tcon,
int rc = 0;
int bytes_returned = 0;
__u16 params, byte_count, data_count, param_offset, offset;
+ size_t cifs_acl_size, bytes_available;
cifs_dbg(FYI, "In SetPosixACL (Unix) for path %s\n", fileName);
setAclRetry:
@@ -3574,8 +3591,7 @@ setAclRetry:
}
params = 6 + name_len;
pSMB->MaxParameterCount = cpu_to_le16(2);
- /* BB find max SMB size from sess */
- pSMB->MaxDataCount = cpu_to_le16(1000);
+ pSMB->MaxDataCount = cpu_to_le16(min_t(unsigned int, CIFSMaxBufSize, USHRT_MAX));
pSMB->MaxSetupCount = 0;
pSMB->Reserved = 0;
pSMB->Flags = 0;
@@ -3587,6 +3603,15 @@ setAclRetry:
parm_data = ((char *)pSMB) + offset;
pSMB->ParameterOffset = cpu_to_le16(param_offset);
+ /* make sure we can fit the larger cifs_posix_aces in the buffer */
+ cifs_acl_size = sizeof(struct cifs_posix_acl) +
+ (acl->a_count * sizeof(struct cifs_posix_ace));
+ bytes_available = (CIFSMaxBufSize + MAX_HEADER_SIZE(tcon->ses->server)) - offset;
+ if (cifs_acl_size > bytes_available || cifs_acl_size > USHRT_MAX) {
+ rc = -E2BIG;
+ goto setACLerrorExit;
+ }
+
/* convert to on the wire format for POSIX ACL */
data_count = posix_acl_to_cifs(parm_data, acl, acl_type);
@@ -6325,8 +6350,10 @@ CIFSSMBSetEA(const unsigned int xid, struct cifs_tcon *tcon,
int name_len;
int rc = 0;
int bytes_returned = 0;
- __u16 params, param_offset, byte_count, offset, count;
+ __u16 params, param_offset;
+ unsigned int byte_count, offset, count;
int remap = cifs_remap(cifs_sb);
+ unsigned int total_len;
cifs_dbg(FYI, "In SetEA\n");
SetEARetry:
@@ -6378,6 +6405,13 @@ SetEARetry:
pSMB->Reserved3 = 0;
pSMB->SubCommand = cpu_to_le16(TRANS2_SET_PATH_INFORMATION);
byte_count = 3 /* pad */ + params + count;
+ if (check_add_overflow(in_len, byte_count, &total_len) ||
+ byte_count > U16_MAX ||
+ total_len > CIFSMaxBufSize + MAX_CIFS_HDR_SIZE) {
+ cifs_dbg(VFS, "EA request too large: %u bytes\n", total_len);
+ cifs_buf_release(pSMB);
+ return -E2BIG;
+ }
pSMB->DataCount = cpu_to_le16(count);
parm_data->list_len = cpu_to_le32(count);
parm_data->list.EA_flags = 0;
diff --git a/fs/smb/client/connect.c b/fs/smb/client/connect.c
index bcd7f1ae99ba..28e1ddeb6182 100644
--- a/fs/smb/client/connect.c
+++ b/fs/smb/client/connect.c
@@ -174,6 +174,8 @@ cifs_signal_cifsd_for_reconnect(struct TCP_Server_Info *server,
nserver = ses->chans[i].server;
if (!nserver)
continue;
+ if (!list_empty(&nserver->rlist))
+ continue;
nserver->srv_count++;
list_add(&nserver->rlist, &reco);
}
@@ -182,11 +184,15 @@ cifs_signal_cifsd_for_reconnect(struct TCP_Server_Info *server,
}
}
+ spin_lock(&cifs_tcp_ses_lock);
list_for_each_entry_safe(server, nserver, &reco, rlist) {
list_del_init(&server->rlist);
set_need_reco(server);
+ spin_unlock(&cifs_tcp_ses_lock);
cifs_put_tcp_session(server, 0);
+ spin_lock(&cifs_tcp_ses_lock);
}
+ spin_unlock(&cifs_tcp_ses_lock);
}
/*
@@ -1067,6 +1073,7 @@ clean_demultiplex_info(struct TCP_Server_Info *server)
spin_unlock(&server->srv_lock);
cancel_delayed_work_sync(&server->echo);
+ cancel_delayed_work_sync(&server->reconnect);
spin_lock(&server->srv_lock);
server->tcpStatus = CifsExiting;
@@ -1823,6 +1830,7 @@ cifs_get_tcp_session(struct smb3_fs_context *ctx,
spin_lock_init(&tcp_ses->mid_counter_lock);
INIT_LIST_HEAD(&tcp_ses->tcp_ses_list);
INIT_LIST_HEAD(&tcp_ses->smb_ses_list);
+ INIT_LIST_HEAD(&tcp_ses->rlist);
INIT_DELAYED_WORK(&tcp_ses->echo, cifs_echo_request);
INIT_DELAYED_WORK(&tcp_ses->reconnect, smb2_reconnect_server);
mutex_init(&tcp_ses->reconnect_mutex);
@@ -1926,6 +1934,7 @@ out_err:
kfree(tcp_ses->leaf_fullpath);
if (tcp_ses->ssocket)
sock_release(tcp_ses->ssocket);
+ smbd_destroy(tcp_ses);
kfree(tcp_ses);
}
return ERR_PTR(rc);
@@ -4189,14 +4198,25 @@ cifs_setup_session(const unsigned int xid, struct cifs_ses *ses,
return rc;
}
-static int
-cifs_set_vol_auth(struct smb3_fs_context *ctx, struct cifs_ses *ses)
+static int set_fs_context_auth(struct smb3_fs_context *ctx,
+ struct cifs_ses *ses)
{
ctx->sectype = ses->sectype;
- /* krb5 is special, since we don't need username or pw */
- if (ctx->sectype == Kerberos)
+ /*
+ * krb5 is special as we might need to pass username (passwordless) down
+ * to cifs.upcall(8) for keytab.
+ */
+ if (ctx->sectype == Kerberos) {
+ if (ses->user_name && ses->user_name[0]) {
+ ctx->username = kstrndup(ses->user_name,
+ CIFS_MAX_USERNAME_LEN,
+ GFP_KERNEL);
+ if (!ctx->username)
+ return -ENOMEM;
+ }
return 0;
+ }
return cifs_set_cifscreds(ctx, ses);
}
@@ -4236,7 +4256,7 @@ cifs_construct_tcon(struct cifs_sb_info *cifs_sb, kuid_t fsuid)
ctx->dfs_root_ses = master_tcon->ses->dfs_root_ses;
ctx->unicode = master_tcon->ses->unicode;
- rc = cifs_set_vol_auth(ctx, master_tcon->ses);
+ rc = set_fs_context_auth(ctx, master_tcon->ses);
if (rc) {
tcon = ERR_PTR(rc);
goto out;
diff --git a/fs/smb/client/dfs_cache.c b/fs/smb/client/dfs_cache.c
index 86dba25b7a5a..29dfd7595941 100644
--- a/fs/smb/client/dfs_cache.c
+++ b/fs/smb/client/dfs_cache.c
@@ -123,6 +123,7 @@ static inline void free_tgts(struct cache_entry *ce)
kfree(t);
}
+ ce->numtgts = 0;
WRITE_ONCE(ce->tgthint, NULL);
}
@@ -365,7 +366,7 @@ static struct cache_dfs_tgt *alloc_target(const char *name, int path_consumed)
{
struct cache_dfs_tgt *t;
- t = kmalloc_obj(*t, GFP_KERNEL);
+ t = kmalloc_obj(*t);
if (!t)
return ERR_PTR(-ENOMEM);
t->name = kstrdup(name, GFP_KERNEL);
@@ -388,13 +389,6 @@ static int copy_ref_data(const struct dfs_info3_param *refs, int numrefs,
struct cache_dfs_tgt *target;
int i;
- ce->ttl = max_t(int, refs[0].ttl, CACHE_MIN_TTL);
- ce->etime = get_expire_time(ce->ttl);
- ce->srvtype = refs[0].server_type;
- ce->hdr_flags = refs[0].flags;
- ce->ref_flags = refs[0].ref_flag;
- ce->path_consumed = refs[0].path_consumed;
-
for (i = 0; i < numrefs; i++) {
struct cache_dfs_tgt *t;
@@ -409,12 +403,19 @@ static int copy_ref_data(const struct dfs_info3_param *refs, int numrefs,
} else {
list_add_tail(&t->list, &ce->tlist);
}
- ce->numtgts++;
}
target = list_first_entry_or_null(&ce->tlist, struct cache_dfs_tgt,
list);
+
WRITE_ONCE(ce->tgthint, target);
+ ce->ttl = max_t(int, refs[0].ttl, CACHE_MIN_TTL);
+ ce->etime = get_expire_time(ce->ttl);
+ ce->srvtype = refs[0].server_type;
+ ce->hdr_flags = refs[0].flags;
+ ce->ref_flags = refs[0].ref_flag;
+ ce->path_consumed = refs[0].path_consumed;
+ ce->numtgts = numrefs;
return 0;
}
@@ -634,7 +635,6 @@ static int update_cache_entry_locked(struct cache_entry *ce, const struct dfs_in
}
free_tgts(ce);
- ce->numtgts = 0;
rc = copy_ref_data(refs, numrefs, ce, th);
diff --git a/fs/smb/client/file.c b/fs/smb/client/file.c
index bdcd54157e6c..0d428517f454 100644
--- a/fs/smb/client/file.c
+++ b/fs/smb/client/file.c
@@ -999,26 +999,50 @@ static int cifs_do_truncate(const unsigned int xid, struct dentry *dentry)
struct cifs_tcon *tcon;
int rc;
- rc = filemap_write_and_wait(inode->i_mapping);
- if (is_interrupt_error(rc))
+ rc = inode_lock_killable(inode);
+ if (rc)
return -ERESTARTSYS;
+
+ filemap_invalidate_lock(inode->i_mapping);
+
+ rc = filemap_write_and_wait(inode->i_mapping);
+ if (is_interrupt_error(rc)) {
+ rc = -ERESTARTSYS;
+ goto out;
+ }
mapping_set_error(inode->i_mapping, rc);
cfile = find_writable_file(cinode, FIND_FSUID_ONLY);
rc = cifs_file_flush(xid, inode, cfile);
if (!rc) {
if (cfile) {
+ struct netfs_inode *ictx = netfs_inode(inode);
+
tcon = tlink_tcon(cfile->tlink);
server = tcon->ses->server;
+ netfs_wb_begin(ictx, false);
rc = server->ops->set_file_size(xid, tcon,
cfile, 0, false);
- }
- if (!rc) {
- netfs_resize_file(&cinode->netfs, 0, true);
- cifs_setsize(inode, 0);
+ if (!rc) {
+ netfs_resize_file(&cinode->netfs, 0, true);
+ cifs_setsize(inode, 0);
+ cifs_invalidate_cache(inode, 0);
+ }
+ netfs_wb_end(ictx);
+ } else {
+ /*
+ * No cached handle; evict stale pages so they can't
+ * be served after the file is later extended; let
+ * the server's O_TRUNC open response set the i_size
+ */
+ truncate_inode_pages(inode->i_mapping, 0);
cifs_invalidate_cache(inode, 0);
}
}
+
+out:
+ filemap_invalidate_unlock(inode->i_mapping);
+ inode_unlock(inode);
if (cfile)
cifsFileInfo_put(cfile);
return rc;
@@ -1491,11 +1515,18 @@ int cifs_close(struct inode *inode, struct file *file)
trace_smb3_close_cached(tcon->tid, tcon->ses->Suid,
cfile->fid.persistent_fid,
cifs_sb->ctx->closetimeo);
- queue_delayed_work(deferredclose_wq,
- &cfile->deferred, cifs_sb->ctx->closetimeo);
- cfile->deferred_close_scheduled = true;
- spin_unlock(&cinode->deferred_lock);
- return 0;
+ /*
+ * Each queued execution owns one reference.
+ * If nothing was queued, the reference of
+ * the closing file is dropped below.
+ */
+ if (queue_delayed_work(deferredclose_wq,
+ &cfile->deferred,
+ cifs_sb->ctx->closetimeo)) {
+ cfile->deferred_close_scheduled = true;
+ spin_unlock(&cinode->deferred_lock);
+ return 0;
+ }
}
spin_unlock(&cinode->deferred_lock);
_cifsFileInfo_put(cfile, true, false);
@@ -3323,9 +3354,12 @@ void cifs_oplock_break(struct work_struct *work)
wait_on_bit(&cinode->flags, CIFS_INODE_PENDING_WRITERS,
TASK_UNINTERRUPTIBLE);
- tlink = cifs_sb_tlink(cifs_sb);
- if (IS_ERR(tlink))
+ tlink = cifs_get_tlink(cfile->tlink);
+ if (IS_ERR_OR_NULL(tlink)) {
+ /* drop the reference taken when the break was queued */
+ _cifsFileInfo_put(cfile, false /* do not wait for ourself */, false);
goto out;
+ }
tcon = tlink_tcon(tlink);
server = tcon->ses->server;
diff --git a/fs/smb/client/inode.c b/fs/smb/client/inode.c
index 12ed8db10e00..1fe0ef0a95db 100644
--- a/fs/smb/client/inode.c
+++ b/fs/smb/client/inode.c
@@ -851,6 +851,7 @@ static void smb311_posix_info_to_fattr(struct cifs_fattr *fattr,
struct smb311_posix_qinfo *info = &data->posix_fi;
struct cifs_sb_info *cifs_sb = CIFS_SB(sb);
struct cifs_tcon *tcon = cifs_sb_master_tcon(cifs_sb);
+ unsigned int sbflags = cifs_sb_flags(cifs_sb);
memset(fattr, 0, sizeof(*fattr));
@@ -895,8 +896,12 @@ out_reparse:
fattr->cf_symlink_target = data->symlink_target;
data->symlink_target = NULL;
}
- sid_to_id(cifs_sb, &data->posix_owner, fattr, SIDOWNER);
- sid_to_id(cifs_sb, &data->posix_group, fattr, SIDGROUP);
+ fattr->cf_uid = cifs_sb->ctx->linux_uid;
+ fattr->cf_gid = cifs_sb->ctx->linux_gid;
+ if (!(sbflags & CIFS_MOUNT_OVERR_UID))
+ sid_to_id(cifs_sb, &data->posix_owner, fattr, SIDOWNER);
+ if (!(sbflags & CIFS_MOUNT_OVERR_GID))
+ sid_to_id(cifs_sb, &data->posix_group, fattr, SIDGROUP);
cifs_dbg(FYI, "POSIX query info: mode 0x%x uniqueid 0x%llx nlink %d\n",
fattr->cf_mode, fattr->cf_uniqueid, fattr->cf_nlink);
@@ -2992,14 +2997,14 @@ int cifs_getattr(struct mnt_idmap *idmap, const struct path *path,
stat->attributes |= STATX_ATTR_ENCRYPTED;
/*
- * If on a multiuser mount without unix extensions or cifsacl being
- * enabled, and the admin hasn't overridden them, set the ownership
- * to the fsuid/fsgid of the current process.
+ * If on a multiuser mount without unix extensions, posix extensions
+ * or cifsacl being enabled, and the admin hasn't overridden them,
+ * set the ownership to the fsuid/fsgid of the current process.
*/
sbflags = cifs_sb_flags(cifs_sb);
if ((sbflags & CIFS_MOUNT_MULTIUSER) &&
!(sbflags & CIFS_MOUNT_CIFS_ACL) &&
- !tcon->unix_ext) {
+ !tcon->unix_ext && !tcon->posix_extensions) {
if (!(sbflags & CIFS_MOUNT_OVERR_UID))
stat->uid = current_fsuid();
if (!(sbflags & CIFS_MOUNT_OVERR_GID))
diff --git a/fs/smb/client/misc.c b/fs/smb/client/misc.c
index 46e1382e8e04..05168284f205 100644
--- a/fs/smb/client/misc.c
+++ b/fs/smb/client/misc.c
@@ -378,10 +378,11 @@ void cifs_queue_oplock_break(struct cifsFileInfo *cfile)
* open_file_lock to enforce the validity of it for the oplock
* break handler. The matching put is done at the end of the
* handler.
+ *
+ * Only take a reference if the work is actually queued.
*/
- cifsFileInfo_get(cfile);
-
- queue_work(cifsoplockd_wq, &cfile->oplock_break);
+ if (queue_work(cifsoplockd_wq, &cfile->oplock_break))
+ cifsFileInfo_get(cfile);
}
void cifs_done_oplock_break(struct cifsInodeInfo *cinode)
@@ -787,7 +788,11 @@ parse_dfs_referrals(struct get_dfs_referral_rsp *rsp, u32 rsp_size,
node->ref_flag = le16_to_cpu(ref->ReferralEntryFlags);
/* copy DfsPath */
- if (le16_to_cpu(ref->DfsPathOffset) > data_end - (char *)ref) {
+ if (le16_to_cpu(ref->DfsPathOffset) < sizeof(*ref) ||
+ le16_to_cpu(ref->DfsPathOffset) > data_end - (char *)ref) {
+ cifs_dbg(VFS, "%s: DfsPathOffset %u out of range [%zu, %td]\n",
+ __func__, le16_to_cpu(ref->DfsPathOffset),
+ sizeof(*ref), data_end - (char *)ref);
rc = -EINVAL;
goto parse_DFS_referrals_exit;
}
@@ -801,7 +806,11 @@ parse_dfs_referrals(struct get_dfs_referral_rsp *rsp, u32 rsp_size,
}
/* copy link target UNC */
- if (le16_to_cpu(ref->NetworkAddressOffset) > data_end - (char *)ref) {
+ if (le16_to_cpu(ref->NetworkAddressOffset) < sizeof(*ref) ||
+ le16_to_cpu(ref->NetworkAddressOffset) > data_end - (char *)ref) {
+ cifs_dbg(VFS, "%s: NetworkAddressOffset %u out of range [%zu, %td]\n",
+ __func__, le16_to_cpu(ref->NetworkAddressOffset),
+ sizeof(*ref), data_end - (char *)ref);
rc = -EINVAL;
goto parse_DFS_referrals_exit;
}
@@ -891,8 +900,14 @@ static void tcon_super_cb(struct super_block *sb, void *arg)
t1->ses->dfs_root_ses == t2->ses->dfs_root_ses) &&
t1->ses->server == t2->ses->server &&
t2->origin_fullpath &&
- dfs_src_pathname_equal(t2->origin_fullpath, t1->origin_fullpath))
+ dfs_src_pathname_equal(t2->origin_fullpath, t1->origin_fullpath)) {
+ /*
+ * Take the active reference while iterate_supers_type() still
+ * holds s_umount shared.
+ */
+ cifs_sb_active(sb);
sd->sb = sb;
+ }
spin_unlock(&t2->tc_lock);
}
@@ -909,15 +924,8 @@ static struct super_block *__cifs_get_super(void (*f)(struct super_block *, void
for (; *fs_type; fs_type++) {
iterate_supers_type(*fs_type, f, &sd);
- if (sd.sb) {
- /*
- * Grab an active reference in order to prevent automounts (DFS links)
- * of expiring and then freeing up our cifs superblock pointer while
- * we're doing failover.
- */
- cifs_sb_active(sd.sb);
+ if (sd.sb)
return sd.sb;
- }
}
pr_warn_once("%s: could not find dfs superblock\n", __func__);
return ERR_PTR(-EINVAL);
diff --git a/fs/smb/client/readdir.c b/fs/smb/client/readdir.c
index 32a75afca8f5..9530e5b01564 100644
--- a/fs/smb/client/readdir.c
+++ b/fs/smb/client/readdir.c
@@ -242,9 +242,11 @@ static void
cifs_posix_to_fattr(struct cifs_fattr *fattr, struct smb2_posix_info *info,
struct cifs_sb_info *cifs_sb)
{
+ unsigned int sbflags = cifs_sb_flags(cifs_sb);
struct smb2_posix_info_parsed parsed;
+ int rc;
- posix_info_parse(info, NULL, &parsed);
+ rc = posix_info_parse(info, NULL, &parsed);
memset(fattr, 0, sizeof(*fattr));
fattr->cf_uniqueid = le64_to_cpu(info->Inode);
@@ -281,8 +283,17 @@ cifs_posix_to_fattr(struct cifs_fattr *fattr, struct smb2_posix_info *info,
le32_to_cpu(info->ReparseTag),
le32_to_cpu(info->Mode));
- sid_to_id(cifs_sb, &parsed.owner, fattr, SIDOWNER);
- sid_to_id(cifs_sb, &parsed.group, fattr, SIDGROUP);
+ fattr->cf_uid = cifs_sb->ctx->linux_uid;
+ fattr->cf_gid = cifs_sb->ctx->linux_gid;
+ if (rc < 0) {
+ cifs_dbg(VFS, "%s: failed to parse SIDs: %d\n",
+ __func__, rc);
+ } else {
+ if (!(sbflags & CIFS_MOUNT_OVERR_UID))
+ sid_to_id(cifs_sb, &parsed.owner, fattr, SIDOWNER);
+ if (!(sbflags & CIFS_MOUNT_OVERR_GID))
+ sid_to_id(cifs_sb, &parsed.group, fattr, SIDGROUP);
+ }
}
static void __dir_info_to_fattr(struct cifs_fattr *fattr, const void *info)
diff --git a/fs/smb/client/reparse.c b/fs/smb/client/reparse.c
index 5cc5b0410d48..3a27773186ae 100644
--- a/fs/smb/client/reparse.c
+++ b/fs/smb/client/reparse.c
@@ -3,6 +3,7 @@
* Copyright (c) 2024 Paulo Alcantara <pc@manguebit.com>
*/
+#include <linux/ctype.h>
#include <linux/fs.h>
#include <linux/stat.h>
#include <linux/slab.h>
@@ -159,15 +160,24 @@ static int create_native_symlink(const unsigned int xid, struct inode *inode,
convert_delimiter(sym, sep);
/*
- * For absolute NT symlinks it is required to pass also leading
- * backslash and to not mangle NT object prefix "\\??\\" and not to
- * mangle colon in drive letter. But cifs_convert_path_to_utf16()
- * removes leading backslash and replaces '?' and ':'. So temporary
- * mask these characters in NT object prefix by '_' and then change
- * them back.
+ * Absolute NT symlinks must retain the leading backslash, "\\??\\"
+ * prefix and drive-letter colon. cifs_convert_path_to_utf16() strips
+ * the leading backslash and maps '?' and ':', so temporarily mask
+ * these characters with '_' and restore them after conversion.
+ *
+ * When symlinkroot is unset, sym comes directly from the caller.
+ * Validate the complete "\\??\\X:" prefix before using fixed offsets
+ * or subtracting the NT prefix length below. Require an ASCII drive
+ * letter so the prefix occupies six characters in UTF-16 too.
*/
- if (!(sbflags & CIFS_MOUNT_POSIX_PATHS) && symname[0] == '/')
+ if (!(sbflags & CIFS_MOUNT_POSIX_PATHS) && symname[0] == '/') {
+ if (!strstarts(sym, "\\??\\") || !isascii(sym[4]) ||
+ !isalpha(sym[4]) || sym[5] != ':') {
+ rc = -EINVAL;
+ goto out;
+ }
sym[0] = sym[1] = sym[2] = sym[5] = '_';
+ }
/*
* On a POSIX paths mount the symlink target is stored verbatim, so
@@ -971,7 +981,8 @@ globalroot:
linux_target[i*3 + 1] = '.';
linux_target[i*3 + 2] = sep;
}
- memcpy(linux_target + levels*3, smb_target+1, smb_target_len); /* +1 to skip leading sep */
+ /* +1 to skip leading sep */
+ memcpy(linux_target + levels*3, smb_target+1, smb_target_len-1);
} else {
/*
* This is either an absolute symlink in POSIX-style format
@@ -1137,25 +1148,31 @@ static bool wsl_to_fattr(struct cifs_open_info_data *data,
struct cifs_sb_info *cifs_sb,
u32 tag, struct cifs_fattr *fattr)
{
+ unsigned int sbflags = cifs_sb_flags(cifs_sb);
+ kuid_t uid = cifs_sb->ctx->linux_uid;
+ kgid_t gid = cifs_sb->ctx->linux_gid;
struct smb2_file_full_ea_info *ea;
bool have_xattr_dev = false;
+ dev_t rdev = 0;
+ umode_t mode;
u32 next = 0;
+ mode = fattr->cf_mode & ~S_IFMT;
switch (tag) {
case IO_REPARSE_TAG_LX_SYMLINK:
- fattr->cf_mode |= S_IFLNK;
+ mode |= S_IFLNK;
break;
case IO_REPARSE_TAG_LX_FIFO:
- fattr->cf_mode |= S_IFIFO;
+ mode |= S_IFIFO;
break;
case IO_REPARSE_TAG_AF_UNIX:
- fattr->cf_mode |= S_IFSOCK;
+ mode |= S_IFSOCK;
break;
case IO_REPARSE_TAG_LX_CHR:
- fattr->cf_mode |= S_IFCHR;
+ mode |= S_IFCHR;
break;
case IO_REPARSE_TAG_LX_BLK:
- fattr->cf_mode |= S_IFBLK;
+ mode |= S_IFBLK;
break;
}
@@ -1177,26 +1194,31 @@ static bool wsl_to_fattr(struct cifs_open_info_data *data,
nlen = ea->ea_name_length;
v = (void *)((u8 *)ea->ea_data + ea->ea_name_length + 1);
- if (!strncmp(name, SMB2_WSL_XATTR_UID, nlen))
- fattr->cf_uid = wsl_make_kuid(cifs_sb, v);
- else if (!strncmp(name, SMB2_WSL_XATTR_GID, nlen))
- fattr->cf_gid = wsl_make_kgid(cifs_sb, v);
- else if (!strncmp(name, SMB2_WSL_XATTR_MODE, nlen)) {
+ if (!strncmp(name, SMB2_WSL_XATTR_UID, nlen)) {
+ if (!(sbflags & CIFS_MOUNT_OVERR_UID))
+ uid = wsl_make_kuid(cifs_sb, v);
+ } else if (!strncmp(name, SMB2_WSL_XATTR_GID, nlen)) {
+ if (!(sbflags & CIFS_MOUNT_OVERR_GID))
+ gid = wsl_make_kgid(cifs_sb, v);
+ } else if (!strncmp(name, SMB2_WSL_XATTR_MODE, nlen)) {
/* File type in reparse point tag and in xattr mode must match. */
- if (S_DT(fattr->cf_mode) != S_DT(le32_to_cpu(*(__le32 *)v)))
+ if (S_DT(mode) != S_DT(get_unaligned_le32(v)))
return false;
- fattr->cf_mode = (umode_t)le32_to_cpu(*(__le32 *)v);
+ mode = get_unaligned_le32(v);
} else if (!strncmp(name, SMB2_WSL_XATTR_DEV, nlen)) {
- fattr->cf_rdev = reparse_mkdev(v);
+ rdev = reparse_mkdev(v);
have_xattr_dev = true;
}
} while (next);
out:
-
/* Major and minor numbers for char and block devices are mandatory. */
if (!have_xattr_dev && (tag == IO_REPARSE_TAG_LX_CHR || tag == IO_REPARSE_TAG_LX_BLK))
return false;
+ fattr->cf_uid = uid;
+ fattr->cf_gid = gid;
+ fattr->cf_mode = mode;
+ fattr->cf_rdev = rdev;
return true;
}
@@ -1205,6 +1227,7 @@ static bool posix_reparse_to_fattr(struct cifs_sb_info *cifs_sb,
struct cifs_open_info_data *data)
{
struct reparse_nfs_data_buffer *buf = (struct reparse_nfs_data_buffer *)data->reparse.buf;
+ umode_t ftype;
if (buf == NULL)
return true;
@@ -1220,7 +1243,7 @@ static bool posix_reparse_to_fattr(struct cifs_sb_info *cifs_sb,
WARN_ON_ONCE(1);
return false;
}
- fattr->cf_mode |= S_IFCHR;
+ ftype = S_IFCHR;
fattr->cf_rdev = reparse_mkdev(buf->DataBuffer);
break;
case NFS_SPECFILE_BLK:
@@ -1228,22 +1251,23 @@ static bool posix_reparse_to_fattr(struct cifs_sb_info *cifs_sb,
WARN_ON_ONCE(1);
return false;
}
- fattr->cf_mode |= S_IFBLK;
+ ftype = S_IFBLK;
fattr->cf_rdev = reparse_mkdev(buf->DataBuffer);
break;
case NFS_SPECFILE_FIFO:
- fattr->cf_mode |= S_IFIFO;
+ ftype = S_IFIFO;
break;
case NFS_SPECFILE_SOCK:
- fattr->cf_mode |= S_IFSOCK;
+ ftype = S_IFSOCK;
break;
case NFS_SPECFILE_LNK:
- fattr->cf_mode |= S_IFLNK;
+ ftype = S_IFLNK;
break;
default:
WARN_ON_ONCE(1);
return false;
}
+ fattr->cf_mode = (fattr->cf_mode & ~S_IFMT) | ftype;
return true;
}
@@ -1271,6 +1295,7 @@ bool cifs_reparse_point_to_fattr(struct cifs_sb_info *cifs_sb,
break;
case 0: /* SMB1 symlink */
case IO_REPARSE_TAG_SYMLINK:
+ fattr->cf_mode &= ~S_IFMT;
fattr->cf_mode |= S_IFLNK;
break;
default:
diff --git a/fs/smb/client/reparse.h b/fs/smb/client/reparse.h
index 49efd85b1e94..05b2cecb4495 100644
--- a/fs/smb/client/reparse.h
+++ b/fs/smb/client/reparse.h
@@ -9,6 +9,7 @@
#include <linux/fs.h>
#include <linux/stat.h>
#include <linux/uidgid.h>
+#include <linux/unaligned.h>
#include "fs_context.h"
#include "cifsglob.h"
#include "../common/smbfsctl.h"
@@ -23,7 +24,7 @@
static inline dev_t reparse_mkdev(void *ptr)
{
- u64 v = le64_to_cpu(*(__le64 *)ptr);
+ u64 v = get_unaligned_le64(ptr);
return MKDEV(v & 0xffffffff, v >> 32);
}
@@ -31,7 +32,7 @@ static inline dev_t reparse_mkdev(void *ptr)
static inline kuid_t wsl_make_kuid(struct cifs_sb_info *cifs_sb,
void *ptr)
{
- u32 uid = le32_to_cpu(*(__le32 *)ptr);
+ u32 uid = get_unaligned_le32(ptr);
if (cifs_sb_flags(cifs_sb) & CIFS_MOUNT_OVERR_UID)
return cifs_sb->ctx->linux_uid;
@@ -41,7 +42,7 @@ static inline kuid_t wsl_make_kuid(struct cifs_sb_info *cifs_sb,
static inline kgid_t wsl_make_kgid(struct cifs_sb_info *cifs_sb,
void *ptr)
{
- u32 gid = le32_to_cpu(*(__le32 *)ptr);
+ u32 gid = get_unaligned_le32(ptr);
if (cifs_sb_flags(cifs_sb) & CIFS_MOUNT_OVERR_GID)
return cifs_sb->ctx->linux_gid;
diff --git a/fs/smb/client/sess.c b/fs/smb/client/sess.c
index 7cf7dd104f7c..e095f41b5882 100644
--- a/fs/smb/client/sess.c
+++ b/fs/smb/client/sess.c
@@ -149,9 +149,9 @@ int cifs_try_adding_channels(struct cifs_ses *ses)
int old_chan_count, new_chan_count;
int left;
int rc = 0;
- int tries = 0;
+ int tries = 0, attempts;
size_t iface_weight = 0, iface_min_speed = 0;
- struct cifs_server_iface *iface = NULL, *niface = NULL;
+ struct cifs_server_iface *iface = NULL, *candidate = NULL;
struct cifs_server_iface *last_iface = NULL;
spin_lock(&ses->chan_lock);
@@ -197,67 +197,89 @@ int cifs_try_adding_channels(struct cifs_ses *ses)
break;
}
- if (!iface)
- iface = list_first_entry(&ses->iface_list, struct cifs_server_iface,
- iface_head);
last_iface = list_last_entry(&ses->iface_list, struct cifs_server_iface,
iface_head);
iface_min_speed = last_iface->speed;
+ spin_unlock(&ses->iface_lock);
- list_for_each_entry_safe_from(iface, niface, &ses->iface_list,
- iface_head) {
- /* do not mix rdma and non-rdma interfaces */
- if (iface->rdma_capable != ses->server->rdma)
- continue;
-
- /* skip ifaces that are unusable */
- if (!iface->is_active ||
- (is_ses_using_iface(ses, iface) &&
- !iface->rss_capable))
- continue;
+ attempts = 0;
+ while (left > 0) {
+ spin_lock(&ses->iface_lock);
- /* check if we already allocated enough channels */
- iface_weight = iface->speed / iface_min_speed;
+ /*
+ * iface_lock must be dropped while opening a channel,
+ * and a concurrent interface refresh may remove and
+ * free entries during that window, so no list entry
+ * may be kept across it without a reference. Scan
+ * the list from the beginning each time and only pass
+ * a referenced candidate to cifs_ses_add_channel();
+ * weight_fulfilled tracks the progress so that no
+ * iface is selected beyond its weight.
+ */
+ candidate = NULL;
+ list_for_each_entry(iface, &ses->iface_list, iface_head) {
+ /* do not mix rdma and non-rdma interfaces */
+ if (iface->rdma_capable != ses->server->rdma)
+ continue;
+
+ /* skip ifaces that are unusable */
+ if (!iface->is_active ||
+ (is_ses_using_iface(ses, iface) &&
+ !iface->rss_capable))
+ continue;
+
+ /* check if we already allocated enough channels */
+ iface_weight = iface->speed / iface_min_speed;
+
+ if (iface->weight_fulfilled >= iface_weight)
+ continue;
+
+ /* take ref before unlock */
+ kref_get(&iface->refcount);
+ candidate = iface;
+ break;
+ }
- if (iface->weight_fulfilled >= iface_weight)
- continue;
+ if (!candidate) {
+ /* no usable iface. reset weight_fulfilled and start over */
+ list_for_each_entry(iface, &ses->iface_list, iface_head)
+ iface->weight_fulfilled = 0;
+ spin_unlock(&ses->iface_lock);
+ break;
+ }
- /* take ref before unlock */
- kref_get(&iface->refcount);
+ attempts++;
+ if (attempts > 3 * ses->chan_max) {
+ kref_put(&candidate->refcount, release_iface);
+ spin_unlock(&ses->iface_lock);
+ break;
+ }
spin_unlock(&ses->iface_lock);
- rc = cifs_ses_add_channel(ses, iface);
+ rc = cifs_ses_add_channel(ses, candidate);
spin_lock(&ses->iface_lock);
if (rc) {
cifs_dbg(VFS, "failed to open extra channel on iface:%pIS rc=%d\n",
- &iface->sockaddr,
+ &candidate->sockaddr,
rc);
/* failure to add chan should increase weight */
- iface->weight_fulfilled++;
- kref_put(&iface->refcount, release_iface);
+ candidate->weight_fulfilled++;
+ kref_put(&candidate->refcount, release_iface);
+ spin_unlock(&ses->iface_lock);
continue;
}
- iface->num_channels++;
- iface->weight_fulfilled++;
+ candidate->num_channels++;
+ candidate->weight_fulfilled++;
cifs_info("successfully opened new channel on iface:%pIS\n",
- &iface->sockaddr);
- break;
- }
-
- /* reached end of list. reset weight_fulfilled and start over */
- if (list_entry_is_head(iface, &ses->iface_list, iface_head)) {
- list_for_each_entry(iface, &ses->iface_list, iface_head)
- iface->weight_fulfilled = 0;
+ &candidate->sockaddr);
spin_unlock(&ses->iface_lock);
- iface = NULL;
- continue;
- }
- spin_unlock(&ses->iface_lock);
- left--;
- new_chan_count++;
+ left--;
+ new_chan_count++;
+ break;
+ }
}
return new_chan_count - old_chan_count;
diff --git a/fs/smb/client/smb2inode.c b/fs/smb/client/smb2inode.c
index 98ea5c6c34af..13fe8e3b48f3 100644
--- a/fs/smb/client/smb2inode.c
+++ b/fs/smb/client/smb2inode.c
@@ -77,6 +77,17 @@ static int parse_posix_sids(struct cifs_open_info_data *data,
sidsbuf = (u8 *)qi + le16_to_cpu(qi->OutputBufferOffset) + qi_len;
sidsbuf_end = sidsbuf + out_len - qi_len;
+ if (sidsbuf_end < sidsbuf) {
+ cifs_dbg(VFS, "%s: server-supplied out_len %u caused pointer wraparound\n",
+ __func__, out_len);
+ return -EINVAL;
+ }
+ if (sidsbuf_end > (u8 *)rsp_iov->iov_base + rsp_iov->iov_len) {
+ cifs_dbg(VFS, "%s: server-supplied out_len %u overruns iov by %td bytes\n",
+ __func__, out_len,
+ sidsbuf_end - ((u8 *)rsp_iov->iov_base + rsp_iov->iov_len));
+ return -EINVAL;
+ }
owner_len = posix_info_sid_size(sidsbuf, sidsbuf_end);
if (owner_len == -1)
@@ -237,7 +248,7 @@ replay_again:
num_rqst = 0;
server = cifs_pick_channel(ses);
- vars = kzalloc_obj(*vars, GFP_KERNEL);
+ vars = kzalloc_obj(*vars);
if (vars == NULL) {
rc = -ENOMEM;
goto out;
diff --git a/fs/smb/client/smb2misc.c b/fs/smb/client/smb2misc.c
index 9068175e57cd..0cfe60ae42c3 100644
--- a/fs/smb/client/smb2misc.c
+++ b/fs/smb/client/smb2misc.c
@@ -85,6 +85,36 @@ static const __le16 smb2_rsp_struct_sizes[NUMBER_OF_SMB2_COMMANDS] = {
/* SMB2_OPLOCK_BREAK */ cpu_to_le16(24)
};
+/*
+ * Minimum received PDU size for commands whose response carries a
+ * variable-length data area. A non-zero entry marks the command as
+ * having one, and gives the length smb2_check_message() requires
+ * before smb2_get_data_area_len() reads the offset and length fields
+ * out of the fixed response struct.
+ */
+static const size_t smb2_min_pdu_len[NUMBER_OF_SMB2_COMMANDS] = {
+ /* SMB2_NEGOTIATE */ sizeof(struct smb2_negotiate_rsp),
+ /* SMB2_SESSION_SETUP */ sizeof(struct smb2_sess_setup_rsp),
+ /* SMB2_LOGOFF */ 0,
+ /* SMB2_TREE_CONNECT */ 0,
+ /* SMB2_TREE_DISCONNECT */ 0,
+ /* SMB2_CREATE */ sizeof(struct smb2_create_rsp),
+ /* SMB2_CLOSE */ 0,
+ /* SMB2_FLUSH */ 0,
+ /* SMB2_READ */ sizeof(struct smb2_read_rsp),
+ /* SMB2_WRITE */ 0,
+ /* SMB2_LOCK */ 0,
+ /* SMB2_IOCTL */ sizeof(struct smb2_ioctl_rsp),
+ /* SMB2_CANCEL */ 0,
+ /* SMB2_ECHO */ 0,
+ /* SMB2_QUERY_DIRECTORY */ sizeof(struct smb2_query_directory_rsp),
+ /* SMB2_CHANGE_NOTIFY */ sizeof(struct smb2_change_notify_rsp),
+ /* SMB2_QUERY_INFO */ sizeof(struct smb2_query_info_rsp),
+ /* SMB2_SET_INFO */ 0,
+ /* SMB2_OPLOCK_BREAK */ 0,
+};
+
+#define smb2_has_data_area(cmd) (smb2_min_pdu_len[cmd] != 0)
#define SMB311_NEGPROT_BASE_SIZE (sizeof(struct smb2_hdr) + sizeof(struct smb2_negotiate_rsp))
static __u32 get_neg_ctxt_len(struct smb2_hdr *hdr, __u32 len,
@@ -233,6 +263,16 @@ smb2_check_message(char *buf, unsigned int pdu_len, unsigned int len,
}
}
+ if ((shdr->Status == STATUS_SUCCESS ||
+ shdr->Status == STATUS_MORE_PROCESSING_REQUIRED ||
+ pdu->StructureSize2 != SMB2_ERROR_STRUCTURE_SIZE2_LE) &&
+ smb2_has_data_area(command) &&
+ len < smb2_min_pdu_len[command]) {
+ cifs_server_dbg(VFS, "SMB2 command %d response too short: %u < %zu\n",
+ command, len, smb2_min_pdu_len[command]);
+ return 1;
+ }
+
have_data = false;
data_area_overlap = false;
calc_len = __smb2_calc_size(buf, &have_data, &data_area_overlap);
@@ -299,33 +339,6 @@ smb2_check_message(char *buf, unsigned int pdu_len, unsigned int len,
}
/*
- * The size of the variable area depends on the offset and length fields
- * located in different fields for various SMB2 responses. SMB2 responses
- * with no variable length info, show an offset of zero for the offset field.
- */
-static const bool has_smb2_data_area[NUMBER_OF_SMB2_COMMANDS] = {
- /* SMB2_NEGOTIATE */ true,
- /* SMB2_SESSION_SETUP */ true,
- /* SMB2_LOGOFF */ false,
- /* SMB2_TREE_CONNECT */ false,
- /* SMB2_TREE_DISCONNECT */ false,
- /* SMB2_CREATE */ true,
- /* SMB2_CLOSE */ false,
- /* SMB2_FLUSH */ false,
- /* SMB2_READ */ true,
- /* SMB2_WRITE */ false,
- /* SMB2_LOCK */ false,
- /* SMB2_IOCTL */ true,
- /* SMB2_CANCEL */ false, /* BB CHECK this not listed in documentation */
- /* SMB2_ECHO */ false,
- /* SMB2_QUERY_DIRECTORY */ true,
- /* SMB2_CHANGE_NOTIFY */ true,
- /* SMB2_QUERY_INFO */ true,
- /* SMB2_SET_INFO */ false,
- /* SMB2_OPLOCK_BREAK */ false
-};
-
-/*
* Returns the pointer to the beginning of the data area. Length of the data
* area and the offset to it (from the beginning of the smb are also returned.
*/
@@ -451,7 +464,7 @@ __smb2_calc_size(void *buf, bool *have_data, bool *data_area_overlap)
*/
len += le16_to_cpu(pdu->StructureSize2);
- if (has_smb2_data_area[le16_to_cpu(shdr->Command)] == false)
+ if (!smb2_has_data_area(le16_to_cpu(shdr->Command)))
goto calc_size_exit;
smb2_get_data_area_len(&offset, &data_length, shdr);
diff --git a/fs/smb/client/smb2ops.c b/fs/smb/client/smb2ops.c
index 7d6738ffcb80..3464470d3297 100644
--- a/fs/smb/client/smb2ops.c
+++ b/fs/smb/client/smb2ops.c
@@ -785,9 +785,9 @@ next_iface:
break;
}
/* Validate that Next doesn't point beyond the buffer */
- if (next > bytes_left) {
- cifs_dbg(VFS, "%s: invalid Next pointer %zu > %zd\n",
- __func__, next, bytes_left);
+ if (next < sizeof(*p) || next > bytes_left) {
+ cifs_dbg(VFS, "%s: invalid Next pointer %zu out of range [%zu, %zd]\n",
+ __func__, next, sizeof(*p), bytes_left);
rc = -EINVAL;
goto out;
}
@@ -1053,8 +1053,9 @@ move_smb2_ea_to_cifs(char *dst, size_t dst_size,
char *name, *value;
size_t buf_size = dst_size;
size_t name_len, value_len, user_name_len;
+ u32 next_off;
- while (src_size > 0) {
+ while (src_size >= sizeof(*src)) {
name_len = (size_t)src->ea_name_length;
value_len = (size_t)le16_to_cpu(src->ea_value_length);
@@ -1110,14 +1111,22 @@ move_smb2_ea_to_cifs(char *dst, size_t dst_size,
if (!src->next_entry_offset)
break;
- if (src_size < le32_to_cpu(src->next_entry_offset)) {
- /* stop before overrun buffer */
- rc = -ERANGE;
- break;
+ next_off = le32_to_cpu(src->next_entry_offset);
+ if (next_off < sizeof(*src) || src_size < next_off) {
+ cifs_dbg(FYI, "EA next_entry_offset %u out of range [%zu, %zu]\n",
+ next_off, sizeof(*src), src_size);
+ rc = smb_EIO2(smb_eio_trace_ea_next_offset,
+ next_off, src_size);
+ goto out;
+ }
+ src_size -= next_off;
+ src = (void *)((char *)src + next_off);
+ if (src_size > 0 && src_size < sizeof(*src)) {
+ cifs_dbg(FYI, "EA next_entry_offset %u left truncated entry (%zu bytes)\n",
+ next_off, src_size);
+ rc = smb_EIO2(smb_eio_trace_ea_next_offset, next_off, src_size);
+ goto out;
}
- src_size -= le32_to_cpu(src->next_entry_offset);
- src = (void *)((char *)src +
- le32_to_cpu(src->next_entry_offset));
}
/* didn't find the named attribute */
@@ -1839,31 +1848,31 @@ free_vars:
*
* @tcon: destination file tcon
* @bytes_left: how many bytes are left to copy
+ * @chunk_size: maximum size of a single chunk
*
* Return: maximum number of chunks with which Chunks[] can be filled.
*/
static inline u32
-calc_chunk_count(struct cifs_tcon *tcon, u64 bytes_left)
+calc_chunk_count(struct cifs_tcon *tcon, u64 bytes_left, u32 chunk_size)
{
u32 max_chunks = READ_ONCE(tcon->max_chunks);
u32 max_bytes_copy = READ_ONCE(tcon->max_bytes_copy);
- u32 max_bytes_chunk = READ_ONCE(tcon->max_bytes_chunk);
u64 need;
u32 allowed;
- if (!max_bytes_chunk || !max_bytes_copy || !max_chunks)
+ if (!chunk_size || !max_bytes_copy || !max_chunks)
return 0;
/* chunks needed for the remaining bytes */
- need = DIV_ROUND_UP_ULL(bytes_left, max_bytes_chunk);
+ need = DIV_ROUND_UP_ULL(bytes_left, chunk_size);
/* chunks allowed per cc request */
- allowed = DIV_ROUND_UP(max_bytes_copy, max_bytes_chunk);
+ allowed = DIV_ROUND_UP(max_bytes_copy, chunk_size);
return (u32)umin(need, umin(max_chunks, allowed));
}
/**
- * smb2_copychunk_range - server-side copy of data range
+ * __smb2_copychunk_range - server-side copy of data range
*
* @xid: transaction id
* @src_file: source file
@@ -1875,15 +1884,15 @@ calc_chunk_count(struct cifs_tcon *tcon, u64 bytes_left)
* Obtains a resume key for @src_file and issues FSCTL_SRV_COPYCHUNK_WRITE
* IOCTLs, splitting the request into chunks limited by tcon->max_*.
*
- * Return: @len on success; negative errno on failure.
+ * Return: 0 on success; negative errno on failure.
*/
-static ssize_t
-smb2_copychunk_range(const unsigned int xid,
- struct cifsFileInfo *src_file,
- struct cifsFileInfo *dst_file,
- u64 src_off,
- u64 len,
- u64 dst_off)
+static int
+__smb2_copychunk_range(const unsigned int xid,
+ struct cifsFileInfo *src_file,
+ struct cifsFileInfo *dst_file,
+ u64 src_off,
+ u64 len,
+ u64 dst_off)
{
int rc = 0;
unsigned int ret_data_len = 0;
@@ -1891,12 +1900,14 @@ smb2_copychunk_range(const unsigned int xid,
struct copychunk_ioctl_rsp *cc_rsp = NULL;
struct cifs_tcon *tcon;
struct srv_copychunk *chunk;
- u32 chunks, chunk_count, chunk_bytes;
+ u32 chunks, chunk_count, chunk_bytes, chunk_size;
u32 copy_bytes, copy_bytes_left;
u32 chunks_written, bytes_written;
u64 total_bytes_left = len;
u64 src_off_prev, dst_off_prev;
+ u64 max_chunk = 0;
u32 retries = 0;
+ bool reverse = false;
tcon = tlink_tcon(dst_file->tlink);
@@ -1904,8 +1915,50 @@ smb2_copychunk_range(const unsigned int xid,
dst_file->fid.volatile_fid, tcon->tid,
tcon->ses->Suid, src_off, dst_off, len);
+ /*
+ * Same-file left shifts are safe in forward order. For a right shift,
+ * let L be the copy length, delta the distance between the source and
+ * destination, and C the normal chunk size:
+ *
+ * delta >= L: copy forwards using C
+ * delta < L:
+ * delta >= C: copy backwards using C
+ * delta < C: copy backwards with chunks limited to delta
+ *
+ * Copying backwards prevents one chunk from overwriting data needed by
+ * a later chunk. Limiting the chunk size to delta prevents an individual
+ * chunk from overlapping itself.
+ * This limit can be removed once all supported servers handle overlapping
+ * descriptors safely.
+ *
+ * A small right shift over a large range may therefore require many
+ * chunks.
+ */
+ if (src_file == dst_file && dst_off > src_off) {
+ u64 delta = dst_off - src_off;
+
+ if (delta < len) {
+ reverse = true;
+ max_chunk = delta;
+ }
+ }
+
+ /*
+ * A backward copy walks the offsets down from the end of the range.
+ * Do this once, outside the retry loop, so a retry does not move the
+ * offsets again.
+ */
+ if (reverse) {
+ src_off += len;
+ dst_off += len;
+ }
+
retry:
- chunk_count = calc_chunk_count(tcon, total_bytes_left);
+ chunk_size = READ_ONCE(tcon->max_bytes_chunk);
+ if (max_chunk && max_chunk < chunk_size)
+ chunk_size = (u32)max_chunk;
+
+ chunk_count = calc_chunk_count(tcon, total_bytes_left, chunk_size);
if (!chunk_count) {
rc = -EOPNOTSUPP;
goto out;
@@ -1946,16 +1999,21 @@ retry:
while (copy_bytes_left > 0 && chunks < chunk_count) {
chunk = &cc_req->Chunks[chunks++];
+ chunk_bytes = umin(copy_bytes_left, chunk_size);
+ if (reverse) {
+ src_off -= chunk_bytes;
+ dst_off -= chunk_bytes;
+ }
+
chunk->SourceOffset = cpu_to_le64(src_off);
chunk->TargetOffset = cpu_to_le64(dst_off);
-
- chunk_bytes = umin(copy_bytes_left, tcon->max_bytes_chunk);
-
chunk->Length = cpu_to_le32(chunk_bytes);
/* Buffer is zeroed, no need to set chunk->Reserved = 0 */
- src_off += chunk_bytes;
- dst_off += chunk_bytes;
+ if (!reverse) {
+ src_off += chunk_bytes;
+ dst_off += chunk_bytes;
+ }
copy_bytes_left -= chunk_bytes;
copy_bytes += chunk_bytes;
@@ -2003,6 +2061,18 @@ retry:
goto out;
}
+ /*
+ * A successful COPYCHUNK should copy every descriptor (MS-SMB2
+ * 3.3.5.15.6). Reject a short backward copy because the rewind
+ * below only supports forward copying.
+ */
+ if (unlikely(reverse && bytes_written < copy_bytes)) {
+ cifs_tcon_dbg(VFS, "Copychunk short write %u/%u (reverse)\n",
+ bytes_written, copy_bytes);
+ rc = -EIO;
+ goto out;
+ }
+
/* Partial write: rewind */
if (bytes_written < copy_bytes) {
u32 delta = copy_bytes - bytes_written;
@@ -2064,10 +2134,27 @@ out:
trace_smb3_copychunk_done(xid, src_file->fid.volatile_fid,
dst_file->fid.volatile_fid, tcon->tid,
tcon->ses->Suid, src_off, dst_off, len);
- return len;
+ return 0;
}
}
+static ssize_t
+smb2_copychunk_range(const unsigned int xid,
+ struct cifsFileInfo *src_file,
+ struct cifsFileInfo *dst_file,
+ u64 src_off,
+ u64 len,
+ u64 dst_off)
+{
+ int rc;
+
+ rc = __smb2_copychunk_range(xid, src_file, dst_file, src_off, len,
+ dst_off);
+ if (rc)
+ return rc;
+ return len;
+}
+
static int
smb2_flush_file(const unsigned int xid, struct cifs_tcon *tcon,
struct cifs_fid *fid)
@@ -2218,7 +2305,7 @@ smb2_duplicate_extents(const unsigned int xid,
trgtfile->fid.volatile_fid, tcon->tid,
tcon->ses->Suid, src_off, dest_off, len);
inode = d_inode(trgtfile->dentry);
- if (inode->i_size < dest_off + len) {
+ if (i_size_read(inode) < dest_off + len) {
rc = smb2_set_file_size(xid, tcon, trgtfile, dest_off + len, false);
if (rc)
goto duplicate_extents_out;
@@ -2235,7 +2322,10 @@ smb2_duplicate_extents(const unsigned int xid,
if (ret_data_len > 0)
cifs_dbg(FYI, "Non-zero response length in duplicate extents\n");
- if (rc == 0) {
+ if (rc) {
+ CIFS_I(inode)->time = 0; /* force reval */
+ cifs_invalidate_cache(inode, 0);
+ } else {
qrc = SMB2_query_info(xid, tcon, trgtfile->fid.persistent_fid,
trgtfile->fid.volatile_fid, &file_inf);
spin_lock(&inode->i_lock);
@@ -2373,8 +2463,14 @@ smb3_enum_snapshots(const unsigned int xid, struct cifs_tcon *tcon,
* and retry the ioctl again with larger array size sufficient
* to hold all of the snapshot GMT tokens on the second try.
*/
- if (snapshot_in.snapshot_array_size < GMT_TOKEN_SIZE)
+ if (snapshot_in.snapshot_array_size < GMT_TOKEN_SIZE) {
+ if (ret_data_len < sizeof(struct smb_snapshot_array)) {
+ rc = -EIO;
+ kfree(retbuf);
+ return rc;
+ }
ret_data_len = sizeof(struct smb_snapshot_array);
+ }
/*
* We return struct SRV_SNAPSHOT_ARRAY, followed by
@@ -3441,6 +3537,13 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
trace_smb3_zero_enter(xid, cfile->fid.persistent_fid, tcon->tid,
ses->Suid, offset, len);
+ new_size = offset + len;
+ if (!keep_size && i_size_read(inode) < new_size) {
+ rc = inode_newsize_ok(inode, new_size);
+ if (rc)
+ goto out;
+ }
+
filemap_invalidate_lock(inode->i_mapping);
netfs_read_sizes(inode, &i_size, &remote_i_size, &zero_point);
@@ -3464,6 +3567,9 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
if (keep_size == false && !CIFS_CACHE_READ(cifsi))
goto zero_range_exit;
+ fscache_invalidate(cifs_inode_cookie(inode), NULL,
+ i_size_read(inode), 0);
+
rc = smb3_zero_data(file, tcon, offset, len, xid);
if (rc < 0)
goto zero_range_exit;
@@ -3471,7 +3577,6 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
/*
* do we also need to change the size of the file?
*/
- new_size = offset + len;
if (keep_size == false && (unsigned long long)i_size_read(inode) < new_size) {
rc = SMB2_set_eof(xid, tcon, cfile->fid.persistent_fid,
cfile->fid.volatile_fid, cfile->pid, new_size);
@@ -3488,6 +3593,7 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
zero_range_exit:
filemap_invalidate_unlock(inode->i_mapping);
+ out:
free_xid(xid);
if (rc)
trace_smb3_zero_err(xid, cfile->fid.persistent_fid, tcon->tid,
@@ -3533,6 +3639,8 @@ static long smb3_punch_hole(struct file *file, struct cifs_tcon *tcon,
*/
truncate_pagecache_range(inode, offset, offset + len - 1);
netfs_wait_for_outstanding_io(inode);
+ fscache_invalidate(cifs_inode_cookie(inode), NULL,
+ i_size_read(inode), 0);
cifs_dbg(FYI, "Offset %lld len %lld\n", offset, len);
@@ -3938,18 +4046,26 @@ static long smb3_collapse_range(struct file *file, struct cifs_tcon *tcon,
}
filemap_invalidate_lock(inode->i_mapping);
- rc = filemap_write_and_wait_range(inode->i_mapping, off, old_eof - 1);
+ rc = filemap_write_and_wait_range(inode->i_mapping,
+ round_down(off, PAGE_SIZE),
+ old_eof - 1);
if (rc < 0)
goto out_2;
- truncate_pagecache_range(inode, off, old_eof);
+ netfs_wait_for_outstanding_io(inode);
+ /*
+ * Invalidate cached folios from the page containing off to EOF before
+ * moving data on the server, so subsequent reads do not see stale data.
+ */
+ truncate_pagecache_range(inode, round_down(off, PAGE_SIZE), -1);
+ fscache_invalidate(cifs_inode_cookie(inode), NULL, old_eof, 0);
+
spin_lock(&inode->i_lock);
netfs_write_zero_point(inode, old_eof);
spin_unlock(&inode->i_lock);
- netfs_wait_for_outstanding_io(inode);
- rc = smb2_copychunk_range(xid, cfile, cfile, off + len,
- old_eof - off - len, off);
+ rc = __smb2_copychunk_range(xid, cfile, cfile, off + len,
+ old_eof - off - len, off);
if (rc < 0)
goto out_2;
@@ -3982,7 +4098,7 @@ static long smb3_insert_range(struct file *file, struct cifs_tcon *tcon,
struct cifsFileInfo *cfile = file->private_data;
struct inode *inode = file_inode(file);
struct cifsInodeInfo *cifsi = CIFS_I(inode);
- __u64 count, old_eof, new_eof;
+ loff_t old_eof, new_eof;
xid = get_xid();
@@ -3992,15 +4108,32 @@ static long smb3_insert_range(struct file *file, struct cifs_tcon *tcon,
goto out;
}
- count = old_eof - off;
- new_eof = old_eof + len;
+ if (check_add_overflow(old_eof, len, &new_eof)) {
+ rc = -EFBIG;
+ goto out;
+ }
+ rc = inode_newsize_ok(inode, new_eof);
+ if (rc)
+ goto out;
+
+ /* SET_ZERO_DATA creates a hole only in a sparse file. */
+ rc = smb2_set_sparse(xid, tcon, cfile, inode, true);
+ if (rc)
+ goto out;
filemap_invalidate_lock(inode->i_mapping);
- rc = filemap_write_and_wait_range(inode->i_mapping, off, new_eof - 1);
+ rc = filemap_write_and_wait_range(inode->i_mapping,
+ round_down(off, PAGE_SIZE),
+ old_eof - 1);
if (rc < 0)
goto out_2;
- truncate_pagecache_range(inode, off, old_eof);
netfs_wait_for_outstanding_io(inode);
+ /*
+ * Invalidate cached folios from the page containing off to EOF before
+ * moving data on the server, so subsequent reads do not see stale data.
+ */
+ truncate_pagecache_range(inode, round_down(off, PAGE_SIZE), -1);
+ fscache_invalidate(cifs_inode_cookie(inode), NULL, old_eof, 0);
rc = SMB2_set_eof(xid, tcon, cfile->fid.persistent_fid,
cfile->fid.volatile_fid, cfile->pid, new_eof);
@@ -4013,7 +4146,12 @@ static long smb3_insert_range(struct file *file, struct cifs_tcon *tcon,
spin_unlock(&inode->i_lock);
fscache_resize_cookie(cifs_inode_cookie(inode), i_size_read(inode));
- rc = smb2_copychunk_range(xid, cfile, cfile, off, count, off + len);
+ /*
+ * Move [off, old_eof) right by len. The helper copies backwards if the
+ * source and destination ranges overlap.
+ */
+ rc = __smb2_copychunk_range(xid, cfile, cfile, off, old_eof - off,
+ off + len);
if (rc < 0)
goto out_2;
spin_lock(&inode->i_lock);
@@ -5242,11 +5380,13 @@ receive_encrypted_standard(struct TCP_Server_Info *server,
length = decrypt_raw_data(server, buf, buf_size, NULL, false);
if (length)
return length;
+ pdu_length = buf_size;
next_is_large = server->large_buf;
one_more:
shdr = (struct smb2_hdr *)buf;
next_cmd = le32_to_cpu(shdr->NextCommand);
+ server->total_read = next_cmd ? next_cmd : pdu_length;
if (*num_mids >= MAX_COMPOUND) {
cifs_server_dbg(VFS, "too many PDUs in compound\n");
@@ -5254,8 +5394,15 @@ one_more:
}
if (next_cmd) {
- if (WARN_ON_ONCE(next_cmd > pdu_length))
+ if (next_cmd < MID_HEADER_SIZE(server) ||
+ next_cmd > pdu_length ||
+ pdu_length - next_cmd < MID_HEADER_SIZE(server)) {
+ unsigned int max_next = pdu_length > (unsigned int)MID_HEADER_SIZE(server) ?
+ pdu_length - (unsigned int)MID_HEADER_SIZE(server) : 0;
+ cifs_server_dbg(VFS, "invalid NextCommand offset %u out of range [%zu, %u]\n",
+ next_cmd, MID_HEADER_SIZE(server), max_next);
return -1;
+ }
if (next_is_large)
next_buffer = (char *)cifs_buf_get();
else
@@ -5291,6 +5438,7 @@ one_more:
server->bigbuf = buf = next_buffer;
else
server->smallbuf = buf = next_buffer;
+ next_buffer = NULL;
goto one_more;
} else if (ret != 0) {
/*
diff --git a/fs/smb/client/smb2pdu.c b/fs/smb/client/smb2pdu.c
index dea05aeb53a1..880ce12f50c4 100644
--- a/fs/smb/client/smb2pdu.c
+++ b/fs/smb/client/smb2pdu.c
@@ -189,18 +189,19 @@ cifs_chan_skip_or_disable(struct cifs_ses *ses,
spin_unlock(&ses->chan_lock);
/*
- * the above reference of server by channel
- * needs to be dropped without holding chan_lock
- * as cifs_put_tcp_session takes a higher lock
- * i.e. cifs_tcp_ses_lock
+ * signal the channel and its primary server to
+ * reconnect before dropping the above reference of
+ * server by channel, which is done without holding
+ * chan_lock as cifs_put_tcp_session takes a higher
+ * lock i.e. cifs_tcp_ses_lock
*/
- cifs_put_tcp_session(server, from_reconnect);
-
cifs_signal_cifsd_for_reconnect(server, false);
/* mark primary server as needing reconnect */
pserver = server->primary_server;
cifs_signal_cifsd_for_reconnect(pserver, false);
+
+ cifs_put_tcp_session(server, from_reconnect);
skip_terminate:
return -EHOSTDOWN;
}
diff --git a/fs/smb/client/trace.h b/fs/smb/client/trace.h
index 12241abb8e2e..bb8d0197cb54 100644
--- a/fs/smb/client/trace.h
+++ b/fs/smb/client/trace.h
@@ -27,6 +27,7 @@
EM(smb_eio_trace_copychunk_overcopy_c, "copychunk_overcopy_c") \
EM(smb_eio_trace_create_rsp_too_small, "create_rsp_too_small") \
EM(smb_eio_trace_dfsref_no_rsp, "dfsref_no_rsp") \
+ EM(smb_eio_trace_ea_next_offset, "ea_next_offset") \
EM(smb_eio_trace_ea_overrun, "ea_overrun") \
EM(smb_eio_trace_extract_will_pin, "extract_will_pin") \
EM(smb_eio_trace_forced_shutdown, "forced_shutdown") \
@@ -79,6 +80,7 @@
EM(smb_eio_trace_qreparse_setup_count, "qreparse_setup_count") \
EM(smb_eio_trace_qreparse_sizes_wrong, "qreparse_sizes_wrong") \
EM(smb_eio_trace_qsym_bcc_too_small, "qsym_bcc_too_small") \
+ EM(smb_eio_trace_read_bad_offset, "read_bad_offset") \
EM(smb_eio_trace_read_mid_state_unknown, "read_mid_state_unknown") \
EM(smb_eio_trace_read_overlarge, "read_overlarge") \
EM(smb_eio_trace_read_rsp_malformed, "read_rsp_malformed") \
diff --git a/fs/smb/client/transport.c b/fs/smb/client/transport.c
index fdf4e50c27ce..e266859818a4 100644
--- a/fs/smb/client/transport.c
+++ b/fs/smb/client/transport.c
@@ -101,12 +101,11 @@ void __release_mid(struct TCP_Server_Info *server, struct mid_q_entry *midEntry)
trace_smb3_slow_rsp(smb_cmd, midEntry->mid, midEntry->pid,
midEntry->when_sent, midEntry->when_received);
if (cifsFYI & CIFS_TIMER) {
- pr_debug("slow rsp: cmd %d mid %llu",
- midEntry->command, midEntry->mid);
- cifs_info("A: 0x%lx S: 0x%lx R: 0x%lx\n",
- now - midEntry->when_alloc,
- now - midEntry->when_sent,
- now - midEntry->when_received);
+ pr_debug("slow rsp: cmd %d mid %llu A: 0x%lx S: 0x%lx R: 0x%lx\n",
+ midEntry->command, midEntry->mid,
+ now - midEntry->when_alloc,
+ now - midEntry->when_sent,
+ now - midEntry->when_received);
}
}
#endif
diff --git a/fs/smb/server/connection.c b/fs/smb/server/connection.c
index 91fdd1ddc61f..d211861ff86f 100644
--- a/fs/smb/server/connection.c
+++ b/fs/smb/server/connection.c
@@ -13,6 +13,7 @@
#include "mgmt/ksmbd_ida.h"
#include "mgmt/user_session.h"
#include "connection.h"
+#include "vfs_cache.h"
#include "compress.h"
#include "transport_tcp.h"
#include "transport_rdma.h"
@@ -21,6 +22,8 @@
static DEFINE_MUTEX(init_lock);
static struct ksmbd_conn_ops default_conn_ops;
+static struct delayed_work session_expiration_work;
+static bool stopping_session_expiration_work;
DEFINE_HASHTABLE(conn_list, CONN_HASH_BITS);
DECLARE_RWSEM(conn_list_lock);
@@ -157,18 +160,28 @@ static void delete_proc_clients(void) {}
static struct workqueue_struct *ksmbd_conn_wq;
+static void ksmbd_session_expiration_worker(struct work_struct *work);
+
int ksmbd_conn_wq_init(void)
{
ksmbd_conn_wq = alloc_workqueue("ksmbd-conn-release",
WQ_UNBOUND | WQ_MEM_RECLAIM, 0);
if (!ksmbd_conn_wq)
return -ENOMEM;
+
+ WRITE_ONCE(stopping_session_expiration_work, false);
+ INIT_DELAYED_WORK(&session_expiration_work,
+ ksmbd_session_expiration_worker);
+ queue_delayed_work(ksmbd_conn_wq, &session_expiration_work,
+ KSMBD_SESSION_EXPIRATION_INTERVAL);
return 0;
}
void ksmbd_conn_wq_destroy(void)
{
if (ksmbd_conn_wq) {
+ WRITE_ONCE(stopping_session_expiration_work, true);
+ cancel_delayed_work_sync(&session_expiration_work);
destroy_workqueue(ksmbd_conn_wq);
ksmbd_conn_wq = NULL;
}
@@ -278,6 +291,7 @@ struct ksmbd_conn *ksmbd_conn_alloc(void)
return NULL;
conn->need_neg = true;
+ conn->creation_time = jiffies;
ksmbd_conn_set_new(conn);
conn->local_nls = load_nls("utf8");
if (!conn->local_nls)
@@ -384,12 +398,12 @@ static void ksmbd_conn_cancel_async_requests(struct ksmbd_conn *conn)
spin_lock(&conn->request_lock);
list_for_each_entry_safe(work, tmp, &conn->async_requests,
async_request_entry) {
- if (work->state != KSMBD_WORK_ACTIVE)
+ if (cmpxchg(&work->state, KSMBD_WORK_ACTIVE,
+ KSMBD_WORK_CANCELLED) != KSMBD_WORK_ACTIVE)
continue;
ksmbd_debug(CONN, "Cancel async request id %d\n",
work->async_id);
- work->state = KSMBD_WORK_CANCELLED;
if (work->cancel_fn)
work->cancel_fn(work->cancel_argv);
}
@@ -473,6 +487,9 @@ retry_idle:
if (retry_count >= max_timeout)
return -EIO;
+ /* A blocked byte-range lock cannot drain until teardown wakes it. */
+ ksmbd_wake_session_blocked_works(sess);
+
down_read(&conn_list_lock);
hash_for_each(conn_list, bkt, conn, hlist) {
if (ksmbd_session_is_bound_to_conn(sess, conn)) {
@@ -584,6 +601,20 @@ bool ksmbd_conn_alive(struct ksmbd_conn *conn)
if (kthread_should_stop())
return false;
+ /*
+ * Stale connections that have not completed NEGOTIATE and SESSION_SETUP
+ * must be disconnected. Do not race a request that is currently
+ * completing authentication.
+ */
+ if (!atomic_read(&conn->req_running) &&
+ time_after(jiffies, conn->creation_time +
+ KSMBD_UNAUTHENTICATED_CONN_TIMEOUT) &&
+ (READ_ONCE(conn->need_neg) ||
+ !ksmbd_conn_has_valid_or_expired_session(conn))) {
+ ksmbd_debug(CONN, "Connection setup timed out\n");
+ return false;
+ }
+
if (atomic_read(&conn->stats.open_files_count) > 0)
return true;
@@ -601,6 +632,51 @@ bool ksmbd_conn_alive(struct ksmbd_conn *conn)
return true;
}
+static void ksmbd_session_expiration_worker(struct work_struct *work)
+{
+ struct ksmbd_conn *conn, *target;
+ int bkt;
+
+ if (!ksmbd_server_running())
+ goto reschedule;
+
+ ksmbd_expire_sessions();
+
+ /*
+ * An old connection without a Valid or Expired session must be
+ * disconnected. Process one connection at a time without holding
+ * conn_list_lock across transport shutdown.
+ */
+again:
+ target = NULL;
+ down_read(&conn_list_lock);
+ hash_for_each(conn_list, bkt, conn, hlist) {
+ if (ksmbd_conn_exiting(conn) || ksmbd_conn_releasing(conn) ||
+ atomic_read(&conn->req_running) ||
+ time_before_eq(jiffies, conn->creation_time +
+ KSMBD_UNAUTHENTICATED_CONN_TIMEOUT) ||
+ (!READ_ONCE(conn->need_neg) &&
+ ksmbd_conn_has_valid_or_expired_session(conn)))
+ continue;
+
+ target = ksmbd_conn_get(conn);
+ break;
+ }
+ up_read(&conn_list_lock);
+
+ if (target) {
+ ksmbd_debug(CONN, "Connection setup timed out\n");
+ ksmbd_conn_abort(target);
+ ksmbd_conn_put(target);
+ goto again;
+ }
+
+reschedule:
+ if (!READ_ONCE(stopping_session_expiration_work))
+ queue_delayed_work(ksmbd_conn_wq, &session_expiration_work,
+ KSMBD_SESSION_EXPIRATION_INTERVAL);
+}
+
/* "+2" for BCC field (ByteCount, 2 bytes) */
#define SMB1_MIN_SUPPORTED_PDU_SIZE (sizeof(struct smb_hdr) + 2)
#define SMB2_MIN_SUPPORTED_PDU_SIZE (sizeof(struct smb2_pdu))
diff --git a/fs/smb/server/connection.h b/fs/smb/server/connection.h
index 63484c8efbbd..371f17b4f02a 100644
--- a/fs/smb/server/connection.h
+++ b/fs/smb/server/connection.h
@@ -77,6 +77,7 @@ struct ksmbd_conn {
struct rw_semaphore session_lock;
/* smb session 1 per user */
struct xarray sessions;
+ unsigned long creation_time;
unsigned long last_active;
/* How many request are running currently */
atomic_t req_running;
@@ -192,6 +193,8 @@ struct ksmbd_transport {
#define KSMBD_TCP_RECV_TIMEOUT (7 * HZ)
#define KSMBD_TCP_SEND_TIMEOUT (5 * HZ)
+#define KSMBD_SESSION_EXPIRATION_INTERVAL (5 * HZ)
+#define KSMBD_UNAUTHENTICATED_CONN_TIMEOUT (45 * HZ)
#define KSMBD_TCP_PEER_SOCKADDR(c) ((struct sockaddr *)&((c)->peer_addr))
#define CONN_HASH_BITS 12
diff --git a/fs/smb/server/ksmbd_work.c b/fs/smb/server/ksmbd_work.c
index f35335307670..d307aefe0aec 100644
--- a/fs/smb/server/ksmbd_work.c
+++ b/fs/smb/server/ksmbd_work.c
@@ -30,7 +30,7 @@ static int ksmbd_reserve_iov(struct ksmbd_work *work, int need_iov_cnt)
} while (new_alloc_cnt < work->iov_cnt + need_iov_cnt);
if (work->iov == work->iov_inline) {
- new = kcalloc(new_alloc_cnt, sizeof(*new), KSMBD_DEFAULT_GFP);
+ new = kzalloc_objs(*new, new_alloc_cnt, KSMBD_DEFAULT_GFP);
if (!new)
return -ENOMEM;
diff --git a/fs/smb/server/ksmbd_work.h b/fs/smb/server/ksmbd_work.h
index 5f1d3ebab4fb..0844aa929f55 100644
--- a/fs/smb/server/ksmbd_work.h
+++ b/fs/smb/server/ksmbd_work.h
@@ -82,7 +82,7 @@ struct ksmbd_work {
/* Contiguous SMB2 compression transform owned by this work item. */
void *compress_buf;
- unsigned char state;
+ unsigned int state;
/* No response for cancelled request */
bool send_no_response:1;
/* Request is encrypted */
diff --git a/fs/smb/server/mgmt/share_config.c b/fs/smb/server/mgmt/share_config.c
index b2d9580bddc6..cc9f18ede80d 100644
--- a/fs/smb/server/mgmt/share_config.c
+++ b/fs/smb/server/mgmt/share_config.c
@@ -146,9 +146,9 @@ static struct ksmbd_share_config *__share_lookup(const char *name)
static int parse_veto_list(struct ksmbd_share_config *share,
char *veto_list,
- int veto_list_sz)
+ size_t veto_list_sz)
{
- int sz = 0;
+ size_t sz;
if (!veto_list_sz)
return 0;
@@ -156,7 +156,7 @@ static int parse_veto_list(struct ksmbd_share_config *share,
while (veto_list_sz > 0) {
struct ksmbd_veto_pattern *p;
- sz = strlen(veto_list);
+ sz = strnlen(veto_list, veto_list_sz);
if (!sz)
break;
@@ -164,7 +164,7 @@ static int parse_veto_list(struct ksmbd_share_config *share,
if (!p)
return -ENOMEM;
- p->pattern = kstrdup(veto_list, KSMBD_DEFAULT_GFP);
+ p->pattern = kstrndup(veto_list, sz, KSMBD_DEFAULT_GFP);
if (!p->pattern) {
kfree(p);
return -ENOMEM;
@@ -172,6 +172,9 @@ static int parse_veto_list(struct ksmbd_share_config *share,
list_add(&p->list, &share->veto_list);
+ if (sz == veto_list_sz)
+ break;
+
veto_list += sz + 1;
veto_list_sz -= (sz + 1);
}
@@ -224,17 +227,28 @@ static struct ksmbd_share_config *share_config_request(struct ksmbd_work *work,
}
if (!test_share_config_flag(share, KSMBD_SHARE_FLAG_PIPE)) {
- int path_len = PATH_MAX;
-
- if (resp->payload_sz)
- path_len = resp->payload_sz - resp->veto_list_sz;
+ size_t path_len;
- share->path = kstrndup(ksmbd_share_config_path(resp), path_len,
- KSMBD_DEFAULT_GFP);
- if (!share->path) {
- ret = -ENOMEM;
+ if (resp->payload_sz <= resp->veto_list_sz) {
+ ret = -EINVAL;
} else {
- ret = 0;
+ path_len = resp->payload_sz - resp->veto_list_sz;
+ if (resp->veto_list_sz)
+ path_len--;
+
+ if (!path_len) {
+ ret = -EINVAL;
+ } else {
+ share->path = kstrndup(
+ ksmbd_share_config_path(resp),
+ path_len, KSMBD_DEFAULT_GFP);
+ if (!share->path)
+ ret = -ENOMEM;
+ else
+ ret = 0;
+ }
+ }
+ if (share->path) {
share->path_sz = strlen(share->path);
while (share->path_sz > 1 &&
share->path[share->path_sz - 1] == '/')
diff --git a/fs/smb/server/mgmt/tree_connect.c b/fs/smb/server/mgmt/tree_connect.c
index 5f63e236267a..dd1db3554cae 100644
--- a/fs/smb/server/mgmt/tree_connect.c
+++ b/fs/smb/server/mgmt/tree_connect.c
@@ -82,6 +82,8 @@ ksmbd_tree_conn_connect(struct ksmbd_work *work, const char *share_name)
down_write(&sess->tree_conns_lock);
ret = xa_err(xa_store(&sess->tree_conns, tree_conn->id, tree_conn,
KSMBD_DEFAULT_GFP));
+ if (!ret)
+ atomic_inc(&tree_conn->refcount);
up_write(&sess->tree_conns_lock);
if (ret) {
status.ret = -ENOMEM;
@@ -129,6 +131,12 @@ int ksmbd_tree_conn_disconnect(struct ksmbd_session *sess,
struct ksmbd_tree_connect *tree_conn)
{
down_write(&sess->tree_conns_lock);
+ if (tree_conn->t_state == TREE_DISCONNECTED ||
+ xa_load(&sess->tree_conns, tree_conn->id) != tree_conn) {
+ up_write(&sess->tree_conns_lock);
+ return -ENOENT;
+ }
+ tree_conn->t_state = TREE_DISCONNECTED;
xa_erase(&sess->tree_conns, tree_conn->id);
up_write(&sess->tree_conns_lock);
diff --git a/fs/smb/server/mgmt/user_session.c b/fs/smb/server/mgmt/user_session.c
index 7022d5d656b4..44dc3f800cd4 100644
--- a/fs/smb/server/mgmt/user_session.c
+++ b/fs/smb/server/mgmt/user_session.c
@@ -22,6 +22,7 @@
static DEFINE_IDA(session_ida);
#define SESSION_HASH_BITS 12
+#define KSMBD_MAX_PENDING_SESSIONS 1
static DEFINE_HASHTABLE(sessions_table, SESSION_HASH_BITS);
static DECLARE_RWSEM(sessions_table_lock);
@@ -432,26 +433,31 @@ struct ksmbd_session *__session_lookup(unsigned long long id)
return NULL;
}
-static void ksmbd_expire_session(struct ksmbd_conn *conn)
+static bool ksmbd_too_many_session_setups(struct ksmbd_conn *conn)
{
unsigned long id;
struct ksmbd_session *sess;
+ unsigned int pending = 0;
down_write(&sessions_table_lock);
down_write(&conn->session_lock);
xa_for_each(&conn->sessions, id, sess) {
+ if (READ_ONCE(sess->state) != SMB2_SESSION_IN_PROGRESS)
+ continue;
+
if (atomic_read(&sess->refcnt) <= 1 &&
- (sess->state != SMB2_SESSION_VALID ||
- time_after(jiffies,
- sess->last_active + SMB2_SESSION_TIMEOUT))) {
+ time_after(jiffies, sess->last_active +
+ KSMBD_UNAUTHENTICATED_CONN_TIMEOUT)) {
xa_erase(&conn->sessions, sess->id);
ksmbd_session_remove_from_table(sess);
ksmbd_session_destroy(sess);
continue;
}
+ pending++;
}
up_write(&conn->session_lock);
up_write(&sessions_table_lock);
+ return pending >= KSMBD_MAX_PENDING_SESSIONS;
}
int ksmbd_session_register(struct ksmbd_conn *conn,
@@ -461,9 +467,12 @@ int ksmbd_session_register(struct ksmbd_conn *conn,
sess->dialect = conn->dialect;
memcpy(sess->ClientGUID, conn->ClientGUID, SMB2_CLIENT_GUID_SIZE);
- ksmbd_expire_session(conn);
- ret = xa_err(xa_store(&conn->sessions, sess->id, sess,
- KSMBD_DEFAULT_GFP));
+ /* Bound abandoned SessionId-zero authentication exchanges. */
+ if (ksmbd_too_many_session_setups(conn))
+ ret = -ENOSPC;
+ else
+ ret = xa_err(xa_store(&conn->sessions, sess->id, sess,
+ KSMBD_DEFAULT_GFP));
if (ret) {
down_write(&sessions_table_lock);
ksmbd_session_remove_from_table(sess);
@@ -474,6 +483,105 @@ int ksmbd_session_register(struct ksmbd_conn *conn,
return ret;
}
+void ksmbd_session_unregister(struct ksmbd_conn *conn,
+ struct ksmbd_session *sess)
+{
+ struct ksmbd_conn *session_conns[KSMBD_MAX_CHANNELS];
+ struct channel *chann;
+ unsigned long index;
+ unsigned int nr_conns = 0, i;
+ bool removed = false;
+
+ down_write(&sessions_table_lock);
+ if (!hlist_unhashed(&sess->hlist)) {
+ /* Keep each channel connection stable under sessions_table_lock. */
+ down_read(&sess->chann_lock);
+ xa_for_each(&sess->ksmbd_chann_list, index, chann) {
+ if (nr_conns == ARRAY_SIZE(session_conns))
+ break;
+ session_conns[nr_conns++] = chann->conn;
+ }
+ up_read(&sess->chann_lock);
+
+ ksmbd_session_remove_from_table(sess);
+ removed = true;
+ }
+
+ down_write(&conn->session_lock);
+ if (xa_load(&conn->sessions, sess->id) == sess)
+ xa_erase(&conn->sessions, sess->id);
+ up_write(&conn->session_lock);
+ for (i = 0; i < nr_conns; i++) {
+ if (session_conns[i] == conn)
+ continue;
+ down_write(&session_conns[i]->session_lock);
+ if (xa_load(&session_conns[i]->sessions, sess->id) == sess)
+ xa_erase(&session_conns[i]->sessions, sess->id);
+ up_write(&session_conns[i]->session_lock);
+ }
+ up_write(&sessions_table_lock);
+
+ if (removed)
+ ksmbd_user_session_put(sess);
+}
+
+bool ksmbd_conn_has_valid_or_expired_session(struct ksmbd_conn *conn)
+{
+ struct ksmbd_session *sess;
+ unsigned long id;
+ int state, bkt;
+ bool found = false;
+
+ down_read(&conn->session_lock);
+ xa_for_each(&conn->sessions, id, sess) {
+ state = READ_ONCE(sess->state);
+ if (state == SMB2_SESSION_VALID ||
+ state == SMB2_SESSION_EXPIRED) {
+ found = true;
+ break;
+ }
+ }
+ up_read(&conn->session_lock);
+ if (found)
+ return true;
+
+ /* A session bound through SMB3 multichannel is not in conn->sessions. */
+ down_read(&sessions_table_lock);
+ hash_for_each(sessions_table, bkt, sess, hlist) {
+ state = READ_ONCE(sess->state);
+ if (state != SMB2_SESSION_VALID &&
+ state != SMB2_SESSION_EXPIRED)
+ continue;
+
+ down_read(&sess->chann_lock);
+ found = xa_load(&sess->ksmbd_chann_list, (long)conn);
+ up_read(&sess->chann_lock);
+ if (found)
+ break;
+ }
+ up_read(&sessions_table_lock);
+ return found;
+}
+
+void ksmbd_expire_sessions(void)
+{
+ struct ksmbd_session *sess;
+ u64 now = ktime_get_real_seconds();
+ int bkt;
+
+ down_read(&sessions_table_lock);
+ hash_for_each(sessions_table, bkt, sess, hlist) {
+ if (READ_ONCE(sess->state) != SMB2_SESSION_VALID ||
+ !sess->kerberos_expiry || now < sess->kerberos_expiry)
+ continue;
+
+ if (cmpxchg(&sess->state, SMB2_SESSION_VALID,
+ SMB2_SESSION_EXPIRED) == SMB2_SESSION_VALID)
+ ksmbd_counter_inc(KSMBD_COUNTER_SESSION_TIMEOUTS);
+ }
+ up_read(&sessions_table_lock);
+}
+
static int ksmbd_chann_del(struct ksmbd_conn *conn, struct ksmbd_session *sess)
{
struct channel *chann;
@@ -488,7 +596,7 @@ static int ksmbd_chann_del(struct ksmbd_conn *conn, struct ksmbd_session *sess)
return 0;
}
-void ksmbd_sessions_deregister(struct ksmbd_conn *conn)
+void ksmbd_conn_sessions_cleanup(struct ksmbd_conn *conn)
{
struct ksmbd_session *sess;
unsigned long id;
@@ -666,10 +774,21 @@ void destroy_previous_session(struct ksmbd_conn *conn,
memcmp(user->passkey, prev_user->passkey, user->passkey_sz))
goto out;
+ down_write(&prev_sess->chann_lock);
+ if (prev_sess->tearing_down) {
+ up_write(&prev_sess->chann_lock);
+ goto out;
+ }
+ prev_sess->tearing_down = true;
+ up_write(&prev_sess->chann_lock);
+
ksmbd_all_conn_set_status(prev_sess, KSMBD_SESS_NEED_RECONNECT);
err = ksmbd_conn_wait_idle_sess(conn, prev_sess);
if (err) {
- ksmbd_all_conn_set_status(prev_sess, KSMBD_SESS_NEED_SETUP);
+ down_write(&prev_sess->chann_lock);
+ prev_sess->tearing_down = false;
+ up_write(&prev_sess->chann_lock);
+ ksmbd_all_conn_set_status(prev_sess, KSMBD_SESS_GOOD);
goto out;
}
diff --git a/fs/smb/server/mgmt/user_session.h b/fs/smb/server/mgmt/user_session.h
index f8a24c33f7fe..217258551d6d 100644
--- a/fs/smb/server/mgmt/user_session.h
+++ b/fs/smb/server/mgmt/user_session.h
@@ -42,6 +42,7 @@ struct ksmbd_session {
bool sign;
bool enc;
+ bool tearing_down;
int state;
__u8 *Preauth_HashValue;
@@ -71,6 +72,8 @@ struct ksmbd_session {
struct rw_semaphore rpc_lock;
};
+#define KSMBD_MAX_CHANNELS 32
+
static inline int test_session_flag(struct ksmbd_session *sess, int bit)
{
return sess->flags & bit;
@@ -97,7 +100,11 @@ bool is_ksmbd_session_in_connection(struct ksmbd_conn *conn,
unsigned long long id);
int ksmbd_session_register(struct ksmbd_conn *conn,
struct ksmbd_session *sess);
-void ksmbd_sessions_deregister(struct ksmbd_conn *conn);
+void ksmbd_session_unregister(struct ksmbd_conn *conn,
+ struct ksmbd_session *sess);
+void ksmbd_conn_sessions_cleanup(struct ksmbd_conn *conn);
+bool ksmbd_conn_has_valid_or_expired_session(struct ksmbd_conn *conn);
+void ksmbd_expire_sessions(void);
struct ksmbd_session *__session_lookup(unsigned long long id);
struct ksmbd_session *ksmbd_session_lookup_all(struct ksmbd_conn *conn,
unsigned long long id);
diff --git a/fs/smb/server/oplock.c b/fs/smb/server/oplock.c
index 58af0fddf39f..1b8c3482d1e4 100644
--- a/fs/smb/server/oplock.c
+++ b/fs/smb/server/oplock.c
@@ -924,31 +924,69 @@ out:
ksmbd_conn_put(conn);
}
+/*
+ * Select and pin the connection used for an oplock break before doing any
+ * allocations which may sleep. The caller of oplock_break() holds a live
+ * reference on ci (a file being opened, a file being operated on, or an
+ * explicit ksmbd_inode_lookup_lock() reference in the parent lease break
+ * paths), so the inode cannot be freed during the call and its lock is
+ * reachable without dereferencing opinfo->o_fp, which is not pinned by
+ * the oplock reference and may be freed by a concurrent close.
+ *
+ * opinfo->conn is cleared under ci->m_lock by session_fd_check() when the
+ * durable handle owning the oplock is disconnected, reassigned by
+ * ksmbd_reopen_durable_fd() under the same lock, and the last
+ * ksmbd_conn_put() of the old connection frees it. Holding the read lock
+ * excludes both writers, so the connection cannot be freed while it is
+ * selected.
+ */
+static struct ksmbd_conn *smb2_oplock_break_conn_get(struct oplock_info *opinfo,
+ struct ksmbd_inode *ci)
+{
+ struct ksmbd_conn *conn;
+
+ down_read(&ci->m_lock);
+ conn = READ_ONCE(opinfo->conn);
+ if (conn && !ksmbd_conn_releasing(conn))
+ conn = ksmbd_conn_get(conn);
+ else
+ conn = NULL;
+ up_read(&ci->m_lock);
+
+ return conn;
+}
+
/**
* smb2_oplock_break_noti() - send smb2 exclusive/batch to level2 oplock
* break command from server to client
* @opinfo: oplock info object
+ * @ci: inode owning the break target's oplock list, pinned by
+ * the caller
*
* Return: 0 on success, otherwise error
*/
-static int smb2_oplock_break_noti(struct oplock_info *opinfo)
+static int smb2_oplock_break_noti(struct oplock_info *opinfo,
+ struct ksmbd_inode *ci)
{
struct ksmbd_conn *conn;
struct oplock_break_info *br_info;
int ret = 0;
struct ksmbd_work *work;
- conn = READ_ONCE(opinfo->conn);
+ conn = smb2_oplock_break_conn_get(opinfo, ci);
if (!conn)
return ksmbd_invalidate_durable_fd(opinfo->fid);
work = ksmbd_alloc_work_struct();
- if (!work)
+ if (!work) {
+ ksmbd_conn_put(conn);
return -ENOMEM;
+ }
br_info = kmalloc_obj(struct oplock_break_info, KSMBD_DEFAULT_GFP);
if (!br_info) {
ksmbd_free_work_struct(work);
+ ksmbd_conn_put(conn);
return -ENOMEM;
}
@@ -957,7 +995,8 @@ static int smb2_oplock_break_noti(struct oplock_info *opinfo)
br_info->open_trunc = opinfo->open_trunc;
work->request_buf = (char *)br_info;
- work->conn = ksmbd_conn_get(conn);
+ /* Transfer the reference acquired by smb2_oplock_break_conn_get(). */
+ work->conn = conn;
work->sess = opinfo->sess;
ksmbd_conn_r_count_inc(conn);
@@ -1154,9 +1193,9 @@ static void wait_lease_breaking(struct oplock_info *opinfo)
}
}
-static int oplock_break(struct oplock_info *brk_opinfo, int req_op_level,
- struct ksmbd_work *in_work, bool share_break,
- bool sync_lease_break)
+static int oplock_break(struct oplock_info *brk_opinfo, struct ksmbd_inode *ci,
+ int req_op_level, struct ksmbd_work *in_work,
+ bool share_break, bool sync_lease_break)
{
int err = 0;
bool sent_interim = false;
@@ -1298,7 +1337,7 @@ again:
}
}
- err = smb2_oplock_break_noti(brk_opinfo);
+ err = smb2_oplock_break_noti(brk_opinfo, ci);
ksmbd_debug(OPLOCK, "oplock granted = %d\n", brk_opinfo->level);
if (brk_opinfo->op_state == OPLOCK_CLOSING)
@@ -1326,13 +1365,14 @@ static int oplock_break_add(struct list_head *head, struct oplock_info *opinfo)
return 0;
}
-static void oplock_break_drain_none(struct list_head *head)
+static void oplock_break_drain_none(struct list_head *head,
+ struct ksmbd_inode *ci)
{
struct oplock_break_entry *ent, *tmp;
list_for_each_entry_safe(ent, tmp, head, list) {
- oplock_break(ent->opinfo, SMB2_OPLOCK_LEVEL_NONE, NULL, false,
- false);
+ oplock_break(ent->opinfo, ci, SMB2_OPLOCK_LEVEL_NONE, NULL,
+ false, false);
list_del(&ent->list);
opinfo_put(ent->opinfo);
kfree(ent);
@@ -1481,7 +1521,7 @@ void smb_send_parent_lease_break_noti(struct ksmbd_file *fp,
}
up_read(&p_ci->m_lock);
- oplock_break_drain_none(&brk_list);
+ oplock_break_drain_none(&brk_list, p_ci);
ksmbd_inode_put(p_ci);
}
@@ -1525,7 +1565,7 @@ void smb_lazy_parent_lease_break_close(struct ksmbd_file *fp)
}
up_read(&p_ci->m_lock);
- oplock_break_drain_none(&brk_list);
+ oplock_break_drain_none(&brk_list, p_ci);
ksmbd_inode_put(p_ci);
}
@@ -1665,7 +1705,7 @@ int smb_grant_oplock(struct ksmbd_work *work, int req_op_level, u64 pid,
prev_durable_detached = prev_op_snapshot.durable_detached;
prev_fid = prev_op_snapshot.fid;
- err = oplock_break(prev_opinfo, break_level, work,
+ err = oplock_break(prev_opinfo, ci, break_level, work,
share_ret < 0 && prev_opinfo->is_lease, false);
if (prev_durable_detached || (prev_durable_open && err == -ENOENT))
ksmbd_invalidate_durable_fd(prev_fid);
@@ -1771,7 +1811,8 @@ static bool smb_break_all_write_oplock(struct ksmbd_work *work,
}
brk_opinfo->open_trunc = is_trunc;
- oplock_break(brk_opinfo, SMB2_OPLOCK_LEVEL_II, work, false, false);
+ oplock_break(brk_opinfo, fp->f_ci, SMB2_OPLOCK_LEVEL_II, work, false,
+ false);
sent_break = true;
opinfo_put(brk_opinfo);
@@ -1863,7 +1904,7 @@ next:
brk_op->op_state = OPLOCK_STATE_NONE;
spin_unlock(&brk_op->state_lock);
} else {
- oplock_break(brk_op,
+ oplock_break(brk_op, ci,
brk_op->is_lease && !is_trunc ?
SMB2_OPLOCK_LEVEL_II : SMB2_OPLOCK_LEVEL_NONE,
send_interim && !sent_interim ? work : NULL,
diff --git a/fs/smb/server/proc.c b/fs/smb/server/proc.c
index 826353ed0553..19f0f2cfbf54 100644
--- a/fs/smb/server/proc.c
+++ b/fs/smb/server/proc.c
@@ -178,6 +178,8 @@ static int proc_show_ksmbd_stats(struct seq_file *m, void *v)
proc_show_runtime_totals(m);
seq_printf(m, "sessions:\t%lld\n",
ksmbd_counter_sum(KSMBD_COUNTER_SESSIONS));
+ seq_printf(m, "session_timeouts:\t%lld\n",
+ ksmbd_counter_sum(KSMBD_COUNTER_SESSION_TIMEOUTS));
seq_printf(m, "tree_connects:\t%lld\n",
ksmbd_counter_sum(KSMBD_COUNTER_TREE_CONNS));
seq_printf(m, "requests:\t%lld\n",
diff --git a/fs/smb/server/server.c b/fs/smb/server/server.c
index 0069d4e6a60a..0827c8c51006 100644
--- a/fs/smb/server/server.c
+++ b/fs/smb/server/server.c
@@ -414,7 +414,7 @@ static int ksmbd_server_process_request(struct ksmbd_conn *conn)
static int ksmbd_server_terminate_conn(struct ksmbd_conn *conn)
{
- ksmbd_sessions_deregister(conn);
+ ksmbd_conn_sessions_cleanup(conn);
destroy_lease_table(conn);
return 0;
}
diff --git a/fs/smb/server/smb2pdu.c b/fs/smb/server/smb2pdu.c
index a8046f477d54..6b8809f67b92 100644
--- a/fs/smb/server/smb2pdu.c
+++ b/fs/smb/server/smb2pdu.c
@@ -85,8 +85,6 @@ struct channel *lookup_chann_list(struct ksmbd_session *sess, struct ksmbd_conn
return chann;
}
-#define KSMBD_MAX_CHANNELS 32
-
static int register_session_channel(struct ksmbd_session *sess,
struct ksmbd_conn *conn,
const char *sess_key)
@@ -97,6 +95,11 @@ static int register_session_channel(struct ksmbd_session *sess,
int rc = 0;
down_write(&sess->chann_lock);
+ if (sess->tearing_down) {
+ rc = -ESHUTDOWN;
+ goto out;
+ }
+
if (xa_load(&sess->ksmbd_chann_list, (long)conn))
goto out;
@@ -873,7 +876,8 @@ int smb2_allocate_rsp_buf(struct ksmbd_work *work)
req = smb_get_msg(work->request_buf);
if ((req->InfoType == SMB2_O_INFO_FILE &&
(req->FileInfoClass == FILE_FULL_EA_INFORMATION ||
- req->FileInfoClass == FILE_ALL_INFORMATION)) ||
+ req->FileInfoClass == FILE_ALL_INFORMATION ||
+ req->FileInfoClass == FILE_NORMALIZED_NAME_INFORMATION)) ||
req->InfoType == SMB2_O_INFO_SECURITY)
sz = large_sz;
}
@@ -927,8 +931,14 @@ static bool smb2_session_expired_cmd_allowed(struct ksmbd_work *work,
static bool smb2_session_kerberos_expired(struct ksmbd_session *sess)
{
- return sess->kerberos_expiry &&
- ktime_get_real_seconds() >= sess->kerberos_expiry;
+ if (!sess->kerberos_expiry ||
+ ktime_get_real_seconds() < sess->kerberos_expiry)
+ return false;
+
+ if (cmpxchg(&sess->state, SMB2_SESSION_VALID,
+ SMB2_SESSION_EXPIRED) == SMB2_SESSION_VALID)
+ ksmbd_counter_inc(KSMBD_COUNTER_SESSION_TIMEOUTS);
+ return true;
}
/**
@@ -963,9 +973,8 @@ int smb2_check_user_session(struct ksmbd_work *work)
if (!work->next_smb2_rcv_hdr_off && sess_id)
work->sess = ksmbd_session_lookup_all_states(conn, sess_id);
if (work->sess) {
- if (smb2_session_kerberos_expired(work->sess)) {
- work->sess->state = SMB2_SESSION_EXPIRED;
- } else if (work->sess->state != SMB2_SESSION_VALID) {
+ if (!smb2_session_kerberos_expired(work->sess) &&
+ work->sess->state != SMB2_SESSION_VALID) {
ksmbd_user_session_put(work->sess);
work->sess = NULL;
}
@@ -990,8 +999,7 @@ int smb2_check_user_session(struct ksmbd_work *work)
sess_id, work->sess->id);
return -EINVAL;
}
- if (smb2_session_kerberos_expired(work->sess))
- work->sess->state = SMB2_SESSION_EXPIRED;
+ smb2_session_kerberos_expired(work->sess);
if (work->sess->state != SMB2_SESSION_VALID) {
pr_err("compound request on a non-valid session (state %d)\n",
work->sess->state);
@@ -1008,7 +1016,6 @@ int smb2_check_user_session(struct ksmbd_work *work)
work->sess = ksmbd_session_lookup_all_states(conn, sess_id);
if (work->sess) {
if (smb2_session_kerberos_expired(work->sess)) {
- work->sess->state = SMB2_SESSION_EXPIRED;
return smb2_session_expired_cmd_allowed(work, cmd) ?
1 : -EKEYEXPIRED;
}
@@ -2430,7 +2437,7 @@ int smb2_sess_setup(struct ksmbd_work *work)
struct ksmbd_conn *conn = work->conn;
struct smb2_sess_setup_req *req;
struct smb2_sess_setup_rsp *rsp;
- struct ksmbd_session *sess;
+ struct ksmbd_session *sess = NULL;
struct negotiate_message *negblob;
unsigned int negblob_len, negblob_off;
int rc = 0;
@@ -2586,6 +2593,9 @@ int smb2_sess_setup(struct ksmbd_work *work)
goto out_err;
}
+ if (work->session_setup_reauth)
+ WRITE_ONCE(sess->state, SMB2_SESSION_IN_PROGRESS);
+
conn->binding = false;
}
work->sess = sess;
@@ -2697,6 +2707,14 @@ out_err:
}
if (rc < 0) {
+ bool setup_in_progress = sess &&
+ READ_ONCE(sess->state) == SMB2_SESSION_IN_PROGRESS &&
+ !(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING);
+
+ /* Authentication errors must not leave the new session published. */
+ if (setup_in_progress)
+ ksmbd_session_unregister(conn, sess);
+
if (sess && conn->dialect == SMB311_PROT_ID &&
(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING)) {
struct preauth_session *preauth_sess;
@@ -2730,7 +2748,8 @@ out_err:
* For binding requests, session belongs to another
* connection. Do not expire it.
*/
- if (!(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING)) {
+ if (!(req->Flags & SMB2_SESSION_REQ_FLAG_BINDING) &&
+ !setup_in_progress) {
sess->last_active = jiffies;
sess->kerberos_expiry = 0;
sess->state = SMB2_SESSION_EXPIRED;
@@ -2784,6 +2803,7 @@ int smb2_tree_connect(struct ksmbd_work *work)
struct ksmbd_session *sess = work->sess;
char *treename = NULL, *name = NULL;
struct ksmbd_tree_conn_status status;
+ struct ksmbd_tree_connect *tree_conn = NULL;
struct ksmbd_share_config *share = NULL;
int rc = -EINVAL;
@@ -2811,6 +2831,7 @@ int smb2_tree_connect(struct ksmbd_work *work)
status = ksmbd_tree_conn_connect(work, name);
if (status.ret == KSMBD_TREE_CONN_STATUS_OK) {
+ tree_conn = status.tree_conn;
rsp->hdr.Id.SyncId.TreeId = cpu_to_le32(status.tree_conn->id);
share = status.tree_conn->share_conf;
@@ -2854,8 +2875,15 @@ int smb2_tree_connect(struct ksmbd_work *work)
status.tree_conn->posix_extensions = true;
down_write(&sess->tree_conns_lock);
- status.tree_conn->t_state = TREE_CONNECTED;
+ if (status.tree_conn->t_state == TREE_DISCONNECTED) {
+ status.ret = KSMBD_TREE_CONN_STATUS_ERROR;
+ share = NULL;
+ } else {
+ status.tree_conn->t_state = TREE_CONNECTED;
+ }
up_write(&sess->tree_conns_lock);
+ if (status.ret != KSMBD_TREE_CONN_STATUS_OK)
+ goto out_err1;
rsp->StructureSize = cpu_to_le16(16);
out_err1:
/*
@@ -2882,9 +2910,6 @@ out_err1:
rc = ksmbd_iov_pin_rsp(work, rsp, sizeof(struct smb2_tree_connect_rsp));
if (rc) {
if (status.ret == KSMBD_TREE_CONN_STATUS_OK) {
- down_write(&sess->tree_conns_lock);
- status.tree_conn->t_state = TREE_DISCONNECTED;
- up_write(&sess->tree_conns_lock);
ksmbd_tree_conn_disconnect(sess, status.tree_conn);
status.tree_conn = NULL;
}
@@ -2925,6 +2950,9 @@ out_err1:
if (status.ret != KSMBD_TREE_CONN_STATUS_OK)
smb2_set_err_rsp(work);
+ if (tree_conn)
+ ksmbd_tree_connect_put(tree_conn);
+
return rc;
}
@@ -3028,17 +3056,6 @@ int smb2_tree_disconnect(struct ksmbd_work *work)
ksmbd_close_tree_conn_fds(work);
- down_write(&sess->tree_conns_lock);
- if (tcon->t_state == TREE_DISCONNECTED) {
- up_write(&sess->tree_conns_lock);
- rsp->hdr.Status = STATUS_NETWORK_NAME_DELETED;
- err = -ENOENT;
- goto err_out;
- }
-
- tcon->t_state = TREE_DISCONNECTED;
- up_write(&sess->tree_conns_lock);
-
err = ksmbd_tree_conn_disconnect(sess, tcon);
if (err) {
rsp->hdr.Status = STATUS_NETWORK_NAME_DELETED;
@@ -3086,17 +3103,41 @@ int smb2_session_logoff(struct ksmbd_work *work)
smb2_set_err_rsp(work);
return -ENOENT;
}
+
+ down_write(&sess->chann_lock);
+ if (sess->tearing_down) {
+ up_write(&sess->chann_lock);
+ ksmbd_conn_unlock(conn);
+ rsp->hdr.Status = STATUS_USER_SESSION_DELETED;
+ smb2_set_err_rsp(work);
+ return -ENOENT;
+ }
+ sess->tearing_down = true;
+ up_write(&sess->chann_lock);
+
ksmbd_all_conn_set_status(sess, KSMBD_SESS_NEED_RECONNECT);
ksmbd_conn_unlock(conn);
+ err = ksmbd_conn_wait_idle_sess(conn, sess);
+ if (err) {
+ down_write(&sess->chann_lock);
+ sess->tearing_down = false;
+ up_write(&sess->chann_lock);
+ ksmbd_all_conn_set_status(sess, KSMBD_SESS_GOOD);
+ rsp->hdr.Status = STATUS_UNEXPECTED_IO_ERROR;
+ smb2_set_err_rsp(work);
+ return err;
+ }
+
ksmbd_close_session_fds(work);
- ksmbd_conn_wait_idle(conn);
if (ksmbd_tree_conn_session_logoff(sess)) {
ksmbd_debug(SMB, "Invalid tid %d\n", req->hdr.Id.SyncId.TreeId);
rsp->hdr.Status = STATUS_NETWORK_NAME_DELETED;
smb2_set_err_rsp(work);
- return -ENOENT;
+ err = -ENOENT;
+ } else {
+ err = 0;
}
down_write(&conn->session_lock);
@@ -3106,6 +3147,9 @@ int smb2_session_logoff(struct ksmbd_work *work)
ksmbd_all_conn_set_status(sess, KSMBD_SESS_NEED_SETUP);
+ if (err)
+ return err;
+
rsp->StructureSize = cpu_to_le16(4);
err = ksmbd_iov_pin_rsp(work, rsp, sizeof(struct smb2_logoff_rsp));
if (err) {
@@ -6231,21 +6275,18 @@ err_out2:
* @reqOutputBufferLength: max buffer length expected in command response
* @fixed_len: minimum fixed response length
* @rsp: query info response buffer contains output buffer length
- * @rsp_org: base response buffer pointer in case of chained response
*
* Return: 0 on success, otherwise error
*/
static int buffer_check_err(int reqOutputBufferLength,
unsigned int fixed_len,
- struct smb2_query_info_rsp *rsp,
- void *rsp_org)
+ struct smb2_query_info_rsp *rsp)
{
unsigned int output_len = le32_to_cpu(rsp->OutputBufferLength);
if (reqOutputBufferLength < fixed_len) {
pr_err("Invalid Buffer Size Requested\n");
rsp->hdr.Status = STATUS_INFO_LENGTH_MISMATCH;
- *(__be32 *)rsp_org = cpu_to_be32(sizeof(struct smb2_hdr));
return -EINVAL;
}
@@ -6256,8 +6297,7 @@ static int buffer_check_err(int reqOutputBufferLength,
return 0;
}
-static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp,
- void *rsp_org)
+static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp)
{
struct smb2_file_standard_info *sinfo;
@@ -6272,8 +6312,7 @@ static void get_standard_info_pipe(struct smb2_query_info_rsp *rsp,
cpu_to_le32(sizeof(struct smb2_file_standard_info));
}
-static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num,
- void *rsp_org)
+static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num)
{
struct smb2_file_internal_info *file_info;
@@ -6287,8 +6326,7 @@ static void get_internal_info_pipe(struct smb2_query_info_rsp *rsp, u64 num,
static int smb2_get_info_file_pipe(struct ksmbd_session *sess,
struct smb2_query_info_req *req,
- struct smb2_query_info_rsp *rsp,
- void *rsp_org)
+ struct smb2_query_info_rsp *rsp)
{
u64 id;
int rc;
@@ -6313,16 +6351,16 @@ static int smb2_get_info_file_pipe(struct ksmbd_session *sess,
switch (req->FileInfoClass) {
case FILE_STANDARD_INFORMATION:
- get_standard_info_pipe(rsp, rsp_org);
+ get_standard_info_pipe(rsp);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
le32_to_cpu(rsp->OutputBufferLength),
- rsp, rsp_org);
+ rsp);
break;
case FILE_INTERNAL_INFORMATION:
- get_internal_info_pipe(rsp, id, rsp_org);
+ get_internal_info_pipe(rsp, id);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
le32_to_cpu(rsp->OutputBufferLength),
- rsp, rsp_org);
+ rsp);
break;
default:
ksmbd_debug(SMB, "smb2_info_file_pipe for %u not supported\n",
@@ -6757,7 +6795,7 @@ static int get_file_normalized_name_info(struct ksmbd_work *work,
{
struct smb2_file_alt_name_info *file_info;
char *filename, *normalized, *stream_name;
- int conv_len, filename_len;
+ int buf_free_len, conv_len, filename_len;
if (work->conn->dialect < SMB311_PROT_ID) {
rsp->hdr.Status = STATUS_NOT_SUPPORTED;
@@ -6781,6 +6819,14 @@ static int get_file_normalized_name_info(struct ksmbd_work *work,
return -ENOMEM;
filename_len = strlen(normalized);
+ buf_free_len = smb2_resp_buf_len(work, sizeof(*rsp) +
+ sizeof(*file_info));
+ if (buf_free_len < 0 ||
+ (size_t)buf_free_len < (filename_len + 1) * sizeof(__le16)) {
+ kfree(normalized);
+ return -EINVAL;
+ }
+
file_info = (struct smb2_file_alt_name_info *)rsp->Buffer;
conv_len = smbConvertToUTF16((__le16 *)file_info->FileName,
normalized, filename_len,
@@ -7150,8 +7196,7 @@ static int smb2_get_info_file(struct ksmbd_work *work,
if (test_share_config_flag(work->tcon->share_conf,
KSMBD_SHARE_FLAG_PIPE)) {
/* smb2 info file called for pipe */
- rc = smb2_get_info_file_pipe(work->sess, req, rsp,
- work->response_buf);
+ rc = smb2_get_info_file_pipe(work->sess, req, rsp);
goto iov_pin_out;
}
@@ -7260,13 +7305,16 @@ static int smb2_get_info_file(struct ksmbd_work *work,
case FILE_ALTERNATE_NAME_INFORMATION:
fixed_len = FILE_ALTERNATE_NAME_INFORMATION_SIZE;
break;
+ case FILE_NORMALIZED_NAME_INFORMATION:
+ fixed_len = FILE_NORMALIZED_NAME_INFORMATION_SIZE;
+ break;
case FILE_STREAM_INFORMATION:
fixed_len = FILE_STREAM_INFORMATION_SIZE;
break;
}
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
fixed_len,
- rsp, work->response_buf);
+ rsp);
}
ksmbd_fd_put(work, fp);
@@ -7444,6 +7492,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
struct object_id_info *info;
info = (struct object_id_info *)(rsp->Buffer);
+ memset(info, 0, sizeof(*info));
if (path.mnt->mnt_sb->s_uuid_len == 16)
memcpy(info->objid, path.mnt->mnt_sb->s_uuid.b,
@@ -7499,6 +7548,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->FreeSpaceStopFiltering = 0;
info->DefaultQuotaThreshold = cpu_to_le64(SMB2_NO_FID);
info->DefaultQuotaLimit = cpu_to_le64(SMB2_NO_FID);
+ info->FileSystemControlFlags = 0;
info->Padding = 0;
rsp->OutputBufferLength = cpu_to_le32(48);
fixed_len = 48;
@@ -7521,6 +7571,9 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
info->UserBlocksAvail = cpu_to_le64(stfs.f_bavail);
info->TotalFileNodes = cpu_to_le64(stfs.f_files);
info->FreeFileNodes = cpu_to_le64(stfs.f_ffree);
+ info->FileSysIdentifier =
+ cpu_to_le64((u64)(u32)stfs.f_fsid.val[1] << 32 |
+ (u32)stfs.f_fsid.val[0]);
rsp->OutputBufferLength = cpu_to_le32(56);
fixed_len = 56;
}
@@ -7532,7 +7585,7 @@ static int smb2_get_info_filesystem(struct ksmbd_work *work,
}
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
fixed_len,
- rsp, work->response_buf);
+ rsp);
path_put(&path);
if (!rc)
@@ -7646,7 +7699,7 @@ release_acl:
rsp->OutputBufferLength = cpu_to_le32(secdesclen);
rc = buffer_check_err(le32_to_cpu(req->OutputBufferLength),
le32_to_cpu(rsp->OutputBufferLength),
- rsp, work->response_buf);
+ rsp);
if (rc)
goto err_out;
@@ -8620,13 +8673,18 @@ static noinline int smb2_read_pipe(struct ksmbd_work *work)
}
aux_payload_buf =
- kvmalloc(rpc_resp->payload_sz, KSMBD_DEFAULT_GFP);
+ kvmalloc(ALIGN(rpc_resp->payload_sz, 8),
+ KSMBD_DEFAULT_GFP);
if (!aux_payload_buf) {
err = -ENOMEM;
goto out;
}
memcpy(aux_payload_buf, rpc_resp->payload, rpc_resp->payload_sz);
+ if (rpc_resp->payload_sz & 7)
+ memset(aux_payload_buf + rpc_resp->payload_sz, 0,
+ ALIGN(rpc_resp->payload_sz, 8) -
+ rpc_resp->payload_sz);
nbytes = rpc_resp->payload_sz;
err = ksmbd_iov_pin_rsp_read(work, (void *)rsp,
@@ -9680,14 +9738,14 @@ int smb2_cancel(struct ksmbd_work *work)
* still on conn->async_requests with a live cancel_fn
* pointing at the freed file_lock.
*/
- if (iter->state != KSMBD_WORK_ACTIVE)
+ if (cmpxchg(&iter->state, KSMBD_WORK_ACTIVE,
+ KSMBD_WORK_CANCELLED) != KSMBD_WORK_ACTIVE)
break;
ksmbd_debug(SMB,
"smb2 with AsyncId %llu cancelled command = 0x%x\n",
le64_to_cpu(hdr->Id.AsyncId),
le16_to_cpu(chdr->Command));
- iter->state = KSMBD_WORK_CANCELLED;
if (iter->cancel_fn == smb2_notify_cancel_fn)
cancelled_notify =
smb2_notify_cancel_claim(iter->cancel_argv);
@@ -9716,11 +9774,16 @@ int smb2_cancel(struct ksmbd_work *work)
iter == work)
continue;
+ if (cmpxchg(&iter->state, KSMBD_WORK_ACTIVE,
+ KSMBD_WORK_CANCELLED) != KSMBD_WORK_ACTIVE)
+ break;
+
ksmbd_debug(SMB,
"smb2 with mid %llu cancelled command = 0x%x\n",
le64_to_cpu(hdr->MessageId),
le16_to_cpu(chdr->Command));
- iter->state = KSMBD_WORK_CANCELLED;
+ if (iter->cancel_fn)
+ iter->cancel_fn(iter->cancel_argv);
break;
}
spin_unlock(&conn->request_lock);
@@ -11766,7 +11829,7 @@ static void smb2_notify_cancel_fn(void **argv)
return;
conn = in_work->conn;
- ctx = kmalloc(sizeof(*ctx), GFP_ATOMIC);
+ ctx = kmalloc_obj(*ctx, GFP_ATOMIC);
if (!ctx) {
/* Can't defer the response -- free without sending one. */
list_del_init(&in_work->async_request_entry);
diff --git a/fs/smb/server/smb2pdu.h b/fs/smb/server/smb2pdu.h
index 3f08d1ca5a38..ca8e27f7b712 100644
--- a/fs/smb/server/smb2pdu.h
+++ b/fs/smb/server/smb2pdu.h
@@ -61,8 +61,6 @@ struct preauth_integrity_info {
#define SMB2_SESSION_IN_PROGRESS BIT(0)
#define SMB2_SESSION_VALID BIT(1)
-#define SMB2_SESSION_TIMEOUT (10 * HZ)
-
/* Apple Defined Contexts */
#define SMB2_CREATE_AAPL "AAPL"
@@ -214,6 +212,7 @@ struct file_sparse {
#define FILE_ALLOCATION_INFORMATION_SIZE 19
#define FILE_END_OF_FILE_INFORMATION_SIZE 20
#define FILE_ALTERNATE_NAME_INFORMATION_SIZE 8
+#define FILE_NORMALIZED_NAME_INFORMATION_SIZE 8
#define FILE_STREAM_INFORMATION_SIZE 32
#define FILE_PIPE_INFORMATION_SIZE 23
#define FILE_PIPE_LOCAL_INFORMATION_SIZE 24
diff --git a/fs/smb/server/smbacl.c b/fs/smb/server/smbacl.c
index 8ad2e5a5cca8..1fad6ccf3a72 100644
--- a/fs/smb/server/smbacl.c
+++ b/fs/smb/server/smbacl.c
@@ -383,10 +383,10 @@ void free_acl_state(struct posix_acl_state *state)
kfree(state->groups);
}
-static void parse_dacl(struct mnt_idmap *idmap,
- struct smb_acl *pdacl, char *end_of_acl,
- struct smb_sid *pownersid, struct smb_sid *pgrpsid,
- struct smb_fattr *fattr)
+static int parse_dacl(struct mnt_idmap *idmap,
+ struct smb_acl *pdacl, char *end_of_acl,
+ struct smb_sid *pownersid, struct smb_sid *pgrpsid,
+ struct smb_fattr *fattr)
{
int i, ret;
u16 num_aces = 0;
@@ -400,13 +400,13 @@ static void parse_dacl(struct mnt_idmap *idmap,
bool owner_found = false, group_found = false, others_found = false;
if (!pdacl)
- return;
+ return 0;
/* validate that we do not go past end of acl */
if (end_of_acl < (char *)pdacl + sizeof(struct smb_acl) ||
end_of_acl < (char *)pdacl + le16_to_cpu(pdacl->size)) {
pr_err("ACL too small to parse DACL\n");
- return;
+ return -EINVAL;
}
ksmbd_debug(SMB, "DACL revision %d size %d num aces %d\n",
@@ -418,31 +418,31 @@ static void parse_dacl(struct mnt_idmap *idmap,
num_aces = le16_to_cpu(pdacl->num_aces);
if (num_aces <= 0)
- return;
+ return 0;
dacl_size = le16_to_cpu(pdacl->size);
if (dacl_size < sizeof(struct smb_acl))
- return;
+ return -EINVAL;
if (num_aces > (dacl_size - sizeof(struct smb_acl)) /
(offsetof(struct smb_ace, sid) +
offsetof(struct smb_sid, sub_auth) + sizeof(__le16)))
- return;
+ return -EINVAL;
ret = init_acl_state(&acl_state, num_aces);
if (ret)
- return;
+ return ret;
ret = init_acl_state(&default_acl_state, num_aces);
if (ret) {
free_acl_state(&acl_state);
- return;
+ return ret;
}
ppace = kmalloc_objs(struct smb_ace *, num_aces, KSMBD_DEFAULT_GFP);
if (!ppace) {
free_acl_state(&default_acl_state);
free_acl_state(&acl_state);
- return;
+ return -ENOMEM;
}
/*
@@ -451,8 +451,10 @@ static void parse_dacl(struct mnt_idmap *idmap,
* user/group/other have no permissions
*/
for (i = 0; i < num_aces; ++i) {
- if (end_of_acl - acl_base < acl_size)
- break;
+ if (end_of_acl - acl_base < acl_size) {
+ ret = -EINVAL;
+ goto out;
+ }
ppace[i] = (struct smb_ace *)(acl_base + acl_size);
acl_base = (char *)ppace[i];
@@ -465,8 +467,10 @@ static void parse_dacl(struct mnt_idmap *idmap,
(end_of_acl - acl_base <
acl_size + sizeof(__le32) * ppace[i]->sid.num_subauth) ||
(le16_to_cpu(ppace[i]->size) <
- acl_size + sizeof(__le32) * ppace[i]->sid.num_subauth))
- break;
+ acl_size + sizeof(__le32) * ppace[i]->sid.num_subauth)) {
+ ret = -EINVAL;
+ goto out;
+ }
acl_size = le16_to_cpu(ppace[i]->size);
ppace[i]->access_req =
@@ -524,8 +528,8 @@ static void parse_dacl(struct mnt_idmap *idmap,
temp_fattr.cf_uid = INVALID_UID;
ret = sid_to_id(idmap, &ppace[i]->sid, SIDOWNER, &temp_fattr);
if (ret || uid_eq(temp_fattr.cf_uid, INVALID_UID)) {
- pr_err("%s: Error %d mapping Owner SID to uid\n",
- __func__, ret);
+ pr_err_ratelimited("%s: Error %d mapping Owner SID to uid\n",
+ __func__, ret);
continue;
}
@@ -541,7 +545,6 @@ static void parse_dacl(struct mnt_idmap *idmap,
((acl_mode & 0700) >> 6) | 0004;
}
}
- kfree(ppace);
if (owner_found) {
/* The owner must be set to at least read-only. */
@@ -584,10 +587,12 @@ static void parse_dacl(struct mnt_idmap *idmap,
fattr->cf_acls =
posix_acl_alloc(acl_state.users->n +
acl_state.groups->n + 4, KSMBD_DEFAULT_GFP);
- if (fattr->cf_acls) {
- cf_pace = fattr->cf_acls->a_entries;
- posix_state_to_acl(&acl_state, cf_pace);
+ if (!fattr->cf_acls) {
+ ret = -ENOMEM;
+ goto out;
}
+ cf_pace = fattr->cf_acls->a_entries;
+ posix_state_to_acl(&acl_state, cf_pace);
}
}
@@ -598,14 +603,20 @@ static void parse_dacl(struct mnt_idmap *idmap,
fattr->cf_dacls =
posix_acl_alloc(default_acl_state.users->n +
default_acl_state.groups->n + 4, KSMBD_DEFAULT_GFP);
- if (fattr->cf_dacls) {
- cf_pdace = fattr->cf_dacls->a_entries;
- posix_state_to_acl(&default_acl_state, cf_pdace);
+ if (!fattr->cf_dacls) {
+ ret = -ENOMEM;
+ goto out;
}
+ cf_pdace = fattr->cf_dacls->a_entries;
+ posix_state_to_acl(&default_acl_state, cf_pdace);
}
}
+ ret = 0;
+out:
+ kfree(ppace);
free_acl_state(&acl_state);
free_acl_state(&default_acl_state);
+ return ret;
}
static void set_posix_acl_entries_dacl(struct mnt_idmap *idmap,
@@ -966,8 +977,10 @@ int parse_sec_desc(struct mnt_idmap *idmap, struct smb_ntsd *pntsd,
if (dacloffset < sizeof(struct smb_ntsd))
return -EINVAL;
- parse_dacl(idmap, dacl_ptr, end_of_acl,
- owner_sid_ptr, group_sid_ptr, fattr);
+ rc = parse_dacl(idmap, dacl_ptr, end_of_acl,
+ owner_sid_ptr, group_sid_ptr, fattr);
+ if (rc)
+ return rc;
}
return 0;
diff --git a/fs/smb/server/stats.h b/fs/smb/server/stats.h
index bc864efa0d46..8b32b8b4e8be 100644
--- a/fs/smb/server/stats.h
+++ b/fs/smb/server/stats.h
@@ -15,6 +15,7 @@
enum {
KSMBD_COUNTER_SESSIONS = 0,
+ KSMBD_COUNTER_SESSION_TIMEOUTS,
KSMBD_COUNTER_TREE_CONNS,
KSMBD_COUNTER_REQUESTS,
KSMBD_COUNTER_STATUS_SUCCESS,
diff --git a/fs/smb/server/transport_ipc.c b/fs/smb/server/transport_ipc.c
index 4b0b572a3e1b..e550aa41ad2c 100644
--- a/fs/smb/server/transport_ipc.c
+++ b/fs/smb/server/transport_ipc.c
@@ -532,14 +532,21 @@ static int ipc_validate_msg(struct ipc_msg_table_entry *entry)
if (entry->msg_sz < sizeof(struct ksmbd_share_config_response))
return -EINVAL;
- if (resp->payload_sz) {
- if (resp->payload_sz < resp->veto_list_sz)
- return -EINVAL;
+ if (strnlen(resp->share_name, sizeof(resp->share_name)) ==
+ sizeof(resp->share_name))
+ return -EINVAL;
- if (check_add_overflow(sizeof(struct ksmbd_share_config_response),
- resp->payload_sz, &msg_sz))
- return -EINVAL;
- }
+ if (resp->veto_list_sz > resp->payload_sz)
+ return -EINVAL;
+
+ if (resp->flags != KSMBD_SHARE_FLAG_INVALID &&
+ !(resp->flags & KSMBD_SHARE_FLAG_PIPE) &&
+ resp->payload_sz <= resp->veto_list_sz)
+ return -EINVAL;
+
+ if (check_add_overflow(sizeof(struct ksmbd_share_config_response),
+ resp->payload_sz, &msg_sz))
+ return -EINVAL;
break;
}
case KSMBD_EVENT_LOGIN_REQUEST_EXT:
diff --git a/fs/smb/server/transport_tcp.c b/fs/smb/server/transport_tcp.c
index 832e93084605..4968cfc1a572 100644
--- a/fs/smb/server/transport_tcp.c
+++ b/fs/smb/server/transport_tcp.c
@@ -39,6 +39,7 @@ struct tcp_transport {
static const struct ksmbd_transport_ops ksmbd_tcp_transport_ops;
static void tcp_stop_kthread(struct task_struct *kthread);
+static void ksmbd_tcp_stop_listener(struct interface *iface);
static struct interface *alloc_iface(char *ifname);
static void ksmbd_tcp_disconnect(struct ksmbd_transport *t);
@@ -321,13 +322,20 @@ static int ksmbd_tcp_run_kthread(struct interface *iface)
int rc;
struct task_struct *kthread;
- kthread = kthread_run(ksmbd_kthread_fn, (void *)iface, "ksmbd-%s",
- iface->name);
+ kthread = kthread_create(ksmbd_kthread_fn, (void *)iface, "ksmbd-%s",
+ iface->name);
if (IS_ERR(kthread)) {
rc = PTR_ERR(kthread);
return rc;
}
+
+ /*
+ * The listener can exit after its socket is shutdown, so keep the
+ * task_struct alive until the caller has stopped it.
+ */
+ get_task_struct(kthread);
iface->ksmbd_kthread = kthread;
+ wake_up_process(kthread);
return 0;
}
@@ -598,12 +606,7 @@ static int ksmbd_netdev_event(struct notifier_block *nb, unsigned long event,
if (iface && iface->state == IFACE_STATE_CONFIGURED) {
ksmbd_debug(CONN, "netdev-down event: netdev(%s) is going down\n",
iface->name);
- kernel_sock_shutdown(iface->ksmbd_socket, SHUT_RDWR);
- tcp_stop_kthread(iface->ksmbd_kthread);
- iface->ksmbd_kthread = NULL;
- sock_release(iface->ksmbd_socket);
- iface->ksmbd_socket = NULL;
-
+ ksmbd_tcp_stop_listener(iface);
iface->state = IFACE_STATE_DOWN;
break;
}
@@ -631,11 +634,25 @@ static void tcp_stop_kthread(struct task_struct *kthread)
if (!kthread)
return;
- ret = kthread_stop(kthread);
+ ret = kthread_stop_put(kthread);
if (ret)
pr_err("failed to stop forker thread\n");
}
+static void ksmbd_tcp_stop_listener(struct interface *iface)
+{
+ if (iface->ksmbd_socket)
+ kernel_sock_shutdown(iface->ksmbd_socket, SHUT_RDWR);
+
+ tcp_stop_kthread(iface->ksmbd_kthread);
+ iface->ksmbd_kthread = NULL;
+
+ if (iface->ksmbd_socket) {
+ sock_release(iface->ksmbd_socket);
+ iface->ksmbd_socket = NULL;
+ }
+}
+
void ksmbd_tcp_destroy(void)
{
struct interface *iface, *tmp;
@@ -643,6 +660,7 @@ void ksmbd_tcp_destroy(void)
unregister_netdevice_notifier(&ksmbd_netdev_notifier);
list_for_each_entry_safe(iface, tmp, &iface_list, entry) {
+ ksmbd_tcp_stop_listener(iface);
list_del(&iface->entry);
kfree(iface->name);
kfree(iface);
diff --git a/fs/smb/server/vfs.c b/fs/smb/server/vfs.c
index d2b524f79cbe..c2c9aaa5de1b 100644
--- a/fs/smb/server/vfs.c
+++ b/fs/smb/server/vfs.c
@@ -2007,6 +2007,11 @@ out:
return ret;
}
+static bool ksmbd_vfs_copy_range_valid(loff_t offset, size_t len)
+{
+ return offset >= 0 && (loff_t)len <= MAX_LFS_FILESIZE - offset;
+}
+
int ksmbd_vfs_copy_file_ranges(struct ksmbd_work *work,
struct ksmbd_file *src_fp,
struct ksmbd_file *dst_fp,
@@ -2042,6 +2047,10 @@ int ksmbd_vfs_copy_file_ranges(struct ksmbd_work *work,
dst_off = le64_to_cpu(chunks[i].TargetOffset);
len = le32_to_cpu(chunks[i].Length);
+ if (!ksmbd_vfs_copy_range_valid(src_off, len) ||
+ !ksmbd_vfs_copy_range_valid(dst_off, len))
+ return -E2BIG;
+
if (check_lock_range(src_fp->filp, src_off,
src_off + len - 1, READ))
return -EAGAIN;
@@ -2134,7 +2143,8 @@ int ksmbd_vfs_copy_file_ranges(struct ksmbd_work *work,
len = le32_to_cpu(chunks[i].Length);
copy_len = len;
- if (src_off < 0)
+ if (!ksmbd_vfs_copy_range_valid(src_off, len) ||
+ !ksmbd_vfs_copy_range_valid(dst_off, len))
return -E2BIG;
if (src_off > src_file_size || len > src_file_size - src_off) {
diff --git a/fs/smb/server/vfs_cache.c b/fs/smb/server/vfs_cache.c
index 81626d204249..fd2c595f0486 100644
--- a/fs/smb/server/vfs_cache.c
+++ b/fs/smb/server/vfs_cache.c
@@ -846,12 +846,25 @@ static void set_close_state_blocked_works(struct ksmbd_file *fp)
spin_lock(&fp->f_lock);
list_for_each_entry(cancel_work, &fp->blocked_works,
fp_entry) {
- cancel_work->state = KSMBD_WORK_CLOSED;
- cancel_work->cancel_fn(cancel_work->cancel_argv);
+ if (xchg(&cancel_work->state, KSMBD_WORK_CLOSED) ==
+ KSMBD_WORK_ACTIVE)
+ cancel_work->cancel_fn(cancel_work->cancel_argv);
}
spin_unlock(&fp->f_lock);
}
+void ksmbd_wake_session_blocked_works(struct ksmbd_session *sess)
+{
+ struct ksmbd_file_table *ft = &sess->file_table;
+ struct ksmbd_file *fp;
+ unsigned int id;
+
+ read_lock(&ft->lock);
+ idr_for_each_entry(ft->idr, fp, id)
+ set_close_state_blocked_works(fp);
+ read_unlock(&ft->lock);
+}
+
int ksmbd_close_fd(struct ksmbd_work *work, u64 id)
{
struct ksmbd_file *fp;
diff --git a/fs/smb/server/vfs_cache.h b/fs/smb/server/vfs_cache.h
index 502efb16f05f..1884f6deb9d0 100644
--- a/fs/smb/server/vfs_cache.h
+++ b/fs/smb/server/vfs_cache.h
@@ -226,6 +226,7 @@ void ksmbd_stop_durable_scavenger(void);
bool ksmbd_durable_scavenger_active(void);
void ksmbd_close_tree_conn_fds(struct ksmbd_work *work);
void ksmbd_close_session_fds(struct ksmbd_work *work);
+void ksmbd_wake_session_blocked_works(struct ksmbd_session *sess);
int ksmbd_close_inode_fds(struct ksmbd_work *work, struct inode *inode);
int ksmbd_init_global_file_table(void);
void ksmbd_free_global_file_table(void);
diff --git a/fs/super.c b/fs/super.c
index 05e443173038..9d4025213521 100644
--- a/fs/super.c
+++ b/fs/super.c
@@ -172,19 +172,6 @@ static void super_wake(struct super_block *sb, unsigned int flag)
}
/*
- * The s_op->nr_cached_objects hooks (used for example by btrfs and xfs)
- * operate on filesystem-global state and ignore sc->memcg. Driving them
- * from per-memcg shrink_slab_memcg() invocations only burns CPU walking
- * per-cpu counters and queueing duplicate work: the actual reclaim happens on
- * the global path (kswapd or root direct reclaim) regardless. Restrict them
- * to that path.
- */
-static inline bool super_fs_objects_eligible(struct shrink_control *sc)
-{
- return !sc->memcg || mem_cgroup_is_root(sc->memcg);
-}
-
-/*
* One thing we have to be careful of with a per-sb shrinker is that we don't
* drop the last active reference to the superblock from within the shrinker.
* If that happens we could trigger unregistering the shrinker from within the
@@ -213,7 +200,7 @@ static unsigned long super_cache_scan(struct shrinker *shrink,
if (!super_trylock_shared(sb))
return SHRINK_STOP;
- if (sb->s_op->nr_cached_objects && super_fs_objects_eligible(sc))
+ if (sb->s_op->nr_cached_objects)
fs_objects = sb->s_op->nr_cached_objects(sb, sc);
inodes = list_lru_shrink_count(&sb->s_inode_lru, sc);
@@ -274,8 +261,7 @@ static unsigned long super_cache_count(struct shrinker *shrink,
return 0;
smp_rmb();
- if (sb->s_op && sb->s_op->nr_cached_objects &&
- super_fs_objects_eligible(sc))
+ if (sb->s_op && sb->s_op->nr_cached_objects)
total_objects = sb->s_op->nr_cached_objects(sb, sc);
total_objects += list_lru_shrink_count(&sb->s_dentry_lru, sc);
@@ -2369,11 +2355,14 @@ static int thaw_super_locked(struct super_block *sb, enum freeze_holder who,
goto out_unlock;
/*
- * All freezers share a single active reference.
- * So just unlock in case there are any left.
+ * All freezers share a single active reference. If other freezers
+ * remain, drop our hold and report success; the superblock stays
+ * frozen until the last holder thaws it.
*/
- if (freeze_dec(sb, who))
+ if (freeze_dec(sb, who)) {
+ error = 0;
goto out_unlock;
+ }
if (sb_rdonly(sb)) {
sb->s_writers.frozen = SB_UNFROZEN;
diff --git a/fs/ufs/cylinder.c b/fs/ufs/cylinder.c
index a2813270c303..b930ee1cf853 100644
--- a/fs/ufs/cylinder.c
+++ b/fs/ufs/cylinder.c
@@ -68,6 +68,16 @@ static bool ufs_read_cylinder(struct super_block *sb,
ucpi->c_clustersumoff = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_clustersumoff);
ucpi->c_clusteroff = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_clusteroff);
ucpi->c_nclusterblks = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_nclusterblks);
+
+ /* these on-disk values become array and bitmap indices */
+ if (ucpi->c_cgx != cgno ||
+ ucpi->c_rotor >= uspi->s_fpg ||
+ ucpi->c_frotor >= uspi->s_fpg ||
+ ucpi->c_irotor >= uspi->s_ipg) {
+ ufs_error(sb, __func__,
+ "inconsistent metadata in cylinder group %u\n", cgno);
+ goto failed;
+ }
UFSD("EXIT\n");
return true;
diff --git a/fs/ufs/dir.c b/fs/ufs/dir.c
index e62fe5667671..ce43cf20b07c 100644
--- a/fs/ufs/dir.c
+++ b/fs/ufs/dir.c
@@ -590,7 +590,7 @@ int ufs_empty_dir(struct inode * inode)
kaddr = ufs_get_folio(inode, i, &folio);
if (IS_ERR(kaddr))
- continue;
+ return 0;
de = (struct ufs_dir_entry *)kaddr;
kaddr += ufs_last_byte(inode, i) - UFS_DIR_REC_LEN(1);
diff --git a/fs/ufs/super.c b/fs/ufs/super.c
index 6dcf6d048cce..3569ac92b065 100644
--- a/fs/ufs/super.c
+++ b/fs/ufs/super.c
@@ -1199,6 +1199,15 @@ magic_found:
sb->s_maxbytes = ufs_max_bytes(sb);
sb->s_max_links = UFS_LINK_MAX;
+ ufs_setup_cstotal(sb);
+ /*
+ * Read cylinder group structures
+ */
+ if (!sb_rdonly(sb))
+ if (!ufs_read_cylinder_structures(sb))
+ goto failed;
+
+ /* create the root dentry last, once UFS_SB(sb) is fully set up */
inode = ufs_iget(sb, UFS_ROOTINO);
if (IS_ERR(inode)) {
ret = PTR_ERR(inode);
@@ -1210,14 +1219,6 @@ magic_found:
goto failed;
}
- ufs_setup_cstotal(sb);
- /*
- * Read cylinder group structures
- */
- if (!sb_rdonly(sb))
- if (!ufs_read_cylinder_structures(sb))
- goto failed;
-
UFSD("EXIT\n");
return 0;
diff --git a/fs/xfs/Makefile b/fs/xfs/Makefile
index 9f7133e02576..399a207f2d0e 100644
--- a/fs/xfs/Makefile
+++ b/fs/xfs/Makefile
@@ -91,6 +91,7 @@ xfs-y += xfs_aops.o \
xfs_healthmon.o \
xfs_icache.o \
xfs_ioctl.o \
+ xfs_ioend.o \
xfs_iomap.o \
xfs_iops.o \
xfs_inode.o \
diff --git a/fs/xfs/libxfs/xfs_btree_mem.c b/fs/xfs/libxfs/xfs_btree_mem.c
index 37136a70e56d..1d83a4251cee 100644
--- a/fs/xfs/libxfs/xfs_btree_mem.c
+++ b/fs/xfs/libxfs/xfs_btree_mem.c
@@ -117,6 +117,7 @@ xfbtree_init(
struct xfs_buftarg *btp,
const struct xfs_btree_ops *ops)
{
+ unsigned long long owner = xfbt->owner;
unsigned int blocklen = xfbtree_rec_bytes(mp, ops);
unsigned int keyptr_len;
int error;
@@ -133,6 +134,7 @@ xfbtree_init(
memset(xfbt, 0, sizeof(*xfbt));
xfbt->target = btp;
+ xfbt->owner = owner;
/* Set up min/maxrecs for this btree. */
keyptr_len = ops->key_len + sizeof(__be64);
diff --git a/fs/xfs/libxfs/xfs_btree_staging.c b/fs/xfs/libxfs/xfs_btree_staging.c
index 7314dab4bcfb..561fd2c2e950 100644
--- a/fs/xfs/libxfs/xfs_btree_staging.c
+++ b/fs/xfs/libxfs/xfs_btree_staging.c
@@ -336,8 +336,10 @@ xfs_btree_bload_prep_block(
xfs_btree_set_sibling(cur, *blockp, &new_ptr, XFS_BB_RIGHTSIB);
ret = xfs_btree_bload_drop_buf(bbl, buffers_list, bpp);
- if (ret)
+ if (ret) {
+ xfs_buf_relse(new_bp);
return ret;
+ }
/* Initialize the new btree block. */
xfs_btree_init_block_cur(cur, new_bp, level, nr_this_block);
diff --git a/fs/xfs/libxfs/xfs_da_btree.c b/fs/xfs/libxfs/xfs_da_btree.c
index f190c088591b..3d02a0d7ba44 100644
--- a/fs/xfs/libxfs/xfs_da_btree.c
+++ b/fs/xfs/libxfs/xfs_da_btree.c
@@ -130,7 +130,7 @@ xfs_da_state_reset(
state->mp = state->args->dp->i_mount;
}
-static inline int xfs_dabuf_nfsb(struct xfs_mount *mp, int whichfork)
+inline int xfs_dabuf_nfsb(struct xfs_mount *mp, int whichfork)
{
if (whichfork == XFS_DATA_FORK)
return mp->m_dir_geo->fsbcount;
@@ -2384,6 +2384,7 @@ xfs_da_grow_inode_int(
}
/* account for newly allocated blocks in reserved blocks total */
+ ASSERT(args->total >= dp->i_nblocks - nblks);
args->total -= dp->i_nblocks - nblks;
out_free_map:
@@ -2746,8 +2747,8 @@ xfs_dabuf_map(
* larger one that needs to be free by the caller.
*/
if (nirecs > 1) {
- map = kcalloc(nirecs, sizeof(struct xfs_buf_map),
- GFP_KERNEL | __GFP_NOLOCKDEP | __GFP_NOFAIL);
+ map = kzalloc_objs(struct xfs_buf_map, nirecs,
+ GFP_KERNEL | __GFP_NOLOCKDEP | __GFP_NOFAIL);
*mapp = map;
}
diff --git a/fs/xfs/libxfs/xfs_da_btree.h b/fs/xfs/libxfs/xfs_da_btree.h
index afcf2d3c7a21..a718b1ceb0aa 100644
--- a/fs/xfs/libxfs/xfs_da_btree.h
+++ b/fs/xfs/libxfs/xfs_da_btree.h
@@ -244,4 +244,6 @@ xfs_failaddr_t xfs_da3_node_header_check(struct xfs_buf *bp, xfs_ino_t owner);
extern struct kmem_cache *xfs_da_state_cache;
+int xfs_dabuf_nfsb(struct xfs_mount *mp, int whichfork);
+
#endif /* __XFS_DA_BTREE_H__ */
diff --git a/fs/xfs/libxfs/xfs_defer.c b/fs/xfs/libxfs/xfs_defer.c
index 89501e8bd2f8..3152acdc335d 100644
--- a/fs/xfs/libxfs/xfs_defer.c
+++ b/fs/xfs/libxfs/xfs_defer.c
@@ -229,6 +229,7 @@ xfs_defer_barrier_cancel_item(
}
static const struct xfs_defer_op_type xfs_barrier_defer_type = {
+ .name = "barrier",
.max_items = 1,
.create_intent = xfs_defer_barrier_create_intent,
.abort_intent = xfs_defer_barrier_abort_intent,
@@ -583,7 +584,7 @@ xfs_defer_finish_one(
const struct xfs_defer_op_type *ops = dfp->dfp_ops;
struct xfs_btree_cur *state = NULL;
struct list_head *li, *n;
- int error;
+ int error = 0;
trace_xfs_defer_pending_finish(tp->t_mountp, dfp);
@@ -655,6 +656,7 @@ xfs_defer_finish_noroll(
struct xfs_trans **tp)
{
struct xfs_defer_pending *dfp = NULL;
+ const char *what = "chain";
int error = 0;
LIST_HEAD(dop_pending);
LIST_HEAD(dop_paused);
@@ -704,9 +706,17 @@ xfs_defer_finish_noroll(
struct xfs_defer_pending, dfp_list);
if (!dfp)
break;
+ what = dfp->dfp_ops->name;
error = xfs_defer_finish_one(*tp, dfp);
if (error && error != -EAGAIN)
goto out_shutdown;
+ /*
+ * A finished item is no longer a candidate for a later
+ * failure. An -EAGAIN one is not finished, so it keeps the
+ * attribution across the roll that completes it.
+ */
+ if (!error)
+ what = "chain";
}
/* Requeue the paused items in the outgoing transaction. */
@@ -718,8 +728,12 @@ xfs_defer_finish_noroll(
out_shutdown:
list_splice_tail_init(&dop_paused, &dop_pending);
xfs_defer_trans_abort(*tp, &dop_pending);
- xfs_force_shutdown((*tp)->t_mountp, SHUTDOWN_CORRUPT_INCORE);
trace_xfs_defer_finish_error(*tp, error);
+ if (!xfs_is_shutdown((*tp)->t_mountp))
+ xfs_alert((*tp)->t_mountp,
+ "deferred %s work failed, error %d, %u blocks reserved",
+ what, error, (*tp)->t_blk_res);
+ xfs_force_shutdown((*tp)->t_mountp, SHUTDOWN_CORRUPT_INCORE);
xfs_defer_cancel_list((*tp)->t_mountp, &dop_pending);
xfs_defer_cancel(*tp);
return error;
diff --git a/fs/xfs/libxfs/xfs_exchmaps.c b/fs/xfs/libxfs/xfs_exchmaps.c
index 3efed37cb98a..49eda8d0994d 100644
--- a/fs/xfs/libxfs/xfs_exchmaps.c
+++ b/fs/xfs/libxfs/xfs_exchmaps.c
@@ -969,6 +969,16 @@ xmi_can_exchange_reflink_flags(
if (req->flags & XFS_EXCHMAPS_INO1_WRITTEN)
return false;
+ /*
+ * The INO1_WRITTEN optimization can skip exchanging hole and
+ * unwritten mappings, which means we cannot guarantee that all
+ * shared extents actually moved to the other file. Clearing the
+ * reflink flag of an inode that still holds shared extents breaks
+ * the CoW write path, so refuse to exchange the flags in that case.
+ */
+ if (req->flags & XFS_EXCHMAPS_INO1_WRITTEN)
+ return false;
+
if (hweight32(reflink_state) != 1)
return false;
if (req->startoff1 != 0 || req->startoff2 != 0)
diff --git a/fs/xfs/libxfs/xfs_parent.c b/fs/xfs/libxfs/xfs_parent.c
index 8d111c9b6527..a2f2f5fa640e 100644
--- a/fs/xfs/libxfs/xfs_parent.c
+++ b/fs/xfs/libxfs/xfs_parent.c
@@ -193,7 +193,7 @@ xfs_parent_addname(
const struct xfs_name *parent_name,
struct xfs_inode *child)
{
- int error;
+ int error, local;
error = xfs_parent_iread_extents(tp, child);
if (error)
@@ -203,6 +203,10 @@ xfs_parent_addname(
xfs_parent_da_args_init(&ppargs->args, tp, &ppargs->rec, child,
I_INO(child), parent_name);
+ /* Growing the attr fork needs a real reservation in args->total. */
+ ppargs->args.total = xfs_attr_calc_size(&ppargs->args, &local);
+ ASSERT(local);
+
return xfs_attr_setname(&ppargs->args, 0);
}
@@ -239,7 +243,7 @@ xfs_parent_replacename(
const struct xfs_name *new_name,
struct xfs_inode *child)
{
- int error;
+ int error, local;
error = xfs_parent_iread_extents(tp, child);
if (error)
@@ -249,6 +253,10 @@ xfs_parent_replacename(
xfs_parent_da_args_init(&ppargs->args, tp, &ppargs->rec, child,
I_INO(child), old_name);
+ /* Growing the attr fork needs a real reservation in args->total. */
+ ppargs->args.total = xfs_attr_calc_size(&ppargs->args, &local);
+ ASSERT(local);
+
xfs_inode_to_parent_rec(&ppargs->new_rec, new_dp);
ppargs->args.new_name = new_name->name;
diff --git a/fs/xfs/libxfs/xfs_rtgroup.h b/fs/xfs/libxfs/xfs_rtgroup.h
index c0b9f9f2c413..fca2eb74908c 100644
--- a/fs/xfs/libxfs/xfs_rtgroup.h
+++ b/fs/xfs/libxfs/xfs_rtgroup.h
@@ -359,7 +359,11 @@ static inline int xfs_initialize_rtgroups(struct xfs_mount *mp,
# define xfs_rtgroup_unlock(rtg, gf) ((void)0)
# define xfs_rtgroup_trans_join(tp, rtg, gf) ((void)0)
# define xfs_update_rtsb(bp, sb_bp) ((void)0)
-# define xfs_log_rtsb(tp, sb_bp) (NULL)
+static inline struct xfs_buf *xfs_log_rtsb(struct xfs_trans *tp,
+ const struct xfs_buf *sb_bp)
+{
+ return NULL;
+}
# define xfs_rtgroup_get_geometry(rtg, rgeo) (-EOPNOTSUPP)
#endif /* CONFIG_XFS_RT */
diff --git a/fs/xfs/libxfs/xfs_rtrefcount_btree.c b/fs/xfs/libxfs/xfs_rtrefcount_btree.c
index 22acc1411aac..e2950dbe2068 100644
--- a/fs/xfs/libxfs/xfs_rtrefcount_btree.c
+++ b/fs/xfs/libxfs/xfs_rtrefcount_btree.c
@@ -489,8 +489,11 @@ xfs_rtrefcountbt_maxlevels_ondisk(void)
minrecs[0] = xfs_rtrefcountbt_block_maxrecs(blocklen, true) / 2;
minrecs[1] = xfs_rtrefcountbt_block_maxrecs(blocklen, false) / 2;
- /* We need at most one record for every block in an rt group. */
- return xfs_btree_compute_maxlevels(minrecs, XFS_MAX_RGBLOCKS);
+ /*
+ * We need at most one record for every block in an rt group, and
+ * one extra level for the inode root.
+ */
+ return xfs_btree_compute_maxlevels(minrecs, XFS_MAX_RGBLOCKS) + 1;
}
int __init
diff --git a/fs/xfs/libxfs/xfs_rtrmap_btree.c b/fs/xfs/libxfs/xfs_rtrmap_btree.c
index c264bc5651c0..a15e460a1ec7 100644
--- a/fs/xfs/libxfs/xfs_rtrmap_btree.c
+++ b/fs/xfs/libxfs/xfs_rtrmap_btree.c
@@ -618,7 +618,7 @@ xfs_rtrmapbt_mem_cursor(
struct xfs_btree_cur *cur;
cur = xfs_btree_alloc_cursor(mp, tp, &xfs_rtrmapbt_mem_ops,
- mp->m_rtrmap_maxlevels, xfs_rtrmapbt_cur_cache);
+ xfs_rtrmapbt_maxlevels_ondisk(), xfs_rtrmapbt_cur_cache);
cur->bc_mem.xfbtree = xfbt;
cur->bc_nlevels = xfbt->nlevels;
cur->bc_group = xfs_group_hold(rtg_group(rtg));
@@ -716,10 +716,12 @@ xfs_rtrmapbt_maxlevels_ondisk(void)
* happens, which means that we must compute the max height based on
* what the btree will look like if it consumes almost all the blocks
* in the data device due to maximal sharing factor.
+ *
+ * Add one extra level for the inode root.
*/
max_dblocks = -1U; /* max ag count */
max_dblocks *= XFS_MAX_CRC_AG_BLOCKS;
- return xfs_btree_space_to_height(minrecs, max_dblocks);
+ return xfs_btree_space_to_height(minrecs, max_dblocks) + 1;
}
int __init
diff --git a/fs/xfs/libxfs/xfs_sb.c b/fs/xfs/libxfs/xfs_sb.c
index 75f2a021ee6d..f0341adbb879 100644
--- a/fs/xfs/libxfs/xfs_sb.c
+++ b/fs/xfs/libxfs/xfs_sb.c
@@ -1470,36 +1470,33 @@ xfs_sync_sb_buf(
bool update_rtsb)
{
struct xfs_trans *tp;
- struct xfs_buf *bp;
- struct xfs_buf *rtsb_bp = NULL;
int error;
error = xfs_trans_alloc(mp, &M_RES(mp)->tr_sb, 0, 0, 0, &tp);
if (error)
return error;
- bp = xfs_trans_getsb(tp);
xfs_log_sb(tp);
- xfs_trans_bhold(tp, bp);
- if (update_rtsb) {
- rtsb_bp = xfs_log_rtsb(tp, bp);
- if (rtsb_bp)
- xfs_trans_bhold(tp, rtsb_bp);
- }
+ if (update_rtsb)
+ xfs_log_rtsb(tp, xfs_trans_getsb(tp));
xfs_trans_set_sync(tp);
error = xfs_trans_commit(tp);
if (error)
- goto out;
- /*
- * write out the sb buffer to get the changes to disk
- */
- error = xfs_bwrite(bp);
- if (!error && rtsb_bp)
- error = xfs_bwrite(rtsb_bp);
-out:
- if (rtsb_bp)
- xfs_buf_relse(rtsb_bp);
- xfs_buf_relse(bp);
+ return error;
+
+ /* Re-acquire and write the sb and rtsb to disk. */
+ xfs_buf_lock(mp->m_sb_bp);
+ error = xfs_bwrite(mp->m_sb_bp);
+ xfs_buf_unlock(mp->m_sb_bp);
+ if (error)
+ return error;
+
+ if (update_rtsb && mp->m_rtsb_bp) {
+ xfs_buf_lock(mp->m_rtsb_bp);
+ error = xfs_bwrite(mp->m_rtsb_bp);
+ xfs_buf_unlock(mp->m_rtsb_bp);
+ }
+
return error;
}
diff --git a/fs/xfs/libxfs/xfs_trans_space.c b/fs/xfs/libxfs/xfs_trans_space.c
index 9b8f495c9049..c4cd547033e5 100644
--- a/fs/xfs/libxfs/xfs_trans_space.c
+++ b/fs/xfs/libxfs/xfs_trans_space.c
@@ -22,8 +22,23 @@ xfs_parent_calc_space_res(
unsigned int namelen)
{
/*
- * Parent pointers are always the first attr in an attr tree, and never
- * larger than a block
+ * A parent pointer is recorded per dirent, so an inode with N links
+ * carries N of them and the attr fork can already be in leaf or node
+ * format when one is added. That does not affect the reservation:
+ * XFS_DAENTER_SPACE_RES covers a split at every level of a
+ * maximum-depth attr dabtree, whatever format the fork is in now.
+ *
+ * The name is a dirent name and the value is a struct xfs_parent_rec,
+ * so the leaf entry is always local and never exceeds 272 bytes.
+ * Parent pointers require V5, hence a 1k minimum block size, so the
+ * entry always stays under half a block and this needs none of the
+ * double split allowance that xfs_attr_calc_size() makes.
+ *
+ * The second term hands a byte count to a macro whose parameter counts
+ * mappings, so it asks for more extent-add allowance than the single
+ * mapping a parent pointer adds - how much more depends on the block
+ * size. It over-reserves either way, which is why it is left alone:
+ * correcting the unit would shrink a reservation that is only generous.
*/
return XFS_DAENTER_SPACE_RES(mp, XFS_ATTR_FORK) +
XFS_NEXTENTADD_SPACE_RES(mp, namelen, XFS_ATTR_FORK);
diff --git a/fs/xfs/scrub/agheader.c b/fs/xfs/scrub/agheader.c
index 1fa66aa68e16..fa5d32ec020a 100644
--- a/fs/xfs/scrub/agheader.c
+++ b/fs/xfs/scrub/agheader.c
@@ -418,6 +418,13 @@ xchk_superblock(
xchk_block_set_corrupt(sc, bp);
}
+ if (xfs_has_zoned(mp)) {
+ if (sb->sb_rtstart != cpu_to_be64(mp->m_sb.sb_rtstart))
+ xchk_block_set_corrupt(sc, bp);
+ if (sb->sb_rtreserved != cpu_to_be64(mp->m_sb.sb_rtreserved))
+ xchk_block_set_corrupt(sc, bp);
+ }
+
/* Everything else must be zero. */
sblen = xchk_superblock_ondisk_size(mp);
if (memchr_inv((char *)sb + sblen, 0, BBTOB(bp->b_length) - sblen))
diff --git a/fs/xfs/scrub/agheader_repair.c b/fs/xfs/scrub/agheader_repair.c
index 2104512f1ee1..a66b611588c4 100644
--- a/fs/xfs/scrub/agheader_repair.c
+++ b/fs/xfs/scrub/agheader_repair.c
@@ -668,14 +668,16 @@ xrep_agfl_init_header(
struct xfs_scrub *sc,
struct xfs_buf *agfl_bp,
struct xagb_bitmap *agfl_extents,
- xfs_agblock_t flcount)
+ xfs_agblock_t flcount,
+ struct xfs_agfl *old_agfl)
{
struct xrep_agfl_fill af = {
.sc = sc,
.flcount = flcount,
};
struct xfs_mount *mp = sc->mp;
- struct xfs_agfl *agfl;
+ struct xfs_agfl *agfl = XFS_BUF_TO_AGFL(agfl_bp);
+ const size_t agfl_sz = BBTOB(agfl_bp->b_length);
int error;
ASSERT(flcount <= xfs_agfl_size(mp));
@@ -684,8 +686,8 @@ xrep_agfl_init_header(
* Start rewriting the header by setting the bno[] array to
* NULLAGBLOCK, then setting AGFL header fields.
*/
- agfl = XFS_BUF_TO_AGFL(agfl_bp);
- memset(agfl, 0xFF, BBTOB(agfl_bp->b_length));
+ memcpy(old_agfl, agfl, agfl_sz);
+ memset(agfl, 0xFF, agfl_sz);
agfl->agfl_magicnum = cpu_to_be32(XFS_AGFL_MAGIC);
agfl->agfl_seqno = cpu_to_be32(pag_agno(sc->sa.pag));
uuid_copy(&agfl->agfl_uuid, &mp->m_sb.sb_meta_uuid);
@@ -697,16 +699,23 @@ xrep_agfl_init_header(
*/
xagb_bitmap_init(&af.used_extents);
af.agfl_bno = xfs_buf_to_agfl_bno(agfl_bp);
- xagb_bitmap_walk(agfl_extents, xrep_agfl_fill, &af);
+ error = xagb_bitmap_walk(agfl_extents, xrep_agfl_fill, &af);
+ if (error && error != -ECANCELED)
+ goto err_undo;
error = xagb_bitmap_disunion(agfl_extents, &af.used_extents);
if (error)
- return error;
+ goto err_undo;
/* Write new AGFL to disk. */
xfs_trans_buf_set_type(sc->tp, agfl_bp, XFS_BLFT_AGFL_BUF);
- xfs_trans_log_buf(sc->tp, agfl_bp, 0, BBTOB(agfl_bp->b_length) - 1);
+ xfs_trans_log_buf(sc->tp, agfl_bp, 0, agfl_sz - 1);
xagb_bitmap_destroy(&af.used_extents);
return 0;
+
+err_undo:
+ xagb_bitmap_destroy(&af.used_extents);
+ memcpy(agfl, old_agfl, agfl_sz);
+ return error;
}
/* Repair the AGFL. */
@@ -718,6 +727,7 @@ xrep_agfl(
struct xfs_mount *mp = sc->mp;
struct xfs_buf *agf_bp;
struct xfs_buf *agfl_bp;
+ struct xfs_agfl *old_agfl;
xfs_agblock_t flcount;
int error;
@@ -725,6 +735,10 @@ xrep_agfl(
if (!xfs_has_rmapbt(mp))
return -EOPNOTSUPP;
+ old_agfl = kzalloc(BBTOB(XFS_FSS_TO_BB(mp, 1)), XCHK_GFP_FLAGS);
+ if (!old_agfl)
+ return -ENOMEM;
+
xagb_bitmap_init(&agfl_extents);
/*
@@ -734,7 +748,7 @@ xrep_agfl(
*/
error = xfs_alloc_read_agf(sc->sa.pag, sc->tp, 0, &agf_bp);
if (error)
- return error;
+ goto err_old_agfl;
/*
* Make sure we have the AGFL buffer, as scrub might have decided it
@@ -745,7 +759,7 @@ xrep_agfl(
XFS_AGFL_DADDR(mp)),
XFS_FSS_TO_BB(mp, 1), 0, &agfl_bp, NULL);
if (error)
- return error;
+ goto err_old_agfl;
agfl_bp->b_ops = &xfs_agfl_buf_ops;
/* Gather all the extents we're going to put on the new AGFL. */
@@ -762,10 +776,11 @@ xrep_agfl(
* we adjust the AGF flcount (which can fail) so avoid updating any
* buffers until we know that part works.
*/
- xrep_agfl_update_agf(sc, agf_bp, flcount);
- error = xrep_agfl_init_header(sc, agfl_bp, &agfl_extents, flcount);
+ error = xrep_agfl_init_header(sc, agfl_bp, &agfl_extents, flcount,
+ old_agfl);
if (error)
goto err;
+ xrep_agfl_update_agf(sc, agf_bp, flcount);
/*
* Ok, the AGFL should be ready to go now. Roll the transaction to
@@ -785,6 +800,8 @@ xrep_agfl(
err:
xagb_bitmap_destroy(&agfl_extents);
+err_old_agfl:
+ kfree(old_agfl);
return error;
}
diff --git a/fs/xfs/scrub/alloc_repair.c b/fs/xfs/scrub/alloc_repair.c
index dce6ab0429dc..95e318e4f3a6 100644
--- a/fs/xfs/scrub/alloc_repair.c
+++ b/fs/xfs/scrub/alloc_repair.c
@@ -571,7 +571,7 @@ xrep_abt_dispose_one(
* allocation, and blocks that didn't get used can be freed via the usual
* (deferred) means.
*/
-STATIC void
+STATIC int
xrep_abt_dispose_reservations(
struct xrep_abt *ra,
int error)
@@ -582,9 +582,13 @@ xrep_abt_dispose_reservations(
goto junkit;
list_for_each_entry_safe(resv, n, &ra->new_bnobt.resv_list, list) {
- error = xrep_abt_dispose_one(ra, resv);
- if (error)
+ int error2 = xrep_abt_dispose_one(ra, resv);
+
+ if (error2) {
+ if (!error)
+ error = error2;
goto junkit;
+ }
}
junkit:
@@ -596,6 +600,7 @@ junkit:
xrep_newbt_cancel(&ra->new_bnobt);
xrep_newbt_cancel(&ra->new_cntbt);
+ return error;
}
/* Retrieve free space data for bulk load. */
@@ -801,7 +806,9 @@ xrep_abt_build_new_trees(
goto err_newbt;
/* Dispose of any unused blocks and the accounting information. */
- xrep_abt_dispose_reservations(ra, error);
+ error = xrep_abt_dispose_reservations(ra, error);
+ if (error)
+ return error;
return xrep_roll_ag_trans(sc);
@@ -812,8 +819,7 @@ err_cur:
xfs_btree_del_cursor(cnt_cur, error);
xfs_btree_del_cursor(bno_cur, error);
err_newbt:
- xrep_abt_dispose_reservations(ra, error);
- return error;
+ return xrep_abt_dispose_reservations(ra, error);
}
/*
diff --git a/fs/xfs/scrub/attr_repair.c b/fs/xfs/scrub/attr_repair.c
index 6e6af142f1fb..28f92e9ba72b 100644
--- a/fs/xfs/scrub/attr_repair.c
+++ b/fs/xfs/scrub/attr_repair.c
@@ -1294,7 +1294,7 @@ xrep_xattr_swap_prep(
.geo = sc->mp->m_attr_geo,
.whichfork = XFS_ATTR_FORK,
.trans = sc->tp,
- .total = 1,
+ .total = xfs_dabuf_nfsb(sc->mp, XFS_ATTR_FORK),
.owner = I_INO(sc->ip),
};
diff --git a/fs/xfs/scrub/bmap.c b/fs/xfs/scrub/bmap.c
index 401c278725d2..4f3c7f681bd9 100644
--- a/fs/xfs/scrub/bmap.c
+++ b/fs/xfs/scrub/bmap.c
@@ -274,7 +274,7 @@ xchk_bmap_xref_rmap_cow(
unsigned long long rmap_end;
uint64_t owner = XFS_RMAP_OWN_COW;
- if (!info->sc->sa.rmap_cur || xchk_skip_xref(info->sc->sm))
+ if (xchk_skip_xref(info->sc->sm))
return;
/* Find the rmap record for this irec. */
@@ -1103,8 +1103,9 @@ xchk_bmap(
* the rmap must match the combined mapping exactly.
*/
while (xchk_bmap_iext_iter(&info, &irec)) {
- if (xchk_should_terminate(sc, &error) ||
- (sc->sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT))
+ if (xchk_should_terminate(sc, &error))
+ return error;
+ if (sc->sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT)
return 0;
if (irec.br_startoff >= endoff) {
diff --git a/fs/xfs/scrub/common.h b/fs/xfs/scrub/common.h
index 9d627fd50687..f0f073a93413 100644
--- a/fs/xfs/scrub/common.h
+++ b/fs/xfs/scrub/common.h
@@ -74,7 +74,6 @@ int xchk_setup_ag_rmapbt(struct xfs_scrub *sc);
int xchk_setup_ag_refcountbt(struct xfs_scrub *sc);
int xchk_setup_inode(struct xfs_scrub *sc);
int xchk_setup_inode_bmap(struct xfs_scrub *sc);
-int xchk_setup_inode_bmap_data(struct xfs_scrub *sc);
int xchk_setup_directory(struct xfs_scrub *sc);
int xchk_setup_xattr(struct xfs_scrub *sc);
int xchk_setup_symlink(struct xfs_scrub *sc);
diff --git a/fs/xfs/scrub/dabtree.h b/fs/xfs/scrub/dabtree.h
index de291e3b77dd..d654c125feb4 100644
--- a/fs/xfs/scrub/dabtree.h
+++ b/fs/xfs/scrub/dabtree.h
@@ -37,8 +37,6 @@ bool xchk_da_process_error(struct xchk_da_btree *ds, int level, int *error);
void xchk_da_set_corrupt(struct xchk_da_btree *ds, int level);
void xchk_da_set_preen(struct xchk_da_btree *ds, int level);
-void xchk_da_set_preen(struct xchk_da_btree *ds, int level);
-
int xchk_da_btree_hash(struct xchk_da_btree *ds, int level, __be32 *hashp);
int xchk_da_btree(struct xfs_scrub *sc, int whichfork,
xchk_da_btree_rec_fn scrub_fn, void *private);
diff --git a/fs/xfs/scrub/dir_repair.c b/fs/xfs/scrub/dir_repair.c
index 1c088cfba10e..2cfcf1c35679 100644
--- a/fs/xfs/scrub/dir_repair.c
+++ b/fs/xfs/scrub/dir_repair.c
@@ -484,18 +484,24 @@ xrep_dir_recover_data(
while (offset < end) {
struct xfs_dir2_data_unused *dup = bp->b_addr + offset;
struct xfs_dir2_data_entry *dep = bp->b_addr + offset;
+ unsigned int advance;
if (xchk_should_terminate(rd->sc, &error))
return error;
/* Skip unused entries. */
if (be16_to_cpu(dup->freetag) == XFS_DIR2_DATA_FREE_TAG) {
+ if (!dup->length)
+ break;
offset += be16_to_cpu(dup->length);
continue;
}
/* Don't walk off the end of the block. */
- offset += xfs_dir2_data_entsize(rd->sc->mp, dep->namelen);
+ advance = xfs_dir2_data_entsize(rd->sc->mp, dep->namelen);
+ if (!advance)
+ break;
+ offset += advance;
if (offset > end)
break;
@@ -721,7 +727,7 @@ xrep_dir_replay_removename(
const struct xfs_name *name,
xfs_extlen_t total)
{
- struct xfs_inode *dp = rd->args.dp;
+ struct xfs_inode *dp = rd->sc->tempip;
ASSERT(S_ISDIR(VFS_I(dp)->i_mode));
@@ -1375,9 +1381,24 @@ xrep_dir_live_update(
if (p->delta > 0)
error = xrep_dir_stash_createname(rd, p->name,
I_INO(p->ip));
- else
- error = xrep_dir_stash_removename(rd, p->name,
+ else {
+ /*
+ * xfs_dentry_to_name in unlink or rename-exchange can
+ * pass us names with ftype FT_UNKNOWN, but we really
+ * must know the ftype of the child that is being
+ * removed so that we can do nlink updates correctly
+ * without holding inode references.
+ */
+ struct xfs_name name = {
+ .name = p->name->name,
+ .len = p->name->len,
+ .type = xfs_mode_to_ftype(
+ VFS_IC(p->ip)->i_mode),
+ };
+
+ error = xrep_dir_stash_removename(rd, &name,
I_INO(p->ip));
+ }
mutex_unlock(&rd->pscan.lock);
if (error)
goto out_abort;
@@ -1467,7 +1488,7 @@ xrep_dir_swap_prep(
.geo = sc->mp->m_dir_geo,
.whichfork = XFS_DATA_FORK,
.trans = sc->tp,
- .total = 1,
+ .total = xfs_dabuf_nfsb(sc->mp, XFS_DATA_FORK),
.owner = I_INO(sc->ip),
};
diff --git a/fs/xfs/scrub/dirtree.c b/fs/xfs/scrub/dirtree.c
index b2cf6e5439d9..9b0ab2316612 100644
--- a/fs/xfs/scrub/dirtree.c
+++ b/fs/xfs/scrub/dirtree.c
@@ -259,6 +259,7 @@ xchk_dirtree_create_path(
dl->nr_paths++;
return 0;
out_path:
+ xino_bitmap_destroy(&path->seen_inodes);
kfree(path);
return error;
}
@@ -368,12 +369,38 @@ xchk_dirpath_step_up(
struct xfs_inode *dp;
xfs_ino_t parent_ino = be64_to_cpu(dl->pptr_rec.p_ino);
unsigned int lock_mode;
- int error;
+ int error = 0;
+
+ if (xchk_should_terminate(sc, &error))
+ return error;
/* Grab and lock the parent directory. */
error = xchk_iget(sc, parent_ino, &dp);
- if (error)
+ switch (error) {
+ case -EINVAL:
+ case -ENOENT:
+ mutex_lock(&dl->lock);
+
+ if (dl->stale) {
+ /* live update detected a change in this path */
+ error = -ESTALE;
+ } else {
+ /* inode doesn't exist, path invalid */
+ error = -EFSCORRUPTED;
+
+ trace_xchk_dirpath_badino(dl->sc, path->path_nr,
+ path->nr_steps, &dl->xname,
+ &dl->pptr_rec);
+ }
+
+ mutex_unlock(&dl->lock);
+ return error;
+ case 0:
+ /* keep going */
+ break;
+ default:
return error;
+ }
lock_mode = xfs_ilock_attr_map_shared(dp);
mutex_lock(&dl->lock);
diff --git a/fs/xfs/scrub/dirtree_repair.c b/fs/xfs/scrub/dirtree_repair.c
index bbf6acf6fd40..1d1eafcf6eb5 100644
--- a/fs/xfs/scrub/dirtree_repair.c
+++ b/fs/xfs/scrub/dirtree_repair.c
@@ -479,6 +479,7 @@ again:
}
if (xfs_has_parent(sc->mp)) {
+ memset(&dl->ppargs, 0, sizeof(dl->ppargs));
error = xfs_parent_removename(sc->tp, &dl->ppargs, dp,
&dl->xname, sc->ip);
if (error)
@@ -618,6 +619,7 @@ xrep_dirtree_create_adoption_path(
return 0;
out_path:
+ xino_bitmap_destroy(&path->seen_inodes);
kfree(path);
return error;
}
diff --git a/fs/xfs/scrub/findparent.c b/fs/xfs/scrub/findparent.c
index 04b6b96b0a30..eab3ac2704be 100644
--- a/fs/xfs/scrub/findparent.c
+++ b/fs/xfs/scrub/findparent.c
@@ -139,32 +139,52 @@ xrep_findparent_dirent(
return 0;
}
-/*
- * If this is a directory, walk the dirents looking for any that point to the
- * scrub target inode.
- */
-STATIC int
-xrep_findparent_walk_directory(
- struct xrep_findparent_info *fpi)
+static inline bool
+xrep_findparent_want_scan_file(
+ const struct xrep_findparent_info *fpi)
{
- struct xfs_scrub *sc = fpi->sc;
- struct xfs_inode *dp = fpi->dp;
- unsigned int lock_mode;
- int error = 0;
+ const struct xfs_scrub *sc = fpi->sc;
+ const struct xfs_inode *dp = fpi->dp;
+
+ /* Only directories can be parents */
+ if (!S_ISDIR(VFS_IC(dp)->i_mode))
+ return false;
/*
* The inode being scanned cannot be its own parent, nor can any
* temporary directory we created to stage this repair.
*/
if (dp == sc->ip || dp == sc->tempip)
- return 0;
+ return false;
/*
* Similarly, temporary files created to stage a repair cannot be the
* parent of this inode.
*/
if (xrep_is_tempfile(dp))
+ return false;
+
+ return true;
+}
+
+/*
+ * If this is a directory, walk the dirents looking for any that point to the
+ * scrub target inode.
+ */
+STATIC int
+xrep_findparent_walk_file(
+ struct xrep_findparent_info *fpi)
+{
+ struct xfs_scrub *sc = fpi->sc;
+ struct xfs_inode *dp = fpi->dp;
+ unsigned int lock_mode;
+ int error = 0;
+
+ if (!xrep_findparent_want_scan_file(fpi)) {
+ if (fpi->parent_scan)
+ xchk_iscan_mark_visited(&fpi->parent_scan->iscan, dp);
return 0;
+ }
/*
* Scan the directory to see if there it contains an entry pointing to
@@ -201,6 +221,8 @@ xrep_findparent_walk_directory(
goto out_unlock;
out_unlock:
+ if (fpi->parent_scan)
+ xchk_iscan_mark_visited(&fpi->parent_scan->iscan, dp);
xfs_iunlock(dp, lock_mode);
return error;
}
@@ -308,11 +330,7 @@ xrep_findparent_scan(
ASSERT(S_ISDIR(VFS_IC(sc->ip)->i_mode));
while ((ret = xchk_iscan_iter(&pscan->iscan, &fpi.dp)) == 1) {
- if (S_ISDIR(VFS_I(fpi.dp)->i_mode))
- ret = xrep_findparent_walk_directory(&fpi);
- else
- ret = 0;
- xchk_iscan_mark_visited(&pscan->iscan, fpi.dp);
+ ret = xrep_findparent_walk_file(&fpi);
xchk_irele(sc, fpi.dp);
if (ret)
break;
@@ -401,7 +419,7 @@ xrep_findparent_confirm(
goto out_rele;
}
- error = xrep_findparent_walk_directory(&fpi);
+ error = xrep_findparent_walk_file(&fpi);
if (error)
goto out_rele;
diff --git a/fs/xfs/scrub/ialloc.c b/fs/xfs/scrub/ialloc.c
index 19c0b1b2a787..9270ad075fe0 100644
--- a/fs/xfs/scrub/ialloc.c
+++ b/fs/xfs/scrub/ialloc.c
@@ -85,6 +85,8 @@ xchk_inobt_xref_finobt(
goto no_record;
error = xfs_inobt_get_rec(cur, &frec, &has_record);
+ if (error)
+ return error;
if (!has_record)
return -EFSCORRUPTED;
@@ -188,6 +190,8 @@ xchk_finobt_xref_inobt(
goto no_record;
error = xfs_inobt_get_rec(cur, &irec, &has_record);
+ if (error)
+ return error;
if (!has_record)
return -EFSCORRUPTED;
diff --git a/fs/xfs/scrub/metapath.c b/fs/xfs/scrub/metapath.c
index ff1ff762b300..e0ee7d9b903f 100644
--- a/fs/xfs/scrub/metapath.c
+++ b/fs/xfs/scrub/metapath.c
@@ -23,6 +23,7 @@
#include "xfs_rtgroup.h"
#include "xfs_rtrmap_btree.h"
#include "xfs_rtrefcount_btree.h"
+#include "xfs_ag.h"
#include "scrub/scrub.h"
#include "scrub/common.h"
#include "scrub/trace.h"
@@ -348,12 +349,78 @@ out_cancel:
}
#ifdef CONFIG_XFS_ONLINE_REPAIR
+/*
+ * Given a directory @dp, an existing inode @ip, and a @name, link @ip into @dp
+ * under the given @name.
+ */
+static int
+xrep_metadir_add_child(
+ struct xchk_metapath *mpath,
+ xfs_ino_t old_dotdot)
+{
+ struct xfs_trans *tp = mpath->sc->tp;
+ struct xfs_dir_update *du = &mpath->du;
+ struct xfs_inode *dp = du->dp;
+ const struct xfs_name *name = du->name;
+ struct xfs_inode *ip = du->ip;
+ struct xfs_mount *mp = tp->t_mountp;
+ const unsigned int resblks = mpath->link_resblks;
+ int error;
+
+ /*
+ * The metadata file shouldn't be on the unlinked list, but we'll fix
+ * it if that is the case.
+ */
+ if (VFS_I(ip)->i_nlink == 0) {
+ struct xfs_perag *pag;
+
+ pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, I_INO(ip)));
+ error = xfs_iunlink_remove(tp, pag, ip);
+ xfs_perag_put(pag);
+ if (error)
+ return error;
+ }
+
+ error = xfs_dir_createname(tp, dp, name, I_INO(ip), resblks);
+ if (error)
+ return error;
+
+ xfs_trans_log_inode(tp, dp, XFS_ILOG_CORE);
+
+ xfs_bumplink(tp, ip);
+
+ /* update dotdot entry in child */
+ if (S_ISDIR(VFS_I(ip)->i_mode)) {
+ xfs_bumplink(tp, dp);
+
+ /* Replace the dotdot entry in the child */
+ if (old_dotdot != I_INO(dp)) {
+ error = xfs_dir_replace(tp, ip, &xfs_name_dotdot,
+ I_INO(dp), resblks);
+ if (error)
+ return error;
+ }
+ }
+
+ /* Update the child's parent pointer */
+ if (du->ppargs) {
+ error = xfs_parent_addname(tp, du->ppargs, dp, name, ip);
+ if (error)
+ return error;
+ }
+
+ xfs_dir_update_hook(dp, ip, 1, name);
+ return 0;
+}
+
/* Create the dirent represented by the final component of the path. */
STATIC int
xrep_metapath_link(
struct xchk_metapath *mpath)
{
struct xfs_scrub *sc = mpath->sc;
+ xfs_ino_t old_dotdot = NULLFSINO;
+ int error;
mpath->du.dp = mpath->dp;
mpath->du.name = &mpath->xname;
@@ -366,7 +433,21 @@ xrep_metapath_link(
trace_xrep_metapath_link(sc, mpath->path, mpath->dp, I_INO(sc->ip));
- return xfs_dir_add_child(sc->tp, mpath->link_resblks, &mpath->du);
+ if (S_ISDIR(VFS_I(sc->ip)->i_mode)) {
+ error = xchk_dir_lookup(sc, sc->ip, &xfs_name_dotdot,
+ &old_dotdot);
+ if (error && error != -ENOENT)
+ return error;
+
+ /*
+ * subdir didn't give us a dotdot entry, so we just give up
+ * and let the repair get marked as failed.
+ */
+ if (old_dotdot == NULLFSINO)
+ return 0;
+ }
+
+ return xrep_metadir_add_child(mpath, old_dotdot);
}
/* Remove the dirent at the final component of the path. */
@@ -397,7 +478,7 @@ xrep_metapath_unlink(
/* Figure out if we're removing a parent pointer too. */
if (xfs_has_parent(mp)) {
- xfs_inode_to_parent_rec(&rec, ip);
+ xfs_inode_to_parent_rec(&rec, mpath->dp);
error = xfs_parent_lookup(sc->tp, ip, &mpath->xname, &rec,
&mpath->pptr_args);
switch (error) {
@@ -556,6 +637,8 @@ xrep_metapath_try_unlink(
error = xchk_metapath_ilock_parent_and_child(mpath, ip);
if (error) {
xchk_trans_cancel(sc);
+ if (ip)
+ xchk_irele(sc, ip);
return error;
}
xfs_trans_ijoin(sc->tp, mpath->dp, 0);
diff --git a/fs/xfs/scrub/quota_repair.c b/fs/xfs/scrub/quota_repair.c
index 487bd4f68ebb..59302e8afc7e 100644
--- a/fs/xfs/scrub/quota_repair.c
+++ b/fs/xfs/scrub/quota_repair.c
@@ -325,7 +325,7 @@ xrep_quota_block(
* If there's nothing that would impede a dqiterate, we're
* done.
*/
- if ((ddq->d_type & XFS_DQTYPE_REC_MASK) != dqtype ||
+ if ((ddq->d_type & XFS_DQTYPE_REC_MASK) == dqtype &&
id == be32_to_cpu(ddq->d_id)) {
xfs_trans_brelse(sc->tp, bp);
return 0;
@@ -363,11 +363,18 @@ xrep_quota_block(
ddq->d_rtbcount, &ddq->d_rtbtimer,
defq->rtb.time);
+ /*
+ * This transaction operates on raw disk buffers, so we don't
+ * have a dquot log item to assign the LSN for us. Instead,
+ * set it to zero so that log recovery will always replay any
+ * logged dquot item atop this buffer.
+ */
+ dqblk->dd_lsn = 0;
+
/* We only support v5 filesystems so always set these. */
uuid_copy(&dqblk->dd_uuid, &sc->mp->m_sb.sb_meta_uuid);
xfs_update_cksum((char *)dqblk, sizeof(struct xfs_dqblk),
XFS_DQUOT_CRC_OFF);
- dqblk->dd_lsn = 0;
}
switch (dqtype) {
case XFS_DQTYPE_USER:
@@ -455,8 +462,7 @@ xrep_quota_data_fork(
if (truncate) {
/* Erase everything after the block containing the max dquot */
- error = xfs_bunmapi_range(&sc->tp, sc->ip, 0,
- max_dqid_off * sc->mp->m_sb.sb_blocksize,
+ error = xfs_bunmapi_range(&sc->tp, sc->ip, 0, max_dqid_off + 1,
XFS_MAX_FILEOFF);
if (error)
goto out;
diff --git a/fs/xfs/scrub/quotacheck.c b/fs/xfs/scrub/quotacheck.c
index c199d128538e..c32030a05440 100644
--- a/fs/xfs/scrub/quotacheck.c
+++ b/fs/xfs/scrub/quotacheck.c
@@ -263,8 +263,10 @@ xqcheck_mod_live_ino_dqtrx(
dqa->tx_id = p->tx_id;
error = rhashtable_insert_fast(&xqc->shadow_dquot_acct,
&dqa->hash, xqcheck_dqacct_hash_params);
- if (error)
+ if (error) {
+ kfree(dqa);
goto out_abort;
+ }
}
/* Find the shadow dqtrx (or an empty slot) here. */
diff --git a/fs/xfs/scrub/reap.c b/fs/xfs/scrub/reap.c
index fcd14c1703ea..f698b9be3dd1 100644
--- a/fs/xfs/scrub/reap.c
+++ b/fs/xfs/scrub/reap.c
@@ -601,7 +601,7 @@ xreap_configure_agextent_limits(
/* Maximum overhead of invalidating one buffer. */
const unsigned int per_binval =
- xfs_buf_inval_log_space(1, XFS_B_TO_FSBT(mp, max_binval));
+ xfs_buf_inval_log_space(1, XFS_FSB_TO_B(mp, max_binval));
/*
* For each transaction in a reap chain, we can delete some number of
@@ -680,7 +680,7 @@ xreap_configure_agcow_limits(
/* Overhead of invalidating one buffer */
const unsigned int per_binval =
- xfs_buf_inval_log_space(1, XFS_B_TO_FSBT(mp, max_binval));
+ xfs_buf_inval_log_space(1, XFS_FSB_TO_B(mp, max_binval));
/*
* For each transaction in a reap chain, we can delete some number of
@@ -1399,7 +1399,7 @@ xreap_bmapi_binval(
* far we've gotten.
*/
if (!xreap_inc_binval(rs)) {
- imap->br_blockcount = agbno_next - bno;
+ imap->br_blockcount = bno - agbno;
goto out;
}
}
diff --git a/fs/xfs/scrub/refcount.c b/fs/xfs/scrub/refcount.c
index 4e1bf23e5b89..f8c51d8fbb3d 100644
--- a/fs/xfs/scrub/refcount.c
+++ b/fs/xfs/scrub/refcount.c
@@ -410,7 +410,7 @@ xchk_refcount_mergeable(
const struct xfs_refcount_irec *r1 = &rrc->prev_rec;
/* Ignore if prev_rec is not yet initialized. */
- if (r1->rc_blockcount > 0)
+ if (r1->rc_blockcount == 0)
return false;
if (r1->rc_domain != r2->rc_domain)
@@ -581,8 +581,12 @@ xchk_xref_is_cow_staging(
if (rc.rc_domain != XFS_REFC_DOMAIN_COW)
xchk_btree_xref_set_corrupt(sc, sc->sa.refc_cur, 0);
+ /* Can't start after bno */
+ if (rc.rc_startblock > agbno)
+ xchk_btree_xref_set_corrupt(sc, sc->sa.refc_cur, 0);
+
/* Must be at least as long as what was passed in */
- if (rc.rc_blockcount < len)
+ if (rc.rc_startblock + rc.rc_blockcount < agbno + len)
xchk_btree_xref_set_corrupt(sc, sc->sa.refc_cur, 0);
}
diff --git a/fs/xfs/scrub/rgsuper.c b/fs/xfs/scrub/rgsuper.c
index 2bd2c0351b35..6e2abe5dc27c 100644
--- a/fs/xfs/scrub/rgsuper.c
+++ b/fs/xfs/scrub/rgsuper.c
@@ -36,8 +36,10 @@ xchk_rgsuperblock_xref(
if (sc->sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT)
return;
- xchk_xref_is_used_rt_space(sc, xfs_rgbno_to_rtb(sc->sr.rtg, 0), 1);
- xchk_xref_is_only_rt_owned_by(sc, 0, 1, &XFS_RMAP_OINFO_FS);
+ xchk_xref_is_used_rt_space(sc, xfs_rgbno_to_rtb(sc->sr.rtg, 0),
+ sc->mp->m_sb.sb_rextsize);
+ xchk_xref_is_only_rt_owned_by(sc, 0, sc->mp->m_sb.sb_rextsize,
+ &XFS_RMAP_OINFO_FS);
}
int
diff --git a/fs/xfs/scrub/rtrefcount.c b/fs/xfs/scrub/rtrefcount.c
index 4e7c540c8d23..3d916d71a135 100644
--- a/fs/xfs/scrub/rtrefcount.c
+++ b/fs/xfs/scrub/rtrefcount.c
@@ -20,6 +20,7 @@
#include "xfs_metafile.h"
#include "xfs_rtrefcount_btree.h"
#include "xfs_rtalloc.h"
+#include "xfs_ag.h"
#include "scrub/scrub.h"
#include "scrub/common.h"
#include "scrub/btree.h"
@@ -375,7 +376,7 @@ xchk_rtrefcount_mergeable(
const struct xfs_refcount_irec *r1 = &rrc->prev_rec;
/* Ignore if prev_rec is not yet initialized. */
- if (r1->rc_blockcount > 0)
+ if (r1->rc_blockcount == 0)
return false;
if (r1->rc_startblock + r1->rc_blockcount != r2->rc_startblock)
@@ -428,7 +429,7 @@ static inline void
xchk_rtrefcountbt_xref_gaps(
struct xfs_scrub *sc,
struct xchk_rtrefcbt_records *rrc,
- xfs_rtblock_t bno)
+ xfs_rgblock_t bno)
{
struct xfs_rmap_irec low;
struct xfs_rmap_irec high;
@@ -504,30 +505,75 @@ xchk_rtrefcountbt_rec(
return 0;
}
+/* Count the number of blocks used by the rtrefcount btree file in this AG. */
+static int
+xchk_rtrefcount_count_agblocks(
+ struct xfs_scrub *sc,
+ xfs_agnumber_t agno,
+ const struct xfs_owner_info *btree_oinfo,
+ xfs_filblks_t *blocks)
+{
+ xfs_filblks_t agblocks = 0;
+ int error;
+
+ error = xchk_ag_init_existing(sc, agno, &sc->sa);
+ if (error)
+ goto out_free;
+
+ /*
+ * If we don't have an rmap cursor, we can't complete the cross
+ * referencing, so return EFSCORRUPTED to end the loop and trigger the
+ * XFAIL flag.
+ */
+ if (!sc->sa.rmap_cur) {
+ error = -EFSCORRUPTED;
+ goto out_free;
+ }
+
+ error = xchk_count_rmap_ownedby_ag(sc, sc->sa.rmap_cur, btree_oinfo,
+ &agblocks);
+ if (error)
+ goto out_free;
+
+ *blocks += agblocks;
+out_free:
+ xchk_ag_free(sc, &sc->sa);
+ return error;
+}
+
/* Make sure we have as many refc blocks as the rmap says. */
STATIC void
-xchk_refcount_xref_rmap(
+xchk_rtrefcount_xref_rmap(
struct xfs_scrub *sc,
const struct xfs_owner_info *btree_oinfo,
xfs_extlen_t cow_blocks)
{
xfs_filblks_t refcbt_blocks = 0;
- xfs_filblks_t blocks;
- int error;
+ xfs_filblks_t blocks = 1; /* one for the iroot */
+ xfs_agnumber_t agno;
+ int error = 0;
- if (!sc->sr.rmap_cur || !sc->sa.rmap_cur || xchk_skip_xref(sc->sm))
+ if (!xfs_has_rmapbt(sc->mp) || xchk_skip_xref(sc->sm))
return;
/* Check that we saw as many refcbt blocks as the rmap knows about. */
error = xfs_btree_count_blocks(sc->sr.refc_cur, &refcbt_blocks);
if (!xchk_btree_process_error(sc, sc->sr.refc_cur, 0, &error))
return;
- error = xchk_count_rmap_ownedby_ag(sc, sc->sa.rmap_cur, btree_oinfo,
- &blocks);
- if (!xchk_should_check_xref(sc, &error, &sc->sa.rmap_cur))
+
+ for (agno = 0; agno < sc->mp->m_sb.sb_agcount; agno++) {
+ error = xchk_rtrefcount_count_agblocks(sc, agno, btree_oinfo,
+ &blocks);
+ if (error)
+ break;
+ }
+ if (!xchk_fblock_xref_process_error(sc, XFS_DATA_FORK, 0, &error))
return;
if (blocks != refcbt_blocks)
- xchk_btree_xref_set_corrupt(sc, sc->sa.rmap_cur, 0);
+ xchk_fblock_xref_set_corrupt(sc, XFS_DATA_FORK, 0);
+
+ if (!sc->sr.rmap_cur || xchk_skip_xref(sc->sm))
+ return;
/* Check that we saw as many cow blocks as the rmap knows about. */
error = xchk_count_rmap_ownedby_ag(sc, sc->sr.rmap_cur,
@@ -538,7 +584,7 @@ xchk_refcount_xref_rmap(
xchk_btree_xref_set_corrupt(sc, sc->sr.rmap_cur, 0);
}
-/* Scrub the refcount btree for some AG. */
+/* Scrub the refcount btree for some rtgroup. */
int
xchk_rtrefcountbt(
struct xfs_scrub *sc)
@@ -564,11 +610,11 @@ xchk_rtrefcountbt(
/*
* Check that all blocks between the last refcount > 1 record and the
- * end of the rt volume have at most one reverse mapping.
+ * end of the rtgroup have at most one reverse mapping.
*/
- xchk_rtrefcountbt_xref_gaps(sc, &rrc, sc->mp->m_sb.sb_rblocks);
-
- xchk_refcount_xref_rmap(sc, &btree_oinfo, rrc.cow_blocks);
+ xchk_rtrefcountbt_xref_gaps(sc, &rrc,
+ xfs_rtx_to_rgbno(sc->sr.rtg, sc->mp->m_sb.sb_rgextents));
+ xchk_rtrefcount_xref_rmap(sc, &btree_oinfo, rrc.cow_blocks);
return 0;
}
@@ -609,8 +655,12 @@ xchk_xref_is_rt_cow_staging(
if (rc.rc_domain != XFS_REFC_DOMAIN_COW)
xchk_btree_xref_set_corrupt(sc, sc->sr.refc_cur, 0);
+ /* Can't start after bno */
+ if (rc.rc_startblock > bno)
+ xchk_btree_xref_set_corrupt(sc, sc->sr.refc_cur, 0);
+
/* Must be at least as long as what was passed in */
- if (rc.rc_blockcount < len)
+ if (rc.rc_startblock + rc.rc_blockcount < bno + len)
xchk_btree_xref_set_corrupt(sc, sc->sr.refc_cur, 0);
}
diff --git a/fs/xfs/scrub/rtsummary_repair.c b/fs/xfs/scrub/rtsummary_repair.c
index f065c3e51ce2..ed763290aec1 100644
--- a/fs/xfs/scrub/rtsummary_repair.c
+++ b/fs/xfs/scrub/rtsummary_repair.c
@@ -164,9 +164,10 @@ xrep_rtsummary(
/*
* Now exchange the contents. Nothing in repair uses the temporary
* buffer, so we can reuse it for the tempfile exchrange information.
+ * Use XFS_MAX_FILEOFF here so that we correct the rtsummary file size.
*/
error = xrep_tempexch_trans_reserve(sc, XFS_DATA_FORK, 0,
- rts->rsumblocks, &rts->tempexch);
+ XFS_MAX_FILEOFF, &rts->tempexch);
if (error)
return error;
diff --git a/fs/xfs/scrub/scrub.c b/fs/xfs/scrub/scrub.c
index 8742445c86f4..12c228b7f477 100644
--- a/fs/xfs/scrub/scrub.c
+++ b/fs/xfs/scrub/scrub.c
@@ -765,8 +765,7 @@ out_nofix:
out_teardown:
error = xchk_teardown(sc, error);
out_sc:
- if (error != -ENOENT)
- xchk_stats_merge(mp, sm, &run);
+ xchk_stats_merge(mp, sm, error, &run);
kfree(sc);
out:
trace_xchk_done(XFS_I(file_inode(file)), sm, error);
diff --git a/fs/xfs/scrub/scrub.h b/fs/xfs/scrub/scrub.h
index 6d7d3523b71f..737a5d6db15f 100644
--- a/fs/xfs/scrub/scrub.h
+++ b/fs/xfs/scrub/scrub.h
@@ -261,7 +261,6 @@ static inline int xchk_nothing(struct xfs_scrub *sc)
}
/* Metadata scrubbers */
-int xchk_tester(struct xfs_scrub *sc);
int xchk_superblock(struct xfs_scrub *sc);
int xchk_agf(struct xfs_scrub *sc);
int xchk_agfl(struct xfs_scrub *sc);
diff --git a/fs/xfs/scrub/stats.c b/fs/xfs/scrub/stats.c
index ef3f6abdb706..3339cae4b39d 100644
--- a/fs/xfs/scrub/stats.c
+++ b/fs/xfs/scrub/stats.c
@@ -29,6 +29,7 @@ struct xchk_scrub_stats {
uint32_t incomplete;
uint32_t warning;
uint32_t retries;
+ uint32_t runtime_errors;
/* repair stats */
uint32_t repair_invocations;
@@ -84,6 +85,7 @@ static const char *name_map[XFS_SCRUB_TYPE_NR] = {
[XFS_SCRUB_TYPE_RGSUPER] = "rgsuper",
[XFS_SCRUB_TYPE_RTRMAPBT] = "rtrmapbt",
[XFS_SCRUB_TYPE_RTREFCBT] = "rtrefcountbt",
+ [XFS_SCRUB_TYPE_HEALTHY] = "healthy",
};
/* Format the scrub stats into a text buffer, similar to pcp style. */
@@ -99,25 +101,32 @@ xchk_stats_format(
int ret = 0;
for (i = 0; i < XFS_SCRUB_TYPE_NR; i++, css++) {
+ struct xchk_scrub_stats fss;
+
if (!name_map[i])
continue;
+ spin_lock(&css->css_lock);
+ memcpy(&fss, css, offsetof(struct xchk_scrub_stats, css_lock));
+ spin_unlock(&css->css_lock);
+
ret = scnprintf(buf, remaining,
- "%s %u %u %u %u %u %u %u %u %u %llu %u %u %llu\n",
+ "%s %u %u %u %u %u %u %u %u %u %llu %u %u %llu %u\n",
name_map[i],
- (unsigned int)css->invocations,
- (unsigned int)css->clean,
- (unsigned int)css->corrupt,
- (unsigned int)css->preen,
- (unsigned int)css->xfail,
- (unsigned int)css->xcorrupt,
- (unsigned int)css->incomplete,
- (unsigned int)css->warning,
- (unsigned int)css->retries,
- (unsigned long long)css->checktime_us,
- (unsigned int)css->repair_invocations,
- (unsigned int)css->repair_success,
- (unsigned long long)css->repairtime_us);
+ (unsigned int)fss.invocations,
+ (unsigned int)fss.clean,
+ (unsigned int)fss.corrupt,
+ (unsigned int)fss.preen,
+ (unsigned int)fss.xfail,
+ (unsigned int)fss.xcorrupt,
+ (unsigned int)fss.incomplete,
+ (unsigned int)fss.warning,
+ (unsigned int)fss.retries,
+ (unsigned long long)fss.checktime_us,
+ (unsigned int)fss.repair_invocations,
+ (unsigned int)fss.repair_success,
+ (unsigned long long)fss.repairtime_us,
+ (unsigned int)fss.runtime_errors);
if (ret <= 0)
break;
@@ -188,31 +197,41 @@ STATIC void
xchk_stats_merge_one(
struct xchk_stats *cs,
const struct xfs_scrub_metadata *sm,
+ int error,
const struct xchk_stats_run *run)
{
struct xchk_scrub_stats *css;
+ unsigned int sm_flags = sm->sm_flags;
if (sm->sm_type >= XFS_SCRUB_TYPE_NR) {
ASSERT(sm->sm_type < XFS_SCRUB_TYPE_NR);
return;
}
+ /* caller applies this same transformation after we return */
+ if (error == -EFSCORRUPTED || error == -EFSBADCRC) {
+ sm_flags |= XFS_SCRUB_OFLAG_CORRUPT;
+ error = 0;
+ }
+
css = &cs->cs_stats[sm->sm_type];
spin_lock(&css->css_lock);
css->invocations++;
- if (!(sm->sm_flags & XFS_SCRUB_OFLAG_UNCLEAN))
+ if (error)
+ css->runtime_errors++;
+ else if (!(sm_flags & XFS_SCRUB_OFLAG_UNCLEAN))
css->clean++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_CORRUPT)
+ if (sm_flags & XFS_SCRUB_OFLAG_CORRUPT)
css->corrupt++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_PREEN)
+ if (sm_flags & XFS_SCRUB_OFLAG_PREEN)
css->preen++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_XFAIL)
+ if (sm_flags & XFS_SCRUB_OFLAG_XFAIL)
css->xfail++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_XCORRUPT)
+ if (sm_flags & XFS_SCRUB_OFLAG_XCORRUPT)
css->xcorrupt++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_INCOMPLETE)
+ if (sm_flags & XFS_SCRUB_OFLAG_INCOMPLETE)
css->incomplete++;
- if (sm->sm_flags & XFS_SCRUB_OFLAG_WARNING)
+ if (sm_flags & XFS_SCRUB_OFLAG_WARNING)
css->warning++;
css->retries += run->retries;
css->checktime_us += howmany_64(run->scrub_ns, NSEC_PER_USEC);
@@ -230,10 +249,14 @@ void
xchk_stats_merge(
struct xfs_mount *mp,
const struct xfs_scrub_metadata *sm,
+ int error,
const struct xchk_stats_run *run)
{
- xchk_stats_merge_one(&global_stats, sm, run);
- xchk_stats_merge_one(mp->m_scrub_stats, sm, run);
+ if (error == -ENOENT)
+ return;
+
+ xchk_stats_merge_one(&global_stats, sm, error, run);
+ xchk_stats_merge_one(mp->m_scrub_stats, sm, error, run);
}
/* debugfs boilerplate */
diff --git a/fs/xfs/scrub/stats.h b/fs/xfs/scrub/stats.h
index b358ad8d8b90..221052b95dd0 100644
--- a/fs/xfs/scrub/stats.h
+++ b/fs/xfs/scrub/stats.h
@@ -27,7 +27,7 @@ void xchk_stats_register(struct xchk_stats *cs, struct dentry *parent);
void xchk_stats_unregister(struct xchk_stats *cs);
void xchk_stats_merge(struct xfs_mount *mp, const struct xfs_scrub_metadata *sm,
- const struct xchk_stats_run *run);
+ int error, const struct xchk_stats_run *run);
static inline u64 xchk_stats_now(void) { return ktime_get_ns(); }
static inline u64 xchk_stats_elapsed_ns(u64 since)
@@ -53,7 +53,7 @@ static inline u64 xchk_stats_elapsed_ns(u64 since)
# define xchk_stats_unregister(cs) ((void)0)
# define xchk_stats_now() (0)
# define xchk_stats_elapsed_ns(x) (0 * (x))
-# define xchk_stats_merge(mp, sm, run) ((void)0)
+# define xchk_stats_merge(mp, sm, error, run) ((void)0)
#endif /* CONFIG_XFS_ONLINE_SCRUB_STATS */
#endif /* __XFS_SCRUB_STATS_H__ */
diff --git a/fs/xfs/scrub/symlink_repair.c b/fs/xfs/scrub/symlink_repair.c
index 91c86ea0e0f1..181961364233 100644
--- a/fs/xfs/scrub/symlink_repair.c
+++ b/fs/xfs/scrub/symlink_repair.c
@@ -291,7 +291,7 @@ xrep_symlink_swap_prep(
if (error)
return error;
- xfs_trans_log_inode(sc->tp, sc->ip, 0);
+ xfs_trans_log_inode(sc->tp, sc->tempip, logflags);
error = xfs_defer_finish(&sc->tp);
if (error)
diff --git a/fs/xfs/scrub/tempfile.c b/fs/xfs/scrub/tempfile.c
index 98820003b929..59a9213a3c7d 100644
--- a/fs/xfs/scrub/tempfile.c
+++ b/fs/xfs/scrub/tempfile.c
@@ -649,6 +649,19 @@ xrep_tempexch_prep_request(
return 0;
}
+static inline unsigned int
+xrep_tempexch_estimate_sf_resblks(
+ struct xfs_scrub *sc,
+ int whichfork)
+{
+ /* repairing a symlink target */
+ if (S_ISLNK(VFS_I(sc->ip)->i_mode) && whichfork == XFS_DATA_FORK)
+ return 1;
+
+ /* everything else is a directory or an xattr structure */
+ return xfs_dabuf_nfsb(sc->mp, whichfork);
+}
+
/*
* Fill out the mapping exchange resource estimation structures in preparation
* for exchanging the contents of a metadata file that we've rebuilt in the
@@ -663,6 +676,8 @@ xrep_tempexch_estimate(
struct xfs_ifork *ifp;
struct xfs_ifork *tifp;
int whichfork = xfs_exchmaps_reqfork(req);
+ unsigned int sf_resblks =
+ xrep_tempexch_estimate_sf_resblks(sc, whichfork);
int state = 0;
/*
@@ -693,9 +708,9 @@ xrep_tempexch_estimate(
* plus the block we converted.
*/
req->ip1_bcount = sc->tempip->i_nblocks;
- req->ip2_bcount = 1;
+ req->ip2_bcount = sf_resblks;
req->nr_exchanges = 1 + tifp->if_nextents;
- req->resblks = 1;
+ req->resblks = sf_resblks;
break;
case 2:
/*
@@ -707,10 +722,10 @@ xrep_tempexch_estimate(
* is (worst case) the extent count of the file being repaired
* plus the block we converted.
*/
- req->ip1_bcount = 1;
+ req->ip1_bcount = sf_resblks;
req->ip2_bcount = sc->ip->i_nblocks;
req->nr_exchanges = 1 + ifp->if_nextents;
- req->resblks = 1;
+ req->resblks = sf_resblks;
break;
case 3:
/*
@@ -722,10 +737,10 @@ xrep_tempexch_estimate(
* fileoff 0. Presumably, the caller could not exchange the
* two inode fork areas directly.
*/
- req->ip1_bcount = 1;
- req->ip2_bcount = 1;
+ req->ip1_bcount = sf_resblks;
+ req->ip2_bcount = sf_resblks;
req->nr_exchanges = 1;
- req->resblks = 2;
+ req->resblks = 2 * sf_resblks;
break;
}
diff --git a/fs/xfs/scrub/tempfile.h b/fs/xfs/scrub/tempfile.h
index 71c1b54599c3..d44ed43bafe0 100644
--- a/fs/xfs/scrub/tempfile.h
+++ b/fs/xfs/scrub/tempfile.h
@@ -39,10 +39,6 @@ int xrep_tempfile_roll_trans(struct xfs_scrub *sc);
void xrep_tempfile_copyout_local(struct xfs_scrub *sc, int whichfork);
bool xrep_is_tempfile(const struct xfs_inode *ip);
#else
-static inline void xrep_tempfile_iolock_both(struct xfs_scrub *sc)
-{
- xchk_ilock(sc, XFS_IOLOCK_EXCL);
-}
# define xrep_is_tempfile(ip) (false)
# define xrep_tempfile_adjust_directory_tree(sc) (0)
# define xrep_tempfile_rele(sc)
diff --git a/fs/xfs/scrub/trace.h b/fs/xfs/scrub/trace.h
index 14aa0ec1f09e..0f5adc293962 100644
--- a/fs/xfs/scrub/trace.h
+++ b/fs/xfs/scrub/trace.h
@@ -1640,7 +1640,7 @@ DECLARE_EVENT_CLASS(xchk_pptr_class,
__entry->dev = ip->i_mount->m_super->s_dev;
__entry->ino = I_INO(ip);
__entry->namelen = name->len;
- memcpy(__get_str(name), name, name->len);
+ memcpy(__get_str(name), name->name, name->len);
__entry->far_ino = far_ino;
),
TP_printk("dev %d:%d ino 0x%llx name '%.*s' far_ino 0x%llx",
@@ -1706,6 +1706,39 @@ DEFINE_EVENT(xchk_dirtree_class, name, \
DEFINE_XCHK_DIRTREE_EVENT(xchk_dirtree_create_path);
DEFINE_XCHK_DIRTREE_EVENT(xchk_dirpath_walk_upwards);
+TRACE_EVENT(xchk_dirpath_badino,
+ TP_PROTO(struct xfs_scrub *sc, unsigned int path_nr,
+ unsigned int step_nr, const struct xfs_name *name,
+ const struct xfs_parent_rec *pptr),
+ TP_ARGS(sc, path_nr, step_nr, name, pptr),
+ TP_STRUCT__entry(
+ __field(dev_t, dev)
+ __field(unsigned int, path_nr)
+ __field(unsigned int, step_nr)
+ __field(xfs_ino_t, parent_ino)
+ __field(unsigned int, parent_gen)
+ __field(unsigned int, namelen)
+ __dynamic_array(char, name, name->len)
+ ),
+ TP_fast_assign(
+ __entry->dev = sc->mp->m_super->s_dev;
+ __entry->path_nr = path_nr;
+ __entry->step_nr = step_nr;
+ __entry->parent_ino = be64_to_cpu(pptr->p_ino);
+ __entry->parent_gen = be32_to_cpu(pptr->p_gen);
+ __entry->namelen = name->len;
+ memcpy(__get_str(name), name->name, name->len);
+ ),
+ TP_printk("dev %d:%d path %u step %u parent_ino 0x%llx parent_gen 0x%x name '%.*s'",
+ MAJOR(__entry->dev), MINOR(__entry->dev),
+ __entry->path_nr,
+ __entry->step_nr,
+ __entry->parent_ino,
+ __entry->parent_gen,
+ __entry->namelen,
+ __get_str(name))
+);
+
DECLARE_EVENT_CLASS(xchk_dirpath_class,
TP_PROTO(struct xfs_scrub *sc, struct xfs_inode *ip,
unsigned int path_nr, unsigned int step_nr,
diff --git a/fs/xfs/xfs_aops.c b/fs/xfs/xfs_aops.c
index 74a6089abadf..8b6119776fb3 100644
--- a/fs/xfs/xfs_aops.c
+++ b/fs/xfs/xfs_aops.c
@@ -20,6 +20,7 @@
#include "xfs_errortag.h"
#include "xfs_error.h"
#include "xfs_icache.h"
+#include "xfs_ioend.h"
#include "xfs_zone_alloc.h"
#include "xfs_rtgroup.h"
#include <linux/bio-integrity.h>
@@ -37,15 +38,6 @@ XFS_WPC(struct iomap_writepage_ctx *ctx)
}
/*
- * Fast and loose check if this write could update the on-disk inode size.
- */
-static inline bool xfs_ioend_is_append(struct iomap_ioend *ioend)
-{
- return ioend->io_offset + ioend->io_size >
- XFS_I(ioend->io_inode)->i_disk_size;
-}
-
-/*
* Update on-disk file size now that data has been written to disk.
*/
int
@@ -80,175 +72,6 @@ xfs_setfilesize(
return xfs_trans_commit(tp);
}
-static void
-xfs_ioend_put_open_zones(
- struct iomap_ioend *ioend)
-{
- struct iomap_ioend *tmp;
-
- /*
- * Put the open zone for all ioends merged into this one (if any).
- */
- list_for_each_entry(tmp, &ioend->io_list, io_list)
- xfs_open_zone_put(tmp->io_private);
-
- /*
- * The main ioend might not have an open zone if the submission failed
- * before xfs_zone_alloc_and_submit got called.
- */
- if (ioend->io_private)
- xfs_open_zone_put(ioend->io_private);
-}
-
-/*
- * IO write completion.
- */
-STATIC void
-xfs_end_ioend_write(
- struct iomap_ioend *ioend)
-{
- struct xfs_inode *ip = XFS_I(ioend->io_inode);
- struct xfs_mount *mp = ip->i_mount;
- bool is_zoned = xfs_is_zoned_inode(ip);
- xfs_off_t offset = ioend->io_offset;
- size_t size = ioend->io_size;
- unsigned int nofs_flag;
- int error;
-
- /*
- * We can allocate memory here while doing writeback on behalf of
- * memory reclaim. To avoid memory allocation deadlocks set the
- * task-wide nofs context for the following operations.
- */
- nofs_flag = memalloc_nofs_save();
-
- /*
- * Just clean up the in-memory structures if the fs has been shut down.
- */
- if (xfs_is_shutdown(mp)) {
- error = -EIO;
- goto done;
- }
-
- /*
- * Clean up all COW blocks and underlying data fork delalloc blocks on
- * I/O error. The delalloc punch is required because this ioend was
- * mapped to blocks in the COW fork and the associated pages are no
- * longer dirty. If we don't remove delalloc blocks here, they become
- * stale and can corrupt free space accounting on unmount.
- */
- error = blk_status_to_errno(ioend->io_bio.bi_status);
- if (unlikely(error)) {
- /*
- * Zoned writes update the in-core open zone accounting before
- * I/O submission. A failed write leaves that state
- * inconsistent, so shut down the filesystem instead of letting
- * later writers wait forever for open zone space to become
- * available.
- */
- if (is_zoned) {
- xfs_force_shutdown(mp, SHUTDOWN_META_IO_ERROR);
- goto done;
- }
- if (ioend->io_flags & IOMAP_IOEND_SHARED) {
- ASSERT(!is_zoned);
- xfs_reflink_cancel_cow_range(ip, offset, size, true);
- xfs_bmap_punch_delalloc_range(ip, XFS_DATA_FORK, offset,
- offset + size, NULL);
- }
- goto done;
- }
-
- /*
- * Success: commit the COW or unwritten blocks if needed.
- */
- if (is_zoned)
- error = xfs_zoned_end_io(ip, offset, size, ioend->io_sector,
- ioend->io_private, NULLFSBLOCK);
- else if (ioend->io_flags & IOMAP_IOEND_SHARED)
- error = xfs_reflink_end_cow(ip, offset, size);
- else if (ioend->io_flags & IOMAP_IOEND_UNWRITTEN)
- error = xfs_iomap_write_unwritten(ip, offset, size, false);
-
- if (!error &&
- !(ioend->io_flags & IOMAP_IOEND_DIRECT) &&
- xfs_ioend_is_append(ioend))
- error = xfs_setfilesize(ip, offset, size);
-done:
- if (is_zoned)
- xfs_ioend_put_open_zones(ioend);
- iomap_finish_ioends(ioend, error);
- memalloc_nofs_restore(nofs_flag);
-}
-
-/*
- * Finish all pending IO completions that require transactional modifications.
- *
- * We try to merge physical and logically contiguous ioends before completion to
- * minimise the number of transactions we need to perform during IO completion.
- * Both unwritten extent conversion and COW remapping need to iterate and modify
- * one physical extent at a time, so we gain nothing by merging physically
- * discontiguous extents here.
- *
- * The ioend chain length that we can be processing here is largely unbound in
- * length and we may have to perform significant amounts of work on each ioend
- * to complete it. Hence we have to be careful about holding the CPU for too
- * long in this loop.
- */
-void
-xfs_end_io(
- struct work_struct *work)
-{
- struct xfs_inode *ip =
- container_of(work, struct xfs_inode, i_ioend_work);
- struct iomap_ioend *ioend;
- struct list_head tmp;
- unsigned long flags;
-
- spin_lock_irqsave(&ip->i_ioend_lock, flags);
- list_replace_init(&ip->i_ioend_list, &tmp);
- spin_unlock_irqrestore(&ip->i_ioend_lock, flags);
-
- iomap_sort_ioends(&tmp);
- while ((ioend = list_first_entry_or_null(&tmp, struct iomap_ioend,
- io_list))) {
- list_del_init(&ioend->io_list);
- iomap_ioend_try_merge(ioend, &tmp);
- if (bio_op(&ioend->io_bio) == REQ_OP_READ)
- iomap_finish_ioends(ioend,
- blk_status_to_errno(ioend->io_bio.bi_status));
- else
- xfs_end_ioend_write(ioend);
- cond_resched();
- }
-}
-
-void
-xfs_end_bio(
- struct bio *bio)
-{
- struct iomap_ioend *ioend = iomap_ioend_from_bio(bio);
- struct xfs_inode *ip = XFS_I(ioend->io_inode);
- struct xfs_mount *mp = ip->i_mount;
- unsigned long flags;
-
- /*
- * For Appends record the actually written block number and set the
- * boundary flag if needed.
- */
- if (IS_ENABLED(CONFIG_XFS_RT) && bio_is_zone_append(bio)) {
- ioend->io_sector = bio->bi_iter.bi_sector;
- xfs_mark_rtg_boundary(ioend);
- }
-
- spin_lock_irqsave(&ip->i_ioend_lock, flags);
- if (list_empty(&ip->i_ioend_list))
- WARN_ON_ONCE(!queue_work(mp->m_unwritten_workqueue,
- &ip->i_ioend_work));
- list_add_tail(&ioend->io_list, &ip->i_ioend_list);
- spin_unlock_irqrestore(&ip->i_ioend_lock, flags);
-}
-
/*
* We cannot cancel the ioend directly on error. We may have already set other
* pages under writeback and hence we have to run I/O completion to mark the
@@ -631,13 +454,8 @@ xfs_zoned_map_blocks(
XFS_BMAPI_REMAP);
xfs_iunlock(ip, XFS_ILOCK_EXCL);
- wpc->iomap.type = IOMAP_MAPPED;
- wpc->iomap.flags = IOMAP_F_DIRTY;
- wpc->iomap.bdev = mp->m_rtdev_targp->bt_bdev;
- wpc->iomap.offset = offset;
- wpc->iomap.length = XFS_FSB_TO_B(mp, count_fsb);
- wpc->iomap.flags = IOMAP_F_ANON_WRITE;
-
+ xfs_iomap_set_anon_write(ip, &wpc->iomap, offset,
+ XFS_FSB_TO_B(mp, count_fsb));
trace_xfs_zoned_map_blocks(ip, offset, wpc->iomap.length);
return 0;
}
diff --git a/fs/xfs/xfs_aops.h b/fs/xfs/xfs_aops.h
index 5a7a0f1a0b49..d5ae5c9d4c26 100644
--- a/fs/xfs/xfs_aops.h
+++ b/fs/xfs/xfs_aops.h
@@ -10,6 +10,5 @@ extern const struct address_space_operations xfs_address_space_operations;
extern const struct address_space_operations xfs_dax_aops;
int xfs_setfilesize(struct xfs_inode *ip, xfs_off_t offset, size_t size);
-void xfs_end_bio(struct bio *bio);
#endif /* __XFS_AOPS_H__ */
diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c
index ee7c2e9c0340..8256c1d13ce2 100644
--- a/fs/xfs/xfs_buf.c
+++ b/fs/xfs/xfs_buf.c
@@ -139,7 +139,7 @@ xfs_buf_free(
ASSERT(list_empty(&bp->b_lru));
if (!xfs_buftarg_is_mem(bp->b_target) && size >= PAGE_SIZE)
- mm_account_reclaimed_pages(howmany(size, PAGE_SHIFT));
+ mm_account_reclaimed_pages(howmany(size, PAGE_SIZE));
if (is_vmalloc_addr(bp->b_addr))
vfree(bp->b_addr);
@@ -176,7 +176,7 @@ xfs_buf_alloc_kmem(
ASSERT(is_power_of_2(size));
ASSERT(size < PAGE_SIZE);
- bp->b_addr = kmalloc(size, gfp_mask);
+ bp->b_addr = kmalloc(size, gfp_mask | __GFP_RECLAIMABLE);
if (!bp->b_addr)
return -ENOMEM;
diff --git a/fs/xfs/xfs_buf_item.h b/fs/xfs/xfs_buf_item.h
index 3159325dd17b..28c79989d725 100644
--- a/fs/xfs/xfs_buf_item.h
+++ b/fs/xfs/xfs_buf_item.h
@@ -60,7 +60,6 @@ static inline void xfs_buf_dquot_iodone(struct xfs_buf *bp)
{
}
#endif /* CONFIG_XFS_QUOTA */
-void xfs_buf_iodone(struct xfs_buf *);
bool xfs_buf_log_check_iovec(struct kvec *iovec);
unsigned int xfs_buf_inval_log_space(unsigned int map_count,
diff --git a/fs/xfs/xfs_exchmaps_item.c b/fs/xfs/xfs_exchmaps_item.c
index c3745d33e54e..dd5d92ca1010 100644
--- a/fs/xfs/xfs_exchmaps_item.c
+++ b/fs/xfs/xfs_exchmaps_item.c
@@ -344,7 +344,17 @@ xfs_xmi_validate(
if (!xfs_verify_fileext(mp, xlf->xmi_startoff1, xlf->xmi_blockcount))
return false;
- return xfs_verify_fileext(mp, xlf->xmi_startoff2, xlf->xmi_blockcount);
+ if (!xfs_verify_fileext(mp, xlf->xmi_startoff2, xlf->xmi_blockcount))
+ return false;
+
+ if (xlf->xmi_flags & XFS_EXCHMAPS_SET_SIZES) {
+ if ((int64_t)xlf->xmi_isize1 < 0)
+ return false;
+ if ((int64_t)xlf->xmi_isize2 < 0)
+ return false;
+ }
+
+ return true;
}
/*
@@ -403,6 +413,13 @@ xfs_xmi_item_recover_intent(
*ipp1 = ip1;
*ipp2 = ip2;
xmi = xfs_exchmaps_init_intent(req);
+
+ /* Restore intended file sizes from recovered logged item */
+ if (req->flags & XFS_EXCHMAPS_SET_SIZES) {
+ xmi->xmi_isize1 = xlf->xmi_isize1;
+ xmi->xmi_isize2 = xlf->xmi_isize2;
+ }
+
xfs_defer_add_item(dfp, &xmi->xmi_list);
return xmi;
diff --git a/fs/xfs/xfs_exchrange.c b/fs/xfs/xfs_exchrange.c
index 94965a6c2187..c69ecd6a19de 100644
--- a/fs/xfs/xfs_exchrange.c
+++ b/fs/xfs/xfs_exchrange.c
@@ -504,6 +504,9 @@ xfs_exchange_range_finish(
{
int error;
+ if (fxr->flags & XFS_EXCHANGE_RANGE_DRY_RUN)
+ return 0;
+
error = file_remove_privs(fxr->file1);
if (error)
return error;
@@ -783,9 +786,12 @@ xfs_exchange_range(
if (ret)
return ret;
- fsnotify_modify(fxr->file1);
- if (fxr->file2 != fxr->file1)
- fsnotify_modify(fxr->file2);
+ if (!(fxr->flags & XFS_EXCHANGE_RANGE_DRY_RUN)) {
+ fsnotify_modify(fxr->file1);
+ if (fxr->file2 != fxr->file1)
+ fsnotify_modify(fxr->file2);
+ }
+
return 0;
}
diff --git a/fs/xfs/xfs_extent_busy.c b/fs/xfs/xfs_extent_busy.c
index 41cf0605ec22..6da8c1f938aa 100644
--- a/fs/xfs/xfs_extent_busy.c
+++ b/fs/xfs/xfs_extent_busy.c
@@ -161,8 +161,8 @@ xfs_extent_busy_update_extent(
xfs_agblock_t fbno,
xfs_extlen_t flen,
bool userdata)
- __releases(&eb->eb_lock)
- __acquires(&eb->eb_lock)
+ __releases(&xg->xg_busy_extents->eb_lock)
+ __acquires(&xg->xg_busy_extents->eb_lock)
{
struct xfs_extent_busy_tree *eb = xg->xg_busy_extents;
xfs_agblock_t fend = fbno + flen;
diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
index 7bff07e31cbd..d8202da15aca 100644
--- a/fs/xfs/xfs_file.c
+++ b/fs/xfs/xfs_file.c
@@ -25,7 +25,7 @@
#include "xfs_iomap.h"
#include "xfs_reflink.h"
#include "xfs_file.h"
-#include "xfs_aops.h"
+#include "xfs_ioend.h"
#include "xfs_zone_alloc.h"
#include "xfs_error.h"
#include "xfs_errortag.h"
@@ -129,9 +129,8 @@ xfs_file_fsync(
int datasync)
{
struct xfs_inode *ip = XFS_I(file->f_mapping->host);
- struct xfs_mount *mp = ip->i_mount;
- int error, err2;
int log_flushed = 0;
+ int error;
trace_xfs_file_fsync(ip);
@@ -139,30 +138,22 @@ xfs_file_fsync(
if (error)
return error;
- if (xfs_is_shutdown(mp))
+ if (xfs_is_shutdown(ip->i_mount))
return -EIO;
xfs_iflags_clear(ip, XFS_ITRUNCATED);
/*
- * If we have an RT and/or log subvolume we need to make sure to flush
- * the write cache the device used for file data first. This is to
- * ensure newly written file data make it to disk before logging the new
- * inode size in case of an extending write.
- */
- if (XFS_IS_REALTIME_INODE(ip) && mp->m_rtdev_targp != mp->m_ddev_targp)
- error = blkdev_issue_flush(mp->m_rtdev_targp->bt_bdev);
- else if (mp->m_logdev_targp != mp->m_ddev_targp)
- error = blkdev_issue_flush(mp->m_ddev_targp->bt_bdev);
-
- /*
- * If the inode has a inode log item attached, it may need the journal
- * flushed to persist any changes the log item might be tracking.
+ * If the inode has a log item attached, we must force the log up to the
+ * last LSN in which the inode was modified to ensure all metadata is
+ * persisted. The log force will flush the caches for all devices
+ * before writing the log records unless it is a no-op because there are
+ * no modifications to this inode that need to be pushed out.
*/
if (ip->i_itemp) {
- err2 = xfs_fsync_flush_log(ip, datasync, &log_flushed);
- if (err2 && !error)
- error = err2;
+ error = xfs_fsync_flush_log(ip, datasync, &log_flushed);
+ if (error)
+ return error;
}
/*
@@ -171,21 +162,11 @@ xfs_file_fsync(
* when no metadata needed to be committed.
*
* Use the inode's actual file data target rather than assuming the
- * main data device. Realtime inodes with a separate realtime device
- * are flushed before the log force, so this fallback only applies
- * when the file data target is the same as the log target.
+ * main data device.
*/
- if (!log_flushed) {
- struct xfs_buftarg *file_targp = xfs_inode_buftarg(ip);
-
- if (mp->m_logdev_targp == file_targp) {
- err2 = blkdev_issue_flush(file_targp->bt_bdev);
- if (err2 && !error)
- error = err2;
- }
- }
-
- return error;
+ if (!log_flushed)
+ return blkdev_issue_flush(xfs_inode_buftarg(ip)->bt_bdev);
+ return 0;
}
static int
diff --git a/fs/xfs/xfs_fsmap.c b/fs/xfs/xfs_fsmap.c
index b6a3bc9f143c..041bb2105ec6 100644
--- a/fs/xfs/xfs_fsmap.c
+++ b/fs/xfs/xfs_fsmap.c
@@ -1174,8 +1174,7 @@ xfs_getfsmap(
if (!xfs_getfsmap_check_keys(&head->fmh_keys[0], &head->fmh_keys[1]))
return -EINVAL;
- use_rmap = xfs_has_rmapbt(mp) &&
- has_capability_noaudit(current, CAP_SYS_ADMIN);
+ use_rmap = xfs_has_rmapbt(mp) && capable_noaudit(CAP_SYS_ADMIN);
head->fmh_entries = 0;
/* Set up our device handlers. */
diff --git a/fs/xfs/xfs_healthmon.c b/fs/xfs/xfs_healthmon.c
index 4521ffdab9f1..c3749675ef19 100644
--- a/fs/xfs/xfs_healthmon.c
+++ b/fs/xfs/xfs_healthmon.c
@@ -87,12 +87,10 @@ xfs_healthmon_put(
struct xfs_healthmon *hm)
{
if (refcount_dec_and_test(&hm->ref)) {
- struct xfs_healthmon_event *event;
- struct xfs_healthmon_event *next = hm->first_event;
+ struct xfs_healthmon_event *event, *s;
- while ((event = next) != NULL) {
+ list_for_each_entry_safe(event, s, &hm->event_list, entry) {
trace_xfs_healthmon_drop(hm, event);
- next = event->next;
kfree(event);
}
@@ -173,9 +171,13 @@ static inline void xfs_healthmon_bump_lost(struct xfs_healthmon *hm)
*/
static bool
xfs_healthmon_merge_events(
- struct xfs_healthmon_event *existing,
+ struct xfs_healthmon *hm,
const struct xfs_healthmon_event *new)
{
+ struct xfs_healthmon_event *existing =
+ list_last_entry_or_null(&hm->event_list, struct
+ xfs_healthmon_event, entry);
+
if (!existing)
return false;
@@ -192,7 +194,7 @@ xfs_healthmon_merge_events(
case XFS_HEALTHMON_LOST:
existing->lostcount += new->lostcount;
- return true;
+ goto out_merge;
case XFS_HEALTHMON_SICK:
case XFS_HEALTHMON_CORRUPT:
@@ -200,19 +202,19 @@ xfs_healthmon_merge_events(
switch (existing->domain) {
case XFS_HEALTHMON_FS:
existing->fsmask |= new->fsmask;
- return true;
+ goto out_merge;
case XFS_HEALTHMON_AG:
case XFS_HEALTHMON_RTGROUP:
if (existing->group == new->group){
existing->grpmask |= new->grpmask;
- return true;
+ goto out_merge;
}
return false;
case XFS_HEALTHMON_INODE:
if (existing->ino == new->ino &&
existing->gen == new->gen) {
existing->imask |= new->imask;
- return true;
+ goto out_merge;
}
return false;
default:
@@ -224,18 +226,18 @@ xfs_healthmon_merge_events(
case XFS_HEALTHMON_SHUTDOWN:
/* yes, we can race to shutdown */
existing->flags |= new->flags;
- return true;
+ goto out_merge;
case XFS_HEALTHMON_MEDIA_ERROR:
/* physically adjacent errors can merge */
if (existing->daddr + existing->bbcount == new->daddr) {
existing->bbcount += new->bbcount;
- return true;
+ goto out_merge;
}
if (new->daddr + new->bbcount == existing->daddr) {
existing->daddr = new->daddr;
existing->bbcount += new->bbcount;
- return true;
+ goto out_merge;
}
return false;
@@ -250,63 +252,58 @@ xfs_healthmon_merge_events(
if (existing->fpos + existing->flen == new->fpos) {
existing->flen += new->flen;
- return true;
+ goto out_merge;
}
if (new->fpos + new->flen == existing->fpos) {
existing->fpos = new->fpos;
existing->flen += new->flen;
- return true;
+ goto out_merge;
}
return false;
}
return false;
+
+out_merge:
+ trace_xfs_healthmon_merge(hm, existing);
+ return true;
}
-/* Insert an event onto the start of the queue. */
+enum insert_where {
+ INSERT_HEAD,
+ INSERT_TAIL,
+};
+
+/* Add an event onto the start or the end of the queue. */
static inline void
__xfs_healthmon_insert(
struct xfs_healthmon *hm,
+ enum insert_where where,
struct xfs_healthmon_event *event)
{
struct timespec64 now;
+ lockdep_assert_held(&hm->lock);
+
ktime_get_coarse_real_ts64(&now);
event->time_ns = (now.tv_sec * NSEC_PER_SEC) + now.tv_nsec;
- event->next = hm->first_event;
- if (!hm->first_event)
- hm->first_event = event;
- if (!hm->last_event)
- hm->last_event = event;
- xfs_healthmon_bump_events(hm);
- wake_up(&hm->wait);
-
- trace_xfs_healthmon_insert(hm, event);
-}
+ switch (where) {
+ case INSERT_HEAD:
+ trace_xfs_healthmon_insert_head(hm, event);
-/* Push an event onto the end of the queue. */
-static inline void
-__xfs_healthmon_push(
- struct xfs_healthmon *hm,
- struct xfs_healthmon_event *event)
-{
- struct timespec64 now;
+ list_add(&event->entry, &hm->event_list);
+ break;
+ case INSERT_TAIL:
+ trace_xfs_healthmon_insert_tail(hm, event);
- ktime_get_coarse_real_ts64(&now);
- event->time_ns = (now.tv_sec * NSEC_PER_SEC) + now.tv_nsec;
+ list_add_tail(&event->entry, &hm->event_list);
+ break;
+ }
- if (!hm->first_event)
- hm->first_event = event;
- if (hm->last_event)
- hm->last_event->next = event;
- hm->last_event = event;
- event->next = NULL;
xfs_healthmon_bump_events(hm);
wake_up(&hm->wait);
-
- trace_xfs_healthmon_push(hm, event);
}
/* Deal with any previously lost events */
@@ -321,8 +318,7 @@ xfs_healthmon_clear_lost_prev(
};
struct xfs_healthmon_event *event = NULL;
- if (xfs_healthmon_merge_events(hm->last_event, &lost_event)) {
- trace_xfs_healthmon_merge(hm, hm->last_event);
+ if (xfs_healthmon_merge_events(hm, &lost_event)) {
wake_up(&hm->wait);
goto cleared;
}
@@ -330,10 +326,12 @@ xfs_healthmon_clear_lost_prev(
if (hm->events < XFS_HEALTHMON_MAX_EVENTS)
event = kmemdup(&lost_event, sizeof(struct xfs_healthmon_event),
GFP_NOFS);
- if (!event)
+ if (!event) {
+ xfs_healthmon_bump_lost(hm);
return -ENOMEM;
+ }
- __xfs_healthmon_push(hm, event);
+ __xfs_healthmon_insert(hm, INSERT_TAIL, event);
cleared:
hm->lost_prev_event = 0;
return 0;
@@ -369,8 +367,7 @@ xfs_healthmon_push(
}
/* Try to merge with the newest event */
- if (xfs_healthmon_merge_events(hm->last_event, template)) {
- trace_xfs_healthmon_merge(hm, hm->last_event);
+ if (xfs_healthmon_merge_events(hm, template)) {
wake_up(&hm->wait);
goto out_unlock;
}
@@ -387,7 +384,7 @@ xfs_healthmon_push(
goto out_unlock;
}
- __xfs_healthmon_push(hm, event);
+ __xfs_healthmon_insert(hm, INSERT_TAIL, event);
out_unlock:
mutex_unlock(&hm->lock);
@@ -415,8 +412,10 @@ xfs_healthmon_unmount(
* There's nothing actionable for userspace after an unmount. Once
* we've inserted the unmount event, hm no longer owns that event.
*/
- __xfs_healthmon_insert(hm, hm->unmount_event);
+ mutex_lock(&hm->lock);
+ __xfs_healthmon_insert(hm, INSERT_HEAD, hm->unmount_event);
hm->unmount_event = NULL;
+ mutex_unlock(&hm->lock);
xfs_healthmon_detach(hm);
xfs_healthmon_put(hm);
@@ -738,6 +737,13 @@ static const unsigned int type_map[] = {
[XFS_HEALTHMON_DATALOST] = XFS_HEALTH_MONITOR_TYPE_DATALOST,
};
+static inline bool
+xfs_healthmon_check_outbuffer_space(const struct xfs_healthmon *hm)
+{
+ return hm->bufhead + sizeof(struct xfs_health_monitor_event) <=
+ hm->bufsize;
+}
+
/* Render event as a V0 structure */
STATIC int
xfs_healthmon_format_v0(
@@ -804,10 +810,10 @@ xfs_healthmon_format_v0(
break;
}
- ASSERT(hm->bufhead + sizeof(hme) <= hm->bufsize);
+ ASSERT(xfs_healthmon_check_outbuffer_space(hm));
/* copy formatted object to the outbuf */
- if (hm->bufhead + sizeof(hme) <= hm->bufsize) {
+ if (xfs_healthmon_check_outbuffer_space(hm)) {
memcpy(hm->buffer + hm->bufhead, &hme, sizeof(hme));
hm->bufhead += sizeof(hme);
}
@@ -890,15 +896,18 @@ xfs_healthmon_format_pop(
{
struct xfs_healthmon_event *event;
- if (hm->bufhead + sizeof(*event) > hm->bufsize)
+ /*
+ * Don't bother if there's not enough space to format even one event in
+ * the outbuffer.
+ */
+ if (!xfs_healthmon_check_outbuffer_space(hm))
return NULL;
mutex_lock(&hm->lock);
- event = hm->first_event;
+ event = list_first_entry_or_null(&hm->event_list,
+ struct xfs_healthmon_event, entry);
if (event) {
- if (hm->last_event == event)
- hm->last_event = NULL;
- hm->first_event = event->next;
+ list_del_init(&event->entry);
hm->events--;
trace_xfs_healthmon_pop(hm, event);
@@ -1198,6 +1207,7 @@ xfs_ioc_health_monitor(
return -ENOMEM;
hm->dev = mp->m_super->s_dev;
refcount_set(&hm->ref, 1);
+ INIT_LIST_HEAD(&hm->event_list);
mutex_init(&hm->lock);
init_waitqueue_head(&hm->wait);
@@ -1213,7 +1223,9 @@ xfs_ioc_health_monitor(
}
running_event->type = XFS_HEALTHMON_RUNNING;
running_event->domain = XFS_HEALTHMON_MOUNT;
- __xfs_healthmon_insert(hm, running_event);
+ mutex_lock(&hm->lock);
+ __xfs_healthmon_insert(hm, INSERT_HEAD, running_event);
+ mutex_unlock(&hm->lock);
/*
* Preallocate the unmount event so that we can't fail to notify the
diff --git a/fs/xfs/xfs_healthmon.h b/fs/xfs/xfs_healthmon.h
index 0e936507037f..fa3deb187a2b 100644
--- a/fs/xfs/xfs_healthmon.h
+++ b/fs/xfs/xfs_healthmon.h
@@ -31,8 +31,7 @@ struct xfs_healthmon {
struct mutex lock;
/* list of event objects */
- struct xfs_healthmon_event *first_event;
- struct xfs_healthmon_event *last_event;
+ struct list_head event_list;
/* preallocated event for unmount */
struct xfs_healthmon_event *unmount_event;
@@ -110,7 +109,7 @@ enum xfs_healthmon_domain {
};
struct xfs_healthmon_event {
- struct xfs_healthmon_event *next;
+ struct list_head entry;
enum xfs_healthmon_type type;
enum xfs_healthmon_domain domain;
diff --git a/fs/xfs/xfs_icache.c b/fs/xfs/xfs_icache.c
index 9d8dd30bd927..82dac88e3c4c 100644
--- a/fs/xfs/xfs_icache.c
+++ b/fs/xfs/xfs_icache.c
@@ -82,24 +82,20 @@ static inline xa_mark_t ici_tag_to_mark(unsigned int tag)
/*
* Allocate and initialise an xfs_inode.
+ *
+ * This can happen in context of already dirtied transactions, so the memory
+ * allocations must not fail.
*/
struct xfs_inode *
xfs_inode_alloc(
struct xfs_mount *mp,
xfs_ino_t ino)
{
+ gfp_t gfp = GFP_KERNEL | __GFP_NOFAIL;
struct xfs_inode *ip;
- /*
- * XXX: If this didn't occur in transactions, we could drop GFP_NOFAIL
- * and return NULL here on ENOMEM.
- */
- ip = alloc_inode_sb(mp->m_super, xfs_inode_cache, GFP_KERNEL | __GFP_NOFAIL);
-
- if (inode_init_always(mp->m_super, VFS_I(ip))) {
- kmem_cache_free(xfs_inode_cache, ip);
- return NULL;
- }
+ ip = alloc_inode_sb(mp->m_super, xfs_inode_cache, gfp);
+ inode_init_always_gfp(mp->m_super, VFS_I(ip), gfp);
VFS_I(ip)->i_ino = ino;
/* VFS doesn't initialise i_mode! */
@@ -501,7 +497,8 @@ xfs_iget_cache_hit(
struct xfs_inode *ip,
xfs_ino_t ino,
int flags,
- int lock_flags) __releases(RCU)
+ int lock_flags)
+ __releases_shared(RCU)
{
struct inode *inode = VFS_I(ip);
struct xfs_mount *mp = ip->i_mount;
diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h
index 34c1038ebfcd..1602027cd0aa 100644
--- a/fs/xfs/xfs_inode.h
+++ b/fs/xfs/xfs_inode.h
@@ -585,7 +585,6 @@ uint xfs_ilock_attr_map_shared(struct xfs_inode *);
int xfs_ifree(struct xfs_trans *, struct xfs_inode *);
int xfs_itruncate_extents_flags(struct xfs_trans **,
struct xfs_inode *, int, xfs_fsize_t, int);
-void xfs_iext_realloc(xfs_inode_t *, int, int);
int xfs_log_force_inode(struct xfs_inode *ip);
void xfs_iunpin_wait(xfs_inode_t *);
diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c
index 1b53701bebea..96ca3e480cb9 100644
--- a/fs/xfs/xfs_ioctl.c
+++ b/fs/xfs/xfs_ioctl.c
@@ -647,7 +647,7 @@ xfs_ioctl_setattr_get_trans(
goto out_error;
error = xfs_trans_alloc_ichange(ip, NULL, NULL, pdqp,
- has_capability_noaudit(current, CAP_FOWNER), &tp);
+ capable_noaudit(CAP_FOWNER), &tp);
if (error)
goto out_error;
diff --git a/fs/xfs/xfs_ioend.c b/fs/xfs/xfs_ioend.c
new file mode 100644
index 000000000000..40695d18dac0
--- /dev/null
+++ b/fs/xfs/xfs_ioend.c
@@ -0,0 +1,184 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Copyright (c) 2016-2025 Christoph Hellwig.
+ * All Rights Reserved.
+ */
+#include "xfs_platform.h"
+#include "xfs_shared.h"
+#include "xfs_format.h"
+#include "xfs_log_format.h"
+#include "xfs_trans_resv.h"
+#include "xfs_mount.h"
+#include "xfs_inode.h"
+#include "xfs_iomap.h"
+#include "xfs_trace.h"
+#include "xfs_bmap_util.h"
+#include "xfs_reflink.h"
+#include "xfs_zone_alloc.h"
+#include "xfs_ioend.h"
+
+static void
+xfs_ioend_put_open_zones(
+ struct iomap_ioend *ioend)
+{
+ struct iomap_ioend *tmp;
+
+ /*
+ * Put the open zone for all ioends merged into this one (if any).
+ */
+ list_for_each_entry(tmp, &ioend->io_list, io_list)
+ xfs_open_zone_put(tmp->io_private);
+
+ /*
+ * The main ioend might not have an open zone if the submission failed
+ * before xfs_zone_alloc_and_submit got called.
+ */
+ if (ioend->io_private)
+ xfs_open_zone_put(ioend->io_private);
+}
+
+static void
+xfs_end_ioend_write(
+ struct iomap_ioend *ioend)
+{
+ struct xfs_inode *ip = XFS_I(ioend->io_inode);
+ struct xfs_mount *mp = ip->i_mount;
+ bool is_zoned = xfs_is_zoned_inode(ip);
+ xfs_off_t offset = ioend->io_offset;
+ size_t size = ioend->io_size;
+ unsigned int nofs_flag;
+ int error;
+
+ /*
+ * We can allocate memory here while doing writeback on behalf of
+ * memory reclaim. To avoid memory allocation deadlocks set the
+ * task-wide nofs context for the following operations.
+ */
+ nofs_flag = memalloc_nofs_save();
+
+ /*
+ * Just clean up the in-memory structures if the fs has been shut down.
+ */
+ if (xfs_is_shutdown(mp)) {
+ error = -EIO;
+ goto done;
+ }
+
+ /*
+ * Clean up all COW blocks and underlying data fork delalloc blocks on
+ * I/O error. The delalloc punch is required because this ioend was
+ * mapped to blocks in the COW fork and the associated pages are no
+ * longer dirty. If we don't remove delalloc blocks here, they become
+ * stale and can corrupt free space accounting on unmount.
+ */
+ error = blk_status_to_errno(ioend->io_bio.bi_status);
+ if (unlikely(error)) {
+ /*
+ * Zoned writes update the in-core open zone accounting before
+ * I/O submission. A failed write leaves that state
+ * inconsistent, so shut down the filesystem instead of letting
+ * later writers wait forever for open zone space to become
+ * available.
+ */
+ if (is_zoned) {
+ xfs_force_shutdown(mp, SHUTDOWN_META_IO_ERROR);
+ goto done;
+ }
+ if (ioend->io_flags & IOMAP_IOEND_SHARED) {
+ ASSERT(!is_zoned);
+ xfs_reflink_cancel_cow_range(ip, offset, size, true);
+ xfs_bmap_punch_delalloc_range(ip, XFS_DATA_FORK, offset,
+ offset + size, NULL);
+ }
+ goto done;
+ }
+
+ /*
+ * Success: commit the COW or unwritten blocks if needed.
+ */
+ if (is_zoned)
+ error = xfs_zoned_end_io(ip, offset, size, ioend->io_sector,
+ ioend->io_private, NULLFSBLOCK);
+ else if (ioend->io_flags & IOMAP_IOEND_SHARED)
+ error = xfs_reflink_end_cow(ip, offset, size);
+ else if (ioend->io_flags & IOMAP_IOEND_UNWRITTEN)
+ error = xfs_iomap_write_unwritten(ip, offset, size, false);
+
+ if (!error &&
+ !(ioend->io_flags & IOMAP_IOEND_DIRECT) &&
+ xfs_ioend_is_append(ioend))
+ error = xfs_setfilesize(ip, offset, size);
+done:
+ if (is_zoned)
+ xfs_ioend_put_open_zones(ioend);
+ iomap_finish_ioends(ioend, error);
+ memalloc_nofs_restore(nofs_flag);
+}
+
+/*
+ * Finish all pending IO completions that require transactional modifications.
+ *
+ * We try to merge physical and logically contiguous ioends before completion to
+ * minimise the number of transactions we need to perform during IO completion.
+ * Both unwritten extent conversion and COW remapping need to iterate and modify
+ * one physical extent at a time, so we gain nothing by merging physically
+ * discontiguous extents here.
+ *
+ * The ioend chain length that we can be processing here is largely unbound in
+ * length and we may have to perform significant amounts of work on each ioend
+ * to complete it. Hence we have to be careful about holding the CPU for too
+ * long in this loop.
+ */
+void
+xfs_end_io(
+ struct work_struct *work)
+{
+ struct xfs_inode *ip =
+ container_of(work, struct xfs_inode, i_ioend_work);
+ struct iomap_ioend *ioend;
+ struct list_head tmp;
+ unsigned long flags;
+
+ spin_lock_irqsave(&ip->i_ioend_lock, flags);
+ list_replace_init(&ip->i_ioend_list, &tmp);
+ spin_unlock_irqrestore(&ip->i_ioend_lock, flags);
+
+ iomap_sort_ioends(&tmp);
+ while ((ioend = list_first_entry_or_null(&tmp, struct iomap_ioend,
+ io_list))) {
+ list_del_init(&ioend->io_list);
+ iomap_ioend_try_merge(ioend, &tmp);
+ if (bio_op(&ioend->io_bio) == REQ_OP_READ)
+ iomap_finish_ioends(ioend,
+ blk_status_to_errno(ioend->io_bio.bi_status));
+ else
+ xfs_end_ioend_write(ioend);
+ cond_resched();
+ }
+}
+
+void
+xfs_end_bio(
+ struct bio *bio)
+{
+ struct iomap_ioend *ioend = iomap_ioend_from_bio(bio);
+ struct xfs_inode *ip = XFS_I(ioend->io_inode);
+ struct xfs_mount *mp = ip->i_mount;
+ unsigned long flags;
+
+ /*
+ * For Appends record the actually written block number and set the
+ * boundary flag if needed.
+ */
+ if (IS_ENABLED(CONFIG_XFS_RT) && bio_is_zone_append(bio)) {
+ ioend->io_sector = bio->bi_iter.bi_sector;
+ xfs_mark_rtg_boundary(ioend);
+ }
+
+ spin_lock_irqsave(&ip->i_ioend_lock, flags);
+ if (list_empty(&ip->i_ioend_list))
+ WARN_ON_ONCE(!queue_work(mp->m_unwritten_workqueue,
+ &ip->i_ioend_work));
+ list_add_tail(&ioend->io_list, &ip->i_ioend_list);
+ spin_unlock_irqrestore(&ip->i_ioend_lock, flags);
+}
diff --git a/fs/xfs/xfs_ioend.h b/fs/xfs/xfs_ioend.h
new file mode 100644
index 000000000000..525865767fca
--- /dev/null
+++ b/fs/xfs/xfs_ioend.h
@@ -0,0 +1,16 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef __XFS_IOEND_H
+#define __XFS_IOEND_H
+
+/*
+ * Fast and loose check if this write could update the on-disk inode size.
+ */
+static inline bool xfs_ioend_is_append(struct iomap_ioend *ioend)
+{
+ return ioend->io_offset + ioend->io_size >
+ XFS_I(ioend->io_inode)->i_disk_size;
+}
+
+void xfs_end_bio(struct bio *bio);
+
+#endif /* __XFS_IOEND_H */
diff --git a/fs/xfs/xfs_iomap.c b/fs/xfs/xfs_iomap.c
index 71c45be8c652..7c6238fed61e 100644
--- a/fs/xfs/xfs_iomap.c
+++ b/fs/xfs/xfs_iomap.c
@@ -1083,12 +1083,7 @@ xfs_zoned_direct_write_iomap_begin(
return error;
}
- iomap->type = IOMAP_MAPPED;
- iomap->flags = IOMAP_F_DIRTY;
- iomap->bdev = ip->i_mount->m_rtdev_targp->bt_bdev;
- iomap->offset = offset;
- iomap->length = length;
- iomap->flags = IOMAP_F_ANON_WRITE;
+ xfs_iomap_set_anon_write(ip, iomap, offset, length);
return 0;
}
diff --git a/fs/xfs/xfs_iomap.h b/fs/xfs/xfs_iomap.h
index cffcec532ea6..f2520a9b3a13 100644
--- a/fs/xfs/xfs_iomap.h
+++ b/fs/xfs/xfs_iomap.h
@@ -29,6 +29,22 @@ int xfs_zero_range(struct xfs_inode *ip, loff_t pos, loff_t len,
int xfs_truncate_page(struct xfs_inode *ip, loff_t pos,
struct xfs_zone_alloc_ctx *ac, bool *did_zero);
+static inline void
+xfs_iomap_set_anon_write(
+ struct xfs_inode *ip,
+ struct iomap *iomap,
+ loff_t offset,
+ loff_t length)
+{
+ iomap->type = IOMAP_MAPPED;
+ iomap->bdev = ip->i_mount->m_rtdev_targp->bt_bdev;
+ iomap->offset = offset;
+ iomap->length = length;
+ iomap->flags = IOMAP_F_ANON_WRITE | IOMAP_F_DIRTY;
+ if (bdev_has_integrity_csum(iomap->bdev))
+ iomap->flags |= IOMAP_F_INTEGRITY;
+}
+
static inline xfs_filblks_t
xfs_aligned_fsb_count(
xfs_fileoff_t offset_fsb,
diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c
index 4a3299abf774..d1306e723899 100644
--- a/fs/xfs/xfs_iops.c
+++ b/fs/xfs/xfs_iops.c
@@ -834,7 +834,7 @@ xfs_setattr_nonsize(
}
error = xfs_trans_alloc_ichange(ip, udqp, gdqp, NULL,
- has_capability_noaudit(current, CAP_FOWNER), &tp);
+ capable_noaudit(CAP_FOWNER), &tp);
if (error)
goto out_dqrele;
diff --git a/fs/xfs/xfs_log.c b/fs/xfs/xfs_log.c
index f807f8f4f705..f4f81d893e8c 100644
--- a/fs/xfs/xfs_log.c
+++ b/fs/xfs/xfs_log.c
@@ -422,6 +422,8 @@ out_error:
static void
xlog_state_shutdown_callbacks(
struct xlog *log)
+ __releases(&log->l_icloglock)
+ __acquires(&log->l_icloglock)
{
struct xlog_in_core *iclog;
LIST_HEAD(cb_list);
@@ -470,6 +472,8 @@ xlog_state_release_iclog(
struct xlog *log,
struct xlog_in_core *iclog,
struct xlog_ticket *ticket)
+ __releases(&log->l_icloglock)
+ __acquires(&log->l_icloglock)
{
bool last_ref;
@@ -744,13 +748,16 @@ xfs_log_mount_cancel(
*/
static inline int
xlog_force_iclog(
+ struct xlog *log,
struct xlog_in_core *iclog)
+ __releases(&log->l_icloglock)
+ __acquires(&log->l_icloglock)
{
atomic_inc(&iclog->ic_refcnt);
iclog->ic_flags |= XLOG_ICL_NEED_FLUSH | XLOG_ICL_NEED_FUA;
if (iclog->ic_state == XLOG_STATE_ACTIVE)
- xlog_state_switch_iclogs(iclog->ic_log, iclog, 0);
- return xlog_state_release_iclog(iclog->ic_log, iclog, NULL);
+ xlog_state_switch_iclogs(log, iclog, 0);
+ return xlog_state_release_iclog(log, iclog, NULL);
}
/*
@@ -778,11 +785,10 @@ xlog_wait_iclog_completion(struct xlog *log)
*/
int
xlog_wait_on_iclog(
+ struct xlog *log,
struct xlog_in_core *iclog)
- __releases(iclog->ic_log->l_icloglock)
+ __releases(log->l_icloglock)
{
- struct xlog *log = iclog->ic_log;
-
trace_xlog_iclog_wait_on(iclog, _RET_IP_);
if (!xlog_is_shutdown(log) &&
iclog->ic_state != XLOG_STATE_ACTIVE &&
@@ -879,8 +885,8 @@ out_err:
spin_lock(&log->l_icloglock);
iclog = log->l_iclog;
- error = xlog_force_iclog(iclog);
- xlog_wait_on_iclog(iclog);
+ error = xlog_force_iclog(log, iclog);
+ xlog_wait_on_iclog(log, iclog);
if (tic) {
trace_xfs_log_umount_write(log, tic);
@@ -1538,6 +1544,35 @@ xlog_bio_end_io(
&iclog->ic_end_io_work);
}
+/*
+ * When using multiple devices, we also need to flush the data and RT device
+ * caches first to ensure that all metadata writeback covered by the LSN in
+ * this iclog is on stable storage. This is slow, but it *must* complete
+ * before we issue the external log IO.
+ *
+ * If the flush fails, we cannot conclude that past metadata writeback from
+ * the log succeeded. Repeating the flush is not possible, hence we must
+ * shut down with log IO error to avoid shutdown re-entering this path and
+ * erroring out again.
+ */
+static int
+xlog_flush_data_caches(
+ struct xlog *log)
+{
+ struct xfs_mount *mp = log->l_mp;
+
+ if (log->l_targ != mp->m_ddev_targp) {
+ if (blkdev_issue_flush(mp->m_ddev_targp->bt_bdev))
+ return -EIO;
+ }
+ if (mp->m_rtdev_targp && mp->m_rtdev_targp != mp->m_ddev_targp) {
+ if (blkdev_issue_flush(mp->m_rtdev_targp->bt_bdev))
+ return -EIO;
+ }
+
+ return 0;
+}
+
STATIC void
xlog_write_iclog(
struct xlog *log,
@@ -1582,21 +1617,9 @@ xlog_write_iclog(
iclog->ic_bio.bi_private = iclog;
if (iclog->ic_flags & XLOG_ICL_NEED_FLUSH) {
- iclog->ic_bio.bi_opf |= REQ_PREFLUSH;
- /*
- * For external log devices, we also need to flush the data
- * device cache first to ensure all metadata writeback covered
- * by the LSN in this iclog is on stable storage. This is slow,
- * but it *must* complete before we issue the external log IO.
- *
- * If the flush fails, we cannot conclude that past metadata
- * writeback from the log succeeded. Repeating the flush is
- * not possible, hence we must shut down with log IO error to
- * avoid shutdown re-entering this path and erroring out again.
- */
- if (log->l_targ != log->l_mp->m_ddev_targp &&
- blkdev_issue_flush(log->l_mp->m_ddev_targp->bt_bdev))
+ if (xlog_flush_data_caches(log))
goto shutdown;
+ iclog->ic_bio.bi_opf |= REQ_PREFLUSH;
}
if (iclog->ic_flags & XLOG_ICL_NEED_FUA)
iclog->ic_bio.bi_opf |= REQ_FUA;
@@ -2741,14 +2764,17 @@ xlog_state_switch_iclogs(
*/
static int
xlog_force_and_check_iclog(
+ struct xlog *log,
struct xlog_in_core *iclog,
bool *completed)
+ __releases(&log->l_icloglock)
+ __acquires(&log->l_icloglock)
{
xfs_lsn_t lsn = be64_to_cpu(iclog->ic_header->h_lsn);
int error;
*completed = false;
- error = xlog_force_iclog(iclog);
+ error = xlog_force_iclog(log, iclog);
if (error)
return error;
@@ -2825,7 +2851,7 @@ xfs_log_force(
/* We have exclusive access to this iclog. */
bool completed;
- if (xlog_force_and_check_iclog(iclog, &completed))
+ if (xlog_force_and_check_iclog(log, iclog, &completed))
goto out_error;
if (completed)
@@ -2850,7 +2876,7 @@ xfs_log_force(
iclog->ic_flags |= XLOG_ICL_NEED_FLUSH | XLOG_ICL_NEED_FUA;
if (flags & XFS_LOG_SYNC)
- return xlog_wait_on_iclog(iclog);
+ return xlog_wait_on_iclog(log, iclog);
out_unlock:
spin_unlock(&log->l_icloglock);
return 0;
@@ -2920,7 +2946,7 @@ xlog_force_lsn(
&log->l_icloglock);
return -EAGAIN;
}
- if (xlog_force_and_check_iclog(iclog, &completed))
+ if (xlog_force_and_check_iclog(log, iclog, &completed))
goto out_error;
if (log_flushed)
*log_flushed = 1;
@@ -2948,7 +2974,7 @@ xlog_force_lsn(
}
if (flags & XFS_LOG_SYNC)
- return xlog_wait_on_iclog(iclog);
+ return xlog_wait_on_iclog(log, iclog);
out_unlock:
spin_unlock(&log->l_icloglock);
return 0;
diff --git a/fs/xfs/xfs_log.h b/fs/xfs/xfs_log.h
index ca66429bf6c9..f715695e8fcb 100644
--- a/fs/xfs/xfs_log.h
+++ b/fs/xfs/xfs_log.h
@@ -105,8 +105,6 @@ int xfs_log_mount(struct xfs_mount *mp,
int num_bblocks);
int xfs_log_mount_finish(struct xfs_mount *mp);
void xfs_log_mount_cancel(struct xfs_mount *);
-xfs_lsn_t xlog_assign_tail_lsn(struct xfs_mount *mp);
-xfs_lsn_t xlog_assign_tail_lsn_locked(struct xfs_mount *mp);
void xfs_log_space_wake(struct xfs_mount *mp);
int xfs_log_reserve(struct xfs_mount *mp, int length, int count,
struct xlog_ticket **ticket, bool permanent);
diff --git a/fs/xfs/xfs_log_cil.c b/fs/xfs/xfs_log_cil.c
index 639f875a8fb2..f9e07a32f60f 100644
--- a/fs/xfs/xfs_log_cil.c
+++ b/fs/xfs/xfs_log_cil.c
@@ -1055,9 +1055,10 @@ xlog_cil_set_ctx_write_state(
spin_unlock(&cil->xc_push_lock);
/*
- * Make sure the metadata we are about to overwrite in the log
- * has been flushed to stable storage before this iclog is
- * issued.
+ * Flush the write cache before writing the start record so that
+ * the metadata we are about to overwrite in the log and the
+ * data that new allocations in this context refer to are
+ * persisted to stable storage before this iclog is written.
*/
spin_lock(&cil->xc_log->l_icloglock);
iclog->ic_flags |= XLOG_ICL_NEED_FLUSH;
@@ -1556,7 +1557,7 @@ xlog_cil_push_work(
* iclogs older than ic_prev. Hence we only need to wait
* on the most recent older iclog here.
*/
- xlog_wait_on_iclog(ctx->commit_iclog->ic_prev);
+ xlog_wait_on_iclog(log, ctx->commit_iclog->ic_prev);
spin_lock(&log->l_icloglock);
}
@@ -1627,6 +1628,7 @@ out_abort_free_ticket:
static void
xlog_cil_push_background(
struct xlog *log)
+ __releases_shared(&log->l_cilp->xc_ctx_lock)
{
struct xfs_cil *cil = log->l_cilp;
int space_used = atomic_read(&cil->xc_ctx->space_used);
diff --git a/fs/xfs/xfs_log_priv.h b/fs/xfs/xfs_log_priv.h
index cf1e4ce61a8c..6d9673c41cdf 100644
--- a/fs/xfs/xfs_log_priv.h
+++ b/fs/xfs/xfs_log_priv.h
@@ -605,8 +605,8 @@ xlog_wait(
remove_wait_queue(wq, &wait);
}
-int xlog_wait_on_iclog(struct xlog_in_core *iclog)
- __releases(iclog->ic_log->l_icloglock);
+int xlog_wait_on_iclog(struct xlog *log, struct xlog_in_core *iclog)
+ __releases(log->l_icloglock);
/* Calculate the distance between two LSNs in bytes */
static inline uint64_t
diff --git a/fs/xfs/xfs_mru_cache.c b/fs/xfs/xfs_mru_cache.c
index d61ec8cb126d..3f3af2e2e31c 100644
--- a/fs/xfs/xfs_mru_cache.c
+++ b/fs/xfs/xfs_mru_cache.c
@@ -520,7 +520,7 @@ xfs_mru_cache_lookup(
if (elem) {
list_del(&elem->list_node);
_xfs_mru_cache_list_insert(mru, elem);
- __release(mru_lock); /* help sparse not be stupid */
+ __release(&mru->lock);
} else
spin_unlock(&mru->lock);
diff --git a/fs/xfs/xfs_platform.h b/fs/xfs/xfs_platform.h
index 59a33c60e0ca..5d542e95fe44 100644
--- a/fs/xfs/xfs_platform.h
+++ b/fs/xfs/xfs_platform.h
@@ -289,15 +289,4 @@ int xfs_rw_bdev(struct block_device *bdev, sector_t sector, unsigned int count,
# define PTR_FMT "%p"
#endif
-/*
- * Helper for IO routines to grab backing pages from allocated kernel memory.
- */
-static inline struct page *
-kmem_to_page(void *addr)
-{
- if (is_vmalloc_addr(addr))
- return vmalloc_to_page(addr);
- return virt_to_page(addr);
-}
-
#endif /* _XFS_PLATFORM_H */
diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
index 4b2eeb7783f7..b24db75eaedc 100644
--- a/fs/xfs/xfs_super.c
+++ b/fs/xfs/xfs_super.c
@@ -445,7 +445,7 @@ xfs_shutdown_devices(
blkdev_issue_flush(mp->m_logdev_targp->bt_bdev);
invalidate_bdev(mp->m_logdev_targp->bt_bdev);
}
- if (mp->m_rtdev_targp) {
+ if (mp->m_rtdev_targp && mp->m_rtdev_targp != mp->m_ddev_targp) {
blkdev_issue_flush(mp->m_rtdev_targp->bt_bdev);
invalidate_bdev(mp->m_rtdev_targp->bt_bdev);
}
diff --git a/fs/xfs/xfs_trace.h b/fs/xfs/xfs_trace.h
index f333c938fbd9..6aa379c2cf0c 100644
--- a/fs/xfs/xfs_trace.h
+++ b/fs/xfs/xfs_trace.h
@@ -6139,8 +6139,8 @@ DEFINE_EVENT(xfs_healthmon_event_class, name, \
TP_PROTO(const struct xfs_healthmon *hm, \
const struct xfs_healthmon_event *event), \
TP_ARGS(hm, event))
-DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_insert);
-DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_push);
+DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_insert_head);
+DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_insert_tail);
DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_pop);
DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_format);
DEFINE_HEALTHMONEVENT_EVENT(xfs_healthmon_format_overflow);
diff --git a/fs/xfs/xfs_trans_ail.c b/fs/xfs/xfs_trans_ail.c
index 99a9bf3762b7..f955479a08fd 100644
--- a/fs/xfs/xfs_trans_ail.c
+++ b/fs/xfs/xfs_trans_ail.c
@@ -33,7 +33,7 @@ STATIC void
xfs_ail_check(
struct xfs_ail *ailp,
struct xfs_log_item *lip)
- __must_hold(&ailp->ail_lock)
+ __must_hold(&ailp->ail_lock)
{
struct xfs_log_item *prev_lip;
struct xfs_log_item *next_lip;
@@ -321,6 +321,7 @@ static void
xfs_ail_delete(
struct xfs_ail *ailp,
struct xfs_log_item *lip)
+ __must_hold(&ailp->ail_lock)
{
xfs_ail_check(ailp, lip);
list_del(&lip->li_ail);
@@ -899,6 +900,7 @@ xfs_lsn_t
xfs_ail_delete_one(
struct xfs_ail *ailp,
struct xfs_log_item *lip)
+ __must_hold(&ailp->ail_lock)
{
struct xfs_log_item *mlip = xfs_ail_min(ailp);
xfs_lsn_t lsn = lip->li_lsn;
diff --git a/fs/xfs/xfs_trans_buf.c b/fs/xfs/xfs_trans_buf.c
index 1e025848811a..a5d25b703dfc 100644
--- a/fs/xfs/xfs_trans_buf.c
+++ b/fs/xfs/xfs_trans_buf.c
@@ -521,7 +521,8 @@ xfs_trans_log_buf(
{
struct xfs_buf_log_item *bip = bp->b_log_item;
- ASSERT(first <= last && last < BBTOB(bp->b_length));
+ ASSERT(first <= last);
+ ASSERT(last < BBTOB(bp->b_length));
ASSERT(!(bip->bli_flags & XFS_BLI_ORDERED));
xfs_trans_dirty_buf(tp, bp);
diff --git a/fs/xfs/xfs_verify_media.c b/fs/xfs/xfs_verify_media.c
index 5ead3976d511..b75c81f8fcc0 100644
--- a/fs/xfs/xfs_verify_media.c
+++ b/fs/xfs/xfs_verify_media.c
@@ -268,6 +268,8 @@ xfs_verify_media(
struct xfs_buftarg *btp = NULL;
struct bio *bio;
struct folio *folio;
+ xfs_daddr_t dev_start = 0;
+ xfs_daddr_t dev_end = 0;
xfs_daddr_t daddr;
uint64_t bbcount;
int error = 0;
@@ -277,24 +279,33 @@ xfs_verify_media(
switch (me->me_dev) {
case XFS_DEV_DATA:
btp = mp->m_ddev_targp;
+ dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_dblocks);
break;
case XFS_DEV_LOG:
- if (mp->m_logdev_targp != mp->m_ddev_targp)
+ if (mp->m_logdev_targp != mp->m_ddev_targp) {
btp = mp->m_logdev_targp;
+ dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_logblocks);
+ }
break;
case XFS_DEV_RT:
btp = mp->m_rtdev_targp;
+ dev_start = XFS_FSB_TO_BB(mp, mp->m_sb.sb_rtstart);
+ dev_end = XFS_FSB_TO_BB(mp, mp->m_sb.sb_rtstart +
+ mp->m_sb.sb_rblocks);
break;
}
if (!btp)
return -ENODEV;
/*
- * If the caller told us to verify beyond the end of the disk, tell the
- * user exactly where that was.
+ * If the caller told us to verify before the start or beyond the end
+ * of the disk volume, tell the user exactly where the volume starts
+ * and ends.
*/
- if (me->me_end_daddr > btp->bt_nr_sectors)
- me->me_end_daddr = btp->bt_nr_sectors;
+ if (me->me_end_daddr > dev_end)
+ me->me_end_daddr = dev_end;
+ if (me->me_start_daddr < dev_start)
+ me->me_start_daddr = dev_start;
/* start and end have to be aligned to the lba size */
if (!IS_ALIGNED(BBTOB(me->me_start_daddr | me->me_end_daddr),
@@ -323,8 +334,7 @@ xfs_verify_media(
* verifying.
*/
daddr = me->me_start_daddr;
- bbcount = min_t(sector_t, me->me_end_daddr, btp->bt_nr_sectors) -
- me->me_start_daddr;
+ bbcount = me->me_end_daddr - me->me_start_daddr;
folio = xfs_verify_alloc_folio(xfs_verify_iosize(me, btp, bbcount));
if (!folio)
diff --git a/fs/xfs/xfs_zone_alloc.c b/fs/xfs/xfs_zone_alloc.c
index 7d13fa7ab30a..28c1e48909fa 100644
--- a/fs/xfs/xfs_zone_alloc.c
+++ b/fs/xfs/xfs_zone_alloc.c
@@ -475,6 +475,8 @@ static struct xfs_open_zone *
xfs_try_open_zone(
struct xfs_mount *mp,
enum rw_hint write_hint)
+ __releases(&mp->m_zone_info->zi_open_zones_lock)
+ __acquires(&mp->m_zone_info->zi_open_zones_lock)
{
struct xfs_zone_info *zi = mp->m_zone_info;
struct xfs_open_zone *oz;
@@ -793,17 +795,35 @@ xfs_get_cached_zone(
rcu_read_lock();
oz = VFS_I(ip)->i_private;
- if (oz) {
- /*
- * GC only steals open zones at mount time, so no GC zones
- * should end up in the cache.
- */
- ASSERT(!oz->oz_is_gc);
- if (!atomic_inc_not_zero(&oz->oz_ref))
+ if (!oz)
+ goto out_unlock;
+
+ /*
+ * GC only steals open zones at mount time, so no GC zones should end up
+ * in the cache.
+ */
+ ASSERT(!oz->oz_is_gc);
+
+ /*
+ * Drop the old cached open zone if it is full.
+ */
+ if (oz->oz_allocated == rtg_blocks(oz->oz_rtg)) {
+ spin_lock(&ip->i_flags_lock);
+ oz = VFS_I(ip)->i_private;
+ if (oz && oz->oz_allocated == rtg_blocks(oz->oz_rtg)) {
+ VFS_I(ip)->i_private = NULL;
+ spin_unlock(&ip->i_flags_lock);
+ xfs_open_zone_put(oz);
oz = NULL;
+ goto out_unlock;
+ }
+ spin_unlock(&ip->i_flags_lock);
}
- rcu_read_unlock();
+ if (!atomic_inc_not_zero(&oz->oz_ref))
+ oz = NULL;
+out_unlock:
+ rcu_read_unlock();
return oz;
}
@@ -818,18 +838,41 @@ xfs_get_cached_zone(
* that were every written to, but significantly simplifies the cached zone
* lookup. Because the open_zone is clearly marked as full when all data
* in the underlying RTG was written, the caching is always safe.
+ *
+ * Called with a reference on @oz held. And returns two references on the
+ * returned zone: one for the caller and one for pinning the zone in
+ * inode->i_private.
*/
-static void
+static struct xfs_open_zone *
xfs_set_cached_zone(
struct xfs_inode *ip,
struct xfs_open_zone *oz)
{
struct xfs_open_zone *old_oz;
+ /*
+ * If the open zone cached in the inode still has free space, use that
+ * instead of the new open zone just selected. This can happen when
+ * multiple threads race to perform zone selection for an inode.
+ * io_uring worker threads seem to be good way to trigger this.
+ *
+ * We need to grab an extra reference to this open zone as the caller
+ * owns a reference in addition to the i_private pointer.
+ */
+ spin_lock(&ip->i_flags_lock);
+ old_oz = VFS_I(ip)->i_private;
+ if (old_oz && old_oz->oz_allocated < rtg_blocks(old_oz->oz_rtg) &&
+ atomic_inc_not_zero(&old_oz->oz_ref)) {
+ spin_unlock(&ip->i_flags_lock);
+ xfs_open_zone_put(oz);
+ return old_oz;
+ }
+ VFS_I(ip)->i_private = oz;
atomic_inc(&oz->oz_ref);
- old_oz = xchg(&VFS_I(ip)->i_private, oz);
+ spin_unlock(&ip->i_flags_lock);
if (old_oz)
xfs_open_zone_put(old_oz);
+ return oz;
}
static void
@@ -873,14 +916,13 @@ xfs_zone_alloc_and_submit(
* the inode is still associated with a zone and use that if so.
*/
if (!*oz)
+select_zone:
*oz = xfs_get_cached_zone(ip);
-
if (!*oz) {
-select_zone:
*oz = xfs_select_zone(mp, write_hint, pack_tight);
if (!*oz)
goto out_error;
- xfs_set_cached_zone(ip, *oz);
+ *oz = xfs_set_cached_zone(ip, *oz);
}
alloc_len = xfs_zone_alloc_blocks(*oz, XFS_B_TO_FSB(mp, ioend->io_size),
diff --git a/fs/xfs/xfs_zone_gc.c b/fs/xfs/xfs_zone_gc.c
index d0b85179a3d2..5fdcf98a2133 100644
--- a/fs/xfs/xfs_zone_gc.c
+++ b/fs/xfs/xfs_zone_gc.c
@@ -869,6 +869,11 @@ xfs_zone_gc_write_chunk(
WRITE_ONCE(chunk->state, XFS_GC_BIO_NEW);
list_move_tail(&chunk->entry, &data->writing);
+ /*
+ * If we run on top of stacked block device, the read I/O might have
+ * reset bi_bdev, restore it to the one we want.
+ */
+ bio_set_dev(&chunk->bio, mp->m_rtdev_targp->bt_bdev);
bio_reuse(&chunk->bio, REQ_OP_WRITE);
while ((split_chunk = xfs_zone_gc_split_write(data, chunk)))
xfs_zone_gc_submit_write(data, split_chunk);
diff --git a/fs/xfs/xfs_zone_space_resv.c b/fs/xfs/xfs_zone_space_resv.c
index 5c6e6ef627e4..7aa3c74fb2e0 100644
--- a/fs/xfs/xfs_zone_space_resv.c
+++ b/fs/xfs/xfs_zone_space_resv.c
@@ -85,13 +85,13 @@ xfs_zoned_add_available(
struct xfs_zone_info *zi = mp->m_zone_info;
struct xfs_zone_reservation *reservation;
- if (list_empty_careful(&zi->zi_reclaim_reservations)) {
- xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb);
+ spin_lock(&zi->zi_reservation_lock);
+ xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb);
+ if (list_empty(&zi->zi_reclaim_reservations)) {
+ spin_unlock(&zi->zi_reservation_lock);
return;
}
- spin_lock(&zi->zi_reservation_lock);
- xfs_add_freecounter(mp, XC_FREE_RTAVAILABLE, count_fsb);
count_fsb = xfs_sum_freecounter(mp, XC_FREE_RTAVAILABLE);
list_for_each_entry(reservation, &zi->zi_reclaim_reservations, entry) {
if (reservation->count_fsb > count_fsb)